{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_7c55e2c2d6d0eb689e76efcef6b964cb8aebaa1f42d34a02b4f034158f12c05e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_7c55e2c2d6d0eb689e76efcef6b964cb8aebaa1f42d34a02b4f034158f12c05e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a86f60c5cb3bc173fecbceaa08f189fa328d60ca1c6f114fcc8326bbe15a5cb5","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"a86f60c5cb3bc173fecbceaa08f189fa328d60ca1c6f114fcc8326bbe15a5cb5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a86f60c5cb3bc173fecbceaa08f189fa328d60ca1c6f114fcc8326bbe15a5cb5","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.15862v2 Announce Type: replace \nAbstract: Large language model (LLM) agents have made rapid progress on short-horizon, well-scoped tasks, yet their ability to sustain coherent decisions in dynamic long-horizon environments remains uncertain. We introduce RetailBench, a data-grounded simulation benchmark for evaluating tool-using LLM agents in single-store supermarket operation. RetailBench models retail management as a partially observable decision process and is designed to support thousand-day-scale simulations. In this environment, agents must manage pricing, replenishment, supplier selection, shelf assortment, inventory aging, customer feedback, external events, and cash-flow constraints. We evaluate seven contemporary LLMs under representative agent frameworks over a 180-day evaluation horizon and compare them with a privileged oracle policy. Results show substantial variation across models: only a small subset survives the full evaluation horizon, and even the stronges","title":"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments","url":"https://arxiv.org/abs/2606.15862","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.15862v2 Announce Type: replace \nAbstract: Large language model (LLM) agents have made rapid progress on short-horizon, well-scoped tasks, yet their ability to sustain coherent decisions in dynamic long-horizon environments remains uncertain. We introduce RetailBench, a data-grounded simulation benchmark for evaluating tool-using LLM agents in single-store supermarket operation. RetailBench models retail management as a partially observable decision process and is designed to support thousand-day-scale simulations. In this environment, agents must manage pricing, replenishment, supplier selection, shelf assortment, inventory aging, customer feedback, external events, and cash-flow constraints. We evaluate seven contemporary LLMs under representative agent frameworks over a 180-day evaluation horizon and compare them with a privileged oracle policy. Results show substantial variation across models: only a small subset survives the full evaluation horizon, and even the stronges","title":"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.15862"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:5e88498426d5733e299cfb59deaba78dfecdb3115197d9489ebf2ac564f283037c461f16feb481c9bc8251357e598b6f086ca42afcbee4a1a70ea830fdf13b0c","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.15862"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f8ec49b3b3fe2592d3656a50c5eb99b1ae694d8e748076661132f007da1a5918","leaf_index":235699,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3c9ce49740bc5ec01a071edf69f010ea50da918580b29d0e381a57385a066e7c","side":"left"},{"sibling":"d5468874881da5c627c68498a821ddc9edd095a26203d1122e834348296003a0","side":"left"},{"sibling":"df9153b4e4ddf75c56beb67c67ae685df5214768b1031c8cce563db4a3d5236f","side":"right"},{"sibling":"5250402b0a5e17460fb769338b450674f4c08254e0eb8accedc49467a7f8273e","side":"right"},{"sibling":"4442bda67975b90d528d28bc8753aa6b9b70fb28df0873891fa8767cb96f90de","side":"left"},{"sibling":"7dc6760aab106e939aa5db80f98209b84ff8e42be64c8d3a64855d28698ac2fc","side":"left"},{"sibling":"d4e4da8292a59c15443bbd4cd8c53ca2bf82cd90b34e48fb7704c5e2e86fe09f","side":"right"},{"sibling":"9af7ea2b04101a414ff44ef903d5d381777f0286b489347e7e959b64158ff697","side":"left"},{"sibling":"aa68ebe8f5e8e96388fc8d1af3aa08be7ccd27ab4cebcf5913560c48d877bc27","side":"right"},{"sibling":"8253d44cf1ed30d3ab19c2b339fb4000a1fa173182c65390e9e8dabf8173b9e9","side":"right"},{"sibling":"e2bf9b60400244c698c0196109f54323457abc5b64dee08ec33ab14cc4faaef7","side":"right"},{"sibling":"86664e7f68ba08b8dfcf77dda51a4dfa7fcfc986d4ad7c704ffb71b669202da7","side":"left"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_7c55e2c2d6d0eb689e76efcef6b964cb8aebaa1f42d34a02b4f034158f12c05e"}}