{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_f79aae8b099d5f3b64d09f129304d25e9ce87c0f863bfd576d52b68a4cc7ac0f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_f79aae8b099d5f3b64d09f129304d25e9ce87c0f863bfd576d52b68a4cc7ac0f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4ee43810bccb2f9dbebd58db1c62131e91732107f2e21a238a24a48f0d617ea2","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"4ee43810bccb2f9dbebd58db1c62131e91732107f2e21a238a24a48f0d617ea2","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4ee43810bccb2f9dbebd58db1c62131e91732107f2e21a238a24a48f0d617ea2","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.04412v1 Announce Type: new \nAbstract: Reinforcement learning (RL) for non-verifiable instruction following increasingly relies on LLM judges with prompt-specific rubrics as reward signals. While recent methods adapt these rubrics to the evolving policy during training, the training prompts themselves remain static, drawn from fixed corpora. This static approach often results in a critical misalignment between prompt difficulty and policy capability, leaving the judge unable to recover a discriminative reward signal when prompts fail to elicit quality variance among rollouts. To address this misalignment, we introduce LLM-as-a-Tutor, a framework that extends the LLM's role from judge to tutor: a single model serves as an examiner that pairwise-compares policy rollouts to detect non-challenging prompts, and as a generator that appends atomic constraints to them. This append-only design monotonically raises difficulty in step with the policy's capability, producing a self-calib","title":"LLM-as-a-Tutor: Policy-Aware Prompt Adaptation for Non-Verifiable RL","url":"https://arxiv.org/abs/2607.04412","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.04412v1 Announce Type: new \nAbstract: Reinforcement learning (RL) for non-verifiable instruction following increasingly relies on LLM judges with prompt-specific rubrics as reward signals. While recent methods adapt these rubrics to the evolving policy during training, the training prompts themselves remain static, drawn from fixed corpora. This static approach often results in a critical misalignment between prompt difficulty and policy capability, leaving the judge unable to recover a discriminative reward signal when prompts fail to elicit quality variance among rollouts. To address this misalignment, we introduce LLM-as-a-Tutor, a framework that extends the LLM's role from judge to tutor: a single model serves as an examiner that pairwise-compares policy rollouts to detect non-challenging prompts, and as a generator that appends atomic constraints to them. This append-only design monotonically raises difficulty in step with the policy's capability, producing a self-calib","title":"LLM-as-a-Tutor: Policy-Aware Prompt Adaptation for Non-Verifiable RL","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.04412"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:7578c63fee591a3a289111dc9200685518feffc3d4f0e07f7e50bf1ea7fa85a0bc35ae6c753ea2e79c30f62d4c685ac32f811348d79233c77b1c64d7cd85be08","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.04412"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"6a221269db82123d14e9f44e9eae3491ec5dda165e612cc62e900e2d74750c9b","leaf_index":288766,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6061cdbba299306ba8dea4c947860c1fe4de26935c7a0e0d1b91b8c1deb1f94a","side":"right"},{"sibling":"38dfa143155b97ff35255db68a874c5b22e9c52fc67827b8366edb3d7e966576","side":"left"},{"sibling":"9fc753066f94d39d5c228d5dfc26bb5c81a57e058b5f984c734588850e2d0061","side":"left"},{"sibling":"b78c4f3c976cfccd8682001674e9025e3cd2312dd9082d741027be89e3ba7168","side":"left"},{"sibling":"5547927da08df8599cc06965777697fcd81850beeacc9c257378b724de456bea","side":"left"},{"sibling":"6ce2490462f2183557d667ea543ad19b275e012200585fa65831bfe675408ade","side":"left"},{"sibling":"61f7272108de819a7a77493153b988651785961143dd7f7cbf21c3d662725b0d","side":"left"},{"sibling":"693d22f9477576140ca2120520ac7b4b633fb122e7f590d3abc01e28b77f9cfe","side":"left"},{"sibling":"2b24ae0b0e86a6bb85711be0b56eabcfea8026b8cad7364829595d5562ea5b79","side":"left"},{"sibling":"66acb8614a0400fe91b4430bfa4ca8f7f359fb95aea361dfea6a7c88a9fe24a7","side":"left"},{"sibling":"b76af82f0e95812185e25016354f4707e41c1bacfe819e4948e5e364745511e8","side":"left"},{"sibling":"175b61fd9088baa970ad449ad7fc5d7babb21d38120cfa8c28053ae9d448ac83","side":"right"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_f79aae8b099d5f3b64d09f129304d25e9ce87c0f863bfd576d52b68a4cc7ac0f"}}