{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LP4YQRV6M7E3L2NQIIYI3KDNS4","short_pith_number":"pith:LP4YQRV6","schema_version":"1.0","canonical_sha256":"5bf98846be67c9b5e9b042308da86d97010064daf950c0d6bbb55222f7f2ad5f","source":{"kind":"arxiv","id":"2607.27610","version":1},"attestation_state":"computed","paper":{"title":"Kalman Meets Curriculum: Efficient Dynamic Prompt Selection for Adaptive RL Finetuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Baochang Zhang, Haiguang Liu, Haodong Zhu, Linlin Yang, Sheng Xu, Yangyang Ren, Yanjing Li","submitted_at":"2026-07-30T02:55:02Z","abstract_excerpt":"Reinforcement learning (RL) finetuning significantly enhances the reasoning capabilities of large language models (LLMs), yet its effectiveness critically depends on selecting prompts of appropriate difficulty for the current policy. This is challenging because prompt difficulty evolves throughout training. Existing online methods therefore face a trade-off: evaluation-based approaches are accurate but expensive, while prediction-based approaches are efficient but typically assume stationary difficulty, making them ill-suited to RL's non-stationary training dynamics. To address these issues, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.27610","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T02:55:02Z","cross_cats_sorted":[],"title_canon_sha256":"34e56cf46f22c09a87f501eb86dfa77d21e348349566dc07fa3b4f63e4fd6543","abstract_canon_sha256":"76a7376fb39fdf1e50cbc1ca2d15fb496d48f93e24cbc6a2a29247ead4d361fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5bf98846be67c9b5e9b042308da86d97010064daf950c0d6bbb55222f7f2ad5f","last_reissued_at":"2026-07-31T01:27:25.845894Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:27:25.845894Z"},"graph_snapshot":{"paper":{"title":"Kalman Meets Curriculum: Efficient Dynamic Prompt Selection for Adaptive RL Finetuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Baochang Zhang, Haiguang Liu, Haodong Zhu, Linlin Yang, Sheng Xu, Yangyang Ren, Yanjing Li","submitted_at":"2026-07-30T02:55:02Z","abstract_excerpt":"Reinforcement learning (RL) finetuning significantly enhances the reasoning capabilities of large language models (LLMs), yet its effectiveness critically depends on selecting prompts of appropriate difficulty for the current policy. This is challenging because prompt difficulty evolves throughout training. Existing online methods therefore face a trade-off: evaluation-based approaches are accurate but expensive, while prediction-based approaches are efficient but typically assume stationary difficulty, making them ill-suited to RL's non-stationary training dynamics. To address these issues, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27610","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27610/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.27610","created_at":"2026-07-31T01:27:25.849120+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.27610v1","created_at":"2026-07-31T01:27:25.849120+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27610","created_at":"2026-07-31T01:27:25.849120+00:00"},{"alias_kind":"pith_short_12","alias_value":"LP4YQRV6M7E3","created_at":"2026-07-31T01:27:25.849120+00:00"},{"alias_kind":"pith_short_16","alias_value":"LP4YQRV6M7E3L2NQ","created_at":"2026-07-31T01:27:25.849120+00:00"},{"alias_kind":"pith_short_8","alias_value":"LP4YQRV6","created_at":"2026-07-31T01:27:25.849120+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4","json":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4.json","graph_json":"https://pith.science/api/pith-number/LP4YQRV6M7E3L2NQIIYI3KDNS4/graph.json","events_json":"https://pith.science/api/pith-number/LP4YQRV6M7E3L2NQIIYI3KDNS4/events.json","paper":"https://pith.science/paper/LP4YQRV6"},"agent_actions":{"view_html":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4","download_json":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4.json","view_paper":"https://pith.science/paper/LP4YQRV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.27610&json=true","fetch_graph":"https://pith.science/api/pith-number/LP4YQRV6M7E3L2NQIIYI3KDNS4/graph.json","fetch_events":"https://pith.science/api/pith-number/LP4YQRV6M7E3L2NQIIYI3KDNS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4/action/storage_attestation","attest_author":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4/action/author_attestation","sign_citation":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4/action/citation_signature","submit_replication":"https://pith.science/pith/LP4YQRV6M7E3L2NQIIYI3KDNS4/action/replication_record"}},"created_at":"2026-07-31T01:27:25.849120+00:00","updated_at":"2026-07-31T01:27:25.849120+00:00"}