{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TPZ7VQAFWFBBMQAO63EN2NBWHT","short_pith_number":"pith:TPZ7VQAF","schema_version":"1.0","canonical_sha256":"9bf3fac005b14216400ef6c8dd34363cc238ed672345a812b330ba4a554c0dc0","source":{"kind":"arxiv","id":"2601.00898","version":3},"attestation_state":"computed","paper":{"title":"Dichotomous Diffusion Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Guang Chen, Hangjun Ye, Jianxiong Li, Jingjing Liu, Jinqiao Wang, Kexin Zheng, Liyuan Mao, Ruiming Liang, Tianyi Tan, Xianyuan Zhan, Yinan Zheng, Zhihao Wang","submitted_at":"2025-12-31T16:56:56Z","abstract_excerpt":"Diffusion-based policies have gained growing popularity in solving a wide range of decision-making tasks due to their superior expressiveness and controllable generation during inference. However, effectively training large diffusion policies using reinforcement learning (RL) remains challenging. Existing methods either suffer from unstable training due to directly maximizing value objectives, or face computational issues due to relying on crude Gaussian likelihood approximation, which requires a large amount of sufficiently small denoising steps. In this work, we propose DIPOLE (Dichotomous d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.00898","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-12-31T16:56:56Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"08d9c9980812b5b5d9343b7050d70be7820b7a1d272e803abac8410f52fe2207","abstract_canon_sha256":"d43820e22b131f18da32c542b09e089e0d7cbe30a6ffb0a104bc02283e48fcec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-20T02:19:17.288012Z","signature_b64":"4B4VL1+DgqNjlEUJpHRvT+tD2RVe7rAr0SFZtB7gpgbo2JDBGeOs+LuGG8MoS98w2aUDOjWIkEGTtgQqj9ASDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9bf3fac005b14216400ef6c8dd34363cc238ed672345a812b330ba4a554c0dc0","last_reissued_at":"2026-07-20T02:19:17.287060Z","signature_status":"signed_v1","first_computed_at":"2026-07-20T02:19:17.287060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dichotomous Diffusion Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Guang Chen, Hangjun Ye, Jianxiong Li, Jingjing Liu, Jinqiao Wang, Kexin Zheng, Liyuan Mao, Ruiming Liang, Tianyi Tan, Xianyuan Zhan, Yinan Zheng, Zhihao Wang","submitted_at":"2025-12-31T16:56:56Z","abstract_excerpt":"Diffusion-based policies have gained growing popularity in solving a wide range of decision-making tasks due to their superior expressiveness and controllable generation during inference. However, effectively training large diffusion policies using reinforcement learning (RL) remains challenging. Existing methods either suffer from unstable training due to directly maximizing value objectives, or face computational issues due to relying on crude Gaussian likelihood approximation, which requires a large amount of sufficiently small denoising steps. In this work, we propose DIPOLE (Dichotomous d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.00898","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.00898/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.00898","created_at":"2026-07-20T02:19:17.287502+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.00898v3","created_at":"2026-07-20T02:19:17.287502+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.00898","created_at":"2026-07-20T02:19:17.287502+00:00"},{"alias_kind":"pith_short_12","alias_value":"TPZ7VQAFWFBB","created_at":"2026-07-20T02:19:17.287502+00:00"},{"alias_kind":"pith_short_16","alias_value":"TPZ7VQAFWFBBMQAO","created_at":"2026-07-20T02:19:17.287502+00:00"},{"alias_kind":"pith_short_8","alias_value":"TPZ7VQAF","created_at":"2026-07-20T02:19:17.287502+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.24742","citing_title":"World Value Models for Robotic Manipulation","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04470","citing_title":"CRAFT: Counterfactual-to-Interactive Reinforcement Fine-Tuning for Driving Policies","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT","json":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT.json","graph_json":"https://pith.science/api/pith-number/TPZ7VQAFWFBBMQAO63EN2NBWHT/graph.json","events_json":"https://pith.science/api/pith-number/TPZ7VQAFWFBBMQAO63EN2NBWHT/events.json","paper":"https://pith.science/paper/TPZ7VQAF"},"agent_actions":{"view_html":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT","download_json":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT.json","view_paper":"https://pith.science/paper/TPZ7VQAF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.00898&json=true","fetch_graph":"https://pith.science/api/pith-number/TPZ7VQAFWFBBMQAO63EN2NBWHT/graph.json","fetch_events":"https://pith.science/api/pith-number/TPZ7VQAFWFBBMQAO63EN2NBWHT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT/action/storage_attestation","attest_author":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT/action/author_attestation","sign_citation":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT/action/citation_signature","submit_replication":"https://pith.science/pith/TPZ7VQAFWFBBMQAO63EN2NBWHT/action/replication_record"}},"created_at":"2026-07-20T02:19:17.287502+00:00","updated_at":"2026-07-20T02:19:17.287502+00:00"}