{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:WN3EH2EAALZNZMFWQWR5323EBO","short_pith_number":"pith:WN3EH2EA","canonical_record":{"source":{"id":"2502.00361","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-01T07:55:06Z","cross_cats_sorted":[],"title_canon_sha256":"4217daa5cc37f0bef8423dae899a16f7ec7a13cfb167b31d2958006b1e71d153","abstract_canon_sha256":"5621d3b5f3850a9557bb78aba5530ca7721a3fb0fdb9f071ec66ac128e84c48e"},"schema_version":"1.0"},"canonical_sha256":"b37643e88002f2dcb0b685a3ddeb640bb58222a966590f53821d285fcbec5b47","source":{"kind":"arxiv","id":"2502.00361","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.00361","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"arxiv_version","alias_value":"2502.00361v4","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00361","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_12","alias_value":"WN3EH2EAALZN","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_16","alias_value":"WN3EH2EAALZNZMFW","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_8","alias_value":"WN3EH2EA","created_at":"2026-07-05T11:29:06Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:WN3EH2EAALZNZMFWQWR5323EBO","target":"record","payload":{"canonical_record":{"source":{"id":"2502.00361","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-01T07:55:06Z","cross_cats_sorted":[],"title_canon_sha256":"4217daa5cc37f0bef8423dae899a16f7ec7a13cfb167b31d2958006b1e71d153","abstract_canon_sha256":"5621d3b5f3850a9557bb78aba5530ca7721a3fb0fdb9f071ec66ac128e84c48e"},"schema_version":"1.0"},"canonical_sha256":"b37643e88002f2dcb0b685a3ddeb640bb58222a966590f53821d285fcbec5b47","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:06.364887Z","signature_b64":"ci5cvlRCK7V9jDgevhq3815bFGWZfxRBdJNRrPR96nOgTyMtUC0U3VMPGJd7Emgp3g5KPXO7IEBMNO4IwlIBBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b37643e88002f2dcb0b685a3ddeb640bb58222a966590f53821d285fcbec5b47","last_reissued_at":"2026-07-05T11:29:06.364296Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:06.364296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.00361","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:29:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iTx8AJJIFtlB1mm6jy4lNUt6W9LEhzLHYR41i9d3jjvD+Mx2tKNFLlXhdzuFJ9XYgm5yAbf0n2juzI6z9YhbDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T07:42:32.312941Z"},"content_sha256":"a79ac86576eda3f71bb12e6656a27be7fe5c35729da358f5e8fb6ae25c90b3cd","schema_version":"1.0","event_id":"sha256:a79ac86576eda3f71bb12e6656a27be7fe5c35729da358f5e8fb6ae25c90b3cd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:WN3EH2EAALZNZMFWQWR5323EBO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Efficient Online Reinforcement Learning for Diffusion Policy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Dai, Haitong Ma, Kai Wang, Na Li, Tianyi Chen","submitted_at":"2025-02-01T07:55:06Z","abstract_excerpt":"Diffusion policies have achieved superior performance in imitation learning and offline reinforcement learning (RL) due to their rich expressiveness. However, the conventional diffusion training procedure requires samples from target distribution, which is impossible in online RL since we cannot sample from the optimal policy. Backpropagating policy gradient through the diffusion process incurs huge computational costs and instability, thus being expensive and not scalable. To enable efficient training of diffusion policies in online RL, we generalize the conventional denoising score matching "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00361","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00361/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:29:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gHJSSJHqOPV6RpTXEXNxDjadhNWcTs+q7A0EQ77szwUM9yjOFulcbx7hAQs4uy9eWBjyWb630IyHUgXPiJN7Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T07:42:32.313417Z"},"content_sha256":"9bf5fd05fbb38286e9115c74cf3bd817919023c7c85ed2625f5a61c840cb6c4f","schema_version":"1.0","event_id":"sha256:9bf5fd05fbb38286e9115c74cf3bd817919023c7c85ed2625f5a61c840cb6c4f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WN3EH2EAALZNZMFWQWR5323EBO/bundle.json","state_url":"https://pith.science/pith/WN3EH2EAALZNZMFWQWR5323EBO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WN3EH2EAALZNZMFWQWR5323EBO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T07:42:32Z","links":{"resolver":"https://pith.science/pith/WN3EH2EAALZNZMFWQWR5323EBO","bundle":"https://pith.science/pith/WN3EH2EAALZNZMFWQWR5323EBO/bundle.json","state":"https://pith.science/pith/WN3EH2EAALZNZMFWQWR5323EBO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WN3EH2EAALZNZMFWQWR5323EBO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:WN3EH2EAALZNZMFWQWR5323EBO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5621d3b5f3850a9557bb78aba5530ca7721a3fb0fdb9f071ec66ac128e84c48e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-01T07:55:06Z","title_canon_sha256":"4217daa5cc37f0bef8423dae899a16f7ec7a13cfb167b31d2958006b1e71d153"},"schema_version":"1.0","source":{"id":"2502.00361","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.00361","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"arxiv_version","alias_value":"2502.00361v4","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00361","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_12","alias_value":"WN3EH2EAALZN","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_16","alias_value":"WN3EH2EAALZNZMFW","created_at":"2026-07-05T11:29:06Z"},{"alias_kind":"pith_short_8","alias_value":"WN3EH2EA","created_at":"2026-07-05T11:29:06Z"}],"graph_snapshots":[{"event_id":"sha256:9bf5fd05fbb38286e9115c74cf3bd817919023c7c85ed2625f5a61c840cb6c4f","target":"graph","created_at":"2026-07-05T11:29:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.00361/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Diffusion policies have achieved superior performance in imitation learning and offline reinforcement learning (RL) due to their rich expressiveness. However, the conventional diffusion training procedure requires samples from target distribution, which is impossible in online RL since we cannot sample from the optimal policy. Backpropagating policy gradient through the diffusion process incurs huge computational costs and instability, thus being expensive and not scalable. To enable efficient training of diffusion policies in online RL, we generalize the conventional denoising score matching ","authors_text":"Bo Dai, Haitong Ma, Kai Wang, Na Li, Tianyi Chen","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-01T07:55:06Z","title":"Efficient Online Reinforcement Learning for Diffusion Policy"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00361","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a79ac86576eda3f71bb12e6656a27be7fe5c35729da358f5e8fb6ae25c90b3cd","target":"record","created_at":"2026-07-05T11:29:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5621d3b5f3850a9557bb78aba5530ca7721a3fb0fdb9f071ec66ac128e84c48e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-01T07:55:06Z","title_canon_sha256":"4217daa5cc37f0bef8423dae899a16f7ec7a13cfb167b31d2958006b1e71d153"},"schema_version":"1.0","source":{"id":"2502.00361","kind":"arxiv","version":4}},"canonical_sha256":"b37643e88002f2dcb0b685a3ddeb640bb58222a966590f53821d285fcbec5b47","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b37643e88002f2dcb0b685a3ddeb640bb58222a966590f53821d285fcbec5b47","first_computed_at":"2026-07-05T11:29:06.364296Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:29:06.364296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ci5cvlRCK7V9jDgevhq3815bFGWZfxRBdJNRrPR96nOgTyMtUC0U3VMPGJd7Emgp3g5KPXO7IEBMNO4IwlIBBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:29:06.364887Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.00361","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a79ac86576eda3f71bb12e6656a27be7fe5c35729da358f5e8fb6ae25c90b3cd","sha256:9bf5fd05fbb38286e9115c74cf3bd817919023c7c85ed2625f5a61c840cb6c4f"],"state_sha256":"997ea8c6a32413f3e93fad8c3edf5a7c25b577044f8200fd840eb9670f3e8850"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6s0UJ61ba9aATZ1BeCCvJ/uQ0q26eIz521TiU8+vCIzt6Yr0IXe/c+wzwTbA+aeuj0GsXic1ucRcv4nB64ksAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T07:42:32.316824Z","bundle_sha256":"7821dbefaaf2d00b4f080b3c5cb3c1f4b85df9d322c0c32cead5de0509d69bb3"}}