{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:GHBXT2PRSA54FBBQ377Z66VGE3","short_pith_number":"pith:GHBXT2PR","canonical_record":{"source":{"id":"2504.18766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-26T02:12:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4eb74de54d537a9add550339ded858a4722c74af10fab50c04e30ba0721f38cb","abstract_canon_sha256":"02802c940a738cbbd6a11141abceb7c83f8c7fb0b90f6a67d960839218863db1"},"schema_version":"1.0"},"canonical_sha256":"31c379e9f1903bc28430dfff9f7aa626da9498d1cce48228116eb18d1683ab65","source":{"kind":"arxiv","id":"2504.18766","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.18766","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"arxiv_version","alias_value":"2504.18766v1","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18766","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_12","alias_value":"GHBXT2PRSA54","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_16","alias_value":"GHBXT2PRSA54FBBQ","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_8","alias_value":"GHBXT2PR","created_at":"2026-07-05T10:54:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:GHBXT2PRSA54FBBQ377Z66VGE3","target":"record","payload":{"canonical_record":{"source":{"id":"2504.18766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-26T02:12:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4eb74de54d537a9add550339ded858a4722c74af10fab50c04e30ba0721f38cb","abstract_canon_sha256":"02802c940a738cbbd6a11141abceb7c83f8c7fb0b90f6a67d960839218863db1"},"schema_version":"1.0"},"canonical_sha256":"31c379e9f1903bc28430dfff9f7aa626da9498d1cce48228116eb18d1683ab65","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:21.071606Z","signature_b64":"yo8nx0TQ4ZV/GYyjIiWJBO0Op5twQNlaaVW8SEFN4Gr02LcbC5drO8txEIY8eBsYM4YFeaRPo6eobncT77opAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31c379e9f1903bc28430dfff9f7aa626da9498d1cce48228116eb18d1683ab65","last_reissued_at":"2026-07-05T10:54:21.071129Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:21.071129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.18766","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iMUbqqoAwMGzF1faHyZJP7ODlR04O11fCtoCUKzItaNm4srf57d0RlMRi0hYDEr/+nn2CQbdFRMXyolGe0SMDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T13:57:00.702611Z"},"content_sha256":"2e48d2fa3e361bdd331a4157b348fbc2f755f8b0f391565c5884ff195ebd5bb9","schema_version":"1.0","event_id":"sha256:2e48d2fa3e361bdd331a4157b348fbc2f755f8b0f391565c5884ff195ebd5bb9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:GHBXT2PRSA54FBBQ377Z66VGE3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dynamic Action Interpolation: A Universal Approach for Accelerating Reinforcement Learning with Expert Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Wenjun Cao","submitted_at":"2025-04-26T02:12:02Z","abstract_excerpt":"Reinforcement learning (RL) suffers from severe sample inefficiency, especially during early training, requiring extensive environmental interactions to perform competently. Existing methods tend to solve this by incorporating prior knowledge, but introduce significant architectural and implementation complexity. We propose Dynamic Action Interpolation (DAI), a universal yet straightforward framework that interpolates expert and RL actions via a time-varying weight $\\alpha(t)$, integrating into any Actor-Critic algorithm with just a few lines of code and without auxiliary networks or additiona"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.18766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4Irh+EqSpxp+5QcA0XjV0NKfkohyRi5Uku/xrBr0C/QZ3Kq+wlqpt2JcjNLpAQj39mzIok1yhcohkhg55dFsDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T13:57:00.703533Z"},"content_sha256":"09c5fdcbae85f05cbb6182ea1dcbc2613544b9ab3d9c60a8ceba18f6777a1b5f","schema_version":"1.0","event_id":"sha256:09c5fdcbae85f05cbb6182ea1dcbc2613544b9ab3d9c60a8ceba18f6777a1b5f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GHBXT2PRSA54FBBQ377Z66VGE3/bundle.json","state_url":"https://pith.science/pith/GHBXT2PRSA54FBBQ377Z66VGE3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GHBXT2PRSA54FBBQ377Z66VGE3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T13:57:00Z","links":{"resolver":"https://pith.science/pith/GHBXT2PRSA54FBBQ377Z66VGE3","bundle":"https://pith.science/pith/GHBXT2PRSA54FBBQ377Z66VGE3/bundle.json","state":"https://pith.science/pith/GHBXT2PRSA54FBBQ377Z66VGE3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GHBXT2PRSA54FBBQ377Z66VGE3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:GHBXT2PRSA54FBBQ377Z66VGE3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"02802c940a738cbbd6a11141abceb7c83f8c7fb0b90f6a67d960839218863db1","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-26T02:12:02Z","title_canon_sha256":"4eb74de54d537a9add550339ded858a4722c74af10fab50c04e30ba0721f38cb"},"schema_version":"1.0","source":{"id":"2504.18766","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.18766","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"arxiv_version","alias_value":"2504.18766v1","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18766","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_12","alias_value":"GHBXT2PRSA54","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_16","alias_value":"GHBXT2PRSA54FBBQ","created_at":"2026-07-05T10:54:21Z"},{"alias_kind":"pith_short_8","alias_value":"GHBXT2PR","created_at":"2026-07-05T10:54:21Z"}],"graph_snapshots":[{"event_id":"sha256:09c5fdcbae85f05cbb6182ea1dcbc2613544b9ab3d9c60a8ceba18f6777a1b5f","target":"graph","created_at":"2026-07-05T10:54:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.18766/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) suffers from severe sample inefficiency, especially during early training, requiring extensive environmental interactions to perform competently. Existing methods tend to solve this by incorporating prior knowledge, but introduce significant architectural and implementation complexity. We propose Dynamic Action Interpolation (DAI), a universal yet straightforward framework that interpolates expert and RL actions via a time-varying weight $\\alpha(t)$, integrating into any Actor-Critic algorithm with just a few lines of code and without auxiliary networks or additiona","authors_text":"Wenjun Cao","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-26T02:12:02Z","title":"Dynamic Action Interpolation: A Universal Approach for Accelerating Reinforcement Learning with Expert Guidance"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18766","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2e48d2fa3e361bdd331a4157b348fbc2f755f8b0f391565c5884ff195ebd5bb9","target":"record","created_at":"2026-07-05T10:54:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"02802c940a738cbbd6a11141abceb7c83f8c7fb0b90f6a67d960839218863db1","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-26T02:12:02Z","title_canon_sha256":"4eb74de54d537a9add550339ded858a4722c74af10fab50c04e30ba0721f38cb"},"schema_version":"1.0","source":{"id":"2504.18766","kind":"arxiv","version":1}},"canonical_sha256":"31c379e9f1903bc28430dfff9f7aa626da9498d1cce48228116eb18d1683ab65","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"31c379e9f1903bc28430dfff9f7aa626da9498d1cce48228116eb18d1683ab65","first_computed_at":"2026-07-05T10:54:21.071129Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:54:21.071129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"yo8nx0TQ4ZV/GYyjIiWJBO0Op5twQNlaaVW8SEFN4Gr02LcbC5drO8txEIY8eBsYM4YFeaRPo6eobncT77opAA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:54:21.071606Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.18766","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2e48d2fa3e361bdd331a4157b348fbc2f755f8b0f391565c5884ff195ebd5bb9","sha256:09c5fdcbae85f05cbb6182ea1dcbc2613544b9ab3d9c60a8ceba18f6777a1b5f"],"state_sha256":"2fdf3eb95d27f94f6402048f1bdf0acb22187b563163303e502c3be365142365"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uFDZD4PZfPiXBbSAEERSSV8AJByYQMNY+EofdV7liZD7U/E04fCVr8ITRL0K4DqUlhMNiDy2rPvQjiJnkyy/Aw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T13:57:00.709297Z","bundle_sha256":"172dfd4f6515ab698d3363d1c6b761605c28d3823f1618e9836d5ccb203263e9"}}