{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:2FQELIDNB3TMTRK7FRLFCB6CWD","short_pith_number":"pith:2FQELIDN","canonical_record":{"source":{"id":"2506.05968","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T10:46:20Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"e8d254704fee5adc0ec1459fbad12152fa154d468ce8efd6558c1bd23f517d46","abstract_canon_sha256":"6cb937935f7a235fa668619d758f7de597297351a863bc3ef1bcf9a1e3417b01"},"schema_version":"1.0"},"canonical_sha256":"d16045a06d0ee6c9c55f2c565107c2b0e16c49f1712e216629656b1358beab10","source":{"kind":"arxiv","id":"2506.05968","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.05968","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"arxiv_version","alias_value":"2506.05968v2","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05968","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_12","alias_value":"2FQELIDNB3TM","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_16","alias_value":"2FQELIDNB3TMTRK7","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_8","alias_value":"2FQELIDN","created_at":"2026-07-05T11:53:04Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:2FQELIDNB3TMTRK7FRLFCB6CWD","target":"record","payload":{"canonical_record":{"source":{"id":"2506.05968","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T10:46:20Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"e8d254704fee5adc0ec1459fbad12152fa154d468ce8efd6558c1bd23f517d46","abstract_canon_sha256":"6cb937935f7a235fa668619d758f7de597297351a863bc3ef1bcf9a1e3417b01"},"schema_version":"1.0"},"canonical_sha256":"d16045a06d0ee6c9c55f2c565107c2b0e16c49f1712e216629656b1358beab10","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:04.123339Z","signature_b64":"JxnVrxPkUCgQhrNfFtqPMWEPUiV2yYx++lh+1cOir5VWcocleF4n9Po2180dhT6unq62USPCMM6InsknQHNxAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d16045a06d0ee6c9c55f2c565107c2b0e16c49f1712e216629656b1358beab10","last_reissued_at":"2026-07-05T11:53:04.122898Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:04.122898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.05968","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:53:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JDEnopD+4/4Wd4IHcl9ydvb3aAaH/MyRvItkQR9J+Jx2De5hJ3SX1Hiu25otrBW+JFWZFKmCN/Pm/kMgpRmKDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T23:46:25.276670Z"},"content_sha256":"a3babd7cde5adf2f8e8ab97860b9befb2b2733b0361d9536d6f3094300bb4207","schema_version":"1.0","event_id":"sha256:a3babd7cde5adf2f8e8ab97860b9befb2b2733b0361d9536d6f3094300bb4207"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:2FQELIDNB3TMTRK7FRLFCB6CWD","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Gradual Transition from Bellman Optimality Operator to Bellman Operator in Online Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Kazuki Ota, Motoki Omura, Takayuki Osa, Tatsuya Harada, Yusuke Mukuta","submitted_at":"2025-06-06T10:46:20Z","abstract_excerpt":"For continuous action spaces, actor-critic methods are widely used in online reinforcement learning (RL). However, unlike RL algorithms for discrete actions, which generally model the optimal value function using the Bellman optimality operator, RL algorithms for continuous actions typically model Q-values for the current policy using the Bellman operator. These algorithms for continuous actions rely exclusively on policy updates for improvement, which often results in low sample efficiency. This study examines the effectiveness of incorporating the Bellman optimality operator into actor-criti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.05968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:53:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9ys0YQWIeLDkRnYps9mtnY792uQ5VP26j5X8dlgemE+GqCZZfTXiguL5nPVaEYkpIRf5Tms20RA866xIcxaJBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T23:46:25.277194Z"},"content_sha256":"0a06c6b40bffdd6ce35139ce02aeb6c5ca8d4efc77586e34d1c0c0ff9d18b08d","schema_version":"1.0","event_id":"sha256:0a06c6b40bffdd6ce35139ce02aeb6c5ca8d4efc77586e34d1c0c0ff9d18b08d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/bundle.json","state_url":"https://pith.science/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T23:46:25Z","links":{"resolver":"https://pith.science/pith/2FQELIDNB3TMTRK7FRLFCB6CWD","bundle":"https://pith.science/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/bundle.json","state":"https://pith.science/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2FQELIDNB3TMTRK7FRLFCB6CWD/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:2FQELIDNB3TMTRK7FRLFCB6CWD","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6cb937935f7a235fa668619d758f7de597297351a863bc3ef1bcf9a1e3417b01","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T10:46:20Z","title_canon_sha256":"e8d254704fee5adc0ec1459fbad12152fa154d468ce8efd6558c1bd23f517d46"},"schema_version":"1.0","source":{"id":"2506.05968","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.05968","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"arxiv_version","alias_value":"2506.05968v2","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05968","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_12","alias_value":"2FQELIDNB3TM","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_16","alias_value":"2FQELIDNB3TMTRK7","created_at":"2026-07-05T11:53:04Z"},{"alias_kind":"pith_short_8","alias_value":"2FQELIDN","created_at":"2026-07-05T11:53:04Z"}],"graph_snapshots":[{"event_id":"sha256:0a06c6b40bffdd6ce35139ce02aeb6c5ca8d4efc77586e34d1c0c0ff9d18b08d","target":"graph","created_at":"2026-07-05T11:53:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.05968/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"For continuous action spaces, actor-critic methods are widely used in online reinforcement learning (RL). However, unlike RL algorithms for discrete actions, which generally model the optimal value function using the Bellman optimality operator, RL algorithms for continuous actions typically model Q-values for the current policy using the Bellman operator. These algorithms for continuous actions rely exclusively on policy updates for improvement, which often results in low sample efficiency. This study examines the effectiveness of incorporating the Bellman optimality operator into actor-criti","authors_text":"Kazuki Ota, Motoki Omura, Takayuki Osa, Tatsuya Harada, Yusuke Mukuta","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T10:46:20Z","title":"Gradual Transition from Bellman Optimality Operator to Bellman Operator in Online Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05968","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a3babd7cde5adf2f8e8ab97860b9befb2b2733b0361d9536d6f3094300bb4207","target":"record","created_at":"2026-07-05T11:53:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6cb937935f7a235fa668619d758f7de597297351a863bc3ef1bcf9a1e3417b01","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T10:46:20Z","title_canon_sha256":"e8d254704fee5adc0ec1459fbad12152fa154d468ce8efd6558c1bd23f517d46"},"schema_version":"1.0","source":{"id":"2506.05968","kind":"arxiv","version":2}},"canonical_sha256":"d16045a06d0ee6c9c55f2c565107c2b0e16c49f1712e216629656b1358beab10","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d16045a06d0ee6c9c55f2c565107c2b0e16c49f1712e216629656b1358beab10","first_computed_at":"2026-07-05T11:53:04.122898Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:53:04.122898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"JxnVrxPkUCgQhrNfFtqPMWEPUiV2yYx++lh+1cOir5VWcocleF4n9Po2180dhT6unq62USPCMM6InsknQHNxAg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:53:04.123339Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.05968","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a3babd7cde5adf2f8e8ab97860b9befb2b2733b0361d9536d6f3094300bb4207","sha256:0a06c6b40bffdd6ce35139ce02aeb6c5ca8d4efc77586e34d1c0c0ff9d18b08d"],"state_sha256":"d0fd9f101d457d82f8fe712c228e191a2a1d96827634b678a6473a8785a32f46"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vkhKYHF4QnxaBqUxuQ8vIxFRAfIVRn2vy2FAenqV+zh6jKxdPZJffofkp9Xp9BQ01WbBgbLDOvfU9foE0OSjDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T23:46:25.281691Z","bundle_sha256":"c6e09db3284b85f796ad3a9f6694f01a2562d8b54540930dd897b1cae00b0e48"}}