{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TGLZL6ARGACO6AKA7L3LE2SR65","short_pith_number":"pith:TGLZL6AR","schema_version":"1.0","canonical_sha256":"999795f8113004ef0140faf6b26a51f7653bbf79b946c489544c13ff5aa6d9ad","source":{"kind":"arxiv","id":"2506.22401","version":1},"attestation_state":"computed","paper":{"title":"Exploration from a Primal-Dual Lens: Value-Incentivized Actor-Critic Methods for Sample-Efficient Online RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Bo Dai, Lin Xiao, Tong Yang, Yuejie Chi","submitted_at":"2025-06-27T17:18:43Z","abstract_excerpt":"Online reinforcement learning (RL) with complex function approximations such as transformers and deep neural networks plays a significant role in the modern practice of artificial intelligence. Despite its popularity and importance, balancing the fundamental trade-off between exploration and exploitation remains a long-standing challenge; in particular, we are still in lack of efficient and practical schemes that are backed by theoretical performance guarantees. Motivated by recent developments in exploration via optimistic regularization, this paper provides an interpretation of the principle"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.22401","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-27T17:18:43Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"2d4f1e6d0154b863c48c8cec373ac39f1d79a70884ed68c212e836f918d64ea7","abstract_canon_sha256":"5bf041922ee17f00b83c898aaef5e9d86fb7b3c3bc639fe0ea08e84efaf4306b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:26.878008Z","signature_b64":"q6F6Nbr8KZyjzZkKnVjuIILKQxzSmzH94PAX3+QJTNrkxKByFVz6Ew1FbNIJgy2hwiPnqzHhf2iD//2ER8mUBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"999795f8113004ef0140faf6b26a51f7653bbf79b946c489544c13ff5aa6d9ad","last_reissued_at":"2026-07-05T11:28:26.877509Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:26.877509Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploration from a Primal-Dual Lens: Value-Incentivized Actor-Critic Methods for Sample-Efficient Online RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Bo Dai, Lin Xiao, Tong Yang, Yuejie Chi","submitted_at":"2025-06-27T17:18:43Z","abstract_excerpt":"Online reinforcement learning (RL) with complex function approximations such as transformers and deep neural networks plays a significant role in the modern practice of artificial intelligence. Despite its popularity and importance, balancing the fundamental trade-off between exploration and exploitation remains a long-standing challenge; in particular, we are still in lack of efficient and practical schemes that are backed by theoretical performance guarantees. Motivated by recent developments in exploration via optimistic regularization, this paper provides an interpretation of the principle"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22401","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.22401","created_at":"2026-07-05T11:28:26.877574+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.22401v1","created_at":"2026-07-05T11:28:26.877574+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22401","created_at":"2026-07-05T11:28:26.877574+00:00"},{"alias_kind":"pith_short_12","alias_value":"TGLZL6ARGACO","created_at":"2026-07-05T11:28:26.877574+00:00"},{"alias_kind":"pith_short_16","alias_value":"TGLZL6ARGACO6AKA","created_at":"2026-07-05T11:28:26.877574+00:00"},{"alias_kind":"pith_short_8","alias_value":"TGLZL6AR","created_at":"2026-07-05T11:28:26.877574+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65","json":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65.json","graph_json":"https://pith.science/api/pith-number/TGLZL6ARGACO6AKA7L3LE2SR65/graph.json","events_json":"https://pith.science/api/pith-number/TGLZL6ARGACO6AKA7L3LE2SR65/events.json","paper":"https://pith.science/paper/TGLZL6AR"},"agent_actions":{"view_html":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65","download_json":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65.json","view_paper":"https://pith.science/paper/TGLZL6AR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.22401&json=true","fetch_graph":"https://pith.science/api/pith-number/TGLZL6ARGACO6AKA7L3LE2SR65/graph.json","fetch_events":"https://pith.science/api/pith-number/TGLZL6ARGACO6AKA7L3LE2SR65/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65/action/storage_attestation","attest_author":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65/action/author_attestation","sign_citation":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65/action/citation_signature","submit_replication":"https://pith.science/pith/TGLZL6ARGACO6AKA7L3LE2SR65/action/replication_record"}},"created_at":"2026-07-05T11:28:26.877574+00:00","updated_at":"2026-07-05T11:28:26.877574+00:00"}