{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:R46QZ374TX7X5PDIQN34MRNVXK","short_pith_number":"pith:R46QZ374","schema_version":"1.0","canonical_sha256":"8f3d0ceffc9dff7ebc688377c645b5baa43267f08b0d41bb0a97e2c86ec4d0c5","source":{"kind":"arxiv","id":"2002.03534","version":2},"attestation_state":"computed","paper":{"title":"Discrete Action On-Policy Learning with Action-Value Critic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Mingyuan Zhou, Mingzhang Yin, Yuguang Yue, Yunhao Tang","submitted_at":"2020-02-10T04:23:09Z","abstract_excerpt":"Reinforcement learning (RL) in discrete action space is ubiquitous in real-world applications, but its complexity grows exponentially with the action-space dimension, making it challenging to apply existing on-policy gradient based deep RL algorithms efficiently. To effectively operate in multidimensional discrete action spaces, we construct a critic to estimate action-value functions, apply it on correlated actions, and combine these critic estimated action values to control the variance of gradient estimation. We follow rigorous statistical analysis to design how to generate and combine thes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.03534","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2020-02-10T04:23:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7ea8a0dba44efae34f2114093f5e971168d77ef984cb930bab1b669834a18831","abstract_canon_sha256":"8c85541ada7a6a060ca341dcb353d2f5c91b8863930432968722a94b6753576b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:42:53.912624Z","signature_b64":"NyvH37ouav9d0ESdEyINxW7TVR4eZUeY7FgjL5Vh0lzQK90LmbfB0Pu9SoiLN8RR4ltvaXRBzG/VpaLFgtWKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f3d0ceffc9dff7ebc688377c645b5baa43267f08b0d41bb0a97e2c86ec4d0c5","last_reissued_at":"2026-07-05T00:42:53.912066Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:42:53.912066Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Discrete Action On-Policy Learning with Action-Value Critic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Mingyuan Zhou, Mingzhang Yin, Yuguang Yue, Yunhao Tang","submitted_at":"2020-02-10T04:23:09Z","abstract_excerpt":"Reinforcement learning (RL) in discrete action space is ubiquitous in real-world applications, but its complexity grows exponentially with the action-space dimension, making it challenging to apply existing on-policy gradient based deep RL algorithms efficiently. To effectively operate in multidimensional discrete action spaces, we construct a critic to estimate action-value functions, apply it on correlated actions, and combine these critic estimated action values to control the variance of gradient estimation. We follow rigorous statistical analysis to design how to generate and combine thes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.03534","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.03534/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.03534","created_at":"2026-07-05T00:42:53.912129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.03534v2","created_at":"2026-07-05T00:42:53.912129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.03534","created_at":"2026-07-05T00:42:53.912129+00:00"},{"alias_kind":"pith_short_12","alias_value":"R46QZ374TX7X","created_at":"2026-07-05T00:42:53.912129+00:00"},{"alias_kind":"pith_short_16","alias_value":"R46QZ374TX7X5PDI","created_at":"2026-07-05T00:42:53.912129+00:00"},{"alias_kind":"pith_short_8","alias_value":"R46QZ374","created_at":"2026-07-05T00:42:53.912129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK","json":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK.json","graph_json":"https://pith.science/api/pith-number/R46QZ374TX7X5PDIQN34MRNVXK/graph.json","events_json":"https://pith.science/api/pith-number/R46QZ374TX7X5PDIQN34MRNVXK/events.json","paper":"https://pith.science/paper/R46QZ374"},"agent_actions":{"view_html":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK","download_json":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK.json","view_paper":"https://pith.science/paper/R46QZ374","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.03534&json=true","fetch_graph":"https://pith.science/api/pith-number/R46QZ374TX7X5PDIQN34MRNVXK/graph.json","fetch_events":"https://pith.science/api/pith-number/R46QZ374TX7X5PDIQN34MRNVXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK/action/storage_attestation","attest_author":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK/action/author_attestation","sign_citation":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK/action/citation_signature","submit_replication":"https://pith.science/pith/R46QZ374TX7X5PDIQN34MRNVXK/action/replication_record"}},"created_at":"2026-07-05T00:42:53.912129+00:00","updated_at":"2026-07-05T00:42:53.912129+00:00"}