{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ZYQ6RPKATPJFMIPZUTMGXXCDHA","short_pith_number":"pith:ZYQ6RPKA","schema_version":"1.0","canonical_sha256":"ce21e8bd409bd25621f9a4d86bdc43383974f7580d0da5d7f3bc8bdda9cdbbb7","source":{"kind":"arxiv","id":"2601.23075","version":2},"attestation_state":"computed","paper":{"title":"RN-D: Discretized Categorical Actors for On-Policy Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Jie Feng, Sicun Gao, Tao Wang, Yijiang Li, Yuanyuan Shi, Yuexin Bian","submitted_at":"2026-01-30T15:24:34Z","abstract_excerpt":"On-policy Reinforcement Learning (RL) remains a dominant paradigm for continuous control, yet standard implementations rely on Gaussian actors and relatively shallow MLP policies, often leading to brittle optimization when gradients are noisy, and policy updates must be conservative. In this paper, we revisit actor policy representation as a first-class design choice for on-policy RL. We study discretized categorical actors, which represent each action dimension as a distribution over discrete bins and induce a policy objective analogous to classification cross-entropy loss. Building on archit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.23075","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-01-30T15:24:34Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"a80b86566aea89bb5ffc33dad17b712c50dce5695ff2d935ffd7245b21cb6504","abstract_canon_sha256":"fff4327f0b5998563062bd46949b8fb9563badc9f3b572456d9bcb41dff731ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-25T00:18:12.593931Z","signature_b64":"B0WUp+rO5fLGPu3HxcVWZe1rkQhgczoNXlOd/7zF7nQlJF/giwEIll0xfFb/R8GFwZ/FEsvVnGIap/65QxGnDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce21e8bd409bd25621f9a4d86bdc43383974f7580d0da5d7f3bc8bdda9cdbbb7","last_reissued_at":"2026-06-25T00:18:12.593405Z","signature_status":"signed_v1","first_computed_at":"2026-06-25T00:18:12.593405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RN-D: Discretized Categorical Actors for On-Policy Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Jie Feng, Sicun Gao, Tao Wang, Yijiang Li, Yuanyuan Shi, Yuexin Bian","submitted_at":"2026-01-30T15:24:34Z","abstract_excerpt":"On-policy Reinforcement Learning (RL) remains a dominant paradigm for continuous control, yet standard implementations rely on Gaussian actors and relatively shallow MLP policies, often leading to brittle optimization when gradients are noisy, and policy updates must be conservative. In this paper, we revisit actor policy representation as a first-class design choice for on-policy RL. We study discretized categorical actors, which represent each action dimension as a distribution over discrete bins and induce a policy objective analogous to classification cross-entropy loss. Building on archit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.23075","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.23075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.23075","created_at":"2026-06-25T00:18:12.593490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.23075v2","created_at":"2026-06-25T00:18:12.593490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.23075","created_at":"2026-06-25T00:18:12.593490+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYQ6RPKATPJF","created_at":"2026-06-25T00:18:12.593490+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYQ6RPKATPJFMIPZ","created_at":"2026-06-25T00:18:12.593490+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYQ6RPKA","created_at":"2026-06-25T00:18:12.593490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18978","citing_title":"Low-Rank Adaptation for Critic Learning in Off-Policy Reinforcement Learning","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA","json":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA.json","graph_json":"https://pith.science/api/pith-number/ZYQ6RPKATPJFMIPZUTMGXXCDHA/graph.json","events_json":"https://pith.science/api/pith-number/ZYQ6RPKATPJFMIPZUTMGXXCDHA/events.json","paper":"https://pith.science/paper/ZYQ6RPKA"},"agent_actions":{"view_html":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA","download_json":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA.json","view_paper":"https://pith.science/paper/ZYQ6RPKA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.23075&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYQ6RPKATPJFMIPZUTMGXXCDHA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYQ6RPKATPJFMIPZUTMGXXCDHA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA/action/storage_attestation","attest_author":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA/action/author_attestation","sign_citation":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA/action/citation_signature","submit_replication":"https://pith.science/pith/ZYQ6RPKATPJFMIPZUTMGXXCDHA/action/replication_record"}},"created_at":"2026-06-25T00:18:12.593490+00:00","updated_at":"2026-06-25T00:18:12.593490+00:00"}