{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SSYCKOKQWRDXBGEH5YONR4YGJF","short_pith_number":"pith:SSYCKOKQ","schema_version":"1.0","canonical_sha256":"94b0253950b447709887ee1cd8f3064950f9bdd6bb64e8f36d0ce0850662c869","source":{"kind":"arxiv","id":"2501.06700","version":1},"attestation_state":"computed","paper":{"title":"Average Reward Reinforcement Learning for Wireless Radio Resource Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NI","eess.SP","math.IT"],"primary_cat":"cs.IT","authors_text":"Cong Shen, Jing Yang, Kun Yang","submitted_at":"2025-01-12T03:45:14Z","abstract_excerpt":"In this paper, we address a crucial but often overlooked issue in applying reinforcement learning (RL) to radio resource management (RRM) in wireless communications: the mismatch between the discounted reward RL formulation and the undiscounted goal of wireless network optimization. To the best of our knowledge, we are the first to systematically investigate this discrepancy, starting with a discussion of the problem formulation followed by simulations that quantify the extent of the gap. To bridge this gap, we introduce the use of average reward RL, a method that aligns more closely with the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.06700","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IT","submitted_at":"2025-01-12T03:45:14Z","cross_cats_sorted":["cs.LG","cs.NI","eess.SP","math.IT"],"title_canon_sha256":"43fa5121e1eac99fef60dd9dca0176bc3df3714148d554b7f4f313ec150ba24d","abstract_canon_sha256":"243b5247ec37296b922700da0b48aa9de04e5a7000f38544b6bcb306c66e6212"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:08.803953Z","signature_b64":"c7WoEGrZaMYwzSJSHFOe6VaQgfZuwuHVbhddeQcuMIWR2YLt2WUFMR6ShKpeB3Ga+cNkfITuYPUGtOSeUFlZDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94b0253950b447709887ee1cd8f3064950f9bdd6bb64e8f36d0ce0850662c869","last_reissued_at":"2026-07-05T10:00:08.803546Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:08.803546Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Average Reward Reinforcement Learning for Wireless Radio Resource Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NI","eess.SP","math.IT"],"primary_cat":"cs.IT","authors_text":"Cong Shen, Jing Yang, Kun Yang","submitted_at":"2025-01-12T03:45:14Z","abstract_excerpt":"In this paper, we address a crucial but often overlooked issue in applying reinforcement learning (RL) to radio resource management (RRM) in wireless communications: the mismatch between the discounted reward RL formulation and the undiscounted goal of wireless network optimization. To the best of our knowledge, we are the first to systematically investigate this discrepancy, starting with a discussion of the problem formulation followed by simulations that quantify the extent of the gap. To bridge this gap, we introduce the use of average reward RL, a method that aligns more closely with the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06700","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06700/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.06700","created_at":"2026-07-05T10:00:08.803601+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.06700v1","created_at":"2026-07-05T10:00:08.803601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06700","created_at":"2026-07-05T10:00:08.803601+00:00"},{"alias_kind":"pith_short_12","alias_value":"SSYCKOKQWRDX","created_at":"2026-07-05T10:00:08.803601+00:00"},{"alias_kind":"pith_short_16","alias_value":"SSYCKOKQWRDXBGEH","created_at":"2026-07-05T10:00:08.803601+00:00"},{"alias_kind":"pith_short_8","alias_value":"SSYCKOKQ","created_at":"2026-07-05T10:00:08.803601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF","json":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF.json","graph_json":"https://pith.science/api/pith-number/SSYCKOKQWRDXBGEH5YONR4YGJF/graph.json","events_json":"https://pith.science/api/pith-number/SSYCKOKQWRDXBGEH5YONR4YGJF/events.json","paper":"https://pith.science/paper/SSYCKOKQ"},"agent_actions":{"view_html":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF","download_json":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF.json","view_paper":"https://pith.science/paper/SSYCKOKQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.06700&json=true","fetch_graph":"https://pith.science/api/pith-number/SSYCKOKQWRDXBGEH5YONR4YGJF/graph.json","fetch_events":"https://pith.science/api/pith-number/SSYCKOKQWRDXBGEH5YONR4YGJF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF/action/storage_attestation","attest_author":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF/action/author_attestation","sign_citation":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF/action/citation_signature","submit_replication":"https://pith.science/pith/SSYCKOKQWRDXBGEH5YONR4YGJF/action/replication_record"}},"created_at":"2026-07-05T10:00:08.803601+00:00","updated_at":"2026-07-05T10:00:08.803601+00:00"}