{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AS447S6FR4MPJVAJJ2TIP4PTXK","short_pith_number":"pith:AS447S6F","schema_version":"1.0","canonical_sha256":"04b9cfcbc58f18f4d4094ea687f1f3baaca07a6f14f49316152f4fdd828ce1dd","source":{"kind":"arxiv","id":"2409.16764","version":2},"attestation_state":"computed","paper":{"title":"Offline and Distributional Reinforcement Learning for Radio Resource Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Eslam Eldeeb, Hirley Alves","submitted_at":"2024-09-25T09:22:23Z","abstract_excerpt":"Reinforcement learning (RL) has proved to have a promising role in future intelligent wireless networks. Online RL has been adopted for radio resource management (RRM), taking over traditional schemes. However, due to its reliance on online interaction with the environment, its role becomes limited in practical, real-world problems where online interaction is not feasible. In addition, traditional RL stands short in front of the uncertainties and risks in real-world stochastic environments. In this manner, we propose an offline and distributional RL scheme for the RRM problem, enabling offline"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16764","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-25T09:22:23Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"80d008e0ce4283cc355793aa866ec385a7836b02449be0b9c513da728684fc2e","abstract_canon_sha256":"b7e05aaeba2156514ea862c2c2cc4bf4b2d0d205d21361ea456d9e9737482130"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:17.755261Z","signature_b64":"mvIgFtRxmuhLtx1TWFqSRibvysDNm33qfSYA7VSaeLCtCxyVv60V3w5WrcEFYtNGpphOC/q2AJnbeQcLcOVrDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04b9cfcbc58f18f4d4094ea687f1f3baaca07a6f14f49316152f4fdd828ce1dd","last_reissued_at":"2026-07-05T10:04:17.754752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:17.754752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline and Distributional Reinforcement Learning for Radio Resource Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Eslam Eldeeb, Hirley Alves","submitted_at":"2024-09-25T09:22:23Z","abstract_excerpt":"Reinforcement learning (RL) has proved to have a promising role in future intelligent wireless networks. Online RL has been adopted for radio resource management (RRM), taking over traditional schemes. However, due to its reliance on online interaction with the environment, its role becomes limited in practical, real-world problems where online interaction is not feasible. In addition, traditional RL stands short in front of the uncertainties and risks in real-world stochastic environments. In this manner, we propose an offline and distributional RL scheme for the RRM problem, enabling offline"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16764","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16764","created_at":"2026-07-05T10:04:17.754808+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16764v2","created_at":"2026-07-05T10:04:17.754808+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16764","created_at":"2026-07-05T10:04:17.754808+00:00"},{"alias_kind":"pith_short_12","alias_value":"AS447S6FR4MP","created_at":"2026-07-05T10:04:17.754808+00:00"},{"alias_kind":"pith_short_16","alias_value":"AS447S6FR4MPJVAJ","created_at":"2026-07-05T10:04:17.754808+00:00"},{"alias_kind":"pith_short_8","alias_value":"AS447S6F","created_at":"2026-07-05T10:04:17.754808+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01268","citing_title":"Resilient UAV Trajectory Planning via Few-Shot Meta-Offline Reinforcement Learning","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK","json":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK.json","graph_json":"https://pith.science/api/pith-number/AS447S6FR4MPJVAJJ2TIP4PTXK/graph.json","events_json":"https://pith.science/api/pith-number/AS447S6FR4MPJVAJJ2TIP4PTXK/events.json","paper":"https://pith.science/paper/AS447S6F"},"agent_actions":{"view_html":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK","download_json":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK.json","view_paper":"https://pith.science/paper/AS447S6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16764&json=true","fetch_graph":"https://pith.science/api/pith-number/AS447S6FR4MPJVAJJ2TIP4PTXK/graph.json","fetch_events":"https://pith.science/api/pith-number/AS447S6FR4MPJVAJJ2TIP4PTXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK/action/storage_attestation","attest_author":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK/action/author_attestation","sign_citation":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK/action/citation_signature","submit_replication":"https://pith.science/pith/AS447S6FR4MPJVAJJ2TIP4PTXK/action/replication_record"}},"created_at":"2026-07-05T10:04:17.754808+00:00","updated_at":"2026-07-05T10:04:17.754808+00:00"}