{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B4FGTPU3POPM5LZ43DOWNJOTV7","short_pith_number":"pith:B4FGTPU3","schema_version":"1.0","canonical_sha256":"0f0a69be9b7b9eceaf3cd8dd66a5d3afe3fe1f525b89c79abb1ce777aa637ec9","source":{"kind":"arxiv","id":"2505.15040","version":1},"attestation_state":"computed","paper":{"title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ivan Smirnov, Shangding Gu","submitted_at":"2025-05-21T02:49:25Z","abstract_excerpt":"Reinforcement learning (RL) has seen significant advancements through the application of various neural network architectures. In this study, we systematically investigate the performance of several neural networks in RL tasks, including Long Short-Term Memory (LSTM), Multi-Layer Perceptron (MLP), Mamba/Mamba-2, Transformer-XL, Gated Transformer-XL, and Gated Recurrent Unit (GRU). Through comprehensive evaluation across continuous control, discrete decision-making, and memory-based environments, we identify architecture-specific strengths and limitations. Our results reveal that: (1) MLPs exce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15040","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-21T02:49:25Z","cross_cats_sorted":[],"title_canon_sha256":"e5310b8e084a7af791a94ad5e8e681fa2984be57c7efc66c79e26caf6ec9a518","abstract_canon_sha256":"abe4bd07caf692474163b919f5ece5b2d1a29b7a4941c24a36e488102258be4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:39.623831Z","signature_b64":"gxhZ9fxJHi+I33O75hL+2CPMLgdybX4dCVX8VJ1e4i5Hiy9YF8z+JrOstGT6xKkj5unpehqWLYBMgkKLCgaCAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f0a69be9b7b9eceaf3cd8dd66a5d3afe3fe1f525b89c79abb1ce777aa637ec9","last_reissued_at":"2026-07-05T11:06:39.623385Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:39.623385Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ivan Smirnov, Shangding Gu","submitted_at":"2025-05-21T02:49:25Z","abstract_excerpt":"Reinforcement learning (RL) has seen significant advancements through the application of various neural network architectures. In this study, we systematically investigate the performance of several neural networks in RL tasks, including Long Short-Term Memory (LSTM), Multi-Layer Perceptron (MLP), Mamba/Mamba-2, Transformer-XL, Gated Transformer-XL, and Gated Recurrent Unit (GRU). Through comprehensive evaluation across continuous control, discrete decision-making, and memory-based environments, we identify architecture-specific strengths and limitations. Our results reveal that: (1) MLPs exce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15040","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15040","created_at":"2026-07-05T11:06:39.623442+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15040v1","created_at":"2026-07-05T11:06:39.623442+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15040","created_at":"2026-07-05T11:06:39.623442+00:00"},{"alias_kind":"pith_short_12","alias_value":"B4FGTPU3POPM","created_at":"2026-07-05T11:06:39.623442+00:00"},{"alias_kind":"pith_short_16","alias_value":"B4FGTPU3POPM5LZ4","created_at":"2026-07-05T11:06:39.623442+00:00"},{"alias_kind":"pith_short_8","alias_value":"B4FGTPU3","created_at":"2026-07-05T11:06:39.623442+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05111","citing_title":"Reward Structure Shapes the Interaction Between Episodic Exploration and Neural Memory in Reinforcement Learning","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7","json":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7.json","graph_json":"https://pith.science/api/pith-number/B4FGTPU3POPM5LZ43DOWNJOTV7/graph.json","events_json":"https://pith.science/api/pith-number/B4FGTPU3POPM5LZ43DOWNJOTV7/events.json","paper":"https://pith.science/paper/B4FGTPU3"},"agent_actions":{"view_html":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7","download_json":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7.json","view_paper":"https://pith.science/paper/B4FGTPU3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15040&json=true","fetch_graph":"https://pith.science/api/pith-number/B4FGTPU3POPM5LZ43DOWNJOTV7/graph.json","fetch_events":"https://pith.science/api/pith-number/B4FGTPU3POPM5LZ43DOWNJOTV7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7/action/storage_attestation","attest_author":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7/action/author_attestation","sign_citation":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7/action/citation_signature","submit_replication":"https://pith.science/pith/B4FGTPU3POPM5LZ43DOWNJOTV7/action/replication_record"}},"created_at":"2026-07-05T11:06:39.623442+00:00","updated_at":"2026-07-05T11:06:39.623442+00:00"}