{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:MOFGBBHGOJQETVG2EBWQNV7A6F","short_pith_number":"pith:MOFGBBHG","schema_version":"1.0","canonical_sha256":"638a6084e6726049d4da206d06d7e0f1794ff2dc642a51eca9d7afea4b8fb412","source":{"kind":"arxiv","id":"2112.04145","version":5},"attestation_state":"computed","paper":{"title":"A Review for Deep Reinforcement Learning in Atari:Benchmarks, Challenges, and Solutions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jiajun Fan","submitted_at":"2021-12-08T06:52:23Z","abstract_excerpt":"The Arcade Learning Environment (ALE) is proposed as an evaluation platform for empirically assessing the generality of agents across dozens of Atari 2600 games. ALE offers various challenging problems and has drawn significant attention from the deep reinforcement learning (RL) community. From Deep Q-Networks (DQN) to Agent57, RL agents seem to achieve superhuman performance in ALE. However, is this the case? In this paper, to explore this problem, we first review the current evaluation metrics in the Atari benchmarks and then reveal that the current evaluation criteria of achieving superhuma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.04145","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-12-08T06:52:23Z","cross_cats_sorted":[],"title_canon_sha256":"2e02310ac04e7ff7c399742f0c04db422dee887365d4254989d2691f06222b99","abstract_canon_sha256":"d6b18dc56bdf8d8bdf79c486c207b6641404cda3e8d7f13bce32228ae5640529"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:45:25.916738Z","signature_b64":"DYJiJWQYPOXuyVHJbOLj7Wn/qfnL9/sk1j9nOnw35HZAnMobXtLPGjQrhhikOiM3mBqxyzVNJElnVqRos51rAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"638a6084e6726049d4da206d06d7e0f1794ff2dc642a51eca9d7afea4b8fb412","last_reissued_at":"2026-07-05T05:45:25.916361Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:45:25.916361Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Review for Deep Reinforcement Learning in Atari:Benchmarks, Challenges, and Solutions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jiajun Fan","submitted_at":"2021-12-08T06:52:23Z","abstract_excerpt":"The Arcade Learning Environment (ALE) is proposed as an evaluation platform for empirically assessing the generality of agents across dozens of Atari 2600 games. ALE offers various challenging problems and has drawn significant attention from the deep reinforcement learning (RL) community. From Deep Q-Networks (DQN) to Agent57, RL agents seem to achieve superhuman performance in ALE. However, is this the case? In this paper, to explore this problem, we first review the current evaluation metrics in the Atari benchmarks and then reveal that the current evaluation criteria of achieving superhuma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.04145","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.04145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.04145","created_at":"2026-07-05T05:45:25.916419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.04145v5","created_at":"2026-07-05T05:45:25.916419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.04145","created_at":"2026-07-05T05:45:25.916419+00:00"},{"alias_kind":"pith_short_12","alias_value":"MOFGBBHGOJQE","created_at":"2026-07-05T05:45:25.916419+00:00"},{"alias_kind":"pith_short_16","alias_value":"MOFGBBHGOJQETVG2","created_at":"2026-07-05T05:45:25.916419+00:00"},{"alias_kind":"pith_short_8","alias_value":"MOFGBBHG","created_at":"2026-07-05T05:45:25.916419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.02831","citing_title":"Reinforcement Learning with Evolving Rubrics as Rewards for Audio Reasoning","ref_index":49,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F","json":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F.json","graph_json":"https://pith.science/api/pith-number/MOFGBBHGOJQETVG2EBWQNV7A6F/graph.json","events_json":"https://pith.science/api/pith-number/MOFGBBHGOJQETVG2EBWQNV7A6F/events.json","paper":"https://pith.science/paper/MOFGBBHG"},"agent_actions":{"view_html":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F","download_json":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F.json","view_paper":"https://pith.science/paper/MOFGBBHG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.04145&json=true","fetch_graph":"https://pith.science/api/pith-number/MOFGBBHGOJQETVG2EBWQNV7A6F/graph.json","fetch_events":"https://pith.science/api/pith-number/MOFGBBHGOJQETVG2EBWQNV7A6F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F/action/storage_attestation","attest_author":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F/action/author_attestation","sign_citation":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F/action/citation_signature","submit_replication":"https://pith.science/pith/MOFGBBHGOJQETVG2EBWQNV7A6F/action/replication_record"}},"created_at":"2026-07-05T05:45:25.916419+00:00","updated_at":"2026-07-05T05:45:25.916419+00:00"}