{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:UUS5F4LZZRQZWTFMJA4X3C53VW","short_pith_number":"pith:UUS5F4LZ","schema_version":"1.0","canonical_sha256":"a525d2f179cc619b4cac48397d8bbbadb266afac7c47a95230df4b622cbfd3ac","source":{"kind":"arxiv","id":"1711.00138","version":5},"attestation_state":"computed","paper":{"title":"Visualizing and Understanding Atari Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alan Fern, Anurag Koul, Jonathan Dodge, Sam Greydanus","submitted_at":"2017-10-31T23:03:17Z","abstract_excerpt":"While deep reinforcement learning (deep RL) agents are effective at maximizing rewards, it is often unclear what strategies they use to do so. In this paper, we take a step toward explaining deep RL agents through a case study using Atari 2600 environments. In particular, we focus on using saliency maps to understand how an agent learns and executes a policy. We introduce a method for generating useful saliency maps and use it to show 1) what strong agents attend to, 2) whether agents are making decisions for the right or wrong reasons, and 3) how agents evolve during learning. We also test ou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1711.00138","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-31T23:03:17Z","cross_cats_sorted":[],"title_canon_sha256":"9ca38b59837320715bbb374620a7e5dc0fa7da050e2c3caba0003867b12351c9","abstract_canon_sha256":"161490d60f5586ae63dfe10db392c79a18bcd4292cab00dd9f480a7f7a0ffa0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:06:05.810882Z","signature_b64":"RyKNi36iSK38TuiO7jaaZ7jrwljedfA9BkrDs3KpcLW5f8Jc2lYNCHTy34LsLzngCkaMzEAtF8q7VCN9UQcXAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a525d2f179cc619b4cac48397d8bbbadb266afac7c47a95230df4b622cbfd3ac","last_reissued_at":"2026-05-18T00:06:05.810269Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:06:05.810269Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visualizing and Understanding Atari Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alan Fern, Anurag Koul, Jonathan Dodge, Sam Greydanus","submitted_at":"2017-10-31T23:03:17Z","abstract_excerpt":"While deep reinforcement learning (deep RL) agents are effective at maximizing rewards, it is often unclear what strategies they use to do so. In this paper, we take a step toward explaining deep RL agents through a case study using Atari 2600 environments. In particular, we focus on using saliency maps to understand how an agent learns and executes a policy. We introduce a method for generating useful saliency maps and use it to show 1) what strong agents attend to, 2) whether agents are making decisions for the right or wrong reasons, and 3) how agents evolve during learning. We also test ou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1711.00138","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1711.00138","created_at":"2026-05-18T00:06:05.810386+00:00"},{"alias_kind":"arxiv_version","alias_value":"1711.00138v5","created_at":"2026-05-18T00:06:05.810386+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1711.00138","created_at":"2026-05-18T00:06:05.810386+00:00"},{"alias_kind":"pith_short_12","alias_value":"UUS5F4LZZRQZ","created_at":"2026-05-18T12:31:49.984773+00:00"},{"alias_kind":"pith_short_16","alias_value":"UUS5F4LZZRQZWTFM","created_at":"2026-05-18T12:31:49.984773+00:00"},{"alias_kind":"pith_short_8","alias_value":"UUS5F4LZ","created_at":"2026-05-18T12:31:49.984773+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2306.03310","citing_title":"LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW","json":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW.json","graph_json":"https://pith.science/api/pith-number/UUS5F4LZZRQZWTFMJA4X3C53VW/graph.json","events_json":"https://pith.science/api/pith-number/UUS5F4LZZRQZWTFMJA4X3C53VW/events.json","paper":"https://pith.science/paper/UUS5F4LZ"},"agent_actions":{"view_html":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW","download_json":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW.json","view_paper":"https://pith.science/paper/UUS5F4LZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1711.00138&json=true","fetch_graph":"https://pith.science/api/pith-number/UUS5F4LZZRQZWTFMJA4X3C53VW/graph.json","fetch_events":"https://pith.science/api/pith-number/UUS5F4LZZRQZWTFMJA4X3C53VW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW/action/storage_attestation","attest_author":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW/action/author_attestation","sign_citation":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW/action/citation_signature","submit_replication":"https://pith.science/pith/UUS5F4LZZRQZWTFMJA4X3C53VW/action/replication_record"}},"created_at":"2026-05-18T00:06:05.810386+00:00","updated_at":"2026-05-18T00:06:05.810386+00:00"}