{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:AN373RUL4D2KU4CCVL6OJZOP6C","short_pith_number":"pith:AN373RUL","schema_version":"1.0","canonical_sha256":"0377fdc68be0f4aa7042aafce4e5cff08348ef2686b3c6aee90efec70c6f069a","source":{"kind":"arxiv","id":"2109.06780","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking the Spectrum of Agent Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Danijar Hafner","submitted_at":"2021-09-14T15:49:31Z","abstract_excerpt":"Evaluating the general abilities of intelligent agents requires complex simulation environments. Existing benchmarks typically evaluate only one narrow task per environment, requiring researchers to perform expensive training runs on many different environments. We introduce Crafter, an open world survival game with visual inputs that evaluates a wide range of general abilities within a single environment. Agents either learn from the provided reward signal or through intrinsic objectives and are evaluated by semantically meaningful achievements that can be unlocked during each episode, such a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.06780","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-09-14T15:49:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"108865ccfaa59902bd78cfde70fe607ee9261e540ee69ce25a75d4ee382204d9","abstract_canon_sha256":"bd6e08db9fdb17df87089c3f8913e6e9f0c35eb528cc397395a2edf44a731607"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:56:16.194809Z","signature_b64":"DVFmu+N0kEP1NKpEl+sZZzxIFD59qQNsLS1Tsb9K8bPuK8k3ImMHnMro6Txkli0WXxsabwX3O/RlQvJyq+uxDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0377fdc68be0f4aa7042aafce4e5cff08348ef2686b3c6aee90efec70c6f069a","last_reissued_at":"2026-07-05T03:56:16.194381Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:56:16.194381Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking the Spectrum of Agent Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Danijar Hafner","submitted_at":"2021-09-14T15:49:31Z","abstract_excerpt":"Evaluating the general abilities of intelligent agents requires complex simulation environments. Existing benchmarks typically evaluate only one narrow task per environment, requiring researchers to perform expensive training runs on many different environments. We introduce Crafter, an open world survival game with visual inputs that evaluates a wide range of general abilities within a single environment. Agents either learn from the provided reward signal or through intrinsic objectives and are evaluated by semantically meaningful achievements that can be unlocked during each episode, such a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.06780","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.06780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.06780","created_at":"2026-07-05T03:56:16.194434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.06780v2","created_at":"2026-07-05T03:56:16.194434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.06780","created_at":"2026-07-05T03:56:16.194434+00:00"},{"alias_kind":"pith_short_12","alias_value":"AN373RUL4D2K","created_at":"2026-07-05T03:56:16.194434+00:00"},{"alias_kind":"pith_short_16","alias_value":"AN373RUL4D2KU4CC","created_at":"2026-07-05T03:56:16.194434+00:00"},{"alias_kind":"pith_short_8","alias_value":"AN373RUL","created_at":"2026-07-05T03:56:16.194434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12384","citing_title":"APPO: Agentic Procedural Policy Optimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01224","citing_title":"AutoMem: Automated Learning of Memory as a Cognitive Skill","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04130","citing_title":"CLAW: Learning Continuous Latent Action World Models via Adversarial Latent Regularization","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27879","citing_title":"Towards Faithful Agentic XAI: A Verification Method and an Open-World Benchmark for Better Model Faithfulness","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23551","citing_title":"Goal-Conditioned Agents that Learn Everything All at Once","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19503","citing_title":"ARC-RL: A Reinforcement Learning Playground Inspired by ARC Raiders","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19503","citing_title":"ARC-RL: A Reinforcement Learning Playground Inspired by ARC Raiders","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07083","citing_title":"Dreamer-CDP: Improving Reconstruction-free World Models Via Continuous Deterministic Representation Prediction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13918","citing_title":"CA2: Code-Aware Agent for Automated Game Testing","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10064","citing_title":"MAGE: Multi-Agent Self-Evolution with Co-Evolutionary Knowledge Graphs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01358","citing_title":"PACE: Parameter Change for Unsupervised Environment Design","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01950","citing_title":"TRAP: Tail-aware Ranking Attack for World-Model Planning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2301.04104","citing_title":"Mastering Diverse Domains through World Models","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C","json":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C.json","graph_json":"https://pith.science/api/pith-number/AN373RUL4D2KU4CCVL6OJZOP6C/graph.json","events_json":"https://pith.science/api/pith-number/AN373RUL4D2KU4CCVL6OJZOP6C/events.json","paper":"https://pith.science/paper/AN373RUL"},"agent_actions":{"view_html":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C","download_json":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C.json","view_paper":"https://pith.science/paper/AN373RUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.06780&json=true","fetch_graph":"https://pith.science/api/pith-number/AN373RUL4D2KU4CCVL6OJZOP6C/graph.json","fetch_events":"https://pith.science/api/pith-number/AN373RUL4D2KU4CCVL6OJZOP6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C/action/storage_attestation","attest_author":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C/action/author_attestation","sign_citation":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C/action/citation_signature","submit_replication":"https://pith.science/pith/AN373RUL4D2KU4CCVL6OJZOP6C/action/replication_record"}},"created_at":"2026-07-05T03:56:16.194434+00:00","updated_at":"2026-07-05T03:56:16.194434+00:00"}