{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BDXRQVBWYF7PLWJ6G7ZI2UGCDI","short_pith_number":"pith:BDXRQVBW","schema_version":"1.0","canonical_sha256":"08ef185436c17ef5d93e37f28d50c21a22de25dbe8d08e6d013409dd0cf238ad","source":{"kind":"arxiv","id":"2409.01374","version":1},"attestation_state":"computed","paper":{"title":"H-ARC: A Robust Estimate of Human Performance on the Abstraction and Reasoning Corpus Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Brenden M. Lake, Solim LeGris, Todd M. Gureckis, Wai Keen Vong","submitted_at":"2024-09-02T17:11:32Z","abstract_excerpt":"The Abstraction and Reasoning Corpus (ARC) is a visual program synthesis benchmark designed to test challenging out-of-distribution generalization in humans and machines. Since 2019, limited progress has been observed on the challenge using existing artificial intelligence methods. Comparing human and machine performance is important for the validity of the benchmark. While previous work explored how well humans can solve tasks from the ARC benchmark, they either did so using only a subset of tasks from the original dataset, or from variants of ARC, and therefore only provided a tentative esti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.01374","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-09-02T17:11:32Z","cross_cats_sorted":[],"title_canon_sha256":"2f265cb60a83049fe09f037627351b4f1788a9001153cdcda19510cc9558ac36","abstract_canon_sha256":"776b2d48e4a4a18c5880b06026472aff20f1f925cbb57cb949a1548ecf0e4367"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:28.330180Z","signature_b64":"0jfUUsvuBvMZyT9oYq49BbATb3tSGOorTzy7InTTGkkx+no02BPSmP4j6EpzB3wzCgWxyyHtphCN/5hGRzGWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08ef185436c17ef5d93e37f28d50c21a22de25dbe8d08e6d013409dd0cf238ad","last_reissued_at":"2026-07-05T09:02:28.329683Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:28.329683Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"H-ARC: A Robust Estimate of Human Performance on the Abstraction and Reasoning Corpus Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Brenden M. Lake, Solim LeGris, Todd M. Gureckis, Wai Keen Vong","submitted_at":"2024-09-02T17:11:32Z","abstract_excerpt":"The Abstraction and Reasoning Corpus (ARC) is a visual program synthesis benchmark designed to test challenging out-of-distribution generalization in humans and machines. Since 2019, limited progress has been observed on the challenge using existing artificial intelligence methods. Comparing human and machine performance is important for the validity of the benchmark. While previous work explored how well humans can solve tasks from the ARC benchmark, they either did so using only a subset of tasks from the original dataset, or from variants of ARC, and therefore only provided a tentative esti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.01374","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.01374/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.01374","created_at":"2026-07-05T09:02:28.329746+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.01374v1","created_at":"2026-07-05T09:02:28.329746+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.01374","created_at":"2026-07-05T09:02:28.329746+00:00"},{"alias_kind":"pith_short_12","alias_value":"BDXRQVBWYF7P","created_at":"2026-07-05T09:02:28.329746+00:00"},{"alias_kind":"pith_short_16","alias_value":"BDXRQVBWYF7PLWJ6","created_at":"2026-07-05T09:02:28.329746+00:00"},{"alias_kind":"pith_short_8","alias_value":"BDXRQVBW","created_at":"2026-07-05T09:02:28.329746+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09026","citing_title":"Structural Grid Descriptors Predict Within-Task Solver Success on ARC-AGI","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07157","citing_title":"Think Fast: Estimating No-CoT Task-Completion Time Horizons of Frontier AI Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07157","citing_title":"Think Fast: Estimating No-CoT Task-Completion Time Horizons of Frontier AI Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2505.11831","citing_title":"ARC-AGI-2: A New Challenge for Frontier AI Reasoning Systems","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI","json":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI.json","graph_json":"https://pith.science/api/pith-number/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/graph.json","events_json":"https://pith.science/api/pith-number/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/events.json","paper":"https://pith.science/paper/BDXRQVBW"},"agent_actions":{"view_html":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI","download_json":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI.json","view_paper":"https://pith.science/paper/BDXRQVBW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.01374&json=true","fetch_graph":"https://pith.science/api/pith-number/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/graph.json","fetch_events":"https://pith.science/api/pith-number/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/action/storage_attestation","attest_author":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/action/author_attestation","sign_citation":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/action/citation_signature","submit_replication":"https://pith.science/pith/BDXRQVBWYF7PLWJ6G7ZI2UGCDI/action/replication_record"}},"created_at":"2026-07-05T09:02:28.329746+00:00","updated_at":"2026-07-05T09:02:28.329746+00:00"}