{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3RMWKVXQEVV5NEAOME6XBVPNYE","short_pith_number":"pith:3RMWKVXQ","schema_version":"1.0","canonical_sha256":"dc596556f0256bd6900e613d70d5edc116519843d24199e9ae8dbfd509ef5230","source":{"kind":"arxiv","id":"2105.07197","version":2},"attestation_state":"computed","paper":{"title":"Are Convolutional Neural Networks or Transformers more like human vision?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Erin Grant, Ishita Dasgupta, Shikhar Tuli, Thomas L. Griffiths","submitted_at":"2021-05-15T10:33:35Z","abstract_excerpt":"Modern machine learning models for computer vision exceed humans in accuracy on specific visual recognition tasks, notably on datasets like ImageNet. However, high accuracy can be achieved in many ways. The particular decision function found by a machine learning system is determined not only by the data to which the system is exposed, but also the inductive biases of the model, which are typically harder to characterize. In this work, we follow a recent trend of in-depth behavioral analyses of neural network models that go beyond accuracy as an evaluation metric by looking at patterns of erro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.07197","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-05-15T10:33:35Z","cross_cats_sorted":[],"title_canon_sha256":"8b1a1e8ee487a759920f2c8325861da64373dd7bdd9dd3bcbac352cd5f4ee28d","abstract_canon_sha256":"fe8a428aa0634e5e21a0640cadeb2bec3b996986a596e2b626b3aea70da6ed8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:54:17.883848Z","signature_b64":"8ifZYDzUMYDZXmGvhhCiNuGbJQ2seGu+VEqveJqWKJZTVowPhu7We6vASco84pb+FvglN+U0wsQ7tYXfOIGGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc596556f0256bd6900e613d70d5edc116519843d24199e9ae8dbfd509ef5230","last_reissued_at":"2026-07-05T02:54:17.883422Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:54:17.883422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Convolutional Neural Networks or Transformers more like human vision?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Erin Grant, Ishita Dasgupta, Shikhar Tuli, Thomas L. Griffiths","submitted_at":"2021-05-15T10:33:35Z","abstract_excerpt":"Modern machine learning models for computer vision exceed humans in accuracy on specific visual recognition tasks, notably on datasets like ImageNet. However, high accuracy can be achieved in many ways. The particular decision function found by a machine learning system is determined not only by the data to which the system is exposed, but also the inductive biases of the model, which are typically harder to characterize. In this work, we follow a recent trend of in-depth behavioral analyses of neural network models that go beyond accuracy as an evaluation metric by looking at patterns of erro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.07197","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.07197/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.07197","created_at":"2026-07-05T02:54:17.883478+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.07197v2","created_at":"2026-07-05T02:54:17.883478+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.07197","created_at":"2026-07-05T02:54:17.883478+00:00"},{"alias_kind":"pith_short_12","alias_value":"3RMWKVXQEVV5","created_at":"2026-07-05T02:54:17.883478+00:00"},{"alias_kind":"pith_short_16","alias_value":"3RMWKVXQEVV5NEAO","created_at":"2026-07-05T02:54:17.883478+00:00"},{"alias_kind":"pith_short_8","alias_value":"3RMWKVXQ","created_at":"2026-07-05T02:54:17.883478+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17410","citing_title":"Attention Alignment Between Humans and Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2408.12974","citing_title":"Accuracy Improvement of Cell Image Segmentation Using Feedback Former","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2412.00666","citing_title":"Explaining Object Detectors via Collective Contribution of Pixels","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23947","citing_title":"Spectral-Adaptive Modulation Networks for Visual Perception","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07245","citing_title":"Zero-Shot Textual Explanations via Translating Decision-Critical Features","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20549","citing_title":"MAPS: A Synthetic Dataset for Probing Vision Models in a Controlled 3D Scene Space","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07462","citing_title":"Do Machines Fail Like Humans? A Human-Centred Out-of-Distribution Spectrum for Mapping Error Alignment","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE","json":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE.json","graph_json":"https://pith.science/api/pith-number/3RMWKVXQEVV5NEAOME6XBVPNYE/graph.json","events_json":"https://pith.science/api/pith-number/3RMWKVXQEVV5NEAOME6XBVPNYE/events.json","paper":"https://pith.science/paper/3RMWKVXQ"},"agent_actions":{"view_html":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE","download_json":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE.json","view_paper":"https://pith.science/paper/3RMWKVXQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.07197&json=true","fetch_graph":"https://pith.science/api/pith-number/3RMWKVXQEVV5NEAOME6XBVPNYE/graph.json","fetch_events":"https://pith.science/api/pith-number/3RMWKVXQEVV5NEAOME6XBVPNYE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE/action/storage_attestation","attest_author":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE/action/author_attestation","sign_citation":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE/action/citation_signature","submit_replication":"https://pith.science/pith/3RMWKVXQEVV5NEAOME6XBVPNYE/action/replication_record"}},"created_at":"2026-07-05T02:54:17.883478+00:00","updated_at":"2026-07-05T02:54:17.883478+00:00"}