{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:VUKQSLKUHP6NWKCC2NNPYEXKJB","short_pith_number":"pith:VUKQSLKU","schema_version":"1.0","canonical_sha256":"ad15092d543bfcdb2842d35afc12ea485384b58a9aa052b0a85167a82f4a38b2","source":{"kind":"arxiv","id":"2209.11737","version":2},"attestation_state":"computed","paper":{"title":"Visual representations in the human brain are aligned with large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-bio.NC"],"primary_cat":"cs.CV","authors_text":"Adrien Doerig, Emily Allen, Ian Charest, Kendrick Kay, Thomas Naselaris, Tim C Kietzmann, Yihan Wu","submitted_at":"2022-09-23T17:34:33Z","abstract_excerpt":"The human brain extracts complex information from visual inputs, including objects, their spatial and semantic interrelations, and their interactions with the environment. However, a quantitative approach for studying this information remains elusive. Here, we test whether the contextual information encoded in large language models (LLMs) is beneficial for modelling the complex visual information extracted by the brain from natural scenes. We show that LLM embeddings of scene captions successfully characterise brain activity evoked by viewing the natural scenes. This mapping captures selectivi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.11737","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-09-23T17:34:33Z","cross_cats_sorted":["cs.LG","q-bio.NC"],"title_canon_sha256":"236619144569d284dac0ad633a5df1a2193d2a646e17ac6a016e0efa88f0ae2c","abstract_canon_sha256":"9ad38d138a43174fb09ce6c6fcfc89feb415673adb0deb256932eacb9f4e4f38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:37.995179Z","signature_b64":"45asRZyuUucmoU/gUkQHZT311s7yZcVAqNj5TuEWNHKPM+9cRv/C7gjAhlRJfIxykeng4TDC1ltdEBz5R0mDBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad15092d543bfcdb2842d35afc12ea485384b58a9aa052b0a85167a82f4a38b2","last_reissued_at":"2026-07-05T08:40:37.994666Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:37.994666Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual representations in the human brain are aligned with large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-bio.NC"],"primary_cat":"cs.CV","authors_text":"Adrien Doerig, Emily Allen, Ian Charest, Kendrick Kay, Thomas Naselaris, Tim C Kietzmann, Yihan Wu","submitted_at":"2022-09-23T17:34:33Z","abstract_excerpt":"The human brain extracts complex information from visual inputs, including objects, their spatial and semantic interrelations, and their interactions with the environment. However, a quantitative approach for studying this information remains elusive. Here, we test whether the contextual information encoded in large language models (LLMs) is beneficial for modelling the complex visual information extracted by the brain from natural scenes. We show that LLM embeddings of scene captions successfully characterise brain activity evoked by viewing the natural scenes. This mapping captures selectivi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.11737","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.11737/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.11737","created_at":"2026-07-05T08:40:37.994725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.11737v2","created_at":"2026-07-05T08:40:37.994725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.11737","created_at":"2026-07-05T08:40:37.994725+00:00"},{"alias_kind":"pith_short_12","alias_value":"VUKQSLKUHP6N","created_at":"2026-07-05T08:40:37.994725+00:00"},{"alias_kind":"pith_short_16","alias_value":"VUKQSLKUHP6NWKCC","created_at":"2026-07-05T08:40:37.994725+00:00"},{"alias_kind":"pith_short_8","alias_value":"VUKQSLKU","created_at":"2026-07-05T08:40:37.994725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.04509","citing_title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08277","citing_title":"Task-conditioned probing of instruction-tuned multimodal LLMs: Region-specific brain alignment patterns under naturalistic stimuli","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04326","citing_title":"A foundation model of vision, audition, and language for in-silico neuroscience","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09817","citing_title":"NeuroFlow: Toward Unified Visual Encoding and Decoding from Neural Activity","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08537","citing_title":"Meta-learning In-Context Enables Training-Free Cross Subject Brain Decoding","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB","json":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB.json","graph_json":"https://pith.science/api/pith-number/VUKQSLKUHP6NWKCC2NNPYEXKJB/graph.json","events_json":"https://pith.science/api/pith-number/VUKQSLKUHP6NWKCC2NNPYEXKJB/events.json","paper":"https://pith.science/paper/VUKQSLKU"},"agent_actions":{"view_html":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB","download_json":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB.json","view_paper":"https://pith.science/paper/VUKQSLKU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.11737&json=true","fetch_graph":"https://pith.science/api/pith-number/VUKQSLKUHP6NWKCC2NNPYEXKJB/graph.json","fetch_events":"https://pith.science/api/pith-number/VUKQSLKUHP6NWKCC2NNPYEXKJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB/action/storage_attestation","attest_author":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB/action/author_attestation","sign_citation":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB/action/citation_signature","submit_replication":"https://pith.science/pith/VUKQSLKUHP6NWKCC2NNPYEXKJB/action/replication_record"}},"created_at":"2026-07-05T08:40:37.994725+00:00","updated_at":"2026-07-05T08:40:37.994725+00:00"}