{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:56P4SMGE3XDYKZXR32BQSB7KER","short_pith_number":"pith:56P4SMGE","schema_version":"1.0","canonical_sha256":"ef9fc930c4ddc78566f1de830907ea244514334afd1d89a71ef63f82e3f46bb9","source":{"kind":"arxiv","id":"2607.16214","version":1},"attestation_state":"computed","paper":{"title":"What Makes Linguistic Representations Good Models of High-Level Visual Perception in the Human Brain?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Anna Bavaresco, Ina Klari\\'c, Marie-Francine Moens, Raquel Fern\\'andez","submitted_at":"2026-05-22T16:43:14Z","abstract_excerpt":"Image descriptions represented with language models (LMs) predict human brain responses to naturalistic images in high-level visual regions, but the factors driving this predictivity remain unclear. To investigate this, we systematically studied how images are described and which language models are used to embed those descriptions. For a common set of images, we considered six caption types -- including human-annotated and multiple machine-generated captions -- differing along several dimensions. Each caption was represented with five LMs, spanning autoregressive LMs trained to predict upcomi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.16214","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-22T16:43:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9f525479b5957ead98a0c4edad92575d9e15fcdbf56a1d9edb15f7a581535b8a","abstract_canon_sha256":"915c07abd4e210a6b1ea21e9b0070ae6fb324d25c9c582035f47c28c9f0e3b18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T00:20:06.135754Z","signature_b64":"U6Yln07XbjpHOGjnI+BWn/4ND8sWoY3cHeKH5a3DYLSafTG/FREasebSCCLEVhtqtEk/uVyqgammAmhqPdDxCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef9fc930c4ddc78566f1de830907ea244514334afd1d89a71ef63f82e3f46bb9","last_reissued_at":"2026-07-21T00:20:06.134863Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T00:20:06.134863Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes Linguistic Representations Good Models of High-Level Visual Perception in the Human Brain?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Anna Bavaresco, Ina Klari\\'c, Marie-Francine Moens, Raquel Fern\\'andez","submitted_at":"2026-05-22T16:43:14Z","abstract_excerpt":"Image descriptions represented with language models (LMs) predict human brain responses to naturalistic images in high-level visual regions, but the factors driving this predictivity remain unclear. To investigate this, we systematically studied how images are described and which language models are used to embed those descriptions. For a common set of images, we considered six caption types -- including human-annotated and multiple machine-generated captions -- differing along several dimensions. Each caption was represented with five LMs, spanning autoregressive LMs trained to predict upcomi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.16214","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.16214/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.16214","created_at":"2026-07-21T00:20:06.135339+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.16214v1","created_at":"2026-07-21T00:20:06.135339+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.16214","created_at":"2026-07-21T00:20:06.135339+00:00"},{"alias_kind":"pith_short_12","alias_value":"56P4SMGE3XDY","created_at":"2026-07-21T00:20:06.135339+00:00"},{"alias_kind":"pith_short_16","alias_value":"56P4SMGE3XDYKZXR","created_at":"2026-07-21T00:20:06.135339+00:00"},{"alias_kind":"pith_short_8","alias_value":"56P4SMGE","created_at":"2026-07-21T00:20:06.135339+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER","json":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER.json","graph_json":"https://pith.science/api/pith-number/56P4SMGE3XDYKZXR32BQSB7KER/graph.json","events_json":"https://pith.science/api/pith-number/56P4SMGE3XDYKZXR32BQSB7KER/events.json","paper":"https://pith.science/paper/56P4SMGE"},"agent_actions":{"view_html":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER","download_json":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER.json","view_paper":"https://pith.science/paper/56P4SMGE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.16214&json=true","fetch_graph":"https://pith.science/api/pith-number/56P4SMGE3XDYKZXR32BQSB7KER/graph.json","fetch_events":"https://pith.science/api/pith-number/56P4SMGE3XDYKZXR32BQSB7KER/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER/action/timestamp_anchor","attest_storage":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER/action/storage_attestation","attest_author":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER/action/author_attestation","sign_citation":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER/action/citation_signature","submit_replication":"https://pith.science/pith/56P4SMGE3XDYKZXR32BQSB7KER/action/replication_record"}},"created_at":"2026-07-21T00:20:06.135339+00:00","updated_at":"2026-07-21T00:20:06.135339+00:00"}