{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WYWY66P7RROBEQG345QM2LSG64","short_pith_number":"pith:WYWY66P7","schema_version":"1.0","canonical_sha256":"b62d8f79ff8c5c1240dbe760cd2e46f72ab517c010ff462a3de06db348fe8cfc","source":{"kind":"arxiv","id":"2311.07766","version":1},"attestation_state":"computed","paper":{"title":"Vision-Language Integration in Multimodal Video Transformers (Partially) Aligns with the Brain","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dota Tianai Dong, Mariya Toneva","submitted_at":"2023-11-13T21:32:37Z","abstract_excerpt":"Integrating information from multiple modalities is arguably one of the essential prerequisites for grounding artificial intelligence systems with an understanding of the real world. Recent advances in video transformers that jointly learn from vision, text, and sound over time have made some progress toward this goal, but the degree to which these models integrate information from modalities still remains unclear. In this work, we present a promising approach for probing a pre-trained multimodal video transformer model by leveraging neuroscientific evidence of multimodal information processin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07766","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-13T21:32:37Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"cbdebbc848b83d858f606460fda160d0911281c59f3a448e9a1aaf5a84b60c65","abstract_canon_sha256":"4183bc88bf1018f9ea3c709865a23f9f07a4b06f193a0afaeed021d8fdecb5b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:23.429830Z","signature_b64":"OnbXKBwdqE4WtGqxT8AgeHg6B/pdSEnfhRBedktqT83nVchCpIInU0RGhAW2Cyye/y7WslnK/JO6R9ggvnaJBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b62d8f79ff8c5c1240dbe760cd2e46f72ab517c010ff462a3de06db348fe8cfc","last_reissued_at":"2026-07-05T07:12:23.429367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:23.429367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision-Language Integration in Multimodal Video Transformers (Partially) Aligns with the Brain","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dota Tianai Dong, Mariya Toneva","submitted_at":"2023-11-13T21:32:37Z","abstract_excerpt":"Integrating information from multiple modalities is arguably one of the essential prerequisites for grounding artificial intelligence systems with an understanding of the real world. Recent advances in video transformers that jointly learn from vision, text, and sound over time have made some progress toward this goal, but the degree to which these models integrate information from modalities still remains unclear. In this work, we present a promising approach for probing a pre-trained multimodal video transformer model by leveraging neuroscientific evidence of multimodal information processin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07766","created_at":"2026-07-05T07:12:23.429423+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07766v1","created_at":"2026-07-05T07:12:23.429423+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07766","created_at":"2026-07-05T07:12:23.429423+00:00"},{"alias_kind":"pith_short_12","alias_value":"WYWY66P7RROB","created_at":"2026-07-05T07:12:23.429423+00:00"},{"alias_kind":"pith_short_16","alias_value":"WYWY66P7RROBEQG3","created_at":"2026-07-05T07:12:23.429423+00:00"},{"alias_kind":"pith_short_8","alias_value":"WYWY66P7","created_at":"2026-07-05T07:12:23.429423+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.08277","citing_title":"Task-conditioned probing of instruction-tuned multimodal LLMs: Region-specific brain alignment patterns under naturalistic stimuli","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19352","citing_title":"Brain alignment of reasoning and action representations from vision-language and action models during naturalistic gameplay","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04326","citing_title":"A foundation model of vision, audition, and language for in-silico neuroscience","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64","json":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64.json","graph_json":"https://pith.science/api/pith-number/WYWY66P7RROBEQG345QM2LSG64/graph.json","events_json":"https://pith.science/api/pith-number/WYWY66P7RROBEQG345QM2LSG64/events.json","paper":"https://pith.science/paper/WYWY66P7"},"agent_actions":{"view_html":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64","download_json":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64.json","view_paper":"https://pith.science/paper/WYWY66P7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07766&json=true","fetch_graph":"https://pith.science/api/pith-number/WYWY66P7RROBEQG345QM2LSG64/graph.json","fetch_events":"https://pith.science/api/pith-number/WYWY66P7RROBEQG345QM2LSG64/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64/action/storage_attestation","attest_author":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64/action/author_attestation","sign_citation":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64/action/citation_signature","submit_replication":"https://pith.science/pith/WYWY66P7RROBEQG345QM2LSG64/action/replication_record"}},"created_at":"2026-07-05T07:12:23.429423+00:00","updated_at":"2026-07-05T07:12:23.429423+00:00"}