{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XGGQE6KSDLMOT2FV6YTKETKZTF","short_pith_number":"pith:XGGQE6KS","schema_version":"1.0","canonical_sha256":"b98d0279521ad8e9e8b5f626a24d59997cfecd0efeea4224a0254aa388e743fb","source":{"kind":"arxiv","id":"2502.15109","version":4},"attestation_state":"computed","paper":{"title":"Social Genome: Grounded Social Reasoning Abilities of Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Leena Mathur, Louis-Philippe Morency, Marian Qian, Paul Pu Liang","submitted_at":"2025-02-21T00:05:40Z","abstract_excerpt":"Social reasoning abilities are crucial for AI systems to effectively interpret and respond to multimodal human communication and interaction within social contexts. We introduce SOCIAL GENOME, the first benchmark for fine-grained, grounded social reasoning abilities of multimodal models. SOCIAL GENOME contains 272 videos of interactions and 1,486 human-annotated reasoning traces related to inferences about these interactions. These traces contain 5,777 reasoning steps that reference evidence from visual cues, verbal cues, vocal cues, and external knowledge (contextual knowledge external to vid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15109","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-21T00:05:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"fbcfc161af0164375ee2a16a5570f04c37d444a0363e7fec7bcf44ee9a87f175","abstract_canon_sha256":"1152b16a393955de0edc59e16645fa08a63327790e804011863b01b28d5f290e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:19.582720Z","signature_b64":"vcPeoNQn9s/XUgYGE84sUrX/qjPmqEKd608H3s/WhPyCLsMzdGRSBypXCxRvf1mCb2PEIpaLZ0DEKbqxd3OYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b98d0279521ad8e9e8b5f626a24d59997cfecd0efeea4224a0254aa388e743fb","last_reissued_at":"2026-07-05T11:15:19.582271Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:19.582271Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Social Genome: Grounded Social Reasoning Abilities of Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Leena Mathur, Louis-Philippe Morency, Marian Qian, Paul Pu Liang","submitted_at":"2025-02-21T00:05:40Z","abstract_excerpt":"Social reasoning abilities are crucial for AI systems to effectively interpret and respond to multimodal human communication and interaction within social contexts. We introduce SOCIAL GENOME, the first benchmark for fine-grained, grounded social reasoning abilities of multimodal models. SOCIAL GENOME contains 272 videos of interactions and 1,486 human-annotated reasoning traces related to inferences about these interactions. These traces contain 5,777 reasoning steps that reference evidence from visual cues, verbal cues, vocal cues, and external knowledge (contextual knowledge external to vid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15109","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15109","created_at":"2026-07-05T11:15:19.582340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15109v4","created_at":"2026-07-05T11:15:19.582340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15109","created_at":"2026-07-05T11:15:19.582340+00:00"},{"alias_kind":"pith_short_12","alias_value":"XGGQE6KSDLMO","created_at":"2026-07-05T11:15:19.582340+00:00"},{"alias_kind":"pith_short_16","alias_value":"XGGQE6KSDLMOT2FV","created_at":"2026-07-05T11:15:19.582340+00:00"},{"alias_kind":"pith_short_8","alias_value":"XGGQE6KS","created_at":"2026-07-05T11:15:19.582340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.13898","citing_title":"Social Human Robot Embodied Conversation (SHREC) Dataset: Benchmarking Foundational Models' Social Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05425","citing_title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06211","citing_title":"PuzzleWorld: A Benchmark for Multimodal, Open-Ended Reasoning in Puzzlehunts","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20765","citing_title":"Looking Beyond the Obvious: A Survey on Abstract Concept Recognition for Video Understanding","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01657","citing_title":"Act2See: Emergent Active Visual Perception for Video Reasoning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF","json":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF.json","graph_json":"https://pith.science/api/pith-number/XGGQE6KSDLMOT2FV6YTKETKZTF/graph.json","events_json":"https://pith.science/api/pith-number/XGGQE6KSDLMOT2FV6YTKETKZTF/events.json","paper":"https://pith.science/paper/XGGQE6KS"},"agent_actions":{"view_html":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF","download_json":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF.json","view_paper":"https://pith.science/paper/XGGQE6KS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15109&json=true","fetch_graph":"https://pith.science/api/pith-number/XGGQE6KSDLMOT2FV6YTKETKZTF/graph.json","fetch_events":"https://pith.science/api/pith-number/XGGQE6KSDLMOT2FV6YTKETKZTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF/action/storage_attestation","attest_author":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF/action/author_attestation","sign_citation":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF/action/citation_signature","submit_replication":"https://pith.science/pith/XGGQE6KSDLMOT2FV6YTKETKZTF/action/replication_record"}},"created_at":"2026-07-05T11:15:19.582340+00:00","updated_at":"2026-07-05T11:15:19.582340+00:00"}