{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:GZAASV5GL4HRU3KEYQRNEQUK5W","short_pith_number":"pith:GZAASV5G","schema_version":"1.0","canonical_sha256":"36400957a65f0f1a6d44c422d2428aed8e7bf85b4b5e6a05eeed01cfa54ebd89","source":{"kind":"arxiv","id":"1904.09728","version":3},"attestation_state":"computed","paper":{"title":"SocialIQA: Commonsense Reasoning about Social Interactions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Social IQa is a 38,000-question benchmark that exposes a greater than 20 percent performance gap between humans and pretrained language models on social commonsense reasoning.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Derek Chen, Hannah Rashkin, Maarten Sap, Ronan LeBras, Yejin Choi","submitted_at":"2019-04-22T05:36:37Z","abstract_excerpt":"We introduce Social IQa, the first largescale benchmark for commonsense reasoning about social situations. Social IQa contains 38,000 multiple choice questions for probing emotional and social intelligence in a variety of everyday situations (e.g., Q: \"Jordan wanted to tell Tracy a secret, so Jordan leaned towards Tracy. Why did Jordan do this?\" A: \"Make sure no one else could hear\"). Through crowdsourcing, we collect commonsense questions along with correct and incorrect answers about social interactions, using a new framework that mitigates stylistic artifacts in incorrect answers by asking "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"1904.09728","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-04-22T05:36:37Z","cross_cats_sorted":[],"title_canon_sha256":"18825dae92aef04eb6bd2f54934a367526413e883fc2b6c7041fb2380b9b7020","abstract_canon_sha256":"247cd33bc4d80ff2cecf07b29b452b3ca136b7853833216a493a2d416f48077a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:58.276623Z","signature_b64":"e00umJph46mK4KPVhfcNKzOAnRdjemZB31WpfG/IDizikZjE0nb8m1S6sxGWMd8B/A3QUUL2AV00+6wK8DtQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36400957a65f0f1a6d44c422d2428aed8e7bf85b4b5e6a05eeed01cfa54ebd89","last_reissued_at":"2026-07-05T00:02:58.276093Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:58.276093Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SocialIQA: Commonsense Reasoning about Social Interactions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Social IQa is a 38,000-question benchmark that exposes a greater than 20 percent performance gap between humans and pretrained language models on social commonsense reasoning.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Derek Chen, Hannah Rashkin, Maarten Sap, Ronan LeBras, Yejin Choi","submitted_at":"2019-04-22T05:36:37Z","abstract_excerpt":"We introduce Social IQa, the first largescale benchmark for commonsense reasoning about social situations. Social IQa contains 38,000 multiple choice questions for probing emotional and social intelligence in a variety of everyday situations (e.g., Q: \"Jordan wanted to tell Tracy a secret, so Jordan leaned towards Tracy. Why did Jordan do this?\" A: \"Make sure no one else could hear\"). Through crowdsourcing, we collect commonsense questions along with correct and incorrect answers about social interactions, using a new framework that mitigates stylistic artifacts in incorrect answers by asking "},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our benchmark is challenging for existing question-answering models based on pretrained language models, compared to human performance (>20% gap). Notably, we further establish Social IQa as a resource for transfer learning of commonsense knowledge, achieving state-of-the-art performance on multiple commonsense reasoning tasks (Winograd Schemas, COPA).","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the crowdsourced questions and answers, even with the new framework to mitigate stylistic artifacts, accurately capture genuine social commonsense without introducing new biases or failing to probe true emotional intelligence.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"SocialIQA is the first large-scale benchmark with 38k crowdsourced questions testing commonsense about social interactions, where pretrained language models trail humans by over 20% but transfer to improve performance on Winograd Schemas and COPA.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Social IQa is a 38,000-question benchmark that exposes a greater than 20 percent performance gap between humans and pretrained language models on social commonsense reasoning.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"78fe54270e47d7902008d8cbe79f87af1e362b9ca3b6c36fa51017e74a48d0e5"},"source":{"id":"1904.09728","kind":"arxiv","version":3},"verdict":{"id":"ab849753-1b51-4b2a-b3cb-79d51241ef77","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T12:18:16.743823Z","strongest_claim":"Our benchmark is challenging for existing question-answering models based on pretrained language models, compared to human performance (>20% gap). Notably, we further establish Social IQa as a resource for transfer learning of commonsense knowledge, achieving state-of-the-art performance on multiple commonsense reasoning tasks (Winograd Schemas, COPA).","one_line_summary":"SocialIQA is the first large-scale benchmark with 38k crowdsourced questions testing commonsense about social interactions, where pretrained language models trail humans by over 20% but transfer to improve performance on Winograd Schemas and COPA.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the crowdsourced questions and answers, even with the new framework to mitigate stylistic artifacts, accurately capture genuine social commonsense without introducing new biases or failing to probe true emotional intelligence.","pith_extraction_headline":"Social IQa is a 38,000-question benchmark that exposes a greater than 20 percent performance gap between humans and pretrained language models on social commonsense reasoning."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.09728/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":140,"sample":[{"doi":"","year":2010,"title":"theory of mind","work_id":"8b01e093-38ea-471f-9c0d-fea577b3e6df","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":1985,"title":"Simon Baron-Cohen, Alan M Leslie, and Uta Frith. 1985. Does the Autistic Child have a ``Theory of Mind''? Cognition, 21(1):37--46","work_id":"2dfad44f-83ce-4c15-a43f-2ca2653fb247","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2015,"title":"Ernest Davis and Gary Marcus. 2015. Commonsense reasoning and commonsense knowledge in artificial intelligence. Commun. ACM, 58:92--103","work_id":"9c3590af-299e-4619-9a79-247635453f77","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT : Pre-training of deep bidirectional transformers for language understanding. In NAACL","work_id":"1d931d50-a17e-43aa-8ad6-e72874c5d4be","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2005,"title":"Espinosa and Henry Lieberman","work_id":"4e5421f6-d39a-476c-a2ee-7bd63cb0eb74","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":140,"snapshot_sha256":"270ce2de5e3cbf7936c8366b22d67ae98c28476abbe9d1629e4b095e2df09985","internal_anchors":4},"formal_canon":{"evidence_count":2,"snapshot_sha256":"f04612728011e45dc59a437fe41104dc28a2da6d544843ebbbc8a0ce97908134"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.09728","created_at":"2026-07-05T00:02:58.276160+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.09728v3","created_at":"2026-07-05T00:02:58.276160+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.09728","created_at":"2026-07-05T00:02:58.276160+00:00"},{"alias_kind":"pith_short_12","alias_value":"GZAASV5GL4HR","created_at":"2026-07-05T00:02:58.276160+00:00"},{"alias_kind":"pith_short_16","alias_value":"GZAASV5GL4HRU3KE","created_at":"2026-07-05T00:02:58.276160+00:00"},{"alias_kind":"pith_short_8","alias_value":"GZAASV5G","created_at":"2026-07-05T00:02:58.276160+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":58,"internal_anchor_count":58,"sample":[{"citing_arxiv_id":"2607.06565","citing_title":"ELSA3D: Elastic Semantic Anchoring for Unified 3D Understanding and Generation","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22249","citing_title":"On the Expressive Power of Weight Quantization in Large Language Models","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":203,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11482","citing_title":"Building Social World Models with Large Language Models","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04050","citing_title":"LiftQuant: Continuous Bit-Width LLM via Dimensional Lifting and Projection","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22297","citing_title":"One LR Doesn't Fit All: Heavy-Tail Guided Layerwise Learning Rates for LLMs","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04050","citing_title":"LiftQuant: Continuous Bit-Width LLM via Dimensional Lifting and Projection","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30036","citing_title":"Teaching Values to Machines: Simulating Human-Like Behavior in LLMs","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23901","citing_title":"LLMs as Noisy Channels: A Shannon Perspective on Model Capacity and Scaling Laws","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2310.06825","citing_title":"Mistral 7B","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2401.04088","citing_title":"Mixtral of Experts","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2406.13621","citing_title":"LaMI: Augmenting Large Language Models via Late Multi-Image Fusion","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2502.12120","citing_title":"LLMs on the Line: Data Determines Loss-to-Loss Scaling Laws","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2503.19786","citing_title":"Gemma 3 Technical Report","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2504.13898","citing_title":"Social Human Robot Embodied Conversation (SHREC) Dataset: Benchmarking Foundational Models' Social Reasoning","ref_index":46,"is_internal_anchor":true},{"citing_arxiv_id":"2505.14990","citing_title":"Language Specific Knowledge: Do Models Know Better in X than in English?","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22297","citing_title":"One LR Doesn't Fit All: Heavy-Tail Guided Layerwise Learning Rates for LLMs","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20189","citing_title":"SOLAR: A Self-Optimizing Open-Ended Autonomous Agent for Lifelong Learning and Continual Adaptation","ref_index":72,"is_internal_anchor":true},{"citing_arxiv_id":"2605.21147","citing_title":"SMoA: Spectrum Modulation Adapter for Parameter-Efficient Fine-Tuning","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16690","citing_title":"UB-SMoE: Universally Balanced Sparse Mixture-of-Experts for Resource-adaptive Federated Fine-tuning of Foundation Models","ref_index":58,"is_internal_anchor":true},{"citing_arxiv_id":"2507.00432","citing_title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","ref_index":206,"is_internal_anchor":true},{"citing_arxiv_id":"2509.21637","citing_title":"BoHA: Blockwise Hadamard Product Adaptation for Parameter-Efficient Fine-Tuning","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2509.18629","citing_title":"HyperAdapt: Simple High-Rank Adaptation","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2509.24552","citing_title":"Short window attention enables long-term memorization","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W","json":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W.json","graph_json":"https://pith.science/api/pith-number/GZAASV5GL4HRU3KEYQRNEQUK5W/graph.json","events_json":"https://pith.science/api/pith-number/GZAASV5GL4HRU3KEYQRNEQUK5W/events.json","paper":"https://pith.science/paper/GZAASV5G"},"agent_actions":{"view_html":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W","download_json":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W.json","view_paper":"https://pith.science/paper/GZAASV5G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.09728&json=true","fetch_graph":"https://pith.science/api/pith-number/GZAASV5GL4HRU3KEYQRNEQUK5W/graph.json","fetch_events":"https://pith.science/api/pith-number/GZAASV5GL4HRU3KEYQRNEQUK5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W/action/storage_attestation","attest_author":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W/action/author_attestation","sign_citation":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W/action/citation_signature","submit_replication":"https://pith.science/pith/GZAASV5GL4HRU3KEYQRNEQUK5W/action/replication_record"}},"created_at":"2026-07-05T00:02:58.276160+00:00","updated_at":"2026-07-05T00:02:58.276160+00:00"}