{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M34CNOPIIEULCMQK7JEJWPLR4F","short_pith_number":"pith:M34CNOPI","schema_version":"1.0","canonical_sha256":"66f826b9e84128b1320afa489b3d71e151a1c801878e4d12ad270a0b52cab47b","source":{"kind":"arxiv","id":"2506.07064","version":1},"attestation_state":"computed","paper":{"title":"Com$^2$: A Causal-Guided Benchmark for Exploring Complex Commonsense Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Qin, Jiaqian Liu, Jinglong Gao, Kai Xiong, Li Du, Ting Liu, Xiao Ding, Yixin Cao, Yufei Zhang, Yuxiong Yan","submitted_at":"2025-06-08T09:53:08Z","abstract_excerpt":"Large language models (LLMs) have mastered abundant simple and explicit commonsense knowledge through pre-training, enabling them to achieve human-like performance in simple commonsense reasoning. Nevertheless, LLMs struggle to reason with complex and implicit commonsense knowledge that is derived from simple ones (such as understanding the long-term effects of certain events), an aspect humans tend to focus on more. Existing works focus on complex tasks like math and code, while complex commonsense reasoning remains underexplored due to its uncertainty and lack of structure. To fill this gap "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07064","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-08T09:53:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b7c428f550396129332be23f1f1073e6be0947665befb675bd1e316a7ff87dfe","abstract_canon_sha256":"e0a2b1dd9edcf3ab0c0a4cbd1943b7ef1b7b03dc4eaffffd80db485002f66516"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:00.733865Z","signature_b64":"I7Y47izLRC9C0hU4IKSUar8zPufXFuIl4AkYm524vrLjrKj0kshfIQuQeyhdbO+T/qxtwr8AfT8xPuRNAhWcBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66f826b9e84128b1320afa489b3d71e151a1c801878e4d12ad270a0b52cab47b","last_reissued_at":"2026-07-05T11:18:00.733419Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:00.733419Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Com$^2$: A Causal-Guided Benchmark for Exploring Complex Commonsense Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Qin, Jiaqian Liu, Jinglong Gao, Kai Xiong, Li Du, Ting Liu, Xiao Ding, Yixin Cao, Yufei Zhang, Yuxiong Yan","submitted_at":"2025-06-08T09:53:08Z","abstract_excerpt":"Large language models (LLMs) have mastered abundant simple and explicit commonsense knowledge through pre-training, enabling them to achieve human-like performance in simple commonsense reasoning. Nevertheless, LLMs struggle to reason with complex and implicit commonsense knowledge that is derived from simple ones (such as understanding the long-term effects of certain events), an aspect humans tend to focus on more. Existing works focus on complex tasks like math and code, while complex commonsense reasoning remains underexplored due to its uncertainty and lack of structure. To fill this gap "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07064","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07064/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07064","created_at":"2026-07-05T11:18:00.733470+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07064v1","created_at":"2026-07-05T11:18:00.733470+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07064","created_at":"2026-07-05T11:18:00.733470+00:00"},{"alias_kind":"pith_short_12","alias_value":"M34CNOPIIEUL","created_at":"2026-07-05T11:18:00.733470+00:00"},{"alias_kind":"pith_short_16","alias_value":"M34CNOPIIEULCMQK","created_at":"2026-07-05T11:18:00.733470+00:00"},{"alias_kind":"pith_short_8","alias_value":"M34CNOPI","created_at":"2026-07-05T11:18:00.733470+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F","json":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F.json","graph_json":"https://pith.science/api/pith-number/M34CNOPIIEULCMQK7JEJWPLR4F/graph.json","events_json":"https://pith.science/api/pith-number/M34CNOPIIEULCMQK7JEJWPLR4F/events.json","paper":"https://pith.science/paper/M34CNOPI"},"agent_actions":{"view_html":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F","download_json":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F.json","view_paper":"https://pith.science/paper/M34CNOPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07064&json=true","fetch_graph":"https://pith.science/api/pith-number/M34CNOPIIEULCMQK7JEJWPLR4F/graph.json","fetch_events":"https://pith.science/api/pith-number/M34CNOPIIEULCMQK7JEJWPLR4F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F/action/storage_attestation","attest_author":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F/action/author_attestation","sign_citation":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F/action/citation_signature","submit_replication":"https://pith.science/pith/M34CNOPIIEULCMQK7JEJWPLR4F/action/replication_record"}},"created_at":"2026-07-05T11:18:00.733470+00:00","updated_at":"2026-07-05T11:18:00.733470+00:00"}