{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:75UWU7L3ZDF3NPN6RO5AHSS2LE","short_pith_number":"pith:75UWU7L3","schema_version":"1.0","canonical_sha256":"ff696a7d7bc8cbb6bdbe8bba03ca5a593237b61b2c27b7deae5dbc0f107cf0f1","source":{"kind":"arxiv","id":"2501.14851","version":2},"attestation_state":"computed","paper":{"title":"JustLogic: A Comprehensive Benchmark for Evaluating Deductive Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.LO"],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Michael K. Chen, Xikun Zhang","submitted_at":"2025-01-24T15:49:10Z","abstract_excerpt":"Logical reasoning is a critical component of Large Language Models (LLMs), and substantial research efforts in recent years have aimed to enhance their deductive reasoning capabilities. However, existing deductive reasoning benchmarks, which are crucial for evaluating and advancing LLMs, are inadequate due to their lack of task complexity, presence of prior knowledge as a confounder, and superficial error analysis. To address these deficiencies, we introduce JustLogic, a synthetically generated deductive reasoning benchmark designed for rigorous evaluation of LLMs. JustLogic is (i) highly comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14851","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-24T15:49:10Z","cross_cats_sorted":["cs.AI","cs.LG","cs.LO"],"title_canon_sha256":"55f827021fe71271e51f874e169ba031f03f158e4cc261031958bafb4a49e91a","abstract_canon_sha256":"da2204811f7050b675037da6afeb40a47315ff387ffae9dc6f71fc1ff3d8ef88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:41.801851Z","signature_b64":"A+21YJbWAf15JVkU9fm1NSh48PawbchMdXyNXNPxlYHioWfFxAdBDbxHA7LbuJtVLO6NHBUWpCFS3wxutzFyAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff696a7d7bc8cbb6bdbe8bba03ca5a593237b61b2c27b7deae5dbc0f107cf0f1","last_reissued_at":"2026-07-05T11:00:41.801311Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:41.801311Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JustLogic: A Comprehensive Benchmark for Evaluating Deductive Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.LO"],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Michael K. Chen, Xikun Zhang","submitted_at":"2025-01-24T15:49:10Z","abstract_excerpt":"Logical reasoning is a critical component of Large Language Models (LLMs), and substantial research efforts in recent years have aimed to enhance their deductive reasoning capabilities. However, existing deductive reasoning benchmarks, which are crucial for evaluating and advancing LLMs, are inadequate due to their lack of task complexity, presence of prior knowledge as a confounder, and superficial error analysis. To address these deficiencies, we introduce JustLogic, a synthetically generated deductive reasoning benchmark designed for rigorous evaluation of LLMs. JustLogic is (i) highly comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14851","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14851/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14851","created_at":"2026-07-05T11:00:41.801378+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14851v2","created_at":"2026-07-05T11:00:41.801378+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14851","created_at":"2026-07-05T11:00:41.801378+00:00"},{"alias_kind":"pith_short_12","alias_value":"75UWU7L3ZDF3","created_at":"2026-07-05T11:00:41.801378+00:00"},{"alias_kind":"pith_short_16","alias_value":"75UWU7L3ZDF3NPN6","created_at":"2026-07-05T11:00:41.801378+00:00"},{"alias_kind":"pith_short_8","alias_value":"75UWU7L3","created_at":"2026-07-05T11:00:41.801378+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05009","citing_title":"DAR: Deontic Reasoning with Agentic Harnesses","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04443","citing_title":"DeonticBench: A Benchmark for Reasoning over Rules","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE","json":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE.json","graph_json":"https://pith.science/api/pith-number/75UWU7L3ZDF3NPN6RO5AHSS2LE/graph.json","events_json":"https://pith.science/api/pith-number/75UWU7L3ZDF3NPN6RO5AHSS2LE/events.json","paper":"https://pith.science/paper/75UWU7L3"},"agent_actions":{"view_html":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE","download_json":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE.json","view_paper":"https://pith.science/paper/75UWU7L3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14851&json=true","fetch_graph":"https://pith.science/api/pith-number/75UWU7L3ZDF3NPN6RO5AHSS2LE/graph.json","fetch_events":"https://pith.science/api/pith-number/75UWU7L3ZDF3NPN6RO5AHSS2LE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE/action/storage_attestation","attest_author":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE/action/author_attestation","sign_citation":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE/action/citation_signature","submit_replication":"https://pith.science/pith/75UWU7L3ZDF3NPN6RO5AHSS2LE/action/replication_record"}},"created_at":"2026-07-05T11:00:41.801378+00:00","updated_at":"2026-07-05T11:00:41.801378+00:00"}