{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IPODHGMUSKNPP4WTJ3TRVJVNPQ","short_pith_number":"pith:IPODHGMU","schema_version":"1.0","canonical_sha256":"43dc339994929af7f2d34ee71aa6ad7c16eb5e7099e0bbff4ed92d5128bdff1b","source":{"kind":"arxiv","id":"2404.15522","version":2},"attestation_state":"computed","paper":{"title":"LogicBench: Towards Systematic Evaluation of Logical Reasoning Ability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arindam Mitra, Chitta Baral, Man Luo, Mihir Parmar, Mutsumi Nakamura, Neeraj Varshney, Nisarg Patel, Santosh Mashetty","submitted_at":"2024-04-23T21:08:49Z","abstract_excerpt":"Recently developed large language models (LLMs) have been shown to perform remarkably well on a wide range of language understanding tasks. But, can they really \"reason\" over the natural language? This question has been receiving significant research attention and many reasoning skills such as commonsense, numerical, and qualitative have been studied. However, the crucial skill pertaining to 'logical reasoning' has remained underexplored. Existing work investigating this reasoning ability of LLMs has focused only on a couple of inference rules (such as modus ponens and modus tollens) of propos"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.15522","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-23T21:08:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2c27a6cc98256fe2b64bf219143d1236b257e8dfe43d62b6536660e6907bfc04","abstract_canon_sha256":"3ff5cf46ee4c734e661680aec2a90a97b4e1af4ae2f2f85bb7df023b79d89076"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:09.445258Z","signature_b64":"O65ayBSUhWD+5/a0WCogwVDDbmtu+1B01xWK5qRovPg5/IkZgTgVdzSfZBetB/gFTxFq2sqlN7jbJvDRQHAfCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43dc339994929af7f2d34ee71aa6ad7c16eb5e7099e0bbff4ed92d5128bdff1b","last_reissued_at":"2026-07-05T08:28:09.444837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:09.444837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LogicBench: Towards Systematic Evaluation of Logical Reasoning Ability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arindam Mitra, Chitta Baral, Man Luo, Mihir Parmar, Mutsumi Nakamura, Neeraj Varshney, Nisarg Patel, Santosh Mashetty","submitted_at":"2024-04-23T21:08:49Z","abstract_excerpt":"Recently developed large language models (LLMs) have been shown to perform remarkably well on a wide range of language understanding tasks. But, can they really \"reason\" over the natural language? This question has been receiving significant research attention and many reasoning skills such as commonsense, numerical, and qualitative have been studied. However, the crucial skill pertaining to 'logical reasoning' has remained underexplored. Existing work investigating this reasoning ability of LLMs has focused only on a couple of inference rules (such as modus ponens and modus tollens) of propos"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.15522","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.15522/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.15522","created_at":"2026-07-05T08:28:09.444892+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.15522v2","created_at":"2026-07-05T08:28:09.444892+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.15522","created_at":"2026-07-05T08:28:09.444892+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPODHGMUSKNP","created_at":"2026-07-05T08:28:09.444892+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPODHGMUSKNPP4WT","created_at":"2026-07-05T08:28:09.444892+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPODHGMU","created_at":"2026-07-05T08:28:09.444892+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.05489","citing_title":"Self-Aligned Reward: Towards Effective and Efficient Reasoners","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ","json":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ.json","graph_json":"https://pith.science/api/pith-number/IPODHGMUSKNPP4WTJ3TRVJVNPQ/graph.json","events_json":"https://pith.science/api/pith-number/IPODHGMUSKNPP4WTJ3TRVJVNPQ/events.json","paper":"https://pith.science/paper/IPODHGMU"},"agent_actions":{"view_html":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ","download_json":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ.json","view_paper":"https://pith.science/paper/IPODHGMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.15522&json=true","fetch_graph":"https://pith.science/api/pith-number/IPODHGMUSKNPP4WTJ3TRVJVNPQ/graph.json","fetch_events":"https://pith.science/api/pith-number/IPODHGMUSKNPP4WTJ3TRVJVNPQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ/action/storage_attestation","attest_author":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ/action/author_attestation","sign_citation":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ/action/citation_signature","submit_replication":"https://pith.science/pith/IPODHGMUSKNPP4WTJ3TRVJVNPQ/action/replication_record"}},"created_at":"2026-07-05T08:28:09.444892+00:00","updated_at":"2026-07-05T08:28:09.444892+00:00"}