{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JMRF4NZXOFD7IIQMOHJWYWFIIJ","short_pith_number":"pith:JMRF4NZX","schema_version":"1.0","canonical_sha256":"4b225e37377147f4220c71d36c58a84267b7d4f16023173c77cb075afceaf482","source":{"kind":"arxiv","id":"2404.01869","version":2},"attestation_state":"computed","paper":{"title":"Beyond Accuracy: Evaluating the Reasoning Behavior of Large Language Models -- A Survey","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Philipp Mondorf","submitted_at":"2024-04-02T11:46:31Z","abstract_excerpt":"Large language models (LLMs) have recently shown impressive performance on tasks involving reasoning, leading to a lively debate on whether these models possess reasoning capabilities similar to humans. However, despite these successes, the depth of LLMs' reasoning abilities remains uncertain. This uncertainty partly stems from the predominant focus on task performance, measured through shallow accuracy metrics, rather than a thorough investigation of the models' reasoning behavior. This paper seeks to address this gap by providing a comprehensive review of studies that go beyond task accuracy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01869","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T11:46:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3131724c5860a036521fc0e42e276aae94e9e478b8405dc1b9ce9cbdcc57775f","abstract_canon_sha256":"eb118e0217072636cd96c6599e6c3b254330e0521dbf64c0a0c2af8d5b4aefa0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:31.360153Z","signature_b64":"f02+CcMMKpfxdm8bCsgslSx1H+BMGg7rDGt4S87pA5mISHsITrzZLhc0h/jBPE3/7aD6exsR+YRH6GgmgNBAAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4b225e37377147f4220c71d36c58a84267b7d4f16023173c77cb075afceaf482","last_reissued_at":"2026-07-05T08:52:31.359738Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:31.359738Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Accuracy: Evaluating the Reasoning Behavior of Large Language Models -- A Survey","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Philipp Mondorf","submitted_at":"2024-04-02T11:46:31Z","abstract_excerpt":"Large language models (LLMs) have recently shown impressive performance on tasks involving reasoning, leading to a lively debate on whether these models possess reasoning capabilities similar to humans. However, despite these successes, the depth of LLMs' reasoning abilities remains uncertain. This uncertainty partly stems from the predominant focus on task performance, measured through shallow accuracy metrics, rather than a thorough investigation of the models' reasoning behavior. This paper seeks to address this gap by providing a comprehensive review of studies that go beyond task accuracy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01869","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01869/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01869","created_at":"2026-07-05T08:52:31.359795+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01869v2","created_at":"2026-07-05T08:52:31.359795+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01869","created_at":"2026-07-05T08:52:31.359795+00:00"},{"alias_kind":"pith_short_12","alias_value":"JMRF4NZXOFD7","created_at":"2026-07-05T08:52:31.359795+00:00"},{"alias_kind":"pith_short_16","alias_value":"JMRF4NZXOFD7IIQM","created_at":"2026-07-05T08:52:31.359795+00:00"},{"alias_kind":"pith_short_8","alias_value":"JMRF4NZX","created_at":"2026-07-05T08:52:31.359795+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22137","citing_title":"Cross-Lingual Consensus: Aligning Multilingual Cultural Knowledge via Multilingual Self-Consistency","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24213","citing_title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04047","citing_title":"TS-Reasoner: Domain-Oriented Time Series Inference Agents for Reasoning and Automated Analysis","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15931","citing_title":"Large Language Model assisted Hybrid Fuzzing","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06211","citing_title":"PuzzleWorld: A Benchmark for Multimodal, Open-Ended Reasoning in Puzzlehunts","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08142","citing_title":"Reasoning emerges from constrained inference manifolds in large language models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13371","citing_title":"Empirical Evidence of Complexity-Induced Limits in Large Language Models on Finite Discrete State-Space Problems with Explicit Validity Constraints","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02442","citing_title":"Measuring AI Reasoning: A Guide for Researchers","ref_index":138,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ","json":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ.json","graph_json":"https://pith.science/api/pith-number/JMRF4NZXOFD7IIQMOHJWYWFIIJ/graph.json","events_json":"https://pith.science/api/pith-number/JMRF4NZXOFD7IIQMOHJWYWFIIJ/events.json","paper":"https://pith.science/paper/JMRF4NZX"},"agent_actions":{"view_html":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ","download_json":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ.json","view_paper":"https://pith.science/paper/JMRF4NZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01869&json=true","fetch_graph":"https://pith.science/api/pith-number/JMRF4NZXOFD7IIQMOHJWYWFIIJ/graph.json","fetch_events":"https://pith.science/api/pith-number/JMRF4NZXOFD7IIQMOHJWYWFIIJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ/action/storage_attestation","attest_author":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ/action/author_attestation","sign_citation":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ/action/citation_signature","submit_replication":"https://pith.science/pith/JMRF4NZXOFD7IIQMOHJWYWFIIJ/action/replication_record"}},"created_at":"2026-07-05T08:52:31.359795+00:00","updated_at":"2026-07-05T08:52:31.359795+00:00"}