{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZQAKAFPHZMKMWAKVF6D3UZTLEW","short_pith_number":"pith:ZQAKAFPH","schema_version":"1.0","canonical_sha256":"cc00a015e7cb14cb01552f87ba666b2582b066ffce419594dc7d677e27dc5795","source":{"kind":"arxiv","id":"2412.13147","version":5},"attestation_state":"computed","paper":{"title":"Are Your LLMs Capable of Stable Reasoning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Hongwei Liu, Junnan Liu, Kai Chen, Kuikun Liu, Linchen Xiao, Songyang Gao, Songyang Zhang, Wenwei Zhang, Ziyi Wang","submitted_at":"2024-12-17T18:12:47Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has shown remarkable progress in complex reasoning tasks. However, a significant disparity exists between benchmark performances and real-world applications. We attribute this gap primarily to current evaluation protocols and metrics, which inadequately capture the full spectrum of LLM capabilities, especially in complex reasoning tasks where both accuracy and consistency are essential. In this paper, we introduce G-Pass@$k$, a novel evaluation metric that continuously assesses model performance across multiple sampling attempts, quantifyin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.13147","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-12-17T18:12:47Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6e383a72bda2ea58cc5f4f0d36d1a12ffa8a13d705825446c66fd5a95b29d5a6","abstract_canon_sha256":"2485e948cd2a4bf03e9e859aea66963d6dc175080770d8294b8da403f9abb8cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:29.548129Z","signature_b64":"R8bKdwGO3+S6OJuDJt2jSbnugrCIAPvWardLbI5kshDv36+/KN5TNhi3CoxoewkExI224+WLMwcwYUW8U+GSBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc00a015e7cb14cb01552f87ba666b2582b066ffce419594dc7d677e27dc5795","last_reissued_at":"2026-07-05T11:50:29.547636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:29.547636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Your LLMs Capable of Stable Reasoning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Hongwei Liu, Junnan Liu, Kai Chen, Kuikun Liu, Linchen Xiao, Songyang Gao, Songyang Zhang, Wenwei Zhang, Ziyi Wang","submitted_at":"2024-12-17T18:12:47Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has shown remarkable progress in complex reasoning tasks. However, a significant disparity exists between benchmark performances and real-world applications. We attribute this gap primarily to current evaluation protocols and metrics, which inadequately capture the full spectrum of LLM capabilities, especially in complex reasoning tasks where both accuracy and consistency are essential. In this paper, we introduce G-Pass@$k$, a novel evaluation metric that continuously assesses model performance across multiple sampling attempts, quantifyin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.13147","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.13147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.13147","created_at":"2026-07-05T11:50:29.547696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.13147v5","created_at":"2026-07-05T11:50:29.547696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.13147","created_at":"2026-07-05T11:50:29.547696+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZQAKAFPHZMKM","created_at":"2026-07-05T11:50:29.547696+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZQAKAFPHZMKMWAKV","created_at":"2026-07-05T11:50:29.547696+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZQAKAFPH","created_at":"2026-07-05T11:50:29.547696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2508.08636","citing_title":"InternBootcamp Technical Report: Boosting LLM Reasoning with Verifiable Task Scaling","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08401","citing_title":"AIPO: Learning to Reason from Active Interaction","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04265","citing_title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08401","citing_title":"AIPO: Learning to Reason from Active Interaction","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01474","citing_title":"ReMedi: Reasoner for Medical Clinical Prediction","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW","json":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW.json","graph_json":"https://pith.science/api/pith-number/ZQAKAFPHZMKMWAKVF6D3UZTLEW/graph.json","events_json":"https://pith.science/api/pith-number/ZQAKAFPHZMKMWAKVF6D3UZTLEW/events.json","paper":"https://pith.science/paper/ZQAKAFPH"},"agent_actions":{"view_html":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW","download_json":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW.json","view_paper":"https://pith.science/paper/ZQAKAFPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.13147&json=true","fetch_graph":"https://pith.science/api/pith-number/ZQAKAFPHZMKMWAKVF6D3UZTLEW/graph.json","fetch_events":"https://pith.science/api/pith-number/ZQAKAFPHZMKMWAKVF6D3UZTLEW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW/action/storage_attestation","attest_author":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW/action/author_attestation","sign_citation":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW/action/citation_signature","submit_replication":"https://pith.science/pith/ZQAKAFPHZMKMWAKVF6D3UZTLEW/action/replication_record"}},"created_at":"2026-07-05T11:50:29.547696+00:00","updated_at":"2026-07-05T11:50:29.547696+00:00"}