{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MVGY62ECNZZ4JJGNC7F4ZCBSQ5","short_pith_number":"pith:MVGY62EC","schema_version":"1.0","canonical_sha256":"654d8f68826e73c4a4cd17cbcc8832874c8b0816dec0793e538910b48d52953f","source":{"kind":"arxiv","id":"2406.12809","version":1},"attestation_state":"computed","paper":{"title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chang Zhou, Jian Yang, Junyang Lin, Tianyu Liu, Yichang Zhang, Zhe Yang, Zhifang Sui","submitted_at":"2024-06-18T17:25:47Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities, but still suffer from inconsistency issues (e.g. LLMs can react differently to disturbances like rephrasing or inconsequential order change). In addition to these inconsistencies, we also observe that LLMs, while capable of solving hard problems, can paradoxically fail at easier ones. To evaluate this hard-to-easy inconsistency, we develop the ConsisEval benchmark, where each entry comprises a pair of questions with a strict order of difficulty. Furthermore, we introduce the concept of consistency score to quantitatively m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12809","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T17:25:47Z","cross_cats_sorted":[],"title_canon_sha256":"8b471aea58df9188e65a32762205b9011f5756cedd9dfbd78a5a990b1b04944f","abstract_canon_sha256":"6b642176a07d178945395fc0f694b677899b9f8f7b7cb8609b482b53c839f010"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:51.544848Z","signature_b64":"qgJmApGpaHC6YC9HF7srmlxedilB9r1DRZPYWK9mGMEY0Oex1idxRsNeUNRRhSB/LlSkOo8mKXX6SNxwYMKKAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"654d8f68826e73c4a4cd17cbcc8832874c8b0816dec0793e538910b48d52953f","last_reissued_at":"2026-07-05T08:33:51.544427Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:51.544427Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chang Zhou, Jian Yang, Junyang Lin, Tianyu Liu, Yichang Zhang, Zhe Yang, Zhifang Sui","submitted_at":"2024-06-18T17:25:47Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities, but still suffer from inconsistency issues (e.g. LLMs can react differently to disturbances like rephrasing or inconsequential order change). In addition to these inconsistencies, we also observe that LLMs, while capable of solving hard problems, can paradoxically fail at easier ones. To evaluate this hard-to-easy inconsistency, we develop the ConsisEval benchmark, where each entry comprises a pair of questions with a strict order of difficulty. Furthermore, we introduce the concept of consistency score to quantitatively m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12809","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12809/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12809","created_at":"2026-07-05T08:33:51.544484+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12809v1","created_at":"2026-07-05T08:33:51.544484+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12809","created_at":"2026-07-05T08:33:51.544484+00:00"},{"alias_kind":"pith_short_12","alias_value":"MVGY62ECNZZ4","created_at":"2026-07-05T08:33:51.544484+00:00"},{"alias_kind":"pith_short_16","alias_value":"MVGY62ECNZZ4JJGN","created_at":"2026-07-05T08:33:51.544484+00:00"},{"alias_kind":"pith_short_8","alias_value":"MVGY62EC","created_at":"2026-07-05T08:33:51.544484+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.07985","citing_title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5","json":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5.json","graph_json":"https://pith.science/api/pith-number/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/graph.json","events_json":"https://pith.science/api/pith-number/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/events.json","paper":"https://pith.science/paper/MVGY62EC"},"agent_actions":{"view_html":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5","download_json":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5.json","view_paper":"https://pith.science/paper/MVGY62EC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12809&json=true","fetch_graph":"https://pith.science/api/pith-number/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/graph.json","fetch_events":"https://pith.science/api/pith-number/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/action/storage_attestation","attest_author":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/action/author_attestation","sign_citation":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/action/citation_signature","submit_replication":"https://pith.science/pith/MVGY62ECNZZ4JJGNC7F4ZCBSQ5/action/replication_record"}},"created_at":"2026-07-05T08:33:51.544484+00:00","updated_at":"2026-07-05T08:33:51.544484+00:00"}