{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V6VTQF42NNSFB76NW3335N2T4O","short_pith_number":"pith:V6VTQF42","schema_version":"1.0","canonical_sha256":"afab38179a6b6450ffcdb6f7beb753e39b5d7c6d5aafa19b843fb94372feca9d","source":{"kind":"arxiv","id":"2408.05093","version":4},"attestation_state":"computed","paper":{"title":"Order Matters in Hallucination: Reasoning Order as Benchmark and Reflexive Prompting for Large-Language-Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Zikai Xie","submitted_at":"2024-08-09T14:34:32Z","abstract_excerpt":"Large language models (LLMs) have generated significant attention since their inception, finding applications across various academic and industrial domains. However, these models often suffer from the \"hallucination problem\", where outputs, though grammatically and logically coherent, lack factual accuracy or are entirely fabricated. A particularly troubling issue discovered and widely discussed recently is the numerical comparison error where multiple LLMs incorrectly infer that \"9.11$>$9.9\". We discovered that the order in which LLMs generate answers and reasoning impacts their consistency."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05093","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-09T14:34:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4cd43fe8e0cb3ac99566f1d0f185f95f871656e86379249c954bf257f5accd31","abstract_canon_sha256":"a1f7b56ea638ece9e793f6bbda5f39582895a4d6b622b7b57ffeab83987d23d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:25.987133Z","signature_b64":"3RcsFnSi0NqxyeMoaAie1B3yPj6fc1GUlwkUKmAGaEHoRfGqze4BrpoaXVAStlsR0iZMnRkT7Y/ECNQF8HLOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afab38179a6b6450ffcdb6f7beb753e39b5d7c6d5aafa19b843fb94372feca9d","last_reissued_at":"2026-07-05T11:01:25.986632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:25.986632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Order Matters in Hallucination: Reasoning Order as Benchmark and Reflexive Prompting for Large-Language-Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Zikai Xie","submitted_at":"2024-08-09T14:34:32Z","abstract_excerpt":"Large language models (LLMs) have generated significant attention since their inception, finding applications across various academic and industrial domains. However, these models often suffer from the \"hallucination problem\", where outputs, though grammatically and logically coherent, lack factual accuracy or are entirely fabricated. A particularly troubling issue discovered and widely discussed recently is the numerical comparison error where multiple LLMs incorrectly infer that \"9.11$>$9.9\". We discovered that the order in which LLMs generate answers and reasoning impacts their consistency."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05093","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05093/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05093","created_at":"2026-07-05T11:01:25.986697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05093v4","created_at":"2026-07-05T11:01:25.986697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05093","created_at":"2026-07-05T11:01:25.986697+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6VTQF42NNSF","created_at":"2026-07-05T11:01:25.986697+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6VTQF42NNSFB76N","created_at":"2026-07-05T11:01:25.986697+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6VTQF42","created_at":"2026-07-05T11:01:25.986697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.16617","citing_title":"AVRT: Audio-Visual Reasoning Transfer through Single-Modality Teachers","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O","json":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O.json","graph_json":"https://pith.science/api/pith-number/V6VTQF42NNSFB76NW3335N2T4O/graph.json","events_json":"https://pith.science/api/pith-number/V6VTQF42NNSFB76NW3335N2T4O/events.json","paper":"https://pith.science/paper/V6VTQF42"},"agent_actions":{"view_html":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O","download_json":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O.json","view_paper":"https://pith.science/paper/V6VTQF42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05093&json=true","fetch_graph":"https://pith.science/api/pith-number/V6VTQF42NNSFB76NW3335N2T4O/graph.json","fetch_events":"https://pith.science/api/pith-number/V6VTQF42NNSFB76NW3335N2T4O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O/action/storage_attestation","attest_author":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O/action/author_attestation","sign_citation":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O/action/citation_signature","submit_replication":"https://pith.science/pith/V6VTQF42NNSFB76NW3335N2T4O/action/replication_record"}},"created_at":"2026-07-05T11:01:25.986697+00:00","updated_at":"2026-07-05T11:01:25.986697+00:00"}