{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UITEON5UN2X74UXSOSPZJY3BQP","short_pith_number":"pith:UITEON5U","schema_version":"1.0","canonical_sha256":"a2264737b46eaffe52f2749f94e36183e221044664148a8ed4d7f321459d88e6","source":{"kind":"arxiv","id":"2506.03923","version":1},"attestation_state":"computed","paper":{"title":"More or Less Wrong: A Benchmark for Directional Bias in LLM Comparative Reasoning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamidreza Saffari, Mohammadamin Shafiei, Nafise Sadat Moosavi","submitted_at":"2025-06-04T13:15:01Z","abstract_excerpt":"Large language models (LLMs) are known to be sensitive to input phrasing, but the mechanisms by which semantic cues shape reasoning remain poorly understood. We investigate this phenomenon in the context of comparative math problems with objective ground truth, revealing a consistent and directional framing bias: logically equivalent questions containing the words ``more'', ``less'', or ``equal'' systematically steer predictions in the direction of the framing term. To study this effect, we introduce MathComp, a controlled benchmark of 300 comparison scenarios, each evaluated under 14 prompt v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03923","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T13:15:01Z","cross_cats_sorted":[],"title_canon_sha256":"2c90340e1b8a33ebe28d283a063033c42982e1f9f17d0562d6d4effbf5f0950a","abstract_canon_sha256":"0e32e8648e3d5f35add5ccd93400eaeb8751c59c7a4925da89daba7ad30a91bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:54.159472Z","signature_b64":"ubs+2dH+EP5M++YYWtJM5UBIs1XxoHrDPpCy1BA1ZadDPtJqiD5Bnvmwgj8jBgk8gusGwc1JLYpDMUxeJ2dTAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2264737b46eaffe52f2749f94e36183e221044664148a8ed4d7f321459d88e6","last_reissued_at":"2026-07-05T11:15:54.159028Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:54.159028Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"More or Less Wrong: A Benchmark for Directional Bias in LLM Comparative Reasoning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamidreza Saffari, Mohammadamin Shafiei, Nafise Sadat Moosavi","submitted_at":"2025-06-04T13:15:01Z","abstract_excerpt":"Large language models (LLMs) are known to be sensitive to input phrasing, but the mechanisms by which semantic cues shape reasoning remain poorly understood. We investigate this phenomenon in the context of comparative math problems with objective ground truth, revealing a consistent and directional framing bias: logically equivalent questions containing the words ``more'', ``less'', or ``equal'' systematically steer predictions in the direction of the framing term. To study this effect, we introduce MathComp, a controlled benchmark of 300 comparison scenarios, each evaluated under 14 prompt v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03923","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03923","created_at":"2026-07-05T11:15:54.159083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03923v1","created_at":"2026-07-05T11:15:54.159083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03923","created_at":"2026-07-05T11:15:54.159083+00:00"},{"alias_kind":"pith_short_12","alias_value":"UITEON5UN2X7","created_at":"2026-07-05T11:15:54.159083+00:00"},{"alias_kind":"pith_short_16","alias_value":"UITEON5UN2X74UXS","created_at":"2026-07-05T11:15:54.159083+00:00"},{"alias_kind":"pith_short_8","alias_value":"UITEON5U","created_at":"2026-07-05T11:15:54.159083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30587","citing_title":"Words Speak Louder Than Code: Investigating Cognitive Heuristics in LLM-Based Code Vulnerability Detection","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP","json":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP.json","graph_json":"https://pith.science/api/pith-number/UITEON5UN2X74UXSOSPZJY3BQP/graph.json","events_json":"https://pith.science/api/pith-number/UITEON5UN2X74UXSOSPZJY3BQP/events.json","paper":"https://pith.science/paper/UITEON5U"},"agent_actions":{"view_html":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP","download_json":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP.json","view_paper":"https://pith.science/paper/UITEON5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03923&json=true","fetch_graph":"https://pith.science/api/pith-number/UITEON5UN2X74UXSOSPZJY3BQP/graph.json","fetch_events":"https://pith.science/api/pith-number/UITEON5UN2X74UXSOSPZJY3BQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP/action/storage_attestation","attest_author":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP/action/author_attestation","sign_citation":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP/action/citation_signature","submit_replication":"https://pith.science/pith/UITEON5UN2X74UXSOSPZJY3BQP/action/replication_record"}},"created_at":"2026-07-05T11:15:54.159083+00:00","updated_at":"2026-07-05T11:15:54.159083+00:00"}