{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W65RQQ6OA7FE32ACDHRBF4ST5V","short_pith_number":"pith:W65RQQ6O","schema_version":"1.0","canonical_sha256":"b7bb1843ce07ca4de80219e212f253ed70c0a527f0a54b10efd67c848a8c69c2","source":{"kind":"arxiv","id":"2502.18435","version":3},"attestation_state":"computed","paper":{"title":"What Makes the Preferred Thinking Direction for LLMs in Multiple-choice Questions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","cs.LG","math.IT"],"primary_cat":"cs.CL","authors_text":"Emmanuel Abbe, Jiatao Gu, Navdeep Jaitly, Richard Bai, Ruixiang Zhang, Samy Bengio, Yizhe Zhang, Zijin Gu","submitted_at":"2025-02-25T18:30:25Z","abstract_excerpt":"Language models usually use left-to-right (L2R) autoregressive factorization. However, L2R factorization may not always be the best inductive bias. Therefore, we investigate whether alternative factorizations of the text distribution could be beneficial in some tasks. We investigate right-to-left (R2L) training as a compelling alternative, focusing on multiple-choice questions (MCQs) as a test bed for knowledge extraction and reasoning. Through extensive experiments across various model sizes (2B-8B parameters) and training datasets, we find that R2L models can significantly outperform L2R mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18435","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-25T18:30:25Z","cross_cats_sorted":["cs.IT","cs.LG","math.IT"],"title_canon_sha256":"e4767edc976c951c50b6b2d590d255f1314557fdbbac9dfd7890dc78e5e621f5","abstract_canon_sha256":"6470465e163f7f4f623e85ebb0e4c01e2e422d7b1db67489ff07503e36a648ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:34.408116Z","signature_b64":"+K0173rJlXJ9RnGsc4d06IQ1gW+UUrQRX+wWKoW0/uODLWFI8fOz+0mT3efjzg8duyiELBpi5kXovt6elT4YDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7bb1843ce07ca4de80219e212f253ed70c0a527f0a54b10efd67c848a8c69c2","last_reissued_at":"2026-07-05T11:28:34.407572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:34.407572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes the Preferred Thinking Direction for LLMs in Multiple-choice Questions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","cs.LG","math.IT"],"primary_cat":"cs.CL","authors_text":"Emmanuel Abbe, Jiatao Gu, Navdeep Jaitly, Richard Bai, Ruixiang Zhang, Samy Bengio, Yizhe Zhang, Zijin Gu","submitted_at":"2025-02-25T18:30:25Z","abstract_excerpt":"Language models usually use left-to-right (L2R) autoregressive factorization. However, L2R factorization may not always be the best inductive bias. Therefore, we investigate whether alternative factorizations of the text distribution could be beneficial in some tasks. We investigate right-to-left (R2L) training as a compelling alternative, focusing on multiple-choice questions (MCQs) as a test bed for knowledge extraction and reasoning. Through extensive experiments across various model sizes (2B-8B parameters) and training datasets, we find that R2L models can significantly outperform L2R mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18435","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18435/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18435","created_at":"2026-07-05T11:28:34.407652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18435v3","created_at":"2026-07-05T11:28:34.407652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18435","created_at":"2026-07-05T11:28:34.407652+00:00"},{"alias_kind":"pith_short_12","alias_value":"W65RQQ6OA7FE","created_at":"2026-07-05T11:28:34.407652+00:00"},{"alias_kind":"pith_short_16","alias_value":"W65RQQ6OA7FE32AC","created_at":"2026-07-05T11:28:34.407652+00:00"},{"alias_kind":"pith_short_8","alias_value":"W65RQQ6O","created_at":"2026-07-05T11:28:34.407652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.08739","citing_title":"Probability Consistency in Large Language Models: Theoretical Foundations Meet Empirical Discrepancies","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V","json":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V.json","graph_json":"https://pith.science/api/pith-number/W65RQQ6OA7FE32ACDHRBF4ST5V/graph.json","events_json":"https://pith.science/api/pith-number/W65RQQ6OA7FE32ACDHRBF4ST5V/events.json","paper":"https://pith.science/paper/W65RQQ6O"},"agent_actions":{"view_html":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V","download_json":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V.json","view_paper":"https://pith.science/paper/W65RQQ6O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18435&json=true","fetch_graph":"https://pith.science/api/pith-number/W65RQQ6OA7FE32ACDHRBF4ST5V/graph.json","fetch_events":"https://pith.science/api/pith-number/W65RQQ6OA7FE32ACDHRBF4ST5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V/action/storage_attestation","attest_author":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V/action/author_attestation","sign_citation":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V/action/citation_signature","submit_replication":"https://pith.science/pith/W65RQQ6OA7FE32ACDHRBF4ST5V/action/replication_record"}},"created_at":"2026-07-05T11:28:34.407652+00:00","updated_at":"2026-07-05T11:28:34.407652+00:00"}