{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZWYXJMNC7WSPOS4SGJJTSFAGMK","short_pith_number":"pith:ZWYXJMNC","schema_version":"1.0","canonical_sha256":"cdb174b1a2fda4f74b9232533914066284795b31b4c2a264d6507a7774057b65","source":{"kind":"arxiv","id":"2505.05408","version":1},"attestation_state":"computed","paper":{"title":"Crosslingual Reasoning through Test-Time Scaling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Carsten Eickhoff, Genta Indra Winata, Jonibek Mansurov, Julia Kreutzer, M. Farid Adilazuarda, Niklas Muennighoff, Ruochen Zhang, Stephen H. Bach, Zheng-Xin Yong","submitted_at":"2025-05-08T16:50:06Z","abstract_excerpt":"Reasoning capabilities of large language models are primarily studied for English, even when pretrained models are multilingual. In this work, we investigate to what extent English reasoning finetuning with long chain-of-thoughts (CoTs) can generalize across languages. First, we find that scaling up inference compute for English-centric reasoning language models (RLMs) improves multilingual mathematical reasoning across many languages including low-resource languages, to an extent where they outperform models twice their size. Second, we reveal that while English-centric RLM's CoTs are natural"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.05408","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-08T16:50:06Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"857b7d1d2b44f9a0ab05b669ec9489c91f6b648d64cea84a86b619cc7d6b14ab","abstract_canon_sha256":"6c300f099924a37ba6410a9855a26c667a652d1e85455bc0e024542057a018f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:23.828833Z","signature_b64":"34iiLOlj11zLlxpw5+1H11Sv0hc1AL74QgTorYbyNTDoc8VMOYhRJuTNwEcPeyV4VHWhQolIxoSzDDM65y5UDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cdb174b1a2fda4f74b9232533914066284795b31b4c2a264d6507a7774057b65","last_reissued_at":"2026-07-05T11:00:23.828248Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:23.828248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Crosslingual Reasoning through Test-Time Scaling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Carsten Eickhoff, Genta Indra Winata, Jonibek Mansurov, Julia Kreutzer, M. Farid Adilazuarda, Niklas Muennighoff, Ruochen Zhang, Stephen H. Bach, Zheng-Xin Yong","submitted_at":"2025-05-08T16:50:06Z","abstract_excerpt":"Reasoning capabilities of large language models are primarily studied for English, even when pretrained models are multilingual. In this work, we investigate to what extent English reasoning finetuning with long chain-of-thoughts (CoTs) can generalize across languages. First, we find that scaling up inference compute for English-centric reasoning language models (RLMs) improves multilingual mathematical reasoning across many languages including low-resource languages, to an extent where they outperform models twice their size. Second, we reveal that while English-centric RLM's CoTs are natural"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.05408","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.05408/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.05408","created_at":"2026-07-05T11:00:23.828302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.05408v1","created_at":"2026-07-05T11:00:23.828302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.05408","created_at":"2026-07-05T11:00:23.828302+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZWYXJMNC7WSP","created_at":"2026-07-05T11:00:23.828302+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZWYXJMNC7WSPOS4S","created_at":"2026-07-05T11:00:23.828302+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZWYXJMNC","created_at":"2026-07-05T11:00:23.828302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26466","citing_title":"Soft Token Alignment for Cross-Lingual Reasoning","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14990","citing_title":"Language Specific Knowledge: Do Models Know Better in X than in English?","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22567","citing_title":"LANG: Reinforcement Learning for Multilingual Reasoning with Language-Adaptive Hint Guidance","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19173","citing_title":"Prompting language influences diagnostic reasoning and accuracy of large language models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21593","citing_title":"Language as a Latent Variable for Reasoning Optimization","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK","json":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK.json","graph_json":"https://pith.science/api/pith-number/ZWYXJMNC7WSPOS4SGJJTSFAGMK/graph.json","events_json":"https://pith.science/api/pith-number/ZWYXJMNC7WSPOS4SGJJTSFAGMK/events.json","paper":"https://pith.science/paper/ZWYXJMNC"},"agent_actions":{"view_html":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK","download_json":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK.json","view_paper":"https://pith.science/paper/ZWYXJMNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.05408&json=true","fetch_graph":"https://pith.science/api/pith-number/ZWYXJMNC7WSPOS4SGJJTSFAGMK/graph.json","fetch_events":"https://pith.science/api/pith-number/ZWYXJMNC7WSPOS4SGJJTSFAGMK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK/action/storage_attestation","attest_author":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK/action/author_attestation","sign_citation":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK/action/citation_signature","submit_replication":"https://pith.science/pith/ZWYXJMNC7WSPOS4SGJJTSFAGMK/action/replication_record"}},"created_at":"2026-07-05T11:00:23.828302+00:00","updated_at":"2026-07-05T11:00:23.828302+00:00"}