{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CLGMRKHV3T2MNYVF544EKNMNQ7","short_pith_number":"pith:CLGMRKHV","schema_version":"1.0","canonical_sha256":"12ccc8a8f5dcf4c6e2a5ef3845358d87f68d903f36fb3e591d1900c67296fb4a","source":{"kind":"arxiv","id":"2505.15508","version":1},"attestation_state":"computed","paper":{"title":"Multilingual Test-Time Scaling via Initial Thought Transfer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Prasoon Bajpai, Tanmoy Chakraborty","submitted_at":"2025-05-21T13:27:38Z","abstract_excerpt":"Test-time scaling has emerged as a widely adopted inference-time strategy for boosting reasoning performance. However, its effectiveness has been studied almost exclusively in English, leaving its behavior in other languages largely unexplored. We present the first systematic study of test-time scaling in multilingual settings, evaluating DeepSeek-R1-Distill-LLama-8B and DeepSeek-R1-Distill-Qwen-7B across both high- and low-resource Latin-script languages. Our findings reveal that the relative gains from test-time scaling vary significantly across languages. Additionally, models frequently swi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15508","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T13:27:38Z","cross_cats_sorted":[],"title_canon_sha256":"5aeee2c4ac12c59b7819f216f87a679b63010372f346aa149ac8be4eae5eda60","abstract_canon_sha256":"3bd041eb4b75d4a387ef28e7b4f5f121a765419dc68bb7676ababb13ba3d72b5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:46.932130Z","signature_b64":"ccXXNoZuRSLZ4QXxqGnzP/565D1qC16uH7R47qPysYKvafNT6dqhjibgmpD6meH01PmUsbkMnb9WWQUv8z6QAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12ccc8a8f5dcf4c6e2a5ef3845358d87f68d903f36fb3e591d1900c67296fb4a","last_reissued_at":"2026-07-05T11:06:46.931585Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:46.931585Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multilingual Test-Time Scaling via Initial Thought Transfer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Prasoon Bajpai, Tanmoy Chakraborty","submitted_at":"2025-05-21T13:27:38Z","abstract_excerpt":"Test-time scaling has emerged as a widely adopted inference-time strategy for boosting reasoning performance. However, its effectiveness has been studied almost exclusively in English, leaving its behavior in other languages largely unexplored. We present the first systematic study of test-time scaling in multilingual settings, evaluating DeepSeek-R1-Distill-LLama-8B and DeepSeek-R1-Distill-Qwen-7B across both high- and low-resource Latin-script languages. Our findings reveal that the relative gains from test-time scaling vary significantly across languages. Additionally, models frequently swi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15508","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15508","created_at":"2026-07-05T11:06:46.931651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15508v1","created_at":"2026-07-05T11:06:46.931651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15508","created_at":"2026-07-05T11:06:46.931651+00:00"},{"alias_kind":"pith_short_12","alias_value":"CLGMRKHV3T2M","created_at":"2026-07-05T11:06:46.931651+00:00"},{"alias_kind":"pith_short_16","alias_value":"CLGMRKHV3T2MNYVF","created_at":"2026-07-05T11:06:46.931651+00:00"},{"alias_kind":"pith_short_8","alias_value":"CLGMRKHV","created_at":"2026-07-05T11:06:46.931651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22567","citing_title":"LANG: Reinforcement Learning for Multilingual Reasoning with Language-Adaptive Hint Guidance","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7","json":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7.json","graph_json":"https://pith.science/api/pith-number/CLGMRKHV3T2MNYVF544EKNMNQ7/graph.json","events_json":"https://pith.science/api/pith-number/CLGMRKHV3T2MNYVF544EKNMNQ7/events.json","paper":"https://pith.science/paper/CLGMRKHV"},"agent_actions":{"view_html":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7","download_json":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7.json","view_paper":"https://pith.science/paper/CLGMRKHV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15508&json=true","fetch_graph":"https://pith.science/api/pith-number/CLGMRKHV3T2MNYVF544EKNMNQ7/graph.json","fetch_events":"https://pith.science/api/pith-number/CLGMRKHV3T2MNYVF544EKNMNQ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7/action/storage_attestation","attest_author":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7/action/author_attestation","sign_citation":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7/action/citation_signature","submit_replication":"https://pith.science/pith/CLGMRKHV3T2MNYVF544EKNMNQ7/action/replication_record"}},"created_at":"2026-07-05T11:06:46.931651+00:00","updated_at":"2026-07-05T11:06:46.931651+00:00"}