{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LBHME7CPVK3SDFH6NBDQ4WF6MI","short_pith_number":"pith:LBHME7CP","schema_version":"1.0","canonical_sha256":"584ec27c4faab72194fe68470e58be62210180dfe22136e90c32894f665e346b","source":{"kind":"arxiv","id":"2607.19226","version":1},"attestation_state":"computed","paper":{"title":"The Price of Reasoning: Cost-Quality Tradeoffs in Reinforcement Learning for Neural Machine Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aixiu An, Michael Jungo","submitted_at":"2026-07-21T15:57:36Z","abstract_excerpt":"Reinforcement learning with verifiable rewards (RLVR) has been established as a viable paradigm for the post-training of Large Language Models (LLMs), including downstream tasks, such as Neural Machine Translation (NMT). With the latest research indicating that RLVR could be the preferred training method for translating legal documents due to the induced reasoning capabilities, it raises the question whether it is really attributed to the reasoning or more generally to the training paradigm. We investigate the importance of including the model's reasoning trace in the generated responses durin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.19226","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-21T15:57:36Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"06899e2fd7f0e1028c04782e1e420241728e05aa953b715a3447f12862c97d36","abstract_canon_sha256":"b5fd477d15921853a1483aac7e78744c501d075e3fdaf9c44368d9b9bd28c2bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-22T01:24:15.280287Z","signature_b64":"wUlH6Jt9OAO/UFCM6DlMzhZeWIP1FVbwryzhTfieqXrG5GzQRwW4HkZyFlAUcKaDGFG3KlZJsYyBVVt3kT+PBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"584ec27c4faab72194fe68470e58be62210180dfe22136e90c32894f665e346b","last_reissued_at":"2026-07-22T01:24:15.279554Z","signature_status":"signed_v1","first_computed_at":"2026-07-22T01:24:15.279554Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Price of Reasoning: Cost-Quality Tradeoffs in Reinforcement Learning for Neural Machine Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aixiu An, Michael Jungo","submitted_at":"2026-07-21T15:57:36Z","abstract_excerpt":"Reinforcement learning with verifiable rewards (RLVR) has been established as a viable paradigm for the post-training of Large Language Models (LLMs), including downstream tasks, such as Neural Machine Translation (NMT). With the latest research indicating that RLVR could be the preferred training method for translating legal documents due to the induced reasoning capabilities, it raises the question whether it is really attributed to the reasoning or more generally to the training paradigm. We investigate the importance of including the model's reasoning trace in the generated responses durin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19226","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.19226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.19226","created_at":"2026-07-22T01:24:15.279965+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.19226v1","created_at":"2026-07-22T01:24:15.279965+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19226","created_at":"2026-07-22T01:24:15.279965+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBHME7CPVK3S","created_at":"2026-07-22T01:24:15.279965+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBHME7CPVK3SDFH6","created_at":"2026-07-22T01:24:15.279965+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBHME7CP","created_at":"2026-07-22T01:24:15.279965+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI","json":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI.json","graph_json":"https://pith.science/api/pith-number/LBHME7CPVK3SDFH6NBDQ4WF6MI/graph.json","events_json":"https://pith.science/api/pith-number/LBHME7CPVK3SDFH6NBDQ4WF6MI/events.json","paper":"https://pith.science/paper/LBHME7CP"},"agent_actions":{"view_html":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI","download_json":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI.json","view_paper":"https://pith.science/paper/LBHME7CP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.19226&json=true","fetch_graph":"https://pith.science/api/pith-number/LBHME7CPVK3SDFH6NBDQ4WF6MI/graph.json","fetch_events":"https://pith.science/api/pith-number/LBHME7CPVK3SDFH6NBDQ4WF6MI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI/action/storage_attestation","attest_author":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI/action/author_attestation","sign_citation":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI/action/citation_signature","submit_replication":"https://pith.science/pith/LBHME7CPVK3SDFH6NBDQ4WF6MI/action/replication_record"}},"created_at":"2026-07-22T01:24:15.279965+00:00","updated_at":"2026-07-22T01:24:15.279965+00:00"}