{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:N47GMDVJ2UJHMZHPUJTEY5NBOU","short_pith_number":"pith:N47GMDVJ","schema_version":"1.0","canonical_sha256":"6f3e660ea9d5127664efa2664c75a1753bb5e03fae5b2d6dbd20b0585a7c0b88","source":{"kind":"arxiv","id":"1804.08771","version":2},"attestation_state":"computed","paper":{"title":"A Call for Clarity in Reporting BLEU Scores","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Matt Post","submitted_at":"2018-04-23T22:54:55Z","abstract_excerpt":"The field of machine translation faces an under-recognized problem because of inconsistency in the reporting of scores from its dominant metric. Although people refer to \"the\" BLEU score, BLEU is in fact a parameterized metric whose values can vary wildly with changes to these parameters. These parameters are often not reported or are hard to find, and consequently, BLEU scores between papers cannot be directly compared. I quantify this variation, finding differences as high as 1.8 between commonly used configurations. The main culprit is different tokenization and normalization schemes applie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1804.08771","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-04-23T22:54:55Z","cross_cats_sorted":[],"title_canon_sha256":"1e57461cd57324d3923f04e19b214675e054d00ecda9130e8a4dafbed059f682","abstract_canon_sha256":"9ac394f4ca39d0cad515b6e620b94f88bad9fea6f7070cd804f13302d6761ed1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:05:56.135370Z","signature_b64":"9CL87ADI88VIYCD8i3A5L+fiQ28yIyjsqPmk6m8YUlBRXOZQQE+PE0Y5c8khl/HK7QTaGBzj7GVjme7rCj13Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f3e660ea9d5127664efa2664c75a1753bb5e03fae5b2d6dbd20b0585a7c0b88","last_reissued_at":"2026-05-18T00:05:56.134896Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:05:56.134896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Call for Clarity in Reporting BLEU Scores","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Matt Post","submitted_at":"2018-04-23T22:54:55Z","abstract_excerpt":"The field of machine translation faces an under-recognized problem because of inconsistency in the reporting of scores from its dominant metric. Although people refer to \"the\" BLEU score, BLEU is in fact a parameterized metric whose values can vary wildly with changes to these parameters. These parameters are often not reported or are hard to find, and consequently, BLEU scores between papers cannot be directly compared. I quantify this variation, finding differences as high as 1.8 between commonly used configurations. The main culprit is different tokenization and normalization schemes applie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1804.08771","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1804.08771","created_at":"2026-05-18T00:05:56.134965+00:00"},{"alias_kind":"arxiv_version","alias_value":"1804.08771v2","created_at":"2026-05-18T00:05:56.134965+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1804.08771","created_at":"2026-05-18T00:05:56.134965+00:00"},{"alias_kind":"pith_short_12","alias_value":"N47GMDVJ2UJH","created_at":"2026-05-18T12:32:40.477152+00:00"},{"alias_kind":"pith_short_16","alias_value":"N47GMDVJ2UJHMZHP","created_at":"2026-05-18T12:32:40.477152+00:00"},{"alias_kind":"pith_short_8","alias_value":"N47GMDVJ","created_at":"2026-05-18T12:32:40.477152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":6,"sample":[{"citing_arxiv_id":"2606.02258","citing_title":"Matter to Mechanism: A Benchmark for AI Co-Scientists in Materials and Battery Research","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"1906.08393","citing_title":"Robust Machine Translation with Domain Sensitive Pseudo-Sources: Baidu-OSU WMT19 MT Robustness Shared Task System Report","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"1906.08885","citing_title":"Low-Resource Corpus Filtering using Multilingual Sentence Embeddings","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"1906.12284","citing_title":"Widening the Representation Bottleneck in Neural Machine Translation with Lexical Shortcuts","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20912","citing_title":"Enhancing Scientific Discourse: Machine Translation for the Scientific Domain","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"1910.07467","citing_title":"Root Mean Square Layer Normalization","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"1808.06226","citing_title":"SentencePiece: A simple and language independent subword tokenizer and detokenizer for Neural Text Processing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"1910.10683","citing_title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20878","citing_title":"AITP: Traffic Accident Responsibility Allocation via Multimodal Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2005.14165","citing_title":"Language Models are Few-Shot Learners","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15390","citing_title":"Analyzing Chain of Thought (CoT) Approaches in Control Flow Code Deobfuscation Tasks","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU","json":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU.json","graph_json":"https://pith.science/api/pith-number/N47GMDVJ2UJHMZHPUJTEY5NBOU/graph.json","events_json":"https://pith.science/api/pith-number/N47GMDVJ2UJHMZHPUJTEY5NBOU/events.json","paper":"https://pith.science/paper/N47GMDVJ"},"agent_actions":{"view_html":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU","download_json":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU.json","view_paper":"https://pith.science/paper/N47GMDVJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1804.08771&json=true","fetch_graph":"https://pith.science/api/pith-number/N47GMDVJ2UJHMZHPUJTEY5NBOU/graph.json","fetch_events":"https://pith.science/api/pith-number/N47GMDVJ2UJHMZHPUJTEY5NBOU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU/action/storage_attestation","attest_author":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU/action/author_attestation","sign_citation":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU/action/citation_signature","submit_replication":"https://pith.science/pith/N47GMDVJ2UJHMZHPUJTEY5NBOU/action/replication_record"}},"created_at":"2026-05-18T00:05:56.134965+00:00","updated_at":"2026-05-18T00:05:56.134965+00:00"}