{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:J5ZPVKHFSWZR33Y56CQYD43QRV","short_pith_number":"pith:J5ZPVKHF","schema_version":"1.0","canonical_sha256":"4f72faa8e595b31def1df0a181f3708d77d78c1c70aad13b98d901d29904beed","source":{"kind":"arxiv","id":"2004.04696","version":5},"attestation_state":"computed","paper":{"title":"BLEURT: Learning Robust Metrics for Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankur P. Parikh, Dipanjan Das, Thibault Sellam","submitted_at":"2020-04-09T17:26:52Z","abstract_excerpt":"Text generation has made significant advances in the last few years. Yet, evaluation metrics have lagged behind, as the most popular choices (e.g., BLEU and ROUGE) may correlate poorly with human judgments. We propose BLEURT, a learned evaluation metric based on BERT that can model human judgments with a few thousand possibly biased training examples. A key aspect of our approach is a novel pre-training scheme that uses millions of synthetic examples to help the model generalize. BLEURT provides state-of-the-art results on the last three years of the WMT Metrics shared task and the WebNLG Comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.04696","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-09T17:26:52Z","cross_cats_sorted":[],"title_canon_sha256":"1255a3a351f709ca33ff001ec1ee0c2d0d807b43ccd20cf72843458d723749a0","abstract_canon_sha256":"3e4d6acea5fa3df82b98e4abcf72dd6166ad1f174d4a9064a21046f5a48c0efe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:04:48.676778Z","signature_b64":"f3lwdo2RvylRaUo36efqQNaRryP/U7XCNRwyMIExJSGfC+IA3tFwfKAcnbEeqNi1selTet/zLLNwWIyB3aHzCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f72faa8e595b31def1df0a181f3708d77d78c1c70aad13b98d901d29904beed","last_reissued_at":"2026-07-05T01:04:48.676001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:04:48.676001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BLEURT: Learning Robust Metrics for Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankur P. Parikh, Dipanjan Das, Thibault Sellam","submitted_at":"2020-04-09T17:26:52Z","abstract_excerpt":"Text generation has made significant advances in the last few years. Yet, evaluation metrics have lagged behind, as the most popular choices (e.g., BLEU and ROUGE) may correlate poorly with human judgments. We propose BLEURT, a learned evaluation metric based on BERT that can model human judgments with a few thousand possibly biased training examples. A key aspect of our approach is a novel pre-training scheme that uses millions of synthetic examples to help the model generalize. BLEURT provides state-of-the-art results on the last three years of the WMT Metrics shared task and the WebNLG Comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.04696","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.04696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.04696","created_at":"2026-07-05T01:04:48.676297+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.04696v5","created_at":"2026-07-05T01:04:48.676297+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.04696","created_at":"2026-07-05T01:04:48.676297+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5ZPVKHFSWZR","created_at":"2026-07-05T01:04:48.676297+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5ZPVKHFSWZR33Y5","created_at":"2026-07-05T01:04:48.676297+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5ZPVKHF","created_at":"2026-07-05T01:04:48.676297+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12198","citing_title":"LLM-Based User Personas for Recommendations at Scale","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16737","citing_title":"Secure LLM Fine-Tuning via Safety-Aware Probing","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2512.24366","citing_title":"On the Factual Consistency of Text-based Explainable Recommendation Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.15907","citing_title":"TabReX : Tabular Referenceless eXplainable Evaluation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03666","citing_title":"MMP-Refer: Multimodal Path Retrieval-augmented LLMs For Explainable Recommendation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03724","citing_title":"Rank, Don't Generate: Statement-level Ranking for Explainable Recommendation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2308.07201","citing_title":"ChatEval: Towards Better LLM-based Evaluators through Multi-Agent Debate","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05902","citing_title":"Evaluating Non-English Developer Support in Machine Learning for Software Engineering","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20334","citing_title":"An Explainable Approach to Document-level Translation Evaluation with Topic Modeling","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10470","citing_title":"From Query to Counsel: Structured Reasoning with a Multi-Agent Framework and Dataset for Legal Consultation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV","json":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV.json","graph_json":"https://pith.science/api/pith-number/J5ZPVKHFSWZR33Y56CQYD43QRV/graph.json","events_json":"https://pith.science/api/pith-number/J5ZPVKHFSWZR33Y56CQYD43QRV/events.json","paper":"https://pith.science/paper/J5ZPVKHF"},"agent_actions":{"view_html":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV","download_json":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV.json","view_paper":"https://pith.science/paper/J5ZPVKHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.04696&json=true","fetch_graph":"https://pith.science/api/pith-number/J5ZPVKHFSWZR33Y56CQYD43QRV/graph.json","fetch_events":"https://pith.science/api/pith-number/J5ZPVKHFSWZR33Y56CQYD43QRV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV/action/storage_attestation","attest_author":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV/action/author_attestation","sign_citation":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV/action/citation_signature","submit_replication":"https://pith.science/pith/J5ZPVKHFSWZR33Y56CQYD43QRV/action/replication_record"}},"created_at":"2026-07-05T01:04:48.676297+00:00","updated_at":"2026-07-05T01:04:48.676297+00:00"}