{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LIGLZA36TBD7WEHQKW3A2WF6AW","short_pith_number":"pith:LIGLZA36","schema_version":"1.0","canonical_sha256":"5a0cbc837e9847fb10f055b60d58be05957de5ea84b169b444a48d06e2cf826f","source":{"kind":"arxiv","id":"2101.06561","version":4},"attestation_state":"computed","paper":{"title":"GENIE: Toward Reproducible and Standardized Human Evaluation for Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Khashabi, Daniel S. Weld, Gabriel Stanovsky, Jonathan Bragg, Jungo Kasai, Nicholas Lourie, Noah A. Smith, Yejin Choi","submitted_at":"2021-01-17T00:40:47Z","abstract_excerpt":"While often assumed a gold standard, effective human evaluation of text generation remains an important, open area for research. We revisit this problem with a focus on producing consistent evaluations that are reproducible -- over time and across different populations. We study this goal in different stages of the human evaluation pipeline. In particular, we consider design choices for the annotation interface used to elicit human judgments and their impact on reproducibility. Furthermore, we develop an automated mechanism for maintaining annotator quality via a probabilistic model that detec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.06561","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-01-17T00:40:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a1ce6c238491f571b144a3accdc0d6a00baf2cc5e9aa074e13c1baa781c4ef0a","abstract_canon_sha256":"143a98940bf0db579cbfdcdb2357bf89233631295d9f6a4a1c631fabea5c2eb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:12:07.607602Z","signature_b64":"NCAALHNdPdUKZFtaYxuwpnq2qACqqypJwx8nzT9Ind8gh2eirO6AADrtj30r+buOzLmUD7FocYURkw0S68EICg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a0cbc837e9847fb10f055b60d58be05957de5ea84b169b444a48d06e2cf826f","last_reissued_at":"2026-07-05T05:12:07.607185Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:12:07.607185Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GENIE: Toward Reproducible and Standardized Human Evaluation for Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Khashabi, Daniel S. Weld, Gabriel Stanovsky, Jonathan Bragg, Jungo Kasai, Nicholas Lourie, Noah A. Smith, Yejin Choi","submitted_at":"2021-01-17T00:40:47Z","abstract_excerpt":"While often assumed a gold standard, effective human evaluation of text generation remains an important, open area for research. We revisit this problem with a focus on producing consistent evaluations that are reproducible -- over time and across different populations. We study this goal in different stages of the human evaluation pipeline. In particular, we consider design choices for the annotation interface used to elicit human judgments and their impact on reproducibility. Furthermore, we develop an automated mechanism for maintaining annotator quality via a probabilistic model that detec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.06561","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.06561/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.06561","created_at":"2026-07-05T05:12:07.607244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.06561v4","created_at":"2026-07-05T05:12:07.607244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.06561","created_at":"2026-07-05T05:12:07.607244+00:00"},{"alias_kind":"pith_short_12","alias_value":"LIGLZA36TBD7","created_at":"2026-07-05T05:12:07.607244+00:00"},{"alias_kind":"pith_short_16","alias_value":"LIGLZA36TBD7WEHQ","created_at":"2026-07-05T05:12:07.607244+00:00"},{"alias_kind":"pith_short_8","alias_value":"LIGLZA36","created_at":"2026-07-05T05:12:07.607244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.19221","citing_title":"Evaluating the Evaluators: Are readability metrics good measures of readability?","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW","json":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW.json","graph_json":"https://pith.science/api/pith-number/LIGLZA36TBD7WEHQKW3A2WF6AW/graph.json","events_json":"https://pith.science/api/pith-number/LIGLZA36TBD7WEHQKW3A2WF6AW/events.json","paper":"https://pith.science/paper/LIGLZA36"},"agent_actions":{"view_html":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW","download_json":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW.json","view_paper":"https://pith.science/paper/LIGLZA36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.06561&json=true","fetch_graph":"https://pith.science/api/pith-number/LIGLZA36TBD7WEHQKW3A2WF6AW/graph.json","fetch_events":"https://pith.science/api/pith-number/LIGLZA36TBD7WEHQKW3A2WF6AW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW/action/storage_attestation","attest_author":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW/action/author_attestation","sign_citation":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW/action/citation_signature","submit_replication":"https://pith.science/pith/LIGLZA36TBD7WEHQKW3A2WF6AW/action/replication_record"}},"created_at":"2026-07-05T05:12:07.607244+00:00","updated_at":"2026-07-05T05:12:07.607244+00:00"}