{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LI2PZCISOCSBVA6PSXN5OVQQLI","short_pith_number":"pith:LI2PZCIS","schema_version":"1.0","canonical_sha256":"5a34fc891270a41a83cf95dbd756105a2340fff01fb9732798435fe2fe6b247a","source":{"kind":"arxiv","id":"2608.01423","version":1},"attestation_state":"computed","paper":{"title":"Scoring Rules! Statistical and Strategic Alignment for Text Evaluation Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT","cs.LG"],"primary_cat":"cs.AI","authors_text":"Grant Schoenebeck, Jason Hartline, Shengwei Xu, Yifan Wu, Yuxuan Lu","submitted_at":"2026-08-02T18:10:01Z","abstract_excerpt":"Reference-based text evaluation metrics, which are widely used to assess natural language generation systems, score a candidate response by comparing it with a reference response. The reliability of an evaluation metric is usually judged by its statistical correlation with human ratings. However, as these metrics are increasingly used as optimization objectives, correlation alone is no longer sufficient: agents may strategically game the evaluation metric. We study this issue through two complementary notions of alignment. A metric is statistically aligned if it correlates with human ratings a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.01423","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-02T18:10:01Z","cross_cats_sorted":["cs.GT","cs.LG"],"title_canon_sha256":"1df4e6b9b6fcb1aa4c82c83c792e2017f36ea942f7a6b31b391c7ffdd33397db","abstract_canon_sha256":"48b382b6365980381915f55be1bcafee479499a9a87908d791c9f011f77dc261"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:05:11.993871Z","signature_b64":"xcvmAzAqnqMM9EE/xuh2MYUuN/5W7/s8KhetPxZvvIK5VPwqaNwUMC4W84+EwOXEAbxwmqDOO75i7cBhTX3mDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a34fc891270a41a83cf95dbd756105a2340fff01fb9732798435fe2fe6b247a","last_reissued_at":"2026-08-04T02:05:11.992355Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:05:11.992355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scoring Rules! Statistical and Strategic Alignment for Text Evaluation Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT","cs.LG"],"primary_cat":"cs.AI","authors_text":"Grant Schoenebeck, Jason Hartline, Shengwei Xu, Yifan Wu, Yuxuan Lu","submitted_at":"2026-08-02T18:10:01Z","abstract_excerpt":"Reference-based text evaluation metrics, which are widely used to assess natural language generation systems, score a candidate response by comparing it with a reference response. The reliability of an evaluation metric is usually judged by its statistical correlation with human ratings. However, as these metrics are increasingly used as optimization objectives, correlation alone is no longer sufficient: agents may strategically game the evaluation metric. We study this issue through two complementary notions of alignment. A metric is statistically aligned if it correlates with human ratings a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.01423","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.01423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.01423","created_at":"2026-08-04T02:05:11.993782+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.01423v1","created_at":"2026-08-04T02:05:11.993782+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.01423","created_at":"2026-08-04T02:05:11.993782+00:00"},{"alias_kind":"pith_short_12","alias_value":"LI2PZCISOCSB","created_at":"2026-08-04T02:05:11.993782+00:00"},{"alias_kind":"pith_short_16","alias_value":"LI2PZCISOCSBVA6P","created_at":"2026-08-04T02:05:11.993782+00:00"},{"alias_kind":"pith_short_8","alias_value":"LI2PZCIS","created_at":"2026-08-04T02:05:11.993782+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI","json":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI.json","graph_json":"https://pith.science/api/pith-number/LI2PZCISOCSBVA6PSXN5OVQQLI/graph.json","events_json":"https://pith.science/api/pith-number/LI2PZCISOCSBVA6PSXN5OVQQLI/events.json","paper":"https://pith.science/paper/LI2PZCIS"},"agent_actions":{"view_html":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI","download_json":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI.json","view_paper":"https://pith.science/paper/LI2PZCIS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.01423&json=true","fetch_graph":"https://pith.science/api/pith-number/LI2PZCISOCSBVA6PSXN5OVQQLI/graph.json","fetch_events":"https://pith.science/api/pith-number/LI2PZCISOCSBVA6PSXN5OVQQLI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI/action/storage_attestation","attest_author":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI/action/author_attestation","sign_citation":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI/action/citation_signature","submit_replication":"https://pith.science/pith/LI2PZCISOCSBVA6PSXN5OVQQLI/action/replication_record"}},"created_at":"2026-08-04T02:05:11.993782+00:00","updated_at":"2026-08-04T02:05:11.993782+00:00"}