{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LMLCV4HIESICXHXKW2EDENCJWO","short_pith_number":"pith:LMLCV4HI","schema_version":"1.0","canonical_sha256":"5b162af0e824902b9eeab688323449b3bed11cb0ca8912b10295b50a1df8d3a7","source":{"kind":"arxiv","id":"2507.06221","version":1},"attestation_state":"computed","paper":{"title":"Aligned Textual Scoring Rules","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.AI","authors_text":"Jason Hartline, Michael J. Curry, Yifan Wu, Yuxuan Lu","submitted_at":"2025-07-08T17:53:22Z","abstract_excerpt":"Scoring rules elicit probabilistic predictions from a strategic agent by scoring the prediction against a ground truth state. A scoring rule is proper if, from the agent's perspective, reporting the true belief maximizes the expected score. With the development of language models, Wu and Hartline (2024) proposes a reduction from textual information elicitation to the numerical (i.e. probabilistic) information elicitation problem, which achieves provable properness for textual elicitation. However, not all proper scoring rules are well aligned with human preference over text. Our paper designs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06221","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-08T17:53:22Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"a6aaec7b32f59993d308f7dc9459cbe58695df0670fb4f11f0061ff4ce6a7025","abstract_canon_sha256":"a8a158dd80c0b77ae17607317e703db76ccf9cb8355803effe64e7c077bb96b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:54.480618Z","signature_b64":"Qp2ewvOAm+sJEuaStTGJE4a5k8GJzJd5NcD8iBT4JP34oBvq6mmzjYqQl4LPq47W869UTdVjL36BwZOSZAAoDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b162af0e824902b9eeab688323449b3bed11cb0ca8912b10295b50a1df8d3a7","last_reissued_at":"2026-07-05T11:33:54.480060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:54.480060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligned Textual Scoring Rules","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.AI","authors_text":"Jason Hartline, Michael J. Curry, Yifan Wu, Yuxuan Lu","submitted_at":"2025-07-08T17:53:22Z","abstract_excerpt":"Scoring rules elicit probabilistic predictions from a strategic agent by scoring the prediction against a ground truth state. A scoring rule is proper if, from the agent's perspective, reporting the true belief maximizes the expected score. With the development of language models, Wu and Hartline (2024) proposes a reduction from textual information elicitation to the numerical (i.e. probabilistic) information elicitation problem, which achieves provable properness for textual elicitation. However, not all proper scoring rules are well aligned with human preference over text. Our paper designs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06221","created_at":"2026-07-05T11:33:54.480121+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06221v1","created_at":"2026-07-05T11:33:54.480121+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06221","created_at":"2026-07-05T11:33:54.480121+00:00"},{"alias_kind":"pith_short_12","alias_value":"LMLCV4HIESIC","created_at":"2026-07-05T11:33:54.480121+00:00"},{"alias_kind":"pith_short_16","alias_value":"LMLCV4HIESICXHXK","created_at":"2026-07-05T11:33:54.480121+00:00"},{"alias_kind":"pith_short_8","alias_value":"LMLCV4HI","created_at":"2026-07-05T11:33:54.480121+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01423","citing_title":"Scoring Rules! Statistical and Strategic Alignment for Text Evaluation Metrics","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO","json":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO.json","graph_json":"https://pith.science/api/pith-number/LMLCV4HIESICXHXKW2EDENCJWO/graph.json","events_json":"https://pith.science/api/pith-number/LMLCV4HIESICXHXKW2EDENCJWO/events.json","paper":"https://pith.science/paper/LMLCV4HI"},"agent_actions":{"view_html":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO","download_json":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO.json","view_paper":"https://pith.science/paper/LMLCV4HI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06221&json=true","fetch_graph":"https://pith.science/api/pith-number/LMLCV4HIESICXHXKW2EDENCJWO/graph.json","fetch_events":"https://pith.science/api/pith-number/LMLCV4HIESICXHXKW2EDENCJWO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO/action/storage_attestation","attest_author":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO/action/author_attestation","sign_citation":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO/action/citation_signature","submit_replication":"https://pith.science/pith/LMLCV4HIESICXHXKW2EDENCJWO/action/replication_record"}},"created_at":"2026-07-05T11:33:54.480121+00:00","updated_at":"2026-07-05T11:33:54.480121+00:00"}