{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K4W2JHZF3GH3I4F6YYFUPLDFJF","short_pith_number":"pith:K4W2JHZF","schema_version":"1.0","canonical_sha256":"572da49f25d98fb470bec60b47ac654966fd200ba364478cc220776881690edc","source":{"kind":"arxiv","id":"2410.18653","version":3},"attestation_state":"computed","paper":{"title":"Towards Better Open-Ended Text Generation: A Multicriteria Evaluation Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christian Heumann, Esteban Garces Arias, Hannah Blocher, Julian Rodemann, Matthias A{\\ss}enmacher, Meimingwei Li","submitted_at":"2024-10-24T11:32:01Z","abstract_excerpt":"Open-ended text generation has become a prominent task in natural language processing due to the rise of powerful (large) language models. However, evaluating the quality of these models and the employed decoding strategies remains challenging due to trade-offs among widely used metrics such as coherence, diversity, and perplexity. This paper addresses the specific problem of multicriteria evaluation for open-ended text generation, proposing novel methods for both relative and absolute rankings of decoding methods. Specifically, we employ benchmarking approaches based on partial orderings and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18653","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:32:01Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ac68fe455c9920d22db3384d429541472d02079cb95d207102510146810858e7","abstract_canon_sha256":"0351d95556f27bc842a0e4e73fafc9516fbd77f4aa3ab38efde17b4844f7fe05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:07.486523Z","signature_b64":"5snES5aOXgvnkBtmjFhsAOZ5lYvqrejJsMm47GgInSfO61XE+SCfYUh7JZcVtv04TQo/oMJfiq45FLHiOXQLCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"572da49f25d98fb470bec60b47ac654966fd200ba364478cc220776881690edc","last_reissued_at":"2026-07-05T11:23:07.485979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:07.485979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Better Open-Ended Text Generation: A Multicriteria Evaluation Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christian Heumann, Esteban Garces Arias, Hannah Blocher, Julian Rodemann, Matthias A{\\ss}enmacher, Meimingwei Li","submitted_at":"2024-10-24T11:32:01Z","abstract_excerpt":"Open-ended text generation has become a prominent task in natural language processing due to the rise of powerful (large) language models. However, evaluating the quality of these models and the employed decoding strategies remains challenging due to trade-offs among widely used metrics such as coherence, diversity, and perplexity. This paper addresses the specific problem of multicriteria evaluation for open-ended text generation, proposing novel methods for both relative and absolute rankings of decoding methods. Specifically, we employ benchmarking approaches based on partial orderings and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18653","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18653/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18653","created_at":"2026-07-05T11:23:07.486034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18653v3","created_at":"2026-07-05T11:23:07.486034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18653","created_at":"2026-07-05T11:23:07.486034+00:00"},{"alias_kind":"pith_short_12","alias_value":"K4W2JHZF3GH3","created_at":"2026-07-05T11:23:07.486034+00:00"},{"alias_kind":"pith_short_16","alias_value":"K4W2JHZF3GH3I4F6","created_at":"2026-07-05T11:23:07.486034+00:00"},{"alias_kind":"pith_short_8","alias_value":"K4W2JHZF","created_at":"2026-07-05T11:23:07.486034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18082","citing_title":"Statistical Multicriteria Evaluation of LLM-Generated Text","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF","json":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF.json","graph_json":"https://pith.science/api/pith-number/K4W2JHZF3GH3I4F6YYFUPLDFJF/graph.json","events_json":"https://pith.science/api/pith-number/K4W2JHZF3GH3I4F6YYFUPLDFJF/events.json","paper":"https://pith.science/paper/K4W2JHZF"},"agent_actions":{"view_html":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF","download_json":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF.json","view_paper":"https://pith.science/paper/K4W2JHZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18653&json=true","fetch_graph":"https://pith.science/api/pith-number/K4W2JHZF3GH3I4F6YYFUPLDFJF/graph.json","fetch_events":"https://pith.science/api/pith-number/K4W2JHZF3GH3I4F6YYFUPLDFJF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF/action/storage_attestation","attest_author":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF/action/author_attestation","sign_citation":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF/action/citation_signature","submit_replication":"https://pith.science/pith/K4W2JHZF3GH3I4F6YYFUPLDFJF/action/replication_record"}},"created_at":"2026-07-05T11:23:07.486034+00:00","updated_at":"2026-07-05T11:23:07.486034+00:00"}