{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NVTZIWXWZHVZROCA5SBR7DEBXF","short_pith_number":"pith:NVTZIWXW","schema_version":"1.0","canonical_sha256":"6d67945af6c9eb98b840ec831f8c81b9537a6a975d9eab0d760e9b89899f7a36","source":{"kind":"arxiv","id":"2507.00769","version":1},"attestation_state":"computed","paper":{"title":"LitBench: A Benchmark and Dataset for Reliable Evaluation of Creative Writing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Fein, Kabir Jolly, Nick Haber, Rafael Rafailov, Sebastian Russo, Violet Xiang","submitted_at":"2025-07-01T14:10:36Z","abstract_excerpt":"Evaluating creative writing generated by large language models (LLMs) remains challenging because open-ended narratives lack ground truths. Without performant automated evaluation methods, off-the-shelf (OTS) language models are employed as zero-shot judges, yet their reliability is unclear in this context. In pursuit of robust evaluation for creative writing, we introduce LitBench, the first standardized benchmark and paired dataset for creative writing verification, comprising a held-out test set of 2,480 debiased, human-labeled story comparisons drawn from Reddit and a 43,827-pair training "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00769","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-01T14:10:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f86a4411eef1f77e82fea79fa8b66f92e094210d7f92666209480410f04b86c7","abstract_canon_sha256":"e0a562325c720db5dac1498587067003fb4cab5b91a7c1e8884154c74ee8059d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:14.196284Z","signature_b64":"f1AIajU9hrxF4tU8FBQn9EhRb2njCwZljE5cdSsCpyNp49Q0r0kvVY/O8SgylRJGGS74wfG0hEZ7p4wyzo0fCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d67945af6c9eb98b840ec831f8c81b9537a6a975d9eab0d760e9b89899f7a36","last_reissued_at":"2026-07-05T11:30:14.195761Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:14.195761Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LitBench: A Benchmark and Dataset for Reliable Evaluation of Creative Writing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Fein, Kabir Jolly, Nick Haber, Rafael Rafailov, Sebastian Russo, Violet Xiang","submitted_at":"2025-07-01T14:10:36Z","abstract_excerpt":"Evaluating creative writing generated by large language models (LLMs) remains challenging because open-ended narratives lack ground truths. Without performant automated evaluation methods, off-the-shelf (OTS) language models are employed as zero-shot judges, yet their reliability is unclear in this context. In pursuit of robust evaluation for creative writing, we introduce LitBench, the first standardized benchmark and paired dataset for creative writing verification, comprising a held-out test set of 2,480 debiased, human-labeled story comparisons drawn from Reddit and a 43,827-pair training "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00769","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00769/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00769","created_at":"2026-07-05T11:30:14.195822+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00769v1","created_at":"2026-07-05T11:30:14.195822+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00769","created_at":"2026-07-05T11:30:14.195822+00:00"},{"alias_kind":"pith_short_12","alias_value":"NVTZIWXWZHVZ","created_at":"2026-07-05T11:30:14.195822+00:00"},{"alias_kind":"pith_short_16","alias_value":"NVTZIWXWZHVZROCA","created_at":"2026-07-05T11:30:14.195822+00:00"},{"alias_kind":"pith_short_8","alias_value":"NVTZIWXW","created_at":"2026-07-05T11:30:14.195822+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24819","citing_title":"HelpBench: Assessing the Ability of LLMs to Provide Privacy, Safety, and Security Advice","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04831","citing_title":"StoryAlign: Evaluating and Training Reward Models for Story Generation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF","json":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF.json","graph_json":"https://pith.science/api/pith-number/NVTZIWXWZHVZROCA5SBR7DEBXF/graph.json","events_json":"https://pith.science/api/pith-number/NVTZIWXWZHVZROCA5SBR7DEBXF/events.json","paper":"https://pith.science/paper/NVTZIWXW"},"agent_actions":{"view_html":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF","download_json":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF.json","view_paper":"https://pith.science/paper/NVTZIWXW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00769&json=true","fetch_graph":"https://pith.science/api/pith-number/NVTZIWXWZHVZROCA5SBR7DEBXF/graph.json","fetch_events":"https://pith.science/api/pith-number/NVTZIWXWZHVZROCA5SBR7DEBXF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF/action/storage_attestation","attest_author":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF/action/author_attestation","sign_citation":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF/action/citation_signature","submit_replication":"https://pith.science/pith/NVTZIWXWZHVZROCA5SBR7DEBXF/action/replication_record"}},"created_at":"2026-07-05T11:30:14.195822+00:00","updated_at":"2026-07-05T11:30:14.195822+00:00"}