{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SNQ4ZZSLZ26BX5U46333XP4CFB","short_pith_number":"pith:SNQ4ZZSL","schema_version":"1.0","canonical_sha256":"9361cce64bcebc1bf69cf6f7bbbf8228435a7d6917f4af55c8ef2b924ede6fea","source":{"kind":"arxiv","id":"2408.14622","version":1},"attestation_state":"computed","paper":{"title":"What Makes a Good Story and How Can We Measure It? A Comprehensive Survey of Story Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dingyi Yang, Qin Jin","submitted_at":"2024-08-26T20:35:42Z","abstract_excerpt":"With the development of artificial intelligence, particularly the success of Large Language Models (LLMs), the quantity and quality of automatically generated stories have significantly increased. This has led to the need for automatic story evaluation to assess the generative capabilities of computing systems and analyze the quality of both automatic-generated and human-written stories. Evaluating a story can be more challenging than other generation evaluation tasks. While tasks like machine translation primarily focus on assessing the aspects of fluency and accuracy, story evaluation demand"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.14622","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-26T20:35:42Z","cross_cats_sorted":[],"title_canon_sha256":"cd62151e1d8755c80f192c0f7eda96781a6038a5cbb33f1e9335b78147af6995","abstract_canon_sha256":"f809db319d362ff75a340c65a3c32a09db39d949150ca6c60f22c0d264f0c00b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:36.510513Z","signature_b64":"BGRyVXRmqAFU3HufEzK17DzbYRVQ/wnrnkcoIkBYi3vhY+yBEtnfLICtx228oOeecdxqnWV/Kd+2hxpn4gZKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9361cce64bcebc1bf69cf6f7bbbf8228435a7d6917f4af55c8ef2b924ede6fea","last_reissued_at":"2026-07-05T08:59:36.510040Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:36.510040Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes a Good Story and How Can We Measure It? A Comprehensive Survey of Story Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dingyi Yang, Qin Jin","submitted_at":"2024-08-26T20:35:42Z","abstract_excerpt":"With the development of artificial intelligence, particularly the success of Large Language Models (LLMs), the quantity and quality of automatically generated stories have significantly increased. This has led to the need for automatic story evaluation to assess the generative capabilities of computing systems and analyze the quality of both automatic-generated and human-written stories. Evaluating a story can be more challenging than other generation evaluation tasks. While tasks like machine translation primarily focus on assessing the aspects of fluency and accuracy, story evaluation demand"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.14622","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.14622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.14622","created_at":"2026-07-05T08:59:36.510095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.14622v1","created_at":"2026-07-05T08:59:36.510095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.14622","created_at":"2026-07-05T08:59:36.510095+00:00"},{"alias_kind":"pith_short_12","alias_value":"SNQ4ZZSLZ26B","created_at":"2026-07-05T08:59:36.510095+00:00"},{"alias_kind":"pith_short_16","alias_value":"SNQ4ZZSLZ26BX5U4","created_at":"2026-07-05T08:59:36.510095+00:00"},{"alias_kind":"pith_short_8","alias_value":"SNQ4ZZSL","created_at":"2026-07-05T08:59:36.510095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29625","citing_title":"Improving Collaborative Storytelling with a Multi-Agent Framework Based on Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22748","citing_title":"AI Fiction in the Wild","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08503","citing_title":"NARRA-Gym for Evaluating Interactive Narrative Agents","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB","json":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB.json","graph_json":"https://pith.science/api/pith-number/SNQ4ZZSLZ26BX5U46333XP4CFB/graph.json","events_json":"https://pith.science/api/pith-number/SNQ4ZZSLZ26BX5U46333XP4CFB/events.json","paper":"https://pith.science/paper/SNQ4ZZSL"},"agent_actions":{"view_html":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB","download_json":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB.json","view_paper":"https://pith.science/paper/SNQ4ZZSL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.14622&json=true","fetch_graph":"https://pith.science/api/pith-number/SNQ4ZZSLZ26BX5U46333XP4CFB/graph.json","fetch_events":"https://pith.science/api/pith-number/SNQ4ZZSLZ26BX5U46333XP4CFB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB/action/storage_attestation","attest_author":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB/action/author_attestation","sign_citation":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB/action/citation_signature","submit_replication":"https://pith.science/pith/SNQ4ZZSLZ26BX5U46333XP4CFB/action/replication_record"}},"created_at":"2026-07-05T08:59:36.510095+00:00","updated_at":"2026-07-05T08:59:36.510095+00:00"}