{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7YPZSVJ7N73F63HDW72DXQ5ZBG","short_pith_number":"pith:7YPZSVJ7","schema_version":"1.0","canonical_sha256":"fe1f99553f6ff65f6ce3b7f43bc3b909a243a498f08b7618f86243fbcde81224","source":{"kind":"arxiv","id":"2403.01061","version":3},"attestation_state":"computed","paper":{"title":"Reading Subtext: Evaluating Large Language Models on Short Story Summarization with Writers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kathleen McKeown, Lydia B. Chilton, Melanie Subbiah, Sean Zhang","submitted_at":"2024-03-02T01:52:14Z","abstract_excerpt":"We evaluate recent Large Language Models (LLMs) on the challenging task of summarizing short stories, which can be lengthy, and include nuanced subtext or scrambled timelines. Importantly, we work directly with authors to ensure that the stories have not been shared online (and therefore are unseen by the models), and to obtain informed evaluations of summary quality using judgments from the authors themselves. Through quantitative and qualitative analysis grounded in narrative theory, we compare GPT-4, Claude-2.1, and LLama-2-70B. We find that all three models make faithfulness mistakes in ov"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.01061","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-02T01:52:14Z","cross_cats_sorted":[],"title_canon_sha256":"f5544f81c1f5456b47d1bb57c3c46a32421c3ee3f4573155300825de871da6a2","abstract_canon_sha256":"c5948d9fdaef4533abd98f45c82bf808d1c6a960b699e258ca96d041b836f9c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:53.722601Z","signature_b64":"6ypI5m5PLSqSZXOWURSIAa4ES4EnEvZaPvu3wNSQK5rA5SF4S0vkfifiyEQ3SpCCcb+t5INFAqgJJwV4U2dsDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe1f99553f6ff65f6ce3b7f43bc3b909a243a498f08b7618f86243fbcde81224","last_reissued_at":"2026-07-05T08:42:53.722160Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:53.722160Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reading Subtext: Evaluating Large Language Models on Short Story Summarization with Writers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kathleen McKeown, Lydia B. Chilton, Melanie Subbiah, Sean Zhang","submitted_at":"2024-03-02T01:52:14Z","abstract_excerpt":"We evaluate recent Large Language Models (LLMs) on the challenging task of summarizing short stories, which can be lengthy, and include nuanced subtext or scrambled timelines. Importantly, we work directly with authors to ensure that the stories have not been shared online (and therefore are unseen by the models), and to obtain informed evaluations of summary quality using judgments from the authors themselves. Through quantitative and qualitative analysis grounded in narrative theory, we compare GPT-4, Claude-2.1, and LLama-2-70B. We find that all three models make faithfulness mistakes in ov"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01061","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.01061/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.01061","created_at":"2026-07-05T08:42:53.722219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.01061v3","created_at":"2026-07-05T08:42:53.722219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01061","created_at":"2026-07-05T08:42:53.722219+00:00"},{"alias_kind":"pith_short_12","alias_value":"7YPZSVJ7N73F","created_at":"2026-07-05T08:42:53.722219+00:00"},{"alias_kind":"pith_short_16","alias_value":"7YPZSVJ7N73F63HD","created_at":"2026-07-05T08:42:53.722219+00:00"},{"alias_kind":"pith_short_8","alias_value":"7YPZSVJ7","created_at":"2026-07-05T08:42:53.722219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08503","citing_title":"NARRA-Gym for Evaluating Interactive Narrative Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20131","citing_title":"Whose Story Gets Told? Positionality and Bias in LLM Summaries of Life Narratives","ref_index":177,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG","json":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG.json","graph_json":"https://pith.science/api/pith-number/7YPZSVJ7N73F63HDW72DXQ5ZBG/graph.json","events_json":"https://pith.science/api/pith-number/7YPZSVJ7N73F63HDW72DXQ5ZBG/events.json","paper":"https://pith.science/paper/7YPZSVJ7"},"agent_actions":{"view_html":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG","download_json":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG.json","view_paper":"https://pith.science/paper/7YPZSVJ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.01061&json=true","fetch_graph":"https://pith.science/api/pith-number/7YPZSVJ7N73F63HDW72DXQ5ZBG/graph.json","fetch_events":"https://pith.science/api/pith-number/7YPZSVJ7N73F63HDW72DXQ5ZBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG/action/storage_attestation","attest_author":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG/action/author_attestation","sign_citation":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG/action/citation_signature","submit_replication":"https://pith.science/pith/7YPZSVJ7N73F63HDW72DXQ5ZBG/action/replication_record"}},"created_at":"2026-07-05T08:42:53.722219+00:00","updated_at":"2026-07-05T08:42:53.722219+00:00"}