{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z3RIK27SW2MLQTCGBDUZKQXAFC","short_pith_number":"pith:Z3RIK27S","schema_version":"1.0","canonical_sha256":"cee2856bf2b698b84c4608e99542e0288466590f9360bd3a55826d9766c42182","source":{"kind":"arxiv","id":"2506.13356","version":1},"attestation_state":"computed","paper":{"title":"StoryBench: A Dynamic Benchmark for Evaluating Long-Term Memory with Multi Turns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Luanbo Wan, Weizhi Ma","submitted_at":"2025-06-16T10:54:31Z","abstract_excerpt":"Long-term memory (LTM) is essential for large language models (LLMs) to achieve autonomous intelligence in complex, evolving environments. Despite increasing efforts in memory-augmented and retrieval-based architectures, there remains a lack of standardized benchmarks to systematically evaluate LLMs' long-term memory abilities. Existing benchmarks still face challenges in evaluating knowledge retention and dynamic sequential reasoning, and in their own flexibility, all of which limit their effectiveness in assessing models' LTM capabilities. To address these gaps, we propose a novel benchmark "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13356","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T10:54:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"445ab08c2d6fa183173f8685730533ffdb27a973292b064b0c8493ca6c230705","abstract_canon_sha256":"90894377721d2fcf89955ccd68e070ac3771e47aead2dc3252eabf431ce4aad0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:09.195945Z","signature_b64":"puZr4A/VQ1VTtBTVWBzzMiOSW515uJ1m0rI5ibXampLTaSk170qGYIX3irv79lkoAPwSO2m64h0Ua9Ks2LWkAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cee2856bf2b698b84c4608e99542e0288466590f9360bd3a55826d9766c42182","last_reissued_at":"2026-07-05T11:22:09.195450Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:09.195450Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StoryBench: A Dynamic Benchmark for Evaluating Long-Term Memory with Multi Turns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Luanbo Wan, Weizhi Ma","submitted_at":"2025-06-16T10:54:31Z","abstract_excerpt":"Long-term memory (LTM) is essential for large language models (LLMs) to achieve autonomous intelligence in complex, evolving environments. Despite increasing efforts in memory-augmented and retrieval-based architectures, there remains a lack of standardized benchmarks to systematically evaluate LLMs' long-term memory abilities. Existing benchmarks still face challenges in evaluating knowledge retention and dynamic sequential reasoning, and in their own flexibility, all of which limit their effectiveness in assessing models' LTM capabilities. To address these gaps, we propose a novel benchmark "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13356","created_at":"2026-07-05T11:22:09.195517+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13356v1","created_at":"2026-07-05T11:22:09.195517+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13356","created_at":"2026-07-05T11:22:09.195517+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z3RIK27SW2ML","created_at":"2026-07-05T11:22:09.195517+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z3RIK27SW2MLQTCG","created_at":"2026-07-05T11:22:09.195517+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z3RIK27S","created_at":"2026-07-05T11:22:09.195517+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20926","citing_title":"MemConflict: Evaluating Long-Term Memory Systems Under Memory Conflicts","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05257","citing_title":"Evaluating Memory in LLM Agents via Incremental Multi-Turn Interactions","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20857","citing_title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09874","citing_title":"EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06716","citing_title":"From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC","json":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC.json","graph_json":"https://pith.science/api/pith-number/Z3RIK27SW2MLQTCGBDUZKQXAFC/graph.json","events_json":"https://pith.science/api/pith-number/Z3RIK27SW2MLQTCGBDUZKQXAFC/events.json","paper":"https://pith.science/paper/Z3RIK27S"},"agent_actions":{"view_html":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC","download_json":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC.json","view_paper":"https://pith.science/paper/Z3RIK27S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13356&json=true","fetch_graph":"https://pith.science/api/pith-number/Z3RIK27SW2MLQTCGBDUZKQXAFC/graph.json","fetch_events":"https://pith.science/api/pith-number/Z3RIK27SW2MLQTCGBDUZKQXAFC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC/action/storage_attestation","attest_author":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC/action/author_attestation","sign_citation":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC/action/citation_signature","submit_replication":"https://pith.science/pith/Z3RIK27SW2MLQTCGBDUZKQXAFC/action/replication_record"}},"created_at":"2026-07-05T11:22:09.195517+00:00","updated_at":"2026-07-05T11:22:09.195517+00:00"}