{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DJ6IF7KPPW6JWYDG2W2TP2TOMM","short_pith_number":"pith:DJ6IF7KP","schema_version":"1.0","canonical_sha256":"1a7c82fd4f7dbc9b6066d5b537ea6e6305f241ed4de23b78c913d139dadfa668","source":{"kind":"arxiv","id":"2506.21605","version":1},"attestation_state":"computed","paper":{"title":"MemBench: Towards More Comprehensive Evaluation on the Memory of LLM-based Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chen Ma, Haoran Tan, Quanyu Dai, Xu Chen, Zeyu Zhang, Zhenhua Dong","submitted_at":"2025-06-20T10:09:23Z","abstract_excerpt":"Recent works have highlighted the significance of memory mechanisms in LLM-based agents, which enable them to store observed information and adapt to dynamic environments. However, evaluating their memory capabilities still remains challenges. Previous evaluations are commonly limited by the diversity of memory levels and interactive scenarios. They also lack comprehensive metrics to reflect the memory capabilities from multiple aspects. To address these problems, in this paper, we construct a more comprehensive dataset and benchmark to evaluate the memory capability of LLM-based agents. Our d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21605","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T10:09:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f5ef6348b294777a76b84c3acbe4c179b8b5e758cc8b3714a39eb3327bd9ccce","abstract_canon_sha256":"36a3b4ef9dad358416babec9c01d724bb0bccd8077c2ad2a61647b66b4c540a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:57.123622Z","signature_b64":"KQmG3OqwwU2x03LTa/mLshjLt100f3BrOoVW8e15GlK7h1x2ACqEj+/59d9TJDEKXsgb7TX1Cu+zZyB2cN1xDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a7c82fd4f7dbc9b6066d5b537ea6e6305f241ed4de23b78c913d139dadfa668","last_reissued_at":"2026-07-05T11:27:57.123104Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:57.123104Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MemBench: Towards More Comprehensive Evaluation on the Memory of LLM-based Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chen Ma, Haoran Tan, Quanyu Dai, Xu Chen, Zeyu Zhang, Zhenhua Dong","submitted_at":"2025-06-20T10:09:23Z","abstract_excerpt":"Recent works have highlighted the significance of memory mechanisms in LLM-based agents, which enable them to store observed information and adapt to dynamic environments. However, evaluating their memory capabilities still remains challenges. Previous evaluations are commonly limited by the diversity of memory levels and interactive scenarios. They also lack comprehensive metrics to reflect the memory capabilities from multiple aspects. To address these problems, in this paper, we construct a more comprehensive dataset and benchmark to evaluate the memory capability of LLM-based agents. Our d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21605","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21605/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21605","created_at":"2026-07-05T11:27:57.123163+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21605v1","created_at":"2026-07-05T11:27:57.123163+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21605","created_at":"2026-07-05T11:27:57.123163+00:00"},{"alias_kind":"pith_short_12","alias_value":"DJ6IF7KPPW6J","created_at":"2026-07-05T11:27:57.123163+00:00"},{"alias_kind":"pith_short_16","alias_value":"DJ6IF7KPPW6JWYDG","created_at":"2026-07-05T11:27:57.123163+00:00"},{"alias_kind":"pith_short_8","alias_value":"DJ6IF7KP","created_at":"2026-07-05T11:27:57.123163+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05844","citing_title":"StateFuse: Deterministic Conflict-Preserving Memory for Multi-Agent Systems","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24428","citing_title":"Escaping the Self-Confirmation Trap: An Execute-Distill-Verify Paradigm for Agentic Experience Learning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27472","citing_title":"Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26870","citing_title":"Persistent AI Agents in Academic Research: A Single-Investigator Implementation Case Study","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24941","citing_title":"Memory-Induced Tool-Drift in LLM Agents","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21768","citing_title":"Memory-R2: Fair Credit Assignment for Long-Horizon Memory-Augmented LLM Agents","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01847","citing_title":"NeuroState-Bench: A Human-Calibrated Benchmark for Commitment Integrity in LLM Agent Profiles","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20857","citing_title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10870","citing_title":"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23194","citing_title":"From Coarse to Fine: Self-Adaptive Hierarchical Planning for LLM Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01847","citing_title":"NeuroState-Bench: A Human-Calibrated Benchmark for Commitment Integrity in LLM Agent Profiles","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM","json":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM.json","graph_json":"https://pith.science/api/pith-number/DJ6IF7KPPW6JWYDG2W2TP2TOMM/graph.json","events_json":"https://pith.science/api/pith-number/DJ6IF7KPPW6JWYDG2W2TP2TOMM/events.json","paper":"https://pith.science/paper/DJ6IF7KP"},"agent_actions":{"view_html":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM","download_json":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM.json","view_paper":"https://pith.science/paper/DJ6IF7KP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21605&json=true","fetch_graph":"https://pith.science/api/pith-number/DJ6IF7KPPW6JWYDG2W2TP2TOMM/graph.json","fetch_events":"https://pith.science/api/pith-number/DJ6IF7KPPW6JWYDG2W2TP2TOMM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM/action/storage_attestation","attest_author":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM/action/author_attestation","sign_citation":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM/action/citation_signature","submit_replication":"https://pith.science/pith/DJ6IF7KPPW6JWYDG2W2TP2TOMM/action/replication_record"}},"created_at":"2026-07-05T11:27:57.123163+00:00","updated_at":"2026-07-05T11:27:57.123163+00:00"}