{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:Z2O75LUU5L5CB5OSHWQ7RX63YN","short_pith_number":"pith:Z2O75LUU","schema_version":"1.0","canonical_sha256":"ce9dfeae94eafa20f5d23da1f8dfdbc36a5ee95f76a8a3d0bbe283f4f0724301","source":{"kind":"arxiv","id":"2608.00017","version":1},"attestation_state":"computed","paper":{"title":"Memory Reward Inflation in Self-Improving LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CE"],"primary_cat":"cs.AI","authors_text":"Amir Amini, Amirfarhad Farhadi, Azadeh Zamanifar, Mohammad Asadolahi, Samira Talebi","submitted_at":"2026-06-29T12:20:14Z","abstract_excerpt":"Self-improving LLM agents increasingly learn from experience without updating any weights. Each episode is stored in an external memory, scored, and retrieved for similar future tasks to shape later behavior. Viewed through a reward lens, the stored score is a proxy reward for an implicit, non-parametric policy. Each retrieved episode then becomes a policy-improvement step whose reliability hinges on how that score is produced. In deployment, ground-truth labels are unavailable, so the stored reward is at best an LLM assessment. This substitution creates a failure mode, the *Echo Gap*, across "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.00017","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-06-29T12:20:14Z","cross_cats_sorted":["cs.CE"],"title_canon_sha256":"44f32269c7cb2f742512d971fc0b0145d718e182930ae812ac32973fc2ccc7ee","abstract_canon_sha256":"23f9d5e65387196880e9797775b816304113b9a6f24a1b0b8ba3642f91528166"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T00:31:38.895772Z","signature_b64":"WBb+ABsSjR5Cx9wetqBJn5k8BoegviyW9rYmE8UL0N1cGrSrs2F0wfnm3fef/EhU74XR/nRKpV4YesXDOW6oBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce9dfeae94eafa20f5d23da1f8dfdbc36a5ee95f76a8a3d0bbe283f4f0724301","last_reissued_at":"2026-08-04T00:31:38.894341Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T00:31:38.894341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memory Reward Inflation in Self-Improving LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CE"],"primary_cat":"cs.AI","authors_text":"Amir Amini, Amirfarhad Farhadi, Azadeh Zamanifar, Mohammad Asadolahi, Samira Talebi","submitted_at":"2026-06-29T12:20:14Z","abstract_excerpt":"Self-improving LLM agents increasingly learn from experience without updating any weights. Each episode is stored in an external memory, scored, and retrieved for similar future tasks to shape later behavior. Viewed through a reward lens, the stored score is a proxy reward for an implicit, non-parametric policy. Each retrieved episode then becomes a policy-improvement step whose reliability hinges on how that score is produced. In deployment, ground-truth labels are unavailable, so the stored reward is at best an LLM assessment. This substitution creates a failure mode, the *Echo Gap*, across "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.00017","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.00017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.00017","created_at":"2026-08-04T00:31:38.895324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.00017v1","created_at":"2026-08-04T00:31:38.895324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.00017","created_at":"2026-08-04T00:31:38.895324+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z2O75LUU5L5C","created_at":"2026-08-04T00:31:38.895324+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z2O75LUU5L5CB5OS","created_at":"2026-08-04T00:31:38.895324+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z2O75LUU","created_at":"2026-08-04T00:31:38.895324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN","json":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN.json","graph_json":"https://pith.science/api/pith-number/Z2O75LUU5L5CB5OSHWQ7RX63YN/graph.json","events_json":"https://pith.science/api/pith-number/Z2O75LUU5L5CB5OSHWQ7RX63YN/events.json","paper":"https://pith.science/paper/Z2O75LUU"},"agent_actions":{"view_html":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN","download_json":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN.json","view_paper":"https://pith.science/paper/Z2O75LUU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.00017&json=true","fetch_graph":"https://pith.science/api/pith-number/Z2O75LUU5L5CB5OSHWQ7RX63YN/graph.json","fetch_events":"https://pith.science/api/pith-number/Z2O75LUU5L5CB5OSHWQ7RX63YN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN/action/storage_attestation","attest_author":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN/action/author_attestation","sign_citation":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN/action/citation_signature","submit_replication":"https://pith.science/pith/Z2O75LUU5L5CB5OSHWQ7RX63YN/action/replication_record"}},"created_at":"2026-08-04T00:31:38.895324+00:00","updated_at":"2026-08-04T00:31:38.895324+00:00"}