{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TSLPS6A4LND7EZGKHFN3VUDAFT","short_pith_number":"pith:TSLPS6A4","schema_version":"1.0","canonical_sha256":"9c96f9781c5b47f264ca395bbad0602cd6ff74e4516cfaa7d0c967be35bdd3fa","source":{"kind":"arxiv","id":"2607.07608","version":1},"attestation_state":"computed","paper":{"title":"Dual Latent Memory in Vision-Language-Action Models for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Hongyu Qu, Jianzhe Gao, Rui Yan, Shaohuan Yang, Shuicheng Yan, Wenguan Wang, Xiangbo Shu, Xiaobin Hu, Xinlei Yu","submitted_at":"2026-07-08T16:26:06Z","abstract_excerpt":"Mainstream Vision-Language-Action (VLA) models predict actions primarily from the current observation under a Markovian assumption, thus struggling with long-horizon, temporally dependent tasks. Existing memory-augmented VLAs either expand the observation window or retrieve history from the memory bank as auxiliary policy-side context. However, they leave memory outside the native latent embedding space of VLA reasoning, preventing historical experience from being fluidly interleaved with multimodal reasoning and action formation. To this end, we introduce LaMem-VLA, a latent-memory-native fra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.07608","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-08T16:26:06Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"aedf79366c896f2c4506ffa4b51bf6e03f120dcc119a8800a88b39336f77829b","abstract_canon_sha256":"daf9130adaec6be41beb4a406be8de178bfabb335680a042e085ea86ababad1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T01:20:36.621298Z","signature_b64":"6OQV948CUfyJreiD680OkRGlWH63CN/1dXP7cUn7JUQG+uDNG18uGN9B5c8afULXzw4X4Dy46YRdbTjRY6MMAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c96f9781c5b47f264ca395bbad0602cd6ff74e4516cfaa7d0c967be35bdd3fa","last_reissued_at":"2026-07-09T01:20:36.620782Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T01:20:36.620782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual Latent Memory in Vision-Language-Action Models for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Hongyu Qu, Jianzhe Gao, Rui Yan, Shaohuan Yang, Shuicheng Yan, Wenguan Wang, Xiangbo Shu, Xiaobin Hu, Xinlei Yu","submitted_at":"2026-07-08T16:26:06Z","abstract_excerpt":"Mainstream Vision-Language-Action (VLA) models predict actions primarily from the current observation under a Markovian assumption, thus struggling with long-horizon, temporally dependent tasks. Existing memory-augmented VLAs either expand the observation window or retrieve history from the memory bank as auxiliary policy-side context. However, they leave memory outside the native latent embedding space of VLA reasoning, preventing historical experience from being fluidly interleaved with multimodal reasoning and action formation. To this end, we introduce LaMem-VLA, a latent-memory-native fra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.07608","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.07608/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.07608","created_at":"2026-07-09T01:20:36.620847+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.07608v1","created_at":"2026-07-09T01:20:36.620847+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.07608","created_at":"2026-07-09T01:20:36.620847+00:00"},{"alias_kind":"pith_short_12","alias_value":"TSLPS6A4LND7","created_at":"2026-07-09T01:20:36.620847+00:00"},{"alias_kind":"pith_short_16","alias_value":"TSLPS6A4LND7EZGK","created_at":"2026-07-09T01:20:36.620847+00:00"},{"alias_kind":"pith_short_8","alias_value":"TSLPS6A4","created_at":"2026-07-09T01:20:36.620847+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.04530","citing_title":"FocusMem: Factorizing Content, Readout, and Trust in Latent GUI Memory","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT","json":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT.json","graph_json":"https://pith.science/api/pith-number/TSLPS6A4LND7EZGKHFN3VUDAFT/graph.json","events_json":"https://pith.science/api/pith-number/TSLPS6A4LND7EZGKHFN3VUDAFT/events.json","paper":"https://pith.science/paper/TSLPS6A4"},"agent_actions":{"view_html":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT","download_json":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT.json","view_paper":"https://pith.science/paper/TSLPS6A4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.07608&json=true","fetch_graph":"https://pith.science/api/pith-number/TSLPS6A4LND7EZGKHFN3VUDAFT/graph.json","fetch_events":"https://pith.science/api/pith-number/TSLPS6A4LND7EZGKHFN3VUDAFT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT/action/storage_attestation","attest_author":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT/action/author_attestation","sign_citation":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT/action/citation_signature","submit_replication":"https://pith.science/pith/TSLPS6A4LND7EZGKHFN3VUDAFT/action/replication_record"}},"created_at":"2026-07-09T01:20:36.620847+00:00","updated_at":"2026-07-09T01:20:36.620847+00:00"}