{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YFCXHLHECHK654CQVAJSWWBS6Y","short_pith_number":"pith:YFCXHLHE","schema_version":"1.0","canonical_sha256":"c14573ace411d5eef050a8132b5832f608b274c72ef51ee31c604534a86bb72f","source":{"kind":"arxiv","id":"2507.09985","version":1},"attestation_state":"computed","paper":{"title":"Demonstrating the Octopi-1.5 Visual-Tactile-Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Harold Soh, Kelvin Lin, Samson Yu","submitted_at":"2025-07-14T07:05:36Z","abstract_excerpt":"Touch is recognized as a vital sense for humans and an equally important modality for robots, especially for dexterous manipulation, material identification, and scenarios involving visual occlusion. Building upon very recent work in touch foundation models, this demonstration will feature Octopi-1.5, our latest visual-tactile-language model. Compared to its predecessor, Octopi-1.5 introduces the ability to process tactile signals from multiple object parts and employs a simple retrieval-augmented generation (RAG) module to improve performance on tasks and potentially learn new objects on-the-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09985","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-07-14T07:05:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ce93ff447ae8c57335d9b781ab4ac887a6714578b1e50abd3d406ecb7b1a9dbd","abstract_canon_sha256":"05a925a6bfd3acec0f8d7c03a9e280ebe5d8df18deb69ac63c3b249487dfdcdf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:43.274391Z","signature_b64":"wAHbY4QecBhhmqIc0rINOWt29Gdp8HDWocNphgMbP3C3d45CSsPeSCqcchMPsW1ygtZ0fsOZDUTv5x0pPjcaBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c14573ace411d5eef050a8132b5832f608b274c72ef51ee31c604534a86bb72f","last_reissued_at":"2026-07-05T11:36:43.273768Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:43.273768Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Demonstrating the Octopi-1.5 Visual-Tactile-Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Harold Soh, Kelvin Lin, Samson Yu","submitted_at":"2025-07-14T07:05:36Z","abstract_excerpt":"Touch is recognized as a vital sense for humans and an equally important modality for robots, especially for dexterous manipulation, material identification, and scenarios involving visual occlusion. Building upon very recent work in touch foundation models, this demonstration will feature Octopi-1.5, our latest visual-tactile-language model. Compared to its predecessor, Octopi-1.5 introduces the ability to process tactile signals from multiple object parts and employs a simple retrieval-augmented generation (RAG) module to improve performance on tasks and potentially learn new objects on-the-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09985","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09985","created_at":"2026-07-05T11:36:43.273832+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09985v1","created_at":"2026-07-05T11:36:43.273832+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09985","created_at":"2026-07-05T11:36:43.273832+00:00"},{"alias_kind":"pith_short_12","alias_value":"YFCXHLHECHK6","created_at":"2026-07-05T11:36:43.273832+00:00"},{"alias_kind":"pith_short_16","alias_value":"YFCXHLHECHK654CQ","created_at":"2026-07-05T11:36:43.273832+00:00"},{"alias_kind":"pith_short_8","alias_value":"YFCXHLHE","created_at":"2026-07-05T11:36:43.273832+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11637","citing_title":"TouchThinker: Scaling Tactile Commonsense Reasoning to the Open World with Large-scale Data and Action-aware Representation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12069","citing_title":"Tac-DINO: Learning Vision-Tactile Features with Patch Alignment","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17336","citing_title":"Tactile-based Multimodal Fusion in Embodied Intelligence: A Survey of Vision, Language, and Contact-Driven Paradigms","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20239","citing_title":"TouchGuide: Inference-Time Steering of Visuomotor Policies via Touch Guidance","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y","json":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y.json","graph_json":"https://pith.science/api/pith-number/YFCXHLHECHK654CQVAJSWWBS6Y/graph.json","events_json":"https://pith.science/api/pith-number/YFCXHLHECHK654CQVAJSWWBS6Y/events.json","paper":"https://pith.science/paper/YFCXHLHE"},"agent_actions":{"view_html":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y","download_json":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y.json","view_paper":"https://pith.science/paper/YFCXHLHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09985&json=true","fetch_graph":"https://pith.science/api/pith-number/YFCXHLHECHK654CQVAJSWWBS6Y/graph.json","fetch_events":"https://pith.science/api/pith-number/YFCXHLHECHK654CQVAJSWWBS6Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y/action/storage_attestation","attest_author":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y/action/author_attestation","sign_citation":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y/action/citation_signature","submit_replication":"https://pith.science/pith/YFCXHLHECHK654CQVAJSWWBS6Y/action/replication_record"}},"created_at":"2026-07-05T11:36:43.273832+00:00","updated_at":"2026-07-05T11:36:43.273832+00:00"}