{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:75WRCGMQUHHL22VLTT6OS4H6VB","short_pith_number":"pith:75WRCGMQ","schema_version":"1.0","canonical_sha256":"ff6d111990a1cebd6aab9cfce970fea860d9dca8a64d56d2c9884841c7b041ea","source":{"kind":"arxiv","id":"2608.01185","version":1},"attestation_state":"computed","paper":{"title":"3DZip: Spatial-Aware Feature Diversity-Guided Token Compression for 3D Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Changwoo Baek, Kyeongbo Kong","submitted_at":"2026-08-02T12:11:51Z","abstract_excerpt":"Recent 3D vision-language models (3D VLMs) construct geometry aware tokens by projecting 2D visual features into world coordinates, enabling spatial reasoning for tasks such as 3D question answering. However, this design generates thousands of tokens per scene, resulting in substantial computational and memory overhead. While token compression has been extensively studied in 2D VLMs, existing approaches rely on semantic relevance or attention-based selection that overlook the structured spatial nature of 3D tokens. Moreover, redundancy in 3D representations cannot be resolved by spatial proxim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.01185","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-02T12:11:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"dc5d722f411767ca78bc8569ad0caf4a61d4274bcf8407bb941e8083c0fd3591","abstract_canon_sha256":"46a44ac43f05e5b1b1cfef72e84c55c707b3b8c85760697daf07a4af9cf9ac8c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:01:53.505298Z","signature_b64":"LdAkNWT6+8kKZO5SHs//cXT4ubkj7KAAlOFeh4eIn271KIYmUZWBDTJyB1itJjfOr4yZdJKcMFIZ+Il6bxx1AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff6d111990a1cebd6aab9cfce970fea860d9dca8a64d56d2c9884841c7b041ea","last_reissued_at":"2026-08-04T02:01:53.503772Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:01:53.503772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"3DZip: Spatial-Aware Feature Diversity-Guided Token Compression for 3D Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Changwoo Baek, Kyeongbo Kong","submitted_at":"2026-08-02T12:11:51Z","abstract_excerpt":"Recent 3D vision-language models (3D VLMs) construct geometry aware tokens by projecting 2D visual features into world coordinates, enabling spatial reasoning for tasks such as 3D question answering. However, this design generates thousands of tokens per scene, resulting in substantial computational and memory overhead. While token compression has been extensively studied in 2D VLMs, existing approaches rely on semantic relevance or attention-based selection that overlook the structured spatial nature of 3D tokens. Moreover, redundancy in 3D representations cannot be resolved by spatial proxim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.01185","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.01185/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.01185","created_at":"2026-08-04T02:01:53.505128+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.01185v1","created_at":"2026-08-04T02:01:53.505128+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.01185","created_at":"2026-08-04T02:01:53.505128+00:00"},{"alias_kind":"pith_short_12","alias_value":"75WRCGMQUHHL","created_at":"2026-08-04T02:01:53.505128+00:00"},{"alias_kind":"pith_short_16","alias_value":"75WRCGMQUHHL22VL","created_at":"2026-08-04T02:01:53.505128+00:00"},{"alias_kind":"pith_short_8","alias_value":"75WRCGMQ","created_at":"2026-08-04T02:01:53.505128+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB","json":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB.json","graph_json":"https://pith.science/api/pith-number/75WRCGMQUHHL22VLTT6OS4H6VB/graph.json","events_json":"https://pith.science/api/pith-number/75WRCGMQUHHL22VLTT6OS4H6VB/events.json","paper":"https://pith.science/paper/75WRCGMQ"},"agent_actions":{"view_html":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB","download_json":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB.json","view_paper":"https://pith.science/paper/75WRCGMQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.01185&json=true","fetch_graph":"https://pith.science/api/pith-number/75WRCGMQUHHL22VLTT6OS4H6VB/graph.json","fetch_events":"https://pith.science/api/pith-number/75WRCGMQUHHL22VLTT6OS4H6VB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB/action/storage_attestation","attest_author":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB/action/author_attestation","sign_citation":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB/action/citation_signature","submit_replication":"https://pith.science/pith/75WRCGMQUHHL22VLTT6OS4H6VB/action/replication_record"}},"created_at":"2026-08-04T02:01:53.505128+00:00","updated_at":"2026-08-04T02:01:53.505128+00:00"}