{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:GXCYWW2IM5MG7YRE43RVCA6WMI","short_pith_number":"pith:GXCYWW2I","schema_version":"1.0","canonical_sha256":"35c58b5b4867586fe224e6e35103d66238d2455181f4d0b387580dc85a556176","source":{"kind":"arxiv","id":"2608.07861","version":1},"attestation_state":"computed","paper":{"title":"How Much Does It Cost to Answer My Question? Benchmarking Cloud VLM-based VQA Systems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC","cs.IR","cs.MM"],"primary_cat":"cs.CV","authors_text":"Guohao Lan, Henri Vanhuynegem, Weitao Xu, Yiran Shen","submitted_at":"2026-08-08T01:58:13Z","abstract_excerpt":"Vision-language models (VLMs) are becoming a practical backend for mobile visual question answering (VQA) systems, enabling smartphones and smart glasses to answer users' questions about the physical world. Since modern VLMs remain difficult to run on mobile and edge devices, VQA systems increasingly offload inference to cloud-based VLMs. This gives mobile devices access to stronger computation, but it also makes visual input preparation a key system variable: how the image is prepared before offloading affects not only answer quality but also payload size, token cost, and system latency. Prop"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.07861","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2026-08-08T01:58:13Z","cross_cats_sorted":["cs.HC","cs.IR","cs.MM"],"title_canon_sha256":"f800a9c2b9048cb62b7ad92164b77da33420a0cddfd50d9207dffc493f1784b6","abstract_canon_sha256":"82f526a287613c989eb52ee0c151918bd3ca95d006651a3c6347d50d23d7d1ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T01:18:44.800265Z","signature_b64":"Pfs1UE6EqdRzQ6SGpojkRCKUXUixINEDrmA+AF7RFjA+kPsn6ysu6YiKVTmw33x8IXkBjuMyg1a6ieg0lbUoDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35c58b5b4867586fe224e6e35103d66238d2455181f4d0b387580dc85a556176","last_reissued_at":"2026-08-11T01:18:44.797707Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T01:18:44.797707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Much Does It Cost to Answer My Question? Benchmarking Cloud VLM-based VQA Systems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC","cs.IR","cs.MM"],"primary_cat":"cs.CV","authors_text":"Guohao Lan, Henri Vanhuynegem, Weitao Xu, Yiran Shen","submitted_at":"2026-08-08T01:58:13Z","abstract_excerpt":"Vision-language models (VLMs) are becoming a practical backend for mobile visual question answering (VQA) systems, enabling smartphones and smart glasses to answer users' questions about the physical world. Since modern VLMs remain difficult to run on mobile and edge devices, VQA systems increasingly offload inference to cloud-based VLMs. This gives mobile devices access to stronger computation, but it also makes visual input preparation a key system variable: how the image is prepared before offloading affects not only answer quality but also payload size, token cost, and system latency. Prop"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.07861","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.07861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.07861","created_at":"2026-08-11T01:18:44.798500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.07861v1","created_at":"2026-08-11T01:18:44.798500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.07861","created_at":"2026-08-11T01:18:44.798500+00:00"},{"alias_kind":"pith_short_12","alias_value":"GXCYWW2IM5MG","created_at":"2026-08-11T01:18:44.798500+00:00"},{"alias_kind":"pith_short_16","alias_value":"GXCYWW2IM5MG7YRE","created_at":"2026-08-11T01:18:44.798500+00:00"},{"alias_kind":"pith_short_8","alias_value":"GXCYWW2I","created_at":"2026-08-11T01:18:44.798500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI","json":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI.json","graph_json":"https://pith.science/api/pith-number/GXCYWW2IM5MG7YRE43RVCA6WMI/graph.json","events_json":"https://pith.science/api/pith-number/GXCYWW2IM5MG7YRE43RVCA6WMI/events.json","paper":"https://pith.science/paper/GXCYWW2I"},"agent_actions":{"view_html":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI","download_json":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI.json","view_paper":"https://pith.science/paper/GXCYWW2I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.07861&json=true","fetch_graph":"https://pith.science/api/pith-number/GXCYWW2IM5MG7YRE43RVCA6WMI/graph.json","fetch_events":"https://pith.science/api/pith-number/GXCYWW2IM5MG7YRE43RVCA6WMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI/action/storage_attestation","attest_author":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI/action/author_attestation","sign_citation":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI/action/citation_signature","submit_replication":"https://pith.science/pith/GXCYWW2IM5MG7YRE43RVCA6WMI/action/replication_record"}},"created_at":"2026-08-11T01:18:44.798500+00:00","updated_at":"2026-08-11T01:18:44.798500+00:00"}