{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6VYWRINOMFMH77DZG26RNHEIBX","short_pith_number":"pith:6VYWRINO","schema_version":"1.0","canonical_sha256":"f57168a1ae61587ffc7936bd169c880dd991ca0af73d1c636fcc2b909f99920f","source":{"kind":"arxiv","id":"2412.15576","version":5},"attestation_state":"computed","paper":{"title":"QUART-Online: Latency-Free Large Multimodal Language Model for Quadruped Robot Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Can Cui, Donglin Wang, Han Zhao, Hongyin Zhang, Mingyang Sun, Pengxiang Ding, Shangke Lyu, Siteng Huang, Wenjie Zhang, Xinyang Tong, Yiguo Fan, Yonghao Dang","submitted_at":"2024-12-20T05:17:06Z","abstract_excerpt":"This paper addresses the inherent inference latency challenges associated with deploying multimodal large language models (MLLM) in quadruped vision-language-action (QUAR-VLA) tasks. Our investigation reveals that conventional parameter reduction techniques ultimately impair the performance of the language foundation model during the action instruction tuning phase, making them unsuitable for this purpose. We introduce a novel latency-free quadruped MLLM model, dubbed QUART-Online, designed to enhance inference efficiency without degrading the performance of the language foundation model. By i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15576","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-12-20T05:17:06Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"2dcc0b646e6a7551a926134d30386ffb695eea28ba98f10051fc8cf6325136d5","abstract_canon_sha256":"9cd55c55ab96384e05c67218c91c5278eb5d507fcba05182abe1939a6508bdc8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:03.448108Z","signature_b64":"h4F/ia66JMqad+LTJt1SdxzanQFlTPNepztW7HsBlSXaegzeqcd8rmsIKMEGhDMowMOnKybIaq4ctkw29kW7Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f57168a1ae61587ffc7936bd169c880dd991ca0af73d1c636fcc2b909f99920f","last_reissued_at":"2026-07-05T11:10:03.447608Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:03.447608Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QUART-Online: Latency-Free Large Multimodal Language Model for Quadruped Robot Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Can Cui, Donglin Wang, Han Zhao, Hongyin Zhang, Mingyang Sun, Pengxiang Ding, Shangke Lyu, Siteng Huang, Wenjie Zhang, Xinyang Tong, Yiguo Fan, Yonghao Dang","submitted_at":"2024-12-20T05:17:06Z","abstract_excerpt":"This paper addresses the inherent inference latency challenges associated with deploying multimodal large language models (MLLM) in quadruped vision-language-action (QUAR-VLA) tasks. Our investigation reveals that conventional parameter reduction techniques ultimately impair the performance of the language foundation model during the action instruction tuning phase, making them unsuitable for this purpose. We introduce a novel latency-free quadruped MLLM model, dubbed QUART-Online, designed to enhance inference efficiency without degrading the performance of the language foundation model. By i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15576","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15576/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15576","created_at":"2026-07-05T11:10:03.447669+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15576v5","created_at":"2026-07-05T11:10:03.447669+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15576","created_at":"2026-07-05T11:10:03.447669+00:00"},{"alias_kind":"pith_short_12","alias_value":"6VYWRINOMFMH","created_at":"2026-07-05T11:10:03.447669+00:00"},{"alias_kind":"pith_short_16","alias_value":"6VYWRINOMFMH77DZ","created_at":"2026-07-05T11:10:03.447669+00:00"},{"alias_kind":"pith_short_8","alias_value":"6VYWRINO","created_at":"2026-07-05T11:10:03.447669+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.10333","citing_title":"ReconVLA: Reconstructive Vision-Language-Action Model as Effective Robot Perceiver","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX","json":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX.json","graph_json":"https://pith.science/api/pith-number/6VYWRINOMFMH77DZG26RNHEIBX/graph.json","events_json":"https://pith.science/api/pith-number/6VYWRINOMFMH77DZG26RNHEIBX/events.json","paper":"https://pith.science/paper/6VYWRINO"},"agent_actions":{"view_html":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX","download_json":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX.json","view_paper":"https://pith.science/paper/6VYWRINO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15576&json=true","fetch_graph":"https://pith.science/api/pith-number/6VYWRINOMFMH77DZG26RNHEIBX/graph.json","fetch_events":"https://pith.science/api/pith-number/6VYWRINOMFMH77DZG26RNHEIBX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX/action/storage_attestation","attest_author":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX/action/author_attestation","sign_citation":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX/action/citation_signature","submit_replication":"https://pith.science/pith/6VYWRINOMFMH77DZG26RNHEIBX/action/replication_record"}},"created_at":"2026-07-05T11:10:03.447669+00:00","updated_at":"2026-07-05T11:10:03.447669+00:00"}