{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:SP4RYAHMXWPN524JKDXREDGUNE","short_pith_number":"pith:SP4RYAHM","schema_version":"1.0","canonical_sha256":"93f91c00ecbd9edeeb8950ef120cd4691b91f61b05d4d4b53693e05951a2a7b0","source":{"kind":"arxiv","id":"2608.04010","version":1},"attestation_state":"computed","paper":{"title":"ParVL: Parallel Scaling and Expandable Compute Allocation for Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hongjie Zhang, Lixin Gu, Mouxiang Chen, Qinyu Zhao, Wenhai Wang, Wenwei Zhang, Xiaohui Li, Yang Yang","submitted_at":"2026-08-04T17:59:58Z","abstract_excerpt":"Existing scaling strategies for Multimodal Large Language Models (MLLMs) typically expand either model parameters or sequential inference computation, incurring substantial memory or latency overhead. More importantly, most existing methods fail to alter the rigid, fixed computation allocation between the Vision Transformer and the Large Language Model components, limiting task-specific optimization. To address this, we introduce the Parallel Vision-Language (ParVL) scaling framework for MLLMs, which scales parallel computation by reusing the existing ViT and LLM backbone parameters across mul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.04010","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-04T17:59:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e3973756c337879ec70313f66880c34d6e3fcaa04d61d29a39087ccf67baab9f","abstract_canon_sha256":"db88aa9c827d3cf1de5e225dcfe7082280c994addc370893b63c7df657cd309e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-05T01:37:41.683043Z","signature_b64":"pZKVreW6uYsAqvYdnlrP2Lg/piKpTKdVT0iI39wWniIiZJzKWUeZN0TDNvKiDvCGSUQ4DkwFrbaLxfEZY9Y5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93f91c00ecbd9edeeb8950ef120cd4691b91f61b05d4d4b53693e05951a2a7b0","last_reissued_at":"2026-08-05T01:37:41.681314Z","signature_status":"signed_v1","first_computed_at":"2026-08-05T01:37:41.681314Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ParVL: Parallel Scaling and Expandable Compute Allocation for Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hongjie Zhang, Lixin Gu, Mouxiang Chen, Qinyu Zhao, Wenhai Wang, Wenwei Zhang, Xiaohui Li, Yang Yang","submitted_at":"2026-08-04T17:59:58Z","abstract_excerpt":"Existing scaling strategies for Multimodal Large Language Models (MLLMs) typically expand either model parameters or sequential inference computation, incurring substantial memory or latency overhead. More importantly, most existing methods fail to alter the rigid, fixed computation allocation between the Vision Transformer and the Large Language Model components, limiting task-specific optimization. To address this, we introduce the Parallel Vision-Language (ParVL) scaling framework for MLLMs, which scales parallel computation by reusing the existing ViT and LLM backbone parameters across mul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.04010","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.04010/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.04010","created_at":"2026-08-05T01:37:41.681897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.04010v1","created_at":"2026-08-05T01:37:41.681897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.04010","created_at":"2026-08-05T01:37:41.681897+00:00"},{"alias_kind":"pith_short_12","alias_value":"SP4RYAHMXWPN","created_at":"2026-08-05T01:37:41.681897+00:00"},{"alias_kind":"pith_short_16","alias_value":"SP4RYAHMXWPN524J","created_at":"2026-08-05T01:37:41.681897+00:00"},{"alias_kind":"pith_short_8","alias_value":"SP4RYAHM","created_at":"2026-08-05T01:37:41.681897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE","json":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE.json","graph_json":"https://pith.science/api/pith-number/SP4RYAHMXWPN524JKDXREDGUNE/graph.json","events_json":"https://pith.science/api/pith-number/SP4RYAHMXWPN524JKDXREDGUNE/events.json","paper":"https://pith.science/paper/SP4RYAHM"},"agent_actions":{"view_html":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE","download_json":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE.json","view_paper":"https://pith.science/paper/SP4RYAHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.04010&json=true","fetch_graph":"https://pith.science/api/pith-number/SP4RYAHMXWPN524JKDXREDGUNE/graph.json","fetch_events":"https://pith.science/api/pith-number/SP4RYAHMXWPN524JKDXREDGUNE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE/action/storage_attestation","attest_author":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE/action/author_attestation","sign_citation":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE/action/citation_signature","submit_replication":"https://pith.science/pith/SP4RYAHMXWPN524JKDXREDGUNE/action/replication_record"}},"created_at":"2026-08-05T01:37:41.681897+00:00","updated_at":"2026-08-05T01:37:41.681897+00:00"}