{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:B7Z77X67BNU73P4QAFPAUA3MTN","short_pith_number":"pith:B7Z77X67","schema_version":"1.0","canonical_sha256":"0ff3ffdfdf0b69fdbf90015e0a036c9b7be79562bcd574baeb44f4c93baa6d01","source":{"kind":"arxiv","id":"2607.25527","version":1},"attestation_state":"computed","paper":{"title":"Argus-Unified: Towards A Compact and Economical Unified Model for Image Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chen Chen, Jiabo Huang, Jingtao Li, Lingjuan Lyu, Sina Sajadmanesh, Weiming Zhuang, Zhizhong Li","submitted_at":"2026-07-28T10:12:06Z","abstract_excerpt":"Unifying visual understanding and generation in one model holds immense promise, but remains challenging and expensive due to heavy compute and data demands and conflicts between the visual features needed for these two capabilities. To address these challenges, we present Argus-Unified, a compact, effective and unified multimodal model built with low demand on computation and data. Instead of aligning modalities from scratch, Argus-Unified effectively leverages pretrained vision-language models (VLMs) that provide strong multimodal priors. Specifically, we introduce hybrid visual tokens that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.25527","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-07-28T10:12:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3ddae83876e2f5220b2e9379119a2b30044c1e2b491815597f3eb63a66572d63","abstract_canon_sha256":"ebf52458faef0eb5b151672ff6159e7869bb6499dfcf245cd7dd7473a8d23aa6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-29T01:25:45.165063Z","signature_b64":"B21CAUmqex2lapEJl6jJPPMcvD/Ca+3O2/G+OFmkUOr5d53zlawoXRtFCPV0rNYW/F6n9q/BZulkcgHmyKt4Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ff3ffdfdf0b69fdbf90015e0a036c9b7be79562bcd574baeb44f4c93baa6d01","last_reissued_at":"2026-07-29T01:25:45.164168Z","signature_status":"signed_v1","first_computed_at":"2026-07-29T01:25:45.164168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Argus-Unified: Towards A Compact and Economical Unified Model for Image Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chen Chen, Jiabo Huang, Jingtao Li, Lingjuan Lyu, Sina Sajadmanesh, Weiming Zhuang, Zhizhong Li","submitted_at":"2026-07-28T10:12:06Z","abstract_excerpt":"Unifying visual understanding and generation in one model holds immense promise, but remains challenging and expensive due to heavy compute and data demands and conflicts between the visual features needed for these two capabilities. To address these challenges, we present Argus-Unified, a compact, effective and unified multimodal model built with low demand on computation and data. Instead of aligning modalities from scratch, Argus-Unified effectively leverages pretrained vision-language models (VLMs) that provide strong multimodal priors. Specifically, we introduce hybrid visual tokens that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.25527","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.25527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.25527","created_at":"2026-07-29T01:25:45.164589+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.25527v1","created_at":"2026-07-29T01:25:45.164589+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.25527","created_at":"2026-07-29T01:25:45.164589+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7Z77X67BNU7","created_at":"2026-07-29T01:25:45.164589+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7Z77X67BNU73P4Q","created_at":"2026-07-29T01:25:45.164589+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7Z77X67","created_at":"2026-07-29T01:25:45.164589+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN","json":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN.json","graph_json":"https://pith.science/api/pith-number/B7Z77X67BNU73P4QAFPAUA3MTN/graph.json","events_json":"https://pith.science/api/pith-number/B7Z77X67BNU73P4QAFPAUA3MTN/events.json","paper":"https://pith.science/paper/B7Z77X67"},"agent_actions":{"view_html":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN","download_json":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN.json","view_paper":"https://pith.science/paper/B7Z77X67","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.25527&json=true","fetch_graph":"https://pith.science/api/pith-number/B7Z77X67BNU73P4QAFPAUA3MTN/graph.json","fetch_events":"https://pith.science/api/pith-number/B7Z77X67BNU73P4QAFPAUA3MTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN/action/storage_attestation","attest_author":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN/action/author_attestation","sign_citation":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN/action/citation_signature","submit_replication":"https://pith.science/pith/B7Z77X67BNU73P4QAFPAUA3MTN/action/replication_record"}},"created_at":"2026-07-29T01:25:45.164589+00:00","updated_at":"2026-07-29T01:25:45.164589+00:00"}