{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MJ72K4A2OCTHYC5IP6F4AZQNSG","short_pith_number":"pith:MJ72K4A2","schema_version":"1.0","canonical_sha256":"627fa5701a70a67c0ba87f8bc0660d919b88428f7738809e7eeaf6ab30ec568e","source":{"kind":"arxiv","id":"2507.10302","version":1},"attestation_state":"computed","paper":{"title":"DisCo: Towards Distinct and Coherent Visual Encapsulation in Video MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Helin Wang, Hengshuang Zhao, Jiahe Zhao, Rongkun Zheng, Yi Wang","submitted_at":"2025-07-14T14:05:19Z","abstract_excerpt":"In video Multimodal Large Language Models (video MLLMs), the visual encapsulation process plays a pivotal role in converting video contents into representative tokens for LLM input. While linear projectors are widely employed for encapsulation, they introduce semantic indistinctness and temporal incoherence when applied to videos. Conversely, the structure of resamplers shows promise in tackling these challenges, but an effective solution remains unexplored. Drawing inspiration from resampler structures, we introduce DisCo, a novel visual encapsulation method designed to yield semantically dis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.10302","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-14T14:05:19Z","cross_cats_sorted":[],"title_canon_sha256":"8f1c94cbdc334c52e8bb5e1f6cc4cd299c0958d1e06760273db6c27850fed9e2","abstract_canon_sha256":"4329a0a22ca78f544241c1dee9e6f6928f50b27942b5abacb92f5c1f31d6f213"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:49.035791Z","signature_b64":"miObVtPCAp0O50utMK+JBDhe5I7lB04mZaPi1o2mxhNQUxusIHU/YSkOwmN3GwS4eu07alGoXlJTgJ4y4zWvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"627fa5701a70a67c0ba87f8bc0660d919b88428f7738809e7eeaf6ab30ec568e","last_reissued_at":"2026-07-05T11:36:49.035257Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:49.035257Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DisCo: Towards Distinct and Coherent Visual Encapsulation in Video MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Helin Wang, Hengshuang Zhao, Jiahe Zhao, Rongkun Zheng, Yi Wang","submitted_at":"2025-07-14T14:05:19Z","abstract_excerpt":"In video Multimodal Large Language Models (video MLLMs), the visual encapsulation process plays a pivotal role in converting video contents into representative tokens for LLM input. While linear projectors are widely employed for encapsulation, they introduce semantic indistinctness and temporal incoherence when applied to videos. Conversely, the structure of resamplers shows promise in tackling these challenges, but an effective solution remains unexplored. Drawing inspiration from resampler structures, we introduce DisCo, a novel visual encapsulation method designed to yield semantically dis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10302","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.10302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.10302","created_at":"2026-07-05T11:36:49.035326+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.10302v1","created_at":"2026-07-05T11:36:49.035326+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10302","created_at":"2026-07-05T11:36:49.035326+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJ72K4A2OCTH","created_at":"2026-07-05T11:36:49.035326+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJ72K4A2OCTHYC5I","created_at":"2026-07-05T11:36:49.035326+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJ72K4A2","created_at":"2026-07-05T11:36:49.035326+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG","json":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG.json","graph_json":"https://pith.science/api/pith-number/MJ72K4A2OCTHYC5IP6F4AZQNSG/graph.json","events_json":"https://pith.science/api/pith-number/MJ72K4A2OCTHYC5IP6F4AZQNSG/events.json","paper":"https://pith.science/paper/MJ72K4A2"},"agent_actions":{"view_html":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG","download_json":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG.json","view_paper":"https://pith.science/paper/MJ72K4A2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.10302&json=true","fetch_graph":"https://pith.science/api/pith-number/MJ72K4A2OCTHYC5IP6F4AZQNSG/graph.json","fetch_events":"https://pith.science/api/pith-number/MJ72K4A2OCTHYC5IP6F4AZQNSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG/action/storage_attestation","attest_author":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG/action/author_attestation","sign_citation":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG/action/citation_signature","submit_replication":"https://pith.science/pith/MJ72K4A2OCTHYC5IP6F4AZQNSG/action/replication_record"}},"created_at":"2026-07-05T11:36:49.035326+00:00","updated_at":"2026-07-05T11:36:49.035326+00:00"}