{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J6TTGZRHPXPIGYQXRVSOPFSPEJ","short_pith_number":"pith:J6TTGZRH","schema_version":"1.0","canonical_sha256":"4fa73366277dde8362178d64e7964f226793a8a241a027c9ee740afa5149f62e","source":{"kind":"arxiv","id":"2405.14213","version":2},"attestation_state":"computed","paper":{"title":"From Text to Pixel: Advancing Long-Context Understanding in MLLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Miguel Eckstein, Tsu-Jui Fu, William Yang Wang, Xiujun Li, Yujie Lu","submitted_at":"2024-05-23T06:17:23Z","abstract_excerpt":"The rapid progress in Multimodal Large Language Models (MLLMs) has significantly advanced their ability to process and understand complex visual and textual information. However, the integration of multiple images and extensive textual contexts remains a challenge due to the inherent limitation of the models' capacity to handle long input sequences efficiently. In this paper, we introduce SEEKER, a multimodal large language model designed to tackle this issue. SEEKER aims to optimize the compact encoding of long text by compressing the text sequence into the visual pixel space via images, enab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14213","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-23T06:17:23Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8168fa732049f9066a5e5cab341f1174bd6e542f3a9de89111d6a4adc907fa32","abstract_canon_sha256":"aac782993b6925e6956b7b3252b7296769a6013db39912e4d646fcac64b675c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:09.564410Z","signature_b64":"TzGZwshZEniQHtnhGvJc+7uUk0mraHKApfCAmmPSKBG9OXc1mPQGJxOb9IE/vZDCWM/ZNoTv+Zpg4iNUsA+ZAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4fa73366277dde8362178d64e7964f226793a8a241a027c9ee740afa5149f62e","last_reissued_at":"2026-07-05T08:59:09.563851Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:09.563851Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Text to Pixel: Advancing Long-Context Understanding in MLLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Miguel Eckstein, Tsu-Jui Fu, William Yang Wang, Xiujun Li, Yujie Lu","submitted_at":"2024-05-23T06:17:23Z","abstract_excerpt":"The rapid progress in Multimodal Large Language Models (MLLMs) has significantly advanced their ability to process and understand complex visual and textual information. However, the integration of multiple images and extensive textual contexts remains a challenge due to the inherent limitation of the models' capacity to handle long input sequences efficiently. In this paper, we introduce SEEKER, a multimodal large language model designed to tackle this issue. SEEKER aims to optimize the compact encoding of long text by compressing the text sequence into the visual pixel space via images, enab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14213","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14213/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14213","created_at":"2026-07-05T08:59:09.563917+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14213v2","created_at":"2026-07-05T08:59:09.563917+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14213","created_at":"2026-07-05T08:59:09.563917+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6TTGZRHPXPI","created_at":"2026-07-05T08:59:09.563917+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6TTGZRHPXPIGYQX","created_at":"2026-07-05T08:59:09.563917+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6TTGZRH","created_at":"2026-07-05T08:59:09.563917+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28338","citing_title":"Memory Shot for Long-Term Dialogue","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28344","citing_title":"PIXELRAG: Web Screenshots Beat Text for Retrieval-Augmented Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06708","citing_title":"Visual Text Compression as Measure Transport","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ","json":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ.json","graph_json":"https://pith.science/api/pith-number/J6TTGZRHPXPIGYQXRVSOPFSPEJ/graph.json","events_json":"https://pith.science/api/pith-number/J6TTGZRHPXPIGYQXRVSOPFSPEJ/events.json","paper":"https://pith.science/paper/J6TTGZRH"},"agent_actions":{"view_html":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ","download_json":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ.json","view_paper":"https://pith.science/paper/J6TTGZRH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14213&json=true","fetch_graph":"https://pith.science/api/pith-number/J6TTGZRHPXPIGYQXRVSOPFSPEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/J6TTGZRHPXPIGYQXRVSOPFSPEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ/action/storage_attestation","attest_author":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ/action/author_attestation","sign_citation":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ/action/citation_signature","submit_replication":"https://pith.science/pith/J6TTGZRHPXPIGYQXRVSOPFSPEJ/action/replication_record"}},"created_at":"2026-07-05T08:59:09.563917+00:00","updated_at":"2026-07-05T08:59:09.563917+00:00"}