{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:54B3DQ5A3MQENR3CO263K2S4NH","short_pith_number":"pith:54B3DQ5A","schema_version":"1.0","canonical_sha256":"ef03b1c3a0db2046c76276bdb56a5c69e97709197dcc267ab310e4c169210e83","source":{"kind":"arxiv","id":"2402.14073","version":1},"attestation_state":"computed","paper":{"title":"Improving Language Understanding from Screenshots","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adithya Bhaskar, Danqi Chen, Tianyu Gao, Zirui Wang","submitted_at":"2024-02-21T19:01:03Z","abstract_excerpt":"An emerging family of language models (LMs), capable of processing both text and images within a single visual view, has the promise to unlock complex tasks such as chart understanding and UI navigation. We refer to these models as screenshot language models. Despite their appeal, existing screenshot LMs substantially lag behind text-only models on language understanding tasks. To close this gap, we adopt a simplified setting where the model inputs are plain-text-rendered screenshots, and we focus on improving the text ability of screenshot LMs. We propose a novel Patch-and-Text Prediction (PT"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14073","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T19:01:03Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"63f606379e269b74921876ee3251eb24e2fb01ea128b7bf8cfcfc417d6abdeb4","abstract_canon_sha256":"f5da7d0ed54cb9977c8dd777aafd34b57ebd11df294360b87f64c7ce02a546c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:18.350569Z","signature_b64":"/WHG/J45f6gR543QR8R4RAAwBkxypjamHJNIy5U87ELjubo9vk7Sb3qMyNOOd2KVBlL8MBn+bMo7QVWPYhckDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef03b1c3a0db2046c76276bdb56a5c69e97709197dcc267ab310e4c169210e83","last_reissued_at":"2026-07-05T07:49:18.350089Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:18.350089Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Language Understanding from Screenshots","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adithya Bhaskar, Danqi Chen, Tianyu Gao, Zirui Wang","submitted_at":"2024-02-21T19:01:03Z","abstract_excerpt":"An emerging family of language models (LMs), capable of processing both text and images within a single visual view, has the promise to unlock complex tasks such as chart understanding and UI navigation. We refer to these models as screenshot language models. Despite their appeal, existing screenshot LMs substantially lag behind text-only models on language understanding tasks. To close this gap, we adopt a simplified setting where the model inputs are plain-text-rendered screenshots, and we focus on improving the text ability of screenshot LMs. We propose a novel Patch-and-Text Prediction (PT"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14073","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14073/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14073","created_at":"2026-07-05T07:49:18.350147+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14073v1","created_at":"2026-07-05T07:49:18.350147+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14073","created_at":"2026-07-05T07:49:18.350147+00:00"},{"alias_kind":"pith_short_12","alias_value":"54B3DQ5A3MQE","created_at":"2026-07-05T07:49:18.350147+00:00"},{"alias_kind":"pith_short_16","alias_value":"54B3DQ5A3MQENR3C","created_at":"2026-07-05T07:49:18.350147+00:00"},{"alias_kind":"pith_short_8","alias_value":"54B3DQ5A","created_at":"2026-07-05T07:49:18.350147+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09630","citing_title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH","json":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH.json","graph_json":"https://pith.science/api/pith-number/54B3DQ5A3MQENR3CO263K2S4NH/graph.json","events_json":"https://pith.science/api/pith-number/54B3DQ5A3MQENR3CO263K2S4NH/events.json","paper":"https://pith.science/paper/54B3DQ5A"},"agent_actions":{"view_html":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH","download_json":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH.json","view_paper":"https://pith.science/paper/54B3DQ5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14073&json=true","fetch_graph":"https://pith.science/api/pith-number/54B3DQ5A3MQENR3CO263K2S4NH/graph.json","fetch_events":"https://pith.science/api/pith-number/54B3DQ5A3MQENR3CO263K2S4NH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH/action/storage_attestation","attest_author":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH/action/author_attestation","sign_citation":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH/action/citation_signature","submit_replication":"https://pith.science/pith/54B3DQ5A3MQENR3CO263K2S4NH/action/replication_record"}},"created_at":"2026-07-05T07:49:18.350147+00:00","updated_at":"2026-07-05T07:49:18.350147+00:00"}