{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:QDIKQO4KTXOTDBCZFPY7VZSKMA","short_pith_number":"pith:QDIKQO4K","schema_version":"1.0","canonical_sha256":"80d0a83b8a9ddd3184592bf1fae64a602db539dc56dc76ba3bf91aa21e2ab45b","source":{"kind":"arxiv","id":"2608.04726","version":1},"attestation_state":"computed","paper":{"title":"When Prompts Become Pixels: Prompt-Region Grounding for Multimodal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Ruizhe Zhou, Xiaodan Liang, Xiaojun Chang, Xuemin Zhao, Yingying Zhu, Yongxin Wang, Yueling Tang","submitted_at":"2026-08-05T11:46:15Z","abstract_excerpt":"Multimodal large language models increasingly reason over screenshots and documents where the task itself may be written in pixels. Yet benchmarks usually place questions in text, leaving it unclear whether models use the same instruction equally well across channels. We introduce Visualized Task Semantics (VTS), a controlled intervention that moves the question into the image while keeping the source problem and answer fixed. Across six MLLMs and four benchmarks, accuracy drops in all 24 model-task pairs, by 17.8 points on average. Models often transcribe the visual question correctly yet fai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.04726","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-05T11:46:15Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"8f440e447e15684025ce2b81330a2c622255a8c2e76d0bb2352b048fb4ea2e27","abstract_canon_sha256":"4228e8b7241fa31a5331c909b1c5b657f69cb8963e83d2703a390bdad859c7fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:46:57.139367Z","signature_b64":"x4dgnEnHEzUn1esFfKtwlYA7EHCERvLkNvEom+WHeaj8E47D+J/9VFEYilnMymGURKvXkE2LLapTimLxc39LBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80d0a83b8a9ddd3184592bf1fae64a602db539dc56dc76ba3bf91aa21e2ab45b","last_reissued_at":"2026-08-06T01:46:57.137960Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:46:57.137960Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Prompts Become Pixels: Prompt-Region Grounding for Multimodal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Ruizhe Zhou, Xiaodan Liang, Xiaojun Chang, Xuemin Zhao, Yingying Zhu, Yongxin Wang, Yueling Tang","submitted_at":"2026-08-05T11:46:15Z","abstract_excerpt":"Multimodal large language models increasingly reason over screenshots and documents where the task itself may be written in pixels. Yet benchmarks usually place questions in text, leaving it unclear whether models use the same instruction equally well across channels. We introduce Visualized Task Semantics (VTS), a controlled intervention that moves the question into the image while keeping the source problem and answer fixed. Across six MLLMs and four benchmarks, accuracy drops in all 24 model-task pairs, by 17.8 points on average. Models often transcribe the visual question correctly yet fai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.04726","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.04726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.04726","created_at":"2026-08-06T01:46:57.139769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.04726v1","created_at":"2026-08-06T01:46:57.139769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.04726","created_at":"2026-08-06T01:46:57.139769+00:00"},{"alias_kind":"pith_short_12","alias_value":"QDIKQO4KTXOT","created_at":"2026-08-06T01:46:57.139769+00:00"},{"alias_kind":"pith_short_16","alias_value":"QDIKQO4KTXOTDBCZ","created_at":"2026-08-06T01:46:57.139769+00:00"},{"alias_kind":"pith_short_8","alias_value":"QDIKQO4K","created_at":"2026-08-06T01:46:57.139769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA","json":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA.json","graph_json":"https://pith.science/api/pith-number/QDIKQO4KTXOTDBCZFPY7VZSKMA/graph.json","events_json":"https://pith.science/api/pith-number/QDIKQO4KTXOTDBCZFPY7VZSKMA/events.json","paper":"https://pith.science/paper/QDIKQO4K"},"agent_actions":{"view_html":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA","download_json":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA.json","view_paper":"https://pith.science/paper/QDIKQO4K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.04726&json=true","fetch_graph":"https://pith.science/api/pith-number/QDIKQO4KTXOTDBCZFPY7VZSKMA/graph.json","fetch_events":"https://pith.science/api/pith-number/QDIKQO4KTXOTDBCZFPY7VZSKMA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA/action/storage_attestation","attest_author":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA/action/author_attestation","sign_citation":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA/action/citation_signature","submit_replication":"https://pith.science/pith/QDIKQO4KTXOTDBCZFPY7VZSKMA/action/replication_record"}},"created_at":"2026-08-06T01:46:57.139769+00:00","updated_at":"2026-08-06T01:46:57.139769+00:00"}