{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ZSBPEEIMCWZTWVR52ZV2I7J2UC","short_pith_number":"pith:ZSBPEEIM","schema_version":"1.0","canonical_sha256":"cc82f2110c15b33b563dd66ba47d3aa096d80ad8658e0a13aba35a1a9da39e36","source":{"kind":"arxiv","id":"2606.03603","version":1},"attestation_state":"computed","paper":{"title":"World Models Meet Language Models: On the Complementarity of Concrete and Abstract Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jianbing Shen, Wei Tao, Yiwen Guo, Yucheng Zhou","submitted_at":"2026-06-02T13:07:49Z","abstract_excerpt":"World models and multimodal large language models (MLLMs) provide complementary capabilities for predicting future outcomes from static visual observations. World models can generate concrete visual rollouts of possible futures, while MLLMs can reason abstractly over questions, goals, and rules. However, generated rollouts are stochastic and may be visually plausible but task-incorrect, making it necessary to determine when visual simulation is useful, whether a rollout is credible, and how it should influence the final answer. We formulate this problem as controlled concrete reasoning, where "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.03603","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-06-02T13:07:49Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5ed10486699b881891ba8b8db4bfe273b0421568a1f559319bb48c437e2cf4fe","abstract_canon_sha256":"c0f94b0cf61be8e1205bdb91f2b23cc7085d7c690b46ad2fcbe519346444526a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-03T01:06:02.104590Z","signature_b64":"D1xRiHqup52JPSYIEeqgIFpuY7ik7Gs54fB6F83Mv9cE2WZ8R/iQEpccjE2i7WuDhyHKzLZGtUO1h6EhF0PgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc82f2110c15b33b563dd66ba47d3aa096d80ad8658e0a13aba35a1a9da39e36","last_reissued_at":"2026-06-03T01:06:02.104101Z","signature_status":"signed_v1","first_computed_at":"2026-06-03T01:06:02.104101Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"World Models Meet Language Models: On the Complementarity of Concrete and Abstract Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jianbing Shen, Wei Tao, Yiwen Guo, Yucheng Zhou","submitted_at":"2026-06-02T13:07:49Z","abstract_excerpt":"World models and multimodal large language models (MLLMs) provide complementary capabilities for predicting future outcomes from static visual observations. World models can generate concrete visual rollouts of possible futures, while MLLMs can reason abstractly over questions, goals, and rules. However, generated rollouts are stochastic and may be visually plausible but task-incorrect, making it necessary to determine when visual simulation is useful, whether a rollout is credible, and how it should influence the final answer. We formulate this problem as controlled concrete reasoning, where "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.03603","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.03603/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.03603","created_at":"2026-06-03T01:06:02.104156+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.03603v1","created_at":"2026-06-03T01:06:02.104156+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.03603","created_at":"2026-06-03T01:06:02.104156+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSBPEEIMCWZT","created_at":"2026-06-03T01:06:02.104156+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSBPEEIMCWZTWVR5","created_at":"2026-06-03T01:06:02.104156+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSBPEEIM","created_at":"2026-06-03T01:06:02.104156+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC","json":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC.json","graph_json":"https://pith.science/api/pith-number/ZSBPEEIMCWZTWVR52ZV2I7J2UC/graph.json","events_json":"https://pith.science/api/pith-number/ZSBPEEIMCWZTWVR52ZV2I7J2UC/events.json","paper":"https://pith.science/paper/ZSBPEEIM"},"agent_actions":{"view_html":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC","download_json":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC.json","view_paper":"https://pith.science/paper/ZSBPEEIM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.03603&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSBPEEIMCWZTWVR52ZV2I7J2UC/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSBPEEIMCWZTWVR52ZV2I7J2UC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC/action/storage_attestation","attest_author":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC/action/author_attestation","sign_citation":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC/action/citation_signature","submit_replication":"https://pith.science/pith/ZSBPEEIMCWZTWVR52ZV2I7J2UC/action/replication_record"}},"created_at":"2026-06-03T01:06:02.104156+00:00","updated_at":"2026-06-03T01:06:02.104156+00:00"}