{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SVSH7KJFI6UGEIPKU7K7XQPBEZ","short_pith_number":"pith:SVSH7KJF","schema_version":"1.0","canonical_sha256":"95647fa92547a86221eaa7d5fbc1e1266d821f390340e6fb8ddd901b79523f28","source":{"kind":"arxiv","id":"2506.07235","version":1},"attestation_state":"computed","paper":{"title":"Multi-Step Visual Reasoning with Visual Tokens Scaling and Verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Binhang Yuan, Bohan Zeng, Conghui He, Fupeng Sun, Guangxin He, Jiantao Qiu, Tianyi Bai, Wentao Zhang, Yizhen Jiang, Zengjie Hu","submitted_at":"2025-06-08T17:38:49Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have achieved remarkable capabilities by integrating visual perception with language understanding, enabling applications such as image-grounded dialogue, visual question answering, and scientific analysis. However, most MLLMs adopt a static inference paradigm, encoding the entire image into fixed visual tokens upfront, which limits their ability to iteratively refine understanding or adapt to context during inference. This contrasts sharply with human perception, which is dynamic, selective, and feedback-driven. In this work, we introduce a novel fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07235","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-08T17:38:49Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c36ac0231399855a151f983505b7bbf7b24b3b9882d1089e64a77d4440a495b7","abstract_canon_sha256":"67eba042a6a7fa9c68d2e4026775b3dfefdad7e536c1b1184584c8e0c335a7d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:04.067753Z","signature_b64":"1lk9L5Y7jNBO6BuSgJ2UTWU7d5ZLts+ZuLZ8KJd8dZpK/tl1ztqCggnev1H0Pdbyl0MgSelcp014RTbIGFXFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95647fa92547a86221eaa7d5fbc1e1266d821f390340e6fb8ddd901b79523f28","last_reissued_at":"2026-07-05T11:18:04.067263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:04.067263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Step Visual Reasoning with Visual Tokens Scaling and Verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Binhang Yuan, Bohan Zeng, Conghui He, Fupeng Sun, Guangxin He, Jiantao Qiu, Tianyi Bai, Wentao Zhang, Yizhen Jiang, Zengjie Hu","submitted_at":"2025-06-08T17:38:49Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have achieved remarkable capabilities by integrating visual perception with language understanding, enabling applications such as image-grounded dialogue, visual question answering, and scientific analysis. However, most MLLMs adopt a static inference paradigm, encoding the entire image into fixed visual tokens upfront, which limits their ability to iteratively refine understanding or adapt to context during inference. This contrasts sharply with human perception, which is dynamic, selective, and feedback-driven. In this work, we introduce a novel fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07235","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07235","created_at":"2026-07-05T11:18:04.067311+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07235v1","created_at":"2026-07-05T11:18:04.067311+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07235","created_at":"2026-07-05T11:18:04.067311+00:00"},{"alias_kind":"pith_short_12","alias_value":"SVSH7KJFI6UG","created_at":"2026-07-05T11:18:04.067311+00:00"},{"alias_kind":"pith_short_16","alias_value":"SVSH7KJFI6UGEIPK","created_at":"2026-07-05T11:18:04.067311+00:00"},{"alias_kind":"pith_short_8","alias_value":"SVSH7KJF","created_at":"2026-07-05T11:18:04.067311+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27959","citing_title":"ROVER: Routing Object-Centric Visual Evidence for Grounded Multi-Image Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22177","citing_title":"Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09299","citing_title":"VABench: A Comprehensive Benchmark for Audio-Video Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11025","citing_title":"Test-time Scaling over Perception: Resolving the Grounding Paradox in Thinking with Images","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04707","citing_title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ","json":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ.json","graph_json":"https://pith.science/api/pith-number/SVSH7KJFI6UGEIPKU7K7XQPBEZ/graph.json","events_json":"https://pith.science/api/pith-number/SVSH7KJFI6UGEIPKU7K7XQPBEZ/events.json","paper":"https://pith.science/paper/SVSH7KJF"},"agent_actions":{"view_html":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ","download_json":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ.json","view_paper":"https://pith.science/paper/SVSH7KJF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07235&json=true","fetch_graph":"https://pith.science/api/pith-number/SVSH7KJFI6UGEIPKU7K7XQPBEZ/graph.json","fetch_events":"https://pith.science/api/pith-number/SVSH7KJFI6UGEIPKU7K7XQPBEZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ/action/storage_attestation","attest_author":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ/action/author_attestation","sign_citation":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ/action/citation_signature","submit_replication":"https://pith.science/pith/SVSH7KJFI6UGEIPKU7K7XQPBEZ/action/replication_record"}},"created_at":"2026-07-05T11:18:04.067311+00:00","updated_at":"2026-07-05T11:18:04.067311+00:00"}