{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MGNCP2ZMGWT2JCWYTUS4JK3D4Y","short_pith_number":"pith:MGNCP2ZM","schema_version":"1.0","canonical_sha256":"619a27eb2c35a7a48ad89d25c4ab63e60edd42dc1356ccd25d45d8bdcc0cb9e9","source":{"kind":"arxiv","id":"2608.00743","version":1},"attestation_state":"computed","paper":{"title":"LUT: Latent Utility Training for Visual Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaxuan Kang, Mingda Li, Mingjie Liu, Siyu Chen, Tianyue Wang, YanChao Hao, Yongheng Zhang, Zhaoyang Wei, Zheng Wei","submitted_at":"2026-08-01T16:27:45Z","abstract_excerpt":"Multimodal large language models have advanced visual understanding, yet perception-intensive reasoning remains challenging. Recent latent visual reasoning methods introduce hidden-space computation before answering, but they often rely on costly intermediate supervision, such as bounding boxes, sketches, or interleaved rationales. These strategies focus on how latent states should be shaped, but do not explicitly assess whether the latent is useful for the final answer. We propose LUT, a latent reasoning framework trained with only standard VQA pairs. LUT centers training on Latent Utility at"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.00743","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-01T16:27:45Z","cross_cats_sorted":[],"title_canon_sha256":"45300dc11d5de4c1d5f54d4b0ced4ddedabcfcdcdb359380f9d639e081f72900","abstract_canon_sha256":"305c4790e39a0540a1d9e3379e3ef2c0e4ab24e3e1488b073c80ea45121f2bd4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T01:54:51.705693Z","signature_b64":"claY5FHLicTmVlGwNRwG2/eIpIeK4LQgCAt/v8dx692B4D8eIcASx9hN2l2fHimnw56Yc5iuKqHYLcvB5KFpCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"619a27eb2c35a7a48ad89d25c4ab63e60edd42dc1356ccd25d45d8bdcc0cb9e9","last_reissued_at":"2026-08-04T01:54:51.704043Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T01:54:51.704043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LUT: Latent Utility Training for Visual Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaxuan Kang, Mingda Li, Mingjie Liu, Siyu Chen, Tianyue Wang, YanChao Hao, Yongheng Zhang, Zhaoyang Wei, Zheng Wei","submitted_at":"2026-08-01T16:27:45Z","abstract_excerpt":"Multimodal large language models have advanced visual understanding, yet perception-intensive reasoning remains challenging. Recent latent visual reasoning methods introduce hidden-space computation before answering, but they often rely on costly intermediate supervision, such as bounding boxes, sketches, or interleaved rationales. These strategies focus on how latent states should be shaped, but do not explicitly assess whether the latent is useful for the final answer. We propose LUT, a latent reasoning framework trained with only standard VQA pairs. LUT centers training on Latent Utility at"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.00743","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.00743/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.00743","created_at":"2026-08-04T01:54:51.705311+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.00743v1","created_at":"2026-08-04T01:54:51.705311+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.00743","created_at":"2026-08-04T01:54:51.705311+00:00"},{"alias_kind":"pith_short_12","alias_value":"MGNCP2ZMGWT2","created_at":"2026-08-04T01:54:51.705311+00:00"},{"alias_kind":"pith_short_16","alias_value":"MGNCP2ZMGWT2JCWY","created_at":"2026-08-04T01:54:51.705311+00:00"},{"alias_kind":"pith_short_8","alias_value":"MGNCP2ZM","created_at":"2026-08-04T01:54:51.705311+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y","json":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y.json","graph_json":"https://pith.science/api/pith-number/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/graph.json","events_json":"https://pith.science/api/pith-number/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/events.json","paper":"https://pith.science/paper/MGNCP2ZM"},"agent_actions":{"view_html":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y","download_json":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y.json","view_paper":"https://pith.science/paper/MGNCP2ZM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.00743&json=true","fetch_graph":"https://pith.science/api/pith-number/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/graph.json","fetch_events":"https://pith.science/api/pith-number/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/action/storage_attestation","attest_author":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/action/author_attestation","sign_citation":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/action/citation_signature","submit_replication":"https://pith.science/pith/MGNCP2ZMGWT2JCWYTUS4JK3D4Y/action/replication_record"}},"created_at":"2026-08-04T01:54:51.705311+00:00","updated_at":"2026-08-04T01:54:51.705311+00:00"}