{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LC5PNUVPGAR6PGEITNGFPUN2CR","short_pith_number":"pith:LC5PNUVP","schema_version":"1.0","canonical_sha256":"58baf6d2af3023e798889b4c57d1ba1475a826aee5f43e707103965f743cd554","source":{"kind":"arxiv","id":"2410.09575","version":2},"attestation_state":"computed","paper":{"title":"Reconstructive Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anlin Zheng, Haochen Wang, Tiancai Wang, Xiangyu Zhang, Yucheng Zhao, Zhaoxiang Zhang, Zheng Ge","submitted_at":"2024-10-12T15:54:29Z","abstract_excerpt":"This paper introduces reconstructive visual instruction tuning (ROSS), a family of Large Multimodal Models (LMMs) that exploit vision-centric supervision signals. In contrast to conventional visual instruction tuning approaches that exclusively supervise text outputs, ROSS prompts LMMs to supervise visual outputs via reconstructing input images. By doing so, it capitalizes on the inherent richness and detail present within input images themselves, which are often lost in pure text supervision. However, producing meaningful feedback from natural images is challenging due to the heavy spatial re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09575","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-12T15:54:29Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"d07816aa976a72eb0e4a47c9c61070367695bd328b400bb0dab00e0e2404c114","abstract_canon_sha256":"5a9db50d9b1e2212bc3713dbadf30d4edb697675292cb4c37270d335006890f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:35.095616Z","signature_b64":"utyKZ4bfPI9a31uLaTXFidXMgstnCegtnEN+CBXwTLrLqlD1Qz51PevJ3Pk7eOEfmbcrw8Mtfrfg2y+df4rWBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58baf6d2af3023e798889b4c57d1ba1475a826aee5f43e707103965f743cd554","last_reissued_at":"2026-07-05T09:55:35.095111Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:35.095111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reconstructive Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anlin Zheng, Haochen Wang, Tiancai Wang, Xiangyu Zhang, Yucheng Zhao, Zhaoxiang Zhang, Zheng Ge","submitted_at":"2024-10-12T15:54:29Z","abstract_excerpt":"This paper introduces reconstructive visual instruction tuning (ROSS), a family of Large Multimodal Models (LMMs) that exploit vision-centric supervision signals. In contrast to conventional visual instruction tuning approaches that exclusively supervise text outputs, ROSS prompts LMMs to supervise visual outputs via reconstructing input images. By doing so, it capitalizes on the inherent richness and detail present within input images themselves, which are often lost in pure text supervision. However, producing meaningful feedback from natural images is challenging due to the heavy spatial re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09575","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09575/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09575","created_at":"2026-07-05T09:55:35.095169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09575v2","created_at":"2026-07-05T09:55:35.095169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09575","created_at":"2026-07-05T09:55:35.095169+00:00"},{"alias_kind":"pith_short_12","alias_value":"LC5PNUVPGAR6","created_at":"2026-07-05T09:55:35.095169+00:00"},{"alias_kind":"pith_short_16","alias_value":"LC5PNUVPGAR6PGEI","created_at":"2026-07-05T09:55:35.095169+00:00"},{"alias_kind":"pith_short_8","alias_value":"LC5PNUVP","created_at":"2026-07-05T09:55:35.095169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18160","citing_title":"Vision Inference Former: Sustaining Visual Consistency in Multimodal Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18018","citing_title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18160","citing_title":"Vision Inference Former: Sustaining Visual Consistency in Multimodal Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11991","citing_title":"VGR: Visual Grounded Reasoning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12796","citing_title":"DriveVLA-W0: World Models Amplify Data Scaling Law in Autonomous Driving","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21343","citing_title":"Latent Denoising Improves Visual Alignment in Large Multimodal Models","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR","json":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR.json","graph_json":"https://pith.science/api/pith-number/LC5PNUVPGAR6PGEITNGFPUN2CR/graph.json","events_json":"https://pith.science/api/pith-number/LC5PNUVPGAR6PGEITNGFPUN2CR/events.json","paper":"https://pith.science/paper/LC5PNUVP"},"agent_actions":{"view_html":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR","download_json":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR.json","view_paper":"https://pith.science/paper/LC5PNUVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09575&json=true","fetch_graph":"https://pith.science/api/pith-number/LC5PNUVPGAR6PGEITNGFPUN2CR/graph.json","fetch_events":"https://pith.science/api/pith-number/LC5PNUVPGAR6PGEITNGFPUN2CR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR/action/storage_attestation","attest_author":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR/action/author_attestation","sign_citation":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR/action/citation_signature","submit_replication":"https://pith.science/pith/LC5PNUVPGAR6PGEITNGFPUN2CR/action/replication_record"}},"created_at":"2026-07-05T09:55:35.095169+00:00","updated_at":"2026-07-05T09:55:35.095169+00:00"}