{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WEKMC6DM5U7U3GX4GT2B6PHIYF","short_pith_number":"pith:WEKMC6DM","schema_version":"1.0","canonical_sha256":"b114c1786ced3f4d9afc34f41f3ce8c145ac2de4993a906f59512be6789acfc3","source":{"kind":"arxiv","id":"2504.07165","version":1},"attestation_state":"computed","paper":{"title":"Perception in Reflection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"En Yu, Haoran Wei, Jianjian Sun, Kangheng Lin, Liang Zhao, Runpei Dong, Vishal M. Patel, Xiangyu Zhang, Yana Wei, Yuang Peng, Zheng Ge","submitted_at":"2025-04-09T17:59:02Z","abstract_excerpt":"We present a perception in reflection paradigm designed to transcend the limitations of current large vision-language models (LVLMs), which are expected yet often fail to achieve perfect perception initially. Specifically, we propose Reflective Perception (RePer), a dual-model reflection mechanism that systematically alternates between policy and critic models, enables iterative refinement of visual perception. This framework is powered by Reflective Perceptual Learning (RPL), which reinforces intrinsic reflective capabilities through a methodically constructed visual reflection dataset and re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07165","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-09T17:59:02Z","cross_cats_sorted":[],"title_canon_sha256":"c8be93da322f916b16c60bd854f75b350e8de4e541b597e12eeff3430a56da45","abstract_canon_sha256":"907f78cf431425ecd7069c8f9e98a85be44efcf102a0db6e8e2d712ecd44ca7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:04.512611Z","signature_b64":"irYltL32xU8rV+KkdYjfxlo/+3htBb6BLNUbMc5MKjrdeng+1ZLonxJd7ew2/JY7nc2AB4bjxFXYGflM0/eiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b114c1786ced3f4d9afc34f41f3ce8c145ac2de4993a906f59512be6789acfc3","last_reissued_at":"2026-07-05T10:47:04.512124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:04.512124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Perception in Reflection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"En Yu, Haoran Wei, Jianjian Sun, Kangheng Lin, Liang Zhao, Runpei Dong, Vishal M. Patel, Xiangyu Zhang, Yana Wei, Yuang Peng, Zheng Ge","submitted_at":"2025-04-09T17:59:02Z","abstract_excerpt":"We present a perception in reflection paradigm designed to transcend the limitations of current large vision-language models (LVLMs), which are expected yet often fail to achieve perfect perception initially. Specifically, we propose Reflective Perception (RePer), a dual-model reflection mechanism that systematically alternates between policy and critic models, enables iterative refinement of visual perception. This framework is powered by Reflective Perceptual Learning (RPL), which reinforces intrinsic reflective capabilities through a methodically constructed visual reflection dataset and re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07165","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07165/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07165","created_at":"2026-07-05T10:47:04.512184+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07165v1","created_at":"2026-07-05T10:47:04.512184+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07165","created_at":"2026-07-05T10:47:04.512184+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEKMC6DM5U7U","created_at":"2026-07-05T10:47:04.512184+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEKMC6DM5U7U3GX4","created_at":"2026-07-05T10:47:04.512184+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEKMC6DM","created_at":"2026-07-05T10:47:04.512184+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31504","citing_title":"SimpleSearch-VL: A Simple Recipe for Multimodal Agentic Deep Search","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30288","citing_title":"VisReflect: Latent Visual Reflection for Fine-Grained Perception in Long Visual Context","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27494","citing_title":"Learning to Focus and Precise Cropping: A Reinforcement Learning Framework with Information Gaps and Grounding Loss for MLLMs","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21718","citing_title":"Building a Precise Video Language with Human-AI Oversight","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF","json":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF.json","graph_json":"https://pith.science/api/pith-number/WEKMC6DM5U7U3GX4GT2B6PHIYF/graph.json","events_json":"https://pith.science/api/pith-number/WEKMC6DM5U7U3GX4GT2B6PHIYF/events.json","paper":"https://pith.science/paper/WEKMC6DM"},"agent_actions":{"view_html":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF","download_json":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF.json","view_paper":"https://pith.science/paper/WEKMC6DM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07165&json=true","fetch_graph":"https://pith.science/api/pith-number/WEKMC6DM5U7U3GX4GT2B6PHIYF/graph.json","fetch_events":"https://pith.science/api/pith-number/WEKMC6DM5U7U3GX4GT2B6PHIYF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF/action/storage_attestation","attest_author":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF/action/author_attestation","sign_citation":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF/action/citation_signature","submit_replication":"https://pith.science/pith/WEKMC6DM5U7U3GX4GT2B6PHIYF/action/replication_record"}},"created_at":"2026-07-05T10:47:04.512184+00:00","updated_at":"2026-07-05T10:47:04.512184+00:00"}