{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UQLRFPPTS4SCGQARACREXLYPF3","short_pith_number":"pith:UQLRFPPT","schema_version":"1.0","canonical_sha256":"a41712bdf3972423401100a24baf0f2ec16fc13e2f2d55ac9cebcaf4b7faa4c9","source":{"kind":"arxiv","id":"2412.20631","version":2},"attestation_state":"computed","paper":{"title":"Slow Perception: Let's Perceive Geometric Figures Step-by-step","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daxin Jiang, Haoran Wei, Jianjian Sun, Jia Wang, Liang Zhao, Xiangyu Zhang, Youyang Yin, Yumeng Li, Zheng Ge","submitted_at":"2024-12-30T00:40:35Z","abstract_excerpt":"Recently, \"visual o1\" began to enter people's vision, with expectations that this slow-thinking design can solve visual reasoning tasks, especially geometric math problems. However, the reality is that current LVLMs (Large Vision Language Models) can hardly even accurately copy a geometric figure, let alone truly understand the complex inherent logic and spatial relationships within geometric shapes. We believe accurate copying (strong perception) is the first step to visual o1. Accordingly, we introduce the concept of \"slow perception\" (SP), which guides the model to gradually perceive basic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.20631","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-30T00:40:35Z","cross_cats_sorted":[],"title_canon_sha256":"d9b22a0d1ba74cb2fb432ba3e1d8a66e8d4cfdacb854838abf38174201cb968f","abstract_canon_sha256":"1475bc96c028a955f7bddb8f7c8d9222e6a539f379ff64879e0931ce8014de99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:15.813154Z","signature_b64":"f3Oxcw/RqUqaQXiwAiF3xmeinhi7fh4LDJ8xhXtjjIs0TXkMtpni1RrQJJCyvMEj9RQXwKlueb1xwXly6w1fBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a41712bdf3972423401100a24baf0f2ec16fc13e2f2d55ac9cebcaf4b7faa4c9","last_reissued_at":"2026-07-05T10:05:15.812646Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:15.812646Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Slow Perception: Let's Perceive Geometric Figures Step-by-step","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daxin Jiang, Haoran Wei, Jianjian Sun, Jia Wang, Liang Zhao, Xiangyu Zhang, Youyang Yin, Yumeng Li, Zheng Ge","submitted_at":"2024-12-30T00:40:35Z","abstract_excerpt":"Recently, \"visual o1\" began to enter people's vision, with expectations that this slow-thinking design can solve visual reasoning tasks, especially geometric math problems. However, the reality is that current LVLMs (Large Vision Language Models) can hardly even accurately copy a geometric figure, let alone truly understand the complex inherent logic and spatial relationships within geometric shapes. We believe accurate copying (strong perception) is the first step to visual o1. Accordingly, we introduce the concept of \"slow perception\" (SP), which guides the model to gradually perceive basic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.20631","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.20631/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.20631","created_at":"2026-07-05T10:05:15.812706+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.20631v2","created_at":"2026-07-05T10:05:15.812706+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.20631","created_at":"2026-07-05T10:05:15.812706+00:00"},{"alias_kind":"pith_short_12","alias_value":"UQLRFPPTS4SC","created_at":"2026-07-05T10:05:15.812706+00:00"},{"alias_kind":"pith_short_16","alias_value":"UQLRFPPTS4SCGQAR","created_at":"2026-07-05T10:05:15.812706+00:00"},{"alias_kind":"pith_short_8","alias_value":"UQLRFPPT","created_at":"2026-07-05T10:05:15.812706+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00148","citing_title":"StemBind: When MLLMs Get Lost Between Rules and Instances in Abstract Visual Reasoning","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14998","citing_title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21164","citing_title":"Concise Geometric Description as a Bridge: Unleashing the Potential of LLM for Plane Geometry Problem Solving","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18234","citing_title":"DeepSeek-OCR: Contexts Optical Compression","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20806","citing_title":"OMIBench: Benchmarking Olympiad-Level Multi-Image Reasoning in Large Vision-Language Model","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3","json":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3.json","graph_json":"https://pith.science/api/pith-number/UQLRFPPTS4SCGQARACREXLYPF3/graph.json","events_json":"https://pith.science/api/pith-number/UQLRFPPTS4SCGQARACREXLYPF3/events.json","paper":"https://pith.science/paper/UQLRFPPT"},"agent_actions":{"view_html":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3","download_json":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3.json","view_paper":"https://pith.science/paper/UQLRFPPT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.20631&json=true","fetch_graph":"https://pith.science/api/pith-number/UQLRFPPTS4SCGQARACREXLYPF3/graph.json","fetch_events":"https://pith.science/api/pith-number/UQLRFPPTS4SCGQARACREXLYPF3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3/action/storage_attestation","attest_author":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3/action/author_attestation","sign_citation":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3/action/citation_signature","submit_replication":"https://pith.science/pith/UQLRFPPTS4SCGQARACREXLYPF3/action/replication_record"}},"created_at":"2026-07-05T10:05:15.812706+00:00","updated_at":"2026-07-05T10:05:15.812706+00:00"}