{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TLZQ2T4O6EIB6UEEHJ2NQAVDLJ","short_pith_number":"pith:TLZQ2T4O","schema_version":"1.0","canonical_sha256":"9af30d4f8ef1101f50843a74d802a35a50fa5be6987ad3f1ea0ebaf6ffb33975","source":{"kind":"arxiv","id":"2602.23615","version":3},"attestation_state":"computed","paper":{"title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anqi Chen, Cong Wang, Feng Miao, Jiacheng Yang, Qi Fan, Wenbin Li, Yang Gao, Yunkai Dang","submitted_at":"2026-02-27T02:43:35Z","abstract_excerpt":"Current Large Multimodal Models (LMMs) struggle with high-resolution visual inputs during the reasoning process, as the number of image tokens increases quadratically with resolution, introducing substantial redundancy and irrelevant information. A common practice is to identify key image regions and refer to their high-resolution counterparts during reasoning, typically trained with external visual supervision. However, such visual supervision cues require costly grounding labels from human annotators. Meanwhile, it remains an open question how to enhance a model's grounding abilities to supp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.23615","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-02-27T02:43:35Z","cross_cats_sorted":[],"title_canon_sha256":"258f2ee60be9f74ac9052578696419e085fc974e04157431ca73d2f78f846d62","abstract_canon_sha256":"c94124f22e39980b959832f69e1177fc255c769b3e6959309c316b61fbce030f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T01:19:52.144097Z","signature_b64":"Xo7+xh89RKWzOsDVVD8R2ZAuQISshKjDhNOW+vp6wGXeSSnrvOmF+G1M93loDim21gIJSBlJV3BUOJaQurNNCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9af30d4f8ef1101f50843a74d802a35a50fa5be6987ad3f1ea0ebaf6ffb33975","last_reissued_at":"2026-07-09T01:19:52.143614Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T01:19:52.143614Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anqi Chen, Cong Wang, Feng Miao, Jiacheng Yang, Qi Fan, Wenbin Li, Yang Gao, Yunkai Dang","submitted_at":"2026-02-27T02:43:35Z","abstract_excerpt":"Current Large Multimodal Models (LMMs) struggle with high-resolution visual inputs during the reasoning process, as the number of image tokens increases quadratically with resolution, introducing substantial redundancy and irrelevant information. A common practice is to identify key image regions and refer to their high-resolution counterparts during reasoning, typically trained with external visual supervision. However, such visual supervision cues require costly grounding labels from human annotators. Meanwhile, it remains an open question how to enhance a model's grounding abilities to supp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.23615","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.23615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.23615","created_at":"2026-07-09T01:19:52.143673+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.23615v3","created_at":"2026-07-09T01:19:52.143673+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.23615","created_at":"2026-07-09T01:19:52.143673+00:00"},{"alias_kind":"pith_short_12","alias_value":"TLZQ2T4O6EIB","created_at":"2026-07-09T01:19:52.143673+00:00"},{"alias_kind":"pith_short_16","alias_value":"TLZQ2T4O6EIB6UEE","created_at":"2026-07-09T01:19:52.143673+00:00"},{"alias_kind":"pith_short_8","alias_value":"TLZQ2T4O","created_at":"2026-07-09T01:19:52.143673+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.13565","citing_title":"UHR-BAT: Budget-Aware Token Compression Vision-Language model for Ultra-High-Resolution Remote Sensing","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ","json":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ.json","graph_json":"https://pith.science/api/pith-number/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/graph.json","events_json":"https://pith.science/api/pith-number/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/events.json","paper":"https://pith.science/paper/TLZQ2T4O"},"agent_actions":{"view_html":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ","download_json":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ.json","view_paper":"https://pith.science/paper/TLZQ2T4O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.23615&json=true","fetch_graph":"https://pith.science/api/pith-number/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/graph.json","fetch_events":"https://pith.science/api/pith-number/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/action/storage_attestation","attest_author":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/action/author_attestation","sign_citation":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/action/citation_signature","submit_replication":"https://pith.science/pith/TLZQ2T4O6EIB6UEEHJ2NQAVDLJ/action/replication_record"}},"created_at":"2026-07-09T01:19:52.143673+00:00","updated_at":"2026-07-09T01:19:52.143673+00:00"}