{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SLG4GJLQJVT765QM4JP2C542UF","short_pith_number":"pith:SLG4GJLQ","schema_version":"1.0","canonical_sha256":"92cdc325704d67ff760ce25fa1779aa144fac721e35607b438bdd17f857927dc","source":{"kind":"arxiv","id":"2410.23570","version":1},"attestation_state":"computed","paper":{"title":"Phrase Decoupling Cross-Modal Hierarchical Matching and Progressive Position Correction for Visual Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dapeng Tao, Huafeng Li, Mengzhao Wang, Minghong Xie, Yafei Zhang, Zhengtao Yu","submitted_at":"2024-10-31T02:25:47Z","abstract_excerpt":"Visual grounding has attracted wide attention thanks to its broad application in various visual language tasks. Although visual grounding has made significant research progress, existing methods ignore the promotion effect of the association between text and image features at different hierarchies on cross-modal matching. This paper proposes a Phrase Decoupling Cross-Modal Hierarchical Matching and Progressive Position Correction Visual Grounding method. It first generates a mask through decoupled sentence phrases, and a text and image hierarchical matching mechanism is constructed, highlighti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23570","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-31T02:25:47Z","cross_cats_sorted":[],"title_canon_sha256":"ab19b91bec1a7c8ebeb4fcdf3a08a6fb2141bb312385b1945421939fa86a676a","abstract_canon_sha256":"2377afdd1c57b033d334cdcf817846deb7b350be326bc7b9db6cf6ebe0522bd6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:02.230042Z","signature_b64":"VDvWCljTwSllTHuUCiXJNlzK2p0qQjwKDCwTkS+foNjeR6M4AxDZRRm4QJnvEIfKw33lUdn05/JuCnVgzighDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92cdc325704d67ff760ce25fa1779aa144fac721e35607b438bdd17f857927dc","last_reissued_at":"2026-07-05T09:29:02.229636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:02.229636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Phrase Decoupling Cross-Modal Hierarchical Matching and Progressive Position Correction for Visual Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dapeng Tao, Huafeng Li, Mengzhao Wang, Minghong Xie, Yafei Zhang, Zhengtao Yu","submitted_at":"2024-10-31T02:25:47Z","abstract_excerpt":"Visual grounding has attracted wide attention thanks to its broad application in various visual language tasks. Although visual grounding has made significant research progress, existing methods ignore the promotion effect of the association between text and image features at different hierarchies on cross-modal matching. This paper proposes a Phrase Decoupling Cross-Modal Hierarchical Matching and Progressive Position Correction Visual Grounding method. It first generates a mask through decoupled sentence phrases, and a text and image hierarchical matching mechanism is constructed, highlighti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23570","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23570","created_at":"2026-07-05T09:29:02.229697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23570v1","created_at":"2026-07-05T09:29:02.229697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23570","created_at":"2026-07-05T09:29:02.229697+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLG4GJLQJVT7","created_at":"2026-07-05T09:29:02.229697+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLG4GJLQJVT765QM","created_at":"2026-07-05T09:29:02.229697+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLG4GJLQ","created_at":"2026-07-05T09:29:02.229697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.13543","citing_title":"Query-centric Audio-Visual Cognition Network for Moment Retrieval, Segmentation and Step-Captioning","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF","json":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF.json","graph_json":"https://pith.science/api/pith-number/SLG4GJLQJVT765QM4JP2C542UF/graph.json","events_json":"https://pith.science/api/pith-number/SLG4GJLQJVT765QM4JP2C542UF/events.json","paper":"https://pith.science/paper/SLG4GJLQ"},"agent_actions":{"view_html":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF","download_json":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF.json","view_paper":"https://pith.science/paper/SLG4GJLQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23570&json=true","fetch_graph":"https://pith.science/api/pith-number/SLG4GJLQJVT765QM4JP2C542UF/graph.json","fetch_events":"https://pith.science/api/pith-number/SLG4GJLQJVT765QM4JP2C542UF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF/action/storage_attestation","attest_author":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF/action/author_attestation","sign_citation":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF/action/citation_signature","submit_replication":"https://pith.science/pith/SLG4GJLQJVT765QM4JP2C542UF/action/replication_record"}},"created_at":"2026-07-05T09:29:02.229697+00:00","updated_at":"2026-07-05T09:29:02.229697+00:00"}