{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JTRKMG4X57NYQX4KBJFUEFUPF7","short_pith_number":"pith:JTRKMG4X","schema_version":"1.0","canonical_sha256":"4ce2a61b97efdb885f8a0a4b42168f2ffdc76aef6cf226e69b50329a79bf7954","source":{"kind":"arxiv","id":"2407.14563","version":1},"attestation_state":"computed","paper":{"title":"Learning Visual Grounding from Generative Vision and Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Taalimi, Chen Sun, Dahun Kim, Shijie Wang, Weicheng Kuo","submitted_at":"2024-07-18T20:29:49Z","abstract_excerpt":"Visual grounding tasks aim to localize image regions based on natural language references. In this work, we explore whether generative VLMs predominantly trained on image-text data could be leveraged to scale up the text annotation of visual grounding data. We find that grounding knowledge already exists in generative VLM and can be elicited by proper prompting. We thus prompt a VLM to generate object-level descriptions by feeding it object regions from existing object detection datasets. We further propose attribute modeling to explicitly capture the important object attributes, and spatial r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.14563","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-18T20:29:49Z","cross_cats_sorted":[],"title_canon_sha256":"6b20f5e8cc7151227820a10c5f4c62f8471cba36e145a937b0587efe747e241c","abstract_canon_sha256":"51811d9e7bd57178ae39f28f6baa4bbe1d65dab40ca7e926868e46dc3bba0377"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:29.375677Z","signature_b64":"uM8o65lqlUIMryg3L4wnVM05j7PLjWYj8Js41ywZ1Anl/MntQpj6+xXKDF9I4b+pI/kZJ6f8uUZn7Ohe5DslAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ce2a61b97efdb885f8a0a4b42168f2ffdc76aef6cf226e69b50329a79bf7954","last_reissued_at":"2026-07-05T08:46:29.375306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:29.375306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Visual Grounding from Generative Vision and Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Taalimi, Chen Sun, Dahun Kim, Shijie Wang, Weicheng Kuo","submitted_at":"2024-07-18T20:29:49Z","abstract_excerpt":"Visual grounding tasks aim to localize image regions based on natural language references. In this work, we explore whether generative VLMs predominantly trained on image-text data could be leveraged to scale up the text annotation of visual grounding data. We find that grounding knowledge already exists in generative VLM and can be elicited by proper prompting. We thus prompt a VLM to generate object-level descriptions by feeding it object regions from existing object detection datasets. We further propose attribute modeling to explicitly capture the important object attributes, and spatial r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.14563","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.14563/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.14563","created_at":"2026-07-05T08:46:29.375376+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.14563v1","created_at":"2026-07-05T08:46:29.375376+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.14563","created_at":"2026-07-05T08:46:29.375376+00:00"},{"alias_kind":"pith_short_12","alias_value":"JTRKMG4X57NY","created_at":"2026-07-05T08:46:29.375376+00:00"},{"alias_kind":"pith_short_16","alias_value":"JTRKMG4X57NYQX4K","created_at":"2026-07-05T08:46:29.375376+00:00"},{"alias_kind":"pith_short_8","alias_value":"JTRKMG4X","created_at":"2026-07-05T08:46:29.375376+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.09456","citing_title":"IAG: Input-aware Backdoor Attack on VLM-based Visual Grounding","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7","json":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7.json","graph_json":"https://pith.science/api/pith-number/JTRKMG4X57NYQX4KBJFUEFUPF7/graph.json","events_json":"https://pith.science/api/pith-number/JTRKMG4X57NYQX4KBJFUEFUPF7/events.json","paper":"https://pith.science/paper/JTRKMG4X"},"agent_actions":{"view_html":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7","download_json":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7.json","view_paper":"https://pith.science/paper/JTRKMG4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.14563&json=true","fetch_graph":"https://pith.science/api/pith-number/JTRKMG4X57NYQX4KBJFUEFUPF7/graph.json","fetch_events":"https://pith.science/api/pith-number/JTRKMG4X57NYQX4KBJFUEFUPF7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7/action/storage_attestation","attest_author":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7/action/author_attestation","sign_citation":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7/action/citation_signature","submit_replication":"https://pith.science/pith/JTRKMG4X57NYQX4KBJFUEFUPF7/action/replication_record"}},"created_at":"2026-07-05T08:46:29.375376+00:00","updated_at":"2026-07-05T08:46:29.375376+00:00"}