{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QJL6DTWEC3R3DJEPX5WBWXJE2V","short_pith_number":"pith:QJL6DTWE","schema_version":"1.0","canonical_sha256":"8257e1cec416e3b1a48fbf6c1b5d24d5446f890ff6b0fe6a050158b882ff994c","source":{"kind":"arxiv","id":"2409.16723","version":2},"attestation_state":"computed","paper":{"title":"EAGLE: Towards Efficient Arbitrary Referring Visual Prompts Comprehension for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiacheng Zhang, Jingjing Chen, Shaoxiang Chen, Yang Jiao, Yu-Gang Jiang","submitted_at":"2024-09-25T08:22:00Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have sparked great research interests owing to their exceptional content-reasoning and instruction-following capabilities. To effectively instruct an MLLM, in addition to conventional language expressions, the practice of referring to objects by painting with brushes on images has emerged as a prevalent tool (referred to as \"referring visual prompts\") due to its efficacy in aligning the user's intention with specific image regions. To accommodate the most common referring visual prompts, namely points, boxes, and masks, existing approaches ini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16723","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-25T08:22:00Z","cross_cats_sorted":[],"title_canon_sha256":"8cff71930cd7d786c0f0fb957f7c71268e90e0b55a998b027ed604c04ef0232c","abstract_canon_sha256":"b648b39f16e9cf03965d8b66ebf2a1f00356f53b1376161beb83c810ffa467e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:10.466006Z","signature_b64":"gqXDZY5enTK/o+MZ5wlUx4QAQ5H6QCHo8r3Lnve5XDE7XQJPvZ/1D2YN2fwKjyYSBVOsqLCXfZV2KYDKsBGGCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8257e1cec416e3b1a48fbf6c1b5d24d5446f890ff6b0fe6a050158b882ff994c","last_reissued_at":"2026-07-05T09:12:10.465511Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:10.465511Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EAGLE: Towards Efficient Arbitrary Referring Visual Prompts Comprehension for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiacheng Zhang, Jingjing Chen, Shaoxiang Chen, Yang Jiao, Yu-Gang Jiang","submitted_at":"2024-09-25T08:22:00Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have sparked great research interests owing to their exceptional content-reasoning and instruction-following capabilities. To effectively instruct an MLLM, in addition to conventional language expressions, the practice of referring to objects by painting with brushes on images has emerged as a prevalent tool (referred to as \"referring visual prompts\") due to its efficacy in aligning the user's intention with specific image regions. To accommodate the most common referring visual prompts, namely points, boxes, and masks, existing approaches ini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16723","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16723/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16723","created_at":"2026-07-05T09:12:10.465569+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16723v2","created_at":"2026-07-05T09:12:10.465569+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16723","created_at":"2026-07-05T09:12:10.465569+00:00"},{"alias_kind":"pith_short_12","alias_value":"QJL6DTWEC3R3","created_at":"2026-07-05T09:12:10.465569+00:00"},{"alias_kind":"pith_short_16","alias_value":"QJL6DTWEC3R3DJEP","created_at":"2026-07-05T09:12:10.465569+00:00"},{"alias_kind":"pith_short_8","alias_value":"QJL6DTWE","created_at":"2026-07-05T09:12:10.465569+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":225,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V","json":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V.json","graph_json":"https://pith.science/api/pith-number/QJL6DTWEC3R3DJEPX5WBWXJE2V/graph.json","events_json":"https://pith.science/api/pith-number/QJL6DTWEC3R3DJEPX5WBWXJE2V/events.json","paper":"https://pith.science/paper/QJL6DTWE"},"agent_actions":{"view_html":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V","download_json":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V.json","view_paper":"https://pith.science/paper/QJL6DTWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16723&json=true","fetch_graph":"https://pith.science/api/pith-number/QJL6DTWEC3R3DJEPX5WBWXJE2V/graph.json","fetch_events":"https://pith.science/api/pith-number/QJL6DTWEC3R3DJEPX5WBWXJE2V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V/action/storage_attestation","attest_author":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V/action/author_attestation","sign_citation":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V/action/citation_signature","submit_replication":"https://pith.science/pith/QJL6DTWEC3R3DJEPX5WBWXJE2V/action/replication_record"}},"created_at":"2026-07-05T09:12:10.465569+00:00","updated_at":"2026-07-05T09:12:10.465569+00:00"}