{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7AIP6NZNJD3EET6WWTSHWUEITA","short_pith_number":"pith:7AIP6NZN","schema_version":"1.0","canonical_sha256":"f810ff372d48f6424fd6b4e47b5088982c814742533ec14340610ac55d5bca48","source":{"kind":"arxiv","id":"2503.23330","version":1},"attestation_state":"computed","paper":{"title":"EagleVision: Object-level Attribute Multimodal LLM for Remote Sensing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guo Chen, Hongxiang Jiang, Jiaqi Feng, Jihao Yin, Qixiong Wang","submitted_at":"2025-03-30T06:13:13Z","abstract_excerpt":"Recent advances in multimodal large language models (MLLMs) have demonstrated impressive results in various visual tasks. However, in remote sensing (RS), high resolution and small proportion of objects pose challenges to existing MLLMs, which struggle with object-centric tasks, particularly in precise localization and fine-grained attribute description for each object. These RS MLLMs have not yet surpassed classical visual perception models, as they only provide coarse image understanding, leading to limited gains in real-world scenarios. To address this gap, we establish EagleVision, an MLLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23330","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-30T06:13:13Z","cross_cats_sorted":[],"title_canon_sha256":"e4fe1ada664953c3451392164ae020cb549500be0d5f971699bfcae902b97326","abstract_canon_sha256":"2cded08b7359ae6a0f391edb1f6efc3f34d200cf02af15f6a96c402ee9892c62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:34.590071Z","signature_b64":"/tbNvH3u8/WEw0lhft+cr49VLkW4WO5FajFiI36bt/h/ldaGgDpVtoOTAJrbEyMCOH43o0FrfLirKLpgz7dIBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f810ff372d48f6424fd6b4e47b5088982c814742533ec14340610ac55d5bca48","last_reissued_at":"2026-07-05T10:41:34.589613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:34.589613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EagleVision: Object-level Attribute Multimodal LLM for Remote Sensing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guo Chen, Hongxiang Jiang, Jiaqi Feng, Jihao Yin, Qixiong Wang","submitted_at":"2025-03-30T06:13:13Z","abstract_excerpt":"Recent advances in multimodal large language models (MLLMs) have demonstrated impressive results in various visual tasks. However, in remote sensing (RS), high resolution and small proportion of objects pose challenges to existing MLLMs, which struggle with object-centric tasks, particularly in precise localization and fine-grained attribute description for each object. These RS MLLMs have not yet surpassed classical visual perception models, as they only provide coarse image understanding, leading to limited gains in real-world scenarios. To address this gap, we establish EagleVision, an MLLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23330","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23330/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23330","created_at":"2026-07-05T10:41:34.589668+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23330v1","created_at":"2026-07-05T10:41:34.589668+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23330","created_at":"2026-07-05T10:41:34.589668+00:00"},{"alias_kind":"pith_short_12","alias_value":"7AIP6NZNJD3E","created_at":"2026-07-05T10:41:34.589668+00:00"},{"alias_kind":"pith_short_16","alias_value":"7AIP6NZNJD3EET6W","created_at":"2026-07-05T10:41:34.589668+00:00"},{"alias_kind":"pith_short_8","alias_value":"7AIP6NZN","created_at":"2026-07-05T10:41:34.589668+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.04978","citing_title":"Aligning Perception, Reasoning, Modeling and Interaction: A Survey on Physical AI","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17243","citing_title":"RemoteShield: Enable Robust Multimodal Large Language Models for Earth Observation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04451","citing_title":"RemoteZero: Geospatial Reasoning with Zero Human Annotations","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA","json":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA.json","graph_json":"https://pith.science/api/pith-number/7AIP6NZNJD3EET6WWTSHWUEITA/graph.json","events_json":"https://pith.science/api/pith-number/7AIP6NZNJD3EET6WWTSHWUEITA/events.json","paper":"https://pith.science/paper/7AIP6NZN"},"agent_actions":{"view_html":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA","download_json":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA.json","view_paper":"https://pith.science/paper/7AIP6NZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23330&json=true","fetch_graph":"https://pith.science/api/pith-number/7AIP6NZNJD3EET6WWTSHWUEITA/graph.json","fetch_events":"https://pith.science/api/pith-number/7AIP6NZNJD3EET6WWTSHWUEITA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA/action/storage_attestation","attest_author":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA/action/author_attestation","sign_citation":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA/action/citation_signature","submit_replication":"https://pith.science/pith/7AIP6NZNJD3EET6WWTSHWUEITA/action/replication_record"}},"created_at":"2026-07-05T10:41:34.589668+00:00","updated_at":"2026-07-05T10:41:34.589668+00:00"}