{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JDQNKYINR7RHQ2E5435MMKDTQI","short_pith_number":"pith:JDQNKYIN","schema_version":"1.0","canonical_sha256":"48e0d5610d8fe278689de6fac628738221f9bf9fa99a039e23c4aec551af93ac","source":{"kind":"arxiv","id":"2505.21868","version":1},"attestation_state":"computed","paper":{"title":"Cross-DINO: Cross the Deep MLP and Transformer for Small Object Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongmei Jiang, Guiping Cao, Jianguo Zhang, Wenjian Huang, Xiangyuan Lan, Yaowei Wang","submitted_at":"2025-05-28T01:33:23Z","abstract_excerpt":"Small Object Detection (SOD) poses significant challenges due to limited information and the model's low class prediction score. While Transformer-based detectors have shown promising performance, their potential for SOD remains largely unexplored. In typical DETR-like frameworks, the CNN backbone network, specialized in aggregating local information, struggles to capture the necessary contextual information for SOD. The multiple attention layers in the Transformer Encoder face difficulties in effectively attending to small objects and can also lead to blurring of features. Furthermore, the mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.21868","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-28T01:33:23Z","cross_cats_sorted":[],"title_canon_sha256":"646b3945520abeacec09adfa9de132b99d0381abb94507121fae98598395fedd","abstract_canon_sha256":"bfc4117b8820a8dc5fb28cc50f3c06397757a1b066e5539ec019d29edd2b8900"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:05.095145Z","signature_b64":"92Ya9R7zvQmUwTH1yvoyl+HrNk1wcLJJcTZoRBKL4CFWQmdSRSAKda/hwtIB5PIMgHgHqVR9NWrDzL/3uSCaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48e0d5610d8fe278689de6fac628738221f9bf9fa99a039e23c4aec551af93ac","last_reissued_at":"2026-07-05T11:11:05.094731Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:05.094731Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-DINO: Cross the Deep MLP and Transformer for Small Object Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongmei Jiang, Guiping Cao, Jianguo Zhang, Wenjian Huang, Xiangyuan Lan, Yaowei Wang","submitted_at":"2025-05-28T01:33:23Z","abstract_excerpt":"Small Object Detection (SOD) poses significant challenges due to limited information and the model's low class prediction score. While Transformer-based detectors have shown promising performance, their potential for SOD remains largely unexplored. In typical DETR-like frameworks, the CNN backbone network, specialized in aggregating local information, struggles to capture the necessary contextual information for SOD. The multiple attention layers in the Transformer Encoder face difficulties in effectively attending to small objects and can also lead to blurring of features. Furthermore, the mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.21868","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.21868/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.21868","created_at":"2026-07-05T11:11:05.094794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.21868v1","created_at":"2026-07-05T11:11:05.094794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.21868","created_at":"2026-07-05T11:11:05.094794+00:00"},{"alias_kind":"pith_short_12","alias_value":"JDQNKYINR7RH","created_at":"2026-07-05T11:11:05.094794+00:00"},{"alias_kind":"pith_short_16","alias_value":"JDQNKYINR7RHQ2E5","created_at":"2026-07-05T11:11:05.094794+00:00"},{"alias_kind":"pith_short_8","alias_value":"JDQNKYIN","created_at":"2026-07-05T11:11:05.094794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.19807","citing_title":"DS-Det: Single-Query Paradigm and Attention Disentangled Learning for Flexible Object Detection","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI","json":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI.json","graph_json":"https://pith.science/api/pith-number/JDQNKYINR7RHQ2E5435MMKDTQI/graph.json","events_json":"https://pith.science/api/pith-number/JDQNKYINR7RHQ2E5435MMKDTQI/events.json","paper":"https://pith.science/paper/JDQNKYIN"},"agent_actions":{"view_html":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI","download_json":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI.json","view_paper":"https://pith.science/paper/JDQNKYIN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.21868&json=true","fetch_graph":"https://pith.science/api/pith-number/JDQNKYINR7RHQ2E5435MMKDTQI/graph.json","fetch_events":"https://pith.science/api/pith-number/JDQNKYINR7RHQ2E5435MMKDTQI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI/action/storage_attestation","attest_author":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI/action/author_attestation","sign_citation":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI/action/citation_signature","submit_replication":"https://pith.science/pith/JDQNKYINR7RHQ2E5435MMKDTQI/action/replication_record"}},"created_at":"2026-07-05T11:11:05.094794+00:00","updated_at":"2026-07-05T11:11:05.094794+00:00"}