{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:FOCF5OECKUNX3UIH2DTI7J7JCI","short_pith_number":"pith:FOCF5OEC","schema_version":"1.0","canonical_sha256":"2b845eb882551b7dd107d0e68fa7e9123f38fc4a23babdf6199ba370f780d1c8","source":{"kind":"arxiv","id":"2111.11429","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Detection Transfer Learning with Vision Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kaiming He, Piotr Dollar, Ross Girshick, Saining Xie, Xinlei Chen, Yanghao Li","submitted_at":"2021-11-22T18:59:15Z","abstract_excerpt":"Object detection is a central downstream task used to test if pre-trained network parameters confer benefits, such as improved accuracy or training speed. The complexity of object detection methods can make this benchmarking non-trivial when new architectures, such as Vision Transformer (ViT) models, arrive. These difficulties (e.g., architectural incompatibility, slow training, high memory consumption, unknown training formulae, etc.) have prevented recent studies from benchmarking detection transfer learning with standard ViT models. In this paper, we present training techniques that overcom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.11429","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-11-22T18:59:15Z","cross_cats_sorted":[],"title_canon_sha256":"da5078aedb08ea4edc80b32a699f0b2440320cffdde3b8a841b255f2c8d26c15","abstract_canon_sha256":"39735951b3f862a0a4a37a361dbf94752297689898cd77102a63bbcc47ff2246"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:34:03.523396Z","signature_b64":"JP7XGYo7NiiToAOVBGbx2exOr1ye9S6zT0/2CmMPWqZV0RY4Z5hNokTDqNOe9FCa3h3+uj1DIPmnQ2PuGp0zBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b845eb882551b7dd107d0e68fa7e9123f38fc4a23babdf6199ba370f780d1c8","last_reissued_at":"2026-07-05T03:34:03.522923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:34:03.522923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Detection Transfer Learning with Vision Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kaiming He, Piotr Dollar, Ross Girshick, Saining Xie, Xinlei Chen, Yanghao Li","submitted_at":"2021-11-22T18:59:15Z","abstract_excerpt":"Object detection is a central downstream task used to test if pre-trained network parameters confer benefits, such as improved accuracy or training speed. The complexity of object detection methods can make this benchmarking non-trivial when new architectures, such as Vision Transformer (ViT) models, arrive. These difficulties (e.g., architectural incompatibility, slow training, high memory consumption, unknown training formulae, etc.) have prevented recent studies from benchmarking detection transfer learning with standard ViT models. In this paper, we present training techniques that overcom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.11429","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.11429/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.11429","created_at":"2026-07-05T03:34:03.522973+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.11429v1","created_at":"2026-07-05T03:34:03.522973+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.11429","created_at":"2026-07-05T03:34:03.522973+00:00"},{"alias_kind":"pith_short_12","alias_value":"FOCF5OECKUNX","created_at":"2026-07-05T03:34:03.522973+00:00"},{"alias_kind":"pith_short_16","alias_value":"FOCF5OECKUNX3UIH","created_at":"2026-07-05T03:34:03.522973+00:00"},{"alias_kind":"pith_short_8","alias_value":"FOCF5OEC","created_at":"2026-07-05T03:34:03.522973+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2407.17491","citing_title":"Robust Adaptation of Foundation Models with Black-Box Visual Prompting","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2410.07442","citing_title":"Self-Supervised Learning for Real-World Object Detection: a Survey","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2302.05543","citing_title":"Adding Conditional Control to Text-to-Image Diffusion Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14632","citing_title":"High-Speed Full-Color HDR Imaging via Unwrapping Modulo-Encoded Spike Streams","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI","json":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI.json","graph_json":"https://pith.science/api/pith-number/FOCF5OECKUNX3UIH2DTI7J7JCI/graph.json","events_json":"https://pith.science/api/pith-number/FOCF5OECKUNX3UIH2DTI7J7JCI/events.json","paper":"https://pith.science/paper/FOCF5OEC"},"agent_actions":{"view_html":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI","download_json":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI.json","view_paper":"https://pith.science/paper/FOCF5OEC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.11429&json=true","fetch_graph":"https://pith.science/api/pith-number/FOCF5OECKUNX3UIH2DTI7J7JCI/graph.json","fetch_events":"https://pith.science/api/pith-number/FOCF5OECKUNX3UIH2DTI7J7JCI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI/action/storage_attestation","attest_author":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI/action/author_attestation","sign_citation":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI/action/citation_signature","submit_replication":"https://pith.science/pith/FOCF5OECKUNX3UIH2DTI7J7JCI/action/replication_record"}},"created_at":"2026-07-05T03:34:03.522973+00:00","updated_at":"2026-07-05T03:34:03.522973+00:00"}