{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SLNTZYPNZSNRRNRAWV4DCOCFXL","short_pith_number":"pith:SLNTZYPN","schema_version":"1.0","canonical_sha256":"92db3ce1edcc9b18b620b578313845bafe800f7c2673edd87e3b9f66b92c3e8e","source":{"kind":"arxiv","id":"2102.01066","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Large-Vocabulary Object Detectors: The Devil is in the Details","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Achal Dave, Alexander Kirillov, Deva Ramanan, Piotr Doll\\'ar, Ross Girshick","submitted_at":"2021-02-01T18:56:02Z","abstract_excerpt":"By design, average precision (AP) for object detection aims to treat all classes independently: AP is computed independently per category and averaged. On one hand, this is desirable as it treats all classes equally. On the other hand, it ignores cross-category confidence calibration, a key property in real-world use cases. Unfortunately, under important conditions (i.e., large vocabulary, high instance counts) the default implementation of AP is neither category independent, nor does it directly reward properly calibrated detectors. In fact, we show that on LVIS the default implementation pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.01066","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-02-01T18:56:02Z","cross_cats_sorted":[],"title_canon_sha256":"6893b1f48391d07c174701537d1b5c176601c579c796d6bb8d1954083a7af300","abstract_canon_sha256":"dc62e572c3c6c87b11c61d67f92bb0f59c8402e5e0ed9a43053486b252c3bd3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:10.761652Z","signature_b64":"yXqjjaaPjlaH3STRByzZxVe5RoNSFReV0AjFCIXpsb0mWuEOlsXoaN597hpWZjpl9B5B65zTFcCtlWQeywI3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92db3ce1edcc9b18b620b578313845bafe800f7c2673edd87e3b9f66b92c3e8e","last_reissued_at":"2026-07-05T04:05:10.761230Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:10.761230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Large-Vocabulary Object Detectors: The Devil is in the Details","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Achal Dave, Alexander Kirillov, Deva Ramanan, Piotr Doll\\'ar, Ross Girshick","submitted_at":"2021-02-01T18:56:02Z","abstract_excerpt":"By design, average precision (AP) for object detection aims to treat all classes independently: AP is computed independently per category and averaged. On one hand, this is desirable as it treats all classes equally. On the other hand, it ignores cross-category confidence calibration, a key property in real-world use cases. Unfortunately, under important conditions (i.e., large vocabulary, high instance counts) the default implementation of AP is neither category independent, nor does it directly reward properly calibrated detectors. In fact, we show that on LVIS the default implementation pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.01066","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.01066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.01066","created_at":"2026-07-05T04:05:10.761292+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.01066v2","created_at":"2026-07-05T04:05:10.761292+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.01066","created_at":"2026-07-05T04:05:10.761292+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLNTZYPNZSNR","created_at":"2026-07-05T04:05:10.761292+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLNTZYPNZSNRRNRA","created_at":"2026-07-05T04:05:10.761292+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLNTZYPN","created_at":"2026-07-05T04:05:10.761292+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11546","citing_title":"VL-DINO: Leveraging CLIP Vision-Language Knowledge for Open-Vocabulary Object Detectio","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03456","citing_title":"VL-SAM-v3: Memory-Guided Visual Priors for Open-World Object Detection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03456","citing_title":"VL-SAM-v3: Memory-Guided Visual Priors for Open-World Object Detection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14684","citing_title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03456","citing_title":"VL-SAM-v3: Memory-Guided Visual Priors for Open-World Object Detection","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL","json":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL.json","graph_json":"https://pith.science/api/pith-number/SLNTZYPNZSNRRNRAWV4DCOCFXL/graph.json","events_json":"https://pith.science/api/pith-number/SLNTZYPNZSNRRNRAWV4DCOCFXL/events.json","paper":"https://pith.science/paper/SLNTZYPN"},"agent_actions":{"view_html":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL","download_json":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL.json","view_paper":"https://pith.science/paper/SLNTZYPN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.01066&json=true","fetch_graph":"https://pith.science/api/pith-number/SLNTZYPNZSNRRNRAWV4DCOCFXL/graph.json","fetch_events":"https://pith.science/api/pith-number/SLNTZYPNZSNRRNRAWV4DCOCFXL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL/action/storage_attestation","attest_author":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL/action/author_attestation","sign_citation":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL/action/citation_signature","submit_replication":"https://pith.science/pith/SLNTZYPNZSNRRNRAWV4DCOCFXL/action/replication_record"}},"created_at":"2026-07-05T04:05:10.761292+00:00","updated_at":"2026-07-05T04:05:10.761292+00:00"}