{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:N6TIL3E2SURIIGX2EXZ24B7H4A","short_pith_number":"pith:N6TIL3E2","schema_version":"1.0","canonical_sha256":"6fa685ec9a9522841afa25f3ae07e7e019e9e8f886fd25250059ec2c880a595a","source":{"kind":"arxiv","id":"2106.06195","version":1},"attestation_state":"computed","paper":{"title":"MlTr: Multi-label Classification with Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dong Shen, Fan Yang, Hezheng Lin, Honglin Liu, Nian Shi, Xiangyu Wu, Xing Cheng, Zhongyuan Wang","submitted_at":"2021-06-11T06:53:09Z","abstract_excerpt":"The task of multi-label image classification is to recognize all the object labels presented in an image. Though advancing for years, small objects, similar objects and objects with high conditional probability are still the main bottlenecks of previous convolutional neural network(CNN) based models, limited by convolutional kernels' representational capacity. Recent vision transformer networks utilize the self-attention mechanism to extract the feature of pixel granularity, which expresses richer local semantic information, while is insufficient for mining global spatial dependence. In this p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.06195","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-11T06:53:09Z","cross_cats_sorted":[],"title_canon_sha256":"7eca7dad740eb423332200a17ce76e8e1ffe1e9ae99ef369f7da491e4e39abda","abstract_canon_sha256":"8ff464c67ab12d3f22d299a4af3fd93929df17c266af963a56f28a85a35ae40b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:27.266707Z","signature_b64":"PAy1Kf8gGgwLUKD6Bdb5/mCsUPyuz7JcRsM8fqxLWjjb8XUuq+voRbx5vKudo191Zzguen2K4k2UAJ1rx1fSCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6fa685ec9a9522841afa25f3ae07e7e019e9e8f886fd25250059ec2c880a595a","last_reissued_at":"2026-07-05T02:48:27.266295Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:27.266295Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MlTr: Multi-label Classification with Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dong Shen, Fan Yang, Hezheng Lin, Honglin Liu, Nian Shi, Xiangyu Wu, Xing Cheng, Zhongyuan Wang","submitted_at":"2021-06-11T06:53:09Z","abstract_excerpt":"The task of multi-label image classification is to recognize all the object labels presented in an image. Though advancing for years, small objects, similar objects and objects with high conditional probability are still the main bottlenecks of previous convolutional neural network(CNN) based models, limited by convolutional kernels' representational capacity. Recent vision transformer networks utilize the self-attention mechanism to extract the feature of pixel granularity, which expresses richer local semantic information, while is insufficient for mining global spatial dependence. In this p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06195","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06195/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.06195","created_at":"2026-07-05T02:48:27.266367+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.06195v1","created_at":"2026-07-05T02:48:27.266367+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06195","created_at":"2026-07-05T02:48:27.266367+00:00"},{"alias_kind":"pith_short_12","alias_value":"N6TIL3E2SURI","created_at":"2026-07-05T02:48:27.266367+00:00"},{"alias_kind":"pith_short_16","alias_value":"N6TIL3E2SURIIGX2","created_at":"2026-07-05T02:48:27.266367+00:00"},{"alias_kind":"pith_short_8","alias_value":"N6TIL3E2","created_at":"2026-07-05T02:48:27.266367+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.14014","citing_title":"EGFormer: Towards Efficient and Generalizable Multimodal Semantic Segmentation","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A","json":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A.json","graph_json":"https://pith.science/api/pith-number/N6TIL3E2SURIIGX2EXZ24B7H4A/graph.json","events_json":"https://pith.science/api/pith-number/N6TIL3E2SURIIGX2EXZ24B7H4A/events.json","paper":"https://pith.science/paper/N6TIL3E2"},"agent_actions":{"view_html":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A","download_json":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A.json","view_paper":"https://pith.science/paper/N6TIL3E2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.06195&json=true","fetch_graph":"https://pith.science/api/pith-number/N6TIL3E2SURIIGX2EXZ24B7H4A/graph.json","fetch_events":"https://pith.science/api/pith-number/N6TIL3E2SURIIGX2EXZ24B7H4A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A/action/storage_attestation","attest_author":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A/action/author_attestation","sign_citation":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A/action/citation_signature","submit_replication":"https://pith.science/pith/N6TIL3E2SURIIGX2EXZ24B7H4A/action/replication_record"}},"created_at":"2026-07-05T02:48:27.266367+00:00","updated_at":"2026-07-05T02:48:27.266367+00:00"}