{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XCOHT6ZHR3CR4JQY6QSTPCXQDJ","short_pith_number":"pith:XCOHT6ZH","schema_version":"1.0","canonical_sha256":"b89c79fb278ec51e2618f425378af01a46182870fad8e65248022cd4a8d481e4","source":{"kind":"arxiv","id":"2109.14483","version":1},"attestation_state":"computed","paper":{"title":"CCTrans: Simplifying and Improving Crowd Counting with Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongpeng Wang, Xiangxiang Chu, Ye Tian","submitted_at":"2021-09-29T15:13:10Z","abstract_excerpt":"Most recent methods used for crowd counting are based on the convolutional neural network (CNN), which has a strong ability to extract local features. But CNN inherently fails in modeling the global context due to the limited receptive fields. However, the transformer can model the global context easily. In this paper, we propose a simple approach called CCTrans to simplify the design pipeline. Specifically, we utilize a pyramid vision transformer backbone to capture the global crowd information, a pyramid feature aggregation (PFA) model to combine low-level and high-level features, an efficie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.14483","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-09-29T15:13:10Z","cross_cats_sorted":[],"title_canon_sha256":"f1692543a22e5594412e9374336afea5df681be47aa6c46ba0ef43f4084dae68","abstract_canon_sha256":"3e4afccd8829d53c38e2d1186cecd4e3724f4ef24a9c03fa985451d91e468492"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:18:35.594495Z","signature_b64":"yINpnEUr21HCO6t8V4EHabbdqFtanLmxcVK37z/KZhELcMaJ2SJWKaxWMsGMFAdTsokqGPQWJyxYcA7QqSVOAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b89c79fb278ec51e2618f425378af01a46182870fad8e65248022cd4a8d481e4","last_reissued_at":"2026-07-05T03:18:35.593926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:18:35.593926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CCTrans: Simplifying and Improving Crowd Counting with Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongpeng Wang, Xiangxiang Chu, Ye Tian","submitted_at":"2021-09-29T15:13:10Z","abstract_excerpt":"Most recent methods used for crowd counting are based on the convolutional neural network (CNN), which has a strong ability to extract local features. But CNN inherently fails in modeling the global context due to the limited receptive fields. However, the transformer can model the global context easily. In this paper, we propose a simple approach called CCTrans to simplify the design pipeline. Specifically, we utilize a pyramid vision transformer backbone to capture the global crowd information, a pyramid feature aggregation (PFA) model to combine low-level and high-level features, an efficie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.14483","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.14483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.14483","created_at":"2026-07-05T03:18:35.594006+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.14483v1","created_at":"2026-07-05T03:18:35.594006+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.14483","created_at":"2026-07-05T03:18:35.594006+00:00"},{"alias_kind":"pith_short_12","alias_value":"XCOHT6ZHR3CR","created_at":"2026-07-05T03:18:35.594006+00:00"},{"alias_kind":"pith_short_16","alias_value":"XCOHT6ZHR3CR4JQY","created_at":"2026-07-05T03:18:35.594006+00:00"},{"alias_kind":"pith_short_8","alias_value":"XCOHT6ZH","created_at":"2026-07-05T03:18:35.594006+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30846","citing_title":"Count Anything","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15325","citing_title":"SoftHGNN: Soft Hypergraph Neural Networks for General Visual Recognition","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ","json":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ.json","graph_json":"https://pith.science/api/pith-number/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/graph.json","events_json":"https://pith.science/api/pith-number/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/events.json","paper":"https://pith.science/paper/XCOHT6ZH"},"agent_actions":{"view_html":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ","download_json":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ.json","view_paper":"https://pith.science/paper/XCOHT6ZH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.14483&json=true","fetch_graph":"https://pith.science/api/pith-number/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/action/storage_attestation","attest_author":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/action/author_attestation","sign_citation":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/action/citation_signature","submit_replication":"https://pith.science/pith/XCOHT6ZHR3CR4JQY6QSTPCXQDJ/action/replication_record"}},"created_at":"2026-07-05T03:18:35.594006+00:00","updated_at":"2026-07-05T03:18:35.594006+00:00"}