{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IB3QJV5X3G7FIVFPDZUMEOIWVY","short_pith_number":"pith:IB3QJV5X","schema_version":"1.0","canonical_sha256":"407704d7b7d9be5454af1e68c23916ae2df97bac681e113e8051924f81076f4f","source":{"kind":"arxiv","id":"2204.12484","version":3},"attestation_state":"computed","paper":{"title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Jing Zhang, Qiming Zhang, Yufei Xu","submitted_at":"2022-04-26T17:55:04Z","abstract_excerpt":"Although no specific domain knowledge is considered in the design, plain vision transformers have shown excellent performance in visual recognition tasks. However, little effort has been made to reveal the potential of such simple structures for pose estimation tasks. In this paper, we show the surprisingly good capabilities of plain vision transformers for pose estimation from various aspects, namely simplicity in model structure, scalability in model size, flexibility in training paradigm, and transferability of knowledge between models, through a simple baseline model called ViTPose. Specif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.12484","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-26T17:55:04Z","cross_cats_sorted":[],"title_canon_sha256":"781e4be59d5764577db993147fb13029b77c350f00096d02e3f780f403ec2250","abstract_canon_sha256":"00519464a1ea383de05a0121bdd3b70cfd7ce938ec2c3c47518d43e6ba4b5ee7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:11.215660Z","signature_b64":"8bdGFuXKuHKiDwpPBPpVJszCmDyoo5EWEaIfcNQIsnlRkBFYo7iCNJiAF3Eh14wcPwi+MEIsUXOL04/4pIThBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"407704d7b7d9be5454af1e68c23916ae2df97bac681e113e8051924f81076f4f","last_reissued_at":"2026-07-05T05:06:11.215174Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:11.215174Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Jing Zhang, Qiming Zhang, Yufei Xu","submitted_at":"2022-04-26T17:55:04Z","abstract_excerpt":"Although no specific domain knowledge is considered in the design, plain vision transformers have shown excellent performance in visual recognition tasks. However, little effort has been made to reveal the potential of such simple structures for pose estimation tasks. In this paper, we show the surprisingly good capabilities of plain vision transformers for pose estimation from various aspects, namely simplicity in model structure, scalability in model size, flexibility in training paradigm, and transferability of knowledge between models, through a simple baseline model called ViTPose. Specif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.12484","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.12484/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.12484","created_at":"2026-07-05T05:06:11.215223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.12484v3","created_at":"2026-07-05T05:06:11.215223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.12484","created_at":"2026-07-05T05:06:11.215223+00:00"},{"alias_kind":"pith_short_12","alias_value":"IB3QJV5X3G7F","created_at":"2026-07-05T05:06:11.215223+00:00"},{"alias_kind":"pith_short_16","alias_value":"IB3QJV5X3G7FIVFP","created_at":"2026-07-05T05:06:11.215223+00:00"},{"alias_kind":"pith_short_8","alias_value":"IB3QJV5X","created_at":"2026-07-05T05:06:11.215223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26215","citing_title":"TaskNPoint: How to Teach Your Humanoid to Hit a Backhand in Minutes","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28237","citing_title":"Unleashing Infinite Motion: Scaling Expressive Quadrupedal Motion via Generative Video Priors","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2312.06409","citing_title":"LiCamPose: Combining Multi-View LiDAR and RGB Cameras for Robust Single-timestamp 3D Human Pose Estimation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03299","citing_title":"MoViD: View-Invariant 3D Human Pose Estimation via Motion-View Disentanglement","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY","json":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY.json","graph_json":"https://pith.science/api/pith-number/IB3QJV5X3G7FIVFPDZUMEOIWVY/graph.json","events_json":"https://pith.science/api/pith-number/IB3QJV5X3G7FIVFPDZUMEOIWVY/events.json","paper":"https://pith.science/paper/IB3QJV5X"},"agent_actions":{"view_html":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY","download_json":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY.json","view_paper":"https://pith.science/paper/IB3QJV5X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.12484&json=true","fetch_graph":"https://pith.science/api/pith-number/IB3QJV5X3G7FIVFPDZUMEOIWVY/graph.json","fetch_events":"https://pith.science/api/pith-number/IB3QJV5X3G7FIVFPDZUMEOIWVY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY/action/storage_attestation","attest_author":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY/action/author_attestation","sign_citation":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY/action/citation_signature","submit_replication":"https://pith.science/pith/IB3QJV5X3G7FIVFPDZUMEOIWVY/action/replication_record"}},"created_at":"2026-07-05T05:06:11.215223+00:00","updated_at":"2026-07-05T05:06:11.215223+00:00"}