{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:A2NSEWPQ6CFVWCNKFCLRSX2I66","short_pith_number":"pith:A2NSEWPQ","schema_version":"1.0","canonical_sha256":"069b2259f0f08b5b09aa2897195f48f7a8715179abd3f7ae72adce4e046ce52f","source":{"kind":"arxiv","id":"2004.03548","version":2},"attestation_state":"computed","paper":{"title":"Temporal Pyramid Network for Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Dai, Bolei Zhou, Ceyuan Yang, Jianping Shi, Yinghao Xu","submitted_at":"2020-04-07T17:17:23Z","abstract_excerpt":"Visual tempo characterizes the dynamics and the temporal scale of an action. Modeling such visual tempos of different actions facilitates their recognition. Previous works often capture the visual tempo through sampling raw videos at multiple rates and constructing an input-level frame pyramid, which usually requires a costly multi-branch network to handle. In this work we propose a generic Temporal Pyramid Network (TPN) at the feature-level, which can be flexibly integrated into 2D or 3D backbone networks in a plug-and-play manner. Two essential components of TPN, the source of features and t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.03548","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-04-07T17:17:23Z","cross_cats_sorted":[],"title_canon_sha256":"262e1c6f397674534262280f78382a1b4dc75f598481a9f1c836f73136c23cee","abstract_canon_sha256":"54fd54dd9dffb1f43e52de6cf319b48ccfca3e10800db598dd10ba2cf4d1c051"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:10:14.758179Z","signature_b64":"i5B8bdI4d/sK15nbxj6xo1bLfjOSfrVXinOJ0qPAi4Aj3paD+MWRCVJ7pLantXlstcp9uZLXcfA6JO1McAmCCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"069b2259f0f08b5b09aa2897195f48f7a8715179abd3f7ae72adce4e046ce52f","last_reissued_at":"2026-07-05T01:10:14.757599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:10:14.757599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Temporal Pyramid Network for Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Dai, Bolei Zhou, Ceyuan Yang, Jianping Shi, Yinghao Xu","submitted_at":"2020-04-07T17:17:23Z","abstract_excerpt":"Visual tempo characterizes the dynamics and the temporal scale of an action. Modeling such visual tempos of different actions facilitates their recognition. Previous works often capture the visual tempo through sampling raw videos at multiple rates and constructing an input-level frame pyramid, which usually requires a costly multi-branch network to handle. In this work we propose a generic Temporal Pyramid Network (TPN) at the feature-level, which can be flexibly integrated into 2D or 3D backbone networks in a plug-and-play manner. Two essential components of TPN, the source of features and t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.03548","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.03548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.03548","created_at":"2026-07-05T01:10:14.757661+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.03548v2","created_at":"2026-07-05T01:10:14.757661+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.03548","created_at":"2026-07-05T01:10:14.757661+00:00"},{"alias_kind":"pith_short_12","alias_value":"A2NSEWPQ6CFV","created_at":"2026-07-05T01:10:14.757661+00:00"},{"alias_kind":"pith_short_16","alias_value":"A2NSEWPQ6CFVWCNK","created_at":"2026-07-05T01:10:14.757661+00:00"},{"alias_kind":"pith_short_8","alias_value":"A2NSEWPQ","created_at":"2026-07-05T01:10:14.757661+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17133","citing_title":"CAM-VFD: Cross-Attention Multimodal Video Forgery Detection","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66","json":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66.json","graph_json":"https://pith.science/api/pith-number/A2NSEWPQ6CFVWCNKFCLRSX2I66/graph.json","events_json":"https://pith.science/api/pith-number/A2NSEWPQ6CFVWCNKFCLRSX2I66/events.json","paper":"https://pith.science/paper/A2NSEWPQ"},"agent_actions":{"view_html":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66","download_json":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66.json","view_paper":"https://pith.science/paper/A2NSEWPQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.03548&json=true","fetch_graph":"https://pith.science/api/pith-number/A2NSEWPQ6CFVWCNKFCLRSX2I66/graph.json","fetch_events":"https://pith.science/api/pith-number/A2NSEWPQ6CFVWCNKFCLRSX2I66/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66/action/storage_attestation","attest_author":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66/action/author_attestation","sign_citation":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66/action/citation_signature","submit_replication":"https://pith.science/pith/A2NSEWPQ6CFVWCNKFCLRSX2I66/action/replication_record"}},"created_at":"2026-07-05T01:10:14.757661+00:00","updated_at":"2026-07-05T01:10:14.757661+00:00"}