{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:GLLDPC5KDBHM46GIGXPRZ72U6V","short_pith_number":"pith:GLLDPC5K","schema_version":"1.0","canonical_sha256":"32d6378baa184ece78c835df1cff54f559d4eaf2e88beeb1ab14ab42d81d0d72","source":{"kind":"arxiv","id":"1907.06987","version":2},"attestation_state":"computed","paper":{"title":"A Short Note on the Kinetics-700 Human Action Dataset","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrew Zisserman, Chloe Hillier, Eric Noland, Joao Carreira","submitted_at":"2019-07-15T12:58:21Z","abstract_excerpt":"We describe an extension of the DeepMind Kinetics human action dataset from 600 classes to 700 classes, where for each class there are at least 600 video clips from different YouTube videos. This paper details the changes introduced for this new release of the dataset, and includes a comprehensive set of statistics as well as baseline results using the I3D neural network architecture."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1907.06987","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2019-07-15T12:58:21Z","cross_cats_sorted":[],"title_canon_sha256":"27238b6550053f3e5e82c6e43652c4cdb9bb7e11219f328ef0f942b1be6ca838","abstract_canon_sha256":"222694ac5df5531223e854cd3cfa5a81d0118a4c1ce3040b8fc78891e08da345"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:21.272113Z","signature_b64":"H8sZjLjMaUEuVETsZyXBpVIK1neVodqnoFxameZ8zZFu3uYTdkxZQpDiev6HGB7ogKEzbekoEXkW2/tkHKuFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32d6378baa184ece78c835df1cff54f559d4eaf2e88beeb1ab14ab42d81d0d72","last_reissued_at":"2026-07-05T05:07:21.271632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:21.271632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Short Note on the Kinetics-700 Human Action Dataset","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrew Zisserman, Chloe Hillier, Eric Noland, Joao Carreira","submitted_at":"2019-07-15T12:58:21Z","abstract_excerpt":"We describe an extension of the DeepMind Kinetics human action dataset from 600 classes to 700 classes, where for each class there are at least 600 video clips from different YouTube videos. This paper details the changes introduced for this new release of the dataset, and includes a comprehensive set of statistics as well as baseline results using the I3D neural network architecture."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.06987","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.06987/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1907.06987","created_at":"2026-07-05T05:07:21.271692+00:00"},{"alias_kind":"arxiv_version","alias_value":"1907.06987v2","created_at":"2026-07-05T05:07:21.271692+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.06987","created_at":"2026-07-05T05:07:21.271692+00:00"},{"alias_kind":"pith_short_12","alias_value":"GLLDPC5KDBHM","created_at":"2026-07-05T05:07:21.271692+00:00"},{"alias_kind":"pith_short_16","alias_value":"GLLDPC5KDBHM46GI","created_at":"2026-07-05T05:07:21.271692+00:00"},{"alias_kind":"pith_short_8","alias_value":"GLLDPC5K","created_at":"2026-07-05T05:07:21.271692+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23118","citing_title":"LUMINA-26: Low-Light Understanding for Modeling and Interpreting Night-time Actions","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20734","citing_title":"Robust Zero-Shot Generalization for Open-Vocabulary Action Recognition via Task Arithmetic","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00054","citing_title":"From Human Videos to Robot Manipulation: A Survey on Scalable Vision-Language-Action Learning with Human-Centric Data","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23045","citing_title":"The TIME Machine: On The Power of Motion for Efficient Perception","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2309.17257","citing_title":"A Survey on Deep Learning Techniques for Action Anticipation","ref_index":198,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22819","citing_title":"Cambrian-P: Pose-Grounded Video Understanding","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04590","citing_title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2212.03191","citing_title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13684","citing_title":"Recurrent Video Masked Autoencoders","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2205.01917","citing_title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2303.15389","citing_title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06158","citing_title":"GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20157","citing_title":"HumanScore: Benchmarking Human Motions in Generated Videos","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V","json":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V.json","graph_json":"https://pith.science/api/pith-number/GLLDPC5KDBHM46GIGXPRZ72U6V/graph.json","events_json":"https://pith.science/api/pith-number/GLLDPC5KDBHM46GIGXPRZ72U6V/events.json","paper":"https://pith.science/paper/GLLDPC5K"},"agent_actions":{"view_html":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V","download_json":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V.json","view_paper":"https://pith.science/paper/GLLDPC5K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1907.06987&json=true","fetch_graph":"https://pith.science/api/pith-number/GLLDPC5KDBHM46GIGXPRZ72U6V/graph.json","fetch_events":"https://pith.science/api/pith-number/GLLDPC5KDBHM46GIGXPRZ72U6V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V/action/storage_attestation","attest_author":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V/action/author_attestation","sign_citation":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V/action/citation_signature","submit_replication":"https://pith.science/pith/GLLDPC5KDBHM46GIGXPRZ72U6V/action/replication_record"}},"created_at":"2026-07-05T05:07:21.271692+00:00","updated_at":"2026-07-05T05:07:21.271692+00:00"}