{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JTHVQ2O76Z7MXNHBCCDW7LOL4U","short_pith_number":"pith:JTHVQ2O7","schema_version":"1.0","canonical_sha256":"4ccf5869dff67ecbb4e110876fadcbe504f4a8c7a79e92c7c99343a594426665","source":{"kind":"arxiv","id":"2401.10831","version":3},"attestation_state":"computed","paper":{"title":"Understanding Video Transformers via Universal Concept Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Achal Dave, Adrien Gaidon, Konstantinos G. Derpanis, Matthew Kowal, Pavel Tokmakov, Rares Ambrus","submitted_at":"2024-01-19T17:27:21Z","abstract_excerpt":"This paper studies the problem of concept-based interpretability of transformer representations for videos. Concretely, we seek to explain the decision-making process of video transformers based on high-level, spatiotemporal concepts that are automatically discovered. Prior research on concept-based interpretability has concentrated solely on image-level tasks. Comparatively, video models deal with the added temporal dimension, increasing complexity and posing challenges in identifying dynamic concepts over time. In this work, we systematically address these challenges by introducing the first"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.10831","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-19T17:27:21Z","cross_cats_sorted":["cs.AI","cs.LG","cs.RO"],"title_canon_sha256":"5229c55c15cdab493a289f698ae3e4ba708a6766d11561c9fd2c4ae776503b1b","abstract_canon_sha256":"7999bef99c863daf0f94db82a15c9eb6bc2fb892987d0f170c8913e115a9a2a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:27.502947Z","signature_b64":"mSTI0hDaelxkWSkFQd/RzfGndt7yycxPtZkmy7Uo9yCRxGy2DS8ENGoGV/pDxBun8FNAndKIDon7LFYaKPtjCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ccf5869dff67ecbb4e110876fadcbe504f4a8c7a79e92c7c99343a594426665","last_reissued_at":"2026-07-05T08:06:27.502461Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:27.502461Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Video Transformers via Universal Concept Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Achal Dave, Adrien Gaidon, Konstantinos G. Derpanis, Matthew Kowal, Pavel Tokmakov, Rares Ambrus","submitted_at":"2024-01-19T17:27:21Z","abstract_excerpt":"This paper studies the problem of concept-based interpretability of transformer representations for videos. Concretely, we seek to explain the decision-making process of video transformers based on high-level, spatiotemporal concepts that are automatically discovered. Prior research on concept-based interpretability has concentrated solely on image-level tasks. Comparatively, video models deal with the added temporal dimension, increasing complexity and posing challenges in identifying dynamic concepts over time. In this work, we systematically address these challenges by introducing the first"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.10831","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.10831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.10831","created_at":"2026-07-05T08:06:27.502516+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.10831v3","created_at":"2026-07-05T08:06:27.502516+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10831","created_at":"2026-07-05T08:06:27.502516+00:00"},{"alias_kind":"pith_short_12","alias_value":"JTHVQ2O76Z7M","created_at":"2026-07-05T08:06:27.502516+00:00"},{"alias_kind":"pith_short_16","alias_value":"JTHVQ2O76Z7MXNHB","created_at":"2026-07-05T08:06:27.502516+00:00"},{"alias_kind":"pith_short_8","alias_value":"JTHVQ2O7","created_at":"2026-07-05T08:06:27.502516+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24690","citing_title":"Learning reusable concepts across different egocentric video understanding tasks","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U","json":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U.json","graph_json":"https://pith.science/api/pith-number/JTHVQ2O76Z7MXNHBCCDW7LOL4U/graph.json","events_json":"https://pith.science/api/pith-number/JTHVQ2O76Z7MXNHBCCDW7LOL4U/events.json","paper":"https://pith.science/paper/JTHVQ2O7"},"agent_actions":{"view_html":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U","download_json":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U.json","view_paper":"https://pith.science/paper/JTHVQ2O7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.10831&json=true","fetch_graph":"https://pith.science/api/pith-number/JTHVQ2O76Z7MXNHBCCDW7LOL4U/graph.json","fetch_events":"https://pith.science/api/pith-number/JTHVQ2O76Z7MXNHBCCDW7LOL4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U/action/storage_attestation","attest_author":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U/action/author_attestation","sign_citation":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U/action/citation_signature","submit_replication":"https://pith.science/pith/JTHVQ2O76Z7MXNHBCCDW7LOL4U/action/replication_record"}},"created_at":"2026-07-05T08:06:27.502516+00:00","updated_at":"2026-07-05T08:06:27.502516+00:00"}