{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7BTCWP5L7SIMDBYKDUBS26QTTP","short_pith_number":"pith:7BTCWP5L","schema_version":"1.0","canonical_sha256":"f8662b3fabfc90c1870a1d032d7a139bee14e10ba5f297bd6b7652a5b1cdc7f0","source":{"kind":"arxiv","id":"2503.07860","version":1},"attestation_state":"computed","paper":{"title":"Video Action Differencing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alejandro Lozano, Anita Rau, James Burgess, Lisa Dunlap, Serena Yeung-Levy, Trevor Darrell, Xiaohan Wang, Yuhui Zhang","submitted_at":"2025-03-10T21:18:32Z","abstract_excerpt":"How do two individuals differ when performing the same action? In this work, we introduce Video Action Differencing (VidDiff), the novel task of identifying subtle differences between videos of the same action, which has many applications, such as coaching and skill learning. To enable development on this new task, we first create VidDiffBench, a benchmark dataset containing 549 video pairs, with human annotations of 4,469 fine-grained action differences and 2,075 localization timestamps indicating where these differences occur. Our experiments demonstrate that VidDiffBench poses a significant"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.07860","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-10T21:18:32Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8b7400d101af4cdeab0ef217ef79b1e9a7d4020f2b2035b7d5e6fe52d9684e03","abstract_canon_sha256":"cef71461ae884b326d14e5409d41bb35c526ca215c0164b62bf7da94c8f52c8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:45.426704Z","signature_b64":"lebuLMsB7Erl8IFmm8jauWIN7dYmQEVJd5gko7MjYB1AsWPrYDrtItTwfswyuW1byeFrxZrIcyl48lHZNuhFBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8662b3fabfc90c1870a1d032d7a139bee14e10ba5f297bd6b7652a5b1cdc7f0","last_reissued_at":"2026-07-05T10:28:45.426211Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:45.426211Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Action Differencing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alejandro Lozano, Anita Rau, James Burgess, Lisa Dunlap, Serena Yeung-Levy, Trevor Darrell, Xiaohan Wang, Yuhui Zhang","submitted_at":"2025-03-10T21:18:32Z","abstract_excerpt":"How do two individuals differ when performing the same action? In this work, we introduce Video Action Differencing (VidDiff), the novel task of identifying subtle differences between videos of the same action, which has many applications, such as coaching and skill learning. To enable development on this new task, we first create VidDiffBench, a benchmark dataset containing 549 video pairs, with human annotations of 4,469 fine-grained action differences and 2,075 localization timestamps indicating where these differences occur. Our experiments demonstrate that VidDiffBench poses a significant"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.07860","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.07860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.07860","created_at":"2026-07-05T10:28:45.426265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.07860v1","created_at":"2026-07-05T10:28:45.426265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.07860","created_at":"2026-07-05T10:28:45.426265+00:00"},{"alias_kind":"pith_short_12","alias_value":"7BTCWP5L7SIM","created_at":"2026-07-05T10:28:45.426265+00:00"},{"alias_kind":"pith_short_16","alias_value":"7BTCWP5L7SIMDBYK","created_at":"2026-07-05T10:28:45.426265+00:00"},{"alias_kind":"pith_short_8","alias_value":"7BTCWP5L","created_at":"2026-07-05T10:28:45.426265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23061","citing_title":"MotionHalluc: Diagnosing Kinematic Hallucinations in Fine-Grained Motion Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02482","citing_title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02482","citing_title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19629","citing_title":"SkillSight: Efficient First-Person Skill Assessment with Gaze","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP","json":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP.json","graph_json":"https://pith.science/api/pith-number/7BTCWP5L7SIMDBYKDUBS26QTTP/graph.json","events_json":"https://pith.science/api/pith-number/7BTCWP5L7SIMDBYKDUBS26QTTP/events.json","paper":"https://pith.science/paper/7BTCWP5L"},"agent_actions":{"view_html":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP","download_json":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP.json","view_paper":"https://pith.science/paper/7BTCWP5L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.07860&json=true","fetch_graph":"https://pith.science/api/pith-number/7BTCWP5L7SIMDBYKDUBS26QTTP/graph.json","fetch_events":"https://pith.science/api/pith-number/7BTCWP5L7SIMDBYKDUBS26QTTP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP/action/storage_attestation","attest_author":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP/action/author_attestation","sign_citation":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP/action/citation_signature","submit_replication":"https://pith.science/pith/7BTCWP5L7SIMDBYKDUBS26QTTP/action/replication_record"}},"created_at":"2026-07-05T10:28:45.426265+00:00","updated_at":"2026-07-05T10:28:45.426265+00:00"}