{"as_of":"2026-08-22T06:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ee3b3349babb83e1f20d5fc7afc242bf74c49508ef8be501d7a26c9a6173ec4c","coverage":[{"denominator":72,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":72,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T10:28:54.659185Z","state":"measured"},{"denominator":72,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":72,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.00085/citation-record","integrity":"/paper/2508.00085/integrity","json":"/paper/2508.00085/citation-record.json","paper":"/paper/2508.00085"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.461100Z","title":"Ez- clip: Efficient zeroshot video action recognition, 2024","venue":null,"work_id":"4605c10b-7682-41e9-945f-f5a35227f1ee","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.442094Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:f18a7f40295b1c54c2307623906d740c7aa4a69881660967d35eff0f0227be5f","observation_id":"4115693f-33fc-4219-8181-005da5fc9ce0","resolution":{"observed_at":"2026-08-06T10:28:55.464879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.450174Z","title":"T2l: Efficient zero-shot action recognition with temporal to- ken learning","venue":null,"work_id":"29e3a0d4-194c-4825-973e-150a2d3cf599","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.445834Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:6c3e6038c24fafa7c81f39a071795ad9bcb09bdc3f8be2162452ae6956831130","observation_id":"405fec8c-1b4a-42ea-b130-d70d1fa5dcc0","resolution":{"observed_at":"2026-08-06T10:28:55.454224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.439202Z","title":"Are visual- language models effective in action recognition? a compara- tive study, 2024","venue":null,"work_id":"f8425beb-3d06-4b07-8c60-3438c0ae6ee0","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.449338Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c21280fae21aaa9831c6f096f8681fef8d323099bab379cae0760c954def1ad9","observation_id":"dffcc542-6085-409e-b5ad-c3a9a9ee7130","resolution":{"observed_at":"2026-08-06T10:28:55.442867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.427107Z","title":"Understanding depth and height percep- tion in large visual-language models","venue":null,"work_id":"838a7233-7ab7-468a-ab72-a4c163ff0cbd","year":2025},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.452523Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:6eef0afc5a84977a978392877506bc2295e80c5c917c87f52e388e5a3c363884","observation_id":"555e39a9-946b-4ec2-bd48-630e798d63c3","resolution":{"observed_at":"2026-08-06T10:28:55.432532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.415055Z","title":"Hierarq: Task-aware hierarchical q-former for enhanced video understanding","venue":null,"work_id":"8f811eab-d475-4614-8d43-6a9fcf53023d","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.456042Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:6dc71bd78364b7ad472633bf432b4c0ff883f3cf959752ce9f1f430019a3a66a","observation_id":"842c855e-97d9-468c-abf2-518e5c1b5ca6","resolution":{"observed_at":"2026-08-06T10:28:55.418886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.405088Z","title":"Rethinking zero-shot video classi- fication: End-to-end training for realistic applications, 2020","venue":null,"work_id":"8c039582-7489-4b7f-8872-3a9d0c1c097a","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.459258Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:0c980ed5753223f787fa265eaa4f933e22b2aba96ca8696d9c839b5cbda0f207","observation_id":"26c3acdc-df14-4c74-8f83-c0dec426f1ef","resolution":{"observed_at":"2026-08-06T10:28:55.408820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.394189Z","title":"Temporal attentive alignment for large-scale video domain adaptation","venue":null,"work_id":"509e2734-303f-4a9c-bf2d-516d4e99c04f","year":2019},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.462416Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:5fac43dd033116dd00a2ae856da8edf70513c39d76cdf612c9f57d4da0fd5f87","observation_id":"40bef970-52e2-428d-9895-065ab544d01e","resolution":{"observed_at":"2026-08-06T10:28:55.397834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.383474Z","title":"Elaborative rehearsal for zero-shot action recognition, 2021","venue":null,"work_id":"caff7615-b193-4c5b-b990-676f15a64707","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.465332Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c37fc5a5694d92d962f8e873174fc3750ecc8a00263d0d8cb5bd8dd73a59c50a","observation_id":"a2c48975-54a3-4ac5-8a6f-265a4387d171","resolution":{"observed_at":"2026-08-06T10:28:55.387335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.371066Z","title":"Unsupervised and semi-supervised domain adaptation for action recognition from drones","venue":null,"work_id":"47269cec-cd3c-4116-bd74-e5fe2ebc98fd","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.468832Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:326305bee992a5ef2d220445f18e7ac4960297269ec0cdd93f7d91fdfd7d75da","observation_id":"b48f7bae-281d-483e-9475-52fee87def90","resolution":{"observed_at":"2026-08-06T10:28:55.376102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.360045Z","title":"Blender: A 3d modelling and rendering package","venue":null,"work_id":"e2ed7762-1a12-434b-a742-4456a918f8c6","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.471879Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ab5fd982401c4ed5d2371148d931a735ef31fe21148a28f29e8f2cbe92582284","observation_id":"8f057fac-3d8c-48de-95e5-9b246f5dd664","resolution":{"observed_at":"2026-08-06T10:28:55.363833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.349195Z","title":"Toyota smarthome: Real-world activities of daily living","venue":null,"work_id":"301e85a1-b72e-40c3-b067-fdd64a92d603","year":2019},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.474848Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:7aa8371cb33297906444dc73dd6fdb8e623f51fa48903527a8349051b39452e7","observation_id":"ddca8ede-8dbf-4066-8def-58ea508acc06","resolution":{"observed_at":"2026-08-06T10:28:55.352878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.337953Z","title":"Tinyvirat: Low-resolution video action recognition","venue":null,"work_id":"9093ef6b-b2b5-4027-9d23-3acb3aa0522e","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.477665Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:2a12fa1e8399b0632b07aa63a6f3c7d51cc7dfeeee71c69a48e160e7c8fa73b7","observation_id":"51ada693-b875-4acd-a6e1-7be1e5956307","resolution":{"observed_at":"2026-08-06T10:28:55.342081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.326695Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale, 2021","venue":null,"work_id":"c164ce32-d52b-439e-9ab5-5e94ecdc0e1e","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.480897Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:169e246fab9c2315e82ec36865d8367ca364acafe27ffef769ed06e0d5fb9e65","observation_id":"f0ff23fe-fe36-4e6f-a321-f3ce0d58b86d","resolution":{"observed_at":"2026-08-06T10:28:55.330466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.314738Z","title":"Video- capsulenet: a simplified network for action detection","venue":null,"work_id":"8b87d0ce-28f3-4ef9-b874-fe35d4fbeb17","year":2018},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.483836Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:25c7b2bb92ec0c04dc24ce7d113dddc4d35e4609a2e0b17e360ca311786850d0","observation_id":"553bceaa-117c-4b9f-aa92-70bc76cb6523","resolution":{"observed_at":"2026-08-06T10:28:55.318689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.303402Z","title":"Pyslowfast","venue":null,"work_id":"85bff2fa-5163-42d0-8d7f-ee6ff709ee7f","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.486818Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:01019d28e33e4fba36207d07a30c8eb3748aed18098bdd2cccd81b8c187c3fa0","observation_id":"b87ed6d3-4280-43d1-94aa-4343d13a4bfa","resolution":{"observed_at":"2026-08-06T10:28:55.306953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.292104Z","title":"Multiscale vision transformers, 2021","venue":null,"work_id":"37e9e66f-9765-4703-8d54-23abf289d689","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.489800Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c330348994aadc6d95b8b5ff9f1596402e6aacf11c399d5f60a92666b8a284b0","observation_id":"33793a46-2486-4553-a5c9-978cf381f416","resolution":{"observed_at":"2026-08-06T10:28:55.295643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.280480Z","title":"X3d: Expanding architectures for efficient video recognition, 2020","venue":null,"work_id":"2c603d49-68a8-456e-b398-521361acab0a","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.492568Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:5892f8754eace73da6bd6c8b5ded97a9854175cbe6ddfba3adec8f95250cc0c4","observation_id":"9e6c1341-cacb-46f8-b59b-8e298e2138f8","resolution":{"observed_at":"2026-08-06T10:28:55.285152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.269354Z","title":"Slowfast networks for video recognition, 2019","venue":null,"work_id":"7a691383-207d-4eef-8910-569d636caf75","year":2019},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.495516Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ed2d7f6fd7b37bd9e1450f9fde3cfa0adbfbad9baf71bd6cedaa03e22b24d0be","observation_id":"2aee4bb9-9796-48e8-9f10-9624da7d811d","resolution":{"observed_at":"2026-08-06T10:28:55.273232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.257305Z","title":"Telling stories for common sense zero-shot action recognition, 2024","venue":null,"work_id":"bc167c7f-c858-4c6b-aa08-73da11a98acf","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.498488Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:b0c33cae9204dd917938fc963e5131fcd1d195ea3c31c05acde5726dfc3a3147","observation_id":"c78e5c46-f59a-4d99-9ff2-a5241bd8123d","resolution":{"observed_at":"2026-08-06T10:28:55.262175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.245013Z","title":"A new split for evaluating true zero-shot action recognition, 2021","venue":null,"work_id":"22886708-3f94-4ce7-b6f2-d5bbdc984874","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.501360Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:cf38137fcaa54e6807f6b00f76e1a96268e0ab68997b4c36017b53d9b98b363a","observation_id":"18d9ac81-4234-4882-bb4f-714c43027814","resolution":{"observed_at":"2026-08-06T10:28:55.249222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.504909Z","title":"The ”something something” video database for learning and evaluating visual common sense,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.504909Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:44e882983c495a8478960d42e33220316b32d2c6e9bac9cda8da7a2c45c660ed","observation_id":"66a92dda-0919-4f64-b1c7-3f55e16c67c2","resolution":{"observed_at":"2026-08-06T10:28:54.504909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.227209Z","title":"Hierar- chical explanations for video action recognition, 2023","venue":null,"work_id":"c48900d0-de50-4b53-bd4b-59cfdabb11c1","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.508185Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:f26b3aa4bf50462223d3e241dfb898441c5e32be7069545c0f461a3773f4153c","observation_id":"9a60b9a3-745a-4636-b1d3-2804aa873f59","resolution":{"observed_at":"2026-08-06T10:28:55.230960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.215083Z","title":"Learn- ing spatio-temporal features with 3d residual networks for action recognition, 2017","venue":null,"work_id":"cb999bf1-07a6-4675-9d32-24dc87e98602","year":2017},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.511059Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ef61beadc12947ce9243a44fe274c5e072144514cd7f81e825491a888a957125","observation_id":"8ed410d2-c1fe-4455-97fd-c8d866d95da5","resolution":{"observed_at":"2026-08-06T10:28:55.218908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.204240Z","title":"Deep residual learning for image recognition, 2015","venue":null,"work_id":"137a9fd1-b1b4-4d7d-9e9b-d6adcd88a8cc","year":2015},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.514019Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:06e39beb17e8f522a0e2beeaff5e3a7e6bce311ab2be1391a04f39518f9fb7e9","observation_id":"2ba387f4-591a-4e58-9223-2267ee05f3c1","resolution":{"observed_at":"2026-08-06T10:28:55.208017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.192622Z","title":"Froster: Frozen clip is a strong teacher for open-vocabulary action recognition, 2024","venue":null,"work_id":"242af2ab-9045-48ba-a553-5d81a96f5c0e","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.517116Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:21f4e82f86c923b8d912b418a29feaf4fee5f8d9b0f34f31912d4ce0b29b697f","observation_id":"e38be4fe-896d-47d8-9618-c7d1c05f16d5","resolution":{"observed_at":"2026-08-06T10:28:55.197411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.181229Z","title":"The kinetics human action video dataset, 2017","venue":null,"work_id":"21fd228b-c78d-41be-8977-bc94e635287d","year":2017},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.520017Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:cb598e1ef53f3485e06698f4c012c41e1d3cf30e04944f7c91789b69be015dcc","observation_id":"eafeab26-b4f5-4b32-b2e7-583e9acd9a08","resolution":{"observed_at":"2026-08-06T10:28:55.185942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.170650Z","title":"Reformulating zero-shot action recognition for multi- label actions","venue":null,"work_id":"3d610d3d-aa9f-4e84-94ef-c1b098a8bd47","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.523289Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:cc74348538fe717e7e8252834f3a7e42dc3e6b5b21489cd754b6b2bdd2fbf199","observation_id":"75c6528c-4b98-47ce-93c0-57a4c8316807","resolution":{"observed_at":"2026-08-06T10:28:55.174693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.160117Z","title":"Learning Cross-Modal Contrastive Features for Video Do- main Adaptation","venue":null,"work_id":"d04dafad-1c50-4a9d-a841-6134a33aecb7","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.526412Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:f06f398738a0a1778fccaf42e85ef927a01f0f4b2dcdf1f9dd18e376d7fd8c76","observation_id":"ccfcab97-9d39-4f05-a3b2-fbf7886a5449","resolution":{"observed_at":"2026-08-06T10:28:55.163962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.149847Z","title":"Uniformer: Unified transformer for efficient spatiotemporal representation learning, 2022","venue":null,"work_id":"bdc9beab-c35e-4c55-9cbd-6b82c69ce871","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.529621Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:fc4370093ee4967a7f62563f2c6f5f45a47561d043f3636ff9a10e58bc5bbca4","observation_id":"708a4cc8-a1d7-4cc9-9875-8f6b1a58a02c","resolution":{"observed_at":"2026-08-06T10:28:55.153544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.138594Z","title":"Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer, 2022","venue":null,"work_id":"44be11cc-8367-4c2b-aeaa-3f3b01fe1c04","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.532725Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:fc8fcbc583992dc0bd65e8f88914298767c2887c4d8c0d9e12a4a31979b374d4","observation_id":"e9110c7b-fb49-4584-9ab5-42bd222e837e","resolution":{"observed_at":"2026-08-06T10:28:55.143098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.127364Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":"f2fdc261-4095-4056-9b19-c5e481271f2f","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.535858Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:1f220147319dfcf22d2e5163c4519d61e7ff4535d7c83c36cc7ad1ffbd7b813c","observation_id":"e4486dd0-6061-4174-a596-993a2f0a080b","resolution":{"observed_at":"2026-08-06T10:28:55.131467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.116214Z","title":"Mvitv2: Improved multiscale vision transformers for classification and detection, 2022","venue":null,"work_id":"47cfa32e-93ce-4f7c-9891-6c41d7489ffc","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.539031Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:3ddec410ddeee9c2a1d24c16b7302ac65db9dfcb972ba42e0c35b57f90a19c0e","observation_id":"d144fa42-07e1-4c0c-ac5a-30270d176b33","resolution":{"observed_at":"2026-08-06T10:28:55.120320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.105586Z","title":"Cross-modal representation learning for zero- shot action recognition, 2022","venue":null,"work_id":"94b31558-cc02-45b9-8191-5c54969e0eb2","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.542082Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:d7e961c4712517aac950f4598461b85dafc95da05f48e43337768ea519451cc7","observation_id":"57936d54-29a2-4820-abf6-72c0a73ad01f","resolution":{"observed_at":"2026-08-06T10:28:55.109500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.093264Z","title":"Diversifying spatial-temporal perception for video domain generalization","venue":null,"work_id":"032e7e76-335f-422e-b7ba-5311770d30b5","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.545030Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:67dcd1140c301b3e021214d8488431a4d8ea7f1239663baac42bcb6312c54a9d","observation_id":"b6e8e80e-5344-418f-b62f-76244a5cb4d8","resolution":{"observed_at":"2026-08-06T10:28:55.097919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.07796","last_updated":"2023-05-16T16:27:08Z","snapshot_observed_at":"2026-08-16T16:08:53.852053Z","submitted_at":"2022-12-13T19:17:36Z","title":"CREPE: Can Vision-Language Foundation Models Reason Compositionally?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.07796","snapshot_observed_at":"2026-08-06T10:28:54.548143Z","title":"Crepe: Can vision-language foundation models reason compositionally? arXiv preprint arXiv:2212.07796, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.548143Z"},"links":{"cited_paper":"/paper/2212.07796","citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:8f308a9fa43e6bef6740608b787410c55c7fbdcd35b91e4f1e27aed913364686","observation_id":"b1e6386c-a252-45eb-92d4-43f9adf7725f","resolution":{"observed_at":"2026-08-06T10:28:54.548143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.081888Z","title":"Reversible vision transformers, 2023","venue":null,"work_id":"1a02f35d-f0fe-48fd-8eff-ce998281be36","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.551508Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:608a2256385f432e22b6deb5a04439f9e30bd331798a5cf7e25723e3e219bb18","observation_id":"22d44eac-6ed7-4c97-b361-bd56c84f71a2","resolution":{"observed_at":"2026-08-06T10:28:55.085656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.070677Z","title":"Something-else: Com- positional action recognition with spatial-temporal interac- tion networks, 2020","venue":null,"work_id":"d8203593-4c1d-4c77-aae1-8fb6d7e281cc","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.554992Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:2fe3e23f36156ca05f481edf8a855915a49261d853d902ed737ffb214fc8cfde","observation_id":"2eeb25f0-5cdd-4b83-991a-4435967d8168","resolution":{"observed_at":"2026-08-06T10:28:55.074646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.059895Z","title":"Video action detection: Analysing limitations and chal- lenges","venue":null,"work_id":"a18e0c2f-5c91-4245-acb0-11c7f6d85a3f","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.557869Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:33f84b097b3e7964dcf538992a47d1be447a7348525056897275c01db650e283","observation_id":"e77c8da6-9f61-4534-a51c-cbc5903eddce","resolution":{"observed_at":"2026-08-06T10:28:55.063652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.048699Z","title":"Verbs in action: Improving verb understanding in video-language models, 2023","venue":null,"work_id":"295461ff-d346-4071-84b4-efc17cea0cfa","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.560757Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:4340ea154346bb1873d86616c9399ff875d6bb5a0f2e74916cbc86b9a5d644cf","observation_id":"892671ef-091a-4b2e-940b-b18ec982919a","resolution":{"observed_at":"2026-08-06T10:28:55.053080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.036755Z","title":"Multi-modal domain adaptation for fine-grained action recognition, 2020","venue":null,"work_id":"d11d3bec-25a1-4367-b156-547fe320deb9","year":2020},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.563918Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ffc463cbc012018f6b883bc5e689eed60ed3cbf227133cc0684ac286873542aa","observation_id":"26b927d0-a2b3-4e39-a858-4d3e2c9fd907","resolution":{"observed_at":"2026-08-06T10:28:55.040682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.024553Z","title":"Expanding language-image pretrained models for gen- eral video recognition, 2022","venue":null,"work_id":"26ac629d-ff9a-430e-a01b-87edba33395b","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.567472Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:0eac65ee3b5ca032514bc29fedf80ecbfe2adb30093eddc59141e23cd7bfaa9d","observation_id":"b6e0b3a6-c9d1-4d8a-a926-3452f9da15e5","resolution":{"observed_at":"2026-08-06T10:28:55.029829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.013137Z","title":"Object-relation reasoning graph for action recognition","venue":null,"work_id":"298cf650-c2e5-454f-81a1-2bfca71fe5b4","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.570569Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:0a00f6de8355dce5d8e6b9ad00aefde6a8b6139c154c761d2bec4c44957f6b31","observation_id":"787d62e3-cb75-4f10-bf79-827980ed49bd","resolution":{"observed_at":"2026-08-06T10:28:55.017675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:55.001141Z","title":"Relative norm align- ment for tackling domain shift in deep multi-modal classifi- cation","venue":null,"work_id":"f02ebe20-58eb-49f7-8c49-187b3b1731f2","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.573761Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:07cc11126ca7801b812ae98264ee00d05e777d476f2d264b89a3d3a363680191","observation_id":"af83d9fd-e4c4-4241-87f6-ad5a6a0a9629","resolution":{"observed_at":"2026-08-06T10:28:55.005872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.989737Z","title":"What can a cook in italy teach a mechanic in in- dia? action recognition generalisation over scenarios and locations","venue":null,"work_id":"7533d0dc-d58b-44ab-8c28-e57a93f59229","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.576855Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:8998ea413f46cac8b30de6ac4009d07f5ea8c923e4068e7b89c9f3e6676597bc","observation_id":"68496f9b-d914-4812-92bb-e028a11de5d2","resolution":{"observed_at":"2026-08-06T10:28:54.994152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.979004Z","title":"Haupt- mann","venue":null,"work_id":"b5498fe1-5a0b-4ae9-8c8a-e699fb8ac1f2","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.580251Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:5cba56d719d133b29df5b315483ce93f5976b712c91ab700efe9334c106a297a","observation_id":"38305df2-6c1c-408d-96c8-cb8443595034","resolution":{"observed_at":"2026-08-06T10:28:54.983155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.968812Z","title":"Learning transferable visual models from natural language supervision, 2021","venue":null,"work_id":"d061f6d9-39fa-444b-9ddf-e409ebf80f0b","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.583114Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:2dd231d6a0cb9111f5cb3e088a2bfcefad87e32b7097344822b755b9325c66c5","observation_id":"79955c59-9471-4acf-a6b2-336cc4f9a065","resolution":{"observed_at":"2026-08-06T10:28:54.972694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.957068Z","title":"Fine-tuned clip models are efficient video learners, 2023","venue":null,"work_id":"b5d5a665-5c7b-4d10-88f4-7abc3b8401dd","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.586075Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:d4840c680e8666aa81e42a8ca47e222a05c47db2dc8e934d202d438af883b029","observation_id":"a7f1584f-51af-4333-bc8e-1fbdaf22e13e","resolution":{"observed_at":"2026-08-06T10:28:54.961749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.944812Z","title":"Towards a fair evaluation of zero-shot action recognition using external data","venue":null,"work_id":"f88d7378-f7d5-453d-a881-d2af85bfda9a","year":2018},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.588742Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:0333b40971de1218e6235617e7d6b04ebd150e3db40d6e8dde442e4176240a51","observation_id":"f68b5a43-938a-499c-a654-944542e65419","resolution":{"observed_at":"2026-08-06T10:28:54.949384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.934031Z","title":"Probing conceptual understanding of large visual-language models","venue":null,"work_id":"3e054d9b-4956-4dae-9265-5cfd403f65f6","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.591622Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c6f24a917446fe441a2f602299b76c244c595c863bcf4921e6be77fd32091e14","observation_id":"32fade93-23ff-4d3e-a88c-f4ba88d9696c","resolution":{"observed_at":"2026-08-06T10:28:54.937728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.923181Z","title":"A Large-Scale Robustness Analysis of Video Action Recognition Models","venue":null,"work_id":"df0ee64d-642f-4a49-aec7-925bcc1209cb","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.594758Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ba7ef6a17113732788c69e1c38882e401e25709d1c29817cc84dfca0826d8c3d","observation_id":"504c5f4c-144e-40eb-9581-96c993488552","resolution":{"observed_at":"2026-08-06T10:28:54.927348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.911961Z","title":"Noisyactions2m: A multimedia dataset for video understanding from noisy la- bels","venue":null,"work_id":"ffab8cc8-c551-4960-8181-1cfe2eac7dcf","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.597449Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:3489c20cf1d4f1d0545588c3561ae56dd87ffb74ef5fcedcc4d9b020cfb9a610","observation_id":"d1362599-b1dc-4c80-8eb9-46476637bc94","resolution":{"observed_at":"2026-08-06T10:28:54.915584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.901300Z","title":"Learning long-term dependencies for action recognition with a biologically-inspired deep network","venue":null,"work_id":"aa889a49-57d3-48b9-9d46-5fa3b220c912","year":2017},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.600228Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:91a380811d4ecf7f112028db0fdd0d543af00380e2221604abad499904aaf562","observation_id":"f1c0c31e-6a0d-41ee-8d07-c595b425f102","resolution":{"observed_at":"2026-08-06T10:28:54.905442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.890655Z","title":"Dvanet: Disentangling view and action features for multi- view action recognition, 2023","venue":null,"work_id":"757ffa5c-0e11-4a59-85e7-10f8ce22b7ce","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.603319Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:bd9b33f2a63615993b1968099cc35ab56158475fe3077f3d7492fb17ff414bd0","observation_id":"867f2761-c42e-44f3-bd2c-413abf879707","resolution":{"observed_at":"2026-08-06T10:28:54.894525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.879854Z","title":"Spatio-temporal contrastive domain adaptation for action recognition","venue":null,"work_id":"ba1c8007-2ab4-48c3-a284-8880274baee7","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.606637Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:b57cae846229c76b278b3eabc1bc554a2e1e92e11ff008b5765dfd7751975633","observation_id":"ad117d2e-3ea7-4eb6-9f30-5bf3734c428e","resolution":{"observed_at":"2026-08-06T10:28:54.883346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.869391Z","title":"Learning That Transfers: Designing Curricu- lum for a Changing World","venue":null,"work_id":"94c406f2-9052-4da1-981b-1e6a800cf319","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.609775Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c2b173c0cbad6b0587f6c7f0f5f7530c5c30ee20824d973e131a0a15b0a460aa","observation_id":"3a8a2d6c-b30e-4d13-b393-c9b9148f1d3f","resolution":{"observed_at":"2026-08-06T10:28:54.873357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.859956Z","title":"Learning spatiotemporal features with 3d convolutional networks, 2015","venue":null,"work_id":"0bb88ccc-ff68-4978-bea8-2693b8e8e648","year":2015},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.612819Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:4da141cbaf3cd6a06979ff8dd1a307a6bcc9ae17829678d369ee37670f7d1ec0","observation_id":"587e19bd-fdd5-4757-abeb-31ff0b16b62a","resolution":{"observed_at":"2026-08-06T10:28:54.863239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.850940Z","title":"Actionclip: A new paradigm for video action recognition, 2021","venue":null,"work_id":"54054a0b-2a91-4dd6-ac93-2450099e0039","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.615730Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ad0589022e671f7d200ac6c34512b82eaf0f13b56042720dce32c0145c828141","observation_id":"9eb3db18-6364-4f5a-ae47-36ee7542ec7a","resolution":{"observed_at":"2026-08-06T10:28:54.854016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.841765Z","title":"An efficient spatio-temporal pyramid transformer for action detection","venue":null,"work_id":"428ae3ef-1016-47e3-a25c-79f35f63cfb2","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.618561Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:2a5ca6c78714e73bdb777285939dbf07ba7fdcbee21a8936fb4ef063571fed52","observation_id":"bb5607ac-3cb2-424f-82c7-cc0bd58ec9af","resolution":{"observed_at":"2026-08-06T10:28:54.844696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.831294Z","title":"Action recognition using attention-based spatio-temporal vlad net- works and adaptive video sequences optimization","venue":null,"work_id":"aca99139-2e57-447a-a39b-23928b218064","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.621464Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:993b043cea4411847405e73560965236baccb272a80489cc24eba62b8eb11a58","observation_id":"11fc7ff5-e0d3-41e4-a802-70a19262aa49","resolution":{"observed_at":"2026-08-06T10:28:54.834618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.821775Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding, 2021","venue":null,"work_id":"0ac0003e-fbf3-4b70-9653-cfcc999ae223","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.624228Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c4ab3bfba7e8a971987e351c464c12c364af4fcc39a205206d044dc32148cb2d","observation_id":"2a226360-e3ad-4405-a521-2893ee49c6d0","resolution":{"observed_at":"2026-08-06T10:28:54.824981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.812775Z","title":"Seman- tic embedding space for zero-shot action recognition, 2015","venue":null,"work_id":"5c93e512-dac6-40d1-9783-410329ff5a4d","year":2015},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.627336Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:2133b495ce344f4562352e6033949f624c72652eff32108d8c9070e2192d7622","observation_id":"28f64da9-fbcc-474d-84b5-18a267dadf6d","resolution":{"observed_at":"2026-08-06T10:28:54.816065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.803956Z","title":"Interact before align: Leveraging cross-modal knowledge for domain adaptive action recognition","venue":null,"work_id":"bd2c2bfa-30dd-40f0-8e7d-1617071e169a","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.630326Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ede9d9523214d26368e7642d886bb96fe4d11293ad82a55dd30c1bfd1f935fb7","observation_id":"3c6b1119-2439-4be1-9dc6-6a20fa1adf27","resolution":{"observed_at":"2026-08-06T10:28:54.806924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.794752Z","title":"Aim: Adapting image models for efficient video action recognition, 2023","venue":null,"work_id":"31247d9e-04d2-4071-9908-02d9f6879c7e","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.633190Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:ed19f20973ea8eff8c79774cf3996b62d114751e09ce5ec4d539bfa9ec0ab919","observation_id":"85357b52-8ea7-4850-a168-325123912e01","resolution":{"observed_at":"2026-08-06T10:28:54.797794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.784012Z","title":"Yu, and Mingsheng Long","venue":null,"work_id":"a4046089-7d34-4f06-a856-b0809cea0b56","year":2021},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.636128Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:844c8e664b6a14f6ca1877f936d3d33006e4400a975e16e9f858c0e0a4b334c7","observation_id":"0cef7622-926d-4dc8-af62-601f20884879","resolution":{"observed_at":"2026-08-06T10:28:54.787184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.774190Z","title":"Action4d: Online action recognition in the crowd and clutter","venue":null,"work_id":"74506c56-dce4-48fb-bdc8-f1b9b11819dd","year":2019},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.638929Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:c0d917737592f46445cfba64395a551a66bbeced6cb3078592589d4aa2c13ef0","observation_id":"7b86106e-d247-4527-934a-8f81ead14cc9","resolution":{"observed_at":"2026-08-06T10:28:54.777616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.764093Z","title":"Eliciting in-context learning in vision-language models for videos through curated data dis- tributional properties","venue":null,"work_id":"55baac99-2b47-44ba-a7f5-aab9de54090e","year":2024},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.641706Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:d407be92e481c12c58fb53eaa733949f36d4209860aab3989f7ddff4f3d1f720","observation_id":"e540d4b5-4806-4caf-856a-be6a38c73265","resolution":{"observed_at":"2026-08-06T10:28:54.767562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.753693Z","title":"Derpa- nis","venue":null,"work_id":"e6ac3f01-e246-4a43-96e7-6d714ed1c724","year":2013},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.644639Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:b6ecfca5c8ccaf643c96af15fb18ee3b76cd3164554e76f6327e3f07145739f9","observation_id":"5cd47e77-f37a-4a85-9fe7-41a933ac0fa3","resolution":{"observed_at":"2026-08-06T10:28:54.756813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.742505Z","title":"Human-object interaction detection via disentangled transformer, 2022","venue":null,"work_id":"d0589337-a29d-4b94-b38a-369055fd902f","year":2022},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.647366Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:09d7e48ae9569aa09be4dc841824626b51624035f10a952086ef2360b94d82b0","observation_id":"bda1ab80-2b39-4446-a4d7-1e173f79c10b","resolution":{"observed_at":"2026-08-06T10:28:54.746720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.732624Z","title":"How can objects help action recognition?, 2023","venue":null,"work_id":"0d6bba58-a70a-4079-a105-b8f113a9087c","year":2023},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.650326Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:27f23b3835fb98f75c64184b4353a0dad62d743b0f324206657bd4953536cbbc","observation_id":"3339db00-70e0-474d-9735-90022abe873b","resolution":{"observed_at":"2026-08-06T10:28:54.735895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.722524Z","title":"Among multi- modal models, we experimented with different variations of CLIP [46] designed for activity recognition","venue":null,"work_id":"c532d24c-174c-4310-9781-0fb76d67b05a","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.652926Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:67f9941e3fc7315615696daa04daa071e19cc8ff60c3671fb56654dd721d5702","observation_id":"137d15bd-2b1c-4d35-b8bc-522730ff76cc","resolution":{"observed_at":"2026-08-06T10:28:54.725793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.712904Z","title":"pre-train, prompt, and fine-tune","venue":null,"work_id":"f57042e5-468e-4500-b98d-a6edad984327","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.656142Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:75925af7a425bc7cabbc127e352ca72264e5d7417ebdace236a0acbd53bfdbb6","observation_id":"85cb6713-3110-4fe6-904a-325892f1c910","resolution":{"observed_at":"2026-08-06T10:28:54.716403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:28:54.699881Z","title":"Pushing”) and the other focuses on fine-grained con- text (e.g., “Pushing something from left to right","venue":null,"work_id":"2f385a6c-8dee-494e-b4ba-fa956c84e5f9","year":null},"citing_paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:54.659185Z"},"links":{"citing_paper":"/paper/2508.00085"},"observation_digest":"sha256:9c5032e6a0b8c77226334d0bb2ab3583b10343bafd787718e3ae34d526f4d5e3","observation_id":"ab13211d-4430-44ab-9f90-072ca0fa2e2e","resolution":{"observed_at":"2026-08-06T10:28:54.704808Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.00085","last_updated":"2025-07-31T18:19:20Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-22T03:24:28.872547Z","submitted_at":"2025-07-31T18:19:20Z","title":"Punching Bag vs. Punching Person: Motion Transferability in Videos"},"reference_resolution":{"displayed":72,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":0,"verified_fuzzy":69},"total_outbound_references":72},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 72 of 72 outbound references and 0 inbound Pith citation observations for arXiv:2508.00085."}