{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FRNJJDKF3K7ME4HKZG2GAUXWBG","short_pith_number":"pith:FRNJJDKF","schema_version":"1.0","canonical_sha256":"2c5a948d45dabec270eac9b46052f6099318a26c5ca3aaa4f0faab4ff28f4ae2","source":{"kind":"arxiv","id":"1910.04744","version":2},"attestation_state":"computed","paper":{"title":"CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deva Ramanan, Rohit Girdhar","submitted_at":"2019-10-10T17:52:19Z","abstract_excerpt":"Computer vision has undergone a dramatic revolution in performance, driven in large part through deep features trained on large-scale supervised datasets. However, much of these improvements have focused on static image analysis; video understanding has seen rather modest improvements. Even though new datasets and spatiotemporal models have been proposed, simple frame-by-frame classification methods often still remain competitive. We posit that current video datasets are plagued with implicit biases over scene and object structure that can dwarf variations in temporal structure. In this work, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.04744","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-10-10T17:52:19Z","cross_cats_sorted":[],"title_canon_sha256":"9837ce799c3941d4d78c34cd7e6ba18ddbc6ca46c5c1bfe828d65e595d4d5353","abstract_canon_sha256":"b1bc38e6a631ff41f5ff0b71b8361850411b1451b7a5874095e370051f5d94ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:52:48.464992Z","signature_b64":"U4YqzXY2Ju9O9nC3o8iEh6IKnhB0zgNq4FIJABp1Ha3/lTniJLPRa7pY8MGe6C/l8mMvoZUVeUHXePwoa+wtCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c5a948d45dabec270eac9b46052f6099318a26c5ca3aaa4f0faab4ff28f4ae2","last_reissued_at":"2026-07-05T00:52:48.464503Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:52:48.464503Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deva Ramanan, Rohit Girdhar","submitted_at":"2019-10-10T17:52:19Z","abstract_excerpt":"Computer vision has undergone a dramatic revolution in performance, driven in large part through deep features trained on large-scale supervised datasets. However, much of these improvements have focused on static image analysis; video understanding has seen rather modest improvements. Even though new datasets and spatiotemporal models have been proposed, simple frame-by-frame classification methods often still remain competitive. We posit that current video datasets are plagued with implicit biases over scene and object structure that can dwarf variations in temporal structure. In this work, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.04744","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.04744/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.04744","created_at":"2026-07-05T00:52:48.464561+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.04744v2","created_at":"2026-07-05T00:52:48.464561+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.04744","created_at":"2026-07-05T00:52:48.464561+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRNJJDKF3K7M","created_at":"2026-07-05T00:52:48.464561+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRNJJDKF3K7ME4HK","created_at":"2026-07-05T00:52:48.464561+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRNJJDKF","created_at":"2026-07-05T00:52:48.464561+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26694","citing_title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07962","citing_title":"ChronoPhyBench: Do MLLMs Truly Understand the World or Merely Exploit Language Priors?","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26694","citing_title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27589","citing_title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22570","citing_title":"VGenST-Bench: A Benchmark for Spatio-Temporal Reasoning via Active Video Synthesis","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2403.00476","citing_title":"TempCompass: Do Video LLMs Really Understand Videos?","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12449","citing_title":"LychSim: A Controllable and Interactive Simulation Framework for Vision Research","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG","json":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG.json","graph_json":"https://pith.science/api/pith-number/FRNJJDKF3K7ME4HKZG2GAUXWBG/graph.json","events_json":"https://pith.science/api/pith-number/FRNJJDKF3K7ME4HKZG2GAUXWBG/events.json","paper":"https://pith.science/paper/FRNJJDKF"},"agent_actions":{"view_html":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG","download_json":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG.json","view_paper":"https://pith.science/paper/FRNJJDKF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.04744&json=true","fetch_graph":"https://pith.science/api/pith-number/FRNJJDKF3K7ME4HKZG2GAUXWBG/graph.json","fetch_events":"https://pith.science/api/pith-number/FRNJJDKF3K7ME4HKZG2GAUXWBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG/action/storage_attestation","attest_author":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG/action/author_attestation","sign_citation":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG/action/citation_signature","submit_replication":"https://pith.science/pith/FRNJJDKF3K7ME4HKZG2GAUXWBG/action/replication_record"}},"created_at":"2026-07-05T00:52:48.464561+00:00","updated_at":"2026-07-05T00:52:48.464561+00:00"}