{"as_of":"2026-08-09T05:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fceba19dd8015e66fb4f22f335603802f388125832a9e6444224f8ba05f579e9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":53,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T20:36:25.525906Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T13:19:51.020711Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-07T12:45:05.096694Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23656","last_updated":"2025-05-29T17:06:44Z","snapshot_observed_at":"2026-08-07T12:38:23.686012Z","submitted_at":"2025-05-29T17:06:44Z","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:05.096694Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2505.23656"},"observation_digest":"sha256:f347c317e044aa14b23ba1ef399b021bd2a7af79f436a4de299661b0caf16da3","observation_id":"101b1ba2-fbfd-4999-b886-9c911a6af147","resolution":{"observed_at":"2026-08-07T12:45:05.096694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-06T16:30:50.339759Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13428","last_updated":"2026-05-26T08:34:24Z","snapshot_observed_at":"2026-08-06T16:22:35.916708Z","submitted_at":"2025-07-17T17:54:09Z","title":"\"PhyWorldBench\": A Comprehensive Evaluation of Physical Realism in Text-to-Video Models","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T16:30:50.339759Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2507.13428"},"observation_digest":"sha256:dd22281ec308878cbe60dba092b177b288b5daa71b99319d12ec7d2d501c713b","observation_id":"5952afe4-b42d-4238-8458-97351e4b5921","resolution":{"observed_at":"2026-08-06T16:30:50.339759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-06T15:27:00.441134Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15824","last_updated":"2025-07-21T17:30:46Z","snapshot_observed_at":"2026-08-07T10:31:41.147971Z","submitted_at":"2025-07-21T17:30:46Z","title":"Can Your Model Separate Yolks with a Water Bottle? Benchmarking Physical Commonsense Understanding in Video Generation Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:27:00.441134Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2507.15824"},"observation_digest":"sha256:e393c49a1eda49b8bfe467cea1ee42af42eca413d37cddd78228a2b4335abebe","observation_id":"5408ffb6-a523-4df4-b901-5475e57ec79a","resolution":{"observed_at":"2026-08-06T15:27:00.441134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-05T10:59:25.198610Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evalua- tion in Video Generation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.03385","last_updated":"2025-09-03T15:02:40Z","snapshot_observed_at":"2026-08-08T21:04:36.444975Z","submitted_at":"2025-09-03T15:02:40Z","title":"Human Preference-Aligned Concept Customization Benchmark via Decomposed Evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T10:59:25.198610Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2509.03385"},"observation_digest":"sha256:d94f4d7cc20ceabdee6f3fd923fb52fb95aab2b720b027cbfee73d05f8097408","observation_id":"8814fc66-f840-41c2-83fd-f5297fd85d97","resolution":{"observed_at":"2026-08-05T10:59:25.198610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2509.24702","last_updated":"2026-04-06T10:12:03Z","snapshot_observed_at":"2026-08-02T11:34:44.120596Z","submitted_at":"2025-09-29T12:32:54Z","title":"Enhancing Physical Plausibility in Video Generation by Reasoning the Implausibility","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T12:55:42.679016Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2509.24702"},"observation_digest":"sha256:12b651d13d638cb7dadf03e62a1324e4945a24192d965a81e4854e97101fb706","observation_id":"3512bfe1-4b8d-4c9d-99a2-9424d273ac60","resolution":{"observed_at":"2026-05-18T12:56:24.343732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2511.00062","last_updated":"2026-02-24T21:52:50Z","snapshot_observed_at":"2026-07-06T22:34:38.619949Z","submitted_at":"2025-10-28T22:44:13Z","title":"World Simulation with Video Foundation Models for Physical AI","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T23:01:13.546110Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2511.00062"},"observation_digest":"sha256:c3b4bf80172350016190d28722958b8bfbad21da528f36401b095b2ce02fdeaa","observation_id":"feea1a15-9db2-4ab4-8253-1d0e571742b9","resolution":{"observed_at":"2026-05-12T23:01:13.918325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2511.18373","last_updated":"2026-04-11T05:44:20Z","snapshot_observed_at":"2026-08-07T02:29:17.852730Z","submitted_at":"2025-11-23T09:43:44Z","title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T05:55:11.495430Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2511.18373"},"observation_digest":"sha256:6b6ca7d5418f9765883a4746ff5897b61fa2244ecce75bec9dc83981f840fbc4","observation_id":"500a90b3-6903-4d27-8f9b-9b60434e040c","resolution":{"observed_at":"2026-05-17T05:59:08.743022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-03T19:11:47.098371Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.01803","last_updated":"2026-07-09T15:41:25Z","snapshot_observed_at":"2026-08-08T13:02:50.552789Z","submitted_at":"2025-12-01T15:36:33Z","title":"Generative Action Tell-Tales: Assessing Human Motion in Synthesized Videos","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T19:11:47.098371Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.01803"},"observation_digest":"sha256:1f2fc05bfaad0c1da4931e2ec76261d215418568741feef83d75cd32891693ac","observation_id":"2eaf1f6a-91c2-4d27-b10a-c018831d97ce","resolution":{"observed_at":"2026-08-03T19:11:47.098371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.01843","last_updated":"2026-05-18T11:10:13Z","snapshot_observed_at":"2026-08-08T17:35:04.569091Z","submitted_at":"2025-12-01T16:28:13Z","title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T17:57:57.263574Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.01843"},"observation_digest":"sha256:c2f81503c66de7262cd1680bfa4fa913dd955a51c87bd456578455cafe135f7f","observation_id":"7661578d-9147-4d74-b3f5-cc93263bf7bd","resolution":{"observed_at":"2026-05-21T18:00:27.237634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.05564","last_updated":"2026-04-12T02:52:57Z","snapshot_observed_at":"2026-07-06T22:37:50.431948Z","submitted_at":"2025-12-05T09:39:26Z","title":"ProPhy: Progressive Physical Alignment for Dynamic World Simulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T01:05:08.087136Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.05564"},"observation_digest":"sha256:fec69f00dcb6d9d3c371bfaf569cf0b3ebd1ebae5963cb5a81bf08611508ed29","observation_id":"8a6a0aeb-f9d3-42e4-80ff-be1b46426c19","resolution":{"observed_at":"2026-05-17T01:08:47.998118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.23292","last_updated":"2026-05-20T15:48:38Z","snapshot_observed_at":"2026-08-01T23:47:11.303976Z","submitted_at":"2025-12-29T08:26:27Z","title":"Agentic Physical AI toward a Domain-Specific Foundation Model for Nuclear Reactor Control","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T16:57:19.490074Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.23292"},"observation_digest":"sha256:39db092cd052a8c4d48ea49f227cfc9f5c42df536311aa49c3f20f92be300399","observation_id":"98814dd8-c6bb-49dc-8692-50fc179782af","resolution":{"observed_at":"2026-05-21T17:00:23.976691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2601.18577","last_updated":"2026-05-20T05:19:37Z","snapshot_observed_at":"2026-07-06T22:43:02.453697Z","submitted_at":"2026-01-26T15:22:27Z","title":"Self-Refining Video Sampling","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T14:37:57.167882Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2601.18577"},"observation_digest":"sha256:6851f5fdfef5e4d3f0658b5c2dd415c9ed3dbc083d96db507bc236e51e3a1e5c","observation_id":"4ed3be5e-8e4f-4170-a9ab-4d69a8c8e3c6","resolution":{"observed_at":"2026-05-21T14:40:14.478534Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.06339","last_updated":"2026-04-07T18:17:05Z","snapshot_observed_at":"2026-07-06T22:54:53.308309Z","submitted_at":"2026-04-07T18:17:05Z","title":"Evolution of Video Generative Foundations","version":1},"reference_index":170,"source":"pdf_text","source_observed_at":"2026-05-10T18:41:38.616611Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.06339"},"observation_digest":"sha256:620df8541c81e1f52fde81aca6b8143bf0b9271478f91c3bddfa167ee34853a4","observation_id":"c657b1b5-45b9-42b6-b311-c1d46f519cb2","resolution":{"observed_at":"2026-05-11T00:05:51.471254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.07348","last_updated":"2026-04-08T17:59:22Z","snapshot_observed_at":"2026-07-06T22:55:37.787732Z","submitted_at":"2026-04-08T17:59:22Z","title":"MoRight: Motion Control Done Right","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:38:03.776766Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.07348"},"observation_digest":"sha256:7faef875b23328fd070547bceaab9233372fb4965388d8e6f44d1b0a9c317ba2","observation_id":"ddbbf14e-72a5-420a-abb7-34edde3c7d7f","resolution":{"observed_at":"2026-05-11T06:26:01.050690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.09415","last_updated":"2026-04-10T15:27:27Z","snapshot_observed_at":"2026-07-06T22:58:17.167367Z","submitted_at":"2026-04-10T15:27:27Z","title":"PhysInOne: Visual Physics Learning and Reasoning in One Suite","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:39:48.066744Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.09415"},"observation_digest":"sha256:9ba8c7072909a78e3811da5321b70de6728a7737477774565ae3f16fdb2f76f4","observation_id":"6f66a35d-6b39-4dba-991d-3817f187d81b","resolution":{"observed_at":"2026-05-11T08:25:59.537119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.19193","last_updated":"2026-04-21T08:04:02Z","snapshot_observed_at":"2026-08-02T22:05:18.077263Z","submitted_at":"2026-04-21T08:04:02Z","title":"How Far Are Video Models from True Multimodal Reasoning?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:44:52.920816Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.19193"},"observation_digest":"sha256:e18d90e4c2dfa21cde6facf868aa311e653cf1f23396297cc0e56fa55dea7a55","observation_id":"9864ebcd-f30e-4cc2-9c58-0e545b7f4650","resolution":{"observed_at":"2026-05-11T12:51:03.643358Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.00873","last_updated":"2026-04-24T21:34:52Z","snapshot_observed_at":"2026-07-06T23:14:11.211164Z","submitted_at":"2026-04-24T21:34:52Z","title":"BRITE: A Benchmark for Reliable and Interpretable T2V Evaluation on Implausible Scenarios","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-09T21:12:33.209353Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.00873"},"observation_digest":"sha256:6094181b4c5721794df270e89801feb8c323b8c764a525c390c12fcf8e063ac9","observation_id":"199032ec-d394-47bc-b93f-e3e2d2bee3c2","resolution":{"observed_at":"2026-05-11T14:41:33.649224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07061","last_updated":"2026-05-29T22:08:19Z","snapshot_observed_at":"2026-08-03T12:53:30.755817Z","submitted_at":"2026-05-08T00:14:07Z","title":"Do Joint Audio-Video Generation Models Understand Physics?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T02:12:04.230076Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07061"},"observation_digest":"sha256:5f82a438463b03c04bd5e6e17945800889ec5d10f54ae5ccb1be694b85f471dd","observation_id":"236bef27-0804-4a57-a7eb-11e53c67c88e","resolution":{"observed_at":"2026-05-11T03:50:56.767419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07061","last_updated":"2026-05-29T22:08:19Z","snapshot_observed_at":"2026-08-03T12:53:30.755817Z","submitted_at":"2026-05-08T00:14:07Z","title":"Do Joint Audio-Video Generation Models Understand Physics?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T23:39:22.070629Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07061"},"observation_digest":"sha256:ac4ba6ab1aeeec8f1f29fe75b1e76346e0ab269a7cc6e15356238b12cf442fa1","observation_id":"68ca9a9f-28a4-45f6-97c7-000e535ba148","resolution":{"observed_at":"2026-06-30T23:45:08.203083Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07800","last_updated":"2026-06-09T17:19:58Z","snapshot_observed_at":"2026-08-02T01:30:37.148519Z","submitted_at":"2026-05-08T14:36:32Z","title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T02:21:52.861714Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07800"},"observation_digest":"sha256:c420ad16d9bf3ae1b3931fafc7b1debeb678b091827222244b61393ff6c03d65","observation_id":"c4348ac0-7666-440b-b235-55241b7a45f7","resolution":{"observed_at":"2026-05-11T03:40:54.609388Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07800","last_updated":"2026-06-09T17:19:58Z","snapshot_observed_at":"2026-08-02T01:30:37.148519Z","submitted_at":"2026-05-08T14:36:32Z","title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T23:13:31.195562Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07800"},"observation_digest":"sha256:11820b7fb63aa9ca5beb7326121b1f9260afdecb4c1f321e3de3feb1e3ed86af","observation_id":"ae39e17c-04af-48de-bf8a-868a40164b10","resolution":{"observed_at":"2026-06-30T23:15:08.007606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-07-06T23:22:42.512039Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:1a2a9e09a105b3e13b3b9d4a156e8bfe75f2f3c6782982eae31891f9cf678f35","observation_id":"42134aa3-572d-4f86-9217-752a2cba9445","resolution":{"observed_at":"2026-05-12T05:21:27.838360Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T06:00:31.582714Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:eb991941240deeeec0f47571a17f0e9daf7d3a2b94c096eda13e200be03b6b3b","observation_id":"c6006889-d1ff-4cf3-818a-0b4733a448b9","resolution":{"observed_at":"2026-05-13T06:02:22.162702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T22:38:43.102769Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:eb97ff63d5e1802a7239c87363f200daa7d7eb024a490e39973fb3d97e52b416","observation_id":"2dac5e39-8a77-4e2b-a560-ced9b61bfc1b","resolution":{"observed_at":"2026-07-01T13:55:45.294459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.15116","last_updated":"2026-05-14T17:29:35Z","snapshot_observed_at":"2026-07-06T23:26:28.006734Z","submitted_at":"2026-05-14T17:29:35Z","title":"DriveCtrl: Conditioned Sim-to-Real Driving Video Generation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-30T21:06:06.548538Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.15116"},"observation_digest":"sha256:f5562747eb7fb2526ce470b8a5c7a71e7e9aa5f56e2c370088c779dda844c626","observation_id":"b5c2cfd1-1775-4c66-bdd3-46d10217d19f","resolution":{"observed_at":"2026-06-30T21:15:04.778745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.18396","last_updated":"2026-05-19T06:23:43Z","snapshot_observed_at":"2026-07-06T23:29:14.916114Z","submitted_at":"2026-05-18T13:42:24Z","title":"NEWTON: Agentic Planning for Physically Grounded Video Generation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T10:56:22.343333Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.18396"},"observation_digest":"sha256:9426c7d6d357c12ee869aaef15996747cfe711a6724b7c377c7ef8379e4aa77f","observation_id":"881ec02c-7a65-4267-a803-556ef6b74eb5","resolution":{"observed_at":"2026-05-20T10:58:13.762531Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.19242","last_updated":"2026-05-19T01:28:52Z","snapshot_observed_at":"2026-08-06T22:30:04.170402Z","submitted_at":"2026-05-19T01:28:52Z","title":"PhyWorld: Physics-Faithful World Model for Video Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-20T07:28:20.248452Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.19242"},"observation_digest":"sha256:e07e2be3f2839351546903cf1d0fb6a7a41641bb887b58e8e0f70ef0208eff97","observation_id":"a98e831c-1b5f-4ca3-91d7-63573d7e2c39","resolution":{"observed_at":"2026-05-20T07:33:07.595472Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.23699","last_updated":"2026-05-22T14:51:22Z","snapshot_observed_at":"2026-07-06T23:33:53.870540Z","submitted_at":"2026-05-22T14:51:22Z","title":"CRONOS: Benchmarking Counterfactual Physical Consistency in Video Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T04:39:22.400458Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.23699"},"observation_digest":"sha256:58afa5b15ad3dd77b40361ff292cdc3249250af6989ee174a051a55055ee831b","observation_id":"5cd09e5e-1276-445c-953d-e2da461e1bdb","resolution":{"observed_at":"2026-05-25T04:40:23.190574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.23878","last_updated":"2026-05-22T17:34:42Z","snapshot_observed_at":"2026-08-02T04:44:48.004184Z","submitted_at":"2026-05-22T17:34:42Z","title":"LaMo: Self-Supervised Latent Motion Priors for Physical Realism in Video Generation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T04:42:32.717968Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.23878"},"observation_digest":"sha256:c7a114ae4b2222b36066be0f940c6a0b1a8e27d2f9b0c147b63798d81c134590","observation_id":"c747a6d3-4c52-4ddd-b01f-5d5acc06334a","resolution":{"observed_at":"2026-05-25T04:45:20.481958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.24962","last_updated":"2026-05-24T09:28:05Z","snapshot_observed_at":"2026-08-02T19:14:30.533296Z","submitted_at":"2026-05-24T09:28:05Z","title":"Tempered Self-Similarity Alignment for Physically Plausible Video Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T11:39:06.597513Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.24962"},"observation_digest":"sha256:9aa552f1ce16245344d1ecba793dc10eb6bcdbf74d773885e4a860d389b1c3cd","observation_id":"3916976f-d93f-406c-9aca-fa3e7b81f9c7","resolution":{"observed_at":"2026-06-30T11:44:38.371299Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.25874","last_updated":"2026-05-25T14:01:31Z","snapshot_observed_at":"2026-07-06T23:35:46.157653Z","submitted_at":"2026-05-25T14:01:31Z","title":"WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-29T22:57:08.381846Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.25874"},"observation_digest":"sha256:293a8d6bdadff7fa1700335bd73adc503941c6694c6e09e26f2ca3193dfe2275","observation_id":"31fa3a2e-2dd7-4854-ba29-969d8e717dd3","resolution":{"observed_at":"2026-06-29T23:14:02.269296Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.27589","last_updated":"2026-05-26T19:02:26Z","snapshot_observed_at":"2026-08-05T09:42:53.562841Z","submitted_at":"2026-05-26T19:02:26Z","title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T18:23:22.987086Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.27589"},"observation_digest":"sha256:7bfc43cbdb02e90f4abc25c761689904bceee32993af73d62a18aae0e16de70a","observation_id":"2dc8fd89-4e71-47a0-8c81-0f5c24d823d7","resolution":{"observed_at":"2026-06-29T18:23:50.403328Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.28230","last_updated":"2026-05-27T09:44:18Z","snapshot_observed_at":"2026-08-07T12:12:27.862303Z","submitted_at":"2026-05-27T09:44:18Z","title":"Proprio: Latent Self-Scoring and Inference-Time Refinement for Physically Plausible Video Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T12:55:24.689338Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.28230"},"observation_digest":"sha256:9faa84054ef1e7edf3ac36a2c72422d2e2b8aaca242590110091ce23002b8cdd","observation_id":"928a2668-cecb-4338-97a9-034a1af2aa1c","resolution":{"observed_at":"2026-06-29T13:03:26.633426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.30346","last_updated":"2026-05-28T17:59:51Z","snapshot_observed_at":"2026-08-04T10:14:05.552405Z","submitted_at":"2026-05-28T17:59:51Z","title":"YoCausal: How Far is Video Generation from World Model? A Causality Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T08:27:03.674229Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.30346"},"observation_digest":"sha256:dd6a3bb1dcd40405a84733a33cb292c1ee32e74831182ceb95ec1528e9f27c20","observation_id":"b092f7a9-95a8-4f84-9bd4-cd2b18a83278","resolution":{"observed_at":"2026-06-29T08:33:15.610723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.00499","last_updated":"2026-05-30T03:13:00Z","snapshot_observed_at":"2026-07-06T23:41:10.825760Z","submitted_at":"2026-05-30T03:13:00Z","title":"OptiWorld: Optimal Control for Video World Generation under Physical Constraints","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T19:02:51.848742Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.00499"},"observation_digest":"sha256:32247abf3804d4e01aa136e0d2984a7e92fbd6cdd157cd564f83f9b565e7781e","observation_id":"2a5c3853-3881-45ff-821b-facdf2267b1e","resolution":{"observed_at":"2026-06-28T19:32:35.562695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.01538","last_updated":"2026-06-11T05:58:14Z","snapshot_observed_at":"2026-08-02T12:35:46.473849Z","submitted_at":"2026-06-01T01:36:44Z","title":"MPMWorlds: Material-Point-Method Simulations for Inferring and Extrapolating Physical Dynamics","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T12:19:27.221596Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.01538"},"observation_digest":"sha256:ad46fddf598063608e29f9f721363c772ae435443963c10f93c50dd49e3ca4a5","observation_id":"c199f755-4fc8-4cc0-8f1d-1a8b4870040a","resolution":{"observed_at":"2026-07-02T01:16:24.684102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.04737","last_updated":"2026-06-03T11:20:00Z","snapshot_observed_at":"2026-07-06T23:44:47.848374Z","submitted_at":"2026-06-03T11:20:00Z","title":"Physics-Informed Video Generation via Mixture-of-Experts Latent Alignment","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:37.291472Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.04737"},"observation_digest":"sha256:3dbb777832f8a021c23176cdc512b5279f3d9ab14567896dd363f884a3f8901c","observation_id":"d8879029-c4bb-428d-9030-f6d732fd5ae2","resolution":{"observed_at":"2026-07-02T07:16:44.758956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.20545","last_updated":"2026-06-18T17:55:15Z","snapshot_observed_at":"2026-07-06T23:55:38.979306Z","submitted_at":"2026-06-18T17:55:15Z","title":"Current World Models Lack a Persistent State Core","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T17:33:41.461245Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.20545"},"observation_digest":"sha256:0ccaac74fc176f78c77c0fde4030bbc6d54f5b240bf0669e0f816116ae3e2022","observation_id":"db7ea51f-8389-43de-bceb-5e4a862c72a0","resolution":{"observed_at":"2026-07-04T03:49:31.017133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.22918","last_updated":"2026-06-22T06:58:39Z","snapshot_observed_at":"2026-07-06T23:57:42.632111Z","submitted_at":"2026-06-22T06:58:39Z","title":"Each Judge Its Own Yardstick: Discovering Per-VLM Taxonomies for Physical Video Evaluation","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T09:04:23.965554Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.22918"},"observation_digest":"sha256:2909bd709100c58d4e95fa23d31f7382d4c6dd218a06382f414cbc441bd43a98","observation_id":"7e2c8aa6-6fec-46e8-a275-c5d9d023e5a7","resolution":{"observed_at":"2026-07-04T10:09:45.415761Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.26916","last_updated":"2026-06-25T11:53:27Z","snapshot_observed_at":"2026-07-07T00:01:10.711037Z","submitted_at":"2026-06-25T11:53:27Z","title":"PhysRAG: Enhancing Physics-Awareness in Video Generation via Retrieval-Augmented Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T05:16:53.011837Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.26916"},"observation_digest":"sha256:1f212dcd6264b51fccb2dbe8ee7b9dca0808dfdb7d66d52d971b850ffbc09279","observation_id":"03148a37-6e8a-4ca4-a273-bb1a937b2c8b","resolution":{"observed_at":"2026-07-04T13:19:51.022182Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T02:03:45.564122Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:f0db85ceaae2d4e0711aa3b25ffe2dd6324296c3a434ef1053571a8bb7189561","observation_id":"5b8ddd1a-9cf9-47f0-88fd-f3abfedb3e5a","resolution":{"observed_at":"2026-07-01T18:25:57.779239Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-01T06:25:58.872140Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:7ce5db633d7deef3a67d3b84fe23b25b9656d3dd6105c7fbc64e6a08c35f6f73","observation_id":"776a5435-d44c-4177-9a53-4db6cf6b2e37","resolution":{"observed_at":"2026-07-01T09:35:40.499609Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-02T20:52:28.444524Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:43676a891f0af778321b4c5a68aa81ca5b2c448441d101c8279937687d6ec78f","observation_id":"9cd71609-d385-4009-b531-d27eaf95c46a","resolution":{"observed_at":"2026-07-02T20:57:22.806256Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-03T22:44:16.272541Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:a79b53961876d1bae098264043984349fbee5e99c8fe84181217ed9de7bd80ae","observation_id":"fbfc599f-82d6-4ac7-9073-49c0a272493f","resolution":{"observed_at":"2026-07-03T22:49:00.899176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-14T17:14:19.770867Z","title":"arXiv preprint arXiv:2503.06800 (2025) 3, 4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-14T17:14:19.770867Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:783b4c186b0956b729aacd120192790dd91c246cf873b2bf2cbcd2a46cfd5a2d","observation_id":"06aee8cc-d380-4e35-81c5-407ff4d20c72","resolution":{"observed_at":"2026-07-14T17:14:19.770867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-02T10:00:09.187494Z","title":"arXiv preprint arXiv:2503.06800 (2025) 3, 4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T10:00:09.187494Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:d631d793d462621e0c5fd3a3b7aba60cf5ced819e84d6f536656c5489b4db3da","observation_id":"911c1311-d029-402b-81d2-40a9d4c148af","resolution":{"observed_at":"2026-08-02T10:00:09.187494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T21:07:22.906584Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16401","last_updated":"2026-07-17T18:00:05Z","snapshot_observed_at":"2026-08-08T02:26:45.240699Z","submitted_at":"2026-07-17T18:00:05Z","title":"Apple-$\\pi$: Benchmarking Thinking with Video Towards Law-Grounded Physical Intelligence","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T21:07:22.906584Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.16401"},"observation_digest":"sha256:065b2d2d67af1a8aeeb1017f15b7ff1b6675251277fc3f56605b359e1b5f0e7c","observation_id":"42136782-eba8-4cb4-8ca0-4f81411e752f","resolution":{"observed_at":"2026-08-01T21:07:22.906584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T19:33:03.502183Z","title":"Videophy-2: A challenging action-centric physical com- monsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16947","last_updated":"2026-07-18T20:03:50Z","snapshot_observed_at":"2026-08-07T14:13:12.484072Z","submitted_at":"2026-07-18T20:03:50Z","title":"When Physical Preferences Meet Semantic Constraints: Physical and Semantic Direct Preference Optimization for Text-to-Video Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T19:33:03.502183Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.16947"},"observation_digest":"sha256:64b5ba2e0c448591edef7a533f363f40801f5de6873d49223e6c56eb74a359ab","observation_id":"087d6028-ce69-4bf7-bf1d-e1f662b0f4e2","resolution":{"observed_at":"2026-08-01T19:33:03.502183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T17:47:06.272600Z","title":"Videophy-2: A challenging action-centric physical com- monsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17523","last_updated":"2026-07-20T03:56:43Z","snapshot_observed_at":"2026-08-07T23:44:26.873349Z","submitted_at":"2026-07-20T03:56:43Z","title":"Thinking in Video: Can Video Generators Really Reason About the Real World?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T17:47:06.272600Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.17523"},"observation_digest":"sha256:804c6b4dffaa4a43bb963cf69bc74d52ce96936961645ca971d6859105a388b4","observation_id":"5920ead6-c2f8-48fb-9f7d-3462801e86f6","resolution":{"observed_at":"2026-08-01T17:47:06.272600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T14:02:10.808123Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18924","last_updated":"2026-07-21T10:06:55Z","snapshot_observed_at":"2026-08-08T05:11:00.363961Z","submitted_at":"2026-07-21T10:06:55Z","title":"Learning Explicit Physical Parameter Control and Benchmarking for Video Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T14:02:10.808123Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.18924"},"observation_digest":"sha256:6f7bb2c7573b3aa70fd2d4173fecc3c22d4ccdf86614795d390102f72bd82004","observation_id":"a3f15e8b-0fbc-4760-a54c-db2a3b675253","resolution":{"observed_at":"2026-08-01T14:02:10.808123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T02:49:57.720734Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25321","last_updated":"2026-07-28T06:13:18Z","snapshot_observed_at":"2026-08-07T09:19:37.277038Z","submitted_at":"2026-07-28T06:13:18Z","title":"Physics-Grounded Fluid Video Generation with a Simulation Dataset and Dual-Stream Optical-Flow Supervision","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T02:49:57.720734Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.25321"},"observation_digest":"sha256:ceb78b21f40e06ad9a3edbfa3ccc84d9d57a58968b0373e48cc38489811ab360","observation_id":"a6a7e5c4-8726-45ce-a12b-a2e89635b934","resolution":{"observed_at":"2026-08-01T02:49:57.720734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T08:28:04.629823Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27380","last_updated":"2026-07-29T18:38:23Z","snapshot_observed_at":"2026-08-02T23:26:16.524563Z","submitted_at":"2026-07-29T18:38:23Z","title":"VideoCoCo: Code-as-CoT for Physically-Consistent Video Generation via an Agentic Dual-Engine System","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T08:28:04.629823Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.27380"},"observation_digest":"sha256:ad452bd5a5185107c7acd7022ba94942dccb56bb1809e7ca648381522b9a8c4f","observation_id":"73a0adac-4e87-49e6-b669-293c9e9c6aea","resolution":{"observed_at":"2026-08-01T08:28:04.629823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-07T20:36:25.525906Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05948","last_updated":"2026-08-06T12:19:38Z","snapshot_observed_at":"2026-08-09T04:35:23.386556Z","submitted_at":"2026-08-06T12:19:38Z","title":"GAUGE: A Measurement-Grounded Benchmark for Physical Fidelity in Simulation Engines and Video World Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T20:36:25.525906Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2608.05948"},"observation_digest":"sha256:067398531dcf9538cf4c5f783760d0ca1f951081433c55d00f618870581b0bf1","observation_id":"824e9a33-f5e3-4323-b33b-9064aae12a26","resolution":{"observed_at":"2026-08-07T20:36:25.525906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.06800/citation-record","integrity":"/paper/2503.06800/integrity","json":"/paper/2503.06800/citation-record.json","paper":"/paper/2503.06800"},"outbound":[],"paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T17:18:26.824890Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 53 inbound Pith citation observations for arXiv:2503.06800."}