{"as_of":"2026-08-08T15:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ee49f8e78f35835a756a73026b08d78d2979d5de13408b79722c4e202b43526","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:34:16.094906Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:20:05.840081Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-08-06T18:51:31.233075Z","title":"Towards physically plau- sible video generation via vlm planning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07202","last_updated":"2025-07-09T18:20:33Z","snapshot_observed_at":"2026-08-07T10:10:34.633901Z","submitted_at":"2025-07-09T18:20:33Z","title":"A Survey on Long-Video Storytelling Generation: Architectures, Consistency, and Cinematic Quality","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-06T18:51:31.233075Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2507.07202"},"observation_digest":"sha256:5cf99691a45835ada6dc4f9e12853d6f9967eddb8270fa7192a3ec3790f49e78","observation_id":"2a742e0c-8837-4f4f-8bb2-41edbf20e9c1","resolution":{"observed_at":"2026-08-06T18:51:31.233075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2509.24702","last_updated":"2026-04-06T10:12:03Z","snapshot_observed_at":"2026-08-02T11:34:44.120596Z","submitted_at":"2025-09-29T12:32:54Z","title":"Enhancing Physical Plausibility in Video Generation by Reasoning the Implausibility","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T12:55:42.679016Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2509.24702"},"observation_digest":"sha256:d05eba02489d1dc46f1e2d5689f8c15a4f934109fcf20c548a233b2054f718ab","observation_id":"84627d0a-877e-42ef-af92-e3f490eb5a14","resolution":{"observed_at":"2026-05-18T12:56:24.394265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2601.18577","last_updated":"2026-05-20T05:19:37Z","snapshot_observed_at":"2026-07-06T22:43:02.453697Z","submitted_at":"2026-01-26T15:22:27Z","title":"Self-Refining Video Sampling","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T14:37:57.167882Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2601.18577"},"observation_digest":"sha256:b22c06e35476b263ffc36106d6751e1eed0e38ebad3d1757b911ba5d63d77288","observation_id":"ff32deee-3011-4c9c-815f-bc563eef382b","resolution":{"observed_at":"2026-05-21T14:40:14.482113Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-15T13:27:51.848177Z","title":"Yu, S., Cho, J., Yadav, P., and Bansal, M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07109","last_updated":"2026-05-30T07:36:55Z","snapshot_observed_at":"2026-08-06T06:50:59.197329Z","submitted_at":"2026-03-07T08:40:09Z","title":"Vision Language Models Cannot Reason About Physical Transformation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-15T13:27:51.848177Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2603.07109"},"observation_digest":"sha256:35d8a33bad7af32faa4879c35915d50e873326648c828905616600ba0ed632da","observation_id":"b431a42f-d677-4464-a42b-c033990e3954","resolution":{"observed_at":"2026-07-15T13:27:51.848177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2604.06339","last_updated":"2026-04-07T18:17:05Z","snapshot_observed_at":"2026-07-06T22:54:53.308309Z","submitted_at":"2026-04-07T18:17:05Z","title":"Evolution of Video Generative Foundations","version":1},"reference_index":300,"source":"pdf_text","source_observed_at":"2026-05-10T18:41:38.616611Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2604.06339"},"observation_digest":"sha256:4d79cec93a752799fc99c80355a5497fd8aa46589cdec002b0c2284f6b6ea8d2","observation_id":"79bd86d6-9257-4fc5-bd67-4474dcee7d62","resolution":{"observed_at":"2026-05-11T00:05:51.719526Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2605.10564","last_updated":"2026-05-11T13:36:51Z","snapshot_observed_at":"2026-07-06T23:22:32.858399Z","submitted_at":"2026-05-11T13:36:51Z","title":"DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving","version":1},"reference_index":137,"source":"arxiv_source","source_observed_at":"2026-05-12T04:13:37.421188Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2605.10564"},"observation_digest":"sha256:69114b8037e056d75395ccd388964607ae0a10a38a63fdb5414bd9e406c01fb0","observation_id":"a083338f-2e36-480f-9393-1af2799bc958","resolution":{"observed_at":"2026-05-12T06:31:26.419010Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2606.00499","last_updated":"2026-05-30T03:13:00Z","snapshot_observed_at":"2026-07-06T23:41:10.825760Z","submitted_at":"2026-05-30T03:13:00Z","title":"OptiWorld: Optimal Control for Video World Generation under Physical Constraints","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T19:02:51.848742Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2606.00499"},"observation_digest":"sha256:aaba5c5c52fcda1589b7897b89d160ce2e37fd146f1d8b5b16208b585698922f","observation_id":"494e971f-48f1-4b80-bfb2-dec0392206a3","resolution":{"observed_at":"2026-06-28T19:32:35.539164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":"2503.23368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-07-04T19:20:05.840081Z","title":"Vlipp: Towards physically plausible video generation with vision and language informed physical prior.arXiv:2503.23368, 2025a","venue":null,"work_id":"f0875956-6feb-4c51-b4de-2612a76520cf","year":2025},"citing_paper":{"arxiv_id":"2606.25306","last_updated":"2026-06-24T02:12:54Z","snapshot_observed_at":"2026-07-06T23:59:47.542695Z","submitted_at":"2026-06-24T02:12:54Z","title":"Physics Question Scene Graph: Fine-grained Evaluation of Physical Plausibility in Text-to-Video Generation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-25T21:33:38.643889Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2606.25306"},"observation_digest":"sha256:81df95c8f07b12ec02d9767c4be3f9ca2cc8b9ee8b5b595e5e68ff89764f141a","observation_id":"8cbd19d9-eb58-466c-ac65-eeab1a9da6da","resolution":{"observed_at":"2026-07-04T19:20:05.841727Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23368","snapshot_observed_at":"2026-08-07T00:34:16.094906Z","title":"VLIPP: Towards physically plausible video generation with vision and language informed physical prior","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04412","last_updated":"2026-08-05T03:47:18Z","snapshot_observed_at":"2026-08-08T15:12:07.178020Z","submitted_at":"2026-08-05T03:47:18Z","title":"muSync-GS: Physics-Synchronized Driving Video Synthesis for Weather and Geometric Road Hazards","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T00:34:16.094906Z"},"links":{"cited_paper":"/paper/2503.23368","citing_paper":"/paper/2608.04412"},"observation_digest":"sha256:0eeb3757ba881dbe6826388640a3a62759c0e81744a5c87fa6ed19b7f16dee1a","observation_id":"32225ca1-f466-4642-acc8-ea40ea9fd754","resolution":{"observed_at":"2026-08-07T00:34:16.094906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.23368/citation-record","integrity":"/paper/2503.23368/integrity","json":"/paper/2503.23368/citation-record.json","paper":"/paper/2503.23368"},"outbound":[],"paper":{"arxiv_id":"2503.23368","last_updated":"2025-04-04T07:23:21Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T15:14:24.995272Z","submitted_at":"2025-03-30T09:03:09Z","title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:2503.23368."}