{"as_of":"2026-08-09T02:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ef3f8dc465d5233a9f1347d1e746c9db115873f2af23a0c5f023dbaa7ea72abe","coverage":[{"denominator":6,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T12:03:42.508698Z","state":"measured"},{"denominator":7,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":7,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T23:16:52.532992Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T23:16:52.629424Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"cited_work":{"arxiv_id":"2605.25195","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.25195","snapshot_observed_at":"2026-08-07T23:16:52.629424Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","venue":"cs.CV","work_id":"b100cb7f-1ae8-4527-8b95-6f471be49d09","year":2026},"citing_paper":{"arxiv_id":"2608.05803","last_updated":"2026-08-06T09:38:30Z","snapshot_observed_at":"2026-08-09T02:11:42.125051Z","submitted_at":"2026-08-06T09:38:30Z","title":"Vorch-Omni: Multi-Task Orchestration of Sight and Sound","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T23:16:52.532992Z"},"links":{"cited_paper":"/paper/2605.25195","citing_paper":"/paper/2608.05803"},"observation_digest":"sha256:2e2d7644b30592d6ac3c6f6f1129497df2a3feef9c51c85f1db9d76c2ca1b236","observation_id":"7d54ed40-72b2-4164-9ea7-829bc0695111","resolution":{"observed_at":"2026-08-07T23:16:52.635253Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.25195/citation-record","integrity":"/paper/2605.25195/integrity","json":"/paper/2605.25195/citation-record.json","paper":"/paper/2605.25195"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.952975Z","title":"We train the MLLM for 10 epochs, with a batch size of 1 per GPU","venue":null,"work_id":"4916896c-dc2a-48c3-bbec-8c3307abc582","year":2025},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:5d67d419dca7c05e980be0fa8a89b74aa55db97beab932e06ceca445b15727b5","observation_id":"e9cb6ad6-884d-4863-b434-47d72316c535","resolution":{"observed_at":"2026-07-09T07:26:03.954910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.955977Z","title":"to buy drinks, tip our tour guide, tip the guy who played the guitar for us. So, now we're we're leaving the fields and we're heading back into town to to get some money","venue":null,"work_id":"e9476b9f-247f-4b33-ac3e-fa6b383bf3f7","year":2025},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:5cc5ab4238489b9816f160dcb4a229367e7091113683bacc89f402d3f0103913","observation_id":"ed45e8d9-ff80-453a-96dd-423caf3c4917","resolution":{"observed_at":"2026-07-09T07:26:03.957894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.948924Z","title":"Whether the camera movement and framing match any cinematographic instructions (e.g., fixed shot, panning, close-up)","venue":null,"work_id":"98617236-b626-4496-b84d-4f1d900b8886","year":null},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:23affe7bf63499365a843895ca8e11f0c07bbfcd45fad1d9ab8f3f0d1a666291","observation_id":"ceb14431-8c4a-4e16-8485-b4f408b493f4","resolution":{"observed_at":"2026-07-09T07:26:03.951367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.961393Z","title":"Whether speech or dialogue has the correct spoken content, speaker identity, language, and tone","venue":null,"work_id":"e1f89dd6-ddbd-4195-b291-387bb5cdb022","year":null},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:93dbc01e98a4f23362fe88c017141e558d5218fda74e12c7b1c53251b208dbd6","observation_id":"5ddb60eb-162e-47e6-a79c-06142db4cff0","resolution":{"observed_at":"2026-07-09T07:26:03.963457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.964233Z","title":"Whether character-environment interactions are physically plausible (e.g., feet touching the ground while walking, hands making contact with surfaces during interactions)","venue":null,"work_id":"8ba92e8c-f00a-4124-89a5-2c8cfe8c86fd","year":null},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:ebf8bbf3f48d9c2b3bc2de863e80ec817d3368ae5fd2eb611d2b5d61da5598e0","observation_id":"304a76d0-46f4-4060-a01f-017ca709fcd9","resolution":{"observed_at":"2026-07-09T07:26:03.966969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:26:03.958767Z","title":"analysis","venue":null,"work_id":"4efda7a8-54e3-4d68-9256-e19d6b20eb1a","year":null},"citing_paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:42.508698Z"},"links":{"citing_paper":"/paper/2605.25195"},"observation_digest":"sha256:c1c49fdea95fb4fde5349657c20600ee7ae7856583669d2724333cf790523919","observation_id":"b4690b04-a1f5-400d-ae3f-af2efb662bcf","resolution":{"observed_at":"2026-07-09T07:26:03.960299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.25195","last_updated":"2026-06-01T01:54:28Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T06:55:53.214062Z","submitted_at":"2026-05-24T17:55:11Z","title":"Baton: Explicit Semantic Blueprints for Joint Video-Audio Generation"},"reference_resolution":{"displayed":6,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":6},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 6 of 6 outbound references and 1 inbound Pith citation observation for arXiv:2605.25195."}