{"as_of":"2026-08-16T13:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:71cac2ac1c5092c3df2026bcb1e394fbb7deeafe05eb858047ac5b5643830db6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T06:02:34.065885Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:09:40.790278Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-08-12T06:02:34.065885Z","title":"Video-star: Self-training enables video instruction tuning with any supervision.arXiv preprint arXiv:2407.06189, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00161","last_updated":"2025-03-30T14:31:41Z","snapshot_observed_at":"2026-08-16T04:20:19.899128Z","submitted_at":"2024-11-29T11:54:55Z","title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T06:02:34.065885Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2412.00161"},"observation_digest":"sha256:c11cb2ffa83a68955a49e92af76ccafc2e52a044a8986edeba46b983e3b793d5","observation_id":"950c5de4-7ac2-446d-abc0-7c45252c75f1","resolution":{"observed_at":"2026-08-12T06:02:34.065885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-08-11T16:11:10.831110Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-15T17:40:10.309874Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-11T16:11:10.831110Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2412.10360"},"observation_digest":"sha256:4407eacaa3a27a39df5b1b716dbb90cb6efe7a23417bc9930a41309c5b562a42","observation_id":"7bbb85ce-bd01-4a79-94a0-db5e973566a4","resolution":{"observed_at":"2026-08-11T16:11:10.831110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-08-08T15:05:21.947334Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:fdfa51bcefe48f88a671c9e6ae00cb394cba58eb2648e3e789c8a12250b7e159","observation_id":"3c592c7c-ab04-44c7-8713-b9e59e2c950a","resolution":{"observed_at":"2026-05-17T07:51:13.330222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-08-10T15:35:30.335661Z","title":"Video-star: Self-training enables video instruction tuning with any supervision.arXiv preprint arXiv:2407.06189, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13919","last_updated":"2025-09-01T07:50:58Z","snapshot_observed_at":"2026-08-15T21:13:34.833748Z","submitted_at":"2025-01-23T18:58:03Z","title":"Temporal Preference Optimization for Long-Form Video Understanding","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T15:35:30.335661Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2501.13919"},"observation_digest":"sha256:c91b9153e8c7747ee1be23fe77c35babbd1d0cfb1fee3dc051e7a3b8a75437b7","observation_id":"cbd74da5-b010-4ae6-9fc1-70794fdbb2eb","resolution":{"observed_at":"2026-08-10T15:35:30.335661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2504.05299","last_updated":"2025-04-07T17:58:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T17:58:57Z","title":"SmolVLM: Redefining small and efficient multimodal models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T20:23:50.552549Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2504.05299"},"observation_digest":"sha256:bc8c022d4f27d5256d5c0699c6759fcdd6eded737a2647a6d7e231e9e28840a0","observation_id":"e637b7cf-0310-4c32-940f-16401aff3cc1","resolution":{"observed_at":"2026-05-13T20:23:51.741475Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-08-03T19:04:29.148776Z","title":"Let’s think step by step","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.02456","last_updated":"2026-06-29T04:23:52Z","snapshot_observed_at":"2026-08-07T22:33:32.154553Z","submitted_at":"2025-12-02T06:30:10Z","title":"See, Think, Learn: A Self-Taught Multimodal Reasoner","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T19:04:29.148776Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2512.02456"},"observation_digest":"sha256:9d47e5fea64806b62270d8c92a6791891f5a3d663315d6e5363d52d7d0c12bd3","observation_id":"de93c1eb-8727-46ec-a8d0-58a2198a708d","resolution":{"observed_at":"2026-08-03T19:04:29.148776Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2604.26707","last_updated":"2026-04-29T14:15:45Z","snapshot_observed_at":"2026-08-10T23:29:27.702941Z","submitted_at":"2026-04-29T14:15:45Z","title":"CurEvo: Curriculum-Guided Self-Evolution for Video Understanding","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-05-07T11:52:26.828633Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2604.26707"},"observation_digest":"sha256:070cd4dc528c20d686a7d52f30d29c73de8504c32e3183a24ad60df32a1fd291","observation_id":"27b97f05-089c-498c-b445-df9a8fa132ef","resolution":{"observed_at":"2026-05-12T09:11:27.690880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2606.05259","last_updated":"2026-06-03T16:14:20Z","snapshot_observed_at":"2026-08-12T17:34:30.426257Z","submitted_at":"2026-06-03T16:14:20Z","title":"VideoKR: Towards Knowledge- and Reasoning-Intensive Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T07:00:21.192082Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2606.05259"},"observation_digest":"sha256:12ae5c6ae5653bc4aed1d2e5bb84bbd9680bbf70ee9abcbed892b4170112903b","observation_id":"082d1ef7-ff9d-4be5-88c6-a5803e6dd4ec","resolution":{"observed_at":"2026-07-02T07:26:45.455092Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2606.22158","last_updated":"2026-06-29T19:38:11Z","snapshot_observed_at":"2026-08-13T21:06:42.652461Z","submitted_at":"2026-06-20T17:33:07Z","title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-26T12:14:35.109298Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2606.22158"},"observation_digest":"sha256:d1b0985260d11ce615c21a5e18bcd5b1820af4236b0ba8d40208ca028ad3c811","observation_id":"8f2e2cf0-ce10-4037-a833-ae03c78aef61","resolution":{"observed_at":"2026-07-04T08:09:40.791756Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":"2407.06189","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-04T08:09:40.790278Z","title":"Video-star: Self-training enables video instruction tuning with any supervision","venue":null,"work_id":"eb92d10e-2aef-4f24-9096-cc5c7cb83089","year":2024},"citing_paper":{"arxiv_id":"2606.22158","last_updated":"2026-06-29T19:38:11Z","snapshot_observed_at":"2026-08-13T21:06:42.652461Z","submitted_at":"2026-06-20T17:33:07Z","title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-01T06:29:51.635039Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2606.22158"},"observation_digest":"sha256:a287595804595459e164577d0fbcd6c516ea6a6afb6eb64e1bdd76675b4b28d2","observation_id":"ffa5e41b-1a77-46a7-8fe8-57b3a5625bdb","resolution":{"observed_at":"2026-07-01T09:35:39.637541Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06189","snapshot_observed_at":"2026-07-13T00:53:20.749426Z","title":"Video-star: Self-training enables video instruction tuning with any supervision.arXiv preprint arXiv:2407.06189, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09029","last_updated":"2026-07-10T01:31:02Z","snapshot_observed_at":"2026-08-15T06:53:44.080367Z","submitted_at":"2026-07-10T01:31:02Z","title":"MOSAIC: Adaptive Inter-layer Composition for Efficient Heterogeneous Vision-Language Models","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-07-13T00:53:20.749426Z"},"links":{"cited_paper":"/paper/2407.06189","citing_paper":"/paper/2607.09029"},"observation_digest":"sha256:4cf6ace8fb891c7c830f3f5ccbe228389051c2fa79b55b4627cd18a2cd3d27a1","observation_id":"0e6c1eb6-3d0a-4e8f-8554-fd73f8a1d567","resolution":{"observed_at":"2026-07-13T00:53:20.749426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.06189/citation-record","integrity":"/paper/2407.06189/integrity","json":"/paper/2407.06189/citation-record.json","paper":"/paper/2407.06189"},"outbound":[],"paper":{"arxiv_id":"2407.06189","last_updated":"2024-07-08T17:59:42Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T02:33:37.224149Z","submitted_at":"2024-07-08T17:59:42Z","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2407.06189."}