{"as_of":"2026-08-09T13:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:426a3d85fb1a0359b053bbdce351a0ffa81870ce0b12253cabcb65b5b479a707","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":32,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T14:29:00.725533Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":95,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T22:41:43.411128Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2306.14289"},"observation_digest":"sha256:2256a6465f5890f0c216060194f19fb3d2daa24b8a377bbc9c130c034cfa53f4","observation_id":"59e406f9-dff0-4aa7-87ed-8e702d7c6fa3","resolution":{"observed_at":"2026-05-17T22:41:43.492900Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2410.04960","last_updated":"2026-06-04T11:59:03Z","snapshot_observed_at":"2026-08-02T12:51:51.839797Z","submitted_at":"2024-10-07T11:59:54Z","title":"On Efficient Variants of Segment Anything Model: A Survey","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T19:42:24.122342Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2410.04960"},"observation_digest":"sha256:bcde926d0073cdbaa15fc65e46ac80ca2ceec0513c99345e39171baa087a12a8","observation_id":"222e03e3-3644-4541-849b-0d4cb32b831b","resolution":{"observed_at":"2026-05-23T19:43:23.551644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2412.03077","last_updated":"2026-05-06T05:19:23Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T07:02:49Z","title":"RoDyGS: Robust Dynamic Gaussian Splatting for Casual Videos","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-23T07:58:54.015749Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2412.03077"},"observation_digest":"sha256:44b71e6ac62fe3f3ae98d85d71f7de6483a2ddbd0d78ca2c99280772eaaa7265","observation_id":"d2c6632d-85fc-494d-a12a-fb32a701b46e","resolution":{"observed_at":"2026-05-23T08:02:43.822894Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-08T14:29:00.725533Z","title":"Track anything: Segment anything meets videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06756","last_updated":"2025-03-16T10:12:23Z","snapshot_observed_at":"2026-08-09T04:04:20.740573Z","submitted_at":"2025-02-10T18:33:15Z","title":"SAMRefiner: Taming Segment Anything Model for Universal Mask Refinement","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-08T14:29:00.725533Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2502.06756"},"observation_digest":"sha256:f25d602ee1cfe33e36be79c2e8baa42c4572c48461d42a46a45c352952621b7b","observation_id":"914e81ab-71f6-44d6-a9ed-d965447b7a9e","resolution":{"observed_at":"2026-08-08T14:29:00.725533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T14:46:19.959554Z","title":"Track anything: Segment anything meets videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17576","last_updated":"2025-07-02T17:54:27Z","snapshot_observed_at":"2026-08-09T08:51:33.434545Z","submitted_at":"2025-05-23T07:35:55Z","title":"CU-Multi: A Dataset for Multi-Robot Data Association","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:46:19.959554Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2505.17576"},"observation_digest":"sha256:e4d4f3968cf3b6cd69bbb62e08805829bf45f23dbbd7217b895542fc3a09f322","observation_id":"5efc0986-52ca-47c7-b1d9-10c3274f858e","resolution":{"observed_at":"2026-08-07T14:46:19.959554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2505.23617","last_updated":"2026-05-10T04:23:42Z","snapshot_observed_at":"2026-08-03T01:40:18.134653Z","submitted_at":"2025-05-29T16:25:35Z","title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-19T12:54:31.765909Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2505.23617"},"observation_digest":"sha256:1e46765844b8c462e5fc95218899880e698ca4b51399050a977e9f40d6d4df6e","observation_id":"59f9617f-a5a4-4939-a621-9dcedf8a1970","resolution":{"observed_at":"2026-05-19T12:57:17.913143Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T11:52:49.056362Z","title":"Track anything: Segment anything meets videos.arXiv:2304.11968, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-07T11:42:31.580083Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.056362Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f5575042a3ca84e9328ae990d2b7f32d8d5becaa5fc2dd31bbbb99b7559dc955","observation_id":"9ac3aeba-a833-4a41-8e9e-70ccdfed6daa","resolution":{"observed_at":"2026-08-07T11:52:49.056362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T10:35:16.361815Z","title":"Track anything: Segment anything meets videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05011","last_updated":"2025-06-05T13:21:09Z","snapshot_observed_at":"2026-08-08T14:48:36.715235Z","submitted_at":"2025-06-05T13:21:09Z","title":"UAV4D: Dynamic Neural Rendering of Human-Centric UAV Imagery using Gaussian Splatting","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T10:35:16.361815Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.05011"},"observation_digest":"sha256:98d7e4dae2787bec7e391d5e55554216ad58c0d8c771140a26ee88cd60ec971d","observation_id":"340f00e3-ff2c-42a8-af1f-05f4ad41d936","resolution":{"observed_at":"2026-08-07T10:35:16.361815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T05:27:06.851144Z","title":"Track anything: Segment anything meets videos.arXiv preprint arXiv:2304.11968, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07996","last_updated":"2025-06-09T17:58:12Z","snapshot_observed_at":"2026-08-09T08:51:26.063777Z","submitted_at":"2025-06-09T17:58:12Z","title":"UA-Pose: Uncertainty-Aware 6D Object Pose Estimation and Online Object Completion with Partial References","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:06.851144Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.07996"},"observation_digest":"sha256:cbdbfe44e0b8e3e491e346363430c823df6c5aaca7bd959998604f883763d74e","observation_id":"3da692c5-cc8d-489c-a1d9-cee14765e42b","resolution":{"observed_at":"2026-08-07T05:27:06.851144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T23:47:57.411759Z","title":"Track anything: Segment anything meets videos,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16262","last_updated":"2025-06-23T13:06:37Z","snapshot_observed_at":"2026-08-08T23:10:28.452710Z","submitted_at":"2025-06-19T12:25:46Z","title":"R3eVision: A Survey on Robust Rendering, Restoration, and Enhancement for 3D Low-Level Vision","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T23:47:57.411759Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.16262"},"observation_digest":"sha256:49bdd06d1baf65320f4b1c8f891f01f3af0d1a896ccef776c7354dcd9b2b4f5b","observation_id":"a3034808-69f0-46d6-b451-7ef6a9d78fd2","resolution":{"observed_at":"2026-08-06T23:47:57.411759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T23:19:48.831105Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18792","last_updated":"2025-06-23T16:01:15Z","snapshot_observed_at":"2026-08-09T08:52:01.296242Z","submitted_at":"2025-06-23T16:01:15Z","title":"ViDAR: Video Diffusion-Aware 4D Reconstruction From Monocular Inputs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:19:48.831105Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.18792"},"observation_digest":"sha256:18ba69ae358736d30868d1fd2481238c8043a071345bb9bc0af0b942af1ecd14","observation_id":"5962e6ba-592a-4d4b-a0da-765b77fa12db","resolution":{"observed_at":"2026-08-06T23:19:48.831105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T19:54:37.920077Z","title":"Track Anything: Segment Anything Meets Videos,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04317","last_updated":"2025-07-06T09:53:31Z","snapshot_observed_at":"2026-08-07T09:40:27.476162Z","submitted_at":"2025-07-06T09:53:31Z","title":"CLIP-RL: Surgical Scene Segmentation Using Contrastive Language-Vision Pretraining & Reinforcement Learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:54:37.920077Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2507.04317"},"observation_digest":"sha256:d244b6a7655a5a10652d098d9a24c16d9dfa47f54e456da06869a66f53e2f028","observation_id":"68a3a034-4946-4916-8007-372c479b0fe2","resolution":{"observed_at":"2026-08-06T19:54:37.920077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T19:39:57.025806Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04959","last_updated":"2025-07-13T09:31:19Z","snapshot_observed_at":"2026-08-07T10:47:44.341783Z","submitted_at":"2025-07-07T13:01:50Z","title":"Hear-Your-Click: Interactive Object-Specific Video-to-Audio Generation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T19:39:57.025806Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2507.04959"},"observation_digest":"sha256:b9c6b202f13ac8c237d14a981be2bcbc7600ab70247303ea7a7b9c491a3e448f","observation_id":"c66b9629-eb4d-4b1e-b7c0-17d197ee9f7f","resolution":{"observed_at":"2026-08-06T19:39:57.025806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T17:58:46.955280Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09577","last_updated":"2025-07-22T15:04:11Z","snapshot_observed_at":"2026-08-09T08:52:33.625915Z","submitted_at":"2025-07-13T11:05:25Z","title":"Memory-Augmented SAM2 for Training-Free Surgical Video Segmentation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:58:46.955280Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2507.09577"},"observation_digest":"sha256:6b112c9d3aa9a58fde9e8656722a7c14915bc54705cc823b81f99fe8ff03c2e8","observation_id":"9543d116-36d2-4807-8b26-bcfeb0505aa5","resolution":{"observed_at":"2026-08-06T17:58:46.955280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T16:43:36.662024Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12763","last_updated":"2025-07-17T03:35:53Z","snapshot_observed_at":"2026-08-09T08:51:32.885548Z","submitted_at":"2025-07-17T03:35:53Z","title":"Continuous Marine Tracking via Autonomous UAV Handoff","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:36.662024Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2507.12763"},"observation_digest":"sha256:d8222c52c64d402ce9b1b31d27ed8940288e75e9d6a99b47935d85708ca98c73","observation_id":"33b1a0ff-8793-40ba-b7e6-137eb96c29a9","resolution":{"observed_at":"2026-08-06T16:43:36.662024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-06T00:51:41.282511Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04224","last_updated":"2025-08-06T09:00:13Z","snapshot_observed_at":"2026-08-09T08:52:01.402900Z","submitted_at":"2025-08-06T09:00:13Z","title":"SplitGaussian: Reconstructing Dynamic Scenes via Visual Geometry Decomposition","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T00:51:41.282511Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2508.04224"},"observation_digest":"sha256:1ac6f2213de0a9496941ee201daeef0a6e8616c71b7b3c1dee7ae6aa41a7fc3b","observation_id":"4d88af1e-0a19-425c-8225-e696cc2273ec","resolution":{"observed_at":"2026-08-06T00:51:41.282511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T22:22:05.096000Z","title":"Track anything: Segment anything meets videos.arXiv preprint arXiv:2304.11968, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.07182","last_updated":"2025-08-10T05:15:57Z","snapshot_observed_at":"2026-08-08T14:22:10.674904Z","submitted_at":"2025-08-10T05:15:57Z","title":"3D Gaussian Representations with Motion Trajectory Field for Dynamic Scene Reconstruction","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T22:22:05.096000Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2508.07182"},"observation_digest":"sha256:3c36a86533e3807e266fd03fae6bc8306268518e3ffcda2c5e471066ad6a53c3","observation_id":"be47a831-ff2e-4435-9969-65cef60567e7","resolution":{"observed_at":"2026-08-05T22:22:05.096000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T21:57:10.519532Z","title":"Track anything: Segment anything meets videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.07747","last_updated":"2025-08-11T08:27:57Z","snapshot_observed_at":"2026-08-05T21:56:40.030550Z","submitted_at":"2025-08-11T08:27:57Z","title":"Grouped Speculative Decoding for Autoregressive Image Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T21:57:10.519532Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2508.07747"},"observation_digest":"sha256:f06c88efe226cde4fd296de9629afca5a4fe36260f3ed9f49d2ddfeebffed0cc","observation_id":"e579c1b4-5e9d-4862-84a0-bcbba4f1976e","resolution":{"observed_at":"2026-08-05T21:57:10.519532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T14:01:18.011590Z","title":"Track anything: Segment anything meets videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21809","last_updated":"2025-08-29T17:43:58Z","snapshot_observed_at":"2026-08-09T01:44:48.919022Z","submitted_at":"2025-08-29T17:43:58Z","title":"VoCap: Video Object Captioning and Segmentation from Any Prompt","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-05T14:01:18.011590Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2508.21809"},"observation_digest":"sha256:629099ee2b4bd59e0d4ec3d71081c200e6871c7b2bcb941ae9562e6c0cdfd2bf","observation_id":"dc07d712-d243-49bb-9167-63d0ab8cde11","resolution":{"observed_at":"2026-08-05T14:01:18.011590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2510.14244","last_updated":"2026-05-12T20:27:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-16T02:55:04Z","title":"Reinforcement Learning for Unsupervised Domain Adaptation in Spatio-Temporal Echocardiography Segmentation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-18T06:53:21.438159Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2510.14244"},"observation_digest":"sha256:a48686938835e0e983067994395ed8c48d323c9e3c84cb48c1905c190555a443","observation_id":"6ebd9c47-9370-431a-b61d-da4d31740695","resolution":{"observed_at":"2026-05-18T06:56:01.690261Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2511.18264","last_updated":"2026-04-23T02:25:07Z","snapshot_observed_at":"2026-07-06T22:36:44.887709Z","submitted_at":"2025-11-23T03:26:57Z","title":"SatSAM2: Motion-Constrained Video Object Tracking in Satellite Imagery using Promptable SAM2 and Kalman Priors","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T06:10:24.636998Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2511.18264"},"observation_digest":"sha256:cd6a96aaf3f79bae125e756d88462df24b3479352e2472450544c6ce5a583198","observation_id":"2cd0e4bc-c030-468d-8127-afb1d449134e","resolution":{"observed_at":"2026-05-17T06:11:34.847331Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2512.17445","last_updated":"2026-04-08T21:08:41Z","snapshot_observed_at":"2026-07-06T22:39:33.919723Z","submitted_at":"2025-12-19T10:57:03Z","title":"LangDriveCTRL: Natural Language Controllable Driving Scene Editing with Multi-modal Agents","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-16T20:56:58.770875Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2512.17445"},"observation_digest":"sha256:a7414d51830aaff845772b960c07e70b678b00c41e93544261d3b3dcd9373a65","observation_id":"0a4b6bb3-d735-4366-98cd-29d83974c513","resolution":{"observed_at":"2026-05-16T20:58:31.849678Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2602.21668","last_updated":"2026-05-04T08:33:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-25T08:04:07Z","title":"Space-Time Forecasting of Dynamic Scenes with Motion-aware Gaussian Grouping","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T19:51:58.756605Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2602.21668"},"observation_digest":"sha256:eb365625c560af2fc19a8f35a84de4cce3e8666535ef459d498271bcacfdce3f","observation_id":"ed35b158-6198-4fd3-b124-f41879769591","resolution":{"observed_at":"2026-05-15T19:56:33.945749Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-07-13T18:24:58.428701Z","title":"arXiv:2304.11968 (2023) 4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.25168","last_updated":"2026-06-26T05:47:15Z","snapshot_observed_at":"2026-07-13T18:24:45.051987Z","submitted_at":"2026-03-26T08:37:32Z","title":"ET-SAM: Efficient Point Prompt Prediction in SAM for Unified Scene Text Detection and Layout Analysis","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-13T18:24:58.428701Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2603.25168"},"observation_digest":"sha256:94bd6dfaf3e43d8c6bd3ef0113078d0211a32dc37d33147a7678cec97b42ea7a","observation_id":"0067fb33-9d7f-4214-83d9-ac1afe7a2e3c","resolution":{"observed_at":"2026-07-13T18:24:58.428701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2604.05415","last_updated":"2026-04-07T04:19:39Z","snapshot_observed_at":"2026-07-06T22:54:08.897942Z","submitted_at":"2026-04-07T04:19:39Z","title":"Learning to Synergize Semantic and Geometric Priors for Limited-Data Wheat Disease Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T19:20:04.690220Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2604.05415"},"observation_digest":"sha256:5db493cf66276b7852d87068a597cfb652c8aff5a06d0c59cd195c03b2d9db3b","observation_id":"97ac1bc3-7562-46b6-a5b3-5d9aec6c6867","resolution":{"observed_at":"2026-05-10T23:05:51.506643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2604.07986","last_updated":"2026-04-09T08:55:39Z","snapshot_observed_at":"2026-07-06T22:57:09.435331Z","submitted_at":"2026-04-09T08:55:39Z","title":"DP-DeGauss: Dynamic Probabilistic Gaussian Decomposition for Egocentric 4D Scene Reconstruction","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T17:20:54.028846Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2604.07986"},"observation_digest":"sha256:733d21ee933b9422b86217760e50fecfefc3eed20245cb3758c28c79bb19def9","observation_id":"8d1f8330-4715-4b6e-8022-fd92d7f194f9","resolution":{"observed_at":"2026-05-11T07:01:00.941092Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2604.11170","last_updated":"2026-04-13T08:29:49Z","snapshot_observed_at":"2026-08-02T16:48:26.324351Z","submitted_at":"2026-04-13T08:29:49Z","title":"Do Instance Priors Help Weakly Supervised Semantic Segmentation?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:23:15.172519Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2604.11170"},"observation_digest":"sha256:0990135f8b762ce2e8ee0c3f4940e8c57ed0070b0b54f658019338dfcc3ba4db","observation_id":"09bf1170-c3c3-41d3-875f-f9003e323fbb","resolution":{"observed_at":"2026-05-11T08:56:03.251934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2604.23173","last_updated":"2026-04-25T06:55:13Z","snapshot_observed_at":"2026-08-06T21:26:24.889698Z","submitted_at":"2026-04-25T06:55:13Z","title":"One Identity, Many Roles: Multimodal Entity Coreference for Enhanced Video Situation Recognition","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-08T08:50:27.871886Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2604.23173"},"observation_digest":"sha256:e4410248aa5e4bd7fe457ea18f2ed726694d510e1b11baeacd0771b45b32c4a0","observation_id":"9f126d71-9aa3-43ad-b798-6358d6c3dcea","resolution":{"observed_at":"2026-05-11T20:31:11.879222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2605.09677","last_updated":"2026-05-10T17:51:05Z","snapshot_observed_at":"2026-07-06T23:21:44.900680Z","submitted_at":"2026-05-10T17:51:05Z","title":"VFM-SDM: A vision foundation model-based framework for training-free, marker-free, and calibration-free structural displacement measurement","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T04:09:34.131634Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2605.09677"},"observation_digest":"sha256:f12fc099fd159a72916654fce56c6b3b304a5d57609b6a1117228ba1e6afc62a","observation_id":"27a0bc2a-a6db-4545-9e7c-e500ce02400a","resolution":{"observed_at":"2026-05-12T04:11:20.977325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2605.09904","last_updated":"2026-05-12T03:09:23Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:47:59Z","title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T04:13:21.487431Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2605.09904"},"observation_digest":"sha256:ce863d39b158a380f96fd13310bc9429d40cb16da54cc64fb2c712f96fe42bf1","observation_id":"fe5c1906-bfee-4183-a40a-62fd8c924c85","resolution":{"observed_at":"2026-05-12T06:31:26.687252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2605.09904","last_updated":"2026-05-12T03:09:23Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:47:59Z","title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T06:53:42.726350Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2605.09904"},"observation_digest":"sha256:f2c436595cbb693838158f7206ce25e5ae41464aa8a731027fd05b934918bfb8","observation_id":"f6be40f8-8ff4-4b30-8f4d-dee55b3d97e6","resolution":{"observed_at":"2026-05-13T06:57:28.258081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":"2304.11968","doi":"10.48550/arxiv.2304.11968","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2304.11968 (2023)","venue":"arXiv (Cornell University)","work_id":"4a7361a1-31cd-4688-90bf-f97bbf6f25c7","year":2023},"citing_paper":{"arxiv_id":"2605.10106","last_updated":"2026-05-11T07:20:09Z","snapshot_observed_at":"2026-07-06T23:22:08.560919Z","submitted_at":"2026-05-11T07:20:09Z","title":"ViSRA: A Video-based Spatial Reasoning Agent for Multi-modal Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-12T04:00:23.681682Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2605.10106"},"observation_digest":"sha256:b914d15a4eb4d5053e7636fa269b871597fb02ccab9d7c6a155f742e8ada0228","observation_id":"f19ecb62-6f13-4fc1-b92c-ab6e4b28bd36","resolution":{"observed_at":"2026-05-12T06:46:36.587306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2304.11968/citation-record","integrity":"/paper/2304.11968/integrity","json":"/paper/2304.11968/citation-record.json","paper":"/paper/2304.11968"},"outbound":[],"paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 32 inbound Pith citation observations for arXiv:2304.11968."}