{"as_of":"2026-08-15T19:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b983a21ee3b4216190dbb02b9e41818e053f34caf000fdfea173409cef7cec11","coverage":[{"denominator":298,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T22:00:28.350003Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:41:18.289865Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-11T00:36:56.481588Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"cited_work":{"arxiv_id":"2606.07433","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.07433","snapshot_observed_at":"2026-08-11T00:36:56.481588Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","venue":"cs.CV","work_id":"f7b3b395-eec3-4c44-ab20-2210b941475b","year":2026},"citing_paper":{"arxiv_id":"2608.07585","last_updated":"2026-08-05T10:25:47Z","snapshot_observed_at":"2026-08-13T23:19:24.171336Z","submitted_at":"2026-08-05T10:25:47Z","title":"LAVE: Latent Visual Evidence-Enhanced Planning for Video Tool-use Agents","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T00:36:56.179636Z"},"links":{"cited_paper":"/paper/2606.07433","citing_paper":"/paper/2608.07585"},"observation_digest":"sha256:d3e38c517f6e65c7b5c1644cf6905186de300b32258e908e244438df9e05484f","observation_id":"2acb1284-9429-4c5d-be3c-714d5ec81a97","resolution":{"observed_at":"2026-08-11T00:36:56.487289Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.07433","snapshot_observed_at":"2026-08-12T13:41:18.289865Z","title":"Watch, remember, reason: Human-view video understanding with mllms.arXiv preprint arXiv:2606.07433,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10949","last_updated":"2026-08-11T14:19:03Z","snapshot_observed_at":"2026-08-15T10:15:30.314816Z","submitted_at":"2026-08-11T14:19:03Z","title":"StreamFlow: Dynamic Memory Flows for Streaming Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T13:41:18.289865Z"},"links":{"cited_paper":"/paper/2606.07433","citing_paper":"/paper/2608.10949"},"observation_digest":"sha256:fb2cb8a3135ee845e7ad6d2130e76215776052e8402dc3a326c56fe4d098ca76","observation_id":"d613dd20-b8ac-47c9-abcb-98a6b7bc44a4","resolution":{"observed_at":"2026-08-12T13:41:18.289865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2606.07433/citation-record","integrity":"/paper/2606.07433/integrity","json":"/paper/2606.07433/citation-record.json","paper":"/paper/2606.07433"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Qwen3.5: Towards native multimodal agents,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:5afcf993aa7efe70779b4bddf369246ad60c21a2e3c6ba4bad8a5d1b3dbc5470","observation_id":"470fec84-5101-4207-a561-20e5f4adaad3","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17765","last_updated":"2025-09-22T13:26:24Z","snapshot_observed_at":"2026-08-11T00:14:34.061687Z","submitted_at":"2025-09-22T13:26:24Z","title":"Qwen3-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2509.17765","doi":"10.48550/arxiv.2509.17765","metadata_source":"pith","pith_arxiv_id":"2509.17765","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-Omni Technical Report","venue":"cs.CL","work_id":"ae43e594-8bab-4471-b6af-92a300f6a048","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2509.17765","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:a28600fa0c52a0c772090f7f9f37275a5541f2fe5faca89fb51f98c0aa7c7192","observation_id":"b1a03ebf-7729-44c5-91ac-d82454cae18e","resolution":{"observed_at":"2026-07-02T17:27:15.631234Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Qwen3-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:88412f2dfca6f4830696e56389ddf8d702342f589f3071f3b53b2be207c5eaba","observation_id":"7c42e758-6134-4eae-a7bd-475225cd6e4f","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2503.20215","doi":"10.48550/arxiv.2503.20215","metadata_source":"pith","pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-Omni Technical Report","venue":"cs.CL","work_id":"438f105c-fa9b-44aa-ad52-43acb8045cda","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:317380219a78cce3a9526ac8b47e8711edd132581f1ef5abf6e1280dcbafb185","observation_id":"6410d80e-c6ff-4ba8-85ad-5d2a89bc9f61","resolution":{"observed_at":"2026-07-02T17:27:15.619936Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19225","last_updated":"2025-06-24T01:19:56Z","snapshot_observed_at":"2026-08-14T14:43:02.885779Z","submitted_at":"2025-06-24T01:19:56Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification","version":1},"cited_work":{"arxiv_id":"2506.19225","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19225","snapshot_observed_at":"2026-07-04T16:39:58.312479Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification.arXiv preprint arXiv:2506.19225, 2025","venue":null,"work_id":"93f911f1-9747-4144-9975-027df0ac6da7","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.19225","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:ef0c89f2f5bfce7e2a1786c66e93d9fa89e0a57ae0a9c474cf7b733563449c65","observation_id":"ae7501ad-991c-4e6e-8fac-cec4f55cf572","resolution":{"observed_at":"2026-07-02T17:27:15.628146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Msr-vtt: A large video descrip- tion dataset for bridging video and language,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:efcc894f1200803f897b8bd29cda5a28e35fe92ff27535e4533d1dbc7465c49a","observation_id":"dcf79bbe-8e66-4445-b164-6e4cb83da00c","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Video question answering via gradually refined attention over appearance and motion,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b3658b4a234db0bb881894ea9ace5787a2c2172c49dda55f723082dc0eeb9f66","observation_id":"04f357ea-a740-4cd0-8c2e-085458660645","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Tgif-qa: Toward spatio-temporal reasoning in visual question answering,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:52dd58dca9a2cf5ae96ce5eea09a62f2a26058158bfdeee2207589ca374b307c","observation_id":"70b39ce8-58f1-439d-9f5e-d2abb3c9de62","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"LongVU: Spatiotemporal adaptive compression for long video- language understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:aa5e74f440b7793bc946021e9c269ce68f99c88f212e2f8d848e5f30bb79134a","observation_id":"be9e0995-0aa6-46e3-a5a4-d021b08da945","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-11T11:51:37.712014Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2602.20913","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.20913","snapshot_observed_at":"2026-07-02T17:27:15.674592Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","venue":"cs.CV","work_id":"c2490cdf-7863-42f1-a614-523cde816d7f","year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2602.20913","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f27a3b86d563c05aa95123247a02b8dee4e42b7934e03b4e0d60ba15358d0c7e","observation_id":"677d658e-a194-4cff-8db7-3ed14280f2c0","resolution":{"observed_at":"2026-07-02T17:27:15.676404Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.23224","last_updated":"2026-05-21T09:05:36Z","snapshot_observed_at":"2026-08-02T12:29:31.448544Z","submitted_at":"2026-01-30T17:47:30Z","title":"Video-o3: Native Interleaved Clue Seeking for Long Video Multi-Hop Reasoning","version":2},"cited_work":{"arxiv_id":"2601.23224","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.23224","snapshot_observed_at":"2026-07-02T17:27:15.492915Z","title":"Video-o3: Native Interleaved Clue Seeking for Long Video Multi-Hop Reasoning","venue":"cs.CV","work_id":"d9291b6d-e10d-4ad4-b474-2cba910d9f4e","year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2601.23224","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:ca53531756ab1e62952c2dba690b1c2e292b5e5bc2acb8a610cc6f959c414576","observation_id":"b9cb58ef-339d-40ea-bf26-e2da5cad6ca0","resolution":{"observed_at":"2026-07-02T17:27:15.494462Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:da938c373e3905e3ffeb26867499f064e354e6cc3fc06500b943b614a1b15f04","observation_id":"85920590-edb0-4d63-8a45-ec6f1905523a","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"A visually grounded language model for fetal ultrasound understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:bee9934aad5d3a80c968b95ea7329f36e90c9645bb42052f21045d8b3bcb20cc","observation_id":"f94a8dbe-cb34-452b-bc54-d7321e746e94","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Streamingvlm: Real-time understanding for infinite video streams,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:00232ea0d31d024a636a68f5656ec25e6d55bdf751197ee52d9952c8bb3585d5","observation_id":"8713d225-e2c1-47b9-9fb7-02420a9ec188","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-08-13T20:05:08.806400Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:0e92a37b3c6028c524bce193d05ef263b86948ac90e8956455c5a71da2a8189f","observation_id":"99f19cce-399a-45ea-af49-7a991abf30fc","resolution":{"observed_at":"2026-07-02T17:27:15.504136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Adaptive keyframe sampling for long video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:1214c7788608f26eb4fb4cdcdf2d037500c6b9b08f6626d796ed0d8160d1370f","observation_id":"32b32525-956f-4587-8a61-1c3826126fd5","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"FrameFusion: Combining similarity and importance for video token reduction on large vision language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:75c9eeeeb56bc4e21a458bb954257eb134c05550304b93fc935e15452f8e8169","observation_id":"448d42e9-20d1-4e16-a889-747a316d02eb","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Moviechat: From dense token to sparse memory for long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b9e80c2b5146302256c4f20bcb2c3d84ce8ec7b3060662be998869dd8a16049e","observation_id":"70f7cc58-b488-4f44-92b4-c764110c54cb","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:8c30d8e3703dbabd4dc3263f6946aadb98ae5cbd16e4b772fdbe9642a2fcedfa","observation_id":"41a8fa6e-0ea2-424e-bad9-8af29e776310","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09149","last_updated":"2025-06-20T07:15:14Z","snapshot_observed_at":"2026-08-15T12:34:01.040304Z","submitted_at":"2025-03-12T08:23:32Z","title":"Memory-enhanced Retrieval Augmentation for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2503.09149","doi":"10.48550/arxiv.2503.09149","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09149","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Memory- enhanced retrieval augmentation for long video understand- ing","venue":"ArXiv.org","work_id":"8b5385f0-d090-497a-b8ee-95495baeddd3","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2503.09149","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:3e6bc62bb5a331c0c42980a472e26c46e96b315a7c2bc559c2e859f4572dccc4","observation_id":"2b2e50ed-d4b3-4c13-8886-42123c4e3bbe","resolution":{"observed_at":"2026-07-02T17:27:15.682433Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-13T01:15:12.087659Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:8c60d636f1df5120475d143472374cf19d5aa942e32499df8f5c2c8aa94ebda2","observation_id":"ace06fdb-019e-44c5-aecf-8cefd999604a","resolution":{"observed_at":"2026-07-02T17:27:15.497535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15717","last_updated":"2025-08-21T16:56:29Z","snapshot_observed_at":"2026-08-15T11:21:12.899320Z","submitted_at":"2025-08-21T16:56:29Z","title":"StreamMem: Query-Agnostic KV Cache Memory for Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2508.15717","doi":"10.48550/arxiv.2508.15717","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.15717","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streammem: Query-agnostic kv cache memory for stream- ing video understanding.arXiv preprint arXiv:2508.15717","venue":"ArXiv.org","work_id":"752bf839-d545-4378-a68d-4a958ba10cee","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2508.15717","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d5f1f4b75e2df03bd313c65e431e2fdfd918213cbf84164d062dbcd7b1751570","observation_id":"71c727b2-aebc-4c6f-ae89-7fa506eb5885","resolution":{"observed_at":"2026-07-02T17:27:15.536238Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":"2503.21776","doi":"10.48550/arxiv.2503.21776","metadata_source":"pith","pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","venue":"cs.CV","work_id":"0ce88332-564c-4361-8e2a-3850eb1ace9c","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:96c596a4364a2f2e19f0f97201c9896f88f32c3f612f1cc4e74e2982ed238cd8","observation_id":"cf307e9e-eb4a-4574-a123-362d95f657cc","resolution":{"observed_at":"2026-07-02T17:27:15.613385Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.12434","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:40:00.905697Z","title":"Videorft: Incentivizing video reasoning capability in mllms via reinforced fine-tuning","venue":null,"work_id":"89c249fe-60da-4724-bb0c-770442676a88","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:c4dfd666a5dc3cbe2600b924d3bd716ddcbae43260b90f9144f6cd8e5138be5e","observation_id":"86725e39-b5db-470b-94c2-6a5be27362f8","resolution":{"observed_at":"2026-07-02T17:27:15.657429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.20579","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.203105Z","title":"Open-o3 video: Grounded video reasoning with explicit spatio-temporal evidence","venue":null,"work_id":"fb25eed0-9409-44e6-af47-191ebb595de1","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:18fc4bad78dd14b5c69694bcd763b6df0c67cae5c18dde7e4bdd1f9cf92b8e3c","observation_id":"d9bc7633-e4fb-4873-8808-3c15b431dd7b","resolution":{"observed_at":"2026-07-02T17:27:15.796210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04416","last_updated":"2025-09-03T07:11:03Z","snapshot_observed_at":"2026-08-15T16:49:33.852456Z","submitted_at":"2025-08-06T13:03:21Z","title":"Thinking With Videos: Multimodal Tool-Augmented Reinforcement Learning for Long Video Reasoning","version":2},"cited_work":{"arxiv_id":"2508.04416","doi":"10.48550/arxiv.2508.04416","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.04416","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Thinking with videos: Multimodal tool-augmented reinforcement learning for long video reasoning","venue":"arXiv (Cornell University)","work_id":"824594ad-21b0-49b7-88f8-fac8553528e3","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2508.04416","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:060c621550be4f9706d318b8a7e60a10da20811bd7bafa6cb4b61943e4773311","observation_id":"26674536-e164-4883-88af-f7ed786e4eae","resolution":{"observed_at":"2026-07-02T17:27:15.802070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Video-language understanding: A survey from model architecture, model training, and data per- spectives,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b791b8c1aa2be625b9de0fdaecca20019592a728167d137a30486f73f97e65cd","observation_id":"534f1959-5ab9-4481-86a3-16af66200fc0","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Video understanding with large language models: A survey,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b27ae012d5702c1693f119b6a8c8e689a8063e69844d73b966dc21f69c5f0fcd","observation_id":"7256a394-d786-4b1a-97b0-e67e3f4961f0","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"A survey on video temporal grounding with multimodal large lan- guage model,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:445fc560564d6723d5f3be66483491d0c3d31a66751dd580db0bad95b945f5c8","observation_id":"49bafe87-e9b4-4b20-b944-ddd57ffc1cf8","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.05034","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.638400Z","title":"Video-lmm post-training: A deep dive into video reasoning with large multimodal models","venue":null,"work_id":"73d36644-47e7-4b81-87f2-f1d58d3ae40d","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:bb892ec190086d0463ce1edda06b0ece7940b624e56970646a1a44224ee5479c","observation_id":"ab7bb81b-feb6-4cb1-b6ad-8dd26adc9593","resolution":{"observed_at":"2026-07-02T17:27:15.640149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"cited_work":{"arxiv_id":"2509.08827","doi":"10.48550/arxiv.2509.08827","metadata_source":"pith","pith_arxiv_id":"2509.08827","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","venue":"cs.CL","work_id":"7618c14b-e527-4268-9926-d4f462ea9925","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2509.08827","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:8249a4437da65620defb274355c9c2e3719afec79f606c8b566faf798f42a0d7","observation_id":"d4c59d4a-3ef9-4e6d-8198-32ccac65e378","resolution":{"observed_at":"2026-07-02T17:27:15.500849Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13564","last_updated":"2026-01-13T09:33:57Z","snapshot_observed_at":"2026-08-13T14:23:02.090859Z","submitted_at":"2025-12-15T17:22:34Z","title":"Memory in the Age of AI Agents","version":2},"cited_work":{"arxiv_id":"2512.13564","doi":"10.48550/arxiv.2512.13564","metadata_source":"pith","pith_arxiv_id":"2512.13564","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Memory in the Age of AI Agents","venue":"cs.CL","work_id":"1ff75a14-302e-4906-aa86-1f96fbcf12ff","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2512.13564","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:a72bd5182b925f73a9254af90ba6cd7efc36ab0acc42d36cc335c9d88f6e21bd","observation_id":"65f00e7b-dba5-4b01-9f86-3ff893e99903","resolution":{"observed_at":"2026-07-02T17:37:14.031096Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:38:06.749638+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T06:38:06.749638+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.18227","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.677725Z","title":"arXiv preprint arXiv:2505.18227 , year=","venue":null,"work_id":"13b6e579-06ad-4c5c-bd41-97a41270caf2","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:538b863b331f0031b3f34869cf616a3d7e001330b6f5d5421a4a6982ddbd399a","observation_id":"ed615186-123c-4b51-a335-d212b8a0ef71","resolution":{"observed_at":"2026-07-02T17:27:15.679384Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.04921","last_updated":"2025-07-06T14:40:27Z","snapshot_observed_at":"2026-08-14T05:24:48.290030Z","submitted_at":"2025-05-08T03:35:23Z","title":"Perception, Reason, Think, and Plan: A Survey on Large Multimodal Reasoning Models","version":2},"cited_work":{"arxiv_id":"2505.04921","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.04921","snapshot_observed_at":"2026-07-08T02:44:28.049569Z","title":"Perception, reason, think, and plan: A survey on large multimodal reasoning models","venue":"cs.CV","work_id":"18ce98be-3cf0-4e03-b05f-ad32a6306ca2","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2505.04921","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:572f89cbc8721aeb22c5532e7a57f4adf5699604a793e783c5bbabcfae3770c1","observation_id":"3db0dc68-3a88-4d5a-b518-32cb56fcfb3b","resolution":{"observed_at":"2026-07-02T17:37:14.037520Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Lita: Language instructed temporal- localization assistant,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:422863741c60586666d4f9928af6b85f799578c2f86594907becaed25c53d7b8","observation_id":"d6210a4c-4b8d-4fb4-97d3-46ea3dde4d8b","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Universal video temporal grounding with generative multi-modal large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f7118ffa37ce359403c11b47e4b9a008db74a9512d35be4337d82d83656b5e66","observation_id":"e096ad9f-002b-4963-94d1-a56e9187972d","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.14698","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:17:29.020428Z","title":"arXiv preprint arXiv:2512.14698 , year=","venue":null,"work_id":"7e995230-85aa-4b56-84dd-314b7ee46b29","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:ad6809e2517f872f78a462948cd32eadbc68591357ed3829795bcc29fcdbf8ac","observation_id":"3d94bdf2-c212-44cd-803b-289b88d55d3f","resolution":{"observed_at":"2026-07-02T17:37:14.030570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Towards one-to-many temporal grounding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:9d5b9ff13ef8f9327c219fcf497a1a4b3e5adf36ba1bc03748b144ae064a3427","observation_id":"6774cf53-d140-4d87-9dfa-02fc6a1bdd31","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-15T03:59:34.260351Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":"2501.04001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-07-10T03:06:43.598138Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","venue":"cs.CV","work_id":"fbffda10-106c-432c-b84e-5d0908666921","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:4d9ad3b97414b4094780360f76bdeff624b8b71bc7374059652f537798aca61b","observation_id":"9d88011c-296d-4909-bf1d-4faf9afd5e13","resolution":{"observed_at":"2026-07-02T17:37:14.041253Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Sama: Towards multi-turn referential grounded video chat with large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d601e35c750852b9ff02fdf8733bc043681d1b1d1047a80ae783e8647d80c0e9","observation_id":"2f565c03-1547-4ca0-a376-663912c12253","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Streaming dense video captioning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:46630cc3068bba75f24ae020ca5adeff9b539c5bb9c42adbdd8dadbfda5a8ee9","observation_id":"e8188500-5fc2-45dc-bdef-1a93e263223d","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Do you remember? dense video captioning with cross-modal memory retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f1f0217a15e928e50bf5c0617361767104b7cdf887f5a9573310a9072260f556","observation_id":"f46ae04d-e8d2-4d1b-bc2f-fa8ddec060a7","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Dibs: Enhancing dense video captioning with unlabeled videos via pseudo boundary en- richment and online refinement,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:18a5241bc3b6e437edc652d6589de8d527d459798d69f3ee0e01ecd3f919bb22","observation_id":"3e7d640b-f394-4356-885a-76ad781b6cae","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-08-13T20:40:43.794560Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":"2404.16994","doi":"10.48550/arxiv.2404.16994","metadata_source":"pith","pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","venue":"cs.CV","work_id":"8949d4db-20f2-47c1-83a6-fcbe041b62ef","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:06c49246a01432c3d4410c4d1dafea4891d14d512292de0e37eaea24e0928a30","observation_id":"1db51307-ad73-434f-9901-01772a1abe2d","resolution":{"observed_at":"2026-07-02T17:27:15.625038Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03051","last_updated":"2025-04-09T06:24:14Z","snapshot_observed_at":"2026-08-14T03:26:31.108831Z","submitted_at":"2024-10-04T00:13:54Z","title":"AuroraCap: Efficient, Performant Video Detailed Captioning and a New Benchmark","version":4},"cited_work":{"arxiv_id":"2410.03051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03051","snapshot_observed_at":"2026-07-04T15:09:54.928668Z","title":"Auroracap: Efficient, performant video detailed captioning and a new benchmark","venue":null,"work_id":"3c75ba15-49f8-4ea3-87a8-6c357f825176","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2410.03051","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b2e290cf99977c693e17436fad8fb7b0b41ae92fb632fc45ce02db7a7678eb85","observation_id":"00fc8319-181c-4837-ad2f-97b1f9e2570d","resolution":{"observed_at":"2026-07-02T17:27:15.575211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07888","last_updated":"2025-01-24T05:16:36Z","snapshot_observed_at":"2026-08-10T20:27:34.844428Z","submitted_at":"2025-01-14T06:54:39Z","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","version":3},"cited_work":{"arxiv_id":"2501.07888","doi":"10.48550/arxiv.2501.07888","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07888","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tarsier2: Advancing large vision-language models from detailed video description to comprehensive video understanding","venue":"ArXiv.org","work_id":"77605c65-9b44-4521-ad8c-79d9c78cf33a","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2501.07888","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:035201b1dbee0c46d52f620f075f222056601bffcf76a129349b255b5ed40be8","observation_id":"d344e5c1-beb9-4a5f-a9ab-a1c298b9f080","resolution":{"observed_at":"2026-07-02T17:27:15.563714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.08565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.518202Z","title":"Baichuan-omni technical report","venue":null,"work_id":"fda3d528-4c1c-43ea-9da0-fa19c174c861","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:0e03cbc3d738c43819fc8f8036e0094592ecc6177acb68b325c87120433ac169","observation_id":"a3e788fe-5b9e-482f-a1bd-9f06988f7cb3","resolution":{"observed_at":"2026-07-02T17:27:15.519812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09344","last_updated":"2025-06-11T02:50:49Z","snapshot_observed_at":"2026-08-08T17:53:50.843512Z","submitted_at":"2025-06-11T02:50:49Z","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","version":1},"cited_work":{"arxiv_id":"2506.09344","doi":"10.48550/arxiv.2506.09344","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09344","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ming-omni: A unified multimodal model for perception and generation","venue":"ArXiv.org","work_id":"eba4dae2-6087-4ecf-ba38-9ab784c1b9a7","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.09344","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:6a4cd402a7579659e5b7db137aad708016696d208d80cbafbfd5031c086cdc1d","observation_id":"02327fc4-30ea-44bc-9b36-2beebe1677fc","resolution":{"observed_at":"2026-07-02T17:37:14.025507Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-08-15T18:25:40.547533Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":"2409.06666","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-07-04T08:29:41.322265Z","title":"Llama-omni: Seamless speech interaction with large language models","venue":null,"work_id":"130f0aad-80b2-40cc-a118-ad7dd036cb72","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d803a027b3684634a05cca95732734e45045b06752e6e5a4a4cdd76e654e9e1e","observation_id":"eb3d44dd-5b67-4797-96c8-e90f5dc44120","resolution":{"observed_at":"2026-07-02T17:27:15.662681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13642","last_updated":"2025-06-22T07:56:58Z","snapshot_observed_at":"2026-08-15T16:30:56.628505Z","submitted_at":"2025-06-16T16:06:45Z","title":"Stream-Omni: Simultaneous Multimodal Interactions with Large Language-Vision-Speech Model","version":2},"cited_work":{"arxiv_id":"2506.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13642","snapshot_observed_at":"2026-07-03T16:38:39.602552Z","title":"Stream-omni: Simultaneous multimodal in- teractions with large language-vision-speech model","venue":null,"work_id":"c01077b2-51f3-43a8-bc1b-a7c40d57abf6","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.13642","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:778e9bdb28657113035a0da887c552aa106f676b0a4bf60c4b5b6ca7a15b27cd","observation_id":"b3a8cf48-e086-43d3-bc8e-60ddc7414b0b","resolution":{"observed_at":"2026-07-02T17:27:15.634543Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07089","last_updated":"2025-06-02T09:38:26Z","snapshot_observed_at":"2026-08-07T21:01:40.617151Z","submitted_at":"2025-04-09T17:58:58Z","title":"OmniCaptioner: One Captioner to Rule Them All","version":3},"cited_work":{"arxiv_id":"2504.07089","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07089","snapshot_observed_at":"2026-07-11T03:17:51.543739Z","title":"Omnicaptioner: One captioner to rule them all","venue":"cs.CV","work_id":"97453bde-de3a-45c0-8f6b-5c08fe830211","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2504.07089","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:45d3799c19e36a7e1021271bc6d5f91d63e49e05c68ab7d93ab890eeaecd257b","observation_id":"f4e8a51d-2404-4008-af5e-42bb65278d08","resolution":{"observed_at":"2026-07-02T17:27:15.021888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.15870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.263774Z","title":"Omnivinci: Enhancing architecture and data for omni-modal understanding llm","venue":null,"work_id":"6179c0bb-4911-4be4-89c8-a387d39c8459","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:76a6cf2b9fdeafc381d0da1fdb470ba52889a67022c13f55b11d2f5ea3991167","observation_id":"ec4f2d40-b768-4f6c-99bc-dd8fe6a0e163","resolution":{"observed_at":"2026-07-02T17:27:15.027098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.22139","last_updated":"2025-07-22T07:42:31Z","snapshot_observed_at":"2026-08-09T21:52:46.816951Z","submitted_at":"2025-06-27T11:30:51Z","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","version":3},"cited_work":{"arxiv_id":"2506.22139","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.22139","snapshot_observed_at":"2026-07-04T16:09:57.149192Z","title":"Q-frame: Query-aware frame selection and multi-resolution adaptation for video-llms","venue":null,"work_id":"2b209583-6ef6-4e41-96ea-7ec74820dd14","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.22139","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:844670388b18a4e769da0481d11e7f4adc7e61e02b57211f0fbf0581c743ace1","observation_id":"de4ae604-e237-4b7a-9565-2c18234cc729","resolution":{"observed_at":"2026-07-02T17:27:15.057718Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"DyCoke: Dynamic compression of tokens for fast video large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:9868fb4ed003bbdbd6c18b83e84226d8f4a128f3638416bb0e7b90901478e8a8","observation_id":"c035664f-04b6-46ee-8fde-057eea9f1ce2","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.02295","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T01:19:20.250607Z","title":"Streamingvlm: Real-time understanding for infinite video streams","venue":null,"work_id":"83fa51a9-b7fe-4d72-b9cd-ef6727cf4b01","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:1ca915a385d71a12bb97cce9c1c46a65b5b9ae9afb754f24477b50fdcf42c696","observation_id":"009b482a-7cd2-4c0a-a22e-50cb3ec90cd4","resolution":{"observed_at":"2026-07-02T17:27:15.041754Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Vtimellm: Empower llm to grasp video moments,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:2c5bc79dac8338b4093c6641b48078e8a410c6e057b4de8aa84f4f7883bdcafa","observation_id":"25a0b264-e534-4b6f-9a11-ec6ff6e28042","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Vtg-llm: Integrating timestamp knowledge into video llms for enhanced video temporal grounding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:9d3ed27843567013ceea9d06003ffcb818b397b9b8ba73e217f1fd37b46c3f0f","observation_id":"122cca4f-cfcf-49b6-b085-ecbf659a35cb","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Distime: Distribution-based time representation for video large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:960227d6a0ba9af8e263c488d679c47953936d0fde5a3c730008ab749556a37c","observation_id":"55f91030-75e7-4d02-9ae3-047ddde04ebb","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Self-chained image- language model for video localization and question answering,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:580693a38953dabfac8df6c4f7854bb30a1e49cfdd79cfbeb5b23c6886b70af2","observation_id":"732cb7e7-3395-41d0-b089-50e7272270c8","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14505","last_updated":"2024-11-21T09:34:23Z","snapshot_observed_at":"2026-08-14T07:55:35.415892Z","submitted_at":"2024-11-21T09:34:23Z","title":"LLaVA-MR: Large Language-and-Vision Assistant for Video Moment Retrieval","version":1},"cited_work":{"arxiv_id":"2411.14505","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14505","snapshot_observed_at":"2026-07-02T17:27:14.988080Z","title":"Llava-mr: Large language-and-vision assistant for video moment retrieval,","venue":null,"work_id":"0e50912d-ef0a-40b0-bffc-61df72e7c85f","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2411.14505","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:1e4b0fec965e81e8bc535b81dac81cfea09f50fdb116e215fad379c95eb74a03","observation_id":"0b2b1842-8324-47dd-9e39-7e9d9215ef89","resolution":{"observed_at":"2026-07-02T17:27:14.990641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:500211947185b664faeb4ef0564e22aa9c917f75550b64d8fcaa23052628d2ed","observation_id":"84a2c07b-a05d-4151-8dfd-a752d5421d95","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Scanning only once: An end-to-end framework for fast temporal grounding in long videos,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:999d28d264fda4a3a9a96d7d83c656c700913ed0439139e6ba35e7c43b078b72","observation_id":"99336aa3-e415-44dc-8026-07c0c700cff4","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Trace: Temporal grounding video llm via causal event modeling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:4adaa6d9590507d5cfe2784248f0efa75015f67eb84cabf4698ee0aff79009ef","observation_id":"f7e3701d-ac02-4841-86b0-48089a54eb63","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.07683","last_updated":"2026-06-29T02:15:03Z","snapshot_observed_at":"2026-08-13T11:58:33.169597Z","submitted_at":"2025-08-11T06:59:32Z","title":"TAR: Temporal Anchor-Constrained Reasoning for Video Temporal Grounding","version":2},"cited_work":{"arxiv_id":"2508.07683","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.07683","snapshot_observed_at":"2026-07-02T17:27:14.974912Z","title":"Tar-tvg: Enhancing vlms with timestamp anchor-constrained reasoning for temporal video grounding","venue":"cs.CV","work_id":"8392709e-ece1-4e93-a81c-3f6b3b4a5fc6","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2508.07683","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:0ef9be58763612dd74402dd01b7e822c8cfc8475bd720f59b6645a60b9263819","observation_id":"c8c49ad2-98fd-4b20-b646-437a84e9f354","resolution":{"observed_at":"2026-07-02T17:27:14.976786Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03290","last_updated":"2025-08-21T05:15:19Z","snapshot_observed_at":"2026-08-12T22:30:54.111905Z","submitted_at":"2024-10-04T10:04:37Z","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2410.03290","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03290","snapshot_observed_at":"2026-07-04T15:09:55.249321Z","title":"Grounded-videollm: Sharpening fine-grained temporal grounding in video large language models","venue":null,"work_id":"6cda7c97-9a94-4872-b214-cddce67fea61","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2410.03290","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f0d225f190162d31b93ae7a21e3a24f0c2036c0a18edf8059ec4f768a17a2f53","observation_id":"43060d12-bce0-4242-89ee-08bd37db8ef6","resolution":{"observed_at":"2026-07-02T17:27:14.973292Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Momentor: Advancing video large language model with fine-grained temporal reasoning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:5dc8b64e0f07ecb4c3906f28fdc88fded94e6f4bf76eeb58bb495732009556e3","observation_id":"bf592336-9039-4686-9448-fc6a9d60a463","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.18823","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:14.981477Z","title":"Luowei Zhou, Chenliang Xu, and Jason J","venue":null,"work_id":"53b55a92-b592-401f-8f79-5a12151e3826","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d3dd3086df2c251af2e243b59a6ed0bdca6fc0343f4f6270ef017c67e41e455b","observation_id":"22782f23-b055-40fe-a3fd-480129dee614","resolution":{"observed_at":"2026-07-02T17:27:14.983423Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13377","last_updated":"2025-06-29T08:11:35Z","snapshot_observed_at":"2026-08-14T02:35:09.273544Z","submitted_at":"2025-03-17T17:04:20Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","version":3},"cited_work":{"arxiv_id":"2503.13377","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.13377","snapshot_observed_at":"2026-07-04T09:49:44.696023Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","venue":"cs.CV","work_id":"ef2c21b6-ae25-436a-bac3-f8d625541320","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2503.13377","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:0eb0b1d43156c759abc2490a51c50cd3569e006c811a2dd3cf4bbe7b4cea9bf9","observation_id":"cf68d152-da0c-4486-9e6e-bc9893b58257","resolution":{"observed_at":"2026-07-02T17:27:15.836506Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02994","last_updated":"2026-06-02T08:33:06Z","snapshot_observed_at":"2026-08-14T12:19:00.856841Z","submitted_at":"2026-02-03T02:05:48Z","title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","version":3},"cited_work":{"arxiv_id":"2602.02994","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.02994","snapshot_observed_at":"2026-07-04T19:20:07.207750Z","title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","venue":"cs.CV","work_id":"137c077f-c9c8-4d5a-8955-1ad1e988d656","year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2602.02994","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:003e0a2c444155d8e768930cff94cdd0f67061149eeb827b48c47eed15d56086","observation_id":"34ae96de-f14d-4c2c-a5ae-ac121e656b05","resolution":{"observed_at":"2026-07-02T17:27:15.826646Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.22315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.830378Z","title":"Videozoomer: Reinforcement-learned temporal focusing for long video reasoning","venue":null,"work_id":"ed057419-0436-4abb-bdc4-60bf5755a879","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:c9a34e32ba3ba26ad49b90f605de19499845f47e327e935d5a70f60277c2e4c4","observation_id":"1c4b1f02-8d40-4948-9db3-8f3101b27eee","resolution":{"observed_at":"2026-07-02T17:27:15.831821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Datasets and recipes for video temporal grounding via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:437f4e3f0f8db55244c32d202b5e6994fdea0ed66a4da56c516dcbc8f4554ebc","observation_id":"d83e18d7-2dd2-4760-ab08-85c1c7832d6b","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20715","last_updated":"2026-04-18T02:55:33Z","snapshot_observed_at":"2026-08-15T07:57:21.988562Z","submitted_at":"2025-05-27T04:50:07Z","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","version":2},"cited_work":{"arxiv_id":"2505.20715","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20715","snapshot_observed_at":"2026-07-02T17:27:15.832913Z","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","venue":"cs.CV","work_id":"35af37e5-9956-4ce9-be10-932e2342d463","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2505.20715","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:501bc3b7f57d50d4af1d6bba47852ee4c4cd2567a35103285afb36a6591d5943","observation_id":"0cba4df5-6108-4d84-b38a-8c1e04efdca5","resolution":{"observed_at":"2026-07-02T17:27:15.834103Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.12798","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T05:27:40.003735Z","title":"arXiv preprint arXiv:2510.12798 (2025)","venue":null,"work_id":"18a07d47-4390-4092-9958-8b1fab5c7aad","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:ef812ab915c4fe099bbc94960f83417420c62e80c53620ac119df18f079e9d2a","observation_id":"d24a5721-fb42-4ae0-a17f-db1a69665c82","resolution":{"observed_at":"2026-07-02T17:27:15.819395Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21375","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.224424Z","title":"arXiv preprint arXiv:2511.21375 (2025)","venue":null,"work_id":"1a1f6a40-fdc8-4cc4-afcf-af374eb098ec","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:78e9dfbfc69424de13e5fbafaab015a8cb23d02b9f6705358ae947203e888f86","observation_id":"9f873f25-c3dc-4730-a153-b7eca279b160","resolution":{"observed_at":"2026-07-02T17:27:15.821898Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Universal instance perception as object discovery and retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:3982b8393d4efaf5f856c6e681ec44ac86e0ecb90d00a364ab2aee1589f7bad5","observation_id":"bc8ac887-80ac-4566-b640-d4e7ba88adf3","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00265","last_updated":"2025-08-05T11:42:44Z","snapshot_observed_at":"2026-08-08T00:36:11.607815Z","submitted_at":"2025-08-01T02:14:00Z","title":"Multimodal Referring Segmentation: A Survey","version":2},"cited_work":{"arxiv_id":"2508.00265","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00265","snapshot_observed_at":"2026-07-02T17:27:15.815532Z","title":"Multimodal referring segmentation: A survey","venue":null,"work_id":"d324c202-01bb-44ba-a653-c67a8629d7c5","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2508.00265","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d971d5eb6723ebec30d1929c110b69521031441b0d54bdddb3a2dbd53d0df2af","observation_id":"38760e07-9fad-4d9c-9ad7-1b5480c8b3c8","resolution":{"observed_at":"2026-07-02T17:27:15.816944Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04159","last_updated":"2021-03-18T03:14:26Z","snapshot_observed_at":"2026-08-13T17:54:01.724774Z","submitted_at":"2020-10-08T17:59:21Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":"2010.04159","doi":"10.48550/arxiv.2010.04159","metadata_source":"pith","pith_arxiv_id":"2010.04159","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","venue":"cs.CV","work_id":"876f9fe8-c712-4550-9c26-cc18ab69abf2","year":2020},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2010.04159","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d07b9d5dee40d782283136b5b946ba5057c6d078c37d06fb0143c23af90a9e47","observation_id":"a6671a1c-94cd-4ff9-8395-227a8aacc08a","resolution":{"observed_at":"2026-07-02T17:27:15.843435Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Lisa: Reasoning segmentation via large language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:d8a574f6b3944babd89e0cb1f06c7747b3b9e878e427d70d2cfb1521e8c60956","observation_id":"8f74da62-f36e-4bda-9e73-0b8dfd4e9c90","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Omg-llava: Bridging image-level, object-level, pixel-level reasoning and understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:372060a834188cabae044d3cf0f1aca94c03a678b799467d9c936fd4ddc9276d","observation_id":"e9037922-680a-4b5e-9933-fda8ab7a1e73","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02555","last_updated":"2026-06-03T14:06:58Z","snapshot_observed_at":"2026-08-13T04:27:17.921109Z","submitted_at":"2024-02-04T16:06:05Z","title":"High-Quality Entity Segmentation and Grounding","version":2},"cited_work":{"arxiv_id":"2402.02555","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.02555","snapshot_observed_at":"2026-07-02T17:27:15.787607Z","title":"Generalizable entity grounding via assistance of large language model","venue":"cs.CV","work_id":"342358fa-0416-4366-b09e-00d7a899a241","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2402.02555","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:53f714ca4d76396e16a26fc010c637227c8935b27ae7039a198f77d2ea56b677","observation_id":"979b6460-4c29-4ef7-be79-b4af886f6a9f","resolution":{"observed_at":"2026-07-02T17:27:15.788784Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":"2408.00714","doi":"10.1038/s41598-025-97590-3","metadata_source":"pith","pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SAM 2: Segment Anything in Images and Videos","venue":"cs.CV","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:7b3016ad18e1651bcb06f53fa04632733b4efc3f232707e705e75b8cc4818a58","observation_id":"17791f4d-f18e-4477-91de-7bb3273a8790","resolution":{"observed_at":"2026-07-02T17:27:15.838832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Unipixel: Unified object referring and segmentation for pixel- level visual reasoning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:475a09632f1ffdda47e9ad96ccaa2b8b20f9db1a52f61bb0bf900b2c19165da2","observation_id":"1a18f4a7-0e7f-4b1d-bee8-ae5063d4a4fb","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.16093","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T00:49:18.003081Z","title":"Samtok: Representing any mask with two words.arXiv preprint arXiv:2601.16093, 2026","venue":null,"work_id":"93b1de9d-7f38-470d-abe4-ddc391f40cdd","year":2026},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:05576d58c25bcd2b115f27198fdb92c2af8ccd89d2f8bbfb2308231f2eaa5b6f","observation_id":"89f988ab-d5e2-40f5-9bf2-850cd4b214bc","resolution":{"observed_at":"2026-07-02T17:37:14.036900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Collecting highly parallel data for paraphrase evaluation,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f5273b3514e8448133b1071db3524d64dce9b608df1c520c3c501ba01bb4b439","observation_id":"d572f914-b166-4af9-b5ea-d87b7587650a","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Video-llava: Learning united visual representation by alignment before projection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:06beda09cf9f4dc58fad4cd7b3c9f63ccf2b9c9d159d343c045808be8cf683dd","observation_id":"e50f0776-31b5-4c2f-afa7-736e19cab602","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:1436bfa16b146cb36e92af645fc35512159f47d5aa5f32c78bcd1d46b40afd7d","observation_id":"325f0750-8865-48cf-b56b-997cb7b9819e","resolution":{"observed_at":"2026-07-02T17:27:15.561094Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Video recap: Recursive captioning of hour-long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:cbe1e4a9bad0fe3cf0643aa5046f72d7c517b78e91acc3be8f67c980534cb97b","observation_id":"1c250328-5078-438c-b812-bd495f678941","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15393","last_updated":"2025-03-01T02:06:59Z","snapshot_observed_at":"2026-08-11T16:06:29.545521Z","submitted_at":"2025-02-21T11:40:23Z","title":"LongCaptioning: Unlocking the Power of Long Video Caption Generation in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":"2502.15393","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15393","snapshot_observed_at":"2026-07-02T17:27:15.761589Z","title":"Longcaptioning: Unlocking the power of long video caption generation in large multimodal models,","venue":null,"work_id":"a9c81210-8e79-43ef-99b5-c0dae6e179ae","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2502.15393","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:c239dd582352bf37dc2b865caa2f8254af690c51603e9c4483d988eade30bf03","observation_id":"74bbc708-9bc2-4185-a15a-3c23914329cf","resolution":{"observed_at":"2026-07-02T17:27:15.763039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16427","last_updated":"2025-07-07T04:25:02Z","snapshot_observed_at":"2026-08-07T17:55:45.818931Z","submitted_at":"2025-02-23T03:59:05Z","title":"Fine-Grained Captioning of Long Videos through Scene Graph Consolidation","version":2},"cited_work":{"arxiv_id":"2502.16427","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.16427","snapshot_observed_at":"2026-07-04T19:50:11.312748Z","title":"arXiv preprint arXiv:2502.16427 (2025)","venue":null,"work_id":"3de2db06-6b7f-4478-832e-2fb48583789e","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2502.16427","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:370a95b778b090511b4c5fb6eeb1dd703bb9e63b378674a8676653f7dae52a77","observation_id":"e9841d58-d9f5-4f29-93ec-bd548fc09fd3","resolution":{"observed_at":"2026-07-02T17:27:15.765503Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00634","last_updated":"2024-09-24T04:41:08Z","snapshot_observed_at":"2026-08-15T02:19:40.902712Z","submitted_at":"2024-06-30T09:21:01Z","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","version":2},"cited_work":{"arxiv_id":"2407.00634","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.00634","snapshot_observed_at":"2026-07-04T17:40:00.906584Z","title":"Tarsier: Recipes for training and evaluating large video description models","venue":null,"work_id":"bf38f75f-fbc4-4c6d-ab2a-6e1bcbc38485","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2407.00634","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:dca06219ea8235e0c9d8f5dd27d83bad1c21b7fde6f159d09b1c3093e96a9c04","observation_id":"6909c904-fb97-4a95-aee1-5bbac09e37f0","resolution":{"observed_at":"2026-07-02T17:27:15.770571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.281790Z","title":"video-SALMONN 2: Caption-enhanced audio-visual large language models","venue":null,"work_id":"a6dea8e4-6736-44e5-b9d0-6e036ce7a7c0","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:49d32d18c1d3bd74528d0be856ab653fbf90bc76ecc9e2ddb14a81a089f24389","observation_id":"577b781c-615a-4add-b747-514c1e9853a1","resolution":{"observed_at":"2026-07-02T17:27:15.539022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01725","last_updated":"2025-06-02T14:30:09Z","snapshot_observed_at":"2026-08-15T09:52:26.987289Z","submitted_at":"2025-06-02T14:30:09Z","title":"VideoCap-R1: Enhancing MLLMs for Video Captioning via Structured Thinking","version":1},"cited_work":{"arxiv_id":"2506.01725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01725","snapshot_observed_at":"2026-07-04T17:40:00.934274Z","title":"Videocap-r1: Enhancing mllms for video captioning via structured thinking","venue":null,"work_id":"cad9c21e-f29f-40a6-affe-ab87ea23cffe","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.01725","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:fa09f6c5e80aeb0affb6feebada98d72b69f4d2a1e56cb94b202f856980972d3","observation_id":"c8439cfd-3612-4e2a-a1f4-fcd335241947","resolution":{"observed_at":"2026-07-02T17:27:15.755390Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18634","last_updated":"2025-08-27T03:53:24Z","snapshot_observed_at":"2026-08-15T16:52:31.092379Z","submitted_at":"2025-08-26T03:18:34Z","title":"OwlCap: Harmonizing Motion-Detail for Video Captioning via HMD-270K and Caption Set Equivalence Reward","version":2},"cited_work":{"arxiv_id":"2508.18634","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.18634","snapshot_observed_at":"2026-07-03T16:38:39.639869Z","title":"describe the subject, background, mo- tion, and camera in detail","venue":null,"work_id":"aa7808c8-b0d6-456e-8680-7e5b982823f8","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2508.18634","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:04982e2477264a573888195ef6fac2962f2eb4e7dde93cb8842fc88e5e3ae62a","observation_id":"a29d4349-76e2-4025-9e38-e08a8d8b8536","resolution":{"observed_at":"2026-07-02T17:27:15.748175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Towards fine-grained human motion video captioning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:b5e99fd38eae83a2cd6cbb9de0d9b430f79f30244d64dcf48d018cc45b0dd050","observation_id":"1d92c611-654f-45e6-ac0f-5eeb48ac7f38","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Sharegpt4video: Improving video understanding and generation with better captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:571e398b7863f9ab026b47904ed5c4ee020353e4a08231cd0cd9eb0098214adc","observation_id":"29488250-0ac9-421f-936e-ed56697a307b","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Panda- 70m: Captioning 70m videos with multiple cross-modality teach- ers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:4c0f924600c08a97b19b2fb3246795d57bda8171c5f555f0a40cbf1286f82ad5","observation_id":"92c72d79-9b4a-4015-9f16-f2bf4b205e71","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Vript: A video is worth thousands of words,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:24c1931ff7e5efbc7a4f0a77c1c39a89916b9aa91482aa9f94361bda047257b4","observation_id":"0be33b41-f6ea-4f99-add3-b7a71a23750c","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.18726","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.579606Z","title":"IF-VidCap: Can video caption models follow instructions?","venue":null,"work_id":"23d3e217-5dd7-461d-bb05-225f15caf638","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:a24b6925956e183cf759c6326a711ca14b96df49121274c95f36dd40f89832cf","observation_id":"695a5423-fe49-4fbe-8a57-6e6b5c895561","resolution":{"observed_at":"2026-07-02T17:27:15.581109Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.12841","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.744095Z","title":"Any- cap project: A unified framework, dataset, and benchmark for controllable omni-modal caption- ing","venue":null,"work_id":"8164eb44-88d5-4036-a422-ed622cfd31a2","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:6c6b35935ea5f1aaa129f12c5c5665dbb215569b1f7903c41fb7ec2c7620f97b","observation_id":"280879c3-079f-47ae-bd7f-b70571d57359","resolution":{"observed_at":"2026-07-02T17:27:15.745659Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:00:28.350003Z","title":"Intentvcnet: Bridging spatio-temporal gaps for intention-oriented controllable video captioning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:f40cd60233e8fe8975534821c545136b0001a63b8547ae40abe490d5f8cf6210","observation_id":"af362dc5-c511-4ee3-b77c-0495ce0eebac","resolution":{"observed_at":"2026-06-27T22:00:28.350003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":8,"parse_uncertain":0,"unresolved":45,"verified_exact":47,"verified_fuzzy":0},"total_outbound_references":298},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 100 of 298 outbound references and 2 inbound Pith citation observations for arXiv:2606.07433."}