{"as_of":"2026-08-08T14:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:55688c47ce49d6b92a28d2c5876ef9839f11287e57c3d4ac52967c4aaf2551d8","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:59:13.889276Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00891/citation-record","integrity":"/paper/2506.00891/integrity","json":"/paper/2506.00891/citation-record.json","paper":"/paper/2506.00891"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:17.228549Z","title":"Dual alignment unsupervised domain adaptation for video-text retrieval,","venue":null,"work_id":"4dc1af16-61be-44f3-ad69-a5361c35ab96","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.514757Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:ab83ac3f1c2562bfe1b1ab83f119313f153f3618f84440d5d0bc53c49ba5bd43","observation_id":"33c92fad-9261-4d56-bf0d-d082fe73b9ae","resolution":{"observed_at":"2026-08-07T11:59:17.273131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:17.094410Z","title":"Uncertainty-aware alignment network for cross-domain video-text retrieval,","venue":null,"work_id":"1d49577e-c1d3-4339-a995-279c126ec908","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.647809Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:80da072ee8c145c6d3b81ff0b553ba429eaa2b9197438652e232a789c9bc1efc","observation_id":"d5f6b0bd-f6da-4e68-bdfd-2a5e15af7dee","resolution":{"observed_at":"2026-08-07T11:59:17.140821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.947162Z","title":"Uatvr: Uncertainty-adaptive text-video retrieval,","venue":null,"work_id":"047000c4-d9c2-44bc-b272-87e9e2f96715","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.791831Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:bf7182cb8946bc5b34ceca3c548d4434a915f39012096fa4909a19346c5e5229","observation_id":"c0d04294-2bec-40df-8d99-81f1f7d8065a","resolution":{"observed_at":"2026-08-07T11:59:17.003651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.824756Z","title":"Unified coarse-to-fine alignment for video-text retrieval,","venue":null,"work_id":"8a517748-6fde-498e-8931-9188dcb79174","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.897250Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:9494237ba0fadcb237a17206e17614b975959847a73cbaaab8d20def1a709a64","observation_id":"064c0b14-3e82-4499-95d4-2c61e04c105b","resolution":{"observed_at":"2026-08-07T11:59:16.880777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.694867Z","title":"Text is mass: Modeling as stochastic embedding for text-video retrieval,","venue":null,"work_id":"2779c44e-4f73-4853-a031-d4f52e710111","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.017241Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:bf6f4d3546e423e13b0a1315a1c92f9785918b54b6aaae3f9e6497652d4a7965","observation_id":"faf0d322-6db3-4479-aed4-332607f4d5e0","resolution":{"observed_at":"2026-08-07T11:59:16.741366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.535933Z","title":"Dgl: Dynamic global-local prompt tuning for text-video retrieval,","venue":null,"work_id":"b2d4a8a7-edd6-40ca-a93b-3034cf2e9794","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.141183Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:eaffe86ae200fa8f66978cd67070c9fb2a0000a0ca32dda939b852bac1034a8d","observation_id":"05f0067b-d255-4df0-a344-77864ccc298c","resolution":{"observed_at":"2026-08-07T11:59:16.630962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12218","last_updated":"2023-05-20T15:48:47Z","snapshot_observed_at":"2026-07-06T15:30:05.820261Z","submitted_at":"2023-05-20T15:48:47Z","title":"Text-Video Retrieval with Disentangled Conceptualization and Set-to-Set Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12218","snapshot_observed_at":"2026-08-07T11:59:12.242279Z","title":"Text-video retrieval with disentangled conceptualization and set-to-set alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.242279Z"},"links":{"cited_paper":"/paper/2305.12218","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:6c8e74e1f935fafd348034ab8466cd97896cdd8e9a1e5b6d8c4738c87baabfa0","observation_id":"e309d534-57c2-4926-abbc-243c2eda77c7","resolution":{"observed_at":"2026-08-07T11:59:12.242279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.386619Z","title":"Multi-feature graph attention network for cross-modal video-text retrieval,","venue":null,"work_id":"10e25cac-2b91-4658-bb62-63109f3fee83","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.298946Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:06aca4ce95b3ea3faf87b4eaa31798381c2be67ad90601a460c9c911f7769d15","observation_id":"eebc9fa1-ac43-411b-bd52-8801a0cc2a57","resolution":{"observed_at":"2026-08-07T11:59:16.458283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.258588Z","title":"What matters: Attentive and relational feature aggregation network for video-text retrieval,","venue":null,"work_id":"f0e1cecd-93ce-40f3-87b4-93b296af548d","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.378562Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:838cbe91c6ccd70a9679450ab57db698f7df775d90364bb308a947da4bf1a40c","observation_id":"b75f6d79-b864-4502-9375-3ff3917229ab","resolution":{"observed_at":"2026-08-07T11:59:16.305160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.038081Z","title":"Partially relevant video retrieval,","venue":null,"work_id":"152ebeca-424a-4039-98db-d76645aa8089","year":2022},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.433909Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:7275d01ae5e6ed823c85650aa94827502f561eeccf86bf5b32d1260af31f33ee","observation_id":"151fd767-ce02-453d-87bd-f3f9fcb4e770","resolution":{"observed_at":"2026-08-07T11:59:16.151729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.13824","last_updated":"2024-05-22T16:55:31Z","snapshot_observed_at":"2026-08-07T23:08:11.817239Z","submitted_at":"2024-05-22T16:55:31Z","title":"GMMFormer v2: An Uncertainty-aware Framework for Partially Relevant Video Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.13824","snapshot_observed_at":"2026-08-07T11:59:12.507772Z","title":"Gmmformer v2: An uncertainty-aware framework for partially relevant video retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.507772Z"},"links":{"cited_paper":"/paper/2405.13824","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:d043b8a5ac8504ee93728dd41aae5f44997dad5593042dab85ea9cb6a198f4fa","observation_id":"7ee8147d-027e-4773-884b-a4288312a559","resolution":{"observed_at":"2026-08-07T11:59:12.507772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.939203Z","title":"Gmmformer: Gaussian-mixture- model based transformer for efficient partially relevant video retrieval,","venue":null,"work_id":"815f9c96-25b6-48e1-8cd2-eed3e8ba24e9","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.633308Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:a68ebd88c4f6fab70d90436856697dc9143b7f5437e1e9ddc9d73f378046480b","observation_id":"1a1dad76-254a-4336-9483-a13d861ed2a3","resolution":{"observed_at":"2026-08-07T11:59:15.982314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.767329Z","title":"Dual learning with dynamic knowledge distillation for partially relevant video retrieval,","venue":null,"work_id":"aa79eb66-f520-44c6-8d80-81d51960a944","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.766279Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:957493603238d724ad5ab64abbf4cf05e99954d49400dfbaab20bedf0b53b2d0","observation_id":"c9668524-38d6-4679-b120-fbb726d98092","resolution":{"observed_at":"2026-08-07T11:59:15.835647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02483","last_updated":"2025-02-12T09:39:06Z","snapshot_observed_at":"2026-07-06T19:10:12.721500Z","submitted_at":"2024-09-04T07:20:01Z","title":"TASAR: Transfer-based Attack on Skeletal Action Recognition","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02483","snapshot_observed_at":"2026-08-07T11:59:12.905371Z","title":"Tasar: Transfer-based attack on skeletal action recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.905371Z"},"links":{"cited_paper":"/paper/2409.02483","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:56f8b6405817d78a17b71e2b0d5ccb6d8cb892d362f52c7543c2ea111fdd5f26","observation_id":"bfc91068-4c81-4b87-a076-1cf354b49220","resolution":{"observed_at":"2026-08-07T11:59:12.905371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.616027Z","title":"Progressive event alignment network for partial relevant video retrieval,","venue":null,"work_id":"01641c12-3b71-46a3-b8cf-a5c6e0db6d50","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.970005Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:992f2942d259e41fe97f1c84fd9a539ef2dbb9610994c41efbaeffa34b42d8d3","observation_id":"401ed5bd-d328-46e0-afbc-448bc0f3596e","resolution":{"observed_at":"2026-08-07T11:59:15.707201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T11:59:13.027251Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.027251Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:51778e7569bd0c8ffb766f96920ace1e1902d51cc4dc28d0f00ffe3cda6eb07a","observation_id":"2fc4f736-9708-4578-9ce3-60917c7ecfe6","resolution":{"observed_at":"2026-08-07T11:59:13.027251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.05612","last_updated":"2018-07-29T19:11:57Z","snapshot_observed_at":"2026-07-06T05:51:36.145913Z","submitted_at":"2017-07-18T13:51:32Z","title":"VSE++: Improving Visual-Semantic Embeddings with Hard Negatives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.05612","snapshot_observed_at":"2026-08-07T11:59:13.112215Z","title":"Improv- ing visual-semantic embeddings with hard negatives,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.112215Z"},"links":{"cited_paper":"/paper/1707.05612","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:e95e6467ce2553e5a8761deea7f66d17fb22d4d16c692ddc598ff6daace49da5","observation_id":"f5f306d7-3da1-4bbe-a49b-0fc97179c124","resolution":{"observed_at":"2026-08-07T11:59:13.112215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.466142Z","title":"Video corpus moment retrieval with contrastive learning,","venue":null,"work_id":"86fb2afe-9fb4-46b6-9b89-f416db0a46cc","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.169725Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:8be9f6d2c747c85b174414fb92093519a14e40bfad60781feadbd41b251a30c5","observation_id":"72a3d49d-6119-4727-a2b8-d4b1a56d04ea","resolution":{"observed_at":"2026-08-07T11:59:15.555960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.342838Z","title":"Dense-captioning events in videos,","venue":null,"work_id":"a67fc9ba-f03b-4430-85fc-410809a5e6b4","year":2017},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.216066Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:8491cc4e87b713524bde089730a74d0b009e48ecf7a2cf1b15f3cf3c66883da8","observation_id":"55195957-2c3f-411c-9139-f85e6e67f6db","resolution":{"observed_at":"2026-08-07T11:59:15.390041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.211257Z","title":"Tvr: A large-scale dataset for video-subtitle moment retrieval,","venue":null,"work_id":"a1a88a2f-a495-439f-81b3-8b694f63bd8a","year":2020},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.312726Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:6a5ec9a83435158a41f97c36a81589ef4d89b65b8bde425d388f61a5060f29a6","observation_id":"98525be6-3e51-46cd-8b86-ef843f729faf","resolution":{"observed_at":"2026-08-07T11:59:15.258407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.074840Z","title":"Dual encoding for zero- example video retrieval,","venue":null,"work_id":"bf573752-2306-495b-9374-77500f3f04ca","year":2019},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.355804Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:a409de20dbb2eeb9af3d8d4cfd471ad565d6d914c0904295eb952bcbf708ae8d","observation_id":"54ebfa3a-0346-40ef-a76d-231b310ec87f","resolution":{"observed_at":"2026-08-07T11:59:15.143266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.889100Z","title":"W2vv++ fully deep learning for ad-hoc video search,","venue":null,"work_id":"89de93e9-70c0-43ed-bcd1-e93c24392a05","year":2019},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.415062Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:ddcdb39a259489f69ef91aa4e4c18546667c58d0856cff08a535a208aaa94472","observation_id":"b8e5ff83-9720-4658-933b-ef60c1bd3f0c","resolution":{"observed_at":"2026-08-07T11:59:14.976715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.717600Z","title":"Cap4video: What can auxiliary captions do for text-video retrieval?,","venue":null,"work_id":"bcd79012-e29d-4bb4-b8af-2310badeef44","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.535395Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:f135bed2e8be90ce9ea8b6bf9309a36dd9cbda4f9c410f2f30e83f5a61727ad1","observation_id":"27bab37a-96cb-461d-9c3d-794850efaa66","resolution":{"observed_at":"2026-08-07T11:59:14.818994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.617763Z","title":"Conquer: Contextual query-aware ranking for video corpus moment retrieval,","venue":null,"work_id":"aa54d8fc-a46e-4110-95dc-574f985342b3","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.629463Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:e2b1bad27252ca6e949010dc6414ac76e3f6616dfbde1d8c4ad1cfcd86c06ea0","observation_id":"2a27c75e-479d-4716-bbde-d8f7efee4bf6","resolution":{"observed_at":"2026-08-07T11:59:14.663358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.451201Z","title":"Listen and look: Multi-modal aggregation and co-attention network for video-audio retrieval,","venue":null,"work_id":"923f4f9e-5db0-4027-997a-26f1c27de76f","year":2022},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.723130Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:da83623b07e67a1e6e736e547ef9e392f7a7a8cc3bfd6459ff4f8b1ecbd27e3c","observation_id":"43946145-345f-4406-ae3d-f924195ddd79","resolution":{"observed_at":"2026-08-07T11:59:14.491209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.265480Z","title":"Mixgen: A new multi-modal data augmentation,","venue":null,"work_id":"77f0dbc8-a798-4735-9537-a11a177b7906","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.813900Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:aeb2cd2181afc923200e4f51a51476601f036380732a05162f67b7ff3efc6ca6","observation_id":"6a79fd9c-a0ac-4762-9197-85ad64bf90ed","resolution":{"observed_at":"2026-08-07T11:59:14.336523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.074927Z","title":"Mapfusion: A novel bev feature fusion network for multi-modal map construction,","venue":null,"work_id":"31db617c-d572-4ddb-b833-6bcb596b68a4","year":2025},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.889276Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:691a3a5d9f5d9a1c83a6318fc490e2e806e49b53049d372de7fea894feb15dc3","observation_id":"b62c90c7-fdcd-42c9-ad9a-9cb15e335668","resolution":{"observed_at":"2026-08-07T11:59:14.141393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":0,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2506.00891."}