{"as_of":"2026-08-21T19:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8d46ecb07cb6953fb15d4f86aa42cdace9e3266e6f80f888b16b9ebbcaac1c1d","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T22:41:45.860765Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.06663/citation-record","integrity":"/paper/2505.06663/integrity","json":"/paper/2505.06663/citation-record.json","paper":"/paper/2505.06663"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.358223Z","title":"Social fabric: Tubelet composi- tions for video relation detection","venue":null,"work_id":"8f25d0d7-80be-42be-8c6e-a13b4bb3b1a2","year":2021},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.721310Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:e5b971566237349b368a4af039cbc93705f9a0377db210f8f7aa3d8e00b7176b","observation_id":"9a44a258-fa6a-4988-9f77-f78ce830bde1","resolution":{"observed_at":"2026-08-15T22:41:46.362871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.328175Z","title":"Stacked hybrid-attention and group collaborative learning for unbiased scene graph generation","venue":null,"work_id":"4e6e0418-a27e-493e-a68d-5db3af082e99","year":2022},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.732086Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:4d69bd4b9c6dfd46bb47529a12efbae13685dca6af7ee0f1548bfd8caedae28f","observation_id":"4865361f-2454-4182-a8ab-0ef5dc6e863b","resolution":{"observed_at":"2026-08-15T22:41:46.332745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00268","last_updated":"2023-02-01T06:20:54Z","snapshot_observed_at":"2026-08-19T10:32:30.335079Z","submitted_at":"2023-02-01T06:20:54Z","title":"Compositional Prompt Tuning with Motion Cues for Open-vocabulary Video Relation Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.00268","snapshot_observed_at":"2026-08-15T22:41:45.746728Z","title":"Compositional prompt tuning with motion cues for open-vocabulary video rela- tion detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.746728Z"},"links":{"cited_paper":"/paper/2302.00268","citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:33fb7f918306a2f0df789677176c4475031195b844f19a9f739cc1c4bf1b2794","observation_id":"f700c645-efb0-4785-bd3e-34714c17aaae","resolution":{"observed_at":"2026-08-15T22:41:45.746728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.254589Z","title":"Align and prompt: Video-and-language pre-training with entity prompts","venue":null,"work_id":"822acd50-8d2c-43cb-80c6-07b5d7fad83c","year":2022},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.762209Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:6ae5d867cbeb2f94006fc50fac59592699d5dd238f7c3d3646e719a94c7ce637","observation_id":"e2b5a2b0-fa23-4e20-a875-732070209d66","resolution":{"observed_at":"2026-08-15T22:41:46.259711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.237966Z","title":"Blip-2: Bootstrapping language-image pre- training with frozen image encoders and large language models","venue":null,"work_id":"98f96b4a-27ac-43ac-81f9-e111c961a28e","year":2023},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.766775Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:8fdf9eea27fa28ca956f654b78a1be0e644df4e41ed8d34acb2610eb70e0d76c","observation_id":"25f9e06b-b62b-40b4-a7f6-cf4e9d0b1d7a","resolution":{"observed_at":"2026-08-15T22:41:46.242947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.222173Z","title":"Open-vocabulary se- mantic segmentation with mask-adapted clip","venue":null,"work_id":"6e16c593-9d8f-46a6-a2e1-0124b9386279","year":2023},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.771129Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:0a14fa06d1fa3172736799d7b37139942ed66bfd8f14b24619784e513180f82b","observation_id":"709f1e32-706a-4a16-b199-8d1e8c122a42","resolution":{"observed_at":"2026-08-15T22:41:46.227306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.205958Z","title":"Microsoft coco: Com- mon objects in context","venue":null,"work_id":"4c05b57e-e5dc-4fc6-87c9-d0fee9b47e6a","year":2014},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.775793Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:dfdd900b2e86497356f3c3943c9b02025fddac2262267bdca75bdcdb9459f429","observation_id":"f489c0d1-f809-4de0-844d-ba474d3a15b2","resolution":{"observed_at":"2026-08-15T22:41:46.211167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.172126Z","title":"Beyond short-term snip- pet: Video relation detection with spatio-temporal global context","venue":null,"work_id":"db08a8ed-17ce-4554-b368-b1b3c407c717","year":2020},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.784530Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:3e7f5faba88a536f9d5c8a3ec81ed4049ef6672d466e13bc9d062d1947dc9445","observation_id":"7a6c256d-2399-4728-a92c-0a3de49f7383","resolution":{"observed_at":"2026-08-15T22:41:46.177273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.142712Z","title":"Decoupled weight decay regularization","venue":null,"work_id":"d1b728a4-ee12-4dda-908d-76ef0ce1999a","year":2019},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.793440Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:05c270f1156477c9473f838eaea4b3765b293f0b5bb253891f67e914799d8964","observation_id":"eda83f8f-7f36-401f-87dd-b04ba1bff1f2","resolution":{"observed_at":"2026-08-15T22:41:46.147348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.111819Z","title":"Video rela- tion detection with spatio-temporal graph","venue":null,"work_id":"a93c8dda-ac44-4412-8d06-7beb027ec1a9","year":2019},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.802118Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:686f896e9f04d27b1fcd6a3bca6167b1d39265cf3506e0e1f7648116aa71b7b6","observation_id":"ed689957-151d-41e3-b2ba-2827023e3c6c","resolution":{"observed_at":"2026-08-15T22:41:46.116985Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.806431Z","title":"Learning transferable visual models from nat- ural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.806431Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:7a4fcbf88637b7de2d460428c9b09964f66d7311433531856582b371f4a33304","observation_id":"724552ac-edcd-44b9-995d-498e94d4fdb2","resolution":{"observed_at":"2026-08-15T22:41:45.806431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.086799Z","title":"Ac- tion scene graphs for long-form understanding of egocen- tric videos","venue":null,"work_id":"b218766f-27fe-46b8-910d-6859dab452f4","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.810849Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:deb489fe40e6bf368bf2155584357fda2d7f3019aeac135c01e8a5b5237ce83a","observation_id":"848d8766-596f-4860-b266-a2baafb88b47","resolution":{"observed_at":"2026-08-15T22:41:46.091473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.071707Z","title":"Video visual relation detection","venue":null,"work_id":"7242f205-492b-45c2-aec7-a831e551e735","year":2017},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.816130Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:4a87eac81d536c1de0cf8ed689ac20d673a59fe80c3d443d5d55828cd1b25460","observation_id":"d2de1639-c2ad-4b07-ad46-b216818c5398","resolution":{"observed_at":"2026-08-15T22:41:46.076451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.040208Z","title":"Video visual relation detection via iterative inference","venue":null,"work_id":"2880e19b-220e-41aa-a0b1-f64957a7a75d","year":2021},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.825138Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:068aa4f5086f72757a430015272a4f652436292424f69411174ffa93230b1344","observation_id":"7df242c1-d4c0-451b-84f0-99c6c7313c7f","resolution":{"observed_at":"2026-08-15T22:41:46.045060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.023731Z","title":"Video relationship reasoning using gated spatio- temporal energy graph","venue":null,"work_id":"31943a5f-2fff-41b1-ba8c-6a83960b1e08","year":2019},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.829618Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:8b321490ba9b56b10f55937785d36d8ee58a9160ae6394fcd20fb0433a492380","observation_id":"c69762b8-6d8c-446f-8880-33a6f93ea173","resolution":{"observed_at":"2026-08-15T22:41:46.029000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.08472","last_updated":"2021-09-17T11:21:34Z","snapshot_observed_at":"2026-08-18T03:30:52.103954Z","submitted_at":"2021-09-17T11:21:34Z","title":"ActionCLIP: A New Paradigm for Video Action Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.08472","snapshot_observed_at":"2026-08-15T22:41:45.834024Z","title":"Actionclip: A new paradigm for video action recognition","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.834024Z"},"links":{"cited_paper":"/paper/2109.08472","citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:ebee153afd30f1fa77172d47b9fee51ef1ce3f3f3875339f3d358fc97ac484a1","observation_id":"1991fd2c-6f0d-4f8f-9897-e1640307b889","resolution":{"observed_at":"2026-08-15T22:41:45.834024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.006183Z","title":"End-to-end open-vocabulary video vi- sual relationship detection using multi-modal prompting","venue":null,"work_id":"493376bc-bf41-45a6-b53e-9e887e2f5641","year":2025},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.838837Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:56eb44cb8484f8e7496d9afe315935a1febaca02a3eb8a740c3f3efe03dd1ee9","observation_id":"bf2e9519-5265-4148-b797-a4d83d40582c","resolution":{"observed_at":"2026-08-15T22:41:46.011496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.989395Z","title":"Meta spatio-temporal debiasing for video scene graph generation","venue":null,"work_id":"658081ab-cbd9-4ca8-804b-a2b44517a208","year":2022},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.843031Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:df3fb1dba60144954c713f2eec4d86b371c9b24e98df12910d0804f91f9b97c3","observation_id":"a3d0b1ad-6a10-4483-a461-3be0f514850e","resolution":{"observed_at":"2026-08-15T22:41:45.994364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.973731Z","title":"Multi-modal prompting for open- vocabulary video visual relationship detection","venue":null,"work_id":"51c4a204-764e-4626-bb95-e7dc3568cbb9","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.847417Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:c73a186d14c01f0923b051c60f3e5ff39a456a8b54f5931235c67b572cab8eb1","observation_id":"ae361d15-f1d6-4979-97f7-c221f6ae364e","resolution":{"observed_at":"2026-08-15T22:41:45.978792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.958119Z","title":"End-to-end video scene graph generation with temporal propagation transformer","venue":null,"work_id":"52e62fb3-c348-41ff-b847-f2c29ff1c76c","year":2023},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.851935Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:857cc1303d003b7b51f39074e20f14c58d51fb42149879e74b8e8f11fe56b8a5","observation_id":"9cb81df1-d2ed-46c6-b49e-e03a493e68ad","resolution":{"observed_at":"2026-08-15T22:41:45.963036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.940223Z","title":"Constructing holistic spatio-temporal scene graph for video semantic role labeling","venue":null,"work_id":"8e2971c9-6fdd-468f-b1bb-d35010ae24eb","year":2023},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.856480Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:1f0020764e33db115239161f1e504ea549b0e01a789cd01023571c636b08d694","observation_id":"ebf5f5b9-b3ea-45ab-9e22-fda78c4abd67","resolution":{"observed_at":"2026-08-15T22:41:45.946261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:45.922151Z","title":"Vrdformer: End-to-end video visual relation detec- tion with transformers","venue":null,"work_id":"3309c657-fe5c-4c32-84f4-930e42c2f3dd","year":2022},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.860765Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:f36036e7482188688390909d015c5a0b64cb77ab72036bb52f6e43cb5e548f53","observation_id":"8b36e4ea-9269-4f8a-9a35-353f3989b246","resolution":{"observed_at":"2026-08-15T22:41:45.929043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.269970Z","title":"Vrdone: One-stage video visual relation detection","venue":null,"work_id":"b9b46670-adcb-469b-919b-1b12c4c5210e","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2002,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.757440Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:fe0695f052759620e95b68a6a8f25aec88d1ac403e53dcb27d97729c35255e19","observation_id":"0aa98295-6574-4cd7-877f-f145672e1a54","resolution":{"observed_at":"2026-08-15T22:41:46.274458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.188914Z","title":"Td2-net: To- ward denoising and debiasing for video scene graph gener- ation","venue":null,"work_id":"cd44425b-7459-488b-bc90-c04cd932068f","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.780232Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:385d105bd791098d82a975cd7f4a8f1e4300012a7a257a444ed8790d96b62703","observation_id":"ea660bf8-62a9-490d-a8f6-ee0baef8ef68","resolution":{"observed_at":"2026-08-15T22:41:46.194323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.056239Z","title":"Relation understanding in videos: A grand challenge overview","venue":null,"work_id":"e08d4c50-2bc6-4612-84ec-62f96d4d1f53","year":2019},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.820571Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:fa0d8d6a36f6511d82a00a80e254113fa55c7e420f1c1f8958bce673e07283de","observation_id":"b00aaeaf-f18d-4761-a688-6366966ec79e","resolution":{"observed_at":"2026-08-15T22:41:46.061364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.127124Z","title":"Hig: Hierarchical interlacement graph approach to scene graph generation in video understand- ing","venue":null,"work_id":"c9e4d530-e64f-45b0-b236-6b787d5e9daf","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.797527Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:27f9d7c5cc69f0c1e2f895b07b214a4e2e18bf4a503c3667fdb66f8809a5e678","observation_id":"19bc3818-e824-4a61-875c-fca8a92bedcf","resolution":{"observed_at":"2026-08-15T22:41:46.131880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.157442Z","title":"Open-vocabulary segmenta- tion with semantic-assisted calibration","venue":null,"work_id":"6da84688-d39e-4ce5-af55-1cd10b6172f8","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.789103Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:e6d327425a3edcb801029479a502bac85424c602999ba5641b2116b25d9cd41d","observation_id":"30c5a471-8bc5-4fd6-9e33-70e234e22a04","resolution":{"observed_at":"2026-08-15T22:41:46.162125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.343259Z","title":"Spatial-temporal transformer for dynamic scene graph generation","venue":null,"work_id":"f810bc9a-e533-447b-a702-54397bc82aae","year":2021},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.727385Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:535be7dce241d50e7a786d86932473edf300e80f99220bdba550fe0f7d96ed77","observation_id":"07f8fa9b-dee5-4aff-bd20-a3b3a5f96e0a","resolution":{"observed_at":"2026-08-15T22:41:46.347981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.313538Z","title":"Active open-vocabulary recognition: Let intelligent moving mitigate clip limitations","venue":null,"work_id":"87aca4a5-2861-4fdd-9ce8-7d7ac0f9c0d8","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.736859Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:43175aa0712d612b23a0793ea16c53e5294efaad373489e64f2171cbbf5a33b2","observation_id":"a83ea7af-c81b-4f15-a635-16f5b66dd6b1","resolution":{"observed_at":"2026-08-15T22:41:46.318507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.284167Z","title":"Stochastic neighbor embedding","venue":null,"work_id":"2bf790a2-2954-4090-987e-2a071a6b5d29","year":2002},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.752680Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:6c834c733e8383b520a498dc56d45c4c0cfa722fab1d2eb0a6dbe9a07ecddff5","observation_id":"1c19a07e-4357-46b9-b8ee-88f0eefa85aa","resolution":{"observed_at":"2026-08-15T22:41:46.288603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:41:46.299213Z","title":"Simple image-level classification improves open-vocabulary object detection","venue":null,"work_id":"31a5adb4-adc6-433b-a469-c24c4a60ebfb","year":2024},"citing_paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T22:41:45.741825Z"},"links":{"citing_paper":"/paper/2505.06663"},"observation_digest":"sha256:236e9b9b7403600def42c211a8bb217e8e5be43719aa063f65036c2b6c515b8b","observation_id":"d9e257df-f28e-4b23-a638-af1d45369923","resolution":{"observed_at":"2026-08-15T22:41:46.304045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.06663","last_updated":"2025-05-10T14:45:43Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-21T11:16:27.730467Z","submitted_at":"2025-05-10T14:45:43Z","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":3,"verified_exact":0,"verified_fuzzy":28},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 0 inbound Pith citation observations for arXiv:2505.06663."}