{"as_of":"2026-08-14T08:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a0a2d081f48cdb61f5d62a4f87773db05985add97b5996e3a32dad70433b249a","coverage":[{"denominator":132,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:27:16.884770Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T04:45:38.441181Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T06:00:58.843605Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.08753","snapshot_observed_at":"2026-08-11T04:45:38.441181Z","title":"Which viewpoint shows it best? language for weakly supervising view selection in multi- view videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18386","last_updated":"2025-04-22T13:23:34Z","snapshot_observed_at":"2026-08-12T23:13:41.357483Z","submitted_at":"2024-12-24T12:16:43Z","title":"Switch-a-View: View Selection Learned from Unlabeled In-the-wild Videos","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T04:45:38.441181Z"},"links":{"cited_paper":"/paper/2411.08753","citing_paper":"/paper/2412.18386"},"observation_digest":"sha256:f0213939fc110ec8050620f7f9fbc389f62942b7e0f75bc15e500f690c0d708a","observation_id":"6cd707ae-c6d6-4b32-b3a1-eec071956336","resolution":{"observed_at":"2026-08-11T04:45:38.441181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"cited_work":{"arxiv_id":"2411.08753","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.08753","snapshot_observed_at":"2026-08-07T06:00:58.843605Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","venue":"cs.CV","work_id":"ff735e17-5ea6-4d37-ba76-90bb159da45b","year":2024},"citing_paper":{"arxiv_id":"2506.06253","last_updated":"2025-06-06T17:25:48Z","snapshot_observed_at":"2026-08-09T07:19:17.693738Z","submitted_at":"2025-06-06T17:25:48Z","title":"Bridging Perspectives: A Survey on Cross-view Collaborative Intelligence with Egocentric-Exocentric Vision","version":1},"reference_index":187,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:58.555825Z"},"links":{"cited_paper":"/paper/2411.08753","citing_paper":"/paper/2506.06253"},"observation_digest":"sha256:434640818de0c776988fd9ef0ee86348d97fd6e337590c5824eec5cf9d84fc99","observation_id":"ebe94ece-1e15-49a0-bcc6-e24f2cfb1f29","resolution":{"observed_at":"2026-08-07T06:00:58.847019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.08753/citation-record","integrity":"/paper/2411.08753/integrity","json":"/paper/2411.08753/citation-record.json","paper":"/paper/2411.08753"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1803.08375","last_updated":"2026-04-14T12:21:53Z","snapshot_observed_at":"2026-08-14T01:11:07.622125Z","submitted_at":"2018-03-22T14:30:17Z","title":"Deep Learning using Rectified Linear Units (ReLU)","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.08375","snapshot_observed_at":"2026-08-12T21:27:16.239636Z","title":"Deep learning using rectified linear units (relu), 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.239636Z"},"links":{"cited_paper":"/paper/1803.08375","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a5ac7ea01f61df1da134db392d01bd1aefdf4067cf8dc5da27e1b0c298bc3ffe","observation_id":"bb025d11-7abc-45fa-941e-be55fa65712e","resolution":{"observed_at":"2026-08-12T21:27:16.239636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.246965Z","title":"McCrae, Kenton Murray, Maria Nadejde, Satoshi Nakamura, Matteo Negri, Ha Nguyen, Jan Niehues, Xing Niu, Atul Kr","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.246965Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c4889203d8f2cfaf9b4de09bfa749972c0d75c09a0c39953362ac7f2919fd7f6","observation_id":"bbbd5838-3012-4a00-bed1-68e5f1a12eb2","resolution":{"observed_at":"2026-08-12T21:27:16.246965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.253155Z","title":"A dataset for develop- ing and benchmarking active vision","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.253155Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:93f726a31e43cb6edfa53212a8bb104935ecab73b72ab33c4fd4ee615b252801","observation_id":"5f7e63c4-d784-42c4-9cd6-80c062bfff39","resolution":{"observed_at":"2026-08-12T21:27:16.253155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.258665Z","title":"Automatic editing of footage from multi- ple social cameras","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.258665Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:af2cea394c9428b9e076c7a4902b0da8c943b5a99fc5979a9ed74b470e1e3bcc","observation_id":"5f35b769-c0b0-401e-b12b-d158305d7e1d","resolution":{"observed_at":"2026-08-12T21:27:16.258665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.266710Z","title":null,"venue":null,"work_id":null,"year":1988},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.266710Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:625beec397a7899370d0d94c0884a55d752af6621126ba7a8ac8e8c8619a120c","observation_id":"2b809dd6-454e-4425-a264-bae399839d18","resolution":{"observed_at":"2026-08-12T21:27:16.266710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.02356","last_updated":"2021-05-28T07:09:48Z","snapshot_observed_at":"2026-08-13T23:53:49.363179Z","submitted_at":"2020-12-04T01:22:05Z","title":"WeaQA: Weak Supervision via Captions for Visual Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.02356","snapshot_observed_at":"2026-08-12T21:27:16.273850Z","title":"Weaqa: Weak supervision via captions for visual question answering","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.273850Z"},"links":{"cited_paper":"/paper/2012.02356","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:4111a0953b7dab713715a17154ce2d2b41e4ca5e8b36079ac204fffb1ebb934a","observation_id":"cc8180e4-325e-4e9e-8930-f4a1fede591c","resolution":{"observed_at":"2026-08-12T21:27:16.273850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.281097Z","title":"METEOR: An auto- matic metric for MT evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.281097Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ba1c13409fb7f625a2b8007ac9a499deafcf2815035f636fc6223641c780a1d9","observation_id":"a3137ed3-695d-457c-b997-5a27ba7b661f","resolution":{"observed_at":"2026-08-12T21:27:16.281097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.287219Z","title":"Is space-time attention all you need for video understanding? In ICML, page 4, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.287219Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d46af5f323db0e70dc125b0de498891c426434dac1760dbcefe8254ad1c7ef40","observation_id":"1fd79043-4e34-4a2c-892b-dd4ab3137967","resolution":{"observed_at":"2026-08-12T21:27:16.287219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.296787Z","title":"High- lightme: Detecting highlights from human-centric videos","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.296787Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c222505ce1895bf2a873477bcbdd6a5d593f67dccaf2152af2c7b481cd7265f8","observation_id":"71bc575b-f19d-4b46-8a5c-939063e2b0b4","resolution":{"observed_at":"2026-08-12T21:27:16.296787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.304503Z","title":"Extreme rotation estimation using dense correlation volumes","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.304503Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:4685664ef4cb8cee0189ffe8fadd60a7fde9625046a05d30bfda11d49307ec67","observation_id":"1970d812-711b-4c45-a19f-73c8b38c4a28","resolution":{"observed_at":"2026-08-12T21:27:16.304503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.312059Z","title":"Davis, and Lei Zhang","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.312059Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:76ca4988a27b124f9fa39dedaea182cec65afccc6bf6e9c9b29d2368e09d3cc6","observation_id":"0a0c9b5d-fb6a-4dba-a770-a30c788473ae","resolution":{"observed_at":"2026-08-12T21:27:16.312059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.318489Z","title":"Enhanced interactive 360° viewing via automatic guidance","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.318489Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:746dee1efe9ae1e52003aca87bd9d85c4937c432cc61df65b19dd6e2eda8a4b6","observation_id":"e8215ca4-bb01-45c4-917f-b677ab0198c0","resolution":{"observed_at":"2026-08-12T21:27:16.318489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.324741Z","title":"Learn- ing sports camera selection from internet videos","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.324741Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8719b722845df5fa942ca8e0ad77ce6fb9f7f199ceeb21854aca47d41ec9be3f","observation_id":"20ccd794-4833-45b6-95bf-6553e0024242","resolution":{"observed_at":"2026-08-12T21:27:16.324741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.331782Z","title":"Wide- baseline relative camera pose estimation with directional learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.331782Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c44c58127f516b0985261dcacef7b196da365c1b04698dbe4c017e3886498b69","observation_id":"9f637a90-1a36-4165-9a78-1202d46acbd4","resolution":{"observed_at":"2026-08-12T21:27:16.331782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.337905Z","title":"Geometry-aware recurrent neural networks for active visual recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.337905Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:36f064cf95dc7307b83dc11a55c4b579595948160722f2df9918ab19b554d1f8","observation_id":"3899a749-9f6c-48a5-bf29-582af09d9ade","resolution":{"observed_at":"2026-08-12T21:27:16.337905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.351292Z","title":"Towards a richer 2d understanding of hands at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.351292Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d6503966ff2f86421278601f719086546986329a0101338c3e779fe6d69b6edd","observation_id":"128e8bf9-6538-4aff-8651-901cd8535971","resolution":{"observed_at":"2026-08-12T21:27:16.351292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.357941Z","title":"Gonzalez, Ion Stoica, and Eric P","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.357941Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a16e3b1959008f65e3645cb2fa5b89e9260b5b41e594e8d039716efd6c8c8838","observation_id":"f512be94-a5f9-47ad-9c48-3fe5ebc5064b","resolution":{"observed_at":"2026-08-12T21:27:16.357941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.08664","last_updated":"2017-11-23T12:06:20Z","snapshot_observed_at":"2026-08-04T01:34:38.975772Z","submitted_at":"2017-11-23T12:06:20Z","title":"Self-view Grounding Given a Narrated 360{\\deg} Video","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.08664","snapshot_observed_at":"2026-08-12T21:27:16.365365Z","title":"Self-view ground- ing given a narrated 360 {\\deg} video","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.365365Z"},"links":{"cited_paper":"/paper/1711.08664","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:55ca3475897c480005e6028038caa490604e3ea341666646b0e42b3414c7bb1f","observation_id":"85501816-c5e0-4811-a324-9150f4030214","resolution":{"observed_at":"2026-08-12T21:27:16.365365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.374896Z","title":"Video co-summarization: Video summarization by visual co- occurrence","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.374896Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0c0bb74c1b16ac770a8d9cc281d8f0713950afe10ef6a88498e54326b89c6544","observation_id":"0b7dded2-0479-4e31-860c-2989caae0cf8","resolution":{"observed_at":"2026-08-12T21:27:16.374896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.381981Z","title":"elochoice","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.381981Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:09246301e6a7e5a7fe9eac8636182d70dc5d46a998090370e8ce8fb1df309439","observation_id":"7ab7f5a7-5e64-401d-b3e0-1b9d56e9b9a2","resolution":{"observed_at":"2026-08-12T21:27:16.381981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.387951Z","title":"Scaling egocentric vision: The epic- kitchens dataset","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.387951Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3b6c09243053e1d50f0c5489a314815e5c82b7306152f68faf6de09cb962ee1f","observation_id":"15cdcf51-af90-48d1-9276-bccd1ddac3ad","resolution":{"observed_at":"2026-08-12T21:27:16.387951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-12T21:27:16.393588Z","title":"Flashattention-2: Faster attention with bet- ter parallelism and work partitioning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.393588Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:472c6abb78334e526262ec1c0a7d5e2be93fe99a127ba9253cffde58ece4a3c7","observation_id":"13f61efd-951b-41cd-a55b-661009310c20","resolution":{"observed_at":"2026-08-12T21:27:16.393588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.399483Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.399483Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:e6507cc070eeaf98bb9b92ba8bfa190383bd8fc22997a6456de83bc6eb7a7b5d","observation_id":"58020f30-05f1-4628-8f05-3ada0da5fc75","resolution":{"observed_at":"2026-08-12T21:27:16.399483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.404733Z","title":"Velastin","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.404733Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d876275bf47ba94f27a02ee1b07246c712a8a3061fdb910fa3b43b83998acf96","observation_id":"9cb4724f-b0dd-49f3-825f-fe2f18162725","resolution":{"observed_at":"2026-08-12T21:27:16.404733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.410436Z","title":"Virtex: Learning visual representations from textual annotations","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.410436Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c0906390f6736cc808a04998e8d7ade2ce6ee7da5babdc671b49e05b5ddbebd0","observation_id":"d0be0b1f-e948-4cb3-b0af-4b86378f439e","resolution":{"observed_at":"2026-08-12T21:27:16.410436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T21:27:16.416234Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.416234Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c815b73d0adef44ec2807a83182abad09315c61b7423d9dfcb1b59e13f508f13","observation_id":"acde62f6-cf94-455a-bc65-539a3d3f779d","resolution":{"observed_at":"2026-08-12T21:27:16.416234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.423479Z","title":"Dense and aligned captions (dac) promote compositional reasoning in vl models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.423479Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:fd3b7fe25e140fa7c63c797f804c4be76ee768606fbd908ce35bce3486f39665","observation_id":"786245bb-34cc-4782-b6da-f8e960b5843f","resolution":{"observed_at":"2026-08-12T21:27:16.423479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.429622Z","title":"Multi-view active fine- grained visual recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.429622Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:71cfd38fd096ca83dc486ed2951d51a3869e3af15efc7b10bbf752ea0450e52b","observation_id":"face0574-4d45-4a9a-b547-52f30f5a2cb4","resolution":{"observed_at":"2026-08-12T21:27:16.429622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.436093Z","title":"Multi- stream dynamic video summarization","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.436093Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0fa1f60ff50aeee498d2d5092461bbe2d0eb598b297d2b0db47261c1c2408f16","observation_id":"ae7829db-5a61-4529-b7b2-53d081487132","resolution":{"observed_at":"2026-08-12T21:27:16.436093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.441602Z","title":"Elson and Mark O","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.441602Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9300c5f88be604ee736d287949439e4b52e9639c67d5681183239a8c772bae60","observation_id":"6c41b363-bd5c-48fc-8940-827deb8740d4","resolution":{"observed_at":"2026-08-12T21:27:16.441602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.450299Z","title":"Foote and D","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.450299Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:eb67ab0d9a526529e1b571d20e0b661ac40b73c35a54055a875aea6357875302","observation_id":"20a10721-ce03-4f24-8f5d-68767cb25750","resolution":{"observed_at":"2026-08-12T21:27:16.450299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.457120Z","title":"Multi-view video summa- rization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.457120Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9bc714ee2370b114e8ba8cc14d1e5ef7110ea561a1ef77078fe3fd6ba899dbe1","observation_id":"0904ea70-ee80-4a76-9aac-ec3227472dea","resolution":{"observed_at":"2026-08-12T21:27:16.457120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.463492Z","title":"Gleicher, Rachel M","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.463492Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:31ef3e323a94f0e0f6e5aa62f95e88ae926e2c04ebb5ea8456d6ae91808ac9c4","observation_id":"aea03a6d-8bf6-41c2-823e-ae6aaa5262d3","resolution":{"observed_at":"2026-08-12T21:27:16.463492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07336","last_updated":"2024-04-10T20:32:24Z","snapshot_observed_at":"2026-08-13T00:33:06.461321Z","submitted_at":"2024-04-10T20:32:24Z","title":"PEAVS: Perceptual Evaluation of Audio-Visual Synchrony Grounded in Viewers' Opinion Scores","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07336","snapshot_observed_at":"2026-08-12T21:27:16.469572Z","title":"Peavs: Perceptual evaluation of audio-visual syn- chrony grounded in viewers’ opinion scores","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.469572Z"},"links":{"cited_paper":"/paper/2404.07336","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f14bd8452c980dc94105a203228fa0484e1ff6dc1cf28bfeedd6e0d1fc62b32a","observation_id":"69cd9eef-ad25-4bde-be6c-0f56da3ca3fa","resolution":{"observed_at":"2026-08-12T21:27:16.469572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.479571Z","title":"Diverse sequential subset selection for supervised video summarization","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.479571Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5ba7f92e4d4c5b6742e684e546054833fb08f7215664a88487fdc9acbf281798","observation_id":"b399e38c-c960-45a5-8172-81faef191d2f","resolution":{"observed_at":"2026-08-12T21:27:16.479571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.485303Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.485303Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:054e7d2129483c6e89e52af49b83529f6a59d4433421d8ade22d536a9f37341d","observation_id":"1fb9a401-425c-4436-beaa-101ae1ca7ace","resolution":{"observed_at":"2026-08-12T21:27:16.485303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18259","last_updated":"2024-09-25T21:55:38Z","snapshot_observed_at":"2026-08-13T05:13:48.123305Z","submitted_at":"2023-11-30T05:21:07Z","title":"Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18259","snapshot_observed_at":"2026-08-12T21:27:16.490447Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.490447Z"},"links":{"cited_paper":"/paper/2311.18259","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:bd6026b94c54661847fb60d0843d6819ecd1722ec2c35937ad10d7ef9e1ef2dd","observation_id":"c0b2dc9a-8d8c-43ed-9738-c519d6a0c022","resolution":{"observed_at":"2026-08-12T21:27:16.490447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.03107","last_updated":"2019-01-02T16:53:13Z","snapshot_observed_at":"2026-08-13T21:52:58.954758Z","submitted_at":"2018-06-08T12:10:58Z","title":"Temporal Difference Variational Auto-Encoder","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.03107","snapshot_observed_at":"2026-08-12T21:27:16.496223Z","title":"Temporal difference varia- tional auto-encoder","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.496223Z"},"links":{"cited_paper":"/paper/1806.03107","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:7e4e5c8896ba30ba1880745232d43532fb43a815e2f2601bf20f7b6f244a6f34","observation_id":"36ada9b3-36b4-4a4b-bdaf-73478b66a597","resolution":{"observed_at":"2026-08-12T21:27:16.496223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10846","last_updated":"2023-05-08T06:04:04Z","snapshot_observed_at":"2026-08-13T13:17:30.274656Z","submitted_at":"2022-12-21T08:39:36Z","title":"From Images to Textual Prompts: Zero-shot VQA with Frozen Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10846","snapshot_observed_at":"2026-08-12T21:27:16.503459Z","title":"From images to textual prompts: Zero-shot vqa with frozen large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.503459Z"},"links":{"cited_paper":"/paper/2212.10846","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2e07b27ba7823a774f7366e245df877f444549ec54afc5eff62d74ba44947650","observation_id":"82cf2f02-60ce-48a7-b2b7-45c7745d63b4","resolution":{"observed_at":"2026-08-12T21:27:16.503459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.510211Z","title":"Using closed captions as supervision for video activity recognition","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.510211Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:26f1b9259a5f6fdd80c963b7846933f454bab9f53407eb3959d51f5cffc7cd5b","observation_id":"bb7f5658-868b-4da4-8003-6fea52ca632f","resolution":{"observed_at":"2026-08-12T21:27:16.510211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.520424Z","title":"Creating summaries from user videos","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.520424Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d5dcaaf6a4f0f8b4c1c672f4fa24e4d67b91b1d26d0509ee724c4fc81a572f8d","observation_id":"ee3b7321-6456-4c50-a586-d310a09ced60","resolution":{"observed_at":"2026-08-12T21:27:16.520424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.533989Z","title":"Video summarization by learning submodular mixtures of objec- tives","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.533989Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ad2c21ab0d3b189d91aec4d8c34cd991fc036503f4c53570d9430d2856e2927d","observation_id":"adf155bf-20d6-4c5c-b6cc-30895c8ea6ba","resolution":{"observed_at":"2026-08-12T21:27:16.533989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.539969Z","title":"Align and attend: Multimodal summarization with dual contrastive losses","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.539969Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:65a51782c753a22db85015e1d45403758c157417d56f51e935e311c47c07c751","observation_id":"51c0c39d-eb07-4531-b1e0-32883fb4dd9b","resolution":{"observed_at":"2026-08-12T21:27:16.539969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.545269Z","title":"Cohen, and David H","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.545269Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:21712ba25eb51eb8361c4a2c82fcad6614965a6f3f9a895ce42402b3af54f14a","observation_id":"ccfe8115-fc24-4cb6-8957-db3605f10956","resolution":{"observed_at":"2026-08-12T21:27:16.545269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.551100Z","title":"Cohen, and David H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.551100Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2f25c9274095826567275e788416826f4ab2c0f5d7a49590a6e5437d3b7d8413","observation_id":"4771a9c2-2ac3-4484-be01-c8bc7857d19c","resolution":{"observed_at":"2026-08-12T21:27:16.551100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.556351Z","title":"Vir- tual videography","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.556351Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d279f8e6e4d586331825467d2ff3fd1efead3d89547010078313e6bd0a0cb150","observation_id":"772d4bde-00fa-4203-a507-cb95825b6041","resolution":{"observed_at":"2026-08-12T21:27:16.556351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-12T21:27:16.561315Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.561315Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8185fdd137410835add490f18218cdb3217b1b8e3fe5effa9872ad0d3551ac3e","observation_id":"00f95f17-6312-4f34-91fb-bd507296c3bd","resolution":{"observed_at":"2026-08-12T21:27:16.561315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.566908Z","title":"Deep 360 pilot: Learning a deep agent for piloting through 360deg sports videos","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.566908Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:394f7219ffa1eae3cce789b608eef53e8e329d32217a9fbd6c8db3dd28ff6c3c","observation_id":"d38383b6-f858-4707-a017-1dede39c461d","resolution":{"observed_at":"2026-08-12T21:27:16.566908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16182","last_updated":"2025-03-06T02:46:51Z","snapshot_observed_at":"2026-08-13T00:46:57.055849Z","submitted_at":"2024-03-24T15:00:44Z","title":"EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16182","snapshot_observed_at":"2026-08-12T21:27:16.572054Z","title":"Egoexolearn: A dataset for bridging asynchronous ego-and exo-centric view of procedural activi- ties in real world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.572054Z"},"links":{"cited_paper":"/paper/2403.16182","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5f6b45bcfa313f87b12317a068fc373c08d608444a0ed68e98111d3aac17e464","observation_id":"eaaccbea-74bd-4aa4-a9a1-0ed54ce64a55","resolution":{"observed_at":"2026-08-12T21:27:16.572054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.577856Z","title":"Batch normalization: accelerating deep network training by reducing internal co- variate shift","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.577856Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2806d5ad66989c43bad3e810ab542d84ed2ebf43d1390c0ca21a20d1147918d5","observation_id":"d3df167d-f2ce-44a1-97d9-fcefb64f1699","resolution":{"observed_at":"2026-08-12T21:27:16.577856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.584139Z","title":"Look-ahead be- fore you leap: end-to-end active recognition by forecasting the effect of motion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.584139Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:046593fad8b9dfecfb57a9e11dfeb498f4cd9f9339291de399a1813985a7a27c","observation_id":"c0397a64-a4a6-40dc-9e1d-f306b0a6255a","resolution":{"observed_at":"2026-08-12T21:27:16.584139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.589659Z","title":"Learning to look around: Intelligently exploring unseen environments for unknown tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.589659Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:34e4bbcecb257b60845b5923787f0a432d5dc0c222ca25ba86a2da0c7971694a","observation_id":"8ea4e95b-7769-44eb-b8c4-b79cb4233be3","resolution":{"observed_at":"2026-08-12T21:27:16.589659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.595297Z","title":"End-to-end policy learning for active visual categorization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.595297Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:7a47b0328a232ea769ad94b3af377e5066b411178d06ac45b5cb3fda372d3bdd","observation_id":"512bb144-e0c5-4bf9-a5f8-977456cfdf53","resolution":{"observed_at":"2026-08-12T21:27:16.595297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1808.07784","last_updated":"2018-10-23T19:22:25Z","snapshot_observed_at":"2026-07-06T06:57:04.785555Z","submitted_at":"2018-08-23T14:52:40Z","title":"Time-Agnostic Prediction: Predicting Predictable Video Frames","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.07784","snapshot_observed_at":"2026-08-12T21:27:16.600840Z","title":"Time-agnostic prediction: Predicting pre- dictable video frames","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.600840Z"},"links":{"cited_paper":"/paper/1808.07784","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d7909430e455d6af88a8f4495bafd38287fa831be26a5cd11529c7c58c2891c3","observation_id":"2835bd81-9f36-41e4-a29d-cbf613bc3efc","resolution":{"observed_at":"2026-08-12T21:27:16.600840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.607025Z","title":"Simglim: Simplifying glimpse based active visual reconstruction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.607025Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2c38a850a907a679760853c70366c6f4c048c77047c863a8ef840918837994c7","observation_id":"5c8021b6-ddd4-4df0-a69b-51c0e2d0c7ce","resolution":{"observed_at":"2026-08-12T21:27:16.607025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.612958Z","title":"Lemma: A multi-view dataset for le arning m ulti-agent m ulti-task a ctivities","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.612958Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d86733e8afa7a3a6b6b93d7f818e5ee117ee96b7b2d0234f79a2e45dd8256a00","observation_id":"d3adfba1-3679-4606-90ce-c0a8591e3d1d","resolution":{"observed_at":"2026-08-12T21:27:16.612958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.07399","last_updated":"2023-07-03T03:06:26Z","snapshot_observed_at":"2026-08-13T12:25:27.314050Z","submitted_at":"2023-03-13T18:26:11Z","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.07399","snapshot_observed_at":"2026-08-12T21:27:16.619160Z","title":"Rtmpose: Real-time multi-person pose estimation based on mmpose","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.619160Z"},"links":{"cited_paper":"/paper/2303.07399","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:056836e73c0edb30ab73f8286500e9ae34086d04287573ea38cf7dcf5e2280b1","observation_id":"75304319-c3b5-4630-854f-269cd0168e43","resolution":{"observed_at":"2026-08-12T21:27:16.619160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.624405Z","title":"Large-scale video summarization using web-image priors","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.624405Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:e2ed776c785afa4f1a768d35137ad8514ebdc3645dba18f132e2071505cb7afc","observation_id":"cece5df7-2619-49f6-be83-94a62c0fa7f0","resolution":{"observed_at":"2026-08-12T21:27:16.624405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.629375Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.629375Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2605223876ba3f4f89ef1f0aed1bdcdb90fc24dfde16acdaf40a2692c676a5bd","observation_id":"4afe155d-955c-48ed-bbcc-329d56768cd9","resolution":{"observed_at":"2026-08-12T21:27:16.629375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.635697Z","title":"Segment any- thing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.635697Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:71a5254dfd4acea9df923b9efa331e2db884ebf207b1336491656178cca93161","observation_id":"cb661ab1-ee16-4e98-81ab-c87070f9aa7b","resolution":{"observed_at":"2026-08-12T21:27:16.635697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05016","last_updated":"2024-04-07T17:06:22Z","snapshot_observed_at":"2026-08-13T00:35:31.505586Z","submitted_at":"2024-04-07T17:06:22Z","title":"Hyperbolic Learning with Synthetic Captions for Open-World Detection","version":1},"cited_work":{"arxiv_id":"2404.05016","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05016","snapshot_observed_at":"2026-08-12T21:27:17.283091Z","title":"Hyperbolic Learning with Synthetic Captions for Open-World Detection","venue":"cs.CV","work_id":"e1ef6378-7af6-4bfa-be0b-d4ca7da54b39","year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.641426Z"},"links":{"cited_paper":"/paper/2404.05016","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a4ee457a8ac1a7abfabdf940e36ec3183d1c5cb5acad17a0420a27ca327d1a00","observation_id":"883c89c2-d078-4ec9-9dee-f367fe283eba","resolution":{"observed_at":"2026-08-12T21:27:17.290381Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.650230Z","title":"A memory network approach for story-based temporal summarization of 360° videos","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.650230Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:11da35eff595815986a60f99c3eb157f2366db645f983b6e438262fad08afa10","observation_id":"41dfb3fc-a295-4cc4-895d-81fc3c3afa56","resolution":{"observed_at":"2026-08-12T21:27:16.650230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.657826Z","title":"Predicting important objects for egocentric video summarization","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.657826Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c12f5d758667f6c5cebbb7907135727150b6ee8b0e009bd820b77d9f0ffed27b","observation_id":"f6e45f0f-680a-4afb-9c6e-ea8b2a8a707a","resolution":{"observed_at":"2026-08-12T21:27:16.657826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-12T21:27:16.663653Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.663653Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a5ea4b51bba89790ddceac789580500b57cd740df0aa6fbe9c960980a37c2b3c","observation_id":"ebfefeba-e6cc-4e98-ad04-5e03f150fd7a","resolution":{"observed_at":"2026-08-12T21:27:16.663653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.669888Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.669888Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f22ddbc7cd8560dc577c12592185b32448bf29340554f0639833b00acc07b085","observation_id":"e15fd5d7-482f-47e8-9912-453c1c504e1c","resolution":{"observed_at":"2026-08-12T21:27:16.669888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.675226Z","title":"How local is the local diversity? reinforcing sequen- tial determinantal point processes with dynamic ground sets for supervised video summarization","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.675226Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:b95a40d23f0bfbc2b3efd1cb400d3f20251a180a0f0f62151edddc729be58b8d","observation_id":"cb28cbc6-1078-4f06-883f-5b54aacfb4cb","resolution":{"observed_at":"2026-08-12T21:27:16.675226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.562150Z","title":"Egocentric video-language pretraining","venue":null,"work_id":"1225a86b-0f2f-4f4a-bb4f-7ba9cc8094bb","year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.680256Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:bf2286e95733f2724dc7a23f9e6e55516e5ae956a922e96cedda4c0a9b97cbdf","observation_id":"756366c4-920a-4fa0-adff-8122bda308df","resolution":{"observed_at":"2026-08-12T21:27:18.569353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-12T21:27:16.685434Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.685434Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f6347551a7b19358bd1cce4d0b6d4b776872ac04d2ba9eb4811d10f1f272a151","observation_id":"cadd1a59-85f1-4e16-aa5a-d1d422355b18","resolution":{"observed_at":"2026-08-12T21:27:16.685434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1608.03983","last_updated":"2017-05-03T16:28:09Z","snapshot_observed_at":"2026-07-06T05:06:55.589962Z","submitted_at":"2016-08-13T13:46:05Z","title":"SGDR: Stochastic Gradient Descent with Warm Restarts","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1608.03983","snapshot_observed_at":"2026-08-12T21:27:16.691686Z","title":"Sgdr: Stochastic gradient descent with warm restarts","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.691686Z"},"links":{"cited_paper":"/paper/1608.03983","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a1183040a4c0a1830f48246f46430bbe39f62756e0b8d56d8e619352a93f9117","observation_id":"d116433c-7261-411f-861d-40dd7605c0ca","resolution":{"observed_at":"2026-08-12T21:27:16.691686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-12T21:27:16.697244Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.697244Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:224247ae6af9ea31e320588611d9fde132d1bd9aad8501087d7ed7c5a2c1f56a","observation_id":"30c3a2d4-49dd-4644-9a7e-f73abf1a3b68","resolution":{"observed_at":"2026-08-12T21:27:16.697244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.541297Z","title":"Story-driven summariza- tion for egocentric video","venue":null,"work_id":"0d18bba4-6de9-4f9b-8661-d4533f4cd747","year":2013},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.702606Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:535a4a9df68b64c2b78a2f5306d3841c0662aa953ebf2fad64d8cb9d78b91b6e","observation_id":"5fefd853-04f0-49ae-bd8c-3265e7192f38","resolution":{"observed_at":"2026-08-12T21:27:18.549057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18386","last_updated":"2025-04-22T13:23:34Z","snapshot_observed_at":"2026-08-12T23:13:41.357483Z","submitted_at":"2024-12-24T12:16:43Z","title":"Switch-a-View: View Selection Learned from Unlabeled In-the-wild Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18386","snapshot_observed_at":"2026-08-12T21:27:16.709927Z","title":"Switch-a-view: Few-shot view selection learned from edited videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.709927Z"},"links":{"cited_paper":"/paper/2412.18386","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1b86c5afc4b4b1d7505a1f36b5eb3b9939cbdc87b97797552ff14090e5890d89","observation_id":"a9ae5bc4-44f2-43e1-b13f-e77415b31e02","resolution":{"observed_at":"2026-08-12T21:27:16.709927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.519942Z","title":"Video summarization via multi- view representative selection","venue":null,"work_id":"309630f7-4d4d-467f-9f89-53b1ccefc186","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.715943Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:de959bf2ca2dc4ff38abb989d49776385b8c5c38871e1b7f991f3e24c0ce19f4","observation_id":"04784fba-8090-4375-bf5d-e45497f5a31e","resolution":{"observed_at":"2026-08-12T21:27:18.525516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.489162Z","title":"Howto100m: Learning a text-video embedding by watching hundred million narrated video clips","venue":null,"work_id":"61e7f163-3347-4789-9dfc-024cc8961744","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.721421Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:23ac24a73dc3db713b71a234c36c0c30b7c0fbf1a82fff46f95576b34cf3edf2","observation_id":"fe0807fc-36ba-4265-b1fc-4fa9b5e49f6a","resolution":{"observed_at":"2026-08-12T21:27:18.496074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.726553Z","title":"Srinivasan, Matthew Tancik, Jonathan T","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.726553Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:954f17338e4e6a9cbc364954c67b0b69911e714adb063423a574160ce014f476","observation_id":"8106cd4f-a3de-45d2-9b5d-29a59d8cef9b","resolution":{"observed_at":"2026-08-12T21:27:16.726553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.446155Z","title":"Automatized summarization of multi- player games","venue":null,"work_id":"f0388c32-2d7d-4c9d-8d2e-bf1d36ead5ee","year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.731426Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a8ccff22b5a80c7db93d9a8fe7ae84cd4c3da666da9268b5b503a48f711a9796","observation_id":"03ae2a1e-6682-4d9f-a1ec-1b14efc1a1e2","resolution":{"observed_at":"2026-08-12T21:27:18.454733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.402279Z","title":"Egoenv: Human- centric environment representations from egocentric video","venue":null,"work_id":"8d29c179-1af8-4966-9803-c580df8936ed","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.745576Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:b489a345db2f9dc9ef54a77c9956f1dd82eab24a23ed16cae1d6db6faf652545","observation_id":"61da99fb-d31c-4453-92b7-c0c58c127885","resolution":{"observed_at":"2026-08-12T21:27:18.410732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.374954Z","title":"Tl; dw? summarizing instructional videos with task relevance and cross-modal saliency","venue":null,"work_id":"1a085b25-6d11-4714-81e6-9902b72f5f7d","year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.750940Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:63488738526613d80008c8923a7ea006e0955eac3b8ab9b27f600bb63ed7ab51","observation_id":"67c3248e-d510-4503-9440-8e7f5cc3a3bb","resolution":{"observed_at":"2026-08-12T21:27:18.385894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.350793Z","title":"Adaptive skip intervals: Temporal abstraction for recurrent dynamical models","venue":null,"work_id":"c394d24c-6d22-412e-afd7-0dcadae7029b","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.756958Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:dd24698bff89b5740dab88fa95d40ac5035d5d85550e08704aaf0cd710bcd923","observation_id":"82e5f804-aefd-4974-b8dc-ef24f38a6c01","resolution":{"observed_at":"2026-08-12T21:27:18.357873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.330411Z","title":"Au- tomatic video summarization by graph modeling","venue":null,"work_id":"9ba41228-c339-4821-a544-f94367cfc06c","year":2003},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.763234Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ba219f2aa456947e6cb0694c8b3ea7e24cac8af0605a7fa984c1a844458fcfe4","observation_id":"3e63b266-ec5b-4f6c-abe9-3c919f19deda","resolution":{"observed_at":"2026-08-12T21:27:18.336833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.307919Z","title":"Collabora- tive summarization of topic-related videos","venue":null,"work_id":"a55202f2-d40a-4264-b2a8-ccf570447f90","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.771983Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1a31dae75c921d3e554efdc9452ed41c8ede41a6f93a6889810e716cdfc60716","observation_id":"ad9bab32-aae4-4415-8174-2a602a225541","resolution":{"observed_at":"2026-08-12T21:27:18.316696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.288571Z","title":"Multi-view surveillance video summarization via joint embedding and sparse optimization","venue":null,"work_id":"d0957fdd-5c26-45e6-932a-136d0763d479","year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.778053Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a92ee250b06fd101853e39ee7dbc499b9ff75acf8298cf65d1a08188a53e6a1a","observation_id":"1deecac3-7788-4b32-ab77-5d47a72a0932","resolution":{"observed_at":"2026-08-12T21:27:18.295299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.268290Z","title":"Roy-Chowdhury","venue":null,"work_id":"fe1484b7-8320-42a1-97dc-210d1f32ed09","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.783830Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:6c455fe1978be7307c572036969314f10a6915a72b6c7120f70d8d3748e844cb","observation_id":"8f1a830f-752c-4357-a063-ac2da8b6188e","resolution":{"observed_at":"2026-08-12T21:27:18.274325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.252163Z","title":"Bleu: a method for automatic evaluation of machine translation","venue":null,"work_id":"fd3cf8ac-07c4-44f9-8c77-6ba51dec67f4","year":2002},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.789486Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d92a079efdf7f45ae6204163bef3e9858377e2cf45b6d2919f243f0af2bd6137","observation_id":"c21a3630-06c2-4b52-af5b-4584cb1ac0f1","resolution":{"observed_at":"2026-08-12T21:27:18.257179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.232407Z","title":"Sumgraph: Video summarization via recursive graph modeling","venue":null,"work_id":"e4c2dbf3-0cae-4153-8cf8-cef52777c5d9","year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.795037Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5a0fead504137fe746fd0970b09b9f1fab38b76232a8babab0a51efaf53105d2","observation_id":"5a1a688b-0e61-4405-8da4-40835fb472d8","resolution":{"observed_at":"2026-08-12T21:27:18.238943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.209438Z","title":"Egovlpv2: Egocentric video-language pre-training with fusion in the backbone","venue":null,"work_id":"5daa8f2d-8057-45d5-a514-89fa67b62cd0","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.799995Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ec29a0ea27df3dadbe02b2e0e38998db37d570c3d6a68b31ac668e0aec63cd23","observation_id":"7d7ac8df-72f4-4c62-9db1-cc869eb1d79d","resolution":{"observed_at":"2026-08-12T21:27:18.215657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.187817Z","title":"Vloc- net++: Deep multitask learning for semantic visual localiza- tion and odometry","venue":null,"work_id":"8a37e01f-6958-4a14-8a34-002e7161b040","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.805015Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2784e87e40b45b23dc6df41d38139b4deca9f4524b54f136f3692ab4d18e1718","observation_id":"378c0f34-b5b3-4334-b4f1-51e9efeac2e5","resolution":{"observed_at":"2026-08-12T21:27:18.193781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.171261Z","title":"Sidekick policy learning for active visual exploration","venue":null,"work_id":"63bddafa-32d9-4ea0-a55d-c59b136c87a9","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.810155Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a53e77bd93e49d8b642019f24ac05c69846339aa5d1acd0676f116b291d1ca2f","observation_id":"a4a08e10-1781-4d09-a398-42aed2923331","resolution":{"observed_at":"2026-08-12T21:27:18.176412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.155042Z","title":"Emergence of exploratory look-around behaviors through active observation completion","venue":null,"work_id":"1d30e913-b3c0-4a66-9672-6ef4314e2b87","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.814977Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a834e195bc503d44da3e69a8f23cb53eb35e77e092503c67d8f8a152a4c8b7af","observation_id":"af605da0-9852-40f0-bac3-870705489a30","resolution":{"observed_at":"2026-08-12T21:27:18.160511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.138359Z","title":"Naq: Leveraging narrations as queries to super- vise episodic memory","venue":null,"work_id":"2cf78ac6-5875-47f0-9a1f-dba3a11bf329","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.820501Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f0736283db3c10ca4b7581483bc8bb70b6ad098b128f402a1983c68f41017d44","observation_id":"e496eb70-24af-4f73-94f6-c2f539eb4e2a","resolution":{"observed_at":"2026-08-12T21:27:18.143654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.121068Z","title":"Video summarization by learning from unpaired data","venue":null,"work_id":"bdf1f7a8-954c-49c9-b09c-62732aba3825","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.829184Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2499e86acb8594bc52366244411c4130bed9535d6c1e4706361cae4ed47d22d2","observation_id":"d19f8cfa-e3cc-4253-ad30-7ea6ae3fa489","resolution":{"observed_at":"2026-08-12T21:27:18.126653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.835818Z","title":"Adaptive video highlight detection by learning from user history","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.835818Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:605031bea5ca7f588dfd930ba06406a127a8f083d83e75f8d7e8085997f70379","observation_id":"0d4725b1-c25f-457f-9870-5f756309db84","resolution":{"observed_at":"2026-08-12T21:27:16.835818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.092546Z","title":"Chowdhury","venue":null,"work_id":"e13412bb-f291-429a-8875-929608a9cdf9","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.841151Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:b5bbe042088be0ceb2737ed086eeb0349704719d670f402d94935390227d166d","observation_id":"2fc92f75-2740-4f13-a654-84a748667606","resolution":{"observed_at":"2026-08-12T21:27:18.097889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.846399Z","title":"Attend and segment: Attention guided active semantic segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.846399Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d2072b2f32f4d9d62effa97c1e4021e56aa1c8833545b81f85c989e7e22c801b","observation_id":"c5a9690f-59df-487e-b0b1-bb2c99d7e9e3","resolution":{"observed_at":"2026-08-12T21:27:16.846399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.064516Z","title":"Glimpse- attend-and-explore: Self-attention for active visual explo- ration","venue":null,"work_id":"a8cc0b43-4ecc-45d6-8ea4-987c85282cf7","year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.851964Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ef5918902104dab6c37099af70de8807915877dac2c516f7ddeefc1d8ad3c21c","observation_id":"121c8f1a-836b-4abb-a24b-3acd089351ec","resolution":{"observed_at":"2026-08-12T21:27:18.070161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.046566Z","title":"Actor and observer: Joint modeling of first and third-person videos","venue":null,"work_id":"29b5912f-8738-4e05-9aee-ae1884bc3021","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.858613Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9b4729b6ba8c0e0d21a9f6ea3143217eb3381414e57aaa084cce7bb69585ce41","observation_id":"d6b9d89a-40b3-4049-8dfd-16e538200d9a","resolution":{"observed_at":"2026-08-12T21:27:18.052780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.029253Z","title":"Tvsum: Summarizing web videos using titles","venue":null,"work_id":"bbba6d49-f813-409c-975c-c5fbfe922965","year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.863934Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2a4b157d07ccd6747b43747a1d0da5217564fda6d599d7e1bc8a6306baa27ef0","observation_id":"8481d850-ddf3-403a-8c8c-56f49059cce9","resolution":{"observed_at":"2026-08-12T21:27:18.034387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.010191Z","title":"Making 360 ° video watchable in 2d: Learning videography for click free view- ing","venue":null,"work_id":"1d195d5d-2012-40ef-be1c-21e2059689ef","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.872398Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9dbebbb485b11b643b84f256c2a97937d2c27df2aa43c1e01add8807cff322b1","observation_id":"6266f243-3dcb-4190-a5e2-0dda93bef23a","resolution":{"observed_at":"2026-08-12T21:27:18.015861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:17.989207Z","title":"Pano2vid: Automatic cinematography for watching 360 videos","venue":null,"work_id":"ff6aef1a-f056-41c3-b682-afc62123442b","year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.878297Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d9bbd81466710a15c7ab349ef670d0e9ed1e3dd78b516a51792e25c7df27a424","observation_id":"67a9f5c6-e0a6-41cf-9c32-e40ba2a166ae","resolution":{"observed_at":"2026-08-12T21:27:17.994590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:17.968863Z","title":"Automatic con- cept discovery from parallel text and visual corpora","venue":null,"work_id":"f21fe035-6211-4deb-85f0-605e55872aa4","year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.884770Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:aa7a12b49320eea3cffe86056ccb26c7d7eeb7fb143b249bbac9e648ec6e24da","observation_id":"fbe32558-5b46-4a57-9a06-394c4ba8a462","resolution":{"observed_at":"2026-08-12T21:27:17.975142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T23:13:31.276750Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":72,"verified_exact":1,"verified_fuzzy":27},"total_outbound_references":132},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 100 of 132 outbound references and 2 inbound Pith citation observations for arXiv:2411.08753."}