{"as_of":"2026-08-08T00:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:34a917070ea05eb4ebeee2560a39373c6155911750244973a440939ae8a54bad","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:57:54.159395Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.16784/citation-record","integrity":"/paper/2505.16784/integrity","json":"/paper/2505.16784/citation-record.json","paper":"/paper/2505.16784"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:57:51.616124Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:51.616124Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:50c787a0cfbec2f71f212a92bf2f53d8fd5697b5ad61b9bb073645fe0a040761","observation_id":"b7f30f4c-34b9-4bc9-82e2-e1e78977242a","resolution":{"observed_at":"2026-08-07T14:57:51.616124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05861","last_updated":"2024-05-31T15:22:58Z","snapshot_observed_at":"2026-08-06T20:09:28.314739Z","submitted_at":"2024-02-08T17:50:22Z","title":"Memory Consolidation Enables Long-Context Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05861","snapshot_observed_at":"2026-08-07T14:57:51.705783Z","title":"Mem- ory consolidation enables long-context video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:51.705783Z"},"links":{"cited_paper":"/paper/2402.05861","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:a269b1a92840bf79b7152d12408008034c65deca21fbe0a72de0452c10e4b6fb","observation_id":"237097a9-daa1-4824-a48d-451b0c10ecb1","resolution":{"observed_at":"2026-08-07T14:57:51.705783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:57.744666Z","title":"Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning.Advances in Neural Infor- mation Processing Systems, 37:110805–110853, 2024","venue":null,"work_id":"131b22d3-84cc-41cc-8afc-bd60fd5aab05","year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:51.821362Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:8e6f387beec88da1fb50b672ffcf12ce48dc19193932b03c2922be339537e498","observation_id":"c2c862b9-381c-4899-8845-45c304aff552","resolution":{"observed_at":"2026-08-07T14:57:57.782215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00937","last_updated":"2023-12-01T21:34:10Z","snapshot_observed_at":"2026-08-02T08:12:01.084894Z","submitted_at":"2023-12-01T21:34:10Z","title":"Zero-Shot Video Question Answering with Procedural Programs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00937","snapshot_observed_at":"2026-08-07T14:57:51.945568Z","title":"Zero-shot video question answering with pro- cedural programs.arXiv preprint arXiv:2312.00937, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:51.945568Z"},"links":{"cited_paper":"/paper/2312.00937","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:1915951974f78492867059b9d45e2c3bc24f8604e24446aa07df8d83a16683da","observation_id":"f2a6cd60-b6d1-428b-b48b-86a2259482d0","resolution":{"observed_at":"2026-08-07T14:57:51.945568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:57.571004Z","title":"Second joint egocentric vision (EgoVis) workshop,","venue":null,"work_id":"4f9ab358-6b75-460e-9da0-fb8be786c920","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.051616Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:56c1483f4319300ce1c1f0bf286c5643a78f45b1bf403d372a76de894e8e505e","observation_id":"7e378511-06e1-4f13-87f1-e1374ef9b207","resolution":{"observed_at":"2026-08-07T14:57:57.653636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:52.215793Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.215793Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:a38c422376500df8da86a72424f2bc1dd39d7d02e4372cd51e75abacfac78a9d","observation_id":"aec9caee-362d-4a53-8e2d-5e691d3c8a0e","resolution":{"observed_at":"2026-08-07T14:57:52.215793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:57.222813Z","title":"Parameter-efficient transfer learning for nlp","venue":null,"work_id":"c070534d-d761-4ca7-9b92-9de02f87625b","year":2019},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.307955Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:fb22cb98260c6db178a9d28f26ecfe14ad64513c0529e77d71748c4107440c5c","observation_id":"7dce8cc0-a22f-4e6d-84f5-ff837bd07762","resolution":{"observed_at":"2026-08-07T14:57:57.294307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:52.473986Z","title":"Lora: Low-rank adaptation of large language models.ICLR, 1(2):3, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.473986Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:4dd9cf4b96f1feed17acac9571fe61709a82509a98df645ac99466af199b9f1d","observation_id":"222e22f9-d8d6-4e07-b303-0b807a0bef2e","resolution":{"observed_at":"2026-08-07T14:57:52.473986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:57.002561Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding.Advances in Neural In- formation Processing Systems, 36:46212–46244, 2023","venue":null,"work_id":"a8e00007-1bd0-4116-adae-2806808ecfbb","year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.557512Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:1d0d3c00ce04587bf499fd9c28c2eb12d5c64a71e3f12eb7a6c30c2fb5f61abc","observation_id":"99c2ae90-6771-4ffb-b426-a3683d83e1b7","resolution":{"observed_at":"2026-08-07T14:57:57.124446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:56.733746Z","title":"A simple recipe for contrastively pre-training video-first en- coders beyond 16 frames","venue":null,"work_id":"f971036f-e966-4bef-9f09-19ed808ba80b","year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.647893Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:98c9896456f30482a423a1ebe82bbab56a0060f46883d3f09a7784b88732ee28","observation_id":"8ce93b11-66d3-4fc9-85f4-742fe0635e5f","resolution":{"observed_at":"2026-08-07T14:57:56.898855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T14:57:52.707951Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of con- text.arXiv preprint arXiv:2403.05530, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.707951Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:1c86ece895710f9a42976743d14acc4bf1cf0a625e974f784ab46bbf659a78d4","observation_id":"9d60e9a6-f6d4-4272-90ec-2b7d60946bb4","resolution":{"observed_at":"2026-08-07T14:57:52.707951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10517","last_updated":"2024-03-15T17:57:52Z","snapshot_observed_at":"2026-08-06T12:50:13.936842Z","submitted_at":"2024-03-15T17:57:52Z","title":"VideoAgent: Long-form Video Understanding with Large Language Model as Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.10517","snapshot_observed_at":"2026-08-07T14:57:52.793038Z","title":"Videoagent: Long-form video understand- ing with large language model as agent.arXiv preprint arXiv:2403.10517, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.793038Z"},"links":{"cited_paper":"/paper/2403.10517","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:fc95526fcfb2d0f78c54c185bdea8ed55ce68b3401e2a8df52617ecfe5accb1e","observation_id":"d418d1fe-bf5c-4005-82bb-888a32e90f20","resolution":{"observed_at":"2026-08-07T14:57:52.793038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05269","last_updated":"2024-11-05T22:08:14Z","snapshot_observed_at":"2026-08-01T08:49:09.735566Z","submitted_at":"2023-12-07T19:19:25Z","title":"LifelongMemory: Leveraging LLMs for Answering Queries in Long-form Egocentric Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05269","snapshot_observed_at":"2026-08-07T14:57:52.884757Z","title":"Lifelongmem- ory: Leveraging llms for answering queries in long-form egocentric videos.arXiv preprint arXiv:2312.05269, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.884757Z"},"links":{"cited_paper":"/paper/2312.05269","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:41ee4c8c2bffd127d4d301c222a9f7d20de3ce3184b522a49df843cc82e345aa","observation_id":"e8b5f916-629f-49a1-813d-e86875a402bc","resolution":{"observed_at":"2026-08-07T14:57:52.884757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:56.451486Z","title":"Internvideo2: Scaling foundation models for mul- timodal video understanding","venue":null,"work_id":"d4887587-5477-4313-8c46-7359d3f0b994","year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.976834Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:8ef45ed0c44b967c50b6d53f381e41789f7cd9d1a6629fcc1c8866daa7dacb03","observation_id":"10a131b8-5899-4cd3-aed9-52b42c428d2b","resolution":{"observed_at":"2026-08-07T14:57:56.570189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-07T14:57:53.094827Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.094827Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:05524aaabd29c47e7807fd2eecb29a02649c11fda1e7435520c4dae3b013bd24","observation_id":"1de8b3d6-773b-4936-b1d7-db7414f37aea","resolution":{"observed_at":"2026-08-07T14:57:53.094827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17235","last_updated":"2024-10-10T05:17:00Z","snapshot_observed_at":"2026-07-06T17:09:24.271573Z","submitted_at":"2023-12-28T18:58:01Z","title":"A Simple LLM Framework for Long-Range Video Question-Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17235","snapshot_observed_at":"2026-08-07T14:57:53.167691Z","title":"A sim- ple llm framework for long-range video question-answering","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.167691Z"},"links":{"cited_paper":"/paper/2312.17235","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:81f85f28788138d863057cdc8f7503ac552c45c0cd130f6ccf52de8b1923ff26","observation_id":"d88c3ee7-3d28-4528-8fc2-6a4f70996126","resolution":{"observed_at":"2026-08-07T14:57:53.167691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15771","last_updated":"2024-10-29T02:38:27Z","snapshot_observed_at":"2026-08-06T07:53:27.979224Z","submitted_at":"2024-06-22T07:20:39Z","title":"HCQA @ Ego4D EgoSchema Challenge 2024","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15771","snapshot_observed_at":"2026-08-07T14:57:53.256239Z","title":"HCQA@Ego4D egoschema challenge 2024.arXiv preprint arXiv:2406.15771, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.256239Z"},"links":{"cited_paper":"/paper/2406.15771","citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:2f6b533c911e1119a449d3961a73230523927c73112cdfde81f302115e3eece3","observation_id":"4f0e4a4e-3f7d-4f44-9eb8-c972c0e149bd","resolution":{"observed_at":"2026-08-07T14:57:53.256239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:56.277745Z","title":"Learning video representations from large lan- guage models","venue":null,"work_id":"bc634f52-d378-4e5e-8435-485400b5a58f","year":2023},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.371228Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:a8b4ce78a88bfe73851351b9237d4a2f91b924ea1bb75c59a53ce27da440ad7d","observation_id":"3d142de2-27ca-4bf8-bf1e-6ce25b3b731a","resolution":{"observed_at":"2026-08-07T14:57:56.346481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:56.117492Z","title":"Among them, the underline indicates the content that needs to be filled based on the sample","venue":null,"work_id":"4a9d0915-0149-4497-95a7-51fe79db1561","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.487549Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:6878c73eed21cfda2dae71eed3fcbe32d72a946a111504b4a1e10bea512eca28","observation_id":"6be8a779-1b88-4a89-9371-1ff0d9c4a7e2","resolution":{"observed_at":"2026-08-07T14:57:56.189409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.947754Z","title":"CAPTION\":","venue":null,"work_id":"4b3c8ed4-76a4-47dc-b96c-7eb5f6ac7e2a","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.604872Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:134f19584fa52d70bdd75167292227f7f4c869dc79a33346bb28b53766e309c8","observation_id":"bab42e37-5b8b-4d53-b67d-09f2899da8d7","resolution":{"observed_at":"2026-08-07T14:57:56.024351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.747346Z","title":"Select the option that best suits the question and video content","venue":null,"work_id":"102aadfb-7e1e-42e8-a62b-6e1d02d23076","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.686850Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:76495ed1a0f532b0d782b12f9a76303835216bb06e4a2f00b3e46f32681c8f2e","observation_id":"eeba713a-7550-452e-b94c-b2890b59027e","resolution":{"observed_at":"2026-08-07T14:57:55.810758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.541666Z","title":"THINK\": [your chain of thoughts]","venue":null,"work_id":"bb52ce4e-9fd0-45f0-9957-fec51fe30e9b","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.747532Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:f13be443da3b1d6ccdef9e9826f28655780dc028d4631e53e410c37eda18fcd9","observation_id":"e2962be8-efa2-4656-b927-f450b3df18f6","resolution":{"observed_at":"2026-08-07T14:57:55.624377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.395847Z","title":null,"venue":null,"work_id":"0763b3da-8350-4945-8fcd-264bbdee5fb0","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.790552Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:017999d5f5d361c6511d7979ac4cd791cb8c4c3706aad264126b9bdd5d9db31c","observation_id":"5284b3cf-236a-4c01-9398-566fd8dd48d8","resolution":{"observed_at":"2026-08-07T14:57:55.433669Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.189018Z","title":null,"venue":null,"work_id":"32c5e1d6-2bc7-479e-bea5-329a5d5b4908","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.857517Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:00818fa1eb31b0578fd5f1dec15705499b672b72e0ab11ab16371aba8e77bec3","observation_id":"258a6297-3a54-45c9-9007-d2c950e9f1b3","resolution":{"observed_at":"2026-08-07T14:57:55.281070Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:55.025390Z","title":null,"venue":null,"work_id":"116346bc-a574-445a-843e-aeffd3379f5a","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.917672Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:88619e5d189e9144a30cfc735200fefe4da9e47bb609702550f02daf2da69b6c","observation_id":"de74d7c0-4222-4679-9c37-d352c3f3050d","resolution":{"observed_at":"2026-08-07T14:57:55.103605Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:54.832552Z","title":"The #C indicates the image seen from your point of view, and the #O indicates the other people in the image you seen","venue":null,"work_id":"f7520cd0-15f5-43de-9ae0-c4fe36b0b654","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:53.956970Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:45b403e16b4e7dbebe748c5cdb76e357eb4ceb6cf90ed2d4f0716fc47859cb2b","observation_id":"2851db99-882d-42c7-a5e6-04d4ad621645","resolution":{"observed_at":"2026-08-07T14:57:54.907490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:54.704820Z","title":null,"venue":null,"work_id":"fa94746e-766c-46a1-8072-010d2538d4a5","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:54.025118Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:acba9de3be24d5ec75aab75a6ce69652baa8165591cb426d7ab0bd4ced6529b7","observation_id":"8b48b877-910e-4466-bd4b-c419fad98cc2","resolution":{"observed_at":"2026-08-07T14:57:54.772522Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:54.604553Z","title":"REASON\". Please provide the results and confidence level in","venue":null,"work_id":"c035526b-bd57-4097-9425-887b4b922030","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:54.095631Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:e43860389925fa2025bb5346410e3f6f2b6071539ad3afa38f8f74dc4f9c61f6","observation_id":"97495255-83fc-4170-8322-f0c75189466e","resolution":{"observed_at":"2026-08-07T14:57:54.663506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:54.448547Z","title":"CAPTION\": {{","venue":null,"work_id":"e67f5251-4ecc-4971-a0df-ba21aa136510","year":null},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:54.159395Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:9c39852fb0f361c2011b1f1f218ae5421cf9d55a598d215e806a03b631d514fb","observation_id":"9c639481-fdc1-41d5-8f6a-c8608a50d836","resolution":{"observed_at":"2026-08-07T14:57:54.515358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:57:57.415893Z","title":null,"venue":null,"work_id":"a6ee0216-25df-4a7e-8aed-47829862174f","year":2025},"citing_paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:52.135230Z"},"links":{"citing_paper":"/paper/2505.16784"},"observation_digest":"sha256:b6054ce6f5cfab799de3ffd152cbfc2d3b373b72279b24683213d5231c7822e9","observation_id":"cc41e618-1c20-4e0a-b741-54fbe96d273e","resolution":{"observed_at":"2026-08-07T14:57:57.483673Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.16784","last_updated":"2025-06-08T02:36:43Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T14:52:57.917127Z","submitted_at":"2025-05-22T15:27:31Z","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2505.16784."}