{"as_of":"2026-08-22T17:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3c384b2618f21d430af933a2c309f3ba325dfde5e05a4ceaac6b0dfbae3eb69e","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:51:00.548967Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-19T11:12:41.130806Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-19T11:13:02.889016Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2411.16761","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16761","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is’ right’right? enhancing object orientation understanding in multimodal language models through egocentric instruction tuning","venue":null,"work_id":"d75495f4-481a-4ec1-a4b5-31d7d825951b","year":2024},"citing_paper":{"arxiv_id":"2506.09082","last_updated":"2026-05-03T04:37:38Z","snapshot_observed_at":"2026-08-18T09:58:59.618504Z","submitted_at":"2025-06-10T05:43:34Z","title":"AVA-Bench: Atomic Visual Ability Benchmark for Vision Foundation Models","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-19T11:12:41.130806Z"},"links":{"cited_paper":"/paper/2411.16761","citing_paper":"/paper/2506.09082"},"observation_digest":"sha256:ca670de5d5fa4b271726640537e0a63933f06f070662a4a5177a2173f63ea3b0","observation_id":"f9d7873b-8f6f-47f2-8402-09bbd1246c7f","resolution":{"observed_at":"2026-05-19T11:13:02.890631Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2411.16761","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16761","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is’ right’right? enhancing object orientation understanding in multimodal language models through egocentric instruction tuning","venue":null,"work_id":"d75495f4-481a-4ec1-a4b5-31d7d825951b","year":2024},"citing_paper":{"arxiv_id":"2604.04746","last_updated":"2026-04-08T01:34:51Z","snapshot_observed_at":"2026-08-20T01:52:18.683653Z","submitted_at":"2026-04-06T15:11:57Z","title":"Think in Strokes, Not Pixels: Process-Driven Image Generation via Interleaved Reasoning","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T19:16:58.323955Z"},"links":{"cited_paper":"/paper/2411.16761","citing_paper":"/paper/2604.04746"},"observation_digest":"sha256:f8ea6c3cc2c6fad486f2f5350f61d252b73f4e53b36b2a0e1d5e9f7d6891a853","observation_id":"390e76f5-04cc-4123-aafa-e0252ab906b1","resolution":{"observed_at":"2026-05-10T23:10:53.787696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.16761/citation-record","integrity":"/paper/2411.16761/integrity","json":"/paper/2411.16761/citation-record.json","paper":"/paper/2411.16761"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T13:51:00.328980Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.328980Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:86cd41dd068a4a2e845cf7cc4efa9c20adcd158841284e6e5c28efb2716ed8fa","observation_id":"98657fc5-a208-4ada-b229-c5b56bdc3602","resolution":{"observed_at":"2026-08-12T13:51:00.328980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.333999Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.333999Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:74163cafb81726e18d380c35f7d9b0ac68d5eee43680b573b453a55b6820a5c6","observation_id":"92df7e73-02b3-4a55-b01c-0f6accf25c7b","resolution":{"observed_at":"2026-08-12T13:51:00.333999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.378416Z","title":"Claude 3.5 sonnet model card addendum","venue":null,"work_id":"09ee95d6-5d14-4728-bd54-554b58392906","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.338497Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:5224f40c0624d828926c08c8092809e36e4b8c4fdf1b4be75f7763d9ffa3d6a1","observation_id":"8d042221-d275-4515-b938-ba29d7cd1c47","resolution":{"observed_at":"2026-08-12T13:51:01.383100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-12T13:51:00.342915Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.342915Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:c5bdc16392f9b6df232bf52ce04477e5a3f9ac52ddbb81d8fa19bfff212a479b","observation_id":"9d7703ce-7359-454d-84ce-77bc2a1a4653","resolution":{"observed_at":"2026-08-12T13:51:00.342915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.347288Z","title":"Egocentric vehicle dense video captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.347288Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:f68350def5d8a125536f397e37dab109742cb81f3662d5ce97b860eb8423e12f","observation_id":"d5c3bd4e-2345-4f79-8f22-a7f799a9f909","resolution":{"observed_at":"2026-08-12T13:51:00.347288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-07T12:15:30.838846Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20330","snapshot_observed_at":"2026-08-12T13:51:00.351623Z","title":"Are we on the right way for evaluating large vision-language models? arXiv preprint arXiv:2403.20330,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.351623Z"},"links":{"cited_paper":"/paper/2403.20330","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:0aa5e57be6787d1645285f863cb0c67652b24fc535861523281d387bd4d89f68","observation_id":"3652958d-4a66-4fb2-abf7-20a1e0836eaf","resolution":{"observed_at":"2026-08-12T13:51:00.351623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.356764Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":"3322855c-2d51-4cfb-9a09-f91716b34a43","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.356234Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:bb8175e56822c190df5782aab58d6eb35b665685dbfe8ef700fc315b05708709","observation_id":"7d66eed1-557e-4b08-afad-00300d730aaa","resolution":{"observed_at":"2026-08-12T13:51:01.361261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.342600Z","title":"A benchmark for reason- ing with spatial prepositions","venue":null,"work_id":"5263abd9-e61d-49d4-964e-9bd02671ac2e","year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.360458Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:ccb3133ca3a6b8ea41bb63b48703b0b60c3534948559298632cd67cc9d4b29eb","observation_id":"e8a86ecc-dfb7-4ca6-81b6-dcb33b12e495","resolution":{"observed_at":"2026-08-12T13:51:01.347413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.364493Z","title":"Instructblip: Towards general- purpose vision-language models with instruction tuning,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.364493Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:fe689cb5ac11cc418f6e6fedf1eea7b5e811a6bc4bee8e46d3dcb11c36f8a03c","observation_id":"dad0ab57-fb61-42bb-8e0a-b7e920424163","resolution":{"observed_at":"2026-08-12T13:51:00.364493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.319655Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":"cc0f1ac7-e9ed-45fd-b2c1-40d960efc261","year":2009},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.368658Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:eb531ab970574db76d71f77ec69e21aa5c04187ef066559390b4d70f689df677","observation_id":"7aa87d5a-031b-4b8e-8a3b-eccbd0dd0010","resolution":{"observed_at":"2026-08-12T13:51:01.324596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.302404Z","title":"Gpv-pose: Category-level object pose estimation via geometry-guided point-wise voting","venue":null,"work_id":"f8bf7b59-aea1-498d-9d72-4ee14e21a67c","year":2022},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.372958Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:0158616b367b2530ca30715b3e54e0462819eeae3314775e4fb7c4541c1d8081","observation_id":"65f975cd-6008-4083-bb41-586171b570ae","resolution":{"observed_at":"2026-08-12T13:51:01.308194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.284505Z","title":"Pedestrian detection: An evaluation of the state of the art","venue":null,"work_id":"3590d5f5-f4bc-4c9e-bb21-439327264bae","year":2011},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.376989Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:c18bf69944a372b96274f64c3df13d0c16266fca560fda05e8abe37c0f2b6896","observation_id":"6260e3ea-0abc-4d22-9a38-23913b96f1df","resolution":{"observed_at":"2026-08-12T13:51:01.289802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.269363Z","title":"Pedestrian movement direction recognition using convolutional neural networks","venue":null,"work_id":"86690dc1-e55d-435b-92c6-0073d29b80b9","year":2017},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.380797Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:c981172bd0457fab45f878d921344931b413ebc2a0bd38e1736f13c611f4be36","observation_id":"0debd8bf-b013-495e-8e81-4a24490c601b","resolution":{"observed_at":"2026-08-12T13:51:01.273896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.254359Z","title":"Palm-e: An embodied multimodal language model","venue":null,"work_id":"18b89616-138f-4ad5-a11b-b0cde276cea0","year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.384377Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:53f3b7924313410305f79653f4cc68061c5fff09f4d6d343c21aee94f37d07ab","observation_id":"468b7edc-3849-497a-90b3-76bd82053dda","resolution":{"observed_at":"2026-08-12T13:51:01.259070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.240840Z","title":"Integrated pedes- trian classification and orientation estimation","venue":null,"work_id":"471c00bf-c73c-4529-a9f0-dda2918cf67c","year":2010},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.388481Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:c0e4829e7d6b522d955d522a094b894617371ac3483542ff27ebc9b3cf46809c","observation_id":"166481fb-dcd3-4cb3-9471-eb55c41bb38f","resolution":{"observed_at":"2026-08-12T13:51:01.245433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-12T13:51:00.392494Z","title":"Mme: A comprehensive evaluation bench- mark for multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.392494Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:c4832560936e97e0723812064f9dd28e62f7837c01441d02136d62142c90c2f4","observation_id":"f6285ba5-5f0a-4c36-a443-9e336c2b9168","resolution":{"observed_at":"2026-08-12T13:51:00.392494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.226875Z","title":"Image based estimation of pedestrian orientation for improving path pre- diction","venue":null,"work_id":"8b5cb283-0854-4723-b482-a9dfad54e1f4","year":2008},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.397035Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:975b37fbf9efb319be3559b76f7318c53f98b698193f93f59b29d0224fc6593f","observation_id":"f4eba972-771a-4015-a955-4f89dc31be2f","resolution":{"observed_at":"2026-08-12T13:51:01.231701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.211168Z","title":"Phys- ically grounded vision-language models for robotic manipu- lation","venue":null,"work_id":"55353539-ce69-46cb-bc45-5355674086de","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.401239Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:2e48b816444b4e6554d6d05e055d985d23507399e71b55a9a00703795b64b9d1","observation_id":"85b136eb-d572-46ef-86b2-6850ecb7323d","resolution":{"observed_at":"2026-08-12T13:51:01.217051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.15125","last_updated":"2024-09-23T15:31:25Z","snapshot_observed_at":"2026-08-20T02:55:28.869949Z","submitted_at":"2024-09-23T15:31:25Z","title":"Detect, Describe, Discriminate: Moving Beyond VQA for MLLM Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.15125","snapshot_observed_at":"2026-08-12T13:51:00.405551Z","title":"Detect, describe, dis- criminate: Moving beyond vqa for mllm evaluation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.405551Z"},"links":{"cited_paper":"/paper/2409.15125","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:07f8e9fcaec44e32945290ebf632f5eacc98413c799b499d30f06ad0414d261c","observation_id":"c2a4c464-c645-4b87-be3f-806d71ab2873","resolution":{"observed_at":"2026-08-12T13:51:00.405551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.194972Z","title":"Human brain dynamics accompanying use of egocentric and allocentric reference frames during navigation","venue":null,"work_id":"16b9a172-84c9-45bb-ac89-612f211cc925","year":2010},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.410766Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:771c9431cb886f1a13e3d0e8858249d7a158b848a30735a621a703116e627e1d","observation_id":"5b3f2ce3-ce83-4c87-9ba7-2635f92b801b","resolution":{"observed_at":"2026-08-12T13:51:01.199835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.415796Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.415796Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:e2b83261bde8cd214fa73756df2db3aa37b610f54be3d9f438dcf2264efdfaff","observation_id":"aa94c7c4-0278-4840-baee-96c27dd51a17","resolution":{"observed_at":"2026-08-12T13:51:00.415796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.173066Z","title":"Loc-zson: Language-driven object-centric zero-shot object retrieval and navigation","venue":null,"work_id":"d06c4d5a-ba88-420c-9818-00b03f2efbc0","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.420453Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:ffcddbc2b4cac30e9ce1841940c958c422a63d3c97a410db8f543ae94bb22d60","observation_id":"3bc83d3c-d248-4095-bb19-9cab35fd42e9","resolution":{"observed_at":"2026-08-12T13:51:01.177741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.155101Z","title":"Lora: Low- rank adaptation of large language models","venue":null,"work_id":"cf9953db-b67e-4722-b66b-2ac20a8ace52","year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.424618Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:f3f1c3da3744f7c2d2f7cf4d30d6dc1568e474a132eee0b038df701338f60812","observation_id":"49d902cb-5626-41ec-b50b-a6afef179bcf","resolution":{"observed_at":"2026-08-12T13:51:01.159976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.139910Z","title":"Egoexolearn: A dataset for bridging asyn- chronous ego-and exo-centric view of procedural activities in real world","venue":null,"work_id":"7fd34db8-9f15-47af-a7ab-99ec6657c9b4","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.429428Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:776ad6105fb97993cf21df4d5deeb7495e87bccced6a56657b4894ca91b3e091","observation_id":"3825dca1-753e-4a4f-83a7-3065d42a2508","resolution":{"observed_at":"2026-08-12T13:51:01.145535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.120390Z","title":"What’s “up” with vision-language models? investigating their strug- gle with spatial reasoning","venue":null,"work_id":"249aa775-d7a4-487c-8d3e-620f3a8d67e0","year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.433937Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:4a5be2442a5cd5af40aa0a7aae82814365aeaab8a7355104368d7938ac2d7334","observation_id":"30254ba8-f78c-4466-a255-87442d5bdf37","resolution":{"observed_at":"2026-08-12T13:51:01.126517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.100859Z","title":"Frames of reference and molyneux’s question: Crosslinguistic evidence","venue":null,"work_id":"2bc91844-958b-470c-a47c-483548bc3aa7","year":1996},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.438133Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:2df1bae19f1b884c67ee3e41afcc0426f259794ba4cd77917bdbed301c22f766","observation_id":"c9f4f3f9-f795-42f9-b4a4-bb8c3b1aff51","resolution":{"observed_at":"2026-08-12T13:51:01.106047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.085452Z","title":"Deeper, broader and artier domain generaliza- tion","venue":null,"work_id":"321ebe69-5673-499f-80d6-6e9e241f222f","year":2017},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.441932Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:80e1a2b95218bdbce67cdfef55785c094f3c803474f522297fbf2ed7ddd0b0df","observation_id":"f01eadcb-cf25-414b-91d5-35854a66354c","resolution":{"observed_at":"2026-08-12T13:51:01.090421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-08-12T18:48:30.326248Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-08-12T13:51:00.446313Z","title":"Evaluating object hallucina- tion in large vision-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.446313Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:2536511aa47db09d22fe9de1170a94aa4aea7ed7e466b34b90f22698a79b799f","observation_id":"f9f762bb-497d-4eca-b956-d567e5a32bc4","resolution":{"observed_at":"2026-08-12T13:51:00.446313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.450789Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.450789Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:7ca7e8cacd0fd00213b32889fa6d3a407a7edc853d0bef2d700bcb37c2c0b18c","observation_id":"2aa98f29-2aae-4a5d-9c04-89eed2b1baf2","resolution":{"observed_at":"2026-08-12T13:51:00.450789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:01.062900Z","title":"Visual instruction tuning","venue":null,"work_id":"80fbc284-c702-40a4-a5b7-31ef71fa13ba","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.454515Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:675d0a5eebe4e93d7b8448ac8e99a5e99a34c378b6d8e16ae8d1a4d2c6215061","observation_id":"0edff087-e152-4151-ab17-17ed8fb502d7","resolution":{"observed_at":"2026-08-12T13:51:01.067161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.458219Z","title":"Decoupled weight de- cay regularization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.458219Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:1c13089226c7d44c02cc0ade34b99ae54b00f481cf161f5213244c8454dc07e9","observation_id":"5f68169d-043f-4477-a9c6-c61676db6a2c","resolution":{"observed_at":"2026-08-12T13:51:00.458219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.462256Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.462256Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:79c687a6c0bdcc96ab39fcb7c86fe4d5d2f61106cb016a7bf6948f7c2d2f15db","observation_id":"a5d6e9c2-2afb-419b-978a-cd830d9868ed","resolution":{"observed_at":"2026-08-12T13:51:00.462256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.917261Z","title":"Embodiedgpt: Vision-language pre-training via embodied chain of thought","venue":null,"work_id":"291ebf59-9280-45b5-b9ca-ee7f0d5b6e64","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.465825Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:94ce16911b5659cc4b5121bd6d8bfc9b4b76b3129c44a95609562cdbe2fd7270","observation_id":"bba0b25c-2550-41d1-9cbb-51a87562aa3f","resolution":{"observed_at":"2026-08-12T13:51:01.034723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.902731Z","title":"Pose es- timation for category specific multiview object localization","venue":null,"work_id":"0376e7ad-0378-4012-8b8a-2ff0dfc6a329","year":2009},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.469613Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:ea9e4cc0066f2da1752c18e5157a582efdae10196fcc5c19bc6b534576584064","observation_id":"9406d84a-1528-4af5-a02f-6e5c1e1fe225","resolution":{"observed_at":"2026-08-12T13:51:00.907779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.13034","last_updated":"2024-06-05T21:47:37Z","snapshot_observed_at":"2026-08-19T22:03:00.722001Z","submitted_at":"2024-05-16T14:20:30Z","title":"Autonomous Workflow for Multimodal Fine-Grained Training Assistants Towards Mixed Reality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.13034","snapshot_observed_at":"2026-08-12T13:51:00.473380Z","title":"Autonomous workflow for multi- modal fine-grained training assistants towards mixed reality","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.473380Z"},"links":{"cited_paper":"/paper/2405.13034","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:0ca402990e70c964bce9b10129e6f70aee7bb3486459de4afedcf3d257bf8991","observation_id":"12005bce-e790-4a9b-af33-c1806d6fdbca","resolution":{"observed_at":"2026-08-12T13:51:00.473380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.477740Z","title":"Moment matching for multi-source domain adaptation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.477740Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:f191f684a1b34272203ced2262d49d4dd25d9c87d30ec29dedebaf66bbf12f7f","observation_id":"eb825cd3-e964-4672-ab7d-5c3a1596af12","resolution":{"observed_at":"2026-08-12T13:51:00.477740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-08-12T12:24:23.815073Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-12T13:51:00.481880Z","title":"Kosmos-2: Ground- ing multimodal large language models to the world","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.481880Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:cbc09a03e74df057c212553c75ef39e5912fd216ccf94f385a80553f457dde3f","observation_id":"05368f0b-6c6e-4a22-8417-e7f4f4db5ee3","resolution":{"observed_at":"2026-08-12T13:51:00.481880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.486107Z","title":"Dreamfusion: Text-to-3d using 2d diffusion","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.486107Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:6d5afe3ed9f585743d634e309c645f19fe6e116582a05affe41da03d13edb8d6","observation_id":"08156588-335c-4d2e-939f-55b195cbde3f","resolution":{"observed_at":"2026-08-12T13:51:00.486107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.490426Z","title":"Laion-5b: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.490426Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:7be54d5ad295eb30c6698c693228539079dba05bc741e830e1a44290937a9473","observation_id":"6b02b073-5dcf-4928-82fb-bf9d71c1e174","resolution":{"observed_at":"2026-08-12T13:51:00.490426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.861339Z","title":"Lmdrive: Closed-loop end-to-end driving with large language models","venue":null,"work_id":"3297b3aa-73b6-4ea1-b8cf-c7fd126f321a","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.494259Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:2b0e8a2348dc16a57807b0325d3570fd491f64b4990f4dc9b35ed61004634028","observation_id":"7fed912b-d058-4c8c-94d0-64fc8a1f78d9","resolution":{"observed_at":"2026-08-12T13:51:00.866393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.845278Z","title":"Gemini 1.5: Unlocking multimodal under- standing across millions of tokens of context (2024)","venue":null,"work_id":"39f6dabc-2bff-4741-8f68-8ad05c7ac2e5","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.498377Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:9ef321d71554abe89a51ac3464c3f23cfc6718443e0e2a585bf843f1f9e4b11f","observation_id":"ea9594f9-1af7-4836-882b-80ef7feae745","resolution":{"observed_at":"2026-08-12T13:51:00.850417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.827907Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":null,"work_id":"6f138ea6-66fd-46ea-969d-2d321625de49","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.502355Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:6dcb6a3a2e8bc2531f79f28a27cc59594ddb9f4f9b74069b6cb3a1c03cb41c0f","observation_id":"8a4acee8-9565-4872-a764-39eaea3d3b19","resolution":{"observed_at":"2026-08-12T13:51:00.834661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.808318Z","title":"A fronto-parietal system for computing the egocentric spatial frame of refer- ence in humans","venue":null,"work_id":"bf2f0c72-d0b5-43dc-8191-0ce4e01c8e19","year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.508084Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:2eb766831ce06e8fe2236b9f4b955b27e7101127ad28f0420fd741f6069864a4","observation_id":"14be0946-ce2c-41d8-b6c2-f9fd0333e926","resolution":{"observed_at":"2026-08-12T13:51:00.816525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.514094Z","title":"Visionllm: Large language model is also an open- ended decoder for vision-centric tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.514094Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:78e0a131382fd204eb21ad911acbd59355fb27966536617a1223996d55fd4ac3","observation_id":"aa019bca-add5-494f-a46d-ca74d7eb0e9d","resolution":{"observed_at":"2026-08-12T13:51:00.514094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.518417Z","title":"Holoassist: an egocen- tric human interaction dataset for interactive ai assistants in the real world","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.518417Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:4ea87c98e3765fd8d82c84fc9ebdf0dc2afde8959b3db3e0ae4cc0656aa2ac37","observation_id":"b662f74c-ac89-4e91-8e0e-f8c78be60371","resolution":{"observed_at":"2026-08-12T13:51:00.518417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.523691Z","title":"Editable scene simulation for autonomous driving via collaborative llm-agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.523691Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:e21514b76bc7c21fefc929c3897ebd6426e8f69093f8328a988235b381211cc3","observation_id":"16c38780-a7d7-45ea-9403-66026202efe5","resolution":{"observed_at":"2026-08-12T13:51:00.523691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.764944Z","title":"Omniobject3d: Large-vocabulary 3d object dataset for realistic perception, reconstruction and generation","venue":null,"work_id":"4ce52e2e-03d3-4a87-8abd-08ffd3dd4d3a","year":2023},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.528072Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:e67a05e67db702e761896d7c17d64e1ca9256b06b94a5519a645eb3e78a9411d","observation_id":"751cf349-11d6-48df-bad6-8f3640c01f88","resolution":{"observed_at":"2026-08-12T13:51:00.769966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.532156Z","title":"Retrieval-augmented egocentric video captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.532156Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:1611625b80fbb9483988e21ef0fffc24b0a4cb1dabc8571647824ed9a54b353e","observation_id":"e6056df6-dfc3-4ef8-93b7-d56e6a06ac5a","resolution":{"observed_at":"2026-08-12T13:51:00.532156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.741009Z","title":"mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration","venue":null,"work_id":"b97990d1-c44a-460d-9227-a8e726ac2c0c","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.536393Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:d62042af37a459cf47229cc1ab0280b59c7cdab6ee824ce9ba23482488eb9364","observation_id":"c85efa5a-0a7d-4395-b640-954d00a9471e","resolution":{"observed_at":"2026-08-12T13:51:00.745525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.726738Z","title":"Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for ex- pert agi","venue":null,"work_id":"8cdf0bf9-b6d0-420d-b4bf-ed9b83d96f48","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.540490Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:0b38f6c855ab9bc113f9c30c2b8568a6d00aa953c62577e3598030a2465572d4","observation_id":"dfe4577d-c8ed-4f46-9b5a-d89befcbdc5b","resolution":{"observed_at":"2026-08-12T13:51:00.731522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.711677Z","title":"The neural basis of the egocentric and allocentric spatial frame of refer- ence","venue":null,"work_id":"389740af-fcc2-415b-8d9c-d204bb1a59eb","year":2007},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.544475Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:f3514bc3cc20b4e7064824a39d04f2bd764951eefbfe1dbcb29c2ae836ce6732","observation_id":"c9e08196-fae1-4fdd-b705-42daaaa2cb96","resolution":{"observed_at":"2026-08-12T13:51:00.716897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:51:00.694813Z","title":"yes” or “no","venue":null,"work_id":"2341b8f8-ad1c-4b6b-b0d1-a672d445f834","year":2024},"citing_paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T13:51:00.548967Z"},"links":{"citing_paper":"/paper/2411.16761"},"observation_digest":"sha256:3a3f5cc7076d7e2e3a84a1b91674b335cb766599e5fe82239379a7d695bdd785","observation_id":"4f05f16d-9d49-4f6e-9fa0-21376ae970bb","resolution":{"observed_at":"2026-08-12T13:51:00.700836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.16761","last_updated":"2025-03-29T09:24:00Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T15:04:46.614569Z","submitted_at":"2024-11-24T15:07:47Z","title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":0,"verified_fuzzy":30},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 2 inbound Pith citation observations for arXiv:2411.16761."}