{"work":{"id":"9c86ed28-ea70-424c-bd56-34f59dcad861","openalex_id":"https://openalex.org/W2776202271","doi":"10.48550/arxiv.1712.05474","arxiv_id":"1712.05474","raw_key":null,"title":"AI2-THOR: An Interactive 3D Environment for Visual AI","authors":null,"authors_text":"Eric Kolve, Roozbeh Mottaghi, Winson Han, Eli VanderBilt, Luca Weihs, Alvaro Herrasti","year":2017,"venue":"cs.CV","abstract":"We introduce The House Of inteRactions (THOR), a framework for visual AI research, available at http://ai2thor.allenai.org. AI2-THOR consists of near photo-realistic 3D indoor scenes, where AI agents can navigate in the scenes and interact with objects to perform tasks. AI2-THOR enables research in many different domains including but not limited to deep reinforcement learning, imitation learning, learning by interaction, planning, visual question answering, unsupervised representation learning, object detection and segmentation, and learning models of cognition. The goal of AI2-THOR is to facilitate building visually intelligent models and push the research forward in this domain.","external_url":"https://arxiv.org/abs/1712.05474","cited_by_count":327,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"1712.05474","created_at":"2026-05-09T22:44:15.208471+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"AI2-THOR: An Interactive 3D Environment for Visual AI","render_title":"AI2-THOR: An Interactive 3D Environment for Visual AI"},"hub":{"state":{"work_id":"9c86ed28-ea70-424c-bd56-34f59dcad861","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":94,"external_cited_by_count":327,"distinct_field_count":4,"first_pith_cited_at":"2018-07-18T03:28:02+00:00","last_pith_cited_at":"2026-07-07T17:02:09+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T20:39:18.755107+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":15},{"context_role":"dataset","n":8}],"polarity_counts":[{"context_polarity":"background","n":17},{"context_polarity":"use_dataset","n":6}],"runs":{"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T21:26:18.308991+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","work_id":"037320f1-b0a9-4cbe-a639-bfb25409ce71","shared_citers":7},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":7},{"title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","work_id":"fa436f46-ec0a-4d2e-a0ff-e697def4a7be","shared_citers":6},{"title":"Gemini: A Family of Highly Capable Multimodal Models","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","shared_citers":6},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":6},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":5},{"title":"Objectnav revisited: On evaluation of embodied agents navigating to objects.CoRR, abs/2006.13171","work_id":"2559f113-5e2a-419e-8aba-d15f8f50737c","shared_citers":5},{"title":"On Evaluation of Embodied Navigation Agents","work_id":"3b074aa9-2ff9-4ad6-8796-6a25689ecfd3","shared_citers":5},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":5},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":5},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":4},{"title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","work_id":"b1e694c6-fe5b-477f-ab8f-801e0fb0412f","shared_citers":4},{"title":"Gemini robotics 1.5: Pushing the frontier of generalist robots with advanced embodied reasoning, thinking, and motion transfer","work_id":"ffe7b7f4-997b-4957-b253-03cfbebf6f9a","shared_citers":4},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":4},{"title":"igibson 2.0: Object-centric simulation for robot learning of everyday household tasks","work_id":"e7427c0a-48e2-4cff-81e8-e2063f12e1da","shared_citers":4},{"title":"Inner Monologue: Embodied Reasoning through Planning with Language Models","work_id":"f6e5e4a1-e34b-4602-a7ad-df0c6103a4d0","shared_citers":4},{"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","shared_citers":4},{"title":"PaLM-E: An Embodied Multimodal Language Model","work_id":"5b99811a-1d93-47e2-9d59-f4045a0b74a2","shared_citers":4},{"title":"RT-1: Robotics Transformer for Real-World Control at Scale","work_id":"e11bda85-8531-46bc-a07f-d0ade3643ab1","shared_citers":4},{"title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","shared_citers":3},{"title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","shared_citers":3},{"title":"arXiv preprint arXiv:2012.02924 (2020),https: //arxiv.org/abs/2012.029243","work_id":"c53628e7-ca07-4cd8-9261-a31871f1b1d9","shared_citers":3},{"title":"Bear, Dan Gutfreund, David Cox, Antonio Torralba, James J","work_id":"0b8d8f1c-f7a4-44cc-832d-8f346e0b1fc4","shared_citers":3},{"title":"Cosmos World Foundation Model Platform for Physical AI","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","shared_citers":3}],"time_series":[{"n":1,"year":2018},{"n":1,"year":2020},{"n":1,"year":2021},{"n":1,"year":2023},{"n":1,"year":2024},{"n":30,"year":2026}],"dependency_candidates":[]},"error":null,"updated_at":"2026-05-14T21:26:18.368163+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T21:26:21.587684+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"AI2-THOR: An Interactive 3D Environment for Visual AI","claims":[{"claim_text":"We introduce The House Of inteRactions (THOR), a framework for visual AI research, available at http://ai2thor.allenai.org. AI2-THOR consists of near photo-realistic 3D indoor scenes, where AI agents can navigate in the scenes and interact with objects to perform tasks. AI2-THOR enables research in many different domains including but not limited to deep reinforcement learning, imitation learning, learning by interaction, planning, visual question answering, unsupervised representation learning, object detection and segmentation, and learning models of cognition. The goal of AI2-THOR is to fac","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks AI2-THOR: An Interactive 3D Environment for Visual AI because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T21:26:18.315038+00:00"}},"summary":{"title":"AI2-THOR: An Interactive 3D Environment for Visual AI","claims":[{"claim_text":"We introduce The House Of inteRactions (THOR), a framework for visual AI research, available at http://ai2thor.allenai.org. AI2-THOR consists of near photo-realistic 3D indoor scenes, where AI agents can navigate in the scenes and interact with objects to perform tasks. AI2-THOR enables research in many different domains including but not limited to deep reinforcement learning, imitation learning, learning by interaction, planning, visual question answering, unsupervised representation learning, object detection and segmentation, and learning models of cognition. The goal of AI2-THOR is to fac","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks AI2-THOR: An Interactive 3D Environment for Visual AI because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","work_id":"037320f1-b0a9-4cbe-a639-bfb25409ce71","shared_citers":7},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":7},{"title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","work_id":"fa436f46-ec0a-4d2e-a0ff-e697def4a7be","shared_citers":6},{"title":"Gemini: A Family of Highly Capable Multimodal Models","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","shared_citers":6},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":6},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":5},{"title":"Objectnav revisited: On evaluation of embodied agents navigating to objects.CoRR, abs/2006.13171","work_id":"2559f113-5e2a-419e-8aba-d15f8f50737c","shared_citers":5},{"title":"On Evaluation of Embodied Navigation Agents","work_id":"3b074aa9-2ff9-4ad6-8796-6a25689ecfd3","shared_citers":5},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":5},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":5},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":4},{"title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","work_id":"b1e694c6-fe5b-477f-ab8f-801e0fb0412f","shared_citers":4},{"title":"Gemini robotics 1.5: Pushing the frontier of generalist robots with advanced embodied reasoning, thinking, and motion transfer","work_id":"ffe7b7f4-997b-4957-b253-03cfbebf6f9a","shared_citers":4},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":4},{"title":"igibson 2.0: Object-centric simulation for robot learning of everyday household tasks","work_id":"e7427c0a-48e2-4cff-81e8-e2063f12e1da","shared_citers":4},{"title":"Inner Monologue: Embodied Reasoning through Planning with Language Models","work_id":"f6e5e4a1-e34b-4602-a7ad-df0c6103a4d0","shared_citers":4},{"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","shared_citers":4},{"title":"PaLM-E: An Embodied Multimodal Language Model","work_id":"5b99811a-1d93-47e2-9d59-f4045a0b74a2","shared_citers":4},{"title":"RT-1: Robotics Transformer for Real-World Control at Scale","work_id":"e11bda85-8531-46bc-a07f-d0ade3643ab1","shared_citers":4},{"title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","shared_citers":3},{"title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","shared_citers":3},{"title":"arXiv preprint arXiv:2012.02924 (2020),https: //arxiv.org/abs/2012.029243","work_id":"c53628e7-ca07-4cd8-9261-a31871f1b1d9","shared_citers":3},{"title":"Bear, Dan Gutfreund, David Cox, Antonio Torralba, James J","work_id":"0b8d8f1c-f7a4-44cc-832d-8f346e0b1fc4","shared_citers":3},{"title":"Cosmos World Foundation Model Platform for Physical AI","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","shared_citers":3}],"time_series":[{"n":1,"year":2018},{"n":1,"year":2020},{"n":1,"year":2021},{"n":1,"year":2023},{"n":1,"year":2024},{"n":30,"year":2026}],"dependency_candidates":[]},"authors":[]}}