{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LVDMRGJPJ3KXY7SLYMI3AIKNPI","short_pith_number":"pith:LVDMRGJP","schema_version":"1.0","canonical_sha256":"5d46c8992f4ed57c7e4bc311b0214d7a293286e78c2177a94d62e4cfa687c60a","source":{"kind":"arxiv","id":"2312.07062","version":2},"attestation_state":"computed","paper":{"title":"ThinkBot: Embodied Instruction Following with Thought Chain Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changliu Liu, Guanxing Lu, Jiwen Lu, Yansong Tang, Ziwei Wang","submitted_at":"2023-12-12T08:30:09Z","abstract_excerpt":"Embodied Instruction Following (EIF) requires agents to complete human instruction by interacting objects in complicated surrounding environments. Conventional methods directly consider the sparse human instruction to generate action plans for agents, which usually fail to achieve human goals because of the instruction incoherence in action descriptions. On the contrary, we propose ThinkBot that reasons the thought chain in human instruction to recover the missing action descriptions, so that the agent can successfully complete human goals by following the coherent instruction. Specifically, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.07062","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-12T08:30:09Z","cross_cats_sorted":[],"title_canon_sha256":"b35559837502dabe29f52a4668a25b6ca513d409915c145a54297faacba67b9c","abstract_canon_sha256":"0e95fed656fbee8f99d62389e09e61492cdfa645f2cb79578a99d424d8aef096"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:02.957038Z","signature_b64":"jGRJokmkpsiWqIVFquCv+9LvVtT+xUL8Tet+gD+Ld7WKW0VKRFOcnBcYeR2juTiru/R9D/BteAtA2f363GUcCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d46c8992f4ed57c7e4bc311b0214d7a293286e78c2177a94d62e4cfa687c60a","last_reissued_at":"2026-07-05T07:24:02.956577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:02.956577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ThinkBot: Embodied Instruction Following with Thought Chain Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changliu Liu, Guanxing Lu, Jiwen Lu, Yansong Tang, Ziwei Wang","submitted_at":"2023-12-12T08:30:09Z","abstract_excerpt":"Embodied Instruction Following (EIF) requires agents to complete human instruction by interacting objects in complicated surrounding environments. Conventional methods directly consider the sparse human instruction to generate action plans for agents, which usually fail to achieve human goals because of the instruction incoherence in action descriptions. On the contrary, we propose ThinkBot that reasons the thought chain in human instruction to recover the missing action descriptions, so that the agent can successfully complete human goals by following the coherent instruction. Specifically, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.07062","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.07062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.07062","created_at":"2026-07-05T07:24:02.956631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.07062v2","created_at":"2026-07-05T07:24:02.956631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.07062","created_at":"2026-07-05T07:24:02.956631+00:00"},{"alias_kind":"pith_short_12","alias_value":"LVDMRGJPJ3KX","created_at":"2026-07-05T07:24:02.956631+00:00"},{"alias_kind":"pith_short_16","alias_value":"LVDMRGJPJ3KXY7SL","created_at":"2026-07-05T07:24:02.956631+00:00"},{"alias_kind":"pith_short_8","alias_value":"LVDMRGJP","created_at":"2026-07-05T07:24:02.956631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":238,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25851","citing_title":"RePlan-Bot: Multi-Level Replanning for Embodied Instruction Following","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14135","citing_title":"GAF: Gaussian Action Field as a 4D Representation for Dynamic World Modeling in Robotic Manipulation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14093","citing_title":"A Survey on Vision-Language-Action Models for Embodied AI","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI","json":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI.json","graph_json":"https://pith.science/api/pith-number/LVDMRGJPJ3KXY7SLYMI3AIKNPI/graph.json","events_json":"https://pith.science/api/pith-number/LVDMRGJPJ3KXY7SLYMI3AIKNPI/events.json","paper":"https://pith.science/paper/LVDMRGJP"},"agent_actions":{"view_html":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI","download_json":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI.json","view_paper":"https://pith.science/paper/LVDMRGJP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.07062&json=true","fetch_graph":"https://pith.science/api/pith-number/LVDMRGJPJ3KXY7SLYMI3AIKNPI/graph.json","fetch_events":"https://pith.science/api/pith-number/LVDMRGJPJ3KXY7SLYMI3AIKNPI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI/action/storage_attestation","attest_author":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI/action/author_attestation","sign_citation":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI/action/citation_signature","submit_replication":"https://pith.science/pith/LVDMRGJPJ3KXY7SLYMI3AIKNPI/action/replication_record"}},"created_at":"2026-07-05T07:24:02.956631+00:00","updated_at":"2026-07-05T07:24:02.956631+00:00"}