{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:BIGOSGWK2SVMWPUOOHQTLI7OI4","short_pith_number":"pith:BIGOSGWK","schema_version":"1.0","canonical_sha256":"0a0ce91acad4aacb3e8e71e135a3ee471bf1da9d453560b1954941bec5793060","source":{"kind":"arxiv","id":"1806.03831","version":1},"attestation_state":"computed","paper":{"title":"Interactive Visual Grounding of Referring Expressions for Human-Robot Interaction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"David Hsu, Mohit Shridhar","submitted_at":"2018-06-11T06:58:19Z","abstract_excerpt":"This paper presents INGRESS, a robot system that follows human natural language instructions to pick and place everyday objects. The core issue here is the grounding of referring expressions: infer objects and their relationships from input images and language expressions. INGRESS allows for unconstrained object categories and unconstrained language expressions. Further, it asks questions to disambiguate referring expressions interactively. To achieve these, we take the approach of grounding by generation and propose a two-stage neural network model for grounding. The first stage uses a neural"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1806.03831","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2018-06-11T06:58:19Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"15f754828c6f44c5d6b635ddeef2457be16d31e0f9ad43a67a8114d15b6cdfcc","abstract_canon_sha256":"e57c33d411e4c58663b523c562dbad6f70ab2957e5329f5667556f1d478e2c1f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:13:41.932109Z","signature_b64":"as8vuSB0KdW95UbhqFgjBOmsyXj0zuNILPd8jy7ezrsEPNJdRgWSQILzk6fsRve6FGS0BO24Y0Q5ONLUY91PBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a0ce91acad4aacb3e8e71e135a3ee471bf1da9d453560b1954941bec5793060","last_reissued_at":"2026-05-18T00:13:41.931369Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:13:41.931369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interactive Visual Grounding of Referring Expressions for Human-Robot Interaction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"David Hsu, Mohit Shridhar","submitted_at":"2018-06-11T06:58:19Z","abstract_excerpt":"This paper presents INGRESS, a robot system that follows human natural language instructions to pick and place everyday objects. The core issue here is the grounding of referring expressions: infer objects and their relationships from input images and language expressions. INGRESS allows for unconstrained object categories and unconstrained language expressions. Further, it asks questions to disambiguate referring expressions interactively. To achieve these, we take the approach of grounding by generation and propose a two-stage neural network model for grounding. The first stage uses a neural"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.03831","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1806.03831","created_at":"2026-05-18T00:13:41.931491+00:00"},{"alias_kind":"arxiv_version","alias_value":"1806.03831v1","created_at":"2026-05-18T00:13:41.931491+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.03831","created_at":"2026-05-18T00:13:41.931491+00:00"},{"alias_kind":"pith_short_12","alias_value":"BIGOSGWK2SVM","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_16","alias_value":"BIGOSGWK2SVMWPUO","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_8","alias_value":"BIGOSGWK","created_at":"2026-05-18T12:32:16.446611+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21250","citing_title":"ACTLLM: Action Consistency Tuned Large Language Model","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4","json":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4.json","graph_json":"https://pith.science/api/pith-number/BIGOSGWK2SVMWPUOOHQTLI7OI4/graph.json","events_json":"https://pith.science/api/pith-number/BIGOSGWK2SVMWPUOOHQTLI7OI4/events.json","paper":"https://pith.science/paper/BIGOSGWK"},"agent_actions":{"view_html":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4","download_json":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4.json","view_paper":"https://pith.science/paper/BIGOSGWK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1806.03831&json=true","fetch_graph":"https://pith.science/api/pith-number/BIGOSGWK2SVMWPUOOHQTLI7OI4/graph.json","fetch_events":"https://pith.science/api/pith-number/BIGOSGWK2SVMWPUOOHQTLI7OI4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4/action/storage_attestation","attest_author":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4/action/author_attestation","sign_citation":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4/action/citation_signature","submit_replication":"https://pith.science/pith/BIGOSGWK2SVMWPUOOHQTLI7OI4/action/replication_record"}},"created_at":"2026-05-18T00:13:41.931491+00:00","updated_at":"2026-05-18T00:13:41.931491+00:00"}