{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:O7IAQ3K5RY2VC2H2INUZJPTQ42","short_pith_number":"pith:O7IAQ3K5","schema_version":"1.0","canonical_sha256":"77d0086d5d8e355168fa436994be70e6b569599875e9886a86c4a57a879c306d","source":{"kind":"arxiv","id":"2303.00855","version":2},"attestation_state":"computed","paper":{"title":"Grounded Decoding: Guiding Text Generation with Grounded Models for Embodied Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Andy Zeng, Brian Ichter, Danny Driess, Dhruv Shah, Fei Xia, Igor Mordatch, Karol Hausman, Pete Florence, Sergey Levine, Wenlong Huang, Yao Lu","submitted_at":"2023-03-01T22:58:50Z","abstract_excerpt":"Recent progress in large language models (LLMs) has demonstrated the ability to learn and leverage Internet-scale knowledge through pre-training with autoregressive models. Unfortunately, applying such models to settings with embodied agents, such as robots, is challenging due to their lack of experience with the physical world, inability to parse non-language observations, and ignorance of rewards or safety constraints that robots may require. On the other hand, language-conditioned robotic policies that learn from interaction data can provide the necessary grounding that allows the agent to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.00855","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-03-01T22:58:50Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"634ee125f7070013b6072d098107a65f6c949c1ba3297ae5be3d37079cf8efa6","abstract_canon_sha256":"f22defb02edef7d90eb95d3dee13bb4b0a9bef5ed6248d0fcb9f57f5da3a0e50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:22:49.528615Z","signature_b64":"Sbx+PQ0yGQj1A3NRp7FtjOwz3sS4B/Twas5FyfG8CD6xJaVkehaDL/lb/VsTWD3yWN3hfDiU84DPeMLUaYIKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77d0086d5d8e355168fa436994be70e6b569599875e9886a86c4a57a879c306d","last_reissued_at":"2026-07-05T07:22:49.526036Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:22:49.526036Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grounded Decoding: Guiding Text Generation with Grounded Models for Embodied Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Andy Zeng, Brian Ichter, Danny Driess, Dhruv Shah, Fei Xia, Igor Mordatch, Karol Hausman, Pete Florence, Sergey Levine, Wenlong Huang, Yao Lu","submitted_at":"2023-03-01T22:58:50Z","abstract_excerpt":"Recent progress in large language models (LLMs) has demonstrated the ability to learn and leverage Internet-scale knowledge through pre-training with autoregressive models. Unfortunately, applying such models to settings with embodied agents, such as robots, is challenging due to their lack of experience with the physical world, inability to parse non-language observations, and ignorance of rewards or safety constraints that robots may require. On the other hand, language-conditioned robotic policies that learn from interaction data can provide the necessary grounding that allows the agent to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.00855","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.00855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.00855","created_at":"2026-07-05T07:22:49.527938+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.00855v2","created_at":"2026-07-05T07:22:49.527938+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.00855","created_at":"2026-07-05T07:22:49.527938+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7IAQ3K5RY2V","created_at":"2026-07-05T07:22:49.527938+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7IAQ3K5RY2VC2H2","created_at":"2026-07-05T07:22:49.527938+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7IAQ3K5","created_at":"2026-07-05T07:22:49.527938+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07999","citing_title":"Efficient Skill Grounding via Code Refactoring with Small Language Models","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10446","citing_title":"VeriGraph: Scene Graphs for Execution Verifiable Robot Planning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16615","citing_title":"LLM-Guided Task- and Affordance-Level Exploration in Reinforcement Learning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42","json":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42.json","graph_json":"https://pith.science/api/pith-number/O7IAQ3K5RY2VC2H2INUZJPTQ42/graph.json","events_json":"https://pith.science/api/pith-number/O7IAQ3K5RY2VC2H2INUZJPTQ42/events.json","paper":"https://pith.science/paper/O7IAQ3K5"},"agent_actions":{"view_html":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42","download_json":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42.json","view_paper":"https://pith.science/paper/O7IAQ3K5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.00855&json=true","fetch_graph":"https://pith.science/api/pith-number/O7IAQ3K5RY2VC2H2INUZJPTQ42/graph.json","fetch_events":"https://pith.science/api/pith-number/O7IAQ3K5RY2VC2H2INUZJPTQ42/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42/action/storage_attestation","attest_author":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42/action/author_attestation","sign_citation":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42/action/citation_signature","submit_replication":"https://pith.science/pith/O7IAQ3K5RY2VC2H2INUZJPTQ42/action/replication_record"}},"created_at":"2026-07-05T07:22:49.527938+00:00","updated_at":"2026-07-05T07:22:49.527938+00:00"}