{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J2BUROLBAOFGIYQZ4BJBKKOUCF","short_pith_number":"pith:J2BUROLB","schema_version":"1.0","canonical_sha256":"4e8348b961038a646219e0521529d4115258185b2f1194d9f47b740315e3283d","source":{"kind":"arxiv","id":"2411.16627","version":2},"attestation_state":"computed","paper":{"title":"Inference-Time Policy Steering through Human Interactions","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Balakumar Sundaralingam, Claudia Perez-D'Arpino, Dieter Fox, Julie Shah, Lirui Wang, Xuning Yang, Yanwei Wang, Yilun Du, Yu-Wei Chao","submitted_at":"2024-11-25T18:03:50Z","abstract_excerpt":"Generative policies trained with human demonstrations can autonomously accomplish multimodal, long-horizon tasks. However, during inference, humans are often removed from the policy execution loop, limiting the ability to guide a pre-trained policy towards a specific sub-goal or trajectory shape among multiple predictions. Naive human intervention may inadvertently exacerbate distribution shift, leading to constraint violations or execution failures. To better align policy output with human intent without inducing out-of-distribution errors, we propose an Inference-Time Policy Steering (ITPS) "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.16627","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.RO","submitted_at":"2024-11-25T18:03:50Z","cross_cats_sorted":["cs.AI","cs.HC","cs.LG"],"title_canon_sha256":"a1878a23ba942fc50e4932f9bb66fef42a2f01d43ddeee378462ce55cc93c708","abstract_canon_sha256":"e91ce7fc1c5819ceac977755c536654994e8567a701eda3ee0ba405822a4136e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:23.140472Z","signature_b64":"/uProjw1wNWbMd2GucKmdJSKlQfms2+RMNSNm4jhdi3wascx+pbjIJsmczdjxmf6/ku4uXu2E89HEE/jrVWpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e8348b961038a646219e0521529d4115258185b2f1194d9f47b740315e3283d","last_reissued_at":"2026-07-05T10:39:23.139917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:23.139917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference-Time Policy Steering through Human Interactions","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Balakumar Sundaralingam, Claudia Perez-D'Arpino, Dieter Fox, Julie Shah, Lirui Wang, Xuning Yang, Yanwei Wang, Yilun Du, Yu-Wei Chao","submitted_at":"2024-11-25T18:03:50Z","abstract_excerpt":"Generative policies trained with human demonstrations can autonomously accomplish multimodal, long-horizon tasks. However, during inference, humans are often removed from the policy execution loop, limiting the ability to guide a pre-trained policy towards a specific sub-goal or trajectory shape among multiple predictions. Naive human intervention may inadvertently exacerbate distribution shift, leading to constraint violations or execution failures. To better align policy output with human intent without inducing out-of-distribution errors, we propose an Inference-Time Policy Steering (ITPS) "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16627","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.16627/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.16627","created_at":"2026-07-05T10:39:23.139973+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.16627v2","created_at":"2026-07-05T10:39:23.139973+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16627","created_at":"2026-07-05T10:39:23.139973+00:00"},{"alias_kind":"pith_short_12","alias_value":"J2BUROLBAOFG","created_at":"2026-07-05T10:39:23.139973+00:00"},{"alias_kind":"pith_short_16","alias_value":"J2BUROLBAOFGIYQZ","created_at":"2026-07-05T10:39:23.139973+00:00"},{"alias_kind":"pith_short_8","alias_value":"J2BUROLB","created_at":"2026-07-05T10:39:23.139973+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21014","citing_title":"BayesFP: Posterior Estimation for Flow-Based Policies via Feynman-Kac Sampling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10180","citing_title":"Flow Control: Steering Vision-Language-Action Models with Simple Real-Time Inputs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15757","citing_title":"You've Got a Golden Ticket: Improving Generative Robot Policies With A Single Noise Vector","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05855","citing_title":"DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF","json":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF.json","graph_json":"https://pith.science/api/pith-number/J2BUROLBAOFGIYQZ4BJBKKOUCF/graph.json","events_json":"https://pith.science/api/pith-number/J2BUROLBAOFGIYQZ4BJBKKOUCF/events.json","paper":"https://pith.science/paper/J2BUROLB"},"agent_actions":{"view_html":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF","download_json":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF.json","view_paper":"https://pith.science/paper/J2BUROLB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.16627&json=true","fetch_graph":"https://pith.science/api/pith-number/J2BUROLBAOFGIYQZ4BJBKKOUCF/graph.json","fetch_events":"https://pith.science/api/pith-number/J2BUROLBAOFGIYQZ4BJBKKOUCF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF/action/storage_attestation","attest_author":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF/action/author_attestation","sign_citation":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF/action/citation_signature","submit_replication":"https://pith.science/pith/J2BUROLBAOFGIYQZ4BJBKKOUCF/action/replication_record"}},"created_at":"2026-07-05T10:39:23.139973+00:00","updated_at":"2026-07-05T10:39:23.139973+00:00"}