{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:DX42UZA7GY5KCOW53KHNIDHGOV","short_pith_number":"pith:DX42UZA7","schema_version":"1.0","canonical_sha256":"1df9aa641f363aa13addda8ed40ce6755c48ab8a5ae9794a9786090161d75919","source":{"kind":"arxiv","id":"1910.04365","version":1},"attestation_state":"computed","paper":{"title":"Asking Easy Questions: A User-Friendly Approach to Active Reward Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Dorsa Sadigh, Dylan P. Losey, Erdem B{\\i}y{\\i}k, Malayandi Palan, Nicholas C. Landolfi","submitted_at":"2019-10-10T04:52:46Z","abstract_excerpt":"Robots can learn the right reward function by querying a human expert. Existing approaches attempt to choose questions where the robot is most uncertain about the human's response; however, they do not consider how easy it will be for the human to answer! In this paper we explore an information gain formulation for optimally selecting questions that naturally account for the human's ability to answer. Our approach identifies questions that optimize the trade-off between robot and human uncertainty, and determines when these questions become redundant or costly. Simulations and a user study sho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.04365","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-10-10T04:52:46Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"46cbb55bd6e1bf05e38013ebaa576d33cab05b2c6650082afcde6192e2a68342","abstract_canon_sha256":"73c14dd7ebc745bed9c0d3b4affb6f766ae4371e045f03d02e0d86d39fdfdac1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:11:07.591336Z","signature_b64":"jFK3X1ZvJEhgTAEP7eC1XoAPHBmBpuGiUo1lQt5EwnNQGcsaCkzsK1Ydhzy2pg4O+gULMTCDOVcpUuGOhXtLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1df9aa641f363aa13addda8ed40ce6755c48ab8a5ae9794a9786090161d75919","last_reissued_at":"2026-07-05T00:11:07.590940Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:11:07.590940Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Asking Easy Questions: A User-Friendly Approach to Active Reward Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Dorsa Sadigh, Dylan P. Losey, Erdem B{\\i}y{\\i}k, Malayandi Palan, Nicholas C. Landolfi","submitted_at":"2019-10-10T04:52:46Z","abstract_excerpt":"Robots can learn the right reward function by querying a human expert. Existing approaches attempt to choose questions where the robot is most uncertain about the human's response; however, they do not consider how easy it will be for the human to answer! In this paper we explore an information gain formulation for optimally selecting questions that naturally account for the human's ability to answer. Our approach identifies questions that optimize the trade-off between robot and human uncertainty, and determines when these questions become redundant or costly. Simulations and a user study sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.04365","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.04365/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.04365","created_at":"2026-07-05T00:11:07.590998+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.04365v1","created_at":"2026-07-05T00:11:07.590998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.04365","created_at":"2026-07-05T00:11:07.590998+00:00"},{"alias_kind":"pith_short_12","alias_value":"DX42UZA7GY5K","created_at":"2026-07-05T00:11:07.590998+00:00"},{"alias_kind":"pith_short_16","alias_value":"DX42UZA7GY5KCOW5","created_at":"2026-07-05T00:11:07.590998+00:00"},{"alias_kind":"pith_short_8","alias_value":"DX42UZA7","created_at":"2026-07-05T00:11:07.590998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26072","citing_title":"Active Query Synthesis for Preference Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17855","citing_title":"QuickLAP: Quick Language-Action Preference Learning for Semi-Autonomous Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17855","citing_title":"QuickLAP: Quick Language-Action Preference Learning for Semi-Autonomous Agents","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV","json":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV.json","graph_json":"https://pith.science/api/pith-number/DX42UZA7GY5KCOW53KHNIDHGOV/graph.json","events_json":"https://pith.science/api/pith-number/DX42UZA7GY5KCOW53KHNIDHGOV/events.json","paper":"https://pith.science/paper/DX42UZA7"},"agent_actions":{"view_html":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV","download_json":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV.json","view_paper":"https://pith.science/paper/DX42UZA7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.04365&json=true","fetch_graph":"https://pith.science/api/pith-number/DX42UZA7GY5KCOW53KHNIDHGOV/graph.json","fetch_events":"https://pith.science/api/pith-number/DX42UZA7GY5KCOW53KHNIDHGOV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV/action/storage_attestation","attest_author":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV/action/author_attestation","sign_citation":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV/action/citation_signature","submit_replication":"https://pith.science/pith/DX42UZA7GY5KCOW53KHNIDHGOV/action/replication_record"}},"created_at":"2026-07-05T00:11:07.590998+00:00","updated_at":"2026-07-05T00:11:07.590998+00:00"}