{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:A4O6ZUW7QFDI7D3AMQMWM3GAUT","short_pith_number":"pith:A4O6ZUW7","schema_version":"1.0","canonical_sha256":"071decd2df81468f8f606419666cc0a4c0989b50bd8dbd20a60ca50d5b098794","source":{"kind":"arxiv","id":"2203.01855","version":3},"attestation_state":"computed","paper":{"title":"Reasoning about Counterfactuals to Improve Human Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.RO","authors_text":"Henny Admoni, Michael S. Lee, Reid Simmons","submitted_at":"2022-03-03T17:06:37Z","abstract_excerpt":"To collaborate well with robots, we must be able to understand their decision making. Humans naturally infer other agents' beliefs and desires by reasoning about their observable behavior in a way that resembles inverse reinforcement learning (IRL). Thus, robots can convey their beliefs and desires by providing demonstrations that are informative for a human learner's IRL. An informative demonstration is one that differs strongly from the learner's expectations of what the robot will do given their current understanding of the robot's decision making. However, standard IRL does not model the l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.01855","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-03-03T17:06:37Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"e36f5fe28554222da2bda89a7b07cd7b916b3aa395a6fb4131914d0fc36a54bd","abstract_canon_sha256":"28b59eaf884dff3d8963ca201afbbefd6e48e61a856a9b9ab8885685f36b0225"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:45:58.169157Z","signature_b64":"kp2vIyOS/MgMlzAxfNdf1OS8H5IL/119tcuCtMj5h4IT+yFI6gvZ6bcIgI33GLFvsWUnrpo3818ZQX/TjBlIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"071decd2df81468f8f606419666cc0a4c0989b50bd8dbd20a60ca50d5b098794","last_reissued_at":"2026-07-05T04:45:58.168655Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:45:58.168655Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning about Counterfactuals to Improve Human Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.RO","authors_text":"Henny Admoni, Michael S. Lee, Reid Simmons","submitted_at":"2022-03-03T17:06:37Z","abstract_excerpt":"To collaborate well with robots, we must be able to understand their decision making. Humans naturally infer other agents' beliefs and desires by reasoning about their observable behavior in a way that resembles inverse reinforcement learning (IRL). Thus, robots can convey their beliefs and desires by providing demonstrations that are informative for a human learner's IRL. An informative demonstration is one that differs strongly from the learner's expectations of what the robot will do given their current understanding of the robot's decision making. However, standard IRL does not model the l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.01855","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.01855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.01855","created_at":"2026-07-05T04:45:58.168713+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.01855v3","created_at":"2026-07-05T04:45:58.168713+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.01855","created_at":"2026-07-05T04:45:58.168713+00:00"},{"alias_kind":"pith_short_12","alias_value":"A4O6ZUW7QFDI","created_at":"2026-07-05T04:45:58.168713+00:00"},{"alias_kind":"pith_short_16","alias_value":"A4O6ZUW7QFDI7D3A","created_at":"2026-07-05T04:45:58.168713+00:00"},{"alias_kind":"pith_short_8","alias_value":"A4O6ZUW7","created_at":"2026-07-05T04:45:58.168713+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT","json":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT.json","graph_json":"https://pith.science/api/pith-number/A4O6ZUW7QFDI7D3AMQMWM3GAUT/graph.json","events_json":"https://pith.science/api/pith-number/A4O6ZUW7QFDI7D3AMQMWM3GAUT/events.json","paper":"https://pith.science/paper/A4O6ZUW7"},"agent_actions":{"view_html":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT","download_json":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT.json","view_paper":"https://pith.science/paper/A4O6ZUW7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.01855&json=true","fetch_graph":"https://pith.science/api/pith-number/A4O6ZUW7QFDI7D3AMQMWM3GAUT/graph.json","fetch_events":"https://pith.science/api/pith-number/A4O6ZUW7QFDI7D3AMQMWM3GAUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT/action/storage_attestation","attest_author":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT/action/author_attestation","sign_citation":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT/action/citation_signature","submit_replication":"https://pith.science/pith/A4O6ZUW7QFDI7D3AMQMWM3GAUT/action/replication_record"}},"created_at":"2026-07-05T04:45:58.168713+00:00","updated_at":"2026-07-05T04:45:58.168713+00:00"}