{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:6UIMEVYZ76RXIBQ3DRIVN6WOLH","short_pith_number":"pith:6UIMEVYZ","schema_version":"1.0","canonical_sha256":"f510c25719ffa374061b1c5156face59dd59ad5ab027b495b8552205a5eef885","source":{"kind":"arxiv","id":"2011.09004","version":1},"attestation_state":"computed","paper":{"title":"Explaining Conditions for Reinforcement Learning Behaviors from Real and Imagined Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.NE","cs.RO"],"primary_cat":"cs.LG","authors_text":"Aastha Acharya, Nisar R. Ahmed, Rebecca Russell","submitted_at":"2020-11-17T23:40:47Z","abstract_excerpt":"The deployment of reinforcement learning (RL) in the real world comes with challenges in calibrating user trust and expectations. As a step toward developing RL systems that are able to communicate their competencies, we present a method of generating human-interpretable abstract behavior models that identify the experiential conditions leading to different task execution strategies and outcomes. Our approach consists of extracting experiential features from state representations, abstracting strategy descriptors from trajectories, and training an interpretable decision tree that identifies th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.09004","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-11-17T23:40:47Z","cross_cats_sorted":["cs.AI","cs.HC","cs.NE","cs.RO"],"title_canon_sha256":"498a0e7a77de87c56b26b55bd9e04f55f877a6b674ee2bcddcf50be1f7e98691","abstract_canon_sha256":"190afd296a76ae24f2ce188abd9891a79e7a1f304f368362066cf061005ba481"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:52:39.640791Z","signature_b64":"hO3j/GOx4X0XXNK5YeW+q9us7bpwglnWv2zRCSx7BKVj3pYjXy06WlvszTIKqVgWxueSzWluyPnGEOQxK7TzAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f510c25719ffa374061b1c5156face59dd59ad5ab027b495b8552205a5eef885","last_reissued_at":"2026-07-05T01:52:39.640410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:52:39.640410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explaining Conditions for Reinforcement Learning Behaviors from Real and Imagined Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.NE","cs.RO"],"primary_cat":"cs.LG","authors_text":"Aastha Acharya, Nisar R. Ahmed, Rebecca Russell","submitted_at":"2020-11-17T23:40:47Z","abstract_excerpt":"The deployment of reinforcement learning (RL) in the real world comes with challenges in calibrating user trust and expectations. As a step toward developing RL systems that are able to communicate their competencies, we present a method of generating human-interpretable abstract behavior models that identify the experiential conditions leading to different task execution strategies and outcomes. Our approach consists of extracting experiential features from state representations, abstracting strategy descriptors from trajectories, and training an interpretable decision tree that identifies th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.09004","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.09004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.09004","created_at":"2026-07-05T01:52:39.640466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.09004v1","created_at":"2026-07-05T01:52:39.640466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.09004","created_at":"2026-07-05T01:52:39.640466+00:00"},{"alias_kind":"pith_short_12","alias_value":"6UIMEVYZ76RX","created_at":"2026-07-05T01:52:39.640466+00:00"},{"alias_kind":"pith_short_16","alias_value":"6UIMEVYZ76RXIBQ3","created_at":"2026-07-05T01:52:39.640466+00:00"},{"alias_kind":"pith_short_8","alias_value":"6UIMEVYZ","created_at":"2026-07-05T01:52:39.640466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12599","citing_title":"A Survey of Explainable Reinforcement Learning: Targets, Methods and Needs","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH","json":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH.json","graph_json":"https://pith.science/api/pith-number/6UIMEVYZ76RXIBQ3DRIVN6WOLH/graph.json","events_json":"https://pith.science/api/pith-number/6UIMEVYZ76RXIBQ3DRIVN6WOLH/events.json","paper":"https://pith.science/paper/6UIMEVYZ"},"agent_actions":{"view_html":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH","download_json":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH.json","view_paper":"https://pith.science/paper/6UIMEVYZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.09004&json=true","fetch_graph":"https://pith.science/api/pith-number/6UIMEVYZ76RXIBQ3DRIVN6WOLH/graph.json","fetch_events":"https://pith.science/api/pith-number/6UIMEVYZ76RXIBQ3DRIVN6WOLH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH/action/storage_attestation","attest_author":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH/action/author_attestation","sign_citation":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH/action/citation_signature","submit_replication":"https://pith.science/pith/6UIMEVYZ76RXIBQ3DRIVN6WOLH/action/replication_record"}},"created_at":"2026-07-05T01:52:39.640466+00:00","updated_at":"2026-07-05T01:52:39.640466+00:00"}