{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2012:GTVDPUULCY3FV5QHQRPUPZIGVI","short_pith_number":"pith:GTVDPUUL","schema_version":"1.0","canonical_sha256":"34ea37d28b16365af607845f47e506aa227c39427b942ae318b1da5d3bfdd872","source":{"kind":"arxiv","id":"1206.6484","version":1},"attestation_state":"computed","paper":{"title":"Apprenticeship Learning for Model Parameters of Partially Observable Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Johane Takeuchi (Honda Research Institute Japan), Takaki Makino (University of Tokyo)","submitted_at":"2012-06-27T19:59:59Z","abstract_excerpt":"We consider apprenticeship learning, i.e., having an agent learn a task by observing an expert demonstrating the task in a partially observable environment when the model of the environment is uncertain. This setting is useful in applications where the explicit modeling of the environment is difficult, such as a dialogue system. We show that we can extract information about the environment model by inferring action selection process behind the demonstration, under the assumption that the expert is choosing optimal actions based on knowledge of the true model of the target environment. Proposed"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1206.6484","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2012-06-27T19:59:59Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"8cf7e2beb8562ab3e6e356b0aba290442d7efb0ac3350af8eec8855f5157e81d","abstract_canon_sha256":"c5887401ad07b2936b78387837d473f2d3ab4903a1d2859ec013f59e89105868"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T03:52:13.694263Z","signature_b64":"rNwgsX/heON4MBAMlIe4LrkMCsHBoDapXfcWRQRmurD7Q/2r6TM/+9TWMvAqakB2lkoEFlv9BPmlg27ICnoBBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34ea37d28b16365af607845f47e506aa227c39427b942ae318b1da5d3bfdd872","last_reissued_at":"2026-05-18T03:52:13.693529Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T03:52:13.693529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Apprenticeship Learning for Model Parameters of Partially Observable Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Johane Takeuchi (Honda Research Institute Japan), Takaki Makino (University of Tokyo)","submitted_at":"2012-06-27T19:59:59Z","abstract_excerpt":"We consider apprenticeship learning, i.e., having an agent learn a task by observing an expert demonstrating the task in a partially observable environment when the model of the environment is uncertain. This setting is useful in applications where the explicit modeling of the environment is difficult, such as a dialogue system. We show that we can extract information about the environment model by inferring action selection process behind the demonstration, under the assumption that the expert is choosing optimal actions based on knowledge of the true model of the target environment. Proposed"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1206.6484","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1206.6484","created_at":"2026-05-18T03:52:13.693657+00:00"},{"alias_kind":"arxiv_version","alias_value":"1206.6484v1","created_at":"2026-05-18T03:52:13.693657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1206.6484","created_at":"2026-05-18T03:52:13.693657+00:00"},{"alias_kind":"pith_short_12","alias_value":"GTVDPUULCY3F","created_at":"2026-05-18T12:27:06.952714+00:00"},{"alias_kind":"pith_short_16","alias_value":"GTVDPUULCY3FV5QH","created_at":"2026-05-18T12:27:06.952714+00:00"},{"alias_kind":"pith_short_8","alias_value":"GTVDPUUL","created_at":"2026-05-18T12:27:06.952714+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04396","citing_title":"Inverse Reinforcement Learning using Revealed Preferences and Passive Stochastic Optimization","ref_index":2016,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI","json":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI.json","graph_json":"https://pith.science/api/pith-number/GTVDPUULCY3FV5QHQRPUPZIGVI/graph.json","events_json":"https://pith.science/api/pith-number/GTVDPUULCY3FV5QHQRPUPZIGVI/events.json","paper":"https://pith.science/paper/GTVDPUUL"},"agent_actions":{"view_html":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI","download_json":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI.json","view_paper":"https://pith.science/paper/GTVDPUUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1206.6484&json=true","fetch_graph":"https://pith.science/api/pith-number/GTVDPUULCY3FV5QHQRPUPZIGVI/graph.json","fetch_events":"https://pith.science/api/pith-number/GTVDPUULCY3FV5QHQRPUPZIGVI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI/action/storage_attestation","attest_author":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI/action/author_attestation","sign_citation":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI/action/citation_signature","submit_replication":"https://pith.science/pith/GTVDPUULCY3FV5QHQRPUPZIGVI/action/replication_record"}},"created_at":"2026-05-18T03:52:13.693657+00:00","updated_at":"2026-05-18T03:52:13.693657+00:00"}