{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:3IIWRSHHLECPUP2XM7GK4J4AXU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"96e4c99c4f091d9363d16f6022b637b384daaeae412a4b85d1ef67ff0651b8af","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-01T22:26:27Z","title_canon_sha256":"911a9fa714ea1d56fa5f51ff8eb1bcfb304ad78852930095137e883e22530ad1"},"schema_version":"1.0","source":{"id":"2412.00985","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2412.00985","created_at":"2026-07-05T10:17:51Z"},{"alias_kind":"arxiv_version","alias_value":"2412.00985v3","created_at":"2026-07-05T10:17:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00985","created_at":"2026-07-05T10:17:51Z"},{"alias_kind":"pith_short_12","alias_value":"3IIWRSHHLECP","created_at":"2026-07-05T10:17:51Z"},{"alias_kind":"pith_short_16","alias_value":"3IIWRSHHLECPUP2X","created_at":"2026-07-05T10:17:51Z"},{"alias_kind":"pith_short_8","alias_value":"3IIWRSHH","created_at":"2026-07-05T10:17:51Z"}],"graph_snapshots":[{"event_id":"sha256:828cf37c268fd84e41a37473f2c0d1c553e19e64db890401a9fe6307c5e59ddb","target":"graph","created_at":"2026-07-05T10:17:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2412.00985/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Partial observability of the underlying states generally presents significant challenges for reinforcement learning (RL). In practice, certain \\emph{privileged information}, e.g., the access to states from simulators, has been exploited in training and has achieved prominent empirical successes. To better understand the benefits of privileged information, we revisit and examine several simple and practically used paradigms in this setting. Specifically, we first formalize the empirical paradigm of \\emph{expert distillation} (also known as \\emph{teacher-student} learning), demonstrating its pit","authors_text":"Argyris Oikonomou, Kaiqing Zhang, Xiangyu Liu, Yang Cai","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-01T22:26:27Z","title":"Provable Partially Observable Reinforcement Learning with Privileged Information"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00985","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ea38d8ae8a1d976e23c4fe285b5599a4a5d842520b67ed0b83ef176f48351a76","target":"record","created_at":"2026-07-05T10:17:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"96e4c99c4f091d9363d16f6022b637b384daaeae412a4b85d1ef67ff0651b8af","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-01T22:26:27Z","title_canon_sha256":"911a9fa714ea1d56fa5f51ff8eb1bcfb304ad78852930095137e883e22530ad1"},"schema_version":"1.0","source":{"id":"2412.00985","kind":"arxiv","version":3}},"canonical_sha256":"da1168c8e75904fa3f5767ccae2780bd3ef93c009d844a295928a693862e9cec","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"da1168c8e75904fa3f5767ccae2780bd3ef93c009d844a295928a693862e9cec","first_computed_at":"2026-07-05T10:17:51.870170Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:17:51.870170Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"2j2qvRxxYUfHJLETKuMdJyaDJLBdzMhQ4At1AyycM/ES+BdeMVE+dVQUiBklN10npqe6NQPtPMUgtLP8NG3ABQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:17:51.870574Z","signed_message":"canonical_sha256_bytes"},"source_id":"2412.00985","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ea38d8ae8a1d976e23c4fe285b5599a4a5d842520b67ed0b83ef176f48351a76","sha256:828cf37c268fd84e41a37473f2c0d1c553e19e64db890401a9fe6307c5e59ddb"],"state_sha256":"1d33dd26fb04491a026049e2f4c6b624ce1069fbafe429d8d5c2b5e375837fef"}