{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:6Y26TW6SUCBC3S5L2YEILVUWOV","short_pith_number":"pith:6Y26TW6S","canonical_record":{"source":{"id":"2007.01612","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-03T11:06:38Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"04d1abd8c39196fbe87ebac1f0e1b8016e8d8f4b891827e50d7ef0a91c6679a3","abstract_canon_sha256":"b7b42a80c23f9000e66347be3cd9e42924d6e70f9cd7e2106f7dd521267f33db"},"schema_version":"1.0"},"canonical_sha256":"f635e9dbd2a0822dcbabd60885d6967568bfe400effbfc2d0b3e2f308ccdb3fe","source":{"kind":"arxiv","id":"2007.01612","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2007.01612","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"arxiv_version","alias_value":"2007.01612v2","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.01612","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_12","alias_value":"6Y26TW6SUCBC","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_16","alias_value":"6Y26TW6SUCBC3S5L","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_8","alias_value":"6Y26TW6S","created_at":"2026-07-05T02:48:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:6Y26TW6SUCBC3S5L2YEILVUWOV","target":"record","payload":{"canonical_record":{"source":{"id":"2007.01612","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-03T11:06:38Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"04d1abd8c39196fbe87ebac1f0e1b8016e8d8f4b891827e50d7ef0a91c6679a3","abstract_canon_sha256":"b7b42a80c23f9000e66347be3cd9e42924d6e70f9cd7e2106f7dd521267f33db"},"schema_version":"1.0"},"canonical_sha256":"f635e9dbd2a0822dcbabd60885d6967568bfe400effbfc2d0b3e2f308ccdb3fe","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:33.519946Z","signature_b64":"DT0CWt+MzuYGF6MqlguWZJHKt26gIxn/SjTU1IJmbr/0zZVNLBvr7zBQ9bYPn6T+4D5r6Sq7qlF4gTfdA8kzBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f635e9dbd2a0822dcbabd60885d6967568bfe400effbfc2d0b3e2f308ccdb3fe","last_reissued_at":"2026-07-05T02:48:33.519499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:33.519499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2007.01612","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:48:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NtxK7NydiLXJFDuHlobnGleaxPpZ7qOdEMPQJvntKxayGbx1/xcs/zX6Ip8rZzMzVqSCwpAcEfmXPdo81/FUAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T22:54:14.968712Z"},"content_sha256":"3e752f1d642885146c8fa07ab9be4bbd4ca1c2ee3df7731654c7efc12f4cae07","schema_version":"1.0","event_id":"sha256:3e752f1d642885146c8fa07ab9be4bbd4ca1c2ee3df7731654c7efc12f4cae07"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:6Y26TW6SUCBC3S5L2YEILVUWOV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Online learning in MDPs with linear function approximation and bandit feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Gergely Neu, Julia Olkhovskaya","submitted_at":"2020-07-03T11:06:38Z","abstract_excerpt":"We consider an online learning problem where the learner interacts with a Markov decision process in a sequence of episodes, where the reward function is allowed to change between episodes in an adversarial manner and the learner only gets to observe the rewards associated with its actions. We allow the state space to be arbitrarily large, but we assume that all action-value functions can be represented as linear functions in terms of a known low-dimensional feature map, and that the learner has access to a simulator of the environment that allows generating trajectories from the true MDP dyna"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.01612","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.01612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:48:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"irM1+fL0VkZ4fC0LysWTzhhP1iGpPa9t+jj2YaQ7oCPDeBU9hOIUiwZGBaISh+swGQoj7VaQY2PXFD/HrRtlAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T22:54:14.969223Z"},"content_sha256":"5af9dfc920e80b105d8cccb1c2760adfc343158ff42f73e8e772ffc07dba0e7a","schema_version":"1.0","event_id":"sha256:5af9dfc920e80b105d8cccb1c2760adfc343158ff42f73e8e772ffc07dba0e7a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/bundle.json","state_url":"https://pith.science/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T22:54:14Z","links":{"resolver":"https://pith.science/pith/6Y26TW6SUCBC3S5L2YEILVUWOV","bundle":"https://pith.science/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/bundle.json","state":"https://pith.science/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6Y26TW6SUCBC3S5L2YEILVUWOV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:6Y26TW6SUCBC3S5L2YEILVUWOV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b7b42a80c23f9000e66347be3cd9e42924d6e70f9cd7e2106f7dd521267f33db","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-03T11:06:38Z","title_canon_sha256":"04d1abd8c39196fbe87ebac1f0e1b8016e8d8f4b891827e50d7ef0a91c6679a3"},"schema_version":"1.0","source":{"id":"2007.01612","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2007.01612","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"arxiv_version","alias_value":"2007.01612v2","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.01612","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_12","alias_value":"6Y26TW6SUCBC","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_16","alias_value":"6Y26TW6SUCBC3S5L","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_8","alias_value":"6Y26TW6S","created_at":"2026-07-05T02:48:33Z"}],"graph_snapshots":[{"event_id":"sha256:5af9dfc920e80b105d8cccb1c2760adfc343158ff42f73e8e772ffc07dba0e7a","target":"graph","created_at":"2026-07-05T02:48:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2007.01612/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We consider an online learning problem where the learner interacts with a Markov decision process in a sequence of episodes, where the reward function is allowed to change between episodes in an adversarial manner and the learner only gets to observe the rewards associated with its actions. We allow the state space to be arbitrarily large, but we assume that all action-value functions can be represented as linear functions in terms of a known low-dimensional feature map, and that the learner has access to a simulator of the environment that allows generating trajectories from the true MDP dyna","authors_text":"Gergely Neu, Julia Olkhovskaya","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-03T11:06:38Z","title":"Online learning in MDPs with linear function approximation and bandit feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.01612","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3e752f1d642885146c8fa07ab9be4bbd4ca1c2ee3df7731654c7efc12f4cae07","target":"record","created_at":"2026-07-05T02:48:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b7b42a80c23f9000e66347be3cd9e42924d6e70f9cd7e2106f7dd521267f33db","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-03T11:06:38Z","title_canon_sha256":"04d1abd8c39196fbe87ebac1f0e1b8016e8d8f4b891827e50d7ef0a91c6679a3"},"schema_version":"1.0","source":{"id":"2007.01612","kind":"arxiv","version":2}},"canonical_sha256":"f635e9dbd2a0822dcbabd60885d6967568bfe400effbfc2d0b3e2f308ccdb3fe","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f635e9dbd2a0822dcbabd60885d6967568bfe400effbfc2d0b3e2f308ccdb3fe","first_computed_at":"2026-07-05T02:48:33.519499Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:48:33.519499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"DT0CWt+MzuYGF6MqlguWZJHKt26gIxn/SjTU1IJmbr/0zZVNLBvr7zBQ9bYPn6T+4D5r6Sq7qlF4gTfdA8kzBw==","signature_status":"signed_v1","signed_at":"2026-07-05T02:48:33.519946Z","signed_message":"canonical_sha256_bytes"},"source_id":"2007.01612","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3e752f1d642885146c8fa07ab9be4bbd4ca1c2ee3df7731654c7efc12f4cae07","sha256:5af9dfc920e80b105d8cccb1c2760adfc343158ff42f73e8e772ffc07dba0e7a"],"state_sha256":"d1b6b2a824c12b3823014aef485cdcf7424981816c32f308f3a7081321cef6ee"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9dDTK82BJbF1qmhKVwBbX21AJuejOpSF3c2NMYGttBH7BjHImf5c+w6iyXY4wp6HmB86rKqwDYI9WAju4YuZAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T22:54:14.973159Z","bundle_sha256":"4a36e8151d3a0740a2ef77fb3ee93cb8223c85027ad7dd976c9cb6fe7d9e4543"}}