{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:PBDEFMWDCXDS5BQA4H3BRR2FVS","short_pith_number":"pith:PBDEFMWD","canonical_record":{"source":{"id":"2405.07637","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-13T10:51:01Z","cross_cats_sorted":[],"title_canon_sha256":"d2522caf3e82f3b524a53934830c620681132900dc98472398a4ddf63ca26872","abstract_canon_sha256":"6661d8fa1d74a3c3be0092d83271141f0bb1c23620bb3b34059f002edbfc45dc"},"schema_version":"1.0"},"canonical_sha256":"784642b2c315c72e8600e1f618c745ac8267f17be50fcdc12dcffbe71102de0e","source":{"kind":"arxiv","id":"2405.07637","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.07637","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"arxiv_version","alias_value":"2405.07637v2","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07637","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_12","alias_value":"PBDEFMWDCXDS","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_16","alias_value":"PBDEFMWDCXDS5BQA","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_8","alias_value":"PBDEFMWD","created_at":"2026-07-05T08:18:53Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:PBDEFMWDCXDS5BQA4H3BRR2FVS","target":"record","payload":{"canonical_record":{"source":{"id":"2405.07637","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-13T10:51:01Z","cross_cats_sorted":[],"title_canon_sha256":"d2522caf3e82f3b524a53934830c620681132900dc98472398a4ddf63ca26872","abstract_canon_sha256":"6661d8fa1d74a3c3be0092d83271141f0bb1c23620bb3b34059f002edbfc45dc"},"schema_version":"1.0"},"canonical_sha256":"784642b2c315c72e8600e1f618c745ac8267f17be50fcdc12dcffbe71102de0e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:18:53.585180Z","signature_b64":"UDU6K2m7vjj1nqKKsueI9njl9XBcfQGTSQ0LbD9lXJPKLN6oaUqAxedj10ZGDrP77PZPhhPivxPq8SIh9b5hDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"784642b2c315c72e8600e1f618c745ac8267f17be50fcdc12dcffbe71102de0e","last_reissued_at":"2026-07-05T08:18:53.584676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:18:53.584676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.07637","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:18:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZRWM00jowlwnWO9isa4//Gi2SS7zH7REN6JMvHvMcgYRp8NP44rQxDQe6grxKK2Bx0BSyIAP78cWtMTNEV//Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:40:19.009009Z"},"content_sha256":"56e769e92edeccf3a103853ec50158328256cd93408475c125a5600a6e345b6a","schema_version":"1.0","event_id":"sha256:56e769e92edeccf3a103853ec50158328256cd93408475c125a5600a6e345b6a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:PBDEFMWDCXDS5BQA4H3BRR2FVS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Near-Optimal Regret in Linear MDPs with Aggregate Bandit Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Asaf Cassel, Aviv Rosenberg, Dmitry Sotnikov, Haipeng Luo","submitted_at":"2024-05-13T10:51:01Z","abstract_excerpt":"In many real-world applications, it is hard to provide a reward signal in each step of a Reinforcement Learning (RL) process and more natural to give feedback when an episode ends. To this end, we study the recently proposed model of RL with Aggregate Bandit Feedback (RL-ABF), where the agent only observes the sum of rewards at the end of an episode instead of each reward individually. Prior work studied RL-ABF only in tabular settings, where the number of states is assumed to be small. In this paper, we extend ABF to linear function approximation and develop two efficient algorithms with near"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07637","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.07637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:18:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9s25st0eHa72N6pI+RTmLWHkw7DWiMgYBlA10Y+4j50sMwB3FX8vvcdACAkmp4R6c/TEJGl00482ZXdLni/ZCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:40:19.009504Z"},"content_sha256":"d1e61ea1601ed7854d5b8d3b9b99dc16bb13ea53873ee921c564b7c04c627fe6","schema_version":"1.0","event_id":"sha256:d1e61ea1601ed7854d5b8d3b9b99dc16bb13ea53873ee921c564b7c04c627fe6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/bundle.json","state_url":"https://pith.science/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T09:40:19Z","links":{"resolver":"https://pith.science/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS","bundle":"https://pith.science/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/bundle.json","state":"https://pith.science/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PBDEFMWDCXDS5BQA4H3BRR2FVS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:PBDEFMWDCXDS5BQA4H3BRR2FVS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6661d8fa1d74a3c3be0092d83271141f0bb1c23620bb3b34059f002edbfc45dc","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-13T10:51:01Z","title_canon_sha256":"d2522caf3e82f3b524a53934830c620681132900dc98472398a4ddf63ca26872"},"schema_version":"1.0","source":{"id":"2405.07637","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.07637","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"arxiv_version","alias_value":"2405.07637v2","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07637","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_12","alias_value":"PBDEFMWDCXDS","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_16","alias_value":"PBDEFMWDCXDS5BQA","created_at":"2026-07-05T08:18:53Z"},{"alias_kind":"pith_short_8","alias_value":"PBDEFMWD","created_at":"2026-07-05T08:18:53Z"}],"graph_snapshots":[{"event_id":"sha256:d1e61ea1601ed7854d5b8d3b9b99dc16bb13ea53873ee921c564b7c04c627fe6","target":"graph","created_at":"2026-07-05T08:18:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.07637/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In many real-world applications, it is hard to provide a reward signal in each step of a Reinforcement Learning (RL) process and more natural to give feedback when an episode ends. To this end, we study the recently proposed model of RL with Aggregate Bandit Feedback (RL-ABF), where the agent only observes the sum of rewards at the end of an episode instead of each reward individually. Prior work studied RL-ABF only in tabular settings, where the number of states is assumed to be small. In this paper, we extend ABF to linear function approximation and develop two efficient algorithms with near","authors_text":"Asaf Cassel, Aviv Rosenberg, Dmitry Sotnikov, Haipeng Luo","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-13T10:51:01Z","title":"Near-Optimal Regret in Linear MDPs with Aggregate Bandit Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07637","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:56e769e92edeccf3a103853ec50158328256cd93408475c125a5600a6e345b6a","target":"record","created_at":"2026-07-05T08:18:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6661d8fa1d74a3c3be0092d83271141f0bb1c23620bb3b34059f002edbfc45dc","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-13T10:51:01Z","title_canon_sha256":"d2522caf3e82f3b524a53934830c620681132900dc98472398a4ddf63ca26872"},"schema_version":"1.0","source":{"id":"2405.07637","kind":"arxiv","version":2}},"canonical_sha256":"784642b2c315c72e8600e1f618c745ac8267f17be50fcdc12dcffbe71102de0e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"784642b2c315c72e8600e1f618c745ac8267f17be50fcdc12dcffbe71102de0e","first_computed_at":"2026-07-05T08:18:53.584676Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:18:53.584676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UDU6K2m7vjj1nqKKsueI9njl9XBcfQGTSQ0LbD9lXJPKLN6oaUqAxedj10ZGDrP77PZPhhPivxPq8SIh9b5hDw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:18:53.585180Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.07637","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:56e769e92edeccf3a103853ec50158328256cd93408475c125a5600a6e345b6a","sha256:d1e61ea1601ed7854d5b8d3b9b99dc16bb13ea53873ee921c564b7c04c627fe6"],"state_sha256":"7543197113f5d02e6142445021989b3662ebc0901a953f73d2678827dd4ad4db"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TfbOmBBtcPiy1sI3KIleuJpPvIEHx0mX2VKg6yJjHM327emAWPG8LC5A08DOwwjP1WbKS3UNhmqxBOI8ex2mBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T09:40:19.013095Z","bundle_sha256":"840921daadee08dc6da107ba4d809e273857c0b90fb4aa09df39a305d88fd286"}}