{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FYKH4W6TUOZDLBHHFU6TZKYUDT","short_pith_number":"pith:FYKH4W6T","schema_version":"1.0","canonical_sha256":"2e147e5bd3a3b23584e72d3d3cab141cdb3ed6a8174c8c4da1f572558a51d0ef","source":{"kind":"arxiv","id":"2412.12641","version":3},"attestation_state":"computed","paper":{"title":"Lagrangian Index Policy for Restless Bandits with Average Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC","math.PR"],"primary_cat":"cs.LG","authors_text":"Konstantin Avrachenkov, Pratik Shah, Vivek S. Borkar","submitted_at":"2024-12-17T08:03:53Z","abstract_excerpt":"We study the Lagrangian Index Policy (LIP) for restless multi-armed bandits with long-run average reward. In particular, we compare the performance of LIP with the performance of the Whittle Index Policy (WIP), both heuristic policies known to be asymptotically optimal under certain natural conditions. Even though in most cases their performances are very similar, in the cases when WIP shows bad performance, LIP continues to perform very well. We then propose reinforcement learning algorithms, both tabular and NN-based, to obtain online learning schemes for LIP in the model-free setting. The p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12641","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-17T08:03:53Z","cross_cats_sorted":["cs.AI","math.OC","math.PR"],"title_canon_sha256":"35ce611467458c9fe143fcd3a396a0a890a379fdf93cd009a2a99f5e1289e967","abstract_canon_sha256":"2b7936738988577cfa52fc9bd858de76270fd9aea012d3e7205cd05bf38823da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-05T01:33:16.168607Z","signature_b64":"15zC4fQKWeJ38JrhvLyk7kDKJUMdPbOfpJkXsx1avIf6VkRGDdAYBs6CM+vJ8BuRGgLkYsonHhbfKxs+LVrUDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e147e5bd3a3b23584e72d3d3cab141cdb3ed6a8174c8c4da1f572558a51d0ef","last_reissued_at":"2026-08-05T01:33:16.167044Z","signature_status":"signed_v1","first_computed_at":"2026-08-05T01:33:16.167044Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lagrangian Index Policy for Restless Bandits with Average Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC","math.PR"],"primary_cat":"cs.LG","authors_text":"Konstantin Avrachenkov, Pratik Shah, Vivek S. Borkar","submitted_at":"2024-12-17T08:03:53Z","abstract_excerpt":"We study the Lagrangian Index Policy (LIP) for restless multi-armed bandits with long-run average reward. In particular, we compare the performance of LIP with the performance of the Whittle Index Policy (WIP), both heuristic policies known to be asymptotically optimal under certain natural conditions. Even though in most cases their performances are very similar, in the cases when WIP shows bad performance, LIP continues to perform very well. We then propose reinforcement learning algorithms, both tabular and NN-based, to obtain online learning schemes for LIP in the model-free setting. The p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12641","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12641/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12641","created_at":"2026-08-05T01:33:16.167425+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12641v3","created_at":"2026-08-05T01:33:16.167425+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12641","created_at":"2026-08-05T01:33:16.167425+00:00"},{"alias_kind":"pith_short_12","alias_value":"FYKH4W6TUOZD","created_at":"2026-08-05T01:33:16.167425+00:00"},{"alias_kind":"pith_short_16","alias_value":"FYKH4W6TUOZDLBHH","created_at":"2026-08-05T01:33:16.167425+00:00"},{"alias_kind":"pith_short_8","alias_value":"FYKH4W6T","created_at":"2026-08-05T01:33:16.167425+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2606.24703","citing_title":"Scheduling jobs with unknown size distribution in a M/G/1 queue: the shifted empirical Gittins","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2604.04101","citing_title":"Restless Bandits with Individual Penalty Constraints: Near-Optimal Indices and Deep Reinforcement Learning","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18077","citing_title":"Lagrange Index based Scheduling for Minimizing Age of Updates from Heterogeneous Sources","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT","json":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT.json","graph_json":"https://pith.science/api/pith-number/FYKH4W6TUOZDLBHHFU6TZKYUDT/graph.json","events_json":"https://pith.science/api/pith-number/FYKH4W6TUOZDLBHHFU6TZKYUDT/events.json","paper":"https://pith.science/paper/FYKH4W6T"},"agent_actions":{"view_html":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT","download_json":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT.json","view_paper":"https://pith.science/paper/FYKH4W6T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12641&json=true","fetch_graph":"https://pith.science/api/pith-number/FYKH4W6TUOZDLBHHFU6TZKYUDT/graph.json","fetch_events":"https://pith.science/api/pith-number/FYKH4W6TUOZDLBHHFU6TZKYUDT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT/action/storage_attestation","attest_author":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT/action/author_attestation","sign_citation":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT/action/citation_signature","submit_replication":"https://pith.science/pith/FYKH4W6TUOZDLBHHFU6TZKYUDT/action/replication_record"}},"created_at":"2026-08-05T01:33:16.167425+00:00","updated_at":"2026-08-05T01:33:16.167425+00:00"}