{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MW5BIQQ5G3FCZBSEA4Q3RCN4QD","short_pith_number":"pith:MW5BIQQ5","schema_version":"1.0","canonical_sha256":"65ba14421d36ca2c86440721b889bc80c09f0851f467f8db8ce925085f14862c","source":{"kind":"arxiv","id":"2402.14807","version":4},"attestation_state":"computed","paper":{"title":"A Decision-Language Model (DLM) for Dynamic Restless Multi-Armed Bandit Tasks in Public Health","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Aparna Taneja, Dheeraj Nagaraj, Edwin Zhang, Milind Tambe, Nikhil Behari, Yunfan Zhao","submitted_at":"2024-02-22T18:58:27Z","abstract_excerpt":"Restless multi-armed bandits (RMAB) have demonstrated success in optimizing resource allocation for large beneficiary populations in public health settings. Unfortunately, RMAB models lack flexibility to adapt to evolving public health policy priorities. Concurrently, Large Language Models (LLMs) have emerged as adept automated planners across domains of robotic control and navigation. In this paper, we propose a Decision Language Model (DLM) for RMABs, enabling dynamic fine-tuning of RMAB policies in public health settings using human-language commands. We propose using LLMs as automated plan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14807","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.MA","submitted_at":"2024-02-22T18:58:27Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e5ccab046b8b6a8e0946f872a7d4d87032d3d4c7445e3e02ea7025e2055882b0","abstract_canon_sha256":"eb39c14c4904531040446cf0a49f5fce4c402d95b5af3a6e87a6aba1417b5eb9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:45.232544Z","signature_b64":"rexvbc0lEy9QObbLNdZkw5qieZYcQWt8eWMbHWk9txmNPpBb8qmAnC5qLdGqaHqrYeXk46okrCDdUcAigDw5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65ba14421d36ca2c86440721b889bc80c09f0851f467f8db8ce925085f14862c","last_reissued_at":"2026-07-05T11:10:45.232046Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:45.232046Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Decision-Language Model (DLM) for Dynamic Restless Multi-Armed Bandit Tasks in Public Health","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Aparna Taneja, Dheeraj Nagaraj, Edwin Zhang, Milind Tambe, Nikhil Behari, Yunfan Zhao","submitted_at":"2024-02-22T18:58:27Z","abstract_excerpt":"Restless multi-armed bandits (RMAB) have demonstrated success in optimizing resource allocation for large beneficiary populations in public health settings. Unfortunately, RMAB models lack flexibility to adapt to evolving public health policy priorities. Concurrently, Large Language Models (LLMs) have emerged as adept automated planners across domains of robotic control and navigation. In this paper, we propose a Decision Language Model (DLM) for RMABs, enabling dynamic fine-tuning of RMAB policies in public health settings using human-language commands. We propose using LLMs as automated plan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14807","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14807","created_at":"2026-07-05T11:10:45.232101+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14807v4","created_at":"2026-07-05T11:10:45.232101+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14807","created_at":"2026-07-05T11:10:45.232101+00:00"},{"alias_kind":"pith_short_12","alias_value":"MW5BIQQ5G3FC","created_at":"2026-07-05T11:10:45.232101+00:00"},{"alias_kind":"pith_short_16","alias_value":"MW5BIQQ5G3FCZBSE","created_at":"2026-07-05T11:10:45.232101+00:00"},{"alias_kind":"pith_short_8","alias_value":"MW5BIQQ5","created_at":"2026-07-05T11:10:45.232101+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.18186","citing_title":"Online Learning of Whittle Indices for Restless Bandits with Non-Stationary Transition Kernels","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD","json":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD.json","graph_json":"https://pith.science/api/pith-number/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/graph.json","events_json":"https://pith.science/api/pith-number/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/events.json","paper":"https://pith.science/paper/MW5BIQQ5"},"agent_actions":{"view_html":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD","download_json":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD.json","view_paper":"https://pith.science/paper/MW5BIQQ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14807&json=true","fetch_graph":"https://pith.science/api/pith-number/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/graph.json","fetch_events":"https://pith.science/api/pith-number/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/action/storage_attestation","attest_author":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/action/author_attestation","sign_citation":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/action/citation_signature","submit_replication":"https://pith.science/pith/MW5BIQQ5G3FCZBSEA4Q3RCN4QD/action/replication_record"}},"created_at":"2026-07-05T11:10:45.232101+00:00","updated_at":"2026-07-05T11:10:45.232101+00:00"}