{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AL3YEW2EDMJRBPJ6U4SEQUEVQN","short_pith_number":"pith:AL3YEW2E","schema_version":"1.0","canonical_sha256":"02f7825b441b1310bd3ea72448509583519428025ff0eb79877ce13dddd3202b","source":{"kind":"arxiv","id":"2412.14355","version":1},"attestation_state":"computed","paper":{"title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Glen Berseth, Gopeshh Subbaraj, Irina Rish, Matthew Riemer","submitted_at":"2024-12-18T21:43:40Z","abstract_excerpt":"Realtime environments change even as agents perform action inference and learning, thus requiring high interaction frequencies to effectively minimize regret. However, recent advances in machine learning involve larger neural networks with longer inference times, raising questions about their applicability in realtime systems where reaction time is crucial. We present an analysis of lower bounds on regret in realtime reinforcement learning (RL) environments to show that minimizing long-term regret is generally impossible within the typical sequential interaction and learning paradigm, but ofte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14355","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-18T21:43:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1489c69e49492beb43d22a5a3d971af8fe56407839d258973a7524f981a7c97f","abstract_canon_sha256":"46d025496ffcfce8f125ed889231573cc9a2679ee442a038c08a3e8b7fe7f2a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:46.083475Z","signature_b64":"/jZK6H7RIwqT015FY6/FVlRDCMpOro0DVP2gm09gOqyHMx+2kUBqw8z6oLpvA//k0dt0KP277Ro8E2YLWaNAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02f7825b441b1310bd3ea72448509583519428025ff0eb79877ce13dddd3202b","last_reissued_at":"2026-07-05T09:51:46.083000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:46.083000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Glen Berseth, Gopeshh Subbaraj, Irina Rish, Matthew Riemer","submitted_at":"2024-12-18T21:43:40Z","abstract_excerpt":"Realtime environments change even as agents perform action inference and learning, thus requiring high interaction frequencies to effectively minimize regret. However, recent advances in machine learning involve larger neural networks with longer inference times, raising questions about their applicability in realtime systems where reaction time is crucial. We present an analysis of lower bounds on regret in realtime reinforcement learning (RL) environments to show that minimizing long-term regret is generally impossible within the typical sequential interaction and learning paradigm, but ofte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14355","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14355/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14355","created_at":"2026-07-05T09:51:46.083058+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14355v1","created_at":"2026-07-05T09:51:46.083058+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14355","created_at":"2026-07-05T09:51:46.083058+00:00"},{"alias_kind":"pith_short_12","alias_value":"AL3YEW2EDMJR","created_at":"2026-07-05T09:51:46.083058+00:00"},{"alias_kind":"pith_short_16","alias_value":"AL3YEW2EDMJRBPJ6","created_at":"2026-07-05T09:51:46.083058+00:00"},{"alias_kind":"pith_short_8","alias_value":"AL3YEW2E","created_at":"2026-07-05T09:51:46.083058+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN","json":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN.json","graph_json":"https://pith.science/api/pith-number/AL3YEW2EDMJRBPJ6U4SEQUEVQN/graph.json","events_json":"https://pith.science/api/pith-number/AL3YEW2EDMJRBPJ6U4SEQUEVQN/events.json","paper":"https://pith.science/paper/AL3YEW2E"},"agent_actions":{"view_html":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN","download_json":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN.json","view_paper":"https://pith.science/paper/AL3YEW2E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14355&json=true","fetch_graph":"https://pith.science/api/pith-number/AL3YEW2EDMJRBPJ6U4SEQUEVQN/graph.json","fetch_events":"https://pith.science/api/pith-number/AL3YEW2EDMJRBPJ6U4SEQUEVQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN/action/storage_attestation","attest_author":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN/action/author_attestation","sign_citation":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN/action/citation_signature","submit_replication":"https://pith.science/pith/AL3YEW2EDMJRBPJ6U4SEQUEVQN/action/replication_record"}},"created_at":"2026-07-05T09:51:46.083058+00:00","updated_at":"2026-07-05T09:51:46.083058+00:00"}