{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6HUZ5I42UFI5JWRWMABVRYDY7R","short_pith_number":"pith:6HUZ5I42","schema_version":"1.0","canonical_sha256":"f1e99ea39aa151d4da36600358e078fc5b1b25ef2f5e57c1b0a660bd54fcd21b","source":{"kind":"arxiv","id":"2310.06793","version":2},"attestation_state":"computed","paper":{"title":"Spectral Entry-wise Matrix Estimation for Low-Rank Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexandre Proutiere, Stefan Stojanovic, Yassir Jedra","submitted_at":"2023-10-10T17:06:41Z","abstract_excerpt":"We study matrix estimation problems arising in reinforcement learning (RL) with low-rank structure. In low-rank bandits, the matrix to be recovered specifies the expected arm rewards, and for low-rank Markov Decision Processes (MDPs), it may for example characterize the transition kernel of the MDP. In both cases, each entry of the matrix carries important information, and we seek estimation methods with low entry-wise error. Importantly, these methods further need to accommodate for inherent correlations in the available data (e.g. for MDPs, the data consists of system trajectories). We inves"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06793","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T17:06:41Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"42fe4028dce6d87aa901b8399a1ca0a2b70bfb464807ea26d2fe2b2dca177711","abstract_canon_sha256":"2b2e96c3c6b7e496c2929d540c2b366af7d42e15e1615debf4fbeb53b126a526"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:05.864923Z","signature_b64":"2SH+e/ocf4nntg69/FfF1jnQf1laONzwsrsx3piD6wt6a6ZZnykSX9k+s6xmm7kD4vvOToRIwkZjm75GIuozCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1e99ea39aa151d4da36600358e078fc5b1b25ef2f5e57c1b0a660bd54fcd21b","last_reissued_at":"2026-07-05T07:06:05.864448Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:05.864448Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spectral Entry-wise Matrix Estimation for Low-Rank Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexandre Proutiere, Stefan Stojanovic, Yassir Jedra","submitted_at":"2023-10-10T17:06:41Z","abstract_excerpt":"We study matrix estimation problems arising in reinforcement learning (RL) with low-rank structure. In low-rank bandits, the matrix to be recovered specifies the expected arm rewards, and for low-rank Markov Decision Processes (MDPs), it may for example characterize the transition kernel of the MDP. In both cases, each entry of the matrix carries important information, and we seek estimation methods with low entry-wise error. Importantly, these methods further need to accommodate for inherent correlations in the available data (e.g. for MDPs, the data consists of system trajectories). We inves"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06793","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06793","created_at":"2026-07-05T07:06:05.864507+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06793v2","created_at":"2026-07-05T07:06:05.864507+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06793","created_at":"2026-07-05T07:06:05.864507+00:00"},{"alias_kind":"pith_short_12","alias_value":"6HUZ5I42UFI5","created_at":"2026-07-05T07:06:05.864507+00:00"},{"alias_kind":"pith_short_16","alias_value":"6HUZ5I42UFI5JWRW","created_at":"2026-07-05T07:06:05.864507+00:00"},{"alias_kind":"pith_short_8","alias_value":"6HUZ5I42","created_at":"2026-07-05T07:06:05.864507+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R","json":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R.json","graph_json":"https://pith.science/api/pith-number/6HUZ5I42UFI5JWRWMABVRYDY7R/graph.json","events_json":"https://pith.science/api/pith-number/6HUZ5I42UFI5JWRWMABVRYDY7R/events.json","paper":"https://pith.science/paper/6HUZ5I42"},"agent_actions":{"view_html":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R","download_json":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R.json","view_paper":"https://pith.science/paper/6HUZ5I42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06793&json=true","fetch_graph":"https://pith.science/api/pith-number/6HUZ5I42UFI5JWRWMABVRYDY7R/graph.json","fetch_events":"https://pith.science/api/pith-number/6HUZ5I42UFI5JWRWMABVRYDY7R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R/action/storage_attestation","attest_author":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R/action/author_attestation","sign_citation":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R/action/citation_signature","submit_replication":"https://pith.science/pith/6HUZ5I42UFI5JWRWMABVRYDY7R/action/replication_record"}},"created_at":"2026-07-05T07:06:05.864507+00:00","updated_at":"2026-07-05T07:06:05.864507+00:00"}