{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:WA6LMMN4UYYYEX6D5C2R3NMVPN","short_pith_number":"pith:WA6LMMN4","canonical_record":{"source":{"id":"2411.14457","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T22:00:29Z","cross_cats_sorted":[],"title_canon_sha256":"cd5853c31b9f4712243d5ab180d407dc2b1e594074147906cac76ea6d3780f20","abstract_canon_sha256":"19a8b7de0a79a00ea61e0cb710c07460a8a7a214dd07b2102afd0328aa7ac31f"},"schema_version":"1.0"},"canonical_sha256":"b03cb631bca631825fc3e8b51db5957b515ebfeef8cf868a4254af313130d2ad","source":{"kind":"arxiv","id":"2411.14457","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.14457","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"arxiv_version","alias_value":"2411.14457v1","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14457","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_12","alias_value":"WA6LMMN4UYYY","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_16","alias_value":"WA6LMMN4UYYYEX6D","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_8","alias_value":"WA6LMMN4","created_at":"2026-07-05T09:38:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:WA6LMMN4UYYYEX6D5C2R3NMVPN","target":"record","payload":{"canonical_record":{"source":{"id":"2411.14457","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T22:00:29Z","cross_cats_sorted":[],"title_canon_sha256":"cd5853c31b9f4712243d5ab180d407dc2b1e594074147906cac76ea6d3780f20","abstract_canon_sha256":"19a8b7de0a79a00ea61e0cb710c07460a8a7a214dd07b2102afd0328aa7ac31f"},"schema_version":"1.0"},"canonical_sha256":"b03cb631bca631825fc3e8b51db5957b515ebfeef8cf868a4254af313130d2ad","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:51.431854Z","signature_b64":"DLAt7kwyL9blnqXhmLfGIGN47ywufbC5jeBeIKyMGtvjke7mrVfnb2aSb2LzvXzHjCDB2D5zJ0uz7f3dW5pEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b03cb631bca631825fc3e8b51db5957b515ebfeef8cf868a4254af313130d2ad","last_reissued_at":"2026-07-05T09:38:51.431388Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:51.431388Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2411.14457","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:38:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/mXduvWrtAnNfjE9wMkFAuoZBvVPsgOlwpf2no3nGcXKAd3QAZor33ejgcSDdw9CIni5gbCD4wizvunYJGSwBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:13:15.980319Z"},"content_sha256":"1794b7edc0cabd26313444bcfe090067654ccb402979c7feeaf7ed60911a3c9e","schema_version":"1.0","event_id":"sha256:1794b7edc0cabd26313444bcfe090067654ccb402979c7feeaf7ed60911a3c9e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:WA6LMMN4UYYYEX6D5C2R3NMVPN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Guiding Reinforcement Learning Using Uncertainty-Aware Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Brent Harrison, Maryam Shoaeinaeini","submitted_at":"2024-11-15T22:00:29Z","abstract_excerpt":"Human guidance in reinforcement learning (RL) is often impractical for large-scale applications due to high costs and time constraints. Large Language Models (LLMs) offer a promising alternative to mitigate RL sample inefficiency and potentially replace human trainers. However, applying LLMs as RL trainers is challenging due to their overconfidence and less reliable solutions in sequential tasks. We address this limitation by introducing a calibrated guidance system that uses Monte Carlo Dropout to enhance LLM advice reliability by assessing prediction variances from multiple forward passes. A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14457","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:38:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hnmlpqpd2Y6piaZ+rDq837VnprkvNKjEHaX+8W0wlZ/dphvDO8OOINofA7xodNTHbjo5vwF9tCG9J6uU0OH+Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:13:15.981280Z"},"content_sha256":"ca6075f3e104f60d634e7757eabc5cd456c53781337673b633b9e94b0eb76aab","schema_version":"1.0","event_id":"sha256:ca6075f3e104f60d634e7757eabc5cd456c53781337673b633b9e94b0eb76aab"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/bundle.json","state_url":"https://pith.science/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T22:13:15Z","links":{"resolver":"https://pith.science/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN","bundle":"https://pith.science/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/bundle.json","state":"https://pith.science/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WA6LMMN4UYYYEX6D5C2R3NMVPN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:WA6LMMN4UYYYEX6D5C2R3NMVPN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"19a8b7de0a79a00ea61e0cb710c07460a8a7a214dd07b2102afd0328aa7ac31f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T22:00:29Z","title_canon_sha256":"cd5853c31b9f4712243d5ab180d407dc2b1e594074147906cac76ea6d3780f20"},"schema_version":"1.0","source":{"id":"2411.14457","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.14457","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"arxiv_version","alias_value":"2411.14457v1","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14457","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_12","alias_value":"WA6LMMN4UYYY","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_16","alias_value":"WA6LMMN4UYYYEX6D","created_at":"2026-07-05T09:38:51Z"},{"alias_kind":"pith_short_8","alias_value":"WA6LMMN4","created_at":"2026-07-05T09:38:51Z"}],"graph_snapshots":[{"event_id":"sha256:ca6075f3e104f60d634e7757eabc5cd456c53781337673b633b9e94b0eb76aab","target":"graph","created_at":"2026-07-05T09:38:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.14457/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Human guidance in reinforcement learning (RL) is often impractical for large-scale applications due to high costs and time constraints. Large Language Models (LLMs) offer a promising alternative to mitigate RL sample inefficiency and potentially replace human trainers. However, applying LLMs as RL trainers is challenging due to their overconfidence and less reliable solutions in sequential tasks. We address this limitation by introducing a calibrated guidance system that uses Monte Carlo Dropout to enhance LLM advice reliability by assessing prediction variances from multiple forward passes. A","authors_text":"Brent Harrison, Maryam Shoaeinaeini","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T22:00:29Z","title":"Guiding Reinforcement Learning Using Uncertainty-Aware Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14457","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1794b7edc0cabd26313444bcfe090067654ccb402979c7feeaf7ed60911a3c9e","target":"record","created_at":"2026-07-05T09:38:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"19a8b7de0a79a00ea61e0cb710c07460a8a7a214dd07b2102afd0328aa7ac31f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T22:00:29Z","title_canon_sha256":"cd5853c31b9f4712243d5ab180d407dc2b1e594074147906cac76ea6d3780f20"},"schema_version":"1.0","source":{"id":"2411.14457","kind":"arxiv","version":1}},"canonical_sha256":"b03cb631bca631825fc3e8b51db5957b515ebfeef8cf868a4254af313130d2ad","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b03cb631bca631825fc3e8b51db5957b515ebfeef8cf868a4254af313130d2ad","first_computed_at":"2026-07-05T09:38:51.431388Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:38:51.431388Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"DLAt7kwyL9blnqXhmLfGIGN47ywufbC5jeBeIKyMGtvjke7mrVfnb2aSb2LzvXzHjCDB2D5zJ0uz7f3dW5pEAA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:38:51.431854Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.14457","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1794b7edc0cabd26313444bcfe090067654ccb402979c7feeaf7ed60911a3c9e","sha256:ca6075f3e104f60d634e7757eabc5cd456c53781337673b633b9e94b0eb76aab"],"state_sha256":"ab8645841cf0be9ee0c481a4b250c2960b1be4b08b71775ac461b056c1ae21e0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2c3bx2Me1DkMqxRSW25nXlsvsrRwiT6AhLT0nDatz3kKdVa/UVHE2rg+KcVOJgximqaVeQTZsNa7UCekAVweCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T22:13:15.987342Z","bundle_sha256":"d0a55fabd3d4b0901cb94f254e35f9979c77f15f738db87d574a7769564fef9b"}}