{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:NEYGMN6WM7LEEN3JGG2HKSIJOS","short_pith_number":"pith:NEYGMN6W","canonical_record":{"source":{"id":"2310.16681","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T14:45:48Z","cross_cats_sorted":[],"title_canon_sha256":"33cdf1adcf01d9d5c05c48e8ac00fb7ea8e95b582318a368a1862d46ba565705","abstract_canon_sha256":"c3f4788e9e3957d0875ba0d67b872e2f9ea9585638f7018f44fede04024436ad"},"schema_version":"1.0"},"canonical_sha256":"69306637d667d642376931b47549097488f543aaf756c23f15ae543610a84bbe","source":{"kind":"arxiv","id":"2310.16681","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.16681","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"arxiv_version","alias_value":"2310.16681v1","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16681","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_12","alias_value":"NEYGMN6WM7LE","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_16","alias_value":"NEYGMN6WM7LEEN3J","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_8","alias_value":"NEYGMN6W","created_at":"2026-07-05T07:05:01Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:NEYGMN6WM7LEEN3JGG2HKSIJOS","target":"record","payload":{"canonical_record":{"source":{"id":"2310.16681","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T14:45:48Z","cross_cats_sorted":[],"title_canon_sha256":"33cdf1adcf01d9d5c05c48e8ac00fb7ea8e95b582318a368a1862d46ba565705","abstract_canon_sha256":"c3f4788e9e3957d0875ba0d67b872e2f9ea9585638f7018f44fede04024436ad"},"schema_version":"1.0"},"canonical_sha256":"69306637d667d642376931b47549097488f543aaf756c23f15ae543610a84bbe","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:01.542625Z","signature_b64":"xBf5AzyplRIjitZv4ILBbL6On+Ufsr4ot458kpUwIFR0wne/ygDwBDSHX51loBfYEgrUpUA5eGOP091hYewVAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69306637d667d642376931b47549097488f543aaf756c23f15ae543610a84bbe","last_reissued_at":"2026-07-05T07:05:01.542111Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:01.542111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.16681","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:05:01Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2jXkI3Y7xLpA11JMuqtnpAWc1xpLpVL5vvPhZqNKvLEZ55BwkoBgRJfHK3JIw/GBGF/3HGGUNDp0M1cYUvs5Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T23:28:41.902224Z"},"content_sha256":"31879449154ae69375a00276e1e158f209c7eec150e165bbb4ba161f263f88ba","schema_version":"1.0","event_id":"sha256:31879449154ae69375a00276e1e158f209c7eec150e165bbb4ba161f263f88ba"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:NEYGMN6WM7LEEN3JGG2HKSIJOS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"BabyStories: Can Reinforcement Learning Teach Baby Language Models to Write Better Stories?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anthony Rios, Sheri Osborn, Tongnian Wang, Xingmeng Zhao","submitted_at":"2023-10-25T14:45:48Z","abstract_excerpt":"Language models have seen significant growth in the size of their corpus, leading to notable performance improvements. Yet, there has been limited progress in developing models that handle smaller, more human-like datasets. As part of the BabyLM shared task, this study explores the impact of reinforcement learning from human feedback (RLHF) on language models pretrained from scratch with a limited training corpus. Comparing two GPT-2 variants, the larger model performs better in storytelling tasks after RLHF fine-tuning. These findings suggest that RLHF techniques may be more advantageous for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16681","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:05:01Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"B6dwi+GKAp9fkDx/jJr9/XnJnbwmrFrvCARipgPSYQhtPnne0Ubb3VRHXo/ml3frpY/zV0NUuVVdoYTYVqIWAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T23:28:41.902745Z"},"content_sha256":"215e24140b69b6aa9be6bab9ee29c7f0345adb789adf7349f9667f412017a81e","schema_version":"1.0","event_id":"sha256:215e24140b69b6aa9be6bab9ee29c7f0345adb789adf7349f9667f412017a81e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/bundle.json","state_url":"https://pith.science/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T23:28:41Z","links":{"resolver":"https://pith.science/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS","bundle":"https://pith.science/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/bundle.json","state":"https://pith.science/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NEYGMN6WM7LEEN3JGG2HKSIJOS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:NEYGMN6WM7LEEN3JGG2HKSIJOS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c3f4788e9e3957d0875ba0d67b872e2f9ea9585638f7018f44fede04024436ad","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T14:45:48Z","title_canon_sha256":"33cdf1adcf01d9d5c05c48e8ac00fb7ea8e95b582318a368a1862d46ba565705"},"schema_version":"1.0","source":{"id":"2310.16681","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.16681","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"arxiv_version","alias_value":"2310.16681v1","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16681","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_12","alias_value":"NEYGMN6WM7LE","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_16","alias_value":"NEYGMN6WM7LEEN3J","created_at":"2026-07-05T07:05:01Z"},{"alias_kind":"pith_short_8","alias_value":"NEYGMN6W","created_at":"2026-07-05T07:05:01Z"}],"graph_snapshots":[{"event_id":"sha256:215e24140b69b6aa9be6bab9ee29c7f0345adb789adf7349f9667f412017a81e","target":"graph","created_at":"2026-07-05T07:05:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.16681/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language models have seen significant growth in the size of their corpus, leading to notable performance improvements. Yet, there has been limited progress in developing models that handle smaller, more human-like datasets. As part of the BabyLM shared task, this study explores the impact of reinforcement learning from human feedback (RLHF) on language models pretrained from scratch with a limited training corpus. Comparing two GPT-2 variants, the larger model performs better in storytelling tasks after RLHF fine-tuning. These findings suggest that RLHF techniques may be more advantageous for ","authors_text":"Anthony Rios, Sheri Osborn, Tongnian Wang, Xingmeng Zhao","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T14:45:48Z","title":"BabyStories: Can Reinforcement Learning Teach Baby Language Models to Write Better Stories?"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16681","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:31879449154ae69375a00276e1e158f209c7eec150e165bbb4ba161f263f88ba","target":"record","created_at":"2026-07-05T07:05:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c3f4788e9e3957d0875ba0d67b872e2f9ea9585638f7018f44fede04024436ad","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T14:45:48Z","title_canon_sha256":"33cdf1adcf01d9d5c05c48e8ac00fb7ea8e95b582318a368a1862d46ba565705"},"schema_version":"1.0","source":{"id":"2310.16681","kind":"arxiv","version":1}},"canonical_sha256":"69306637d667d642376931b47549097488f543aaf756c23f15ae543610a84bbe","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"69306637d667d642376931b47549097488f543aaf756c23f15ae543610a84bbe","first_computed_at":"2026-07-05T07:05:01.542111Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:05:01.542111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"xBf5AzyplRIjitZv4ILBbL6On+Ufsr4ot458kpUwIFR0wne/ygDwBDSHX51loBfYEgrUpUA5eGOP091hYewVAA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:05:01.542625Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.16681","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:31879449154ae69375a00276e1e158f209c7eec150e165bbb4ba161f263f88ba","sha256:215e24140b69b6aa9be6bab9ee29c7f0345adb789adf7349f9667f412017a81e"],"state_sha256":"051cc2e328947c842e68dad15f2717e5f9d09afbbc1c7e74f825c6c521d16f52"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ijH9Dv1pql/ENtcOm8i/5jKrRYab+FBJWDBi0cpuxLONT+YjtPpTMe0c6SmVS0ehlSMO881+EXhZGctc8luhAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T23:28:41.907948Z","bundle_sha256":"edfe734559b0517256cc62491fd7de6077f78f873fffe74570a6805c17c3aa2e"}}