{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:NKFJQDTMOVTBHZU5SQV5POY4LO","short_pith_number":"pith:NKFJQDTM","canonical_record":{"source":{"id":"2501.11651","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f28df3ee434cbda6af1645b07025d1f223d5425f4019ac8111a97cfb28b63f0d","abstract_canon_sha256":"81b3fb41acecd9f74f2fd42a2728a8ca717380f4b45fc230e658f500327a6f03"},"schema_version":"1.0"},"canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","source":{"kind":"arxiv","id":"2501.11651","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.11651","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"arxiv_version","alias_value":"2501.11651v2","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11651","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_12","alias_value":"NKFJQDTMOVTB","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_16","alias_value":"NKFJQDTMOVTBHZU5","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_8","alias_value":"NKFJQDTM","created_at":"2026-07-05T11:20:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:NKFJQDTMOVTBHZU5SQV5POY4LO","target":"record","payload":{"canonical_record":{"source":{"id":"2501.11651","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f28df3ee434cbda6af1645b07025d1f223d5425f4019ac8111a97cfb28b63f0d","abstract_canon_sha256":"81b3fb41acecd9f74f2fd42a2728a8ca717380f4b45fc230e658f500327a6f03"},"schema_version":"1.0"},"canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:49.856172Z","signature_b64":"GxE+K5jRyDnhWmhAZnTeY5I4X4HxcQZ6IpopJDXuutf/l4tNHEyrCBmxwG6e81Y/c/krL1MDU8w7bb3+fNQoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","last_reissued_at":"2026-07-05T11:20:49.855663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:49.855663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2501.11651","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:20:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+C/51sqVlQipynUZdqYRpPTgDdbWF9aLA0kxIb5fDFfBuvhMPnzgJnWp6s+3GpREsMkOz2L0vCnoGofmA4ErAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:05:00.571162Z"},"content_sha256":"dcdbeb97688f3a7b47c08fe6ef8a9ecfd203a8613ef914a2fcf0888a6a292bbb","schema_version":"1.0","event_id":"sha256:dcdbeb97688f3a7b47c08fe6ef8a9ecfd203a8613ef914a2fcf0888a6a292bbb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:NKFJQDTMOVTBHZU5SQV5POY4LO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"T1: Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jiajie Zhang, Jie Tang, Juanzi Li, Rui Lu, Xin Lv, Yujiang Li, Yuxiao Dong, Zhenyu Hou, Zijun Yao","submitted_at":"2025-01-20T18:33:33Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in complex reasoning tasks. However, existing approaches mainly rely on imitation learning and struggle to achieve effective test-time scaling. While reinforcement learning (RL) holds promise for enabling self-exploration, recent attempts yield modest improvements in complex reasoning. In this paper, we present T1 to scale RL by encouraging exploration and understand inference scaling. We first initialize the LLM using synthesized chain-of-thought data that integrates trial-and-error and self-verification. To scale RL train"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11651","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:20:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8kzWeqJUSlUnTXEyh8afI+IbcqypN8XU8lGrGAjOWhR+149teg8I+/xw1wWWeux5CvcJNRf630I1EjYC4l3hBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:05:00.571649Z"},"content_sha256":"245f3538aa8b99f166c4a69ada53b8fb179f38315f5eecdc6dd26e1b438196bc","schema_version":"1.0","event_id":"sha256:245f3538aa8b99f166c4a69ada53b8fb179f38315f5eecdc6dd26e1b438196bc"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/bundle.json","state_url":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T10:05:00Z","links":{"resolver":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO","bundle":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/bundle.json","state":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:NKFJQDTMOVTBHZU5SQV5POY4LO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"81b3fb41acecd9f74f2fd42a2728a8ca717380f4b45fc230e658f500327a6f03","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","title_canon_sha256":"f28df3ee434cbda6af1645b07025d1f223d5425f4019ac8111a97cfb28b63f0d"},"schema_version":"1.0","source":{"id":"2501.11651","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.11651","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"arxiv_version","alias_value":"2501.11651v2","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11651","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_12","alias_value":"NKFJQDTMOVTB","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_16","alias_value":"NKFJQDTMOVTBHZU5","created_at":"2026-07-05T11:20:49Z"},{"alias_kind":"pith_short_8","alias_value":"NKFJQDTM","created_at":"2026-07-05T11:20:49Z"}],"graph_snapshots":[{"event_id":"sha256:245f3538aa8b99f166c4a69ada53b8fb179f38315f5eecdc6dd26e1b438196bc","target":"graph","created_at":"2026-07-05T11:20:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2501.11651/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in complex reasoning tasks. However, existing approaches mainly rely on imitation learning and struggle to achieve effective test-time scaling. While reinforcement learning (RL) holds promise for enabling self-exploration, recent attempts yield modest improvements in complex reasoning. In this paper, we present T1 to scale RL by encouraging exploration and understand inference scaling. We first initialize the LLM using synthesized chain-of-thought data that integrates trial-and-error and self-verification. To scale RL train","authors_text":"Jiajie Zhang, Jie Tang, Juanzi Li, Rui Lu, Xin Lv, Yujiang Li, Yuxiao Dong, Zhenyu Hou, Zijun Yao","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","title":"T1: Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11651","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:dcdbeb97688f3a7b47c08fe6ef8a9ecfd203a8613ef914a2fcf0888a6a292bbb","target":"record","created_at":"2026-07-05T11:20:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"81b3fb41acecd9f74f2fd42a2728a8ca717380f4b45fc230e658f500327a6f03","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","title_canon_sha256":"f28df3ee434cbda6af1645b07025d1f223d5425f4019ac8111a97cfb28b63f0d"},"schema_version":"1.0","source":{"id":"2501.11651","kind":"arxiv","version":2}},"canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","first_computed_at":"2026-07-05T11:20:49.855663Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:20:49.855663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"GxE+K5jRyDnhWmhAZnTeY5I4X4HxcQZ6IpopJDXuutf/l4tNHEyrCBmxwG6e81Y/c/krL1MDU8w7bb3+fNQoCg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:20:49.856172Z","signed_message":"canonical_sha256_bytes"},"source_id":"2501.11651","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:dcdbeb97688f3a7b47c08fe6ef8a9ecfd203a8613ef914a2fcf0888a6a292bbb","sha256:245f3538aa8b99f166c4a69ada53b8fb179f38315f5eecdc6dd26e1b438196bc"],"state_sha256":"760c00e0e3260efa6dcce996e70b8a14c8163eca6af7be7659d5f793026e814a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JQXFP2lWHxYxFfqjjy81i4dWj9xIplEVj6xBq+5IKpFsHmjp+xnMlKuhrRLRHpQxoSmHiAI0D5uMla6F3J9xCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T10:05:00.575068Z","bundle_sha256":"d4f3530b6fa7d14d9094c2d436d1733a8082643c79423a4869a5d2e30ed31208"}}