{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LY2CT3VU4N4BAGJU5H24J4Q5QA","short_pith_number":"pith:LY2CT3VU","schema_version":"1.0","canonical_sha256":"5e3429eeb4e378101934e9f5c4f21d8038dd236234df99b9c382021dee6d0864","source":{"kind":"arxiv","id":"2509.22310","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Policy Backbone via Shared Network","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bumgeun Park, Donghwan Lee","submitted_at":"2025-09-26T13:14:03Z","abstract_excerpt":"Reinforcement learning (RL) has achieved impressive results across domains, yet learning an optimal policy typically requires extensive interaction data, limiting practical deployment. A common remedy is to leverage priors, such as pre-collected datasets or reference policies, but their utility degrades under task mismatch between training and deployment. While prior work has sought to address this mismatch, it has largely been restricted to in-distribution settings. To address this challenge, we propose Adaptive Policy Backbone (APB), a meta-transfer RL method that inserts lightweight linear "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.22310","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T13:14:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8dee3cb3024c6e9615d8b8a48ee510bd5c586585048616edba15a79490d9ee26","abstract_canon_sha256":"1fd96662393f71b669954c28a588c0227dbe88f5079416b5291149f82b27f5f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-03T01:25:44.464309Z","signature_b64":"j3UWdO/238zfWpjAk5j2z9Bhj8O/JLmjVW2Nj/lY9HzQzfBZb6i4ZvyJ+snaTkPxcveSToTWqphtR9IPFnVwAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e3429eeb4e378101934e9f5c4f21d8038dd236234df99b9c382021dee6d0864","last_reissued_at":"2026-08-03T01:25:44.462556Z","signature_status":"signed_v1","first_computed_at":"2026-08-03T01:25:44.462556Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Policy Backbone via Shared Network","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bumgeun Park, Donghwan Lee","submitted_at":"2025-09-26T13:14:03Z","abstract_excerpt":"Reinforcement learning (RL) has achieved impressive results across domains, yet learning an optimal policy typically requires extensive interaction data, limiting practical deployment. A common remedy is to leverage priors, such as pre-collected datasets or reference policies, but their utility degrades under task mismatch between training and deployment. While prior work has sought to address this mismatch, it has largely been restricted to in-distribution settings. To address this challenge, we propose Adaptive Policy Backbone (APB), a meta-transfer RL method that inserts lightweight linear "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.22310","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.22310/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.22310","created_at":"2026-08-03T01:25:44.463476+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.22310v2","created_at":"2026-08-03T01:25:44.463476+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.22310","created_at":"2026-08-03T01:25:44.463476+00:00"},{"alias_kind":"pith_short_12","alias_value":"LY2CT3VU4N4B","created_at":"2026-08-03T01:25:44.463476+00:00"},{"alias_kind":"pith_short_16","alias_value":"LY2CT3VU4N4BAGJU","created_at":"2026-08-03T01:25:44.463476+00:00"},{"alias_kind":"pith_short_8","alias_value":"LY2CT3VU","created_at":"2026-08-03T01:25:44.463476+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA","json":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA.json","graph_json":"https://pith.science/api/pith-number/LY2CT3VU4N4BAGJU5H24J4Q5QA/graph.json","events_json":"https://pith.science/api/pith-number/LY2CT3VU4N4BAGJU5H24J4Q5QA/events.json","paper":"https://pith.science/paper/LY2CT3VU"},"agent_actions":{"view_html":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA","download_json":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA.json","view_paper":"https://pith.science/paper/LY2CT3VU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.22310&json=true","fetch_graph":"https://pith.science/api/pith-number/LY2CT3VU4N4BAGJU5H24J4Q5QA/graph.json","fetch_events":"https://pith.science/api/pith-number/LY2CT3VU4N4BAGJU5H24J4Q5QA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA/action/storage_attestation","attest_author":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA/action/author_attestation","sign_citation":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA/action/citation_signature","submit_replication":"https://pith.science/pith/LY2CT3VU4N4BAGJU5H24J4Q5QA/action/replication_record"}},"created_at":"2026-08-03T01:25:44.463476+00:00","updated_at":"2026-08-03T01:25:44.463476+00:00"}