{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VNW5EEBGRGOXNSBC7XCQ2U2QKV","short_pith_number":"pith:VNW5EEBG","schema_version":"1.0","canonical_sha256":"ab6dd21026899d76c822fdc50d53505570bc6d27dd8ef2034287a5dc3690e40a","source":{"kind":"arxiv","id":"2306.09509","version":3},"attestation_state":"computed","paper":{"title":"Granger Causal Interaction Skill Chains","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Aditya Arjun, Caleb Chuck, Kevin Black, Scott Niekum, Yuke Zhu","submitted_at":"2023-06-15T21:06:54Z","abstract_excerpt":"Reinforcement Learning (RL) has demonstrated promising results in learning policies for complex tasks, but it often suffers from low sample efficiency and limited transferability. Hierarchical RL (HRL) methods aim to address the difficulty of learning long-horizon tasks by decomposing policies into skills, abstracting states, and reusing skills in new tasks. However, many HRL methods require some initial task success to discover useful skills, which paradoxically may be very unlikely without access to useful skills. On the other hand, reward-free HRL methods often need to learn far too many sk"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.09509","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-06-15T21:06:54Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"1b99bbebfe29ef6efd2747abcdf8e2f68b08a66ce45c044c818f06bd7cbdf054","abstract_canon_sha256":"a442cdcb44089d5ab103a771de194f082a872278e2e3a2a4c62df900650f62cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:59.552823Z","signature_b64":"8319tm94iKC+WGIu0ZeQbpkfk8qZOFEpYV7hUHpOHYB1touEq7aIUDbYFrcgzeFNh59yEc4u0gf1dBNezFQnDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab6dd21026899d76c822fdc50d53505570bc6d27dd8ef2034287a5dc3690e40a","last_reissued_at":"2026-07-05T09:22:59.552377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:59.552377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Granger Causal Interaction Skill Chains","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Aditya Arjun, Caleb Chuck, Kevin Black, Scott Niekum, Yuke Zhu","submitted_at":"2023-06-15T21:06:54Z","abstract_excerpt":"Reinforcement Learning (RL) has demonstrated promising results in learning policies for complex tasks, but it often suffers from low sample efficiency and limited transferability. Hierarchical RL (HRL) methods aim to address the difficulty of learning long-horizon tasks by decomposing policies into skills, abstracting states, and reusing skills in new tasks. However, many HRL methods require some initial task success to discover useful skills, which paradoxically may be very unlikely without access to useful skills. On the other hand, reward-free HRL methods often need to learn far too many sk"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09509","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.09509","created_at":"2026-07-05T09:22:59.552436+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.09509v3","created_at":"2026-07-05T09:22:59.552436+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09509","created_at":"2026-07-05T09:22:59.552436+00:00"},{"alias_kind":"pith_short_12","alias_value":"VNW5EEBGRGOX","created_at":"2026-07-05T09:22:59.552436+00:00"},{"alias_kind":"pith_short_16","alias_value":"VNW5EEBGRGOXNSBC","created_at":"2026-07-05T09:22:59.552436+00:00"},{"alias_kind":"pith_short_8","alias_value":"VNW5EEBG","created_at":"2026-07-05T09:22:59.552436+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11525","citing_title":"Learning Object Manipulation from Scratch via Contrastive Interaction","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV","json":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV.json","graph_json":"https://pith.science/api/pith-number/VNW5EEBGRGOXNSBC7XCQ2U2QKV/graph.json","events_json":"https://pith.science/api/pith-number/VNW5EEBGRGOXNSBC7XCQ2U2QKV/events.json","paper":"https://pith.science/paper/VNW5EEBG"},"agent_actions":{"view_html":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV","download_json":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV.json","view_paper":"https://pith.science/paper/VNW5EEBG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.09509&json=true","fetch_graph":"https://pith.science/api/pith-number/VNW5EEBGRGOXNSBC7XCQ2U2QKV/graph.json","fetch_events":"https://pith.science/api/pith-number/VNW5EEBGRGOXNSBC7XCQ2U2QKV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV/action/storage_attestation","attest_author":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV/action/author_attestation","sign_citation":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV/action/citation_signature","submit_replication":"https://pith.science/pith/VNW5EEBGRGOXNSBC7XCQ2U2QKV/action/replication_record"}},"created_at":"2026-07-05T09:22:59.552436+00:00","updated_at":"2026-07-05T09:22:59.552436+00:00"}