{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SZNJJZEWDV3G36VR7AEV4WIESK","short_pith_number":"pith:SZNJJZEW","schema_version":"1.0","canonical_sha256":"965a94e4961d766dfab1f8095e590492a985f939a4b703a9c3116afa868c019c","source":{"kind":"arxiv","id":"2308.14947","version":2},"attestation_state":"computed","paper":{"title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.RO","authors_text":"Adam Sigal, AJung Moon, Hsiu-Chin Lin","submitted_at":"2023-08-29T00:00:18Z","abstract_excerpt":"In order for autonomous mobile robots to navigate in human spaces, they must abide by our social norms. Reinforcement learning (RL) has emerged as an effective method to train sequential decision-making policies that are able to respect these norms. However, a large portion of existing work in the field conducts both RL training and testing in simplistic environments. This limits the generalization potential of these models to unseen environments, and the meaningfulness of their reported results. We propose a method to improve the generalization performance of RL social navigation methods usin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.14947","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-08-29T00:00:18Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"102ece94a4a73dd143fa6251dc66968d33b42010a52c6ef5cf6858e874e1f49b","abstract_canon_sha256":"4d2eff0ded619708b98318fa35c22b78a9fdd774f9656d4968fd620e42fedb75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:20.230457Z","signature_b64":"MoP/JhtMGjWA03/i+hwzAGhpeHZbOd0801j3ioxWoj4vQG3PoXYHOVgt5bhYQnF38JWrW5nOtbqJtAfcAUV0AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"965a94e4961d766dfab1f8095e590492a985f939a4b703a9c3116afa868c019c","last_reissued_at":"2026-07-05T07:50:20.230038Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:20.230038Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.RO","authors_text":"Adam Sigal, AJung Moon, Hsiu-Chin Lin","submitted_at":"2023-08-29T00:00:18Z","abstract_excerpt":"In order for autonomous mobile robots to navigate in human spaces, they must abide by our social norms. Reinforcement learning (RL) has emerged as an effective method to train sequential decision-making policies that are able to respect these norms. However, a large portion of existing work in the field conducts both RL training and testing in simplistic environments. This limits the generalization potential of these models to unseen environments, and the meaningfulness of their reported results. We propose a method to improve the generalization performance of RL social navigation methods usin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.14947","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.14947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.14947","created_at":"2026-07-05T07:50:20.230103+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.14947v2","created_at":"2026-07-05T07:50:20.230103+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.14947","created_at":"2026-07-05T07:50:20.230103+00:00"},{"alias_kind":"pith_short_12","alias_value":"SZNJJZEWDV3G","created_at":"2026-07-05T07:50:20.230103+00:00"},{"alias_kind":"pith_short_16","alias_value":"SZNJJZEWDV3G36VR","created_at":"2026-07-05T07:50:20.230103+00:00"},{"alias_kind":"pith_short_8","alias_value":"SZNJJZEW","created_at":"2026-07-05T07:50:20.230103+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.01539","citing_title":"HALO: Human Preference Aligned Offline Reward Learning for Robot Navigation","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK","json":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK.json","graph_json":"https://pith.science/api/pith-number/SZNJJZEWDV3G36VR7AEV4WIESK/graph.json","events_json":"https://pith.science/api/pith-number/SZNJJZEWDV3G36VR7AEV4WIESK/events.json","paper":"https://pith.science/paper/SZNJJZEW"},"agent_actions":{"view_html":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK","download_json":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK.json","view_paper":"https://pith.science/paper/SZNJJZEW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.14947&json=true","fetch_graph":"https://pith.science/api/pith-number/SZNJJZEWDV3G36VR7AEV4WIESK/graph.json","fetch_events":"https://pith.science/api/pith-number/SZNJJZEWDV3G36VR7AEV4WIESK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK/action/storage_attestation","attest_author":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK/action/author_attestation","sign_citation":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK/action/citation_signature","submit_replication":"https://pith.science/pith/SZNJJZEWDV3G36VR7AEV4WIESK/action/replication_record"}},"created_at":"2026-07-05T07:50:20.230103+00:00","updated_at":"2026-07-05T07:50:20.230103+00:00"}