{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OJKCLLS5CX4I37YGFFQBSHKJCQ","short_pith_number":"pith:OJKCLLS5","schema_version":"1.0","canonical_sha256":"725425ae5d15f88dff062960191d491423aeef019f2accf4418fe4b0882efcb8","source":{"kind":"arxiv","id":"2305.16960","version":3},"attestation_state":"computed","paper":{"title":"Training Socially Aligned Language Models on Simulated Social Interactions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.HC"],"primary_cat":"cs.CL","authors_text":"Andrew M. Dai, Chenyan Jia, Denny Zhou, Diyi Yang, Ge Zhang, Ruibo Liu, Ruixin Yang, Soroush Vosoughi","submitted_at":"2023-05-26T14:17:36Z","abstract_excerpt":"Social alignment in AI systems aims to ensure that these models behave according to established societal values. However, unlike humans, who derive consensus on value judgments through social interaction, current language models (LMs) are trained to rigidly replicate their training corpus in isolation, leading to subpar generalization in unfamiliar scenarios and vulnerability to adversarial attacks. This work presents a novel training paradigm that permits LMs to learn from simulated social interactions. In comparison to existing methodologies, our approach is considerably more scalable and ef"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16960","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-26T14:17:36Z","cross_cats_sorted":["cs.AI","cs.CY","cs.HC"],"title_canon_sha256":"2814c88a678c3628d4e249e7e5a1f31d3e3cf9844316eb1db88ec21fe95a3130","abstract_canon_sha256":"f9acad03896e82517e26eea721ff35176229903a400f556b84651506566757ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:28.168064Z","signature_b64":"dnuW7uYZi6Lfor7eAmEhz1dV5vJ8OGqgVz+2CBK49hj03WMe3JS8WK9d7bvA+01Fvk0bYH+cDzGsWLvJAq3/Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"725425ae5d15f88dff062960191d491423aeef019f2accf4418fe4b0882efcb8","last_reissued_at":"2026-07-05T07:06:28.167573Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:28.167573Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Socially Aligned Language Models on Simulated Social Interactions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.HC"],"primary_cat":"cs.CL","authors_text":"Andrew M. Dai, Chenyan Jia, Denny Zhou, Diyi Yang, Ge Zhang, Ruibo Liu, Ruixin Yang, Soroush Vosoughi","submitted_at":"2023-05-26T14:17:36Z","abstract_excerpt":"Social alignment in AI systems aims to ensure that these models behave according to established societal values. However, unlike humans, who derive consensus on value judgments through social interaction, current language models (LMs) are trained to rigidly replicate their training corpus in isolation, leading to subpar generalization in unfamiliar scenarios and vulnerability to adversarial attacks. This work presents a novel training paradigm that permits LMs to learn from simulated social interactions. In comparison to existing methodologies, our approach is considerably more scalable and ef"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16960","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16960/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16960","created_at":"2026-07-05T07:06:28.167632+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16960v3","created_at":"2026-07-05T07:06:28.167632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16960","created_at":"2026-07-05T07:06:28.167632+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJKCLLS5CX4I","created_at":"2026-07-05T07:06:28.167632+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJKCLLS5CX4I37YG","created_at":"2026-07-05T07:06:28.167632+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJKCLLS5","created_at":"2026-07-05T07:06:28.167632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2502.08691","citing_title":"AgentSociety: Large-Scale Simulation of LLM-Driven Generative Agents Advances Understanding of Human Behaviors and Society","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15207","citing_title":"TeamTR: Trust-Region Fine-Tuning for Multi-Agent LLM Coordination","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2308.05374","citing_title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2309.02427","citing_title":"Cognitive Architectures for Language Agents","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2308.11432","citing_title":"A Survey on Large Language Model based Autonomous Agents","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2312.13010","citing_title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2308.07201","citing_title":"ChatEval: Towards Better LLM-based Evaluators through Multi-Agent Debate","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12195","citing_title":"Representing expertise accelerates learning from pedagogical interaction data","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":196,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ","json":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ.json","graph_json":"https://pith.science/api/pith-number/OJKCLLS5CX4I37YGFFQBSHKJCQ/graph.json","events_json":"https://pith.science/api/pith-number/OJKCLLS5CX4I37YGFFQBSHKJCQ/events.json","paper":"https://pith.science/paper/OJKCLLS5"},"agent_actions":{"view_html":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ","download_json":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ.json","view_paper":"https://pith.science/paper/OJKCLLS5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16960&json=true","fetch_graph":"https://pith.science/api/pith-number/OJKCLLS5CX4I37YGFFQBSHKJCQ/graph.json","fetch_events":"https://pith.science/api/pith-number/OJKCLLS5CX4I37YGFFQBSHKJCQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ/action/storage_attestation","attest_author":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ/action/author_attestation","sign_citation":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ/action/citation_signature","submit_replication":"https://pith.science/pith/OJKCLLS5CX4I37YGFFQBSHKJCQ/action/replication_record"}},"created_at":"2026-07-05T07:06:28.167632+00:00","updated_at":"2026-07-05T07:06:28.167632+00:00"}