{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DFIU34BCCQOZB2JAZKRJU5LQHU","short_pith_number":"pith:DFIU34BC","schema_version":"1.0","canonical_sha256":"19514df022141d90e920caa29a75703d28b3d6939cb366a928f241e37564e11d","source":{"kind":"arxiv","id":"2505.11063","version":3},"attestation_state":"computed","paper":{"title":"Think Twice Before You Act: Enhancing Agent Behavioral Safety with Thought Correction","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Changyue Jiang, Geng Hong, Min Yang, Wenqi Zhang, Xudong Pan","submitted_at":"2025-05-16T10:00:15Z","abstract_excerpt":"LLM-based agents solve complex tasks through iterative reasoning, tool use, and environment interaction, where each intermediate thought directly shapes subsequent actions. Small deviations in these thoughts can therefore propagate into unsafe behaviors, yet existing guardrails typically operate only on final outputs or require intrusive model modifications. We introduce Thought-Aligner, a lightweight plug-in safety model that performs causal correction on unsafe thoughts before action execution, without altering the underlying agent. The corrected thoughts are fed back into the agent, steerin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11063","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-16T10:00:15Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"711b59bb075838dcc34c55d068321e3b16b81ccab1ea61fb1e6bff2a9d3b0d72","abstract_canon_sha256":"d23495679e14b62432017021b0edf3af2e642ef6dd299162375d8688d9397698"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-27T01:05:34.702024Z","signature_b64":"xOht33MC0SegRxVRNM+tjAhkfNFwlOEq054MzpMAl/VKMx07cyA3V8Ziy7w018JPqo9VwyZ6fJUN/15fdNKfAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19514df022141d90e920caa29a75703d28b3d6939cb366a928f241e37564e11d","last_reissued_at":"2026-05-27T01:05:34.701297Z","signature_status":"signed_v1","first_computed_at":"2026-05-27T01:05:34.701297Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Think Twice Before You Act: Enhancing Agent Behavioral Safety with Thought Correction","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Changyue Jiang, Geng Hong, Min Yang, Wenqi Zhang, Xudong Pan","submitted_at":"2025-05-16T10:00:15Z","abstract_excerpt":"LLM-based agents solve complex tasks through iterative reasoning, tool use, and environment interaction, where each intermediate thought directly shapes subsequent actions. Small deviations in these thoughts can therefore propagate into unsafe behaviors, yet existing guardrails typically operate only on final outputs or require intrusive model modifications. We introduce Thought-Aligner, a lightweight plug-in safety model that performs causal correction on unsafe thoughts before action execution, without altering the underlying agent. The corrected thoughts are fed back into the agent, steerin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11063","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11063/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11063","created_at":"2026-05-27T01:05:34.701390+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11063v3","created_at":"2026-05-27T01:05:34.701390+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11063","created_at":"2026-05-27T01:05:34.701390+00:00"},{"alias_kind":"pith_short_12","alias_value":"DFIU34BCCQOZ","created_at":"2026-05-27T01:05:34.701390+00:00"},{"alias_kind":"pith_short_16","alias_value":"DFIU34BCCQOZB2JA","created_at":"2026-05-27T01:05:34.701390+00:00"},{"alias_kind":"pith_short_8","alias_value":"DFIU34BC","created_at":"2026-05-27T01:05:34.701390+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.20874","citing_title":"Governance by Construction for Generalist Agents","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.08612","citing_title":"ATAAT: Adaptive Threat-Aware Adversarial Tuning Framework against Backdoor Attacks on Vision-Language-Action Models","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU","json":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU.json","graph_json":"https://pith.science/api/pith-number/DFIU34BCCQOZB2JAZKRJU5LQHU/graph.json","events_json":"https://pith.science/api/pith-number/DFIU34BCCQOZB2JAZKRJU5LQHU/events.json","paper":"https://pith.science/paper/DFIU34BC"},"agent_actions":{"view_html":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU","download_json":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU.json","view_paper":"https://pith.science/paper/DFIU34BC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11063&json=true","fetch_graph":"https://pith.science/api/pith-number/DFIU34BCCQOZB2JAZKRJU5LQHU/graph.json","fetch_events":"https://pith.science/api/pith-number/DFIU34BCCQOZB2JAZKRJU5LQHU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU/action/storage_attestation","attest_author":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU/action/author_attestation","sign_citation":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU/action/citation_signature","submit_replication":"https://pith.science/pith/DFIU34BCCQOZB2JAZKRJU5LQHU/action/replication_record"}},"created_at":"2026-05-27T01:05:34.701390+00:00","updated_at":"2026-05-27T01:05:34.701390+00:00"}