{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:MGCOCDUT374323PA6Y6JDIYHWG","short_pith_number":"pith:MGCOCDUT","schema_version":"1.0","canonical_sha256":"6184e10e93dff9bd6de0f63c91a307b197ea249db79937a8ed7ad5bc497aba8a","source":{"kind":"arxiv","id":"1908.06177","version":2},"attestation_state":"computed","paper":{"title":"CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jin Dong, Joelle Pineau, Koustuv Sinha, Shagun Sodhani, William L. Hamilton","submitted_at":"2019-08-16T21:12:15Z","abstract_excerpt":"The recent success of natural language understanding (NLU) systems has been troubled by results highlighting the failure of these models to generalize in a systematic and robust way. In this work, we introduce a diagnostic benchmark suite, named CLUTRR, to clarify some key issues related to the robustness and systematicity of NLU systems. Motivated by classic work on inductive logic programming, CLUTRR requires that an NLU system infer kinship relations between characters in short stories. Successful performance on this task requires both extracting relationships between entities, as well as i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.06177","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-16T21:12:15Z","cross_cats_sorted":["cs.CL","cs.LO","stat.ML"],"title_canon_sha256":"16f6a4c8f6dc5c3f64a4c8be3248fb30313e7de384b10104421d01e2dd4c93aa","abstract_canon_sha256":"98e37fb48eb8569266324a17253a97e9b347dde5dbb95b9316d46fbe7c330e35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:14.437558Z","signature_b64":"FE0sBzwXDRWV2swVVIAzE4eDC0x7Pqv+XZrXYyT8wYIPy3oyMkS6rOlWLzPJ5gLnt5lUNfQYACaS9QQzDmjhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6184e10e93dff9bd6de0f63c91a307b197ea249db79937a8ed7ad5bc497aba8a","last_reissued_at":"2026-07-05T00:02:14.437162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:14.437162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jin Dong, Joelle Pineau, Koustuv Sinha, Shagun Sodhani, William L. Hamilton","submitted_at":"2019-08-16T21:12:15Z","abstract_excerpt":"The recent success of natural language understanding (NLU) systems has been troubled by results highlighting the failure of these models to generalize in a systematic and robust way. In this work, we introduce a diagnostic benchmark suite, named CLUTRR, to clarify some key issues related to the robustness and systematicity of NLU systems. Motivated by classic work on inductive logic programming, CLUTRR requires that an NLU system infer kinship relations between characters in short stories. Successful performance on this task requires both extracting relationships between entities, as well as i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.06177","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.06177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.06177","created_at":"2026-07-05T00:02:14.437217+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.06177v2","created_at":"2026-07-05T00:02:14.437217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.06177","created_at":"2026-07-05T00:02:14.437217+00:00"},{"alias_kind":"pith_short_12","alias_value":"MGCOCDUT3743","created_at":"2026-07-05T00:02:14.437217+00:00"},{"alias_kind":"pith_short_16","alias_value":"MGCOCDUT374323PA","created_at":"2026-07-05T00:02:14.437217+00:00"},{"alias_kind":"pith_short_8","alias_value":"MGCOCDUT","created_at":"2026-07-05T00:02:14.437217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29278","citing_title":"The Complexity Ceiling Benchmark: A Multi-Domain Evaluation of Sequential Reasoning Under Depth Scaling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02627","citing_title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2302.04023","citing_title":"A Multitask, Multilingual, Multimodal Evaluation of ChatGPT on Reasoning, Hallucination, and Interactivity","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12426","citing_title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG","json":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG.json","graph_json":"https://pith.science/api/pith-number/MGCOCDUT374323PA6Y6JDIYHWG/graph.json","events_json":"https://pith.science/api/pith-number/MGCOCDUT374323PA6Y6JDIYHWG/events.json","paper":"https://pith.science/paper/MGCOCDUT"},"agent_actions":{"view_html":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG","download_json":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG.json","view_paper":"https://pith.science/paper/MGCOCDUT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.06177&json=true","fetch_graph":"https://pith.science/api/pith-number/MGCOCDUT374323PA6Y6JDIYHWG/graph.json","fetch_events":"https://pith.science/api/pith-number/MGCOCDUT374323PA6Y6JDIYHWG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG/action/storage_attestation","attest_author":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG/action/author_attestation","sign_citation":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG/action/citation_signature","submit_replication":"https://pith.science/pith/MGCOCDUT374323PA6Y6JDIYHWG/action/replication_record"}},"created_at":"2026-07-05T00:02:14.437217+00:00","updated_at":"2026-07-05T00:02:14.437217+00:00"}