{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5US5UV4UMZTC5VKUHQLCC6CREE","short_pith_number":"pith:5US5UV4U","schema_version":"1.0","canonical_sha256":"ed25da579466662ed5543c162178512110d8fe027aafe05d9ac78bfe7c3d8a03","source":{"kind":"arxiv","id":"2112.00086","version":1},"attestation_state":"computed","paper":{"title":"Dyna-bAbI: unlocking bAbI's potential with dynamic synthetic benchmarking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aviad Sar-Shalom, Dafna Shahaf, Kyle Richardson, Nelson Liu, Noam Kahlon, Reut Tsarfaty, Ronen Tamari","submitted_at":"2021-11-30T20:36:56Z","abstract_excerpt":"While neural language models often perform surprisingly well on natural language understanding (NLU) tasks, their strengths and limitations remain poorly understood. Controlled synthetic tasks are thus an increasingly important resource for diagnosing model behavior. In this work we focus on story understanding, a core competency for NLU systems. However, the main synthetic resource for story understanding, the bAbI benchmark, lacks such a systematic mechanism for controllable task generation. We develop Dyna-bAbI, a dynamic framework providing fine-grained control over task generation in bAbI"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.00086","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-11-30T20:36:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4eb5206b09a3244f5d22e61985ad0573dd8906b1bc9e55d8fe36c90dcc3043ea","abstract_canon_sha256":"72fa9937854b7ed53232a2ade4f185f256b249879149a84d87368f40bdcd4b43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:36:42.367404Z","signature_b64":"pRE0hUSU8XjfjI/wPUqJ9iE6seC2FoFprDkr7vPHwG0p8HLxt3xDo8vDBpnlLf+7LSDhTOQTURQBxsfgzyIoBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed25da579466662ed5543c162178512110d8fe027aafe05d9ac78bfe7c3d8a03","last_reissued_at":"2026-07-05T03:36:42.366984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:36:42.366984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dyna-bAbI: unlocking bAbI's potential with dynamic synthetic benchmarking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aviad Sar-Shalom, Dafna Shahaf, Kyle Richardson, Nelson Liu, Noam Kahlon, Reut Tsarfaty, Ronen Tamari","submitted_at":"2021-11-30T20:36:56Z","abstract_excerpt":"While neural language models often perform surprisingly well on natural language understanding (NLU) tasks, their strengths and limitations remain poorly understood. Controlled synthetic tasks are thus an increasingly important resource for diagnosing model behavior. In this work we focus on story understanding, a core competency for NLU systems. However, the main synthetic resource for story understanding, the bAbI benchmark, lacks such a systematic mechanism for controllable task generation. We develop Dyna-bAbI, a dynamic framework providing fine-grained control over task generation in bAbI"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.00086","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.00086/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.00086","created_at":"2026-07-05T03:36:42.367042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.00086v1","created_at":"2026-07-05T03:36:42.367042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.00086","created_at":"2026-07-05T03:36:42.367042+00:00"},{"alias_kind":"pith_short_12","alias_value":"5US5UV4UMZTC","created_at":"2026-07-05T03:36:42.367042+00:00"},{"alias_kind":"pith_short_16","alias_value":"5US5UV4UMZTC5VKU","created_at":"2026-07-05T03:36:42.367042+00:00"},{"alias_kind":"pith_short_8","alias_value":"5US5UV4U","created_at":"2026-07-05T03:36:42.367042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.02940","citing_title":"Towards a Comparative Framework for Compositional AI Models","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE","json":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE.json","graph_json":"https://pith.science/api/pith-number/5US5UV4UMZTC5VKUHQLCC6CREE/graph.json","events_json":"https://pith.science/api/pith-number/5US5UV4UMZTC5VKUHQLCC6CREE/events.json","paper":"https://pith.science/paper/5US5UV4U"},"agent_actions":{"view_html":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE","download_json":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE.json","view_paper":"https://pith.science/paper/5US5UV4U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.00086&json=true","fetch_graph":"https://pith.science/api/pith-number/5US5UV4UMZTC5VKUHQLCC6CREE/graph.json","fetch_events":"https://pith.science/api/pith-number/5US5UV4UMZTC5VKUHQLCC6CREE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE/action/storage_attestation","attest_author":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE/action/author_attestation","sign_citation":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE/action/citation_signature","submit_replication":"https://pith.science/pith/5US5UV4UMZTC5VKUHQLCC6CREE/action/replication_record"}},"created_at":"2026-07-05T03:36:42.367042+00:00","updated_at":"2026-07-05T03:36:42.367042+00:00"}