{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QLRLDKGT7MF2UYRAP2LWAFIASZ","short_pith_number":"pith:QLRLDKGT","schema_version":"1.0","canonical_sha256":"82e2b1a8d3fb0baa62207e97601500964f2c8c1c24a1d70359db83ceb161f394","source":{"kind":"arxiv","id":"2101.12127","version":2},"attestation_state":"computed","paper":{"title":"tf.data: A Machine Learning Data Processing Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MS"],"primary_cat":"cs.LG","authors_text":"Ana Klimovic, Derek G. Murray, Ihor Indyk, Jiri Simsa","submitted_at":"2021-01-28T17:16:46Z","abstract_excerpt":"Training machine learning models requires feeding input data for models to ingest. Input pipelines for machine learning jobs are often challenging to implement efficiently as they require reading large volumes of data, applying complex transformations, and transferring data to hardware accelerators while overlapping computation and communication to achieve optimal performance. We present tf.data, a framework for building and executing efficient input pipelines for machine learning jobs. The tf.data API provides operators which can be parameterized with user-defined computation, composed, and r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.12127","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-01-28T17:16:46Z","cross_cats_sorted":["cs.MS"],"title_canon_sha256":"dfdf7bc9eb41314d361409fa41c7b13b667b8421ed1f1dba23477f7ca42d6762","abstract_canon_sha256":"d06d916117afd5ed91d1a4f398d574b07eebb52d84b788cd8e6253c6d042b0e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:17:56.097202Z","signature_b64":"TXhBC6DrsVyLtOshJWj2vsioq6GlTpQbIHKxOqLlBb3z9V3QLm32n8MysSqWu0q6YiCtwmk7DHDIp+cr+4blAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"82e2b1a8d3fb0baa62207e97601500964f2c8c1c24a1d70359db83ceb161f394","last_reissued_at":"2026-07-05T02:17:56.096712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:17:56.096712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"tf.data: A Machine Learning Data Processing Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MS"],"primary_cat":"cs.LG","authors_text":"Ana Klimovic, Derek G. Murray, Ihor Indyk, Jiri Simsa","submitted_at":"2021-01-28T17:16:46Z","abstract_excerpt":"Training machine learning models requires feeding input data for models to ingest. Input pipelines for machine learning jobs are often challenging to implement efficiently as they require reading large volumes of data, applying complex transformations, and transferring data to hardware accelerators while overlapping computation and communication to achieve optimal performance. We present tf.data, a framework for building and executing efficient input pipelines for machine learning jobs. The tf.data API provides operators which can be parameterized with user-defined computation, composed, and r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.12127","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.12127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.12127","created_at":"2026-07-05T02:17:56.096771+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.12127v2","created_at":"2026-07-05T02:17:56.096771+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.12127","created_at":"2026-07-05T02:17:56.096771+00:00"},{"alias_kind":"pith_short_12","alias_value":"QLRLDKGT7MF2","created_at":"2026-07-05T02:17:56.096771+00:00"},{"alias_kind":"pith_short_16","alias_value":"QLRLDKGT7MF2UYRA","created_at":"2026-07-05T02:17:56.096771+00:00"},{"alias_kind":"pith_short_8","alias_value":"QLRLDKGT","created_at":"2026-07-05T02:17:56.096771+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.09844","citing_title":"MegaScale-Data: Scaling Dataloader for Multisource Large Foundation Model Training","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ","json":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ.json","graph_json":"https://pith.science/api/pith-number/QLRLDKGT7MF2UYRAP2LWAFIASZ/graph.json","events_json":"https://pith.science/api/pith-number/QLRLDKGT7MF2UYRAP2LWAFIASZ/events.json","paper":"https://pith.science/paper/QLRLDKGT"},"agent_actions":{"view_html":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ","download_json":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ.json","view_paper":"https://pith.science/paper/QLRLDKGT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.12127&json=true","fetch_graph":"https://pith.science/api/pith-number/QLRLDKGT7MF2UYRAP2LWAFIASZ/graph.json","fetch_events":"https://pith.science/api/pith-number/QLRLDKGT7MF2UYRAP2LWAFIASZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ/action/storage_attestation","attest_author":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ/action/author_attestation","sign_citation":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ/action/citation_signature","submit_replication":"https://pith.science/pith/QLRLDKGT7MF2UYRAP2LWAFIASZ/action/replication_record"}},"created_at":"2026-07-05T02:17:56.096771+00:00","updated_at":"2026-07-05T02:17:56.096771+00:00"}