{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KOIXCZ3R3RRMIXBL253554O7YF","short_pith_number":"pith:KOIXCZ3R","schema_version":"1.0","canonical_sha256":"5391716771dc62c45c2bd777def1dfc1540b80c0a1bb3b28de6724a948f2851a","source":{"kind":"arxiv","id":"2506.21718","version":1},"attestation_state":"computed","paper":{"title":"Performance Prediction for Large Systems via Text-to-Text Regression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PF","cs.SE","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Adrian N. Reyes, Arissa Wongpanich, Bangding Yang, Bryan Lewandowski, Cheng-Hsi Lin, Grant C. Forbes, Mohamed S. Abdelfattah, Sagi Perel, Xingyou Song, Yash Akhauri","submitted_at":"2025-06-26T19:10:08Z","abstract_excerpt":"In many industries, predicting metric outcomes of large systems is a fundamental problem, driven largely by traditional tabular regression. However, such methods struggle on complex systems data in the wild such as configuration files or system logs, where feature engineering is often infeasible. We propose text-to-text regression as a general, scalable alternative. For predicting resource efficiency on Borg, Google's massive compute cluster scheduling system, a 60M parameter encoder-decoder, trained from random initialization, achieves up to a near perfect 0.99 (0.9 average) rank correlation "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21718","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T19:10:08Z","cross_cats_sorted":["cs.AI","cs.PF","cs.SE","cs.SY","eess.SY"],"title_canon_sha256":"844b673fde7ede44699e503539f858f4d175fde94694b0f37a19df2fa588b703","abstract_canon_sha256":"77bc5a532800583de599740dc451756f105367f70000cdaa11f21165febc682c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:59.973102Z","signature_b64":"5R/HvWqFufKnTDffZ7HQ0D0i6l8V5sqYsL4AaK+1g6hBhgmfDPdYiLIjlqAH4QQQhRWQmfj1mxk1e1ge7jAdDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5391716771dc62c45c2bd777def1dfc1540b80c0a1bb3b28de6724a948f2851a","last_reissued_at":"2026-07-05T11:27:59.972641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:59.972641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Performance Prediction for Large Systems via Text-to-Text Regression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PF","cs.SE","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Adrian N. Reyes, Arissa Wongpanich, Bangding Yang, Bryan Lewandowski, Cheng-Hsi Lin, Grant C. Forbes, Mohamed S. Abdelfattah, Sagi Perel, Xingyou Song, Yash Akhauri","submitted_at":"2025-06-26T19:10:08Z","abstract_excerpt":"In many industries, predicting metric outcomes of large systems is a fundamental problem, driven largely by traditional tabular regression. However, such methods struggle on complex systems data in the wild such as configuration files or system logs, where feature engineering is often infeasible. We propose text-to-text regression as a general, scalable alternative. For predicting resource efficiency on Borg, Google's massive compute cluster scheduling system, a 60M parameter encoder-decoder, trained from random initialization, achieves up to a near perfect 0.99 (0.9 average) rank correlation "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21718","created_at":"2026-07-05T11:27:59.972697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21718v1","created_at":"2026-07-05T11:27:59.972697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21718","created_at":"2026-07-05T11:27:59.972697+00:00"},{"alias_kind":"pith_short_12","alias_value":"KOIXCZ3R3RRM","created_at":"2026-07-05T11:27:59.972697+00:00"},{"alias_kind":"pith_short_16","alias_value":"KOIXCZ3R3RRMIXBL","created_at":"2026-07-05T11:27:59.972697+00:00"},{"alias_kind":"pith_short_8","alias_value":"KOIXCZ3R","created_at":"2026-07-05T11:27:59.972697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20740","citing_title":"Distribution-Aware Reward: Reinforcement Learning over Predictive Distributions for LLM Regression","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF","json":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF.json","graph_json":"https://pith.science/api/pith-number/KOIXCZ3R3RRMIXBL253554O7YF/graph.json","events_json":"https://pith.science/api/pith-number/KOIXCZ3R3RRMIXBL253554O7YF/events.json","paper":"https://pith.science/paper/KOIXCZ3R"},"agent_actions":{"view_html":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF","download_json":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF.json","view_paper":"https://pith.science/paper/KOIXCZ3R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21718&json=true","fetch_graph":"https://pith.science/api/pith-number/KOIXCZ3R3RRMIXBL253554O7YF/graph.json","fetch_events":"https://pith.science/api/pith-number/KOIXCZ3R3RRMIXBL253554O7YF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF/action/storage_attestation","attest_author":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF/action/author_attestation","sign_citation":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF/action/citation_signature","submit_replication":"https://pith.science/pith/KOIXCZ3R3RRMIXBL253554O7YF/action/replication_record"}},"created_at":"2026-07-05T11:27:59.972697+00:00","updated_at":"2026-07-05T11:27:59.972697+00:00"}