{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LQJ7KMTA2OMBMCHNA6I6VTGANQ","short_pith_number":"pith:LQJ7KMTA","schema_version":"1.0","canonical_sha256":"5c13f53260d3981608ed0791eaccc06c1ccbb88756c5300a76b70cb536a6b20e","source":{"kind":"arxiv","id":"2110.03549","version":2},"attestation_state":"computed","paper":{"title":"Bias-Variance Tradeoffs in Single-Sample Binary Gradient Estimators","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Alexander Shekhovtsov","submitted_at":"2021-10-07T15:16:07Z","abstract_excerpt":"Discrete and especially binary random variables occur in many machine learning models, notably in variational autoencoders with binary latent states and in stochastic binary networks. When learning such models, a key tool is an estimator of the gradient of the expected loss with respect to the probabilities of binary variables. The straight-through (ST) estimator gained popularity due to its simplicity and efficiency, in particular in deep networks where unbiased estimators are impractical. Several techniques were proposed to improve over ST while keeping the same low computational complexity:"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.03549","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-07T15:16:07Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"b8fe6a55a8a9ee60ac31f43de8072f3ec6acca324368524ff10bfff1ad229fab","abstract_canon_sha256":"ead8ef0a47b09a30ed99d58750468a102e38ad93cb97e1694dd5da803edadeaa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:22:58.994811Z","signature_b64":"89YB+CZECVGeAM6CUA8kpFjXj5CJb14vMxU4FGCP9qA3urXwIqAuHYKYdrTtysC5W3WfgnIpRnF2V0T2DZ14Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c13f53260d3981608ed0791eaccc06c1ccbb88756c5300a76b70cb536a6b20e","last_reissued_at":"2026-07-05T03:22:58.994371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:22:58.994371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bias-Variance Tradeoffs in Single-Sample Binary Gradient Estimators","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Alexander Shekhovtsov","submitted_at":"2021-10-07T15:16:07Z","abstract_excerpt":"Discrete and especially binary random variables occur in many machine learning models, notably in variational autoencoders with binary latent states and in stochastic binary networks. When learning such models, a key tool is an estimator of the gradient of the expected loss with respect to the probabilities of binary variables. The straight-through (ST) estimator gained popularity due to its simplicity and efficiency, in particular in deep networks where unbiased estimators are impractical. Several techniques were proposed to improve over ST while keeping the same low computational complexity:"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.03549","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.03549/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.03549","created_at":"2026-07-05T03:22:58.994432+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.03549v2","created_at":"2026-07-05T03:22:58.994432+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.03549","created_at":"2026-07-05T03:22:58.994432+00:00"},{"alias_kind":"pith_short_12","alias_value":"LQJ7KMTA2OMB","created_at":"2026-07-05T03:22:58.994432+00:00"},{"alias_kind":"pith_short_16","alias_value":"LQJ7KMTA2OMBMCHN","created_at":"2026-07-05T03:22:58.994432+00:00"},{"alias_kind":"pith_short_8","alias_value":"LQJ7KMTA","created_at":"2026-07-05T03:22:58.994432+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.17962","citing_title":"A Principled Bayesian Framework for Training Binary and Spiking Neural Networks","ref_index":2021,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ","json":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ.json","graph_json":"https://pith.science/api/pith-number/LQJ7KMTA2OMBMCHNA6I6VTGANQ/graph.json","events_json":"https://pith.science/api/pith-number/LQJ7KMTA2OMBMCHNA6I6VTGANQ/events.json","paper":"https://pith.science/paper/LQJ7KMTA"},"agent_actions":{"view_html":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ","download_json":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ.json","view_paper":"https://pith.science/paper/LQJ7KMTA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.03549&json=true","fetch_graph":"https://pith.science/api/pith-number/LQJ7KMTA2OMBMCHNA6I6VTGANQ/graph.json","fetch_events":"https://pith.science/api/pith-number/LQJ7KMTA2OMBMCHNA6I6VTGANQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ/action/storage_attestation","attest_author":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ/action/author_attestation","sign_citation":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ/action/citation_signature","submit_replication":"https://pith.science/pith/LQJ7KMTA2OMBMCHNA6I6VTGANQ/action/replication_record"}},"created_at":"2026-07-05T03:22:58.994432+00:00","updated_at":"2026-07-05T03:22:58.994432+00:00"}