{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LIJM6NQO3E7AP3CMCPQNIP2VK2","short_pith_number":"pith:LIJM6NQO","schema_version":"1.0","canonical_sha256":"5a12cf360ed93e07ec4c13e0d43f55568655c07fab0770e3a71944aebca7debd","source":{"kind":"arxiv","id":"2503.09202","version":1},"attestation_state":"computed","paper":{"title":"Token Weighting for Long-Range Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Falko Helm, Iryna Gurevych, Nico Daheim","submitted_at":"2025-03-12T09:46:59Z","abstract_excerpt":"Many applications of large language models (LLMs) require long-context understanding, but models continue to struggle with such tasks. We hypothesize that conventional next-token prediction training could contribute to this, because each token is assigned equal weight. Yet, intuitively, the amount of context needed to predict the next token accurately varies greatly across different data. To reflect this, we propose various novel token-weighting schemes that assign different weights to each training token in the loss, thereby generalizing existing works. For this, we categorize token-weighting"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09202","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-12T09:46:59Z","cross_cats_sorted":[],"title_canon_sha256":"41decc99981a8c43e53a04dd4cead0ba6a08a5085227ed04a00a8afeb8f922d4","abstract_canon_sha256":"17c47bc9f36e42d6a4f2d2e06bf462276e3e23cd1056c6b4088d4c32da5b1137"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:29:51.378898Z","signature_b64":"ZmQO3o3hRTLQeeK73POdFWfDCuOApPW2B5OLLMpXubbbHz08csgOv/SW2xWBmJsEIgArFeg3dxWNV/X+C+ZwBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a12cf360ed93e07ec4c13e0d43f55568655c07fab0770e3a71944aebca7debd","last_reissued_at":"2026-07-05T10:29:51.378059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:29:51.378059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token Weighting for Long-Range Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Falko Helm, Iryna Gurevych, Nico Daheim","submitted_at":"2025-03-12T09:46:59Z","abstract_excerpt":"Many applications of large language models (LLMs) require long-context understanding, but models continue to struggle with such tasks. We hypothesize that conventional next-token prediction training could contribute to this, because each token is assigned equal weight. Yet, intuitively, the amount of context needed to predict the next token accurately varies greatly across different data. To reflect this, we propose various novel token-weighting schemes that assign different weights to each training token in the loss, thereby generalizing existing works. For this, we categorize token-weighting"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09202","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09202","created_at":"2026-07-05T10:29:51.378161+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09202v1","created_at":"2026-07-05T10:29:51.378161+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09202","created_at":"2026-07-05T10:29:51.378161+00:00"},{"alias_kind":"pith_short_12","alias_value":"LIJM6NQO3E7A","created_at":"2026-07-05T10:29:51.378161+00:00"},{"alias_kind":"pith_short_16","alias_value":"LIJM6NQO3E7AP3CM","created_at":"2026-07-05T10:29:51.378161+00:00"},{"alias_kind":"pith_short_8","alias_value":"LIJM6NQO","created_at":"2026-07-05T10:29:51.378161+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02340","citing_title":"Not All Denoising Steps Are Equal: Model Scheduling for Faster Masked Diffusion Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10544","citing_title":"Where Does Long-Context Supervision Actually Go? Effective-Context Exposure Balancing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08290","citing_title":"Tokalator: A Context Engineering Toolkit for Artificial Intelligence Coding Assistants","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2","json":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2.json","graph_json":"https://pith.science/api/pith-number/LIJM6NQO3E7AP3CMCPQNIP2VK2/graph.json","events_json":"https://pith.science/api/pith-number/LIJM6NQO3E7AP3CMCPQNIP2VK2/events.json","paper":"https://pith.science/paper/LIJM6NQO"},"agent_actions":{"view_html":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2","download_json":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2.json","view_paper":"https://pith.science/paper/LIJM6NQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09202&json=true","fetch_graph":"https://pith.science/api/pith-number/LIJM6NQO3E7AP3CMCPQNIP2VK2/graph.json","fetch_events":"https://pith.science/api/pith-number/LIJM6NQO3E7AP3CMCPQNIP2VK2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2/action/storage_attestation","attest_author":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2/action/author_attestation","sign_citation":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2/action/citation_signature","submit_replication":"https://pith.science/pith/LIJM6NQO3E7AP3CMCPQNIP2VK2/action/replication_record"}},"created_at":"2026-07-05T10:29:51.378161+00:00","updated_at":"2026-07-05T10:29:51.378161+00:00"}