{"as_of":"2026-07-22T02:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:463af48ca2bda2c27cc4f94cfb93808057292e810242cc1189534e8906805439","coverage":[{"denominator":46,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":111,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":111,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-21T06:31:05.380196+00:00","state":"measured"},{"denominator":65,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T19:58:05.484186Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"1906.08237","last_updated":"2020-01-02T12:48:08Z","snapshot_observed_at":"2026-07-06T08:01:30.099981Z","submitted_at":"2019-06-19T17:35:48Z","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T01:29:27.427361Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/1906.08237"},"observation_digest":"sha256:a460d937f7566843d3dd68c5c1aaf8b87254f6ddd896a8974acffb23be33010e","observation_id":"348f1962-2f41-446e-92d7-0ff9483a756e","resolution":{"observed_at":"2026-05-18T01:29:27.505121Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T18:34:44.807534Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/1909.08053"},"observation_digest":"sha256:e7a34d846f973ea778977c50634bfad127117445fd2ae1de36de51729330d7c1","observation_id":"e1a008e8-a390-4f8b-a586-78eec5e7c04f","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"1910.03771","last_updated":"2020-07-14T03:42:34Z","snapshot_observed_at":"2026-07-06T08:27:58.343233Z","submitted_at":"2019-10-09T03:23:22Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","version":5},"reference_index":165,"source":"arxiv_source","source_observed_at":"2026-05-11T14:53:58.963468Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/1910.03771"},"observation_digest":"sha256:22e8a23735d39f8eb7d0f5f2b73913677ef050fb5810d4b8709fcbe71e388fe8","observation_id":"79d32172-39ef-41d7-9338-83b4420f3841","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"1910.10683","last_updated":"2023-09-19T15:14:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-10-23T17:37:36Z","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-12T05:37:55.083206Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/1910.10683"},"observation_digest":"sha256:e381fbbb42c95211b7395e0ac0c9fbd6f90b01710ed1aba088f070a86e188d07","observation_id":"fe772a33-0bff-4a56-ac37-57e2a2200c90","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"1910.13461","last_updated":"2019-10-29T18:01:00Z","snapshot_observed_at":"2026-07-06T08:33:12.534026Z","submitted_at":"2019-10-29T18:01:00Z","title":"BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T00:14:58.134513Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/1910.13461"},"observation_digest":"sha256:2ff23875bd3723d01e543b3270f2252666db1bef30ea9ee648e2995ee29d8b97","observation_id":"43abd102-fdbc-4303-9c9a-baf1645b609c","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2002.08910","last_updated":"2020-10-05T21:26:45Z","snapshot_observed_at":"2026-07-06T08:58:52.372748Z","submitted_at":"2020-02-10T18:55:58Z","title":"How Much Knowledge Can You Pack Into the Parameters of a Language Model?","version":4},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-15T02:00:28.055865Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2002.08910"},"observation_digest":"sha256:f2661322ed1be0d35603e114d14162cb366e2b15efdc25e72ddf3b25021326b1","observation_id":"c313e21d-6253-4baa-b47a-6e562b1ad14b","resolution":{"observed_at":"2026-05-15T02:00:28.126284Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2003.10555","last_updated":"2020-03-23T21:17:42Z","snapshot_observed_at":"2026-07-06T09:06:50.593178Z","submitted_at":"2020-03-23T21:17:42Z","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T10:26:47.593122Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2003.10555"},"observation_digest":"sha256:b6328e8bc6963d0fd8151a9921a1f9488d90bb7689da724764d83b5001fd6fa5","observation_id":"cd6df6d1-5564-4a5d-8212-5c165be9f5bb","resolution":{"observed_at":"2026-05-16T10:26:47.627568Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T12:05:38.045330Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2005.14165"},"observation_digest":"sha256:c4d5b7a9facb6223642c7f218e0882462e4a02d322131aa666bf10de5eb91003","observation_id":"0a33c019-557e-4e52-8c37-63ec5e5dae45","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2008.02275","last_updated":"2023-02-17T16:08:22Z","snapshot_observed_at":"2026-07-06T09:44:51.724718Z","submitted_at":"2020-08-05T17:59:16Z","title":"Aligning AI With Shared Human Values","version":6},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T14:41:26.484482Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2008.02275"},"observation_digest":"sha256:4b599559c52630ef3789d3f176285d3952c4b89cc62fcf3b5a6e96708f1ae269","observation_id":"7461f0ec-f048-4eda-aa73-2b8edd0f55ff","resolution":{"observed_at":"2026-05-21T14:41:26.574987Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-10T12:43:44.359247Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2009.03300"},"observation_digest":"sha256:01c8a5be481b4c118df03a5694ba632c647f8c4eeb49191309cface88ae30b77","observation_id":"2c4f527f-93f9-4577-a0d3-c7fabf69b817","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2102.01293","last_updated":"2021-02-02T04:07:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-02-02T04:07:38Z","title":"Scaling Laws for Transfer","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-18T00:58:13.116663Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2102.01293"},"observation_digest":"sha256:58d3c79f0315f9ecb0464fc2978f897662cb7b52875838596166ef26b393198c","observation_id":"d57ef7d3-4d6f-46e1-b97e-02d77a6bbc21","resolution":{"observed_at":"2026-05-18T00:58:13.665122Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2112.00861","last_updated":"2021-12-09T21:40:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-01T22:24:34Z","title":"A General Language Assistant as a Laboratory for Alignment","version":3},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-11T14:22:57.925354Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2112.00861"},"observation_digest":"sha256:7667c8d88eb430b1d0ac413a327dc7a76e5debffb52fc044651eef92874c7762","observation_id":"bb695082-9b61-4fc2-9d91-e1cd3d5fd78d","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2112.09118","last_updated":"2022-08-29T12:17:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-16T18:57:37Z","title":"Unsupervised Dense Information Retrieval with Contrastive Learning","version":4},"reference_index":150,"source":"arxiv_source","source_observed_at":"2026-05-12T13:21:16.921001Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2112.09118"},"observation_digest":"sha256:539d2583257f52e95a140a886a5419730817870a7a64996bc032ee7d23378269","observation_id":"cdb8a1f6-b239-49d1-a065-c86102c7db02","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"reference_index":138,"source":"arxiv_source","source_observed_at":"2026-05-10T15:42:47.274448Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2207.05221"},"observation_digest":"sha256:d8c39b95ee26b5ae33b39cd919a628b575d81483ad0b0d4bc1f9b7ecef2a3be1","observation_id":"bc762b44-7106-4a0c-b898-4de815c5834e","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2208.03299","last_updated":"2022-11-16T16:38:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-08-05T17:39:22Z","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","version":3},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-05-16T13:48:43.024120Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2208.03299"},"observation_digest":"sha256:bff19b7f67130f4a6ab820ce5dcd228d318bb4701d8494d905619e02a21ca253","observation_id":"28d8cefa-2b5e-4b2c-b19d-163c7e3f53a1","resolution":{"observed_at":"2026-05-16T13:48:43.320356Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2309.05922","last_updated":"2023-09-12T02:34:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-12T02:34:06Z","title":"A Survey of Hallucination in Large Foundation Models","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-16T15:21:00.778049Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2309.05922"},"observation_digest":"sha256:8018ffc6c8c70d866d968d4105314536944be54b33656622ea8326c2c652b70c","observation_id":"390cdd92-1934-404c-9b32-df2a8f459c85","resolution":{"observed_at":"2026-05-16T15:21:00.849987Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2311.04799","last_updated":"2026-04-15T19:55:08Z","snapshot_observed_at":"2026-07-06T16:44:47.529213Z","submitted_at":"2023-11-08T16:18:32Z","title":"DA-Cramming: Enhancing Cost-Effective Language Model Pretraining with Dependency Agreement Integration","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-24T05:31:30.519530Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2311.04799"},"observation_digest":"sha256:166113f6fe7c098f589f9749a91dc3bb960e236b9df8ff8b2b69df0c5cae804b","observation_id":"315d09d6-7960-4d66-a59f-36a4830f35f9","resolution":{"observed_at":"2026-05-24T05:33:56.730707Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2402.06196","last_updated":"2025-03-23T14:51:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-09T05:37:09Z","title":"Large Language Models: A Survey","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-11T15:22:54.023279Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2402.06196"},"observation_digest":"sha256:02758f18c1f567713fd426673bc902f52128b8dab125fb1346030b8ef48294f5","observation_id":"5893cdea-30fd-4245-bb92-3e171a9690e5","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2408.11871","last_updated":"2026-04-11T07:42:24Z","snapshot_observed_at":"2026-07-06T19:04:13.760549Z","submitted_at":"2024-08-19T13:27:07Z","title":"MegaFake: A Theory-Driven Dataset of Fake News Generated by Large Language Models","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-23T21:31:27.597981Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2408.11871"},"observation_digest":"sha256:cf29eeeeb7768d75d8a53fab6828a91f162b99f0b4bfb4c87168c4af39d5a3b1","observation_id":"e8a4fcd5-2afa-48b1-91c1-ca995fb0db62","resolution":{"observed_at":"2026-05-23T21:33:28.543628Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2410.15761","last_updated":"2026-06-03T09:17:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-21T08:21:00Z","title":"Optimal Query Allocation in Extractive QA with LLMs: A Learning-to-Defer Framework with Theoretical Guarantees","version":4},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-23T18:49:39.718108Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2410.15761"},"observation_digest":"sha256:d094180c690b140de97dcf7c0d759b3432481bbdbbfd1f850fe01766c4b4b297","observation_id":"3c3345c9-be49-4f1a-ba30-efcf643eb282","resolution":{"observed_at":"2026-05-23T18:53:21.384938Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-07-06T19:57:55.925634Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-19T11:08:27.472508Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2411.18279"},"observation_digest":"sha256:801da0e358b3925ba94ab6ee5a67ffce06c8329c7771663986b26fa989379fcc","observation_id":"b8446253-35cc-41f2-8c82-e32fe812070f","resolution":{"observed_at":"2026-05-19T11:08:27.629084Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-12T15:39:40.845703Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2502.05171"},"observation_digest":"sha256:139efbb9c0c972b181a2143aa7167d5183c802fd724813d76678986a57a9c4fd","observation_id":"3d03bda9-24de-42a2-9084-49fe10225c64","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2503.08223","last_updated":"2026-04-09T15:28:27Z","snapshot_observed_at":"2026-07-06T20:50:33.513259Z","submitted_at":"2025-03-11T09:41:29Z","title":"Will LLMs Scaling Hit the Wall? Breaking Barriers via Distributed Resources on Massive Edge Devices","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-23T01:03:26.037233Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2503.08223"},"observation_digest":"sha256:46a5b499ee2a63c2070d1e237afbec9744709e062c04f532ad03b7de285f06b0","observation_id":"a7afe8d8-bded-4296-af6c-f402c080a3ff","resolution":{"observed_at":"2026-05-23T01:05:16.551221Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2504.13898","last_updated":"2026-05-12T14:42:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T06:27:02Z","title":"Social Human Robot Embodied Conversation (SHREC) Dataset: Benchmarking Foundational Models' Social Reasoning","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T21:14:13.351140Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2504.13898"},"observation_digest":"sha256:96491fd2cd18773e2e9426185ca380872298f7cdd068e53e92065b6a8a89940d","observation_id":"a07483af-13f6-4cab-bc01-0e33da3959e1","resolution":{"observed_at":"2026-05-22T21:15:09.437230Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2508.03949","last_updated":"2026-04-14T17:16:30Z","snapshot_observed_at":"2026-07-06T22:08:27.007876Z","submitted_at":"2025-08-05T22:32:32Z","title":"Model Compression vs. Adversarial Robustness: An Empirical Study on Language Models for Code","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-19T00:01:42.228190Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2508.03949"},"observation_digest":"sha256:36735de6d96ea6808b000a0c77607df83de95468f88eadc2c22e9bfb90ebcd16","observation_id":"5c652df5-fc40-43d2-b339-ed05704ea8ec","resolution":{"observed_at":"2026-05-19T00:01:55.794337Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2509.22055","last_updated":"2026-04-11T08:49:59Z","snapshot_observed_at":"2026-07-06T22:30:50.642382Z","submitted_at":"2025-09-26T08:36:45Z","title":"RedNote-Vibe: A Dataset for Capturing Temporal Dynamics of AI-Generated Text in Lifestyle Social Media","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T13:46:48.348859Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2509.22055"},"observation_digest":"sha256:f2f8be461ed6ffe40a669765bb31411afe970f0496144b60ee0d3e8964b0f1af","observation_id":"dfab75da-e01f-4ddd-9d97-62252bb2b2ac","resolution":{"observed_at":"2026-05-18T13:51:26.114119Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T07:43:11.620446Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:ad4b038b79b9a7cefbf825853fd5f6f2baf0e7386e0bdae0ea214df6f5d47897","observation_id":"ccba7afb-e5bb-45cb-9809-489565fc3efd","resolution":{"observed_at":"2026-05-15T07:43:11.697979Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2603.29057","last_updated":"2026-05-12T10:01:22Z","snapshot_observed_at":"2026-07-06T22:51:13.690351Z","submitted_at":"2026-03-30T22:49:42Z","title":"LA-Sign: Looped Transformers with Geometry-aware Alignment for Skeleton-based Sign Language Recognition","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T21:09:08.571994Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2603.29057"},"observation_digest":"sha256:b93a3262a66f21cdc5b89a1a63f4566ab016d6fe13b065ee8ab8da83938607ff","observation_id":"844771d1-ec0e-4f43-b67e-14bf10a24070","resolution":{"observed_at":"2026-05-14T21:09:28.585075Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.15499","last_updated":"2026-04-16T20:18:12Z","snapshot_observed_at":"2026-07-06T23:03:02.938732Z","submitted_at":"2026-04-16T20:18:12Z","title":"SecureRouter: Encrypted Routing for Efficient Secure Inference","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T10:30:11.745479Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.15499"},"observation_digest":"sha256:f60bc6bacf8a9b8caff11690865fb2266ff370462045f4c39187cb2fa492d4ff","observation_id":"ee03a2f9-c2e3-4b51-900e-b5da11e958c9","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.17313","last_updated":"2026-04-19T08:02:57Z","snapshot_observed_at":"2026-07-06T23:04:28.260463Z","submitted_at":"2026-04-19T08:02:57Z","title":"GuardPhish: Securing Open-Source LLMs from Phishing Abuse","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T06:25:13.699171Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.17313"},"observation_digest":"sha256:b36018d577cd4f5f6dd2dc90cae762e4ca1698d11f9b4e9b861ddc0e6742fea8","observation_id":"5afc2b8f-dd0c-4878-8399-7ddd8c80c522","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.19028","last_updated":"2026-04-21T03:23:34Z","snapshot_observed_at":"2026-07-06T23:05:44.499005Z","submitted_at":"2026-04-21T03:23:34Z","title":"Learning Posterior Predictive Distributions for Node Classification from Synthetic Graph Priors","version":1},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-05-10T03:50:44.626261Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.19028"},"observation_digest":"sha256:1c8ac68b864a6777ad9d99fe984fe6815b8c6f745d818e302341353b18225b14","observation_id":"8496bbea-a89b-42cd-aff0-b8f5ba32e62f","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.21254","last_updated":"2026-07-02T01:19:03Z","snapshot_observed_at":"2026-07-06T23:07:51.979267Z","submitted_at":"2026-04-23T03:46:14Z","title":"Hyperloop Transformers","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T23:09:35.640412Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.21254"},"observation_digest":"sha256:76363bd515c645ab0d79cd16b14643f0fbca4f13b466adab549313b6da0cb08c","observation_id":"451affbb-fa82-42ba-8b79-beb6ab40eebf","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.22906","last_updated":"2026-04-24T16:56:53Z","snapshot_observed_at":"2026-07-06T23:09:14.698490Z","submitted_at":"2026-04-24T16:56:53Z","title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-08T09:45:57.201837Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.22906"},"observation_digest":"sha256:41c1d9dbaf0ce2038d1d41afdc8d658f6356caa9c6aa9f5feadabf783dca5dc7","observation_id":"d53288ac-5709-4b1d-ac6c-a2b6dd50dd96","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2604.24940","last_updated":"2026-04-29T09:02:13Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:29:33Z","title":"ADE: Adaptive Dictionary Embeddings -- Scaling Multi-Anchor Representations to Large Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T03:42:38.627716Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2604.24940"},"observation_digest":"sha256:420d43769678b1cbd2c70a2754f8836309518ce3109168bbb92a414097a8357a","observation_id":"7ab6801d-2cde-45ce-bc44-6131fce81e64","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.02374","last_updated":"2026-05-04T09:16:57Z","snapshot_observed_at":"2026-07-06T23:15:27.816549Z","submitted_at":"2026-05-04T09:16:57Z","title":"Fight Poison with Poison: Enhancing Robustness in Few-shot Machine-Generated Text Detection with Adversarial Training","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-08T17:49:39.514484Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.02374"},"observation_digest":"sha256:300f0d6882527add8c101787a1b166cfd12cd49ba322f6687d9205689b71850f","observation_id":"0106785d-b97d-45b8-bb56-37942e0df2b8","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.03799","last_updated":"2026-05-09T13:39:51Z","snapshot_observed_at":"2026-07-06T23:16:40.201701Z","submitted_at":"2026-05-05T14:25:48Z","title":"Natural Language Processing: A Comprehensive Practical Guide from Tokenisation to RLHF","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-07T16:29:57.984030Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.03799"},"observation_digest":"sha256:d1f2bb438b2b63a6b247a553250e625d821975cdfe7d619d4bacfd9cdec7d076","observation_id":"99179e0f-d6cc-47c2-ba7b-958fd3ffc20b","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.03799","last_updated":"2026-05-09T13:39:51Z","snapshot_observed_at":"2026-07-06T23:16:40.201701Z","submitted_at":"2026-05-05T14:25:48Z","title":"Natural Language Processing: A Comprehensive Practical Guide from Tokenisation to RLHF","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T01:48:40.194847Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.03799"},"observation_digest":"sha256:3e38ae65f350f2d6eb3fcf754c014463166f0c98c81e367a1361e3fcda0911d5","observation_id":"8c1c1728-bb8a-4254-a769-882c12488a47","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.05495","last_updated":"2026-05-06T22:31:59Z","snapshot_observed_at":"2026-07-06T23:18:07.750358Z","submitted_at":"2026-05-06T22:31:59Z","title":"Shortcut Solutions Learned by Transformers Impair Continual Compositional Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T16:42:08.322419Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.05495"},"observation_digest":"sha256:1577c78dfac18f2003cf04b80c31c2fb1a58148e7e746579fd01d9b54b28b069","observation_id":"ce6cfcde-8d78-4fbc-a12d-d6465e1bde54","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.06322","last_updated":"2026-05-27T19:46:02Z","snapshot_observed_at":"2026-07-06T23:18:46.145104Z","submitted_at":"2026-05-07T14:21:26Z","title":"SMolLM: Small Language Models Learn Small Molecular Grammar","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-08T12:57:47.721361Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.06322"},"observation_digest":"sha256:1a997408a3406c1d5d1c2764a765dc46a839450ffe9bc5e443868267a374ce28","observation_id":"f4932a72-0da3-4049-8871-40cadcdfa74c","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.06322","last_updated":"2026-05-27T19:46:02Z","snapshot_observed_at":"2026-07-06T23:18:46.145104Z","submitted_at":"2026-05-07T14:21:26Z","title":"SMolLM: Small Language Models Learn Small Molecular Grammar","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.06322"},"observation_digest":"sha256:deae34ec4d469569495b9a83c5045b7b392ee9835ceaafd73f776f41f730985e","observation_id":"7e58169c-88f2-4eb1-a11f-a275a0a0b2a3","resolution":{"observed_at":"2026-06-30T23:35:06.787310Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.12139","last_updated":"2026-05-12T13:58:20Z","snapshot_observed_at":"2026-07-06T23:23:54.142937Z","submitted_at":"2026-05-12T13:58:20Z","title":"BoolXLLM: LLM-Assisted Explainability for Boolean Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-13T05:54:32.665189Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.12139"},"observation_digest":"sha256:2acd7f824e759be428b5161a73f9aec5c8728e090c7ec7e31bd18e069e2bd81d","observation_id":"3fa985c6-8cbb-4770-a3ef-af27da6862e3","resolution":{"observed_at":"2026-05-13T12:26:58.164513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.16234","last_updated":"2026-05-18T08:12:08Z","snapshot_observed_at":"2026-07-06T23:27:24.957966Z","submitted_at":"2026-05-15T17:43:16Z","title":"No Free Swap: Protocol-Dependent Layer Redundancy in Transformers","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T20:29:38.004776Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.16234"},"observation_digest":"sha256:908ac88bb60493bb715e6844775a4eab44204fd96ee8098087083729efce2ab7","observation_id":"e18ca552-f838-4900-ac99-20a0031063e4","resolution":{"observed_at":"2026-05-20T20:33:43.483704Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.16343","last_updated":"2026-05-08T01:45:28Z","snapshot_observed_at":"2026-07-06T23:27:29.931962Z","submitted_at":"2026-05-08T01:45:28Z","title":"LoopQ: Quantization for Recursive Transformers","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T22:41:55.787556Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.16343"},"observation_digest":"sha256:ce56c5f67871e95af569d97c45cdc5fb53e6450f2b8b5a89768e5b45f923065f","observation_id":"66188369-62e9-405e-91ff-9a14b5d6b1a9","resolution":{"observed_at":"2026-05-20T22:43:50.882056Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.18007","last_updated":"2026-05-18T08:03:02Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:03:02Z","title":"Semantic Reranking at Inference Time for Hard Examples in Rhetorical Role Labeling","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-20T11:27:30.720693Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.18007"},"observation_digest":"sha256:a3f24ca833e4e7e6f75efb0bc8c5ede0f37a83a04de0a43f35ac18f18748dc35","observation_id":"5da13bfc-7216-4d17-b787-a671a00a101a","resolution":{"observed_at":"2026-05-20T11:28:14.284668Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.19376","last_updated":"2026-05-20T08:03:28Z","snapshot_observed_at":"2026-07-06T23:30:06.876413Z","submitted_at":"2026-05-19T05:20:56Z","title":"Generative Recursive Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-20T05:58:23.543870Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.19376"},"observation_digest":"sha256:9522dd200237ac83dbf3ee093724476adefafd3f32ea7e90a9a54147778a8640","observation_id":"7ad9d184-757b-47b8-a03e-ed673aeb08f3","resolution":{"observed_at":"2026-05-20T06:03:05.419635Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.19376","last_updated":"2026-05-20T08:03:28Z","snapshot_observed_at":"2026-07-06T23:30:06.876413Z","submitted_at":"2026-05-19T05:20:56Z","title":"Generative Recursive Reasoning","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-21T07:40:30.199868Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.19376"},"observation_digest":"sha256:9b6db270e82fe56472a9231dfffe3da4af5a932c23cba8a5f1d3211a39e66deb","observation_id":"401497c5-9092-4267-bd21-a0bfe723bfbe","resolution":{"observed_at":"2026-05-21T07:44:03.134899Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.24020","last_updated":"2026-05-20T06:11:25Z","snapshot_observed_at":"2026-07-06T23:34:08.786038Z","submitted_at":"2026-05-20T06:11:25Z","title":"Machine Intelligence that Understands Visual and Linguistic Information and Interacts with Humans and Environments","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-06-30T17:40:33.082748Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.24020"},"observation_digest":"sha256:fe77c658b4e43560856f056d8634509767f15da93af5997ede3742fe6d4bde7f","observation_id":"a2420ab7-ffca-4c6e-96bd-7b68f0cb0878","resolution":{"observed_at":"2026-06-30T17:44:57.706468Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.26106","last_updated":"2026-05-25T17:58:24Z","snapshot_observed_at":"2026-07-06T23:36:01.564747Z","submitted_at":"2026-05-25T17:58:24Z","title":"Looped Diffusion Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T23:13:12.343355Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.26106"},"observation_digest":"sha256:0b4caeceb3e6422af5999a21927af64ff139cd89f97183f357dc055fd4f8ef91","observation_id":"4af2994b-5caf-485c-9281-d60170bf5b9e","resolution":{"observed_at":"2026-06-29T23:14:01.148365Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.29705","last_updated":"2026-05-28T10:04:02Z","snapshot_observed_at":"2026-07-06T23:39:05.858969Z","submitted_at":"2026-05-28T10:04:02Z","title":"BitTP: The Lightweight Trajectory Prediction Model with BitLLM for Edge-Devices","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T07:56:00.180707Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.29705"},"observation_digest":"sha256:52ce088bde6897702d73f2fc7fe04f42cb702ce72b73f5aa6c8c0b4aa58617d6","observation_id":"481ef6f9-aca8-4774-8940-7cbdfe593769","resolution":{"observed_at":"2026-06-29T08:03:14.565027Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.30022","last_updated":"2026-05-28T14:42:25Z","snapshot_observed_at":"2026-07-06T23:39:24.682426Z","submitted_at":"2026-05-28T14:42:25Z","title":"Give it Space! Explicit Disentangling of Positional and Semantic Representations in Encoders","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-29T07:24:41.424801Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.30022"},"observation_digest":"sha256:8f5480a1f13903468968c229a883fa5203ada704977406785bb20042aa038b31","observation_id":"70793cb5-4e56-420d-b141-3b8bca12ad98","resolution":{"observed_at":"2026-06-29T07:33:13.053708Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2605.31003","last_updated":"2026-05-29T08:36:08Z","snapshot_observed_at":"2026-07-06T23:40:12.939143Z","submitted_at":"2026-05-29T08:36:08Z","title":"Graph-GRPO: Dependency-Aware Credit Assignment for Generative E-commerce Search Relevance","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T21:21:46.537474Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2605.31003"},"observation_digest":"sha256:117909bb00dddbee6d92559fad3d7ba05231524c1539f766be76aa98e846964f","observation_id":"3bf1aa60-f177-4932-b8bb-c0c1c4670806","resolution":{"observed_at":"2026-07-01T20:16:12.105693Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.01671","last_updated":"2026-06-01T04:28:44Z","snapshot_observed_at":"2026-07-06T23:42:12.125247Z","submitted_at":"2026-06-01T04:28:44Z","title":"When Meaning Travels: A Granular Lens on Hybrid-MoE's Role in Idiomatic Understanding for Language Models","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-06-28T15:19:26.983760Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.01671"},"observation_digest":"sha256:1c660c19bccf92179f0ceb178a5c77a3844ca7cd50611819e50507a28ca23add","observation_id":"1214c1bb-7c94-4ecb-bd33-f50e8f615428","resolution":{"observed_at":"2026-07-01T22:36:16.753493Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.02100","last_updated":"2026-06-01T11:32:02Z","snapshot_observed_at":"2026-07-06T23:42:35.063977Z","submitted_at":"2026-06-01T11:32:02Z","title":"PortBERT: Navigating the Depths of Portuguese Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-28T14:43:28.401687Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.02100"},"observation_digest":"sha256:89a788e8f1c26414530f95ebf04c38a55182a3857ea2a747022438e71f5d3a4b","observation_id":"5c45aff0-e75c-4d8e-9ea9-df27574e4e89","resolution":{"observed_at":"2026-07-01T23:06:20.163538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.04032","last_updated":"2026-06-04T17:08:43Z","snapshot_observed_at":"2026-07-06T23:44:14.261541Z","submitted_at":"2026-06-01T20:59:05Z","title":"Do Transformers Need Three Projections? Systematic Study of QKV Variants","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-06-28T15:14:49.475561Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.04032"},"observation_digest":"sha256:12736772aa3b48d914f3d2ce67b57dda820b21e6b08fb38c0e74fc3cbb95a7a2","observation_id":"5886366b-87f8-43b0-8a1e-da79875e9e5a","resolution":{"observed_at":"2026-07-01T22:36:17.309433Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.04438","last_updated":"2026-06-03T04:38:12Z","snapshot_observed_at":"2026-07-06T23:44:33.095008Z","submitted_at":"2026-06-03T04:38:12Z","title":"LoopMoE: Unifying Iterative Computation with Mixture-of-Experts for Language Modeling","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-28T07:19:31.075298Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.04438"},"observation_digest":"sha256:fc244bb1e692bd4e9ef643bba81eb4a3a2b43cfcf54a90b480cd4e5400938e7c","observation_id":"60040f26-7b32-4074-8093-cf2814c21b19","resolution":{"observed_at":"2026-07-02T06:36:44.154453Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.04678","last_updated":"2026-06-03T10:01:45Z","snapshot_observed_at":"2026-07-06T23:44:47.848374Z","submitted_at":"2026-06-03T10:01:45Z","title":"Test-Time Compute Scaling for ASR with Depth-Conditioned Looped Transformers","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T06:56:57.423463Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.04678"},"observation_digest":"sha256:23711575a9e8f29e6838c5bbe980a4239814a7d29f1880483dc9199034572ce2","observation_id":"f42feea1-f9b2-4af7-ae3e-5369cb46b773","resolution":{"observed_at":"2026-07-02T07:26:45.914456Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.08728","last_updated":"2026-07-21T00:41:45Z","snapshot_observed_at":"2026-07-22T01:22:43.314970Z","submitted_at":"2026-06-07T16:50:07Z","title":"Artificial Intelligence for Mathematical Reasoning: An Integrated Survey of Language Models, Neuro-symbolic Systems, and Verified Discovery","version":1},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-06-27T18:39:44.696961Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.08728"},"observation_digest":"sha256:b57306d193752def327c11efb1bae3f8dbf26be340f2ccd69c6a0518856c0b45","observation_id":"1df5ee24-76cd-4931-ba51-d3e614a54480","resolution":{"observed_at":"2026-07-02T22:47:25.990002Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.09357","last_updated":"2026-06-08T11:34:02Z","snapshot_observed_at":"2026-07-06T23:48:44.159537Z","submitted_at":"2026-06-08T11:34:02Z","title":"Rethinking Depth: A study of the Recursive-Transformer for Speech Recognition","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T15:01:02.462134Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.09357"},"observation_digest":"sha256:0488f10cb2282013c1de2f4101f4e3b479a122506a3ee85e976536811dab0a21","observation_id":"f42f2463-9612-4e46-a190-5844c87ce574","resolution":{"observed_at":"2026-07-03T03:37:35.900827Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.12876","last_updated":"2026-06-11T04:06:02Z","snapshot_observed_at":"2026-07-06T23:51:45.306189Z","submitted_at":"2026-06-11T04:06:02Z","title":"Multi-Bitwidth Quantization for LLMs Using Additive Codebooks","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-27T07:29:14.923431Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.12876"},"observation_digest":"sha256:5d821c0b377157f9e5a4e5857c8a0067366b041eceb52010193d733903a0d0da","observation_id":"cfcfb36e-9e46-4c86-a2eb-f902d7f0dd76","resolution":{"observed_at":"2026-07-03T13:48:21.343692Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.20737","last_updated":"2026-06-23T11:42:49Z","snapshot_observed_at":"2026-07-06T23:55:53.019607Z","submitted_at":"2026-06-17T15:52:08Z","title":"Repeated Shared Access Enables Grokking, but Edit Propagation Depends on an Addressable Memory","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T20:43:55.740662Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.20737"},"observation_digest":"sha256:f2bb68a77e32cc8e72f3c0bdb1617d5a11eab21dc211ef05cab44c619a85c4a7","observation_id":"70b97758-830b-4ec1-a4fe-04d18e74ed7a","resolution":{"observed_at":"2026-07-04T01:09:18.541074Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.20751","last_updated":"2026-07-13T19:51:21Z","snapshot_observed_at":"2026-07-17T23:18:05.388463Z","submitted_at":"2026-06-18T03:07:47Z","title":"From Sentiment to Actionable Insights: Public Sentiment Analysis of Advanced Air Mobility","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-26T17:52:25.796384Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.20751"},"observation_digest":"sha256:eafc19d8036d476db4aeec1cd17e8c9a21b13f2e67193a4aa69fece34c98581d","observation_id":"a548ef0b-c1c1-4362-9d9d-2e521077712f","resolution":{"observed_at":"2026-07-04T03:39:29.758997Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.27274","last_updated":"2026-06-27T09:56:16Z","snapshot_observed_at":"2026-07-07T00:01:29.987067Z","submitted_at":"2026-06-25T16:51:53Z","title":"BetXplain: An Explanation-Annotated Dataset for Detecting Manipulative Betting Advertisements on Social Media","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-26T05:18:36.311573Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.27274"},"observation_digest":"sha256:dcbdf94332488c3207092d8da18dd3df90eae866e2abcdbd46ce6cf6bef4ce42","observation_id":"a5d311ef-5027-43c0-9362-030b91382fb1","resolution":{"observed_at":"2026-07-04T13:19:50.819128Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.27274","last_updated":"2026-06-27T09:56:16Z","snapshot_observed_at":"2026-07-07T00:01:29.987067Z","submitted_at":"2026-06-25T16:51:53Z","title":"BetXplain: An Explanation-Annotated Dataset for Detecting Manipulative Betting Advertisements on Social Media","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T09:36:12.630813Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.27274"},"observation_digest":"sha256:7b551eebbf88055170cd7ed78941ee48ad064e695602d65067db10eacb05206b","observation_id":"f03bb98d-8d22-4a43-8659-cc26969aa977","resolution":{"observed_at":"2026-06-30T09:44:37.891725Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":"1909.11942","doi":"10.48550/arxiv.1909.11942","metadata_source":"pith","pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","venue":"cs.CL","work_id":"aedf7950-7c35-4e28-a32d-bec290f51669","year":2019},"citing_paper":{"arxiv_id":"2606.30625","last_updated":"2026-06-29T17:55:40Z","snapshot_observed_at":"2026-07-07T00:04:29.879877Z","submitted_at":"2026-06-29T17:55:40Z","title":"Optimization Dynamics Imprint Semantic Specificity in Contrastive Embedding Norms","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T03:27:31.177489Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2606.30625"},"observation_digest":"sha256:5a9357920bf19820451bf923cabd081b431648556c1afc1f360612aa90af0b46","observation_id":"686bfc91-d010-4940-ac07-0445e0724e1c","resolution":{"observed_at":"2026-06-30T03:34:13.354248Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11942","snapshot_observed_at":"2026-07-11T19:58:05.484186Z","title":"Albert: A lite bert for self-supervised learning of language representations,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.04339","last_updated":"2026-07-05T14:45:29Z","snapshot_observed_at":"2026-07-11T19:58:04.775698Z","submitted_at":"2026-07-05T14:45:29Z","title":"One Framework for All: Cross-Modal Membership Inference for Generative Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-11T19:58:05.484186Z"},"links":{"cited_paper":"/paper/1909.11942","citing_paper":"/paper/2607.04339"},"observation_digest":"sha256:c4177d259481371350b9f2d6c024894ff0f45f7cf6fe5c0338b2df04989ddc79","observation_id":"5ac44ec5-8e0d-49b1-b87d-93553b003102","resolution":{"observed_at":"2026-07-11T19:58:05.484186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1909.11942/citation-record","integrity":"/paper/1909.11942/integrity","json":"/paper/1909.11942/citation-record.json","paper":"/paper/1909.11942"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1809.10853","last_updated":"2019-02-22T23:41:46Z","snapshot_observed_at":"2026-07-06T07:04:49.995800Z","submitted_at":"2018-09-28T04:30:11Z","title":"Adaptive Input Representations for Neural Language Modeling","version":3},"cited_work":{"arxiv_id":"1809.10853","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.10853","snapshot_observed_at":"2026-07-03T17:38:43.607419Z","title":"Adaptive Input Representations for Neural Language Modeling","venue":"cs.CL","work_id":"84b64f8d-525e-4a27-b130-855b07cc501f","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1809.10853","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:0df6a64d548b5f503bd8cbe15ca09b19060b2d0d8ba4d116fe9a261cf7265759","observation_id":"2846432e-76eb-47a8-845e-cdb4cf856dfe","resolution":{"observed_at":"2026-05-13T12:26:58.132357Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SemEval-2017 task 1: Semantic textual similarity multilingual and crosslingual focused evaluation","venue":null,"work_id":"707c87fc-2d50-4287-ad65-74b53a75de94","year":2017},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:82521d5faea2b8559ffad52fd4865590f285a536fb694bf6b0ce5cb5e969ba24","observation_id":"80d91f50-ef8c-4e25-bcbc-933ca398fc26","resolution":{"observed_at":"2026-05-13T12:26:58.149141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1604.06174","last_updated":"2016-04-22T19:21:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-04-21T04:15:27Z","title":"Training Deep Nets with Sublinear Memory Cost","version":2},"cited_work":{"arxiv_id":"1604.06174","doi":"10.48550/arxiv.1604.06174","metadata_source":"pith","pith_arxiv_id":"1604.06174","snapshot_observed_at":"2026-07-11T01:57:49.960653Z","title":"Training Deep Nets with Sublinear Memory Cost","venue":"cs.LG","work_id":"f2c5c287-a500-40e4-a136-e7e3172db1d7","year":2016},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1604.06174","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:130bc41234615c82642d4e6ea5503d72794ed09e1d56c6c6ee3796ab045bc125","observation_id":"c5b1f213-39f8-492a-8c48-6f001fc616bf","resolution":{"observed_at":"2026-05-13T12:26:58.058134Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-05-21T12:26:33.666045+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T12:26:33.666045+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.10509","last_updated":"2019-04-23T19:29:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-04-23T19:29:47Z","title":"Generating Long Sequences with Sparse Transformers","version":1},"cited_work":{"arxiv_id":"1904.10509","doi":"10.48550/arxiv.1904.10509","metadata_source":"pith","pith_arxiv_id":"1904.10509","snapshot_observed_at":"2026-07-11T03:27:46.891605Z","title":"Generating Long Sequences with Sparse Transformers","venue":"cs.LG","work_id":"c5b81688-45ee-4a9a-b095-e6290f45cb6c","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1904.10509","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:40d393520dde064cfb1335cba9712e63694ca868195f8071be48957fc84e69df","observation_id":"eb11c3ec-3ea9-49f9-b837-6703c3ce4e70","resolution":{"observed_at":"2026-05-13T12:26:58.087696Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:59.71041+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:59.71041+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.04829","last_updated":"2019-07-10T17:14:47Z","snapshot_observed_at":"2026-07-06T08:06:52.501752Z","submitted_at":"2019-07-10T17:14:47Z","title":"BAM! Born-Again Multi-Task Networks for Natural Language Understanding","version":1},"cited_work":{"arxiv_id":"1907.04829","doi":null,"metadata_source":"pith","pith_arxiv_id":"1907.04829","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BAM! Born-Again Multi-Task Networks for Natural Language Understanding","venue":"cs.CL","work_id":"7a93b3b5-995f-434e-bb41-0fb424f4c911","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1907.04829","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:58199a26f4e833534eb0e01b7f8beac5c567091a89a106e2d331fb456ff85d28","observation_id":"5378f9bd-b2d2-4cf4-964e-3cd06d974947","resolution":{"observed_at":"2026-05-13T12:26:58.091119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.02860","last_updated":"2019-06-02T21:21:48Z","snapshot_observed_at":"2026-07-06T07:25:48.468658Z","submitted_at":"2019-01-09T18:28:19Z","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","version":3},"cited_work":{"arxiv_id":"1901.02860","doi":"10.48550/arxiv.1901.02860","metadata_source":"pith","pith_arxiv_id":"1901.02860","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","venue":"cs.LG","work_id":"eb970d64-41ff-4e44-afd0-e3bb975e0dc4","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1901.02860","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:be991d3cef1d33ef8380a0ee870f36b278dccf5d471c6665216f12970b6b29be","observation_id":"e7accb3d-3e92-4a23-9276-b6e43547d0c1","resolution":{"observed_at":"2026-05-13T12:26:58.094017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03819","last_updated":"2019-03-05T16:46:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-07-10T18:39:15Z","title":"Universal Transformers","version":3},"cited_work":{"arxiv_id":"1807.03819","doi":"10.48550/arxiv.1807.03819","metadata_source":"pith","pith_arxiv_id":"1807.03819","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Universal Transformers","venue":"cs.CL","work_id":"8e5baefe-d209-411c-aefc-5acaa9275c8a","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1807.03819","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:9d20147e2568a9ad10f001fdebe4d4804eabb2f0661c11806e6f8dfe06b4647c","observation_id":"bac4346d-144e-4da9-90d9-aee9049d18c9","resolution":{"observed_at":"2026-05-13T12:26:58.096951Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BERT: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":"a78966f6-f3bf-46e2-bb6c-df564d02d07a","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:12d7b97baefdec8f57030435b32b349e88cacdb95335b3987a8802e1e30092da","observation_id":"60b83144-7e5d-40b4-abbc-258ca1245e1e","resolution":{"observed_at":"2026-05-13T12:26:58.142849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1423","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T01:27:44.847366Z","title":"BERT : Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"Proceedings of the 2019 Conference of the North","work_id":"3e3c8ac8-b858-4b22-af32-393d98c883e0","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:5c14cccd53cdc53990cef02b82b5f7b774c6edabc37daa230c7c1205f797a08d","observation_id":"3bc352f4-bd3e-4617-8466-be4790e7b9a4","resolution":{"observed_at":"2026-05-13T12:26:58.063828Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-17T19:51:02.335584+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T19:51:02.335584+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zhe Gan, Yunchen Pu, Ricardo Henao, Chunyuan Li, Xiaodong He, and Lawrence Carin","venue":null,"work_id":"7ce7e1af-71e8-46b5-8a2e-8ffae2beb4c4","year":2017},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:a69b80c6e9ab99ecdbf57ce29d7fbf22ae0034577018df771cacbd935416a1ce","observation_id":"f400c0ff-6a82-4881-8637-feb88c10273d","resolution":{"observed_at":"2026-05-13T12:26:58.162084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d17-1254","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"doi: 10.18653/v1/D17-1254","venue":null,"work_id":"8f919868-bce7-45c8-8d08-769fa4bf8663","year":2020},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:4d12cb250afc8d23258d5c771fc6d85704b4579afa200b8f278219364ee521e0","observation_id":"5573041e-c1a7-4aff-9d22-eb2de78ac287","resolution":{"observed_at":"2026-05-13T12:26:58.075185Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"25ec2bda-fe63-443a-b304-b8197c2ee7d0","year":2003},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:e646ab0a8a4a1ea6cf068d14d70d156bf8f20d5397af037e0fa41dad15ff1414","observation_id":"0b20a585-a65d-489c-8828-36a4078a5906","resolution":{"observed_at":"2026-05-13T12:26:58.159306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Modeling recurrence for transformer","venue":null,"work_id":"9296604d-8243-4019-a8e4-eb7b49be2de3","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:336c4f7cb849d59875e43fca1b1e3c14a0bf252edf686a75875067d4535dadae","observation_id":"d18e1156-dd6e-4045-af02-1c501a9a9ccf","resolution":{"observed_at":"2026-05-13T12:26:58.153057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.08415","last_updated":"2023-06-06T01:53:32Z","snapshot_observed_at":"2026-07-06T05:01:27.910364Z","submitted_at":"2016-06-27T19:20:40Z","title":"Gaussian Error Linear Units (GELUs)","version":5},"cited_work":{"arxiv_id":"1606.08415","doi":"10.18653/v1/n19-1122","metadata_source":"pith","pith_arxiv_id":"1606.08415","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Gaussian Error Linear Units (GELUs)","venue":"cs.LG","work_id":"0466fd22-03a1-4a61-af0a-a900e77bb023","year":2016},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1606.08415","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:07f3035c036b10e87414a327dcbdb1aa857413c08665007fb1525ab1818ec6da","observation_id":"b6b8f450-d7fb-406d-a3f8-3e5955e89b8e","resolution":{"observed_at":"2026-05-13T12:26:58.082116Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning distributed representations of sentences from unlabelled data","venue":null,"work_id":"7b5d0613-bd6b-4550-a716-9ab9d52b557d","year":2016},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:31fe741071540d5c9b5d25411a8e690d5202fae1f835fc17529a3e930f517415","observation_id":"46c28f21-48d0-46c5-bd8f-dc0203fafabe","resolution":{"observed_at":"2026-05-13T12:26:58.151195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n16-1162","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning Distributed Representations of Sentences from Unlabelled Data","venue":null,"work_id":"26af5314-13d7-47c3-8101-381216f64101","year":2016},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:7b7cd1d52b411b8d611426a9345f492fd0659b4a64af4e17028839300ad813ff","observation_id":"5cc08e6c-5f7e-42b8-8ffc-fcc1e5deb2b2","resolution":{"observed_at":"2026-05-13T12:26:58.044456Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1801.06146","last_updated":"2018-05-23T09:23:47Z","snapshot_observed_at":"2026-07-06T06:19:16.799773Z","submitted_at":"2018-01-18T17:54:52Z","title":"Universal Language Model Fine-tuning for Text Classification","version":5},"cited_work":{"arxiv_id":"1801.06146","doi":null,"metadata_source":"pith","pith_arxiv_id":"1801.06146","snapshot_observed_at":"2026-07-04T07:39:38.353471Z","title":"Universal Language Model Fine-tuning for Text Classification","venue":"cs.CL","work_id":"9990d84d-84b4-4fd4-acb5-da451524d2f4","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1801.06146","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:28016c00c3de65acce732cc56564ca4129968485158db7597ff3ca824bdda2ea","observation_id":"d25bf620-c39b-4c59-a704-e0ddf37ed746","resolution":{"observed_at":"2026-05-13T12:26:58.100035Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.00557","last_updated":"2017-04-23T09:15:35Z","snapshot_observed_at":"2026-07-06T05:40:03.091136Z","submitted_at":"2017-04-23T09:15:35Z","title":"Discourse-Based Objectives for Fast Unsupervised Sentence Representation Learning","version":1},"cited_work":{"arxiv_id":"1705.00557","doi":null,"metadata_source":"pith","pith_arxiv_id":"1705.00557","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Discourse-Based Objectives for Fast Unsupervised Sentence Representation Learning","venue":"cs.CL","work_id":"8f57ca00-9515-4b6c-a5db-754089f9e265","year":2017},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1705.00557","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:093ed7595df1eaf3d922b8349b2c7585320cab61ee8b5032766e9adc4c319ba7","observation_id":"e12a2347-d35a-4c2f-bff3-a11d7c165bdc","resolution":{"observed_at":"2026-05-13T12:26:58.103117Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.10529","last_updated":"2020-01-18T03:53:04Z","snapshot_observed_at":"2026-07-06T08:09:54.115258Z","submitted_at":"2019-07-24T15:43:40Z","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","version":3},"cited_work":{"arxiv_id":"1907.10529","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1907.10529","snapshot_observed_at":"2026-07-01T23:06:20.155007Z","title":"SpanBERT: Improving pre-training by representing and predicting spans","venue":null,"work_id":"1d90c45c-05d0-44dc-909b-2b6e2a406c24","year":1907},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1907.10529","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:da736d662743acede4c1f6fe7056cb22f04f958fba1aac5148a1870f0bd3ec14","observation_id":"7b8c01b3-3773-4ba2-a8b8-bda919905e0f","resolution":{"observed_at":"2026-05-13T12:26:58.106271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9442.29696","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"URL http://dl.acm.org/citation.cfm?id= 2969442.2969607","venue":null,"work_id":"3e6f6678-6180-4d1f-94df-cce62e154f17","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:1c63364e7775385366fd38e11bc0e897848012e895e1647e1c2f2695a7d10a16","observation_id":"a1d1eb7d-d3af-4278-84c0-2cf9a08503c1","resolution":{"observed_at":"2026-05-13T12:26:58.109164Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.06226","last_updated":"2018-08-19T16:49:06Z","snapshot_observed_at":"2026-07-06T06:56:20.935388Z","submitted_at":"2018-08-19T16:49:06Z","title":"SentencePiece: A simple and language independent subword tokenizer and detokenizer for Neural Text Processing","version":1},"cited_work":{"arxiv_id":"1808.06226","doi":"10.18653/v1/d18-2012","metadata_source":"doi_reference","pith_arxiv_id":"1808.06226","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Sentencepiece: A simple and language independent subword tokenizer and detokenizer for neural text processing","venue":"cs.CL","work_id":"81a6320b-c2e1-4d74-a03e-9e1ff6bbed8d","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1808.06226","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:b89c8d73d704b784c4d610bc1cbdf9e1b87d2c5a0a183de1cfa8fa8b3763dc76","observation_id":"1f01520c-34d6-4b2b-9e63-9ec218db7afd","resolution":{"observed_at":"2026-05-13T12:26:58.066166Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-11T12:19:12.876508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T12:19:12.876508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d17-1082","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:03.412321Z","title":"doi: 10.18653/v1/D17-1082","venue":null,"work_id":"4e65f57b-0562-4172-b6d3-b4ea2e5e62fb","year":2017},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:c7abb877326e3ea331b27505f746bf9cb4067cce1bc214bdf9ba0f5c09672e4e","observation_id":"30b331f6-48f2-43ee-9139-770b26d0140b","resolution":{"observed_at":"2026-05-13T12:26:58.068913Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:40.495585+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:40.495585+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.11692","last_updated":"2019-07-26T17:48:29Z","snapshot_observed_at":"2026-07-06T08:10:31.480621Z","submitted_at":"2019-07-26T17:48:29Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","version":1},"cited_work":{"arxiv_id":"1907.11692","doi":"10.1007/s10489-02203627-9","metadata_source":"pith","pith_arxiv_id":"1907.11692","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","venue":"cs.CL","work_id":"41fe12c4-e538-4890-a244-480650ed3078","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1907.11692","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:7549180d2ea7076037aa0c6c37c0ad3cae6dae478f285221f22dad983f0281ab","observation_id":"bce180f1-1614-4035-8dc6-6d1a86c910b6","resolution":{"observed_at":"2026-05-13T12:26:58.111796Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/p19-1442","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"doi: 10.18653/v1/P19-1442","venue":null,"work_id":"9accf31e-8854-4edf-b20f-5886a1abb332","year":2014},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:edba0e6bab0593b8b94434004583fa3bdd1ed1f2fb02dc94241813d6e1524f4b","observation_id":"c0594520-95f7-4f37-9d9b-6be9fd12deda","resolution":{"observed_at":"2026-05-13T12:26:58.073386Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3115/v1/d14-1162","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:57:06.848728Z","title":"Pennington, R","venue":null,"work_id":"21113f4d-d545-4d27-bde3-02650634d4fe","year":2014},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:7cbe6dae98343b7613ab8319c00a3475637e56d33f16bdd7c8377b392e808d05","observation_id":"9258dcf0-5460-4293-b0fe-15e6f5d1a2a5","resolution":{"observed_at":"2026-05-13T12:26:58.040563Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-12T01:21:19.147622+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T01:21:19.147622+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n18-1202","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T14:51:10.065907Z","title":"and Neumann, Mark and Iyyer, Mohit and Gardner, Matt and Clark, Christopher and Lee, Kenton and Zettlemoyer, Luke","venue":"Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers)","work_id":"a17f7dc7-836c-4e5e-bd49-fe40b25680aa","year":2026},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:e215805c07d53d495b653f643fb86bb160f6e36735e435e8c4e4c6c5ca0c4beb","observation_id":"210b3f8f-f27b-417a-8c5d-bc4b38bda0c0","resolution":{"observed_at":"2026-05-13T12:26:58.079492Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-12T14:49:33.219884+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T14:49:33.219884+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.10683","last_updated":"2023-09-19T15:14:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-10-23T17:37:36Z","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","version":4},"cited_work":{"arxiv_id":"1910.10683","doi":"10.18653/v1/2020.acl-main.259","metadata_source":"pith","pith_arxiv_id":"1910.10683","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","venue":"cs.LG","work_id":"50e3b368-0243-4726-8186-233869802ad1","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1910.10683","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:730d8c9e02883ba28efd1004335234729a0d7f55a6f7a5f915b6183a9a3d2a90","observation_id":"80cfaf98-4c99-4610-8aa6-4cec11c53c3c","resolution":{"observed_at":"2026-05-13T12:26:58.114416Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SQuAD: 100,000+ questions for machine comprehension of text","venue":null,"work_id":"e044be97-68c9-4045-83c8-3ee40767dad0","year":2020},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:c5203db4943ff37e575311c5fb6e4bf7cf51b2f383722513932b3227939d49b1","observation_id":"5fd26503-17a8-43c8-803e-a5bd5f886252","resolution":{"observed_at":"2026-05-13T12:26:58.157456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d16-1264","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T20:15:34.199731Z","title":"SQ u AD : 100,000+ questions for machine comprehension of text","venue":"Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing","work_id":"8e6a63f7-90ad-4b5e-8493-c26145f74b69","year":2016},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:8d482230ddd4e3609e15584f6398dce9623d63e36bcfd216002e505e16ee0bad","observation_id":"07a5e651-4502-4527-93de-54cec2c1c6ec","resolution":{"observed_at":"2026-05-13T12:26:58.077179Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-11T05:49:42.935924+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T05:49:42.935924+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/p18-2124","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T22:17:24.924441Z","title":"In: Gurevych, I., Miyao, Y","venue":null,"work_id":"9fa70caa-364b-4753-b402-0f2e3ac53239","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:2e5014bd221775da4c521c2cc19c1521d157f40afc2e5325e35c385d44f4fa55","observation_id":"2080e1a9-bfde-4d57-8ad5-4851ad5ca384","resolution":{"observed_at":"2026-05-13T12:26:58.070938Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.00857","last_updated":"2018-04-03T07:41:10Z","snapshot_observed_at":"2026-07-06T06:31:29.247744Z","submitted_at":"2018-04-03T07:41:10Z","title":"Bi-Directional Block Self-Attention for Fast and Memory-Efficient Sequence Modeling","version":1},"cited_work":{"arxiv_id":"1804.00857","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1804.00857","snapshot_observed_at":"2026-07-04T22:40:08.461553Z","title":"Bi-directional block self- attention for fast and memory-efﬁcient sequence modeling","venue":null,"work_id":"738626e1-7d76-43a2-b226-172433270218","year":null},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1804.00857","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:cbfc05513b4aa8b51033c1adb073b75fe81ed268670772a7405afffaadd9245a","observation_id":"b211c2e1-5f7c-404a-99ca-ae72d7b1f399","resolution":{"observed_at":"2026-07-04T22:40:08.461553Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Manning, Andrew Ng, and Christopher Potts","venue":null,"work_id":"dc71a1e7-1d93-4212-b6bb-df8cfbf0a3f8","year":2013},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:fbaff94bdb0f5ad2671eb3975fc6d945cedd192d71d7d9c990a68c09108046e8","observation_id":"e47206dd-127b-4a92-90fb-ea363092fc87","resolution":{"observed_at":"2026-05-13T12:26:58.134616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.09355","last_updated":"2019-08-25T16:13:24Z","snapshot_observed_at":"2026-07-06T08:16:39.915426Z","submitted_at":"2019-08-25T16:13:24Z","title":"Patient Knowledge Distillation for BERT Model Compression","version":1},"cited_work":{"arxiv_id":"1908.09355","doi":"10.48550/arxiv.1908.09355","metadata_source":"pith","pith_arxiv_id":"1908.09355","snapshot_observed_at":"2026-07-10T21:17:35.850483Z","title":"arXiv preprint arXiv:1908.09355 (2019)","venue":"cs.CL","work_id":"15b142b1-fae1-4e20-b8d4-3af74cb8825a","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1908.09355","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:19f39a1a38dc98da3525067af633b42517505388899f9eaf157146d6bcc8c1be","observation_id":"b50a58fd-6a2e-4d9d-b589-059f7575d5b6","resolution":{"observed_at":"2026-05-13T12:26:58.120212Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-07-06T08:16:28.022100Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":null,"work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:0a8a8463f1cab5113ba2f7440383783a9536f5e3e0e5136e782b8101fb557fe8","observation_id":"df108f19-e6b9-407a-9853-abd23d00e31d","resolution":{"observed_at":"2026-05-13T12:26:58.123489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding","venue":null,"work_id":"0fcd8bae-8a14-4343-a111-57d02286c392","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:ded54a74b219a180b0364db401aaff8e74b1ec08a1876527d61b16bb0d91aab3","observation_id":"1637cc64-f285-4a00-9e37-ca6d4f3c6731","resolution":{"observed_at":"2026-05-13T12:26:58.140843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/w18-5446","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:57:06.760757Z","title":"Proceedings of the 2018","venue":null,"work_id":"24f74631-3b99-4e7e-ab15-b562e791542e","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:c2e62c9c04e4ce56efbd372bb3fc792ef5d9f61c15a153ad589ff1d7c80d88b5","observation_id":"a1bbdbf3-79ae-41b6-8572-dbc1a6270184","resolution":{"observed_at":"2026-05-13T12:26:58.060994Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-09T07:48:42.118348+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T07:48:42.118348+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.12471","last_updated":"2019-10-01T18:41:05Z","snapshot_observed_at":"2026-07-06T06:42:15.167440Z","submitted_at":"2018-05-31T13:52:06Z","title":"Neural Network Acceptability Judgments","version":3},"cited_work":{"arxiv_id":"1805.12471","doi":null,"metadata_source":"pith","pith_arxiv_id":"1805.12471","snapshot_observed_at":"2026-07-09T21:16:34.237758Z","title":"Neural Network Ac- ceptability Judgments","venue":"cs.CL","work_id":"ee6536e2-986a-4e85-877b-cd6cf6b9219e","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1805.12471","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:ec5095a84ef6573fdce4a0de85b8b1afdebd48e93d4e925a1de27987e726021e","observation_id":"e3e36867-c1b9-46a2-8274-49604f21458a","resolution":{"observed_at":"2026-05-13T12:26:58.126204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A broad-coverage challenge corpus for sen- tence understanding through inference","venue":null,"work_id":"f9f80303-14f8-4b20-bf6a-24285de09bc7","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:33317d5511bffda24a5fbe84766c5c22aff1da9d7d888bd57ced6d9e0c5186f4","observation_id":"9efdea7b-bc1f-487b-89b6-f9fcfcc8fda4","resolution":{"observed_at":"2026-05-13T12:26:58.147092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.05426","last_updated":"2018-02-19T19:19:51Z","snapshot_observed_at":"2026-07-06T05:38:16.162925Z","submitted_at":"2017-04-18T17:10:13Z","title":"A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference","version":4},"cited_work":{"arxiv_id":"1704.05426","doi":"10.18653/v1/n18-1101","metadata_source":"pith","pith_arxiv_id":"1704.05426","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference","venue":"cs.CL","work_id":"9737cdf0-fd48-4485-ab04-c8ec9386e782","year":2017},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1704.05426","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:3e5c7e14189fd006e96b344f6db32957423077db16aefc670021d5b2808ee051","observation_id":"514921cb-81e1-434c-b27e-9583c1319b41","resolution":{"observed_at":"2026-05-13T12:26:58.048845Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:49:40.633782+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:49:40.633782+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.00962","last_updated":"2020-01-03T06:53:00Z","snapshot_observed_at":"2026-07-06T07:43:06.427641Z","submitted_at":"2019-04-01T16:53:35Z","title":"Large Batch Optimization for Deep Learning: Training BERT in 76 minutes","version":5},"cited_work":{"arxiv_id":"1904.00962","doi":"10.48550/arxiv.1904.00962","metadata_source":"pith","pith_arxiv_id":"1904.00962","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Large Batch Optimization for Deep Learning: Training BERT in 76 minutes","venue":"cs.LG","work_id":"206d2a89-691e-4467-8958-5630dacf4765","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1904.00962","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:79ef00d0c0cb53fb40974754fcbe369ccc54f71080f41801db3e42f4275e47f7","observation_id":"ad8e6848-54e4-4a7c-a963-187f4beef389","resolution":{"observed_at":"2026-05-21T21:39:00.104164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-16T20:21:35.755083+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-16T20:21:35.755083+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.11511","last_updated":"2020-01-16T07:56:14Z","snapshot_observed_at":"2026-07-06T08:17:53.144974Z","submitted_at":"2019-08-30T02:30:28Z","title":"DCMN+: Dual Co-Matching Network for Multi-choice Reading Comprehension","version":4},"cited_work":{"arxiv_id":"1908.11511","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1908.11511","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DCMN+: Dual co-matching network for multi-choice reading comprehension","venue":null,"work_id":"fed344b5-411f-4f3c-a2b2-74867e0770c8","year":1908},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1908.11511","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:be3a9c9e9f6200f56a3bfb645f9af3e3b10abacdffcddc3d1f65802d5c17f574","observation_id":"7e4e67ab-6190-4789-827c-35808a87b747","resolution":{"observed_at":"2026-05-13T12:26:58.084998Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e404b5d8-9cfe-495d-9b18-0195ba8b0095","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:89dc0834d482737eef890659d1afd39726c6fb267623037d67f7f95fd171c5d5","observation_id":"f00aad56-1d64-4ea3-a027-3581ccbbba0b","resolution":{"observed_at":"2026-05-13T12:26:58.136510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We conclude that, when sharing all cross-layer parameters (ALBERT-style), there is no need for models deeper than a 12-layer conﬁguration","venue":null,"work_id":"75f2ed56-8323-444e-b7db-5b230e71bf65","year":2018},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:5b8452fdb07fcf85e94798cf580b5bb29873e9d21da030d5a73e539c83c22ff8","observation_id":"6c17a5da-4842-45c6-94d1-d4b3612acbe1","resolution":{"observed_at":"2026-05-13T12:26:58.138814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"It focuses on evaluating model capabilities for natural language understanding","venue":null,"work_id":"f371548a-b0b9-49b9-8dca-6f8bcd59ec99","year":2012},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:de159343025095179ceb639a14f52e83c5d3238914cf599b290ec3887bf95f6e","observation_id":"3033668d-1465-4ed9-9a09-76130e119fe8","resolution":{"observed_at":"2026-05-13T12:26:58.144988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For test set submissions, we perform task-speciﬁc modiﬁcations for WNLI and QNLI as described by Liu et al","venue":null,"work_id":"6ce5acee-cd23-4aa7-95ba-bf2a9f97f0ea","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:4e173963ab3b09336dc84bebbcc0025435755f87baa10a85fc367227a50770d1","observation_id":"de762cd1-7f5f-45ed-a798-772b86d0e962","resolution":{"observed_at":"2026-05-13T12:26:58.155438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(2019), Devlin et al","venue":null,"work_id":"e1be3129-f53f-4535-be4e-7b995ca2c375","year":2019},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:1ba558b3e81447000d9cf55e280c9e6a633f4c8c1f342e2e8d4a1adcac8e7181","observation_id":"9651ef17-4623-48d7-836c-056055e683a6","resolution":{"observed_at":"2026-05-13T12:26:58.163798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations"},"reference_resolution":{"displayed":46,"state_counts":{"malformed_identifier":0,"metadata_mismatch":14,"parse_uncertain":0,"unresolved":0,"verified_exact":17,"verified_fuzzy":15},"total_outbound_references":46},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"thesis":"As of 22 July 2026, this Paper Citation Record lists 46 of 46 outbound references and 65 inbound Pith citation observations for arXiv:1909.11942."}