{"work":{"id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","openalex_id":"https://openalex.org/W4414359541","doi":"10.24963/ijcai.2025/706","arxiv_id":"2307.09288","raw_key":null,"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","authors":null,"authors_text":"Hugo Touvron, Louis Martin, Kevin Stone, et al","year":2023,"venue":"cs.CL","abstract":"In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on our work and contribute to the responsible development of LLMs.","external_url":"https://arxiv.org/abs/2307.09288","cited_by_count":0,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2307.09288","created_at":"2026-05-08T18:13:54.069555+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","render_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models"},"hub":{"state":{"work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","tier":"mega_hub","tier_reason":"1,000+ Pith inbound or 100,000+ external citations","pith_inbound_count":1105,"external_cited_by_count":0,"distinct_field_count":37,"first_pith_cited_at":"2023-02-03T06:06:27+00:00","last_pith_cited_at":"2026-07-09T12:28:39+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"needed","recognition_status":"needed","updated_at":"2026-08-20T16:39:20.917328+00:00","tier_text":"mega_hub"},"tier":"mega_hub","role_counts":[{"context_role":"background","n":154},{"context_role":"method","n":21},{"context_role":"dataset","n":9},{"context_role":"baseline","n":6},{"context_role":"other","n":3},{"context_role":"extension","n":1}],"polarity_counts":[{"context_polarity":"background","n":148},{"context_polarity":"use_method","n":21},{"context_polarity":"use_dataset","n":9},{"context_polarity":"unclear","n":7},{"context_polarity":"baseline","n":6},{"context_polarity":"support","n":2},{"context_polarity":"extend","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","claims":[{"claim_text":"In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on o","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Llama 2: Open Foundation and Fine-Tuned Chat Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T18:43:30.020349+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"6af248bf-4977-4496-b743-94d90183b7a5","orcid":null,"display_name":"Hugo Touvron"},{"id":"1fe8ecee-1c46-42b9-87fb-3b160972fb65","orcid":null,"display_name":"Louis Martin"},{"id":"643d0963-d7dd-4acc-ad85-741c731071d1","orcid":null,"display_name":"Kevin Stone"},{"id":"535def4b-a2c7-4c75-ad8b-db3e939f2bc9","orcid":null,"display_name":"et al"}]},"error":null,"updated_at":"2026-05-13T18:43:30.018391+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T18:43:29.640696+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":111},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":95},{"title":"Mistral 7B","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","shared_citers":62},{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":54},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":52},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":51},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":49},{"title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","shared_citers":42},{"title":"Gemini: A Family of Highly Capable Multimodal Models","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","shared_citers":40},{"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","shared_citers":40},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":31},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":30},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":29},{"title":"Code Llama: Open Foundation Models for Code","work_id":"e73bffa4-7620-47ac-9327-259a60db52ca","shared_citers":25},{"title":"Gemma: Open Models Based on Gemini Research and Technology","work_id":"a9ea2870-df28-40b8-a9e0-a7e9a116f793","shared_citers":25},{"title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","work_id":"3322fa86-1768-4677-8425-dd326b45e078","shared_citers":25},{"title":"Qwen2 Technical Report","work_id":"a1857881-ab9b-4b80-9b5f-9ae4b5c2566d","shared_citers":24},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":23},{"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","shared_citers":23},{"title":"Constitutional AI: Harmlessness from AI Feedback","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","shared_citers":22},{"title":"Measuring Massive Multitask Language Understanding","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","shared_citers":22},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":22},{"title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","work_id":"ed240a10-5b19-406c-baa5-30803f465785","shared_citers":21},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":20}],"time_series":[{"n":19,"year":2023},{"n":39,"year":2024},{"n":7,"year":2025},{"n":274,"year":2026}]},"error":null,"updated_at":"2026-05-13T17:25:56.537419+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T18:43:29.014608+00:00"},"reader_index":{"job_type":"reader_index","status":"succeeded","result":{"note":"annotated reader requires full-text/OA fetch; shell is wired for mega hubs","status":"reader queued"},"error":null,"updated_at":"2026-07-03T00:33:14.325722+00:00"},"recognition_alignment":{"job_type":"recognition_alignment","status":"succeeded","result":{"modules":["IndisputableMonolith.Chemistry.VanDerWaals","IndisputableMonolith.Cosmology.DarkEnergy","IndisputableMonolith.Physics.GrandUnificationFromRS","IndisputableMonolith.StandardModel.StrongCP","IndisputableMonolith.Linguistics.PhonemeInventoryBandFromRS","IndisputableMonolith.QFT.ElectroweakScaleStructure","IndisputableMonolith.Sociology.DunbarFromBandwidth","IndisputableMonolith.Foundation.AlexanderDuality"],"query_chars":714},"error":null,"updated_at":"2026-07-03T00:33:34.008194+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","claims":[{"claim_text":"In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on o","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Llama 2: Open Foundation and Fine-Tuned Chat Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T18:43:29.644957+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","claims":[{"claim_text":"In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on o","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Llama 2: Open Foundation and Fine-Tuned Chat Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T17:25:52.728519+00:00"}},"summary":{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","claims":[{"claim_text":"In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on o","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Llama 2: Open Foundation and Fine-Tuned Chat Models because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":111},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":95},{"title":"Mistral 7B","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","shared_citers":62},{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":54},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":52},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":51},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":49},{"title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","shared_citers":42},{"title":"Gemini: A Family of Highly Capable Multimodal Models","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","shared_citers":40},{"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","shared_citers":40},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":31},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":30},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":29},{"title":"Code Llama: Open Foundation Models for Code","work_id":"e73bffa4-7620-47ac-9327-259a60db52ca","shared_citers":25},{"title":"Gemma: Open Models Based on Gemini Research and Technology","work_id":"a9ea2870-df28-40b8-a9e0-a7e9a116f793","shared_citers":25},{"title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","work_id":"3322fa86-1768-4677-8425-dd326b45e078","shared_citers":25},{"title":"Qwen2 Technical Report","work_id":"a1857881-ab9b-4b80-9b5f-9ae4b5c2566d","shared_citers":24},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":23},{"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","shared_citers":23},{"title":"Constitutional AI: Harmlessness from AI Feedback","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","shared_citers":22},{"title":"Measuring Massive Multitask Language Understanding","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","shared_citers":22},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":22},{"title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","work_id":"ed240a10-5b19-406c-baa5-30803f465785","shared_citers":21},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":20}],"time_series":[{"n":19,"year":2023},{"n":39,"year":2024},{"n":7,"year":2025},{"n":274,"year":2026}]},"authors":[{"id":"535def4b-a2c7-4c75-ad8b-db3e939f2bc9","orcid":null,"display_name":"et al","source":"manual","import_confidence":0.72},{"id":"6af248bf-4977-4496-b743-94d90183b7a5","orcid":null,"display_name":"Hugo Touvron","source":"manual","import_confidence":0.72},{"id":"643d0963-d7dd-4acc-ad85-741c731071d1","orcid":null,"display_name":"Kevin Stone","source":"manual","import_confidence":0.72},{"id":"1fe8ecee-1c46-42b9-87fb-3b160972fb65","orcid":null,"display_name":"Louis Martin","source":"manual","import_confidence":0.72}]},"citers":{"total":1105,"items":[{"citing_arxiv_id":"2607.08403","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Game Theory Driven Multi-Agent Framework Mitigates Language Model Hallucination","primary_cat":"cs.AI","submitted_at":"2026-07-09T12:28:39+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A game-framed multi-agent system synthesizes large chemistry CoT/QA corpora and trains OmniChem-7B to near GPT-4o-mini performance with a large reported drop in hallucinations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08399","ref_index":52,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Prompt Compression via Activation Aggregation","primary_cat":"cs.CL","submitted_at":"2026-07-09T12:21:44+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A learned weighted sum of intermediate-layer activations compresses an instruction prompt into a single patch vector that, injected at an early layer, recovers task accuracy within ~2% of the full prompt.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07953","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Linear Attention Architectures: Mechanisms, Trade-offs, and Cross-Layer Routing","primary_cat":"cs.LG","submitted_at":"2026-07-08T22:14:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.5,"formal_verification":"none","one_line_summary":"Among 350M models trained for 15B tokens, Kimi Delta Attention with Muon has the best validation loss, pure Gated DeltaNet is fastest, and Cross-Layer Value Routing modestly lowers loss for DeltaNet-style memories.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07779","ref_index":226,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"From Solvers to Research: Large Language Model-Driven Formal Mathematics at the Research Frontier","primary_cat":"cs.CL","submitted_at":"2026-07-08T17:46:36+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLM formal provers must shift from competition solvers to research agents that handle open-ended, under-specified frontier mathematics under machine-checked rigor.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07678","ref_index":38,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"How Data Shapes RoPE Frequency Usage: From Positional Scale Matching to Length Generalization","primary_cat":"cs.LG","submitted_at":"2026-07-08T17:38:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"RoPE frequency usage is determined by a data-induced dependency width W, with the optimal frequency scaling as π/W, explaining both learned spectra and the success of position interpolation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07619","ref_index":81,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Rethinking Code Performance Benchmarks for LLMs","primary_cat":"cs.SE","submitted_at":"2026-07-08T16:36:03+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Re-evaluating four LLM code-efficiency benchmarks with 30-run statistical testing shows 93.89% of 'performant' implementations are indistinguishable from baselines; a multi-agent test-generation framework reveals hidden significant improvements in ~24% of previously non-significant tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07608","ref_index":61,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Dual Latent Memory in Vision-Language-Action Models for Robotic Manipulation","primary_cat":"cs.RO","submitted_at":"2026-07-08T16:26:06+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LaMem-VLA reconstructs robotic history into dual short-term and long-term latent memory tokens that are woven directly into a VLA model's reasoning sequence to improve long-horizon manipulation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07388","ref_index":30,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"TF-Engram: A Train-Free Engram with SSD-Backed Memory for Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-07-08T13:19:52+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A train-free phrase memory system for LLMs uses SSD-backed hierarchical storage and early-exit predictive prefetching to improve downstream accuracy without backbone training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07380","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Interpretable Uncertainty for Adaptive Retrieval and Reasoning in Question Answering","primary_cat":"cs.IR","submitted_at":"2026-07-08T13:12:39+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Regression probes on LLM hidden states estimate knowledge insufficiency and ambiguity to guide adaptive retrieval and reasoning in QA.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07235","ref_index":46,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ORCAID: Oblique Rule-Based Continuous-Action Interpretation for Deep RL Policies","primary_cat":"cs.LG","submitted_at":"2026-07-08T10:16:17+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Oblique decision trees with local linear models can approximate deep RL policies with continuous actions using far fewer parameters than axis-aligned trees while retaining task performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07740","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Jet-Long: Efficient Long-Context Extension with Dynamic Bifocal RoPE","primary_cat":"cs.LG","submitted_at":"2026-07-08T06:23:42+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Jet-Long is a tuning-free bifocal RoPE method that dynamically sets remote group size from sequence length, recovering the base model within the pretrained window and beating prior zero-shot extenders on RULER, HELMET-RAG, and PG-19 with near-FA2 throughput.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06564","ref_index":66,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Lift3D-VLA: Lifting VLA Models to 3D Geometry and Dynamics-Aware Manipulation","primary_cat":"cs.RO","submitted_at":"2026-07-07T17:59:47+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Lift3D-VLA integrates 3D point cloud encoding and temporal action modeling into Vision-Language-Action models, achieving higher success rates on simulated and real-world robotic manipulation tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06157","ref_index":63,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability","primary_cat":"cs.CL","submitted_at":"2026-07-07T11:34:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A benchmark for LLM agents in partially observable joint decision-making reveals that deliberation challenges current models but can enable reflection and error correction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06120","ref_index":38,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"AEGIS: A Mechanism-Guided Defense against Visual Synonym Jailbreaks in Text-to-Image Models","primary_cat":"cs.CV","submitted_at":"2026-07-07T10:27:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":6.0,"formal_verification":"none","one_line_summary":"AEGIS localizes sparse semantic-injecting attention heads in diffusion models and applies similarity-aware repulsion at those heads to block visual synonym jailbreaks while preserving benign generation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05916","ref_index":47,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond the Syntax: Do Security Experts Trust LLMs for NIDS Rule Engineering?","primary_cat":"cs.CR","submitted_at":"2026-07-07T07:09:56+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A user study with 10 security experts reveals that while large LLMs (≥70B) generate syntactically valid NIDS rules, experts deem only 37.5% deployable due to low specificity and logic hallucinations, viewing LLMs as support tools rather than autonomous rule generators.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02266","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"HERMES: A Multi-Granularity Labeling Substrate for Pre-training Data Mixtures","primary_cat":"cs.LG","submitted_at":"2026-07-02T14:51:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"HERMES provides a reusable hierarchical labeling substrate for pre-training data that reveals granularity-specific effects in data mixing rules during model training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02182","ref_index":32,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Bayesian Sparse Low-Rank Adaptation for Large Language Model Uncertainty Estimation","primary_cat":"cs.LG","submitted_at":"2026-07-02T13:52:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"DALorRA applies variational Bayesian sparse masking to LoRA ranks to calibrate LLM uncertainty while preserving accuracy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01883","ref_index":74,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PairCoder++: Pair Programming as a Universal Paradigm for Verified Code-Driven Multimodal and Structured-Artifact Generation","primary_cat":"cs.CL","submitted_at":"2026-07-02T08:36:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PairCoder is a two-agent pair-programming method that leverages toolchain verification oracles to improve LLM generation of verifiable structured artifacts on 17 benchmarks across seven models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01733","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Rethinking Speech-LLM Integration for ASR: Effective Joint Speech-Text Training by Interleaving","primary_cat":"cs.CL","submitted_at":"2026-07-02T05:42:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"JSTIP interleaves speech and text sequences during pretraining on 38k hours of ASR data to improve entity accuracy over ASR-only and simple joint-training baselines while matching performance from domain text.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01689","ref_index":60,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Model Merging as Probabilistic Inference in Fine-Tuning Parameter Space","primary_cat":"cs.LG","submitted_at":"2026-07-02T04:30:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Model merging is cast as PoE inference with EBM experts, revealing Gaussian assumptions in prior work and proposing convergent Cauchy experts that improve empirical performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01678","ref_index":43,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SCAPE: Accurate and Efficient LLM Training with Extreme Sparse Communication","primary_cat":"cs.LG","submitted_at":"2026-07-02T04:10:42+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SCAPE enables 90-99% sparse gradient communication in sharded Adam-style LLM training by deriving masks from first-moment statistics, achieving up to 43.3% faster pre-training on Llama-500M with no loss in validation loss or downstream accuracy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01394","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The Wiola Architecture for Efficient Small Language Models","primary_cat":"cs.AI","submitted_at":"2026-07-01T18:52:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Wiola introduces five new architectural components for small language models and releases models from 120M to 1.5B parameters compatible with Hugging Face.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01392","ref_index":61,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Multi-Objective Exploration and Preference Optimization via Mutual Information","primary_cat":"cs.CL","submitted_at":"2026-07-01T18:50:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Maximizing joint conditional mutual information I(Y; C_Z, W, Z | X) decomposes multi-objective LLM alignment into preference-specific DPO terms plus an I(Y;W|X) exploration term that reduces reward-distribution overlap.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01144","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Sequentially-Controlled Interactive Multi-Particle Flow-Maps for Online Feedback-Driven Search","primary_cat":"cs.LG","submitted_at":"2026-07-01T16:27:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"IMPFM is a multi-particle flow-map sampling method with sequential posterior sharing and interaction-aware correction that targets a KL-tilted distribution for global exploration in online feedback search.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01065","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","primary_cat":"cs.LG","submitted_at":"2026-07-01T15:25:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"GSRQ applies a gain-shape variant of K-means inside residual quantization to improve directional fidelity, raising LongBench accuracy from 11.34 to 33.54 at 1-bit on LLaMA-3-8B.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00816","ref_index":44,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Towards High-Resolution Visual Perception via Hierarchical Entity Exploration","primary_cat":"cs.CV","submitted_at":"2026-07-01T11:41:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"HEE is a training-free, model-agnostic method for high-resolution visual perception in MLLMs using hierarchical entity exploration with dual scoring, detection, clustering, and backtracking.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00511","ref_index":98,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Large Language Models for Multi-Lingual Equivalent Mutant Detection: An Extended Empirical Study","primary_cat":"cs.SE","submitted_at":"2026-07-01T06:46:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLM-based methods achieve higher F1-scores than traditional approaches for equivalent mutant detection in Java and C, with fine-tuned code embeddings performing best and showing cross-lingual generalization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00486","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PAPA: Online Personalized Active Preference Alignment","primary_cat":"cs.LG","submitted_at":"2026-07-01T06:14:22+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PAPA directly optimizes diffusion models via real-time user feedback for personalized preference alignment, drawing from variational inference, with an efficiency-enhanced variant EPAPA.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00302","ref_index":44,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Wake up for Touch! Mask-isolated Tactile Alignment Learning in MLLMs","primary_cat":"cs.CV","submitted_at":"2026-07-01T01:02:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Splash partitions MLLM parameters into dormant and critical subspaces via significance quantification, updating only the dormant subspace for tactile alignment while preserving general capabilities and achieving SOTA on visuo-tactile benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00254","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Query-Centric Optimization of AI Workflows via Approximate Query Processing and Proxy Models","primary_cat":"cs.DB","submitted_at":"2026-06-30T23:05:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Query-centric AQP and proxy-model strategies reduce expensive model calls by 60-90% with under 10% error on TPC-DS and LLM tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00151","ref_index":65,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SmoothAgent: Efficient Long-Horizon LLM-Based Agent Serving with Lookahead Context Engineering","primary_cat":"cs.DC","submitted_at":"2026-06-30T20:27:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SmoothAgent introduces lookahead context engineering to eliminate transformation overhead in LLM agents, reducing TTFT by up to 11.9x through proactive KV cache preparation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.32016","ref_index":156,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"FedLAB: Traceable Semantic Codebooks for Federated Multimodal Graph Foundation Learning","primary_cat":"cs.LG","submitted_at":"2026-06-30T17:47:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"FedLAB organizes multimodal graph knowledge into typed hierarchical codebooks for modality evidence, node semantics, and topology context via federated semantic barycenter pre-training, improving performance by up to 7.53% on benchmarks while enabling semantic traceability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31963","ref_index":1,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Signed-Permutation Coordinate Transport for RMSNorm Transformers","primary_cat":"cs.LG","submitted_at":"2026-06-30T17:02:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Introduces signed-permutation gauge B_d for RMSNorm models and sign-marginalized Hungarian matching, showing improved coordinate recovery along fine-tuning trajectories and better transfer of SAEs and steering vectors.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31717","ref_index":47,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Nonlinearity-Aware LoRA: Structured Gate Adaptation under Low-Rank Constraints","primary_cat":"cs.LG","submitted_at":"2026-06-30T14:21:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"NA-LoRA introduces derivative-based temporal-importance masks and activation-specific step scaling to LoRA to reduce selection misalignment in self-gated FFNs, with reported gains on language and vision-language fine-tuning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31464","ref_index":24,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Team MKC at CLPsych 2026: Capturing and Characterizing Mental Health Changes through Social Media Timeline Dynamics","primary_cat":"cs.CL","submitted_at":"2026-06-30T10:43:22+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":2.0,"formal_verification":"none","one_line_summary":"LLM pipeline for joint post-level assessment and user-level temporal modeling of mental health from ordered social media posts in a shared task.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31397","ref_index":87,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Mixture-of-Control: State-Aware Fine-Tuning for Transformer-based Models","primary_cat":"cs.LG","submitted_at":"2026-06-30T09:25:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Mixture-of-Control adaptively combines local and global control states in transformer fine-tuning by treating per-block states as experts in a sparse MoE setup to improve cross-block communication while keeping memory and compute costs comparable to prior state-based methods.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31382","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Revisiting Parameter Redundancy in Vision-Language-Action Models: Insights from VLM-to-VLA Adaptation","primary_cat":"cs.RO","submitted_at":"2026-06-30T09:10:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VLA models from VLM adaptation can be pruned 12-30% via multi-module joint scheme based on divergence signals while keeping ~90% performance on LIBERO without post-pruning recovery, unlike standard criteria that collapse.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31247","ref_index":119,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","primary_cat":"cs.SD","submitted_at":"2026-06-30T07:24:10+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FlexiSLM is the first spoken language model supporting dynamic and controllable frame rates on speech input and output, outperforming fixed-rate 7B models at high quality and enabling faster inference at lower rates like 6.25 Hz.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31209","ref_index":59,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Long-term Traffic Simulation via Structured Autoregressive Modeling","primary_cat":"cs.AI","submitted_at":"2026-06-30T06:41:28+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RosettaSim adapts frozen LLMs via structured autoregressive modeling of scene topology and agent states to reach SOTA short- and long-term traffic simulation on WOSAC, paired with RTE evaluation that correlates better with human-like fidelity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31163","ref_index":7,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ComplianceGate: Classifier-Gated Multi-Tier LLM Routing for Inference in Regulated Industries","primary_cat":"cs.LG","submitted_at":"2026-06-30T05:49:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A classifier before any LLM inference routes PII queries to local endpoints and simple queries to small models, reporting 39% latency reduction and 33-52% cost savings on 600 queries with 99.2% classifier accuracy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31093","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Omni-Flow: A Unified Workflow Orchestration and Distributed KV Cache Sharing Framework for Multimodal Inference","primary_cat":"cs.DC","submitted_at":"2026-06-30T03:29:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Omni-Flow introduces a three-layer abstraction (Control Flow, Data Flow, Compute Flow) for unified orchestration and KV cache sharing in multimodal inference pipelines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00052","ref_index":73,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"AGE: Adaptive-masking for Graph Embedding in Graph Retrieval-Augmented Generation","primary_cat":"cs.IR","submitted_at":"2026-06-30T01:23:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"AGE applies adaptive masking via a learnable sampler in Transformer-based SSL to align graph and text embeddings, yielding higher accuracy on four GraphQA benchmarks for non-parametric GraphRAG.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30814","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Calibration Rankings Reverse: Accuracy-Controlled Evaluation for Fair Comparison of LLMs","primary_cat":"cs.CL","submitted_at":"2026-06-29T18:37:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Global calibration metrics like ECE are confounded by accuracy; the proposed ACE framework with three accuracy-controlled views shows many prior calibration advantages weaken or reverse.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30790","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Indi-RomCoM: Code-Mixed Benchmark for Evaluating LLMs on Romanized Indic-English Instructions","primary_cat":"cs.CL","submitted_at":"2026-06-29T18:19:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Introduces Indi-RomCoM benchmark for evaluating LLMs on Romanized code-mixed Indic-English instructions across seven tasks, four languages, and three mixing levels.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30460","ref_index":14,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","primary_cat":"cs.LG","submitted_at":"2026-06-29T15:26:55+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"HSAP introduces a hierarchical framework and sequence-aware algorithm with JIT-optimized NCCL communication to enable correct causal attention computation on hybrid-context packed sequences without limiting parallelism.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30077","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Online Data Selection for Instruction Tuning via Gaussian Processes","primary_cat":"cs.LG","submitted_at":"2026-06-29T10:08:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"GAIA models continuous utility with Gaussian processes across semantic space and applies fixed-share Hedge updates to achieve dynamic regret guarantees while outperforming baselines on three datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29462","ref_index":43,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MIRROR: Aligning Semantic Relations from Language to Image via Gromov--Wasserstein","primary_cat":"cs.CV","submitted_at":"2026-06-28T15:39:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MIRROR derives a closed-form Semi-Inverse Gromov-Wasserstein loss to align language-derived relational priors with visual representations inside decoder-only Transformers.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29431","ref_index":47,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"FADE: Mitigating Hallucinations by Reducing Language-Prior Dominance in Large Vision-Language Models","primary_cat":"cs.AI","submitted_at":"2026-06-28T14:48:08+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Attenuating FFN outputs at mid-to-late transformer layers reduces language-prior dominance and thereby mitigates object hallucinations in LVLMs while preserving efficiency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29407","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LC-ICL: Label-Guided Contrastive In-Context Learning for Robust Information Extraction","primary_cat":"cs.CL","submitted_at":"2026-06-28T14:01:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LC-ICL improves few-shot NER and RE by using label-guided contrastive demonstrations that pair positive samples with error-annotated negative samples.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29223","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Depth Exploration for LLM Decoding","primary_cat":"cs.LG","submitted_at":"2026-06-28T06:22:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DEX replaces single-depth selection with parallel exploration over multiple candidate depths, committing the final-depth token while collapsing reusable states to reduce per-token computation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28754","ref_index":51,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SHIFT: Dynamic Compute Relocation Framework for Communication-Aware Chiplet-Based Systems","primary_cat":"cs.AR","submitted_at":"2026-06-27T06:02:23+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Runtime compute relocation via utility chiplets cuts communication cost on multi-bandwidth chiplet fabrics, delivering multi-x gains on LLM inference in simulation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28707","ref_index":58,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","primary_cat":"cs.AI","submitted_at":"2026-06-27T03:25:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"BV-Blend blends prompt-local and semantic-cluster historical reward statistics via SEM-derived weights to stabilize critic-free RL advantage estimation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28620","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Reproducing FACTER: Fairness via Conformal Thresholding and Prompt Repair","primary_cat":"cs.IR","submitted_at":"2026-06-26T21:37:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Reproduction of FACTER across architectures and sparsity levels shows static fairness instructions match dynamic prompt repair on semantic parity in fixed-candidate re-ranking, with code released.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28192","ref_index":27,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PA-BiCoop: A Primary-Auxiliary Cooperative Framework for General Bimanual Manipulation","primary_cat":"cs.RO","submitted_at":"2026-06-26T15:38:00+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"PA-BiCoop introduces a single-model bimanual framework with primary-auxiliary arm differentiation, specialized decoders, and dynamic role assignment that reports 48% average gains on RLBench2 and over 50% in real-world tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28117","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When One Adapter Speaks for Many: Discovering Low-Rank Redundancy in Continual Fine-Tuning","primary_cat":"cs.LG","submitted_at":"2026-06-26T14:19:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Task-specific LoRA adapters in continual learning exhibit significant low-rank subspace overlap, enabling LiteLoRA's learned gating to reduce active adapters by 20-70% while matching or exceeding prior performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28076","ref_index":50,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Ontology-Guided Evidence Path Inference for Multi-hop Knowledge Graph Question Answering","primary_cat":"cs.AI","submitted_at":"2026-06-26T13:40:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"OPI introduces a relation-centric ontology graph enabling bidirectional retrieval and iterative refinement, yielding Hit@1/F1 gains of 4.6/5.0 on WebQSP and 8.9/3.3 on CWQ plus near-saturated Hit@1 on MetaQA.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27755","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Drop-Then-Recovery: How Redundant Are Vision-Language-Action Models?","primary_cat":"cs.RO","submitted_at":"2026-06-26T06:22:17+00:00","verdict":"ACCEPT","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VLA language backbones show high redundancy on manipulation benchmarks, with half the LLM blocks removable and even two blocks sufficient to recover baseline performance after fine-tuning, unlike vision and action pathways.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27705","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Mitigating Position Bias in Transformers via Layer-Specific Positional Embedding Scaling","primary_cat":"cs.CL","submitted_at":"2026-06-26T04:07:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LPES uses per-layer scaling factors optimized by a genetic algorithm with Bézier curves to balance attention and improve long-context LLM performance by up to 11.2% on key-value retrieval.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27683","ref_index":38,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CBD: API-Only LLM Black-Box Unlearning through Controlled Behavioral Divergence","primary_cat":"cs.LG","submitted_at":"2026-06-26T03:29:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"CBD is an API-only black-box unlearning method for LLMs that creates controlled behavioral divergence with auxiliary models and uses a Fisher-matrix-derived discriminative basis to balance forgetting target data with retained utility.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27632","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","primary_cat":"cs.CL","submitted_at":"2026-06-26T01:12:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Yuvion LLM applies adversarially aware training and introduces the YLRE benchmark set, claiming superior safety robustness over larger models on multiple tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27617","ref_index":43,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Masked Language Flow Models","primary_cat":"cs.CL","submitted_at":"2026-06-26T00:16:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MLFMs combine masking with continuous flows to scale flow-based language models to reasoning and instruction-following tasks on GSM8K and MT-Bench.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27527","ref_index":25,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Large Language Model Teaches Visual Students: Cross-Modality Transfer of Fine-Grained Conceptual Knowledge","primary_cat":"cs.CV","submitted_at":"2026-06-25T20:19:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LaViD distills LLM conceptual knowledge to vision models via LLM-generated MCQ soft labels, outperforming vision-language distillation baselines on fine-grained benchmarks while improving robustness on spurious correlation datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27199","ref_index":37,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Forecasting With LLMs: Improved Generalization Through Feature Steering","primary_cat":"cs.CL","submitted_at":"2026-06-25T15:59:20+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27161","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"TOPS: First-Principles Visual Token Pruning via Constructing Token Optimal Preservation Sets for Efficient MLLM Inference","primary_cat":"cs.AI","submitted_at":"2026-06-25T15:29:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"TOPS formulates visual token pruning as constructing Token Optimal Preservation Sets using three information-theoretic principles and demonstrates superior performance on MLLM benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27147","ref_index":22,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Safe Autoregressive Image Generation with Iterative Self-Improving Codebooks","primary_cat":"cs.CV","submitted_at":"2026-06-25T15:18:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Iterative self-improving codebooks enhance safety in autoregressive multimodal models by self-identifying unsafe generations and updating the codebook to eliminate harmful visual token mappings without external feedback.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26981","ref_index":59,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"In-Context Model Predictive Generation: Open-Vocabulary Motion Synthesis from Language Models to Physics","primary_cat":"cs.RO","submitted_at":"2026-06-25T12:50:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ICMPG combines LLM-based candidate generation with MPC-style physical simulation and semantic scoring to produce text-driven human motions that are both plausible and faithful.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26923","ref_index":77,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GAVEL: Grounded Caption Error Verification and Localization","primary_cat":"cs.CL","submitted_at":"2026-06-25T12:00:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"GAVEL introduces a joint task, dataset, and benchmark for verifying, explaining, and localizing caption-image misalignments, with a supervised baseline that improves grounding and explanation metrics over strong closed-source models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26749","ref_index":103,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","primary_cat":"cs.LG","submitted_at":"2026-06-25T08:33:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Semantic geometry emerges transiently early in next-token prediction training before collapsing to Neural Collapse symmetry in synthetic settings with latent semantic factors.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26650","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CAT-Q: Cost-efficient and Accurate Ternary Quantization for LLMs","primary_cat":"cs.CL","submitted_at":"2026-06-25T06:24:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CAT-Q performs post-training ternary quantization of 1.7B-235B LLMs with 512 samples via learnable modulation and softened ternarization, outperforming BitNet v1/v2 models trained on 100B tokens.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26633","ref_index":60,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Simulating Unified Tensor Resharding in heterogeneous AI systems","primary_cat":"cs.DC","submitted_at":"2026-06-25T05:50:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Xsim is a heterogeneity-aware simulator for distributed LLM training supporting load balancing, customized collectives, tensor resharding, and pluggable network simulation, reporting under 5% error in training time predictions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26583","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Preference Optimization Drives Monoculture in LLM Prediction Markets","primary_cat":"cs.CE","submitted_at":"2026-06-25T04:16:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DPO fine-tuning causes LLM agents to share output distributions with pairwise error correlations of ρ=0.70, reducing ten agents to the effective power of ≈1.4 independent forecasters.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26566","ref_index":64,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","primary_cat":"cs.CR","submitted_at":"2026-06-25T03:32:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A narrative survey that catalogs fifty papers on diffusion-based adversarial techniques across text, vision, and vision-language models, proposes a six-class taxonomy of diffusion roles plus a unified five-dimension evaluation framework, and releases a companion catalog.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26454","ref_index":36,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Data-driven Machine Learning Cannot Reach Symbolic-level Logical Reasoning -- The Limit of the Scaling Law","primary_cat":"cs.AI","submitted_at":"2026-06-24T23:30:08+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Perfect test accuracy does not imply symbolic-level logical reasoning: GPT-5 gives 100% right syllogism decisions with 25 wrong explanations, and the image-based SupEN stalls at 50-97.8% per mood.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.25800","ref_index":25,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ROAD-VLA: Robust Online Adaptation via Self-Distillation for Vision-Language-Action Models","primary_cat":"cs.LG","submitted_at":"2026-06-24T13:17:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ROAD-VLA constructs an advantage-perturbed proximal teacher in action space to convert sparse rewards into dense supervision for online VLA adaptation and reports outperformance versus PPO across seven manipulation environments.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.25325","ref_index":140,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","primary_cat":"cs.AI","submitted_at":"2026-06-24T02:43:26+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A cue-coverage reward plus a modality-token KL penalty makes multimodal emotion-reasoning models cite more real visual/audio evidence and hallucinate less, with reported SoTA on emotion benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24790","ref_index":61,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Grad Detect: Gradient-Based Hallucination Detection in LLMs","primary_cat":"cs.LG","submitted_at":"2026-06-23T16:46:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Grad Detect uses internal gradient patterns from one inference pass to predict LLM hallucinations and abstention, outperforming confidence and sampling baselines on Q&A benchmarks with most signal in the final five layers.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.25001","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Erased, but Not Gone: Output Forgetting Is Not True Forgetting","primary_cat":"cs.LG","submitted_at":"2026-06-23T16:23:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Output forgetting in machine unlearning overestimates success because unlearned models exhibit structured representation mismatches relative to retraining from scratch.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24734","ref_index":185,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Task Decomposition for Efficient Annotation","primary_cat":"cs.CL","submitted_at":"2026-06-23T15:58:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Decomposing annotation tasks using centers from centering theory reduces aggregate inferential load via a degrees-of-freedom model and enables better sub-task allocation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24970","ref_index":54,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Don't Go Breaking My LLM: The Impact of Pruning Attention Layers on Explanation Faithfulness and Confidence Calibration","primary_cat":"cs.LG","submitted_at":"2026-06-23T11:07:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Pruning attention layers in five LLMs across eight datasets maintains accuracy but degrades faithfulness and calibration.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24331","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Transformer-Based Language Models Across Domain Verticals: Architectures, Applications and Critical Assessment","primary_cat":"cs.CL","submitted_at":"2026-06-23T09:09:55+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":2.0,"formal_verification":"none","one_line_summary":"A survey paper that taxonomizes transformer architectures, reviews domain applications, and critically assesses deployment trade-offs including parameter-energy costs and alignment issues.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24320","ref_index":22,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ZONOS2 Technical Report","primary_cat":"cs.SD","submitted_at":"2026-06-23T08:57:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"ZONOS2 8B is a scaled MoE TTS model with 900M active parameters trained on 6M hours of data that reports competitive SOTA results on naturalness, speaker similarity, WER, and a new ZTTS1-Eval benchmark while releasing weights and code.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23626","ref_index":115,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DiT-Reward: Generative Representations for Text-to-Image Reward Modeling","primary_cat":"cs.LG","submitted_at":"2026-06-22T17:19:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DiT-Reward converts pretrained DiT models into reward predictors that outperform HPSv3 on four benchmarks while providing 1.65x inference speedup.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28386","ref_index":206,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Data Provenance for Image Auto-Regressive Generation","primary_cat":"cs.CV","submitted_at":"2026-06-22T17:06:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A post-hoc detection framework exploits generation-induced patterns in autoregressive image outputs to enable provenance tracing across multiple IAR models without altering the generation process.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23030","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Have You Ever Seen Them? Entity-level Membership Inference through Interrogating Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-06-22T08:40:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Entity-level membership inference determines whether information about a target real-world entity was used in LLM training, using only black-box generated text and achieving AUC up to 0.97 on person entities.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23001","ref_index":61,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"EnerInfer: Energy-Aware On-Device LLM Inference","primary_cat":"cs.SE","submitted_at":"2026-06-22T08:16:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"EnerInfer uses model-structure-aware predictions and online feedback to select energy-efficient NPU and memory frequencies for on-device LLM inference while preserving QoE and managing thermal limits.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23000","ref_index":49,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MotionMAR: Multi-scale Auto-Regressive Human Motion Reconstruction from Sparse Observations","primary_cat":"cs.CV","submitted_at":"2026-06-22T08:15:56+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22913","ref_index":31,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Intend, Reflect, Refine: An Adaptive Multimodal Reflection Framework for Autonomous Driving","primary_cat":"cs.CV","submitted_at":"2026-06-22T06:53:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"IRR-Drive adds an adaptive multimodal reflection step (text intention plus predicted future BEV) that lets a VLA model self-correct its trajectory plan according to scene complexity and reports SOTA on NAVSIM.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22873","ref_index":47,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","primary_cat":"cs.CV","submitted_at":"2026-06-22T05:37:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"SingGuard introduces a policy-adaptive multimodal LLM guardrail with dynamic reasoning regimes and SingGuard-Bench, reporting SOTA F1 scores across 35 datasets and improved policy-following accuracy under runtime shifts.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22794","ref_index":42,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"UniFS: Unified Fast-to-Slow Hierarchical Architecture for Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-06-22T03:10:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"UniFS achieves 98.3% success on LIBERO with 2.1x lower latency than prior fast-slow VLA models by stratifying VLM layer update frequencies, inverting latent interactions, and applying multi-level supervision.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22766","ref_index":51,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"READ More than What You See: Reinforcement Learning for Accurate and Coherent Audio Description Generations","primary_cat":"cs.CV","submitted_at":"2026-06-22T02:05:38+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"READ is the first reinforcement-learning framework for training audio-description generators, using sequence-level rewards for reference match, length, format, and context-aware coherence.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22610","ref_index":9,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement","primary_cat":"cs.AI","submitted_at":"2026-06-21T17:37:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PAPERCLAW is a multi-agent system for end-to-end autonomous research paper generation from literature to output, with human refinement and LLM-judge evaluation showing strong results.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22326","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Adam Converges in Nonsmooth Nonconvex Optimization","primary_cat":"math.OC","submitted_at":"2026-06-21T04:00:35+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"The paper establishes the first finite-time convergence rate of 1/T^{2/13} for classical Adam (with bias correction, no extra steps) in nonsmooth nonconvex optimization under heavy-tailed noise with β1=β2.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22303","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"FlowDPG: Deterministic Policy Gradient on Flow Matching Policies for Real-World Manipulation","primary_cat":"cs.RO","submitted_at":"2026-06-21T02:10:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"FlowDPG distills critic gradients into flow matching velocity fields to enable BPTT-free DDPG-style policy improvement and reports 92% success on a real-world dual-arm AirPods assembly task.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21925","ref_index":52,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PhiBE-Q-Learning: Bridging Off-Policy Reinforcement Learning and Continuous-Time Control","primary_cat":"math.OC","submitted_at":"2026-06-20T07:47:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Introduces a new Q-function definition for continuous-time RL and convergent off-policy algorithms under linear function approximation in model-based and model-free settings.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21734","ref_index":200,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","primary_cat":"cs.CV","submitted_at":"2026-06-19T20:43:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"HPP decouples perception from reasoning in long-video VLMs by having an LLM run iterative programmatic probes on hierarchically segmented video, reporting gains on LongVideoBench, EgoSchema, VideoMME, and MLVU.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21496","ref_index":41,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Decoupling the Declarative from the Procedural in Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-06-19T14:43:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"w²VLA restructures VLA information flow to decouple declarative semantics from procedural skills, enabling zero-shot transfer to novel objects.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21372","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"NAC: Neural Action Codec for Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-06-19T12:24:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"NAC adapts multi-scale RVQGAN audio codecs with kinematic-specific losses to produce ordered action tokens that yield lower reconstruction error and higher task success than prior tokenizers in VLA models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21255","ref_index":42,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SCOPE: Sequential Conformal Probing for Reliable OOD Rejection in LLM Services","primary_cat":"cs.CL","submitted_at":"2026-06-19T09:31:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SCOPE selects readable hidden layers, constructs conformal gates with IND calibration, and uses supermartingale e-processes to certify persistent service-boundary evidence, improving rejection over final-layer detectors across multiple LLMs and boundary conditions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21249","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Does RoPE Prevent or Degrade Retrieval Heads? A Mechanistic Analysis Across Model Families","primary_cat":"cs.LG","submitted_at":"2026-06-19T09:23:59+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Causal head-masking and dimension-zeroing experiments show retrieval heads are necessary for long-context recall and that low-frequency RoPE components within them drive performance across five models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.20310","ref_index":42,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Through the PRISM: Preference Representation in Intermediate States of Video Diffusion Models","primary_cat":"cs.CV","submitted_at":"2026-06-18T14:44:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PRISM shows video diffusion models inherently encode preference information in noisy latents, achieving SOTA accuracy and enabling noise-robust early-stage sampling with a correlation to generative performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":100,"offset":0}}