{
  "site": "https://sigpulse.com",
  "license": "CC BY 4.0 — cite the source URL",
  "notice": "Periodically updated translated digests of arXiv papers via Chinese-language commentary — NOT measurements. First-party measured numbers live in /data.json and never here.",
  "count": 12,
  "papers_covered": 146,
  "entries": [
    {
      "slug": "2026-09-12-kv-cache-step-sizes-watts",
      "title": "When the Bottleneck Moves From Weights to Memory and Power: 24 arXiv Papers on the Machinery Layer",
      "description": "KV cache becomes the central inference bottleneck, quantization theory catches up with practice, watts become a design variable — 24 papers.",
      "issue": "Batch 5",
      "tags": [
        "inference systems",
        "KV cache",
        "quantization",
        "optimization",
        "training methods",
        "arXiv digest"
      ],
      "date": "2026-09-12",
      "original_date": "2026-09-12",
      "original_title": "量化为什么没压坏模型？5篇论文拆推理栈 / GD的步长还能挖多久？13篇优化新论文 / Muon还不稳？优化器军备赛的新答卷",
      "papers": [
        {
          "id": "2609.11744",
          "title": "Building py-kvcache: A Performance Characterization of External KV Caching for vLLM with NVMe SSDs",
          "url": "https://arxiv.org/abs/2609.11744"
        },
        {
          "id": "2609.11716",
          "title": "Why Does Post-Training Quantization Work?",
          "url": "https://arxiv.org/abs/2609.11716"
        },
        {
          "id": "2609.11687",
          "title": "Structured Transforms for Low-Overhead Quantization of Language Models",
          "url": "https://arxiv.org/abs/2609.11687"
        },
        {
          "id": "2609.11582",
          "title": "OmniKVQuant: KV Cache Quantization for Omni-LLMs",
          "url": "https://arxiv.org/abs/2609.11582"
        },
        {
          "id": "2609.11542",
          "title": "Characterizing Job Power Elasticity for Power-Flexible AI Training",
          "url": "https://arxiv.org/abs/2609.11542"
        },
        {
          "id": "2609.11829",
          "title": "WFDroneBench: A Benchmark for Sensor Placement and Drone Routing for Wildfire Detection",
          "url": "https://arxiv.org/abs/2609.11829"
        },
        {
          "id": "2609.11817",
          "title": "Constrained Deep Inventory Management Using Forward-Backward SDEs",
          "url": "https://arxiv.org/abs/2609.11817"
        },
        {
          "id": "2609.11788",
          "title": "Optimal Recursive Composition and Dyadic Phase Laws for Gradient Descent with Predetermined Stepsizes",
          "url": "https://arxiv.org/abs/2609.11788"
        },
        {
          "id": "2609.11784",
          "title": "Constant Steps Are s-Composable: An Exact Interpolation Certificate for Gradient Descent",
          "url": "https://arxiv.org/abs/2609.11784"
        },
        {
          "id": "2609.11751",
          "title": "Stochastic Gradient Methods with Online Scaling",
          "url": "https://arxiv.org/abs/2609.11751"
        },
        {
          "id": "2609.11509",
          "title": "Extending SMT Solving with Non-Ground Clause Learning",
          "url": "https://arxiv.org/abs/2609.11509"
        },
        {
          "id": "2609.11339",
          "title": "InFlow: entropic stochastic dual dynamic programming with HJB cross-certification for seasonal energy storage",
          "url": "https://arxiv.org/abs/2609.11339"
        },
        {
          "id": "2609.11273",
          "title": "Minimax-Optimal Early Stopping for Continuous-Time SGD via the Discrepancy Principle",
          "url": "https://arxiv.org/abs/2609.11273"
        },
        {
          "id": "2609.11207",
          "title": "Convex Optimization with Nested Evolving Feasible Sets (CONES) under Time-Varying Loss Functions",
          "url": "https://arxiv.org/abs/2609.11207"
        },
        {
          "id": "2609.11183",
          "title": "Single-Loop Gradient Algorithms for Pessimistic Bilevel Optimization Problems",
          "url": "https://arxiv.org/abs/2609.11183"
        },
        {
          "id": "2609.11105",
          "title": "The best approximation tuple: an extension of the Cheney-Goldstein algorithm and results to the multiple sets case",
          "url": "https://arxiv.org/abs/2609.11105"
        },
        {
          "id": "2609.10866",
          "title": "Certifying Lower Bounds for Risk-Sensitive Reinforcement Learning under Adversarial State Perturbations",
          "url": "https://arxiv.org/abs/2609.10866"
        },
        {
          "id": "2609.10732",
          "title": "Operator Splitting Methods with Online Scaling",
          "url": "https://arxiv.org/abs/2609.10732"
        },
        {
          "id": "2609.11867",
          "title": "AdamX: Cosine similarity meets gradient descent",
          "url": "https://arxiv.org/abs/2609.11867"
        },
        {
          "id": "2609.11801",
          "title": "Thinking with Looped Flows",
          "url": "https://arxiv.org/abs/2609.11801"
        },
        {
          "id": "2609.11739",
          "title": "LOCUS: Task-Aware Low-Rank Post-Training for Token-Efficient Language Generation",
          "url": "https://arxiv.org/abs/2609.11739"
        },
        {
          "id": "2609.11699",
          "title": "Negative Self-Distillation: Learning to Reason by Avoiding Flaws",
          "url": "https://arxiv.org/abs/2609.11699"
        },
        {
          "id": "2609.11655",
          "title": "Musec: MomentUm SpEctral Clipping for Stable Muon-type Training",
          "url": "https://arxiv.org/abs/2609.11655"
        },
        {
          "id": "2609.10980",
          "title": "EGGROLL, Unrolled: Understanding and Improving Low-Rank Evolution Strategies at Scale",
          "url": "https://arxiv.org/abs/2609.10980"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-09-12-kv-cache-step-sizes-watts/",
      "md_url": "https://sigpulse.com/papers/2026-09-12-kv-cache-step-sizes-watts.md"
    },
    {
      "slug": "2026-09-12-martingales-knowledge-frontiers",
      "title": "What Can't Be Tokenized? 31 arXiv Papers on Foundations and Frontiers",
      "description": "An eleven-year conjecture falls, attention gets statistical-mechanical phase diagrams, knowledge placement becomes token economics. 31 papers.",
      "issue": "Batch 6",
      "tags": [
        "learning theory",
        "knowledge management",
        "RAG",
        "model routing",
        "multimodal",
        "arXiv digest"
      ],
      "date": "2026-09-12",
      "original_date": "2026-09-12",
      "original_title": "从鞅到张量：16篇统计学习理论新论文 / 该检索还是该路由？知识栈重构的6篇论文 / 语音补课、拓扑开路：多模态的9道新题",
      "papers": [
        {
          "id": "2609.11918",
          "title": "General Quantification of Covariate and Concept Shifts",
          "url": "https://arxiv.org/abs/2609.11918"
        },
        {
          "id": "2609.11845",
          "title": "$β$-Skewed Maximal Spanning Forests",
          "url": "https://arxiv.org/abs/2609.11845"
        },
        {
          "id": "2609.11807",
          "title": "Near-Optimal Reinforcement Learning with Multi-Step Transition Lookahead",
          "url": "https://arxiv.org/abs/2609.11807"
        },
        {
          "id": "2609.11795",
          "title": "Small-Ball Marginals Do Not Control Restricted Eigenvalues by Euclidean Gaussian Width",
          "url": "https://arxiv.org/abs/2609.11795"
        },
        {
          "id": "2609.11740",
          "title": "From Good Starts to Optimal Inference: Generalized Latent Factor Models with Missingness and Implicit Regularization",
          "url": "https://arxiv.org/abs/2609.11740"
        },
        {
          "id": "2609.11712",
          "title": "Generalization Analysis of Distributed Kernel-based Robust Gradient Descent Algorithms",
          "url": "https://arxiv.org/abs/2609.11712"
        },
        {
          "id": "2609.11606",
          "title": "Identifiability of Nonnegative Tensor Decompositions via Positive Scattering",
          "url": "https://arxiv.org/abs/2609.11606"
        },
        {
          "id": "2609.11557",
          "title": "Martingale central limit theorems in $p$-Wasserstein distance",
          "url": "https://arxiv.org/abs/2609.11557"
        },
        {
          "id": "2609.10976",
          "title": "Phases in a class of associative memories via hidden neurons",
          "url": "https://arxiv.org/abs/2609.10976"
        },
        {
          "id": "2609.10928",
          "title": "AUC Maximization from Biased Positive-unlabeled Data with Confidence",
          "url": "https://arxiv.org/abs/2609.10928"
        },
        {
          "id": "2609.10904",
          "title": "Agnostic Model-Assisted Estimation with Machine Learning for Survey Data",
          "url": "https://arxiv.org/abs/2609.10904"
        },
        {
          "id": "2609.10886",
          "title": "Relatively Smart II: Tractable or Semi-Supervised Instance-Optimal Learning",
          "url": "https://arxiv.org/abs/2609.10886"
        },
        {
          "id": "2609.10879",
          "title": "Learning Orthogonal Multi-Index Models Beyond Small Initialization: Incremental Learning, Competitive Dynamics and Symmetry",
          "url": "https://arxiv.org/abs/2609.10879"
        },
        {
          "id": "2609.10767",
          "title": "Weighted Empirical Risk Minimization for Machine Learning under Long-Range Dependence: Exact Pathwise Rates and Learning-Error Geometry",
          "url": "https://arxiv.org/abs/2609.10767"
        },
        {
          "id": "2609.10729",
          "title": "A Quantum-Inspired Dequantization Method for Diagonally Weighted Matrix Functions: Application to Learning with Optimized Random Features",
          "url": "https://arxiv.org/abs/2609.10729"
        },
        {
          "id": "2609.10534",
          "title": "Likelihood-free inference with nuisance parameters through normalizing flows",
          "url": "https://arxiv.org/abs/2609.10534"
        },
        {
          "id": "2609.11859",
          "title": "From Parameters to Answers: How LLMs Retrieve and Use Their Internal Knowledge",
          "url": "https://arxiv.org/abs/2609.11859"
        },
        {
          "id": "2609.11656",
          "title": "Learnware and AI Model Management System",
          "url": "https://arxiv.org/abs/2609.11656"
        },
        {
          "id": "2609.11569",
          "title": "Enabling Knowledge Graph Understanding at Scale with the EXplore Your Graphs ENgine (EXYGEN)",
          "url": "https://arxiv.org/abs/2609.11569"
        },
        {
          "id": "2609.11414",
          "title": "SWRouter: Similarity-Contractive Window Routing for Multi-Turn Large Language Model Conversations",
          "url": "https://arxiv.org/abs/2609.11414"
        },
        {
          "id": "2609.11393",
          "title": "Beyond Confidence: Stability-Aware Test-Time Adaptation for LLM Reasoning",
          "url": "https://arxiv.org/abs/2609.11393"
        },
        {
          "id": "2609.11390",
          "title": "VikingRAG: Accurate and Token-efficient Retrieval-augmented Generation over Structured Documents",
          "url": "https://arxiv.org/abs/2609.11390"
        },
        {
          "id": "2609.11900",
          "title": "MindTopo: Can Foundation Models Reason in Topological Space?",
          "url": "https://arxiv.org/abs/2609.11900"
        },
        {
          "id": "2609.11899",
          "title": "Caption-once, Frames-on-Demand: Visual-Need Routing for Budget-Aware Agentic Long Video Understanding",
          "url": "https://arxiv.org/abs/2609.11899"
        },
        {
          "id": "2609.11892",
          "title": "Nuha-Speech: Building General-Purpose Arabic Speech-LLMs",
          "url": "https://arxiv.org/abs/2609.11892"
        },
        {
          "id": "2609.11864",
          "title": "RetroThinker: Enabling Retrospective Thinking in Speech LLMs",
          "url": "https://arxiv.org/abs/2609.11864"
        },
        {
          "id": "2609.11772",
          "title": "Whisper-Based Speech Transcription from Videos Across Multiple Languages for Cross-Cultural Understanding",
          "url": "https://arxiv.org/abs/2609.11772"
        },
        {
          "id": "2609.11762",
          "title": "Component-Aware Differential Privacy for Federated Multilingual Speech-LLMs",
          "url": "https://arxiv.org/abs/2609.11762"
        },
        {
          "id": "2609.11708",
          "title": "Language-Augmented Semantic Priors for B-Spline Surface Fitting",
          "url": "https://arxiv.org/abs/2609.11708"
        },
        {
          "id": "2609.11499",
          "title": "Recursive Code World Models: Building Complex Worlds through Recursive Scene Programs",
          "url": "https://arxiv.org/abs/2609.11499"
        },
        {
          "id": "2609.11412",
          "title": "X-AuT: Progressive Audio-Encoder Compression for Speech LLMs with Cross-Scale Distillation",
          "url": "https://arxiv.org/abs/2609.11412"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-09-12-martingales-knowledge-frontiers/",
      "md_url": "https://sigpulse.com/papers/2026-09-12-martingales-knowledge-frontiers.md"
    },
    {
      "slug": "2026-09-12-who-brakes-the-agents",
      "title": "Who Brakes a Continuously Running Agent? 26 arXiv Papers on Control, Grading, and Day Jobs",
      "description": "Control logic becomes a research object, evaluation lags what it measures, and regulated industries hire LLMs as embeddable components. 26 papers.",
      "issue": "Batch 4",
      "tags": [
        "AI agents",
        "agent control",
        "AI safety",
        "evaluation",
        "LLM applications",
        "arXiv digest"
      ],
      "date": "2026-09-12",
      "original_date": "2026-09-12",
      "original_title": "谁来踩智能体的刹车？11篇论文看控制权 / 幻觉检测F1到0.915，就安全了吗？ / LLM上岗7个行业：CRISPR、SEC文件与病历",
      "papers": [
        {
          "id": "2609.11911",
          "title": "Artificial Id: Drive and Persistent Alignment in Agentic AI",
          "url": "https://arxiv.org/abs/2609.11911"
        },
        {
          "id": "2609.11873",
          "title": "The Last AI Built by Humans: Toward Genuine Recursive Self-Improvement",
          "url": "https://arxiv.org/abs/2609.11873"
        },
        {
          "id": "2609.11737",
          "title": "ORCH: Organizational Principles Enable Collective Intelligence in Embodied AI",
          "url": "https://arxiv.org/abs/2609.11737"
        },
        {
          "id": "2609.11709",
          "title": "When Agents Disagree: Bayesian Backward Reasoning as a Label-Free Anchor for Multi-Agent Collective Decision-Making",
          "url": "https://arxiv.org/abs/2609.11709"
        },
        {
          "id": "2609.11682",
          "title": "COBRA-Skills: Contextual Bandit-Guided Evolution for Agent Skill Optimization",
          "url": "https://arxiv.org/abs/2609.11682"
        },
        {
          "id": "2609.11677",
          "title": "Ecdysis: Efficient and Effective Training of Runtime Harnesses for LLM Agents",
          "url": "https://arxiv.org/abs/2609.11677"
        },
        {
          "id": "2609.11660",
          "title": "Autonomy, Social Norms, and Alignment: Towards a Developmental Framework for Autonomous Artificial Agents",
          "url": "https://arxiv.org/abs/2609.11660"
        },
        {
          "id": "2609.11636",
          "title": "MAPLE: Memory-Augmented Planning with Language and Evolution",
          "url": "https://arxiv.org/abs/2609.11636"
        },
        {
          "id": "2609.11489",
          "title": "The Convention Gap: Towards Measuring Implicit Communication in Cooperative AI Evaluation",
          "url": "https://arxiv.org/abs/2609.11489"
        },
        {
          "id": "2609.11452",
          "title": "RouteRepair: Instance-Level Failure Diagnosis and Targeted Repair in LLM-Based Automated Heuristic Design for Routing Optimization",
          "url": "https://arxiv.org/abs/2609.11452"
        },
        {
          "id": "2609.11381",
          "title": "Agent-Integrated Software: Interaction Contracts and Continuous Assurance",
          "url": "https://arxiv.org/abs/2609.11381"
        },
        {
          "id": "2609.11897",
          "title": "CausalArena: Benchmarking Causal Discovery in the Foundation Model Era",
          "url": "https://arxiv.org/abs/2609.11897"
        },
        {
          "id": "2609.11878",
          "title": "Domain-Specific Hallucination Detection in Large Language Models",
          "url": "https://arxiv.org/abs/2609.11878"
        },
        {
          "id": "2609.11799",
          "title": "SpecGuard: Inference-Time Backdoor Detection For Free",
          "url": "https://arxiv.org/abs/2609.11799"
        },
        {
          "id": "2609.11770",
          "title": "The widening evaluation gap in medical large language model research 2023 to 2026",
          "url": "https://arxiv.org/abs/2609.11770"
        },
        {
          "id": "2609.11769",
          "title": "Recognizing Is Not Reversing: A Controlled Inversion Test of Fact-Preserving News Framing",
          "url": "https://arxiv.org/abs/2609.11769"
        },
        {
          "id": "2609.11758",
          "title": "RAG-Safety-Bench: Reliable Evaluation of Retrieval-Augmented LLM Safety",
          "url": "https://arxiv.org/abs/2609.11758"
        },
        {
          "id": "2609.11498",
          "title": "ActMap: Single-Pass Uncertainty Quantification from Generation-Time Activation Maps",
          "url": "https://arxiv.org/abs/2609.11498"
        },
        {
          "id": "2609.11399",
          "title": "TransClean: A Benchmark for Detecting and Extracting Clean Translations from Large Language Model Outputs",
          "url": "https://arxiv.org/abs/2609.11399"
        },
        {
          "id": "2609.11877",
          "title": "Biology-in-the-loop: Amortized Adaptive Hit Discovery in CRISPR Screens",
          "url": "https://arxiv.org/abs/2609.11877"
        },
        {
          "id": "2609.11860",
          "title": "Explainability Assistant: A Conversational XAI Interface for Interpreting Energy Consumption Models",
          "url": "https://arxiv.org/abs/2609.11860"
        },
        {
          "id": "2609.11620",
          "title": "A Training-Free, Alignment-Free Approach to Corporate Intelligence: Application to SEC Filings",
          "url": "https://arxiv.org/abs/2609.11620"
        },
        {
          "id": "2609.11607",
          "title": "Making Alternative Data Work: Context-Augmented LLMs for Financial Forecasting",
          "url": "https://arxiv.org/abs/2609.11607"
        },
        {
          "id": "2609.11493",
          "title": "From Document Silos to Process Intelligence: A Multi-Layer Knowledge Graph for CMC Process Development",
          "url": "https://arxiv.org/abs/2609.11493"
        },
        {
          "id": "2609.11450",
          "title": "Cross-Lingual Clinical Annotation Projection as Constrained Text Generation: A Six-Language Study",
          "url": "https://arxiv.org/abs/2609.11450"
        },
        {
          "id": "2609.11431",
          "title": "LLMs as Post-hoc Auditors of Physiological Plausibility in Symbolic Regression: A Clinician-Evaluated Case Study",
          "url": "https://arxiv.org/abs/2609.11431"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-09-12-who-brakes-the-agents/",
      "md_url": "https://sigpulse.com/papers/2026-09-12-who-brakes-the-agents.md"
    },
    {
      "slug": "2026-09-04-cheaper-deeper-hired",
      "title": "Cheaper, Deeper, Hired: One Day of arXiv on Training Economics, the Math Underneath, and AI's Eight New Jobs",
      "description": "Distillation from a single training example, lower bounds on 'free' acceleration, a sovereign banking model, no-training kidney screening — one arXiv day.",
      "issue": "Batch 3",
      "tags": [
        "training efficiency",
        "distillation",
        "learning theory",
        "LLM applications",
        "arXiv digest"
      ],
      "date": "2026-09-04",
      "original_date": "2026-09-04",
      "original_title": "一条数据能训出大模型？先学后练、4bit瘦身的新学问 / 看不懂的论文有什么用？AI的地基就埋在今天这17篇里 / 不训练模型就早筛肾病？LLM已潜入8个行业上班",
      "papers": [
        {
          "id": "2609.04172",
          "title": "Rethinking On-Policy Distillation of Large Language Models II: One Training Example",
          "url": "https://arxiv.org/abs/2609.04172"
        },
        {
          "id": "2609.04108",
          "title": "Sequential Beats Joint: On the Interplay between On-Policy Distillation and RLVR",
          "url": "https://arxiv.org/abs/2609.04108"
        },
        {
          "id": "2609.04180",
          "title": "Knowledge Acquisition During Pre-training? Large Language Models Learn Better With Auxiliary Views",
          "url": "https://arxiv.org/abs/2609.04180"
        },
        {
          "id": "2609.04063",
          "title": "Spurious Advantage Hidden in GRPO",
          "url": "https://arxiv.org/abs/2609.04063"
        },
        {
          "id": "2609.04098",
          "title": "Why Gated DeltaNet Survives 4-Bit Quantization: NVFP4 W4A4 for the Recurrent Half of a Hybrid 27B LLM",
          "url": "https://arxiv.org/abs/2609.04098"
        },
        {
          "id": "2609.04010",
          "title": "Unlocking Lossless Speedups in LLMs via Discrete Diffusion",
          "url": "https://arxiv.org/abs/2609.04010"
        },
        {
          "id": "2609.04032",
          "title": "Stronger Lower Bounds for (Non-)Anytime Acceleration of Gradient Descent",
          "url": "https://arxiv.org/abs/2609.04032"
        },
        {
          "id": "2609.03626",
          "title": "Residual neural networks overcome the curse of dimensionality for semilinear heat equations",
          "url": "https://arxiv.org/abs/2609.03626"
        },
        {
          "id": "2609.03846",
          "title": "EF1-Constrained Nash Social Welfare with Identical Additive Valuations: Complexity, Guarantees, and Experiments",
          "url": "https://arxiv.org/abs/2609.03846"
        },
        {
          "id": "2609.04189",
          "title": "Robust PAC Learning of Concurrent Stochastic Games",
          "url": "https://arxiv.org/abs/2609.04189"
        },
        {
          "id": "2609.03858",
          "title": "High-Dimensional Learning Dynamics of Attention-Indexed Models",
          "url": "https://arxiv.org/abs/2609.03858"
        },
        {
          "id": "2609.04013",
          "title": "LLM4CKD: Large Language Models for Early Stage Chronic Kidney Disease Screening",
          "url": "https://arxiv.org/abs/2609.04013"
        },
        {
          "id": "2609.03960",
          "title": "FiMI Banking: A Sovereign Model for Indian Retail Banking",
          "url": "https://arxiv.org/abs/2609.03960"
        },
        {
          "id": "2609.03967",
          "title": "Investigating the Ability of Large Language Models to Analyze Recipes for Diabetes",
          "url": "https://arxiv.org/abs/2609.03967"
        },
        {
          "id": "2609.04070",
          "title": "Continuous Actions from Discrete Minds: Latent-Aligned Planning for End-to-End Autonomous Driving",
          "url": "https://arxiv.org/abs/2609.04070"
        },
        {
          "id": "2609.04030",
          "title": "IRWOZ 2.0: A Large Language Model-driven Dialogue Dataset for Industrial Robot Conversations",
          "url": "https://arxiv.org/abs/2609.04030"
        },
        {
          "id": "2609.03871",
          "title": "Bioinfoysis Technical Report",
          "url": "https://arxiv.org/abs/2609.03871"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-09-04-cheaper-deeper-hired/",
      "md_url": "https://sigpulse.com/papers/2026-09-04-cheaper-deeper-hired.md"
    },
    {
      "slug": "2026-09-04-trust-papers-cheating-judge-hooks",
      "title": "100 AI Researchers, Zero Humans: Cheating and Whistleblowing Emerged Unscripted — the Trust Papers of Sept 4",
      "description": "A 100-agent lab where cheating and whistleblowing emerged unscripted, LLM judges with 0.40 ranking consistency, hook updates as attack surface — Sept 4's arXiv.",
      "issue": "Batch 3",
      "tags": [
        "multi-agent systems",
        "LLM as judge",
        "AI security",
        "evaluation",
        "arXiv digest"
      ],
      "date": "2026-09-04",
      "original_date": "2026-09-04",
      "original_title": "100个AI一起搞科研会怎样？作弊自己冒出来，吹哨的也是AI / AI考官集体翻车：5.3万次审计，排名一致性只有0.40 / AI助手会在你背后执行命令？新暗门藏在生命周期钩子里",
      "papers": [
        {
          "id": "2609.04170",
          "title": "A Case Study on Emergent Cheating and Whistleblowing in Autonomous Research Swarms",
          "url": "https://arxiv.org/abs/2609.04170"
        },
        {
          "id": "2609.04198",
          "title": "Clean Engineering, Unstable Measurement: A Preregistered Reliability Failure of Black-Box LLM Observers on Shared Endpoints",
          "url": "https://arxiv.org/abs/2609.04198"
        },
        {
          "id": "2609.04194",
          "title": "Legibility is Not Interpretability: Comparing Judged and Actual Importance in Chain-Of-Thought Reasoning",
          "url": "https://arxiv.org/abs/2609.04194"
        },
        {
          "id": "2609.03966",
          "title": "Interface-Induced Trajectory Censoring",
          "url": "https://arxiv.org/abs/2609.03966"
        },
        {
          "id": "2609.04047",
          "title": "The Dice Roll Method: A Standardized Protocol for Repeated-Query Auditing of LLM Brand Recommendations",
          "url": "https://arxiv.org/abs/2609.04047"
        },
        {
          "id": "2609.03953",
          "title": "Beyond Majority Vote: Multi-Perspective Adjudication for Medical Hallucination Detection",
          "url": "https://arxiv.org/abs/2609.03953"
        },
        {
          "id": "2609.03884",
          "title": "A Blind Trust, the Bloody Thrust: When Attacker-Controlled Hook Updates Steer AI Agent Harnesses towards Malicious Behaviors",
          "url": "https://arxiv.org/abs/2609.03884"
        },
        {
          "id": "2609.03887",
          "title": "Beyond Shallow Alignment: How Post-Training Methods Determine Refusal Circuits And Steering Robustness",
          "url": "https://arxiv.org/abs/2609.03887"
        },
        {
          "id": "2609.04022",
          "title": "Representational alignment yields generalizable safety in language models",
          "url": "https://arxiv.org/abs/2609.04022"
        },
        {
          "id": "2609.03322",
          "title": "How Perturbations Propagate: A Multi-Level Analysis of Robustness in Large Language Models",
          "url": "https://arxiv.org/abs/2609.03322"
        },
        {
          "id": "2609.03844",
          "title": "Flip, Don't Shuffle: Watermarking LLMs at the Speed of Inference",
          "url": "https://arxiv.org/abs/2609.03844"
        },
        {
          "id": "2609.04127",
          "title": "Epistemic Warrant for LLM Recommendations: Characterizing the Basis for Reliance When Ground Truth Is Unavailable",
          "url": "https://arxiv.org/abs/2609.04127"
        },
        {
          "id": "2609.04148",
          "title": "Terminal-Universe: Turning Agent Trajectories into Scalable Terminal Environments",
          "url": "https://arxiv.org/abs/2609.04148"
        },
        {
          "id": "2609.04128",
          "title": "Environment Evolution for Terminal Agents",
          "url": "https://arxiv.org/abs/2609.04128"
        },
        {
          "id": "2609.03787",
          "title": "DNative-Twin: Decision Graphs and Digital Twins for Reconstructable Agentic Decisions",
          "url": "https://arxiv.org/abs/2609.03787"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-09-04-trust-papers-cheating-judge-hooks/",
      "md_url": "https://sigpulse.com/papers/2026-09-04-trust-papers-cheating-judge-hooks.md"
    },
    {
      "slug": "2026-08-30-eighty-six-papers-four-signals",
      "title": "What Did 86 arXiv Papers in 48 Hours Say About Where AI Is Heading? Four Signals",
      "description": "Self-improvement fragility, hospitals hiring AI as a formatter, judges without stable values, and everyone wanting to send AI on dates — 86 papers, 4 signals.",
      "issue": "Batch 2",
      "tags": [
        "self-improving agents",
        "AI safety",
        "evaluation",
        "AI in medicine",
        "arXiv digest"
      ],
      "date": "2026-08-30",
      "original_date": "2026-08-20",
      "original_title": "AI越自学越强，也越容易被黑？86篇新论文里的4个信号",
      "papers": [
        {
          "id": "2608.18066",
          "title": "On the Fragility of Self-Improving Agents (CMU/UCSD replication)",
          "url": "https://arxiv.org/abs/2608.18066"
        },
        {
          "id": "2608.18027",
          "title": "Chain-of-Experience: continual improvement at inference time",
          "url": "https://arxiv.org/abs/2608.18027"
        },
        {
          "id": "2608.17684",
          "title": "Auditing self-evolving financial agents",
          "url": "https://arxiv.org/abs/2608.17684"
        },
        {
          "id": "2608.18072",
          "title": "Radiology report structuring and quality assurance (638 CT reports)",
          "url": "https://arxiv.org/abs/2608.18072"
        },
        {
          "id": "2608.18017",
          "title": "Explaining flight-safety events down to pilot actions",
          "url": "https://arxiv.org/abs/2608.18017"
        },
        {
          "id": "2608.17644",
          "title": "Inconsistency of LLM numeric preference judgments",
          "url": "https://arxiv.org/abs/2608.17644"
        },
        {
          "id": "2608.17938",
          "title": "Grading needs rubrics, not intelligence",
          "url": "https://arxiv.org/abs/2608.17938"
        },
        {
          "id": "2608.18058",
          "title": "Delegation asymmetry on a dating platform (two surveys, 5,000+)",
          "url": "https://arxiv.org/abs/2608.18058"
        },
        {
          "id": "2608.17665",
          "title": "GraphWake: memory-mediated polarization cascades in agent communities",
          "url": "https://arxiv.org/abs/2608.17665"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-08-30-eighty-six-papers-four-signals/",
      "md_url": "https://sigpulse.com/papers/2026-08-30-eighty-six-papers-four-signals.md"
    },
    {
      "slug": "2026-08-30-mimir-permissible-data-1b",
      "title": "Can a 1B Model Trained Only on Legal Data Compete? Denmark's Mimir v1 Says Yes",
      "description": "One billion parameters, 161 permissible datasets, no copyright gray zones — and state-of-the-art Danish plus near-parity with 4B models on English.",
      "issue": "Batch 2",
      "tags": [
        "small models",
        "permissible data",
        "HRM architecture",
        "open weights",
        "low-resource languages"
      ],
      "date": "2026-08-30",
      "original_date": "2026-08-15",
      "original_title": "1B小模型打赢大模型？丹麦人只喂\"干净合法\"的数据",
      "papers": [
        {
          "id": "2608.13517",
          "title": "DFM Mimir v1: An Open HRM Delivering Frontier Performance at 1B Parameters Using Only Permissible Post-Training Data",
          "url": "https://arxiv.org/abs/2608.13517"
        }
      ],
      "sources": [
        {
          "label": "Hugging Face: danish-foundation-models/DFM-Mimir (open weights)",
          "url": "https://huggingface.co/danish-foundation-models/DFM-Mimir"
        },
        {
          "label": "Danish Foundation Models: Mimir v1 release note",
          "url": "https://www.foundationmodels.dk/news/2026/08/14/mimir-1-release-note.html"
        }
      ],
      "url": "https://sigpulse.com/papers/2026-08-30-mimir-permissible-data-1b/",
      "md_url": "https://sigpulse.com/papers/2026-08-30-mimir-permissible-data-1b.md"
    },
    {
      "slug": "2026-08-30-seventy-more-papers-self-exam",
      "title": "AI Writes Its Own Exam Questions and Grades Its Own Full Marks — Who Audits It? The Next 70 Papers",
      "description": "Hidden multi-agent coordination, self-play environments that train execution but lock strategy, verifiable abstention in sewers and dentistry — the second 70.",
      "issue": "Batch 2",
      "tags": [
        "multi-agent systems",
        "self-play",
        "abstention",
        "distributed inference",
        "arXiv digest"
      ],
      "date": "2026-08-30",
      "original_date": "2026-08-20",
      "original_title": "AI自己出题、自己判满分，谁管得住它？",
      "papers": [
        {
          "id": "2608.19161",
          "title": "Detecting covert coordination among agents via hidden states",
          "url": "https://arxiv.org/abs/2608.19161"
        },
        {
          "id": "2608.18795",
          "title": "Decomposing self-consistency error consensus (GPT-4.1 case study)",
          "url": "https://arxiv.org/abs/2608.18795"
        },
        {
          "id": "2608.19197",
          "title": "SPADE: self-play in adaptively synthesized environments",
          "url": "https://arxiv.org/abs/2608.19197"
        },
        {
          "id": "2608.19072",
          "title": "What AI-for-AI training lacks: early strategy lock-in",
          "url": "https://arxiv.org/abs/2608.19072"
        },
        {
          "id": "2608.18836",
          "title": "Verifiable abstention for water-network leak localization",
          "url": "https://arxiv.org/abs/2608.18836"
        },
        {
          "id": "2608.18878",
          "title": "DentAgent: evidence-traceable dental diagnosis",
          "url": "https://arxiv.org/abs/2608.18878"
        },
        {
          "id": "2608.18726",
          "title": "Atmospheric-science benchmark: multiple choice inflates accuracy ≥12 points",
          "url": "https://arxiv.org/abs/2608.18726"
        },
        {
          "id": "2608.19083",
          "title": "Fluency vs source-content preservation in AI translation (N=306)",
          "url": "https://arxiv.org/abs/2608.19083"
        },
        {
          "id": "2608.19147",
          "title": "Distributed LLM inference across idle office AI PCs (Intel)",
          "url": "https://arxiv.org/abs/2608.19147"
        },
        {
          "id": "2608.19009",
          "title": "L0–L5 levels of verification autonomy",
          "url": "https://arxiv.org/abs/2608.19009"
        },
        {
          "id": "2608.19125",
          "title": "Corrections must be managed like configuration: versioned, monitored, retired",
          "url": "https://arxiv.org/abs/2608.19125"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-08-30-seventy-more-papers-self-exam/",
      "md_url": "https://sigpulse.com/papers/2026-08-30-seventy-more-papers-self-exam.md"
    },
    {
      "slug": "2026-08-30-vtoken-vram-virtualization",
      "title": "What Actually Makes LLM Serving Expensive? VRAM — and vToken Brings OS-Style Virtual Memory to the KV Cache",
      "description": "The KV cache fills your VRAM while the GPU waits; vToken virtualizes reclamation down to single tokens, like paging did for operating systems.",
      "issue": "Batch 2",
      "tags": [
        "KV cache",
        "inference economics",
        "VRAM",
        "serving",
        "systems"
      ],
      "date": "2026-08-30",
      "original_date": "2026-08-15",
      "original_title": "大模型贵在哪？显存。有人给它装了套\"虚拟内存\"",
      "papers": [
        {
          "id": "2608.13263",
          "title": "vToken: Token-Level Virtualization for Reclaimable KV Caches",
          "url": "https://arxiv.org/abs/2608.13263"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-08-30-vtoken-vram-virtualization/",
      "md_url": "https://sigpulse.com/papers/2026-08-30-vtoken-vram-virtualization.md"
    },
    {
      "slug": "2026-08-29-agents-that-write-reviews",
      "title": "Can AI Agents Learn From Experience Now? Wikis, Red Teams, and Failure Mining — the Papers Read Inside China",
      "description": "Agent experience compiled into evolving wikis, red-team agents that learn from attacks, small-model failures tutoring big models — plus Google's ReasoningBank.",
      "issue": "Batch 1",
      "tags": [
        "agents",
        "memory",
        "red-teaming",
        "experience reuse",
        "weak-to-strong"
      ],
      "date": "2026-08-29",
      "original_date": "2026-08-29",
      "original_title": "AI学会复盘了，经验还值几年钱？",
      "papers": [
        {
          "id": "2608.27454",
          "title": "WikiSkill: Compiling Agent Experience into Persistent Knowledge for Skill Evolution",
          "url": "https://arxiv.org/abs/2608.27454"
        },
        {
          "id": "2608.27439",
          "title": "RedEvoAgent: Automatic Red-Teaming Agent with Experience-Driven Skill Evolution",
          "url": "https://arxiv.org/abs/2608.27439"
        },
        {
          "id": "2608.27455",
          "title": "CritICL: Inference-Time Weak-to-Strong Generalization from Small Language Model Failure Modes",
          "url": "https://arxiv.org/abs/2608.27455"
        }
      ],
      "sources": [
        {
          "label": "Google Research blog (ReasoningBank — agents learning from experience; project page linked from the blog index)",
          "url": "https://research.google/blog/"
        }
      ],
      "url": "https://sigpulse.com/papers/2026-08-29-agents-that-write-reviews/",
      "md_url": "https://sigpulse.com/papers/2026-08-29-agents-that-write-reviews.md"
    },
    {
      "slug": "2026-08-29-benchmarks-move-to-the-road-test",
      "title": "Are Static AI Benchmarks Dying? Digital Cities, Multi-Round Code Review, and AI-Built Test Tracks — the Papers Read Inside China",
      "description": "A digital Hong Kong for agent exams, defect-aware code review, world models as zero-shot simulators — and who owns the exam hall.",
      "issue": "Batch 1",
      "tags": [
        "benchmarks",
        "evaluation",
        "world models",
        "agents",
        "auditability"
      ],
      "date": "2026-08-29",
      "original_date": "2026-08-29",
      "original_title": "AI的路考来了：谁出题，谁监考，谁说了算",
      "papers": [
        {
          "id": "2608.27456",
          "title": "UrbanGround: From Local Perception to Spatial Agency in a Real-Scale City",
          "url": "https://arxiv.org/abs/2608.27456"
        },
        {
          "id": "2608.27442",
          "title": "From Static to Dynamic: Benchmarking Real-World Code Review with MCR-Bench",
          "url": "https://arxiv.org/abs/2608.27442"
        },
        {
          "id": "2608.27406",
          "title": "CLAP: Cross-Embodiment Video World Models are Zero-Shot Physical Simulators",
          "url": "https://arxiv.org/abs/2608.27406"
        },
        {
          "id": "2608.27427",
          "title": "Persona-Execution Separation for Governed Agent Auditing",
          "url": "https://arxiv.org/abs/2608.27427"
        }
      ],
      "sources": [],
      "url": "https://sigpulse.com/papers/2026-08-29-benchmarks-move-to-the-road-test/",
      "md_url": "https://sigpulse.com/papers/2026-08-29-benchmarks-move-to-the-road-test.md"
    },
    {
      "slug": "2026-08-29-less-is-more-four-papers",
      "title": "Do New arXiv Papers Really Show That Less Training Data Beats More? Four August Papers, Read Inside China",
      "description": "SWE-Prime's 10% subsets beating full data, weak models rescuing strong ones, label-free test-time optimization, 2% of attention heads — 'less is more'.",
      "issue": "Batch 1",
      "tags": [
        "data selection",
        "RLVR",
        "test-time optimization",
        "interpretability",
        "training efficiency"
      ],
      "date": "2026-08-29",
      "original_date": "2026-08-29",
      "original_title": "烧钱堆算力过时了？AI圈流行起\"少即是多\"",
      "papers": [
        {
          "id": "2608.27449",
          "title": "SWE-Prime: Fewer Trajectories, Better Performance",
          "url": "https://arxiv.org/abs/2608.27449"
        },
        {
          "id": "2608.27420",
          "title": "Boosting LLM Exploration via Weak-Model Guidance in RLVR",
          "url": "https://arxiv.org/abs/2608.27420"
        },
        {
          "id": "2608.27448",
          "title": "TTPO: Test-Time Policy Optimization",
          "url": "https://arxiv.org/abs/2608.27448"
        },
        {
          "id": "2608.27417",
          "title": "Retrieval Heads Meet Vision: Uncovering How VLMs Locate and Extract Visual Information",
          "url": "https://arxiv.org/abs/2608.27417"
        }
      ],
      "sources": [
        {
          "label": "Deloitte TMT Predictions (the counterpoint the take cites: compute demand keeps rising)",
          "url": "https://www.deloitte.com/us/en/insights/industry/technology/technology-media-telecom-predictions.html"
        }
      ],
      "url": "https://sigpulse.com/papers/2026-08-29-less-is-more-four-papers/",
      "md_url": "https://sigpulse.com/papers/2026-08-29-less-is-more-four-papers.md"
    }
  ]
}