{
  "version": "https://jsonfeed.org/version/1.1",
  "title": "LLM Inference Signals \u2014 VQV.me",
  "home_page_url": "https://vqv.me/t/llm-inference/",
  "feed_url": "https://vqv.me/t/llm-inference/feed.json",
  "description": "Latest meaningful AI signals tracked under LLM Inference by VQV.me.",
  "language": "en",
  "items": [
    {
      "id": "vqv:signal:mmcRI",
      "url": "https://vqv.me/s/mmcRI/",
      "external_url": "https://jwlabs.vercel.app/post/biting-the-bullet",
      "title": "Discussion on Predictive Speculative KV Replication for Bursty LLM Inference",
      "summary": "A Hacker News discussion with 41 points and 4 comments explores predictive speculative key-value replication techniques to handle bursty large language model inference workloads. The conversation highlights community interest in improving LLM inference efficiency under variable demand.",
      "date_published": "2026-07-31T19:55:41+00:00",
      "date_modified": "2026-07-31T19:55:41+00:00",
      "tags": [
        "LLM Inference",
        "SOURCE-BACKED",
        "TECHNICAL",
        "USEFUL NOW"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "jwlabs.vercel.app",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 79
      }
    },
    {
      "id": "vqv:signal:9zkTd",
      "url": "https://vqv.me/s/9zkTd/",
      "external_url": "https://systems.seas.harvard.edu/blog/burstiness-is-all-you-need/",
      "title": "Bursty Arrivals Can Accelerate LLM Inference",
      "summary": "A Hacker News discussion highlights that bursty input patterns can speed up large language model (LLM) inference. This insight is based on analysis shared in a Harvard systems blog post.",
      "date_published": "2026-07-31T18:35:24+00:00",
      "date_modified": "2026-07-31T18:35:24+00:00",
      "tags": [
        "LLM Inference",
        "SOURCE-BACKED",
        "TECHNICAL",
        "USEFUL NOW"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "systems.seas.harvard.edu",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 79
      }
    },
    {
      "id": "vqv:signal:7xFFV",
      "url": "https://vqv.me/s/7xFFV/",
      "external_url": "https://twitter.com/Akashi203/status/2083213197171884505",
      "title": "Discussion on LLM Inference Costs on Hacker News",
      "summary": "A Hacker News thread discusses the costs associated with large language model (LLM) inference, highlighting two key points but no comments. The conversation reflects early-stage community engagement on this topic.",
      "date_published": "2026-07-31T18:11:45+00:00",
      "date_modified": "2026-07-31T18:11:45+00:00",
      "tags": [
        "LLM Inference",
        "SOURCE-BACKED",
        "TECHNICAL",
        "USEFUL NOW"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "twitter.com",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 79
      }
    },
    {
      "id": "vqv:signal:CslIm",
      "url": "https://vqv.me/s/CslIm/",
      "external_url": "https://localai.io/blog/why-we-write-our-own-engines/",
      "title": "Why we write our own C and C++ inference engines",
      "summary": "Hacker News surfaced this AI signal from localai.io: Why we write our own C and C++ inference engines.",
      "date_published": "2026-07-31T16:17:04+00:00",
      "date_modified": "2026-07-31T16:17:04+00:00",
      "tags": [
        "LLM Inference",
        "SOURCE-BACKED",
        "TECHNICAL",
        "USEFUL NOW"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "localai.io",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 78
      }
    },
    {
      "id": "vqv:signal:bQmTR",
      "url": "https://vqv.me/s/bQmTR/",
      "external_url": "http://arxiv.org/abs/2607.28418v1",
      "title": "WIDE: Adaptive Token-level Dynamic Width Pruning for Efficient LLM Inference",
      "summary": "WIDE introduces token-level dynamic width pruning to improve LLM inference efficiency by adapting computation to individual inputs, addressing accuracy loss in static pruning methods. This approach balances throughput gains with quality retention under aggressive sparsity.",
      "date_published": "2026-07-30T16:01:03+00:00",
      "date_modified": "2026-07-30T16:01:03+00:00",
      "tags": [
        "LLM Inference",
        "ROBOTS & HARDWARE",
        "SOURCE-BACKED",
        "TECHNICAL"
      ],
      "_vqv": {
        "source_name": "arXiv",
        "source_domain": "arxiv.org",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 95
      }
    },
    {
      "id": "vqv:signal:8KeCH",
      "url": "https://vqv.me/s/8KeCH/",
      "external_url": "https://marcobambini.substack.com/p/the-waste-inference-engine",
      "title": "New inference engine runs Kimi K3 2.78T parameter model with 29GB RAM",
      "summary": "A new inference engine has been developed that can run the Kimi K3 model, which has 2.78 trillion parameters, using only 29GB of RAM. This was discussed in a Hacker News thread with several points and comments.",
      "date_published": "2026-07-30T14:01:44+00:00",
      "date_modified": "2026-07-30T14:01:44+00:00",
      "tags": [
        "LLM Inference",
        "SOURCE-BACKED",
        "TECHNICAL",
        "USEFUL NOW"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "marcobambini.substack.com",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 78
      }
    },
    {
      "id": "vqv:signal:iJGrk",
      "url": "https://vqv.me/s/iJGrk/",
      "external_url": "http://arxiv.org/abs/2607.28150v1",
      "title": "SmartGen Enables Efficient Disaggregated LLM Inference with Selective KV Cache Transfer",
      "summary": "SmartGen addresses the challenge of transferring large key-value (KV) caches between disaggregated nodes in LLM inference by enabling selective KV cache transfer. This approach improves performance for self-hosted LLM deployments on rented cloud instances with limited inter-node network bandwidth.",
      "date_published": "2026-07-30T12:56:05+00:00",
      "date_modified": "2026-07-30T12:56:05+00:00",
      "tags": [
        "LLM Inference",
        "RESEARCH",
        "SOURCE-BACKED",
        "TECHNICAL"
      ],
      "_vqv": {
        "source_name": "arXiv",
        "source_domain": "arxiv.org",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 95
      }
    },
    {
      "id": "vqv:signal:qdThR",
      "url": "https://vqv.me/s/qdThR/",
      "external_url": "http://arxiv.org/abs/2607.27704v1",
      "title": "LightRot: Lightweight Rotation Scheme for Efficient Low-Bit LLM Inference",
      "summary": "LightRot introduces a lightweight rotation scheme and dedicated hardware accelerator to improve energy efficiency and accuracy in low-bit large language model inference. It incorporates Grouped Local Rotation (GLR) and Outlier Direction techniques to optimize performance.",
      "date_published": "2026-07-30T05:39:58+00:00",
      "date_modified": "2026-07-30T05:39:58+00:00",
      "tags": [
        "LLM Inference",
        "ROBOTS & HARDWARE",
        "SOURCE-BACKED",
        "TECHNICAL"
      ],
      "_vqv": {
        "source_name": "arXiv",
        "source_domain": "arxiv.org",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 95
      }
    },
    {
      "id": "vqv:signal:5caZg",
      "url": "https://vqv.me/s/5caZg/",
      "external_url": "https://github.com/FedericoTs/quantprobe",
      "title": "Show HN: I run 30B 22tok/s, 109tok/s not novel,6GB/16GB RAM overcoming llama.cpp",
      "summary": "Hacker News surfaced this AI signal from github.com: Show HN: I run 30B 22tok/s, 109tok/s not novel,6GB/16GB RAM overcoming llama.cpp.",
      "date_published": "2026-07-29T03:47:58+00:00",
      "date_modified": "2026-07-29T03:47:58+00:00",
      "tags": [
        "BIG MOVE",
        "LLM Inference",
        "Llama",
        "TECHNICAL",
        "WATCH"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "github.com",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "WATCH",
        "signal_strength": 76
      }
    },
    {
      "id": "vqv:signal:hnzY1",
      "url": "https://vqv.me/s/hnzY1/",
      "external_url": "https://github.com/mraza007/minute",
      "title": "Show HN: Minute \u2013 Offline meeting notes on macOS with Whisper and llama.cpp",
      "summary": "Hacker News surfaced this AI signal from github.com: Show HN: Minute \u2013 Offline meeting notes on macOS with Whisper and llama.cpp.",
      "date_published": "2026-07-28T19:31:17+00:00",
      "date_modified": "2026-07-28T19:31:17+00:00",
      "tags": [
        "BIG MOVE",
        "LLM Inference",
        "Llama",
        "TECHNICAL",
        "WATCH"
      ],
      "_vqv": {
        "source_name": "Hacker News",
        "source_domain": "github.com",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "WATCH",
        "signal_strength": 76
      }
    },
    {
      "id": "vqv:signal:nW39H",
      "url": "https://vqv.me/s/nW39H/",
      "external_url": "http://arxiv.org/abs/2607.24062v1",
      "title": "ACRL addresses training-inference discrepancy in LLM reinforcement learning",
      "summary": "The paper identifies instability in reinforcement learning for LLMs caused by discrepancies between training and inference, due to architectural differences and precision gaps. It proposes Adaptive Control of Training-Inference Discrepancy (ACRL) to stabilize RL training by mitigating these factors.",
      "date_published": "2026-07-27T07:05:10+00:00",
      "date_modified": "2026-07-27T07:05:10+00:00",
      "tags": [
        "LLM Inference",
        "RESEARCH",
        "SOURCE-BACKED",
        "TECHNICAL"
      ],
      "_vqv": {
        "source_name": "arXiv",
        "source_domain": "arxiv.org",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "SOURCE-BACKED",
        "signal_strength": 95
      }
    },
    {
      "id": "vqv:signal:tDcBq",
      "url": "https://vqv.me/s/tDcBq/",
      "external_url": "https://huggingface.co/blog/native-speed-vllm-transformers-backend",
      "title": "Native-speed vLLM transformers modeling backend",
      "summary": "Native-speed vLLM transformers modeling backend",
      "date_published": "2026-07-08T00:00:00+00:00",
      "date_modified": "2026-07-08T00:00:00+00:00",
      "tags": [
        "Hugging Face",
        "LLM Inference",
        "TECHNICAL",
        "WATCH"
      ],
      "_vqv": {
        "source_name": "Hugging Face Blog",
        "source_domain": "huggingface.co",
        "topic": "LLM Inference",
        "topic_slug": "llm-inference",
        "technical_label": "WATCH",
        "signal_strength": 88
      }
    }
  ]
}
