<?xml version="1.0" encoding="UTF-8" ?>
<rss version="2.0">
  <channel>
    <title>Freedom.Tech: llama.cpp</title>
    <link>https://freedom.tech/project/llama-cpp/</link>
    <description>Every llama.cpp release and update tracked by Freedom.Tech.</description>
    <language>en</language>
    <copyright>Freedom.Tech is a Foundation project. https://foundation.xyz</copyright>
    <generator>Freedom.Tech, a Foundation project (https://foundation.xyz)</generator>
    <item>
      <title>llama.cpp b10644</title>
      <link>https://freedom.tech/posts/2026-08-27-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-27-llama-cpp/</guid>
      <pubDate>Thu, 27 Aug 2026 05:22:38 GMT</pubDate>
      <description>Adds nanbeige4.2-3B model support</description>
    </item>
    <item>
      <title>llama.cpp b10636</title>
      <link>https://freedom.tech/posts/2026-08-26-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-26-llama-cpp/</guid>
      <pubDate>Wed, 26 Aug 2026 12:44:43 GMT</pubDate>
      <description>Ui: disable the npm UI build (LLAMA BUILD UI=OFF)</description>
    </item>
    <item>
      <title>llama.cpp b10625</title>
      <link>https://freedom.tech/posts/2026-08-25-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-25-llama-cpp/</guid>
      <pubDate>Tue, 25 Aug 2026 17:26:21 GMT</pubDate>
      <description>Scopes Qwen3-coder chat workarounds</description>
    </item>
    <item>
      <title>llama.cpp 0.3.0</title>
      <link>https://freedom.tech/posts/2026-08-25-llama-cpp-0-3-0/</link>
      <guid>https://freedom.tech/posts/2026-08-25-llama-cpp-0-3-0/</guid>
      <pubDate>Tue, 25 Aug 2026 10:22:58 GMT</pubDate>
      <description>Llama.cpp 0.3.0 adds dots3-note multimodal support, tensor-split for DeepSeek 4, multi-token prediction for GLM-4.5-Air, and WebP image decoding</description>
    </item>
    <item>
      <title>llama.cpp b10610</title>
      <link>https://freedom.tech/posts/2026-08-24-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-24-llama-cpp/</guid>
      <pubDate>Mon, 24 Aug 2026 13:16:46 GMT</pubDate>
      <description>Metal flash-attention tuning: 53 new f16 instantiations, per-device dispatch tables</description>
    </item>
    <item>
      <title>llama.cpp b10603</title>
      <link>https://freedom.tech/posts/2026-08-23-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-23-llama-cpp/</guid>
      <pubDate>Sun, 23 Aug 2026 18:45:59 GMT</pubDate>
      <description>GLM-4.5-Air MTP support added to llama.cpp for improved model inference efficiency</description>
    </item>
    <item>
      <title>llama.cpp b10588</title>
      <link>https://freedom.tech/posts/2026-08-22-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-22-llama-cpp/</guid>
      <pubDate>Sat, 22 Aug 2026 23:40:48 GMT</pubDate>
      <description>Fixes clang LTO build issue</description>
    </item>
    <item>
      <title>llama.cpp b10568</title>
      <link>https://freedom.tech/posts/2026-08-21-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-21-llama-cpp/</guid>
      <pubDate>Fri, 21 Aug 2026 22:56:01 GMT</pubDate>
      <description>CI switches to LLVM OpenMP on Windows; Drops non-redist debug build</description>
    </item>
    <item>
      <title>llama.cpp 0.2.0</title>
      <link>https://freedom.tech/posts/2026-08-21-llama-cpp-0-2-0/</link>
      <guid>https://freedom.tech/posts/2026-08-21-llama-cpp-0-2-0/</guid>
      <pubDate>Fri, 21 Aug 2026 18:32:48 GMT</pubDate>
      <description>Llama.cpp 0.2.0 syncs ggml to 0.21.0, adds backend optimizations for CUDA/Metal/SYCL/OpenCL, and improves quantization kernel performance across hardware</description>
    </item>
    <item>
      <title>llama.cpp b10505</title>
      <link>https://freedom.tech/posts/2026-08-20-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-20-llama-cpp/</guid>
      <pubDate>Thu, 20 Aug 2026 02:36:34 GMT</pubDate>
      <description>Adds dedup-cache-models preset option to server</description>
    </item>
    <item>
      <title>llama.cpp b10486</title>
      <link>https://freedom.tech/posts/2026-08-18-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-18-llama-cpp/</guid>
      <pubDate>Tue, 18 Aug 2026 10:43:36 GMT</pubDate>
      <description>Fixes LFM2 image tiling threshold; Cross-platform refactor</description>
    </item>
    <item>
      <title>llama.cpp 0.1.2</title>
      <link>https://freedom.tech/posts/2026-08-18-llama-cpp-0-1-2/</link>
      <guid>https://freedom.tech/posts/2026-08-18-llama-cpp-0-1-2/</guid>
      <pubDate>Tue, 18 Aug 2026 10:23:10 GMT</pubDate>
      <description>Llama.cpp synced GGML 0.20.2, added integer tokenizer scores, improved CUDA performance on DGX, and fixed xcframework builds</description>
    </item>
    <item>
      <title>llama.cpp b10470</title>
      <link>https://freedom.tech/posts/2026-08-17-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-17-llama-cpp/</guid>
      <pubDate>Mon, 17 Aug 2026 13:59:43 GMT</pubDate>
      <description>CI; Explicitly pushes release tag before creating GitHub release</description>
    </item>
    <item>
      <title>llama.cpp b10452</title>
      <link>https://freedom.tech/posts/2026-08-16-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-16-llama-cpp/</guid>
      <pubDate>Sun, 16 Aug 2026 11:21:52 GMT</pubDate>
      <description>Removes some ggml concat calls; Model-layer refactor</description>
    </item>
    <item>
      <title>llama.cpp b10447</title>
      <link>https://freedom.tech/posts/2026-08-15-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-15-llama-cpp/</guid>
      <pubDate>Sat, 15 Aug 2026 20:14:19 GMT</pubDate>
      <description>Adds Kimi-K3 text-model support with hybrid KDA and MLA attention</description>
    </item>
    <item>
      <title>llama.cpp b10434</title>
      <link>https://freedom.tech/posts/2026-08-14-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-14-llama-cpp/</guid>
      <pubDate>Fri, 14 Aug 2026 19:22:14 GMT</pubDate>
      <description>Llama.cpp now passes reasoning_effort parameter through chat templates, enabling fine-grained control over model reasoning depth in inference</description>
    </item>
    <item>
      <title>llama.cpp b10419</title>
      <link>https://freedom.tech/posts/2026-08-13-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-13-llama-cpp/</guid>
      <pubDate>Thu, 13 Aug 2026 22:11:34 GMT</pubDate>
      <description>OpenVINO backend gains Qwen3.5 support, memory optimizations for GPU inference, and fixes for stateful RoPE accuracy and recurrent state handling</description>
    </item>
    <item>
      <title>llama.cpp b10369</title>
      <link>https://freedom.tech/posts/2026-08-12-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-12-llama-cpp/</guid>
      <pubDate>Wed, 12 Aug 2026 04:52:57 GMT</pubDate>
      <description>Build b10375 tightens bare function parsing for Qwen models</description>
    </item>
    <item>
      <title>llama.cpp b10359</title>
      <link>https://freedom.tech/posts/2026-08-11-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-11-llama-cpp/</guid>
      <pubDate>Tue, 11 Aug 2026 10:48:48 GMT</pubDate>
      <description>Switches ROCm target to 7.14, the first production release using TheRock build system</description>
    </item>
    <item>
      <title>llama.cpp b10344</title>
      <link>https://freedom.tech/posts/2026-08-10-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-10-llama-cpp/</guid>
      <pubDate>Mon, 10 Aug 2026 16:24:12 GMT</pubDate>
      <description>Added MTP (Multi-Token Prediction) support for Nemotron models, enabling faster inference for this architecture</description>
    </item>
    <item>
      <title>llama.cpp b10329</title>
      <link>https://freedom.tech/posts/2026-08-08-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-08-llama-cpp/</guid>
      <pubDate>Sat, 08 Aug 2026 16:00:17 GMT</pubDate>
      <description>Corrects server info endpoint to report the isolate working directory when a tools runtime is configured</description>
    </item>
    <item>
      <title>llama.cpp b10326</title>
      <link>https://freedom.tech/posts/2026-08-07-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-07-llama-cpp/</guid>
      <pubDate>Fri, 07 Aug 2026 21:23:15 GMT</pubDate>
      <description>TTS timing now accounts for full vocoder pass instead of single trailing window</description>
    </item>
    <item>
      <title>llama.cpp b10290</title>
      <link>https://freedom.tech/posts/2026-08-06-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-06-llama-cpp/</guid>
      <pubDate>Thu, 06 Aug 2026 01:33:36 GMT</pubDate>
      <description>AMD ROCm CI support for gfx1151; No inference changes</description>
    </item>
    <item>
      <title>llama.cpp b10289</title>
      <link>https://freedom.tech/posts/2026-08-05-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-05-llama-cpp/</guid>
      <pubDate>Wed, 05 Aug 2026 20:02:56 GMT</pubDate>
      <description>Fixed directory traversal vulnerabilities in file search, improved path handling on Windows with UTF-8 conversion, and added cache expiration for UI picker searches</description>
    </item>
    <item>
      <title>llama.cpp b10271</title>
      <link>https://freedom.tech/posts/2026-08-04-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-04-llama-cpp/</guid>
      <pubDate>Tue, 04 Aug 2026 18:56:20 GMT</pubDate>
      <description>Adds a per-conversation working directory with file picker; Agents can now treat path-like queries as directory navigation</description>
    </item>
    <item>
      <title>llama.cpp b10236</title>
      <link>https://freedom.tech/posts/2026-08-03-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-03-llama-cpp/</guid>
      <pubDate>Mon, 03 Aug 2026 05:11:26 GMT</pubDate>
      <description>Implements DSv4 Lightning Indexer and f16 Lightning Indexer for Metal; Affects 128-dimensional 64-head inputs on Apple silicon</description>
    </item>
    <item>
      <title>llama.cpp b10232</title>
      <link>https://freedom.tech/posts/2026-08-02-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-02-llama-cpp/</guid>
      <pubDate>Sun, 02 Aug 2026 18:57:43 GMT</pubDate>
      <description>WebGPU backend adds f16 repeat support; MacOS KleidiAI build disabled</description>
    </item>
    <item>
      <title>llama.cpp b10219</title>
      <link>https://freedom.tech/posts/2026-08-01-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-08-01-llama-cpp/</guid>
      <pubDate>Sat, 01 Aug 2026 16:46:07 GMT</pubDate>
      <description>Llama.cpp now preserves reasoning content in chat history, enabling multi-turn conversations to leverage previous model reasoning with --reasoning-preserve flag</description>
    </item>
    <item>
      <title>llama.cpp b10212</title>
      <link>https://freedom.tech/posts/2026-07-31-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-07-31-llama-cpp/</guid>
      <pubDate>Fri, 31 Jul 2026 19:25:45 GMT</pubDate>
      <description>Adds driver-version check for Intel GPUs on Windows to avoid crashes; No action unless running that config</description>
    </item>
    <item>
      <title>llama.cpp b10195</title>
      <link>https://freedom.tech/posts/2026-07-30-llama-cpp/</link>
      <guid>https://freedom.tech/posts/2026-07-30-llama-cpp/</guid>
      <pubDate>Thu, 30 Jul 2026 17:21:07 GMT</pubDate>
      <description>Tests alternative convolution layout; Extends layout checks for conv2d kernel</description>
    </item>
  </channel>
</rss>
