<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>MaaS on Wissen Xue Blog</title>
    <link>https://h.acethon.com/tags/maas/</link>
    <description>Recent content in MaaS on Wissen Xue Blog</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Tue, 22 Sep 2026 18:28:18 +0800</lastBuildDate>
    <atom:link href="https://h.acethon.com/tags/maas/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>From llama.cpp to vLLM: Inside the Architecture of a Production LLM MaaS Platform</title>
      <link>https://h.acethon.com/posts/llama-cpp-vllm-maas/</link>
      <pubDate>Tue, 22 Sep 2026 18:28:18 +0800</pubDate>
      <guid>https://h.acethon.com/posts/llama-cpp-vllm-maas/</guid>
      <description>&lt;h1 id=&#34;from-llamacpp-to-vllm-inside-the-architecture-of-a-production-llm-maas-platform&#34;&gt;From llama.cpp to vLLM: Inside the Architecture of a Production LLM MaaS Platform&lt;/h1&gt;&#xA;&lt;p&gt;When we first start working with open-source LLMs, the problem looks deceptively simple:&lt;/p&gt;&#xA;&lt;blockquote&gt;&#xA;&lt;p&gt;&lt;strong&gt;How do I run this model?&lt;/strong&gt;&lt;/p&gt;&lt;/blockquote&gt;&#xA;&lt;p&gt;Download a model, load it into an inference runtime, send a prompt, and get a response.&lt;/p&gt;&#xA;&lt;p&gt;Projects such as &lt;a href=&#34;https://github.com/ggml-org/llama.cpp&#34;&gt;llama.cpp&lt;/a&gt; have made this experience remarkably accessible. Its goal is to provide efficient LLM and VLM inference with minimal setup across a wide range of hardware, including CPUs, Apple Silicon, NVIDIA and AMD GPUs, and other accelerators. It also supports low-bit quantization and CPU+GPU hybrid inference. (&lt;a href=&#34;https://github.com/ggml-org/llama.cpp?pubDate=20260313&amp;amp;utm_source=chatgpt.com&#34; title=&#34;GitHub - ggml-org/llama.cpp: LLM inference in C/C++ · GitHub&#34;&gt;GitHub&lt;/a&gt;)&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
