<?xml version="1.0" encoding="UTF-8"?><rss version="2.0"
	xmlns:content="http://purl.org/rss/1.0/modules/content/"
	xmlns:wfw="http://wellformedweb.org/CommentAPI/"
	xmlns:dc="http://purl.org/dc/elements/1.1/"
	xmlns:atom="http://www.w3.org/2005/Atom"
	xmlns:sy="http://purl.org/rss/1.0/modules/syndication/"
	xmlns:slash="http://purl.org/rss/1.0/modules/slash/"
	>

<channel>
	<title>edge ai &#8211; Sigma Blogger</title>
	<atom:link href="https://www.sigmablogger.com/tag/edge-ai/feed/" rel="self" type="application/rss+xml" />
	<link>https://www.sigmablogger.com</link>
	<description>Technology, Business &#38; Intelligent Living</description>
	<lastBuildDate>Sun, 13 Sep 2026 06:34:32 +0000</lastBuildDate>
	<language>en-US</language>
	<sy:updatePeriod>
	hourly	</sy:updatePeriod>
	<sy:updateFrequency>
	1	</sy:updateFrequency>
	

<image>
	<url>https://www.sigmablogger.com/wp-content/uploads/2019/10/cropped-Logo-2-32x32.png</url>
	<title>edge ai &#8211; Sigma Blogger</title>
	<link>https://www.sigmablogger.com</link>
	<width>32</width>
	<height>32</height>
</image> 
	<item>
		<title>Building and Deploying vllm.cpp: NVIDIA CUDA, Vulkan, AMD ROCm (RDNA3 / RDNA3.5), Qwen Deployment, RadixTree Caching, and Production Systemd Automation</title>
		<link>https://www.sigmablogger.com/building-and-deploying-vllm-cpp-nvidia-cuda-vulkan-amd-rocm-rdna3-rdna3-5-qwen-deployment-radixtree-caching-and-production-systemd-automation/</link>
		
		<dc:creator><![CDATA[sigma9999]]></dc:creator>
		<pubDate>Sun, 13 Sep 2026 06:33:31 +0000</pubDate>
				<category><![CDATA[All articles]]></category>
		<category><![CDATA[Business]]></category>
		<category><![CDATA[Technology]]></category>
		<category><![CDATA[4gb vram]]></category>
		<category><![CDATA[ai inference]]></category>
		<category><![CDATA[c++20]]></category>
		<category><![CDATA[cmake]]></category>
		<category><![CDATA[continuous batching]]></category>
		<category><![CDATA[Cuda]]></category>
		<category><![CDATA[edge ai]]></category>
		<category><![CDATA[Fedora Linux]]></category>
		<category><![CDATA[fp8 quantization]]></category>
		<category><![CDATA[gguf]]></category>
		<category><![CDATA[gpu acceleration]]></category>
		<category><![CDATA[hip]]></category>
		<category><![CDATA[Linux Tutorial]]></category>
		<category><![CDATA[llama.cpp]]></category>
		<category><![CDATA[llm inference]]></category>
		<category><![CDATA[local ai]]></category>
		<category><![CDATA[model serving]]></category>
		<category><![CDATA[open source llm]]></category>
		<category><![CDATA[paged kv cache]]></category>
		<category><![CDATA[prefix caching]]></category>
		<category><![CDATA[qwen]]></category>
		<category><![CDATA[radixattention]]></category>
		<category><![CDATA[rdna3]]></category>
		<category><![CDATA[rocm]]></category>
		<category><![CDATA[self hosted ai]]></category>
		<category><![CDATA[sglang]]></category>
		<category><![CDATA[systemd]]></category>
		<category><![CDATA[vllm]]></category>
		<category><![CDATA[vllm.cpp]]></category>
		<category><![CDATA[Vulkan]]></category>
		<guid isPermaLink="false">https://www.sigmablogger.com/?p=1302</guid>

					<description><![CDATA[<p class="wp-block-paragraph">While upstream Python-based vLLM requires complex container layers, heavy virtual environments, and multi-gigabyte PyTorch dependencies (Kwon et al., 2023; Paszke et al., 2019), <strong><code>mudler/vllm.cpp</code></strong> compiles directly via modern CMake into an embeddable, standalone C++20 serving engine with zero Python dependencies (Di Giacinto, 2026).</p>



<p class="wp-block-paragraph">Below is &#8230;&#8230; <a href="https://www.sigmablogger.com/building-and-deploying-vllm-cpp-nvidia-cuda-vulkan-amd-rocm-rdna3-rdna3-5-qwen-deployment-radixtree-caching-and-production-systemd-automation/" class="read-more">Read More <span class="screen-reader-text"> &#8220;Building and Deploying vllm.cpp: NVIDIA CUDA, Vulkan, AMD ROCm (RDNA3 / RDNA3.5), Qwen Deployment, RadixTree Caching, and Production Systemd Automation&#8221;</span></a></p>]]></description>
		
		
		
			</item>
	</channel>
</rss>
