
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Distributed Systems Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Production notes and source-backed analysis on distributed systems, messaging infrastructure, open-source internals, and model engineering.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Fri, 31 Jul 2026 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/inference/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/kimi-k3/05-api-vs-self-hosting-cost</guid>
			<title>Kimi K3 Cost Model: Hosted API Versus Self-Hosting</title>
			<link>https://josedavidbaena.com/blog/kimi-k3/05-api-vs-self-hosting-cost</link>
			<description>Model Kimi K3 API and self-hosting costs with real token mixes, topology quotes, measured throughput, redundancy, storage, and engineering overhead.</description>
			<pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>kimi-k3</category><category>api-pricing</category><category>self-hosting</category><category>cost-optimization</category><category>inference</category>
		</item>
	
		<item>
			<guid>https://josedavidbaena.com/blog/kimi-k3/01-hardware-requirements</guid>
			<title>Kimi K3 Hardware Requirements: Why 8 H100s Are Not Enough</title>
			<link>https://josedavidbaena.com/blog/kimi-k3/01-hardware-requirements</link>
			<description>Kimi K3 activates about 104.2B parameters per token, but its 1.561 TB snapshot still needs a multi-GPU memory, cache, storage, and network plan.</description>
			<pubDate>Mon, 27 Jul 2026 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>kimi-k3</category><category>gpu</category><category>inference</category><category>mixture-of-experts</category><category>hardware</category>
		</item>
	
		<item>
			<guid>https://josedavidbaena.com/blog/frontier-model-engineering/03-long-context-storage-system</guid>
			<title>LLM Long Context: Storage, KV Cache, and Retrieval</title>
			<link>https://josedavidbaena.com/blog/frontier-model-engineering/03-long-context-storage-system</link>
			<description>Separate context limits from resident state, prefix reuse, retrieval, compaction, token cost, eviction, and measured task value.</description>
			<pubDate>Fri, 24 Jul 2026 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>long-context</category><category>kv-cache</category><category>retrieval</category><category>inference</category><category>observability</category>
		</item>
	
			</channel>
		</rss>
	