
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Distributed Systems Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Production notes and source-backed analysis on distributed systems, messaging infrastructure, open-source internals, and model engineering.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Wed, 09 Sep 2026 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/preference-optimization/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/distilled-engineering/distilled-dpo-on-policy-alignment</guid>
			<title>Distilled DPO: Training a Small Model on Its Own Mistakes</title>
			<link>https://josedavidbaena.com/blog/distilled-engineering/distilled-dpo-on-policy-alignment</link>
			<description>Build a bounded on-policy loop that turns a distilled student&#39;s validated failures into DPO pairs without leaking holdouts or erasing safety behavior.</description>
			<pubDate>Wed, 09 Sep 2026 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>distillation</category><category>rlhf</category><category>dpo</category><category>preference-optimization</category><category>fine-tuning</category><category>llm</category><category>safety</category>
		</item>
	
			</channel>
		</rss>
	