<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
	<channel>
		<title>Evaluation on dplabs — Software Engineering &amp; Technology Consultancy</title>
		<link>https://dplabs.tech/tags/evaluation/</link>
		<description>Recent content in Evaluation on dplabs — Software Engineering &amp; Technology Consultancy</description>
		<generator>Hugo</generator>
		<language>en-us</language>
		
		
		
		
			<lastBuildDate>Mon, 26 Jan 2026 00:00:00 +0000</lastBuildDate>
		
			<atom:link href="https://dplabs.tech/tags/evaluation/index.xml" rel="self" type="application/rss+xml" />
			<item>
				<title>Evaluating LLM Applications: Beyond Vibe Checks</title>
				<link>https://dplabs.tech/blog/evaluating-llm-applications/</link>
				<pubDate>Mon, 26 Jan 2026 00:00:00 +0000</pubDate>
				<guid>https://dplabs.tech/blog/evaluating-llm-applications/</guid>
				<description>&lt;p&gt;Most teams evaluate their LLM applications by asking them a few questions and deciding whether the answers look right. This is not evaluation — it&amp;rsquo;s a vibe check. It doesn&amp;rsquo;t scale, doesn&amp;rsquo;t catch regressions, and doesn&amp;rsquo;t provide any basis for measuring improvement over time.&lt;/p&gt;&#xA;&lt;p&gt;Systematic LLM evaluation is harder than evaluating deterministic software. The outputs are probabilistic, quality is multidimensional, and the correct answer often isn&amp;rsquo;t a single string. These are difficulties, not reasons to skip evaluation.&lt;/p&gt;</description>
			</item>
	</channel>
</rss>
