<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>LLMJury Blog</title>
    <link>https://llmjury.com/blog</link>
    <description>Run live A/B tests on your prompts and models, version prompts outside your codebase, and get a plain-English verdict backed by real statistics — LLM evaluation with proof, not gut feeling.</description>
    <language>en</language>
    <lastBuildDate>Thu, 16 Jul 2026 12:00:00 GMT</lastBuildDate>
    <atom:link href="https://llmjury.com/rss.xml" rel="self" type="application/rss+xml"/>
    <item>
      <title>Offline evals aren’t enough: why prompts need testing on live traffic</title>
      <link>https://llmjury.com/blog/offline-evals-are-not-enough</link>
      <guid isPermaLink="true">https://llmjury.com/blog/offline-evals-are-not-enough</guid>
      <pubDate>Thu, 16 Jul 2026 12:00:00 GMT</pubDate>
      <description>Golden datasets are unit tests, not proof. Distribution drift, eval overfitting, and the metrics that only exist in production — latency, cost, and what users do next.</description>
      <category>evaluation</category>
      <category>LLM-as-judge</category>
      <category>production</category>
    </item>
    <item>
      <title>Why prompt changes deserve A/B tests, not vibes</title>
      <link>https://llmjury.com/blog/why-ab-test-prompts</link>
      <guid isPermaLink="true">https://llmjury.com/blog/why-ab-test-prompts</guid>
      <pubDate>Thu, 16 Jul 2026 12:00:00 GMT</pubDate>
      <description>Ten playground outputs can’t tell you what a prompt change does to real traffic. The case for measuring every prompt edit the way you measure every other production change.</description>
      <category>A/B testing</category>
      <category>prompt engineering</category>
      <category>experimentation</category>
    </item>
    <item>
      <title>Announcing LLMJury: statistically defensible A/B testing for LLM products</title>
      <link>https://llmjury.com/blog/announcing-llmjury</link>
      <guid isPermaLink="true">https://llmjury.com/blog/announcing-llmjury</guid>
      <pubDate>Wed, 01 Jul 2026 12:00:00 GMT</pubDate>
      <description>Why “it feels better” is not an eval strategy, and how SRM gates, FDR correction, and an LLM judge make prompt experiments trustworthy.</description>
      <category>announcements</category>
      <category>statistics</category>
      <category>SRM</category>
      <category>FDR</category>
    </item>
  </channel>
</rss>
