<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
  <url>
    <loc>https://llmasajudge.hashnode.dev</loc>
    <lastmod>2026-09-11T14:38:07.601Z</lastmod>
    <changefreq>always</changefreq>
    <priority>1.0</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/dividing-your-rag-score-by-retrieval-recall-overstates-your-generation-quality-and-here-is-by-how-much</loc>
    <lastmod>2026-08-25T20:50:06.527Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/six-prompt-optimization-frameworks-what-matters-when-you-run-them-on-the-same-task</loc>
    <lastmod>2026-08-18T18:32:41.559Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/a-judge-that-agrees-with-your-humans-92-percent-of-the-time-can-be-at-60-percent-where-the-gate-actually-decides</loc>
    <lastmod>2026-08-17T17:40:10.631Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/run-forty-experiments-against-one-eval-set-and-you-will-find-an-improvement-that-is-not-there</loc>
    <lastmod>2026-08-12T14:08:12.471Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/i-read-the-metric-libraries-of-five-widely-used-eval-tools-the-metric-was-never-the-hard-part</loc>
    <lastmod>2026-08-11T18:54:34.217Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-eval-monitor-fired-on-four-days-this-week-at-your-sample-size-that-was-the-most-likely-count</loc>
    <lastmod>2026-08-07T19:00:37.658Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/upgrading-the-judge-ends-one-score-series-and-starts-another</loc>
    <lastmod>2026-08-06T19:16:10.489Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/a-noisy-judge-does-not-just-add-error-bars-it-shrinks-the-effect-you-are-trying-to-measure</loc>
    <lastmod>2026-08-05T19:28:02.858Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-eval-s-confidence-interval-assumes-independent-examples-yours-are-clustered</loc>
    <lastmod>2026-07-28T21:24:03.621Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/an-llm-judge-is-a-biased-instrument-not-a-measurement</loc>
    <lastmod>2026-07-22T19:27:11.827Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-eval-dashboard-has-30-metrics-when-one-moves-that-is-usually-arithmetic-not-a-regression</loc>
    <lastmod>2026-07-21T14:39:59.571Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-eval-pass-rate-is-98-percent-your-confidence-interval-is-probably-wrong</loc>
    <lastmod>2026-07-16T15:22:28.752Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/comparing-two-eval-runs-by-their-average-pass-rate-is-the-wrong-test</loc>
    <lastmod>2026-07-14T16:54:06.326Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/one-average-eval-score-was-hiding-two-different-failure-modes</loc>
    <lastmod>2026-07-08T17:04:57.933Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-llm-as-judge-has-a-position-bias-you-are-not-measuring</loc>
    <lastmod>2026-07-07T16:42:35.164Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/i-reviewed-six-operator-ready-checklists-for-ai-agents-none-of-them-define-the-problem-correctly</loc>
    <lastmod>2026-07-01T16:06:46.476Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/we-added-synthetic-data-to-our-eval-set-the-pass-rate-rose-and-so-did-our-production-incidents</loc>
    <lastmod>2026-06-29T17:02:15.234Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/i-checked-six-llm-as-judge-tools-against-human-labels-the-scoreboard-was-the-wrong-thing-to-read</loc>
    <lastmod>2026-06-25T17:54:43.128Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/i-benchmarked-6-prompt-optimization-frameworks-on-the-same-task-here-is-what-each-one-actually-optimizes</loc>
    <lastmod>2026-06-19T17:28:38.873Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/llm-as-judge-tools-compared-the-question-is-not-which-one-scores-it-is-which-one-you-can-trust</loc>
    <lastmod>2026-06-17T17:08:18.833Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/stratified-sampling-for-llm-eval-sets-why-your-aggregate-pass-rate-hides-the-regressions-that-matter</loc>
    <lastmod>2026-06-16T17:23:42.103Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/we-put-confidence-intervals-on-our-llm-judge-scores-the-error-bars-ate-three-weeks-of-trend</loc>
    <lastmod>2026-06-11T19:20:14.335Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/more-eval-traces-will-not-stabilize-your-kappa-stratify-the-ones-you-have</loc>
    <lastmod>2026-06-09T18:44:52.653Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/calibration-set-size-for-llm-as-judge-when-50-traces-is-enough-and-when-200-is-mandatory</loc>
    <lastmod>2026-06-04T17:08:25.517Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/your-llm-as-judge-eval-set-is-too-small-here-is-the-math</loc>
    <lastmod>2026-05-27T11:56:22.002Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/archive</loc>
    <lastmod>2026-09-11T14:38:07.601Z</lastmod>
    <changefreq>daily</changefreq>
    <priority>0.5</priority>
  </url>
  <url>
    <loc>https://llmasajudge.hashnode.dev/recommendations</loc>
    <lastmod>2026-09-11T14:38:07.601Z</lastmod>
    <changefreq>weekly</changefreq>
    <priority>0.4</priority>
  </url>
</urlset>