<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Benchmark on Programmer.ie: Modern AI programming</title>
    <link>http://programmer.ie/tags/benchmark/</link>
    <description>Recent content in Benchmark on Programmer.ie: Modern AI programming</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Mon, 07 Sep 2026 12:00:00 +0000</lastBuildDate>
    <atom:link href="http://programmer.ie/tags/benchmark/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>How Do You Evaluate an Embedding?</title>
      <link>http://programmer.ie/books/embeddings-from-first-principles/13-chapter/</link>
      <pubDate>Mon, 07 Sep 2026 12:00:00 +0000</pubDate>
      <guid>http://programmer.ie/books/embeddings-from-first-principles/13-chapter/</guid>
      <description>&lt;p&gt;&lt;em&gt;Part IV — Measuring the Representation&lt;/em&gt;&lt;/p&gt;&#xA;&lt;h2 id=&#34;the-leaderboard-model-that-lost&#34;&gt;The leaderboard model that lost&lt;/h2&gt;&#xA;&lt;p&gt;A team picks the top model on a public embedding leaderboard. It scores well on 50-plus tasks. In their product — retrieval over dense technical documentation with heavy entity and version-number queries — it underperforms a smaller, older model. Nothing was misconfigured. The benchmark measured a population of tasks; the product is one task, and not one the benchmark weighted heavily.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
