<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Process Reward Models on Programmer.ie</title>
    <link>http://programmer.ie/tags/process-reward-models/</link>
    <description>Recent content in Process Reward Models on Programmer.ie</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Sat, 08 Aug 2026 17:26:00 +0100</lastBuildDate>
    <atom:link href="http://programmer.ie/tags/process-reward-models/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Trajectory Search</title>
      <link>http://programmer.ie/books/agents-from-first-principles/09-chapter/</link>
      <pubDate>Sat, 08 Aug 2026 17:26:00 +0100</pubDate>
      <guid>http://programmer.ie/books/agents-from-first-principles/09-chapter/</guid>
      <description>&lt;p&gt;An agent can make every local decision look reasonable and still lose the task.&lt;/p&gt;&#xA;&lt;p&gt;A coding agent sees that &lt;code&gt;test_checkout_redirect&lt;/code&gt; is failing, concludes there is an implementation bug in &lt;code&gt;checkout.py&lt;/code&gt;, and then behaves impeccably for twenty steps: it reads the file, edits it, runs the tests, repairs the new failures its edit introduced, rewrites the patch, and runs the tests again. Every one of those steps is defensible given the step before it.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
