<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Skill Evaluation Field Notes</title>
    <link>https://peaty.app/blog</link>
    <description>Methods, guides, and findings for evaluating AI agent skills against real workflows.</description>
    <language>en</language>
    <atom:link href="https://peaty.app/feed.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Replay the work, not the prompt</title>
      <link>https://peaty.app/blog/replay-the-work-not-the-prompt</link>
      <guid isPermaLink="true">https://peaty.app/blog/replay-the-work-not-the-prompt</guid>
      <pubDate>Fri, 28 Aug 2026 12:00:00 GMT</pubDate>
      <category>Methods</category>
      <description>A skill changes the path an agent takes. A fair evaluation preserves the work around it, then changes one thing.</description>
    </item>
    <item>
      <title>Skills need contracts, not more prose</title>
      <link>https://peaty.app/blog/skills-need-contracts-not-more-prose</link>
      <guid isPermaLink="true">https://peaty.app/blog/skills-need-contracts-not-more-prose</guid>
      <pubDate>Wed, 26 Aug 2026 12:00:00 GMT</pubDate>
      <category>Skills</category>
      <description>Longer instructions often hide the real interface. A useful skill states when it applies, what it owns, and how success can be checked.</description>
    </item>
    <item>
      <title>A pass rate is not a verdict</title>
      <link>https://peaty.app/blog/a-pass-rate-is-not-a-verdict</link>
      <guid isPermaLink="true">https://peaty.app/blog/a-pass-rate-is-not-a-verdict</guid>
      <pubDate>Fri, 21 Aug 2026 12:00:00 GMT</pubDate>
      <category>Field notes</category>
      <description>Tests can pass while the workflow gets slower, noisier, or less trustworthy. Review the evidence that sits around the score.</description>
    </item>
    <item>
      <title>A practical method for evaluating tool-use skills</title>
      <link>https://peaty.app/blog/evaluating-tool-use-skills</link>
      <guid isPermaLink="true">https://peaty.app/blog/evaluating-tool-use-skills</guid>
      <pubDate>Tue, 18 Aug 2026 12:00:00 GMT</pubDate>
      <category>Methods</category>
      <description>Test whether an agent chose the right tool, used it with the right scope, and turned its result into a defensible outcome.</description>
    </item>
    <item>
      <title>A taxonomy of skill failures in multi-step workflows</title>
      <link>https://peaty.app/blog/a-taxonomy-of-skill-failures</link>
      <guid isPermaLink="true">https://peaty.app/blog/a-taxonomy-of-skill-failures</guid>
      <pubDate>Wed, 12 Aug 2026 12:00:00 GMT</pubDate>
      <category>Field notes</category>
      <description>A useful failure label identifies the decision that broke, not just the symptom visible at the end of the run.</description>
    </item>
    <item>
      <title>Turn one failure into a durable test case</title>
      <link>https://peaty.app/blog/turn-one-failure-into-a-test-case</link>
      <guid isPermaLink="true">https://peaty.app/blog/turn-one-failure-into-a-test-case</guid>
      <pubDate>Wed, 05 Aug 2026 12:00:00 GMT</pubDate>
      <category>Methods</category>
      <description>Preserve the smallest real context that reproduces the failure, then keep it in the suite after the immediate fix ships.</description>
    </item>
  </channel>
</rss>
