<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Developers Digest - Benchmark</title>
    <link>https://www.developersdigest.tech/tags/benchmark</link>
    <description>2 items tagged Benchmark on Developers Digest - blog posts, tools, guides, and tutorials.</description>
    <language>en</language>
    <lastBuildDate>Fri, 31 Jul 2026 20:04:59 GMT</lastBuildDate>
    <atom:link href="https://www.developersdigest.tech/tags/benchmark/feed.xml" rel="self" type="application/rss+xml" />
    <image>
      <url>https://avatars.githubusercontent.com/u/124798203?v=4</url>
      <title>Developers Digest - Benchmark</title>
      <link>https://www.developersdigest.tech/tags/benchmark</link>
    </image>
    <item>
      <title><![CDATA[Change2Task: The Assembly Line for Coding Agent Training Data]]></title>
      <link>https://www.developersdigest.tech/blog/change2task-repo-changes-to-coding-agent-tasks</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/change2task-repo-changes-to-coding-agent-tasks</guid>
      <description><![CDATA[Microsoft's Change2Task turns merged pull requests into verified, executable coding agent tasks: 79.6% construction success across 1,130 repo changes, 29.2% more verified tasks than PR baselines, and tasks that stay current with the codebase.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>Research</category>
    </item>
    <item>
      <title><![CDATA[SWE-NFI: The Benchmark That Catches What Coding Agents Miss]]></title>
      <link>https://www.developersdigest.tech/blog/swe-nfi-coding-agents-quality-benchmark</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/swe-nfi-coding-agents-quality-benchmark</guid>
      <description><![CDATA[A new 188-task benchmark for non-functional improvements finds coding agents hit 70% on functional correctness but lag humans on refactors and structural changes - the quality gap that becomes tech debt.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>Code Quality</category>
      <category>Research</category>
    </item>
  </channel>
</rss>