<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Benchmark - Developers Digest</title>
    <link>https://www.developersdigest.tech/blog/tags/benchmark</link>
    <description>Articles about Benchmark on Developers Digest</description>
    <language>en</language>
    <lastBuildDate>Fri, 14 Aug 2026 23:22:38 GMT</lastBuildDate>
    <atom:link href="https://www.developersdigest.tech/blog/tags/benchmark/feed.xml" rel="self" type="application/rss+xml" />
    <item>
      <title><![CDATA[The ICML 2026 Agent Reproduction Audit: 23% of Examined Papers Had Falsified or Contested Claims]]></title>
      <link>https://www.developersdigest.tech/blog/icml-2026-reproduction-audit</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/icml-2026-reproduction-audit</guid>
      <description><![CDATA[Hugging Face's open challenge used 1,200+ participants and their coding agents to attempt 2,226 ICML 2026 papers claim by claim. 51% had claims independently verified, 23% had a falsified or contested claim, and four documented falsifications include a spotlight theorem that fails after step 224.]]></description>
      <pubDate>Fri, 14 Aug 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Research</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/ai-agent-pmf-cost-control/hero.webp" type="image/webp" />
    </item>
    <item>
      <title><![CDATA[Change2Task: The Assembly Line for Coding Agent Training Data]]></title>
      <link>https://www.developersdigest.tech/blog/change2task-repo-changes-to-coding-agent-tasks</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/change2task-repo-changes-to-coding-agent-tasks</guid>
      <description><![CDATA[Microsoft's Change2Task turns merged pull requests into verified, executable coding agent tasks: 79.6% construction success across 1,130 repo changes, 29.2% more verified tasks than PR baselines, and tasks that stay current with the codebase.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>Research</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/agent-context-reduction-pattern/hero.webp" type="image/webp" />
    </item>
    <item>
      <title><![CDATA[ORCA-bench: Frontier Agents Score 10% on Hard Oncall RCA]]></title>
      <link>https://www.developersdigest.tech/blog/orca-bench-oncall-rca-agents-not-ready</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/orca-bench-oncall-rca-agents-not-ready</guid>
      <description><![CDATA[A new benchmark drops five frontier coding agents into a live OpenTelemetry microservice system with real Prometheus, Jaeger, and OpenSearch telemetry. Best RCA accuracy: 25.3% on Medium, 10.0% on Hard. Even Claude Fable 5 is far from oncall-ready.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>SRE</category>
      <category>Observability</category>
      <category>Research</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/400-dollar-overnight-bill-agent-finops/hero.webp" type="image/webp" />
    </item>
    <item>
      <title><![CDATA[PAIChecker: 13.6% of SWE-bench Verified Instances Have Misaligned PR-Issue Pairs]]></title>
      <link>https://www.developersdigest.tech/blog/paichecker-swe-bench-pr-issue-misalignment</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/paichecker-swe-bench-pr-issue-misalignment</guid>
      <description><![CDATA[A systematic audit of SWE-bench Verified finds 68 of 500 instances (13.6%) pair a pull request with an issue it does not actually resolve, penalizing agents that correctly solve the stated problem. PAIChecker, a three-phase multi-agent checker, flags them with up to 92.12% binary accuracy.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>Code Review</category>
      <category>Research</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/agent-evals-need-baseline-receipts/hero.webp" type="image/webp" />
    </item>
    <item>
      <title><![CDATA[SWE-NFI: The Benchmark That Catches What Coding Agents Miss]]></title>
      <link>https://www.developersdigest.tech/blog/swe-nfi-coding-agents-quality-benchmark</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/swe-nfi-coding-agents-quality-benchmark</guid>
      <description><![CDATA[A new 188-task benchmark for non-functional improvements finds coding agents hit 70% on functional correctness but lag humans on refactors and structural changes - the quality gap that becomes tech debt.]]></description>
      <pubDate>Fri, 31 Jul 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>AI Agents</category>
      <category>Benchmark</category>
      <category>Code Quality</category>
      <category>Research</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/agent-evals-need-baseline-receipts/hero.webp" type="image/webp" />
    </item>
  </channel>
</rss>