{"schema_version":"newruntime-agent-readable-v0.2","type":"post","stable_id":"post:evocode-bench-multi-turn-regressions","slug":"evocode-bench-multi-turn-regressions","title":"EvoCode-Bench Exposes Multi-Turn Regression Risk","description":"EvoCode-Bench tests coding agents across persistent workspaces and evolving requirements, where regressions become the dominant failure mode.","retrieval_nugget":"EvoCode-Bench tests coding agents across persistent workspaces and evolving requirements, where regressions become the dominant failure mode. EvoCode-Bench is a useful correction to how coding agents are normally evaluated. A single prompt with a final pass/fail result misses the real failure mode of agentic coding: the agent has to keep working in the same codebase as requirements evolve.","status":"published","published_at":"2026-08-03","updated_at":"2026-08-03","record_date":"2026-08-03","date_kind":"published_at","topics":["coding-agents","evals","benchmarks","regressions"],"source_urls":["https://www.philschmid.de/evocode-bench"],"visuals":[{"id":"evocode-bench-multi-turn-regressions","kind":"editorial-diagram","role":"hero","src":"https://newruntime.com/images/posts/evocode-bench-multi-turn-regressions.webp","alt":"Whiteboard diagram contrasting single-turn evaluation with multi-turn coding tasks and cumulative regression tests.","caption":"New Runtime synthesis: coding-agent evals need persistent workspaces and cumulative tests because regressions compound over turns.","credit":"New Runtime synthesis","source_url":"https://www.philschmid.de/evocode-bench","generated_with":"gemini-3.1-flash-image","width":1600,"height":900,"legend":[]}],"routes":{"html":"https://newruntime.com/posts/evocode-bench-multi-turn-regressions/","markdown":"https://newruntime.com/posts/evocode-bench-multi-turn-regressions.md","json":"https://newruntime.com/posts/evocode-bench-multi-turn-regressions.json"},"source_format":"markdown","next_reads":[{"type":"topic","path":"/topics/coding-agents/","reason":"Explore the coding agents topic hub.","url":"https://newruntime.com/topics/coding-agents/","title":"Coding agents - New Runtime","media_type":"text/html"},{"type":"topic","path":"/topics/evals/","reason":"Explore the evals topic hub.","url":"https://newruntime.com/topics/evals/","title":"Agent evals - New Runtime","media_type":"text/html"},{"type":"related_material","path":"/posts/agentic-sdlc-software-factory-loop/","reason":"Shares coding agents and evals.","url":"https://newruntime.com/posts/agentic-sdlc-software-factory-loop/","title":"A Software Factory Connects Agents Through Verified Outcomes","media_type":"text/html"},{"type":"related_material","path":"/posts/claude-code-auto-mode-action-gate/","reason":"Shares coding agents and evals.","url":"https://newruntime.com/posts/claude-code-auto-mode-action-gate/","title":"Claude Code Auto Mode Gates Actions Instead Of Explanations","media_type":"text/html"},{"type":"related_material","path":"/posts/langchain-reviewbench-review-agent-evals/","reason":"Shares coding agents and evals.","url":"https://newruntime.com/posts/langchain-reviewbench-review-agent-evals/","title":"ReviewBench Turns Code Review Into An Agent Eval","media_type":"text/html"}]}
