{"schema_version":"newruntime-agent-readable-v0.2","type":"post","stable_id":"post:google-agent-evaluations-offline-online","slug":"google-agent-evaluations-offline-online","title":"Google Runs The Same Agent Metrics Before And After Launch","description":"Gemini Enterprise Agent Platform makes experiments, adaptive rubrics, trace review, simulations, online monitors, and drift alerts generally available on one evaluation engine.","retrieval_nugget":"Gemini Enterprise Agent Platform makes experiments, adaptive rubrics, trace review, simulations, online monitors, and drift alerts generally available on one evaluation engine. Google's Agent and Model Evaluations release closes a common measurement gap: the test suite used before launch and the monitoring system used afterward can now run the same metrics.","status":"published","published_at":"2026-08-01","updated_at":"2026-08-01","record_date":"2026-08-01","date_kind":"published_at","topics":["evals","agent-observability","simulation","quality-systems"],"source_urls":["https://developers.googleblog.com/agent-and-model-evaluations-in-gemini-enterprise-agent-platform-are-now-ga"],"visuals":[{"id":"google-agent-evaluations-offline-online","kind":"editorial-diagram","role":"hero","src":"https://newruntime.com/images/posts/google-agent-evaluations-offline-online.webp","alt":"Hand-drawn evaluation loop where local cases, simulated users, and mocked tools use the same metric registry as sampled production traces and drift alerts.","caption":"Google connects pre-launch experiments and post-launch monitoring through one versioned metric registry.","credit":"New Runtime synthesis from Google Developers Blog","source_url":"https://developers.googleblog.com/agent-and-model-evaluations-in-gemini-enterprise-agent-platform-are-now-ga","generated_with":"gemini-3.1-flash-image","width":1600,"height":900,"legend":[{"label":"Experiment","description":"Datasets, simulated users, and mocked environments produce reproducible development traces."},{"label":"Metric registry","description":"Code checks and LLM judges are versioned once and reused across agents and stages."},{"label":"Monitor","description":"Sampled live traces receive the same scores, producing trends, clusters, and drift alerts."}]}],"telegram_message_id":2906,"telegram_url":"https://t.me/qwgai/2906","telegram_message_ids":[2906,2907],"telegram_delivery_mode":"text_then_media","telegram_media_url":"https://t.me/qwgai/2907","routes":{"html":"https://newruntime.com/posts/google-agent-evaluations-offline-online/","markdown":"https://newruntime.com/posts/google-agent-evaluations-offline-online.md","json":"https://newruntime.com/posts/google-agent-evaluations-offline-online.json"},"source_format":"markdown","next_reads":[{"type":"topic","path":"/topics/agent-observability/","reason":"Explore the agent observability topic hub.","url":"https://newruntime.com/topics/agent-observability/","title":"Agent Observability - New Runtime","media_type":"text/html"},{"type":"topic","path":"/topics/evals/","reason":"Explore the evals topic hub.","url":"https://newruntime.com/topics/evals/","title":"Agent evals - New Runtime","media_type":"text/html"},{"type":"related_material","path":"/posts/digibee-opik-prompt-versioning-loop/","reason":"Shares agent observability and evals.","url":"https://newruntime.com/posts/digibee-opik-prompt-versioning-loop/","title":"Prompt Versioning Is Becoming Agent Operations","media_type":"text/html"},{"type":"related_material","path":"/posts/agentic-sdlc-software-factory-loop/","reason":"Shares evals.","url":"https://newruntime.com/posts/agentic-sdlc-software-factory-loop/","title":"A Software Factory Connects Agents Through Verified Outcomes","media_type":"text/html"},{"type":"related_material","path":"/posts/cerebras-moe-router-gradient-null-expert/","reason":"Shares evals.","url":"https://newruntime.com/posts/cerebras-moe-router-gradient-null-expert/","title":"A Balanced MoE Router Can Still Be Functionally Dead","media_type":"text/html"}]}
