{"schema_version":"newruntime-agent-readable-v0.2","type":"raw_signal","stable_id":"signal:openai-a-benchmark-score-reflects-the-model-as-well-as-the-harness-and-settings","id":"x-2082616641834422740","slug":"openai-a-benchmark-score-reflects-the-model-as-well-as-the-harness-and-settings","title":"OpenAI: A benchmark score reflects the model as well as the harness and settings used to run it.","description":"A public X post from OpenAI with a linked primary source flags A benchmark score reflects the model as well as the harness and settings used to run it. For long-running agents, retaining reasoning and compacting context lets the model build o...","retrieval_nugget":"A public X post from OpenAI with a linked primary source flags A benchmark score reflects the model as well as the harness and settings used to run it. For long-running agents, retaining reasoning and compacting context lets the model build o...","observed_at":"2026-07-29","record_date":"2026-07-29","date_kind":"observed_at","why_it_matters":"This X-discovered record adds fresh evidence to the agent ready software exposes capabilities, model answers become task interfaces lens and lets the site trend graph move as public product and research signals arrive.","novelty":"structural","verification_level":"source-linked","signal_type":"research","evidence_kind":"creator-source","status":"published","source_platform":"x","source_record_id":"2082616641834422740","source_url":"https://x.com/OpenAI/status/2082616641834422740","topics":["generated-ui","evals","api-design","models","context-engineering","agents","interfaces"],"entities":["OpenAI"],"related_patterns":["agent-ready-software-exposes-capabilities","model-answers-become-task-interfaces","harness-architecture-outlives-model-choice","skills-become-portable-capability-layer","company-memory-needs-write-loops","verification-bandwidth-is-the-scarce-resource","agent-economics-moves-to-completed-work","goal-scoped-loops-replace-manual-continuation"],"source_urls":["https://x.com/OpenAI/status/2082616641834422740","https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores/"],"import_batch":"x-analyzer-20260730-084757","routes":{"html":"https://newruntime.com/signals/openai-a-benchmark-score-reflects-the-model-as-well-as-the-harness-and-settings/","markdown":"https://newruntime.com/signals/openai-a-benchmark-score-reflects-the-model-as-well-as-the-harness-and-settings.md","json":"https://newruntime.com/signals/openai-a-benchmark-score-reflects-the-model-as-well-as-the-harness-and-settings.json"},"source_format":"x-api-normalized-json","next_reads":[{"type":"pattern","path":"/patterns/agent-ready-software-exposes-capabilities/","reason":"Pattern connected to this observed signal.","url":"https://newruntime.com/patterns/agent-ready-software-exposes-capabilities/","title":"Agent-ready software exposes capabilities","media_type":"text/html"},{"type":"pattern","path":"/patterns/model-answers-become-task-interfaces/","reason":"Pattern connected to this observed signal.","url":"https://newruntime.com/patterns/model-answers-become-task-interfaces/","title":"Model answers become task interfaces","media_type":"text/html"},{"type":"pattern","path":"/patterns/harness-architecture-outlives-model-choice/","reason":"Pattern connected to this observed signal.","url":"https://newruntime.com/patterns/harness-architecture-outlives-model-choice/","title":"Harness architecture outlives model choice","media_type":"text/html"},{"type":"pattern","path":"/patterns/skills-become-portable-capability-layer/","reason":"Pattern connected to this observed signal.","url":"https://newruntime.com/patterns/skills-become-portable-capability-layer/","title":"Skills become a portable capability layer","media_type":"text/html"},{"type":"pattern","path":"/patterns/company-memory-needs-write-loops/","reason":"Pattern connected to this observed signal.","url":"https://newruntime.com/patterns/company-memory-needs-write-loops/","title":"Company memory needs write and correction loops","media_type":"text/html"}]}
