{
  "schema_version": "newruntime-agent-readable-v0.1",
  "type": "raw_signal",
  "id": "tg-2707",
  "slug": "agent-skills-need-behavioral-evals",
  "title": "Agent skills need behavioral evals, not prose review",
  "description": "Benchmarks show that expert-authored skills can help while self-generated skills can underperform a no-skill baseline.",
  "observed_at": "2026-07-16",
  "why_it_matters": "A skill is executable behavior packaged as files, so its quality has to be measured across tasks, models, and harnesses instead of judged by how polished the instructions look.",
  "novelty": "structural",
  "verification_level": "source-inspected",
  "signal_type": "field-report",
  "evidence_kind": "mixed",
  "status": "published",
  "telegram_message_id": 2707,
  "telegram_url": "https://t.me/qwgai/2707",
  "topics": [
    "agent-skills",
    "evals",
    "context-engineering"
  ],
  "entities": [
    "SkillsBench",
    "Claude"
  ],
  "related_patterns": [],
  "source_urls": [
    "https://arxiv.org/abs/2602.12670",
    "https://arxiv.org/abs/2603.29919",
    "https://arxiv.org/abs/2605.24050",
    "https://github.com/benchflow-ai/skillsbench",
    "https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices"
  ],
  "import_batch": "telegram-2026-07-17-v1",
  "routes": {
    "html": "https://newruntime.com/signals/agent-skills-need-behavioral-evals/",
    "markdown": "https://newruntime.com/signals/agent-skills-need-behavioral-evals.md",
    "json": "https://newruntime.com/signals/agent-skills-need-behavioral-evals.json"
  },
  "source_format": "telegram-export-normalized-json"
}
