{"type":"post","stable_id":"post:model-fit-needs-a-task-taxonomy-not-one-leaderboard","slug":"model-fit-needs-a-task-taxonomy-not-one-leaderboard","title":"Model Fit Needs a Task Taxonomy, Not One Leaderboard","description":"A mathematical taxonomy, an individual writing preference, and a small-model catalog show why asking which model is best is meaningless without task class, budget, and an accepted result.","retrieval_nugget":"The sources do not form a benchmark. Their value is a qualified guide for designing one: define the task, constraints, evaluation artifact, and deployment boundary first.","published_at":"2026-08-14","updated_at":"2026-08-15","record_date":"2026-08-14","date_kind":"discovered_at","topics":["evals","new-models","reasoning"],"entities":["Tim Gowers","DeepSeek","AFM-4.5B-Preview"],"source_urls":["https://gowers.wordpress.com/2026/08/12/what-sort-of-maths-are-llms-good-at","https://deepakness.com/raw/deepseek-better-for-writing","https://together.ai/models/afm-4-5b-preview"],"source_format":"article","editorial_timing":{"lane":"regular_hourly","scheduled_at":"2026-08-22T14:00:00+03:00","real_news_delta":"owner-selected cross-source synthesis"},"origin":{"basket_id":"5854f7b2-5954-4d2a-8997-81596a49da74","basket_revision":1,"target_kind":"synthesis","target_id":"b53f4d99-9e77-4aed-b9c5-96aa1b803430","owner_selection":"45-68 long-tail synthesis","route":"hermes"},"visual_decision":{"outcome":"text_only","status":"not_applicable","reason_code":"concise_text_sufficient","owner_reviewed":true,"reviewed_by":"owner-and-codex"},"schema_version":"newruntime-agent-readable-v0.2","status":"published","visuals":[],"routes":{"html":"https://newruntime.com/posts/model-fit-needs-a-task-taxonomy-not-one-leaderboard/","markdown":"https://newruntime.com/posts/model-fit-needs-a-task-taxonomy-not-one-leaderboard.md","json":"https://newruntime.com/posts/model-fit-needs-a-task-taxonomy-not-one-leaderboard.json"}}
