{
  "schema_version": "newruntime-agent-readable-v0.1",
  "type": "raw_signal",
  "id": "tg-2699",
  "slug": "colibri-runs-a-744b-model-in-25-gb-of-memory",
  "title": "Colibri Runs a 744B Model in 25 GB of Memory",
  "description": "An experimental C inference engine uses aggressive storage and memory techniques to make a very large GLM model runnable without a GPU.",
  "observed_at": "2026-07-15",
  "why_it_matters": "This dated record adds public evidence to the harness architecture outlives model choice analysis and keeps the claim auditable as the underlying products and practices change.",
  "novelty": "notable",
  "verification_level": "source-linked",
  "signal_type": "tool",
  "evidence_kind": "mixed",
  "status": "published",
  "telegram_message_id": 2699,
  "telegram_url": "https://t.me/qwgai/2699",
  "topics": [
    "local-models",
    "inference",
    "efficiency"
  ],
  "entities": [],
  "related_patterns": [
    "harness-architecture-outlives-model-choice"
  ],
  "source_urls": [
    "https://github.com/JustVugg/colibri",
    "https://tomshardware.com/tech-industry/artificial-intelligence/colibri-proof-of-concept-gains-frontier-level-1-5-tb-ai-model-novel-approach-runs-on-only-25gb-of-ram-and-shows-promise-for-local-ai-setups"
  ],
  "import_batch": "telegram-2026-07-17-new-sources-v3",
  "routes": {
    "html": "https://newruntime.com/signals/colibri-runs-a-744b-model-in-25-gb-of-memory/",
    "markdown": "https://newruntime.com/signals/colibri-runs-a-744b-model-in-25-gb-of-memory.md",
    "json": "https://newruntime.com/signals/colibri-runs-a-744b-model-in-25-gb-of-memory.json"
  },
  "source_format": "telegram-export-normalized-json"
}
