{
  "id":"model-foundry",
  "title":"Model Foundry",
  "status":"completed",
  "tagline":"Measure open models locally under one reproducible harness.",
  "summary":"A bounded generation-harness comparison of Llama 3.2 1B and 3B under identical runtime limits. This run measures execution and output behavior, not benchmark accuracy.",
  "hypothesis":"Local 1B and 3B models can serve as reproducible controls for future routing experiments without replacing subscription models.",
  "method":["Pin exact model identifiers and 4-bit quantization.","Run the same 32 routing-format examples sequentially through the GPU queue.","Preserve native outputs and failure rows.","Do not publish a best-model score because this run had no leak-free accuracy scorer."],
  "metrics":[
    {"label":"Completed model/task cells","value":2,"unit":"count","status":"measured"},
    {"label":"1B runtime","value":48.249,"unit":"seconds","status":"measured"},
    {"label":"3B runtime","value":134.077,"unit":"seconds","status":"measured"},
    {"label":"Peak VRAM","value":3346,"unit":"MiB","status":"measured"},
    {"label":"Total wallclock","value":3.11,"unit":"minutes","status":"measured"}
  ],
  "timeline":[{"when":"2026-08-04T05:25:15Z","event":"Campaign authorized."},{"when":"2026-08-04T06:29:53Z","event":"First corrected run exposed a Transformers 5.5 quantization API mismatch."},{"when":"2026-08-04T06:34:37Z","event":"Compatibility fix verified; both model cells completed."}],
  "artifacts":[{"name":"open-eval-results.json","kind":"private-native-results","available":true}],
  "notes":"Both models produced routing JSON but often continued with explanatory prose. No accuracy headline is claimed because the generation fixture included expected responses and is valid only as a harness smoke test.",
  "last_updated":"2026-08-04T06:34:37Z"
}
