{
  "evaluation_id": "invoice-extraction-v0.1",
  "completed_at": "2026-08-10T19:11:38.199Z",
  "status": "three_candidates_scored_fixture_only",
  "headline": "Luna matched Sol on all 12 records at one twenty-fifth of the measured API cost",
  "recommendation": {
    "hosted_default_candidate": "gpt-5.6-luna",
    "reason": "Luna and Sol both achieved 12/12 whole-record exact match and 100% schema validity. Luna cost $0.0012278 versus $0.030695 for Sol and had lower median request latency in this run.",
    "frontier_escalation_evidence": "None in this fixture: there was no record Sol passed that Luna failed.",
    "production_decision": "Not established. Define an error budget and test a larger representative sample, including OCR and real document variation, before production use."
  },
  "results": [
    {
      "candidate_id": "luna-hosted",
      "model": "gpt-5.6-luna",
      "route": "cheap_hosted",
      "whole_records_correct": 12,
      "whole_records_total": 12,
      "whole_record_exact_match": 1,
      "schema_validity": 1,
      "errors": 0,
      "automatic_retries": 0,
      "input_tokens": 2539,
      "output_tokens": 600,
      "measured_cost_usd": 0.0012278,
      "median_latency_ms": 1355,
      "minimum_latency_ms": 996,
      "maximum_latency_ms": 2439,
      "run_id": "luna-hosted-2026-08-10T19-08-46.619Z",
      "artifacts": {
        "manifest": "runs/luna-hosted-2026-08-10T19-08-46.619Z/manifest.json",
        "predictions": "runs/luna-hosted-2026-08-10T19-08-46.619Z/predictions.jsonl",
        "raw_records": "runs/luna-hosted-2026-08-10T19-08-46.619Z/raw-records.jsonl",
        "score": "runs/luna-hosted-2026-08-10T19-08-46.619Z/score.json"
      }
    },
    {
      "candidate_id": "sol-hosted",
      "model": "gpt-5.6-sol",
      "route": "frontier_hosted",
      "whole_records_correct": 12,
      "whole_records_total": 12,
      "whole_record_exact_match": 1,
      "schema_validity": 1,
      "errors": 0,
      "automatic_retries": 0,
      "input_tokens": 2539,
      "output_tokens": 600,
      "measured_cost_usd": 0.030695,
      "median_latency_ms": 1576,
      "minimum_latency_ms": 1227,
      "maximum_latency_ms": 4695,
      "run_id": "sol-hosted-2026-08-10T19-09-16.222Z",
      "artifacts": {
        "manifest": "runs/sol-hosted-2026-08-10T19-09-16.222Z/manifest.json",
        "predictions": "runs/sol-hosted-2026-08-10T19-09-16.222Z/predictions.jsonl",
        "raw_records": "runs/sol-hosted-2026-08-10T19-09-16.222Z/raw-records.jsonl",
        "score": "runs/sol-hosted-2026-08-10T19-09-16.222Z/score.json"
      }
    },
    {
      "candidate_id": "gemma4-local",
      "model": "gemma4:12b",
      "route": "laptop_open_weight",
      "whole_records_correct": 11,
      "whole_records_total": 12,
      "whole_record_exact_match": 0.9166666666666666,
      "schema_validity": 1,
      "errors": 0,
      "automatic_retries": 0,
      "input_tokens": 2289,
      "output_tokens": 937,
      "measured_api_cost_usd": 0,
      "hardware_energy_and_operations_cost": "not measured",
      "median_latency_ms": 3794,
      "minimum_latency_ms": 3617,
      "maximum_latency_ms": 67453,
      "failure": {
        "case_id": "inv-010",
        "field": "issue_date",
        "expected": "2026-08-03",
        "received": "2026-03-08",
        "note": "The local model reversed the day and month in the spaced UK-format date."
      },
      "run_id": "gemma4-local-2026-08-10T19-09-48.959Z",
      "artifacts": {
        "manifest": "runs/gemma4-local-2026-08-10T19-09-48.959Z/manifest.json",
        "predictions": "runs/gemma4-local-2026-08-10T19-09-48.959Z/predictions.jsonl",
        "raw_records": "runs/gemma4-local-2026-08-10T19-09-48.959Z/raw-records.jsonl",
        "score": "runs/gemma4-local-2026-08-10T19-09-48.959Z/score.json"
      }
    }
  ],
  "comparisons": {
    "sol_to_luna_measured_cost_ratio": 25,
    "luna_measured_saving_vs_sol_percent": 96,
    "luna_median_latency_advantage_vs_sol_percent": 14.02,
    "combined_hosted_spend_usd": 0.0319228,
    "illustrative_uncached_cost_at_same_mean_token_shape": {
      "invoices": 100000,
      "luna_usd": 10.23,
      "sol_usd": 255.79,
      "difference_usd": 245.56,
      "boundary": "Linear illustration from this run's mean token counts; excludes prompt caching, Batch discounts, retries, OCR, validation, and integration costs."
    }
  },
  "sources_checked_2026_08_10": {
    "gpt_5_6_luna_model_and_pricing": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
    "gpt_5_6_sol_model_and_pricing": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
    "muse_glimmer_gguf_model_card": "https://huggingface.co/meta-models/Muse-Glimmer-30B-GGUF",
    "ollama_muse_glimmer_release_status": "https://github.com/ollama/ollama/issues/17645"
  },
  "open_weight_hardware_gate": {
    "model": "Muse Glimmer 30B K-Quant-17GB",
    "status": "not_scored_on_this_machine",
    "vendor_target": "24 GB VRAM",
    "machine_gpu": "NVIDIA GeForce RTX 5060 Laptop GPU with 8151 MiB",
    "runtime_note": "Installed Ollama 0.32.7 predates the released GGUF path; Ollama maintainers stated the non-MLX model requires 0.32.8.",
    "decision": "Do not spend a 16.76 GB download to test a configuration outside the vendor hardware target."
  },
  "boundaries": [
    "Twelve synthetic text records test the harness; they do not establish production reliability.",
    "The fixture does not test image rendering or OCR quality.",
    "No production error budget or pass threshold has been set.",
    "Hosted costs use measured uncached tokens and prices locked on 10 August 2026; Batch and caching were not used.",
    "The local route has no API bill in this run, but hardware, energy, maintenance, and concurrency costs were not measured."
  ]
}
