{
  "schema_version": "sommelier.hebrew_teacher_selection_evidence.v1",
  "created_at": "2026-07-13T20:47:08Z",
  "updated_at": "2026-07-14T08:00:40Z",
  "diagnostic_only": true,
  "source": {
    "dataset": "Salesforce/xlam-function-calling-60k",
    "dataset_license": "CC-BY-4.0",
    "root_rows_sha256": "4a9c16ed02e6c2042e7ac0df2c8b8a1948026371e518894b631c554a589ee91f"
  },
  "selection": {
    "method": "union of 10 rows mechanically rejected by both MADLAD400-10B and Qwen3-30B-context diagnostics, plus 12 rows with hard semantic errors in the complete accepted DictaLM3 diagnostic audit; one row overlaps",
    "selected_rows": 21,
    "mechanical_drop_intersection_source_ids_sha256": "cb9c1093f857b886f1291d8eab38cb3ab54b1ce9469fda7c356b1c456c6981d6",
    "union_source_ids_sha256": "50a459c26f0fe7dff2a02ffac239051081dc1410cf1374f5b070af3cbc4bc653",
    "derivation_journals": {
      "madlad400_10b_translation_progress_sha256": "9a689169c0975149b341249064498688c50b3825f6bb88983d9adbb36aef8fda",
      "qwen3_30b_context_translation_progress_sha256": "fb851977f7bd9469ed68fcfbab3f8af2d5ac2f61b4f344eca9043e87287ac707"
    }
  },
  "candidates": {
    "qwen3_next_80b": {
      "model_id": "Qwen/Qwen3-Next-80B-A3B-Instruct-FP8",
      "model_revision": "c5f5f263bdd5cc134092897864e8905d8fe7b928",
      "runtime": "vLLM 0.24.0 on Modal H200",
      "smoke_cohort_rows": 140,
      "mechanically_accepted_rows": 125,
      "mechanically_rejected_rows": 15,
      "translated_rows_sha256": "056bf9aa928c13254614919a3c7392c72153db3adac596fd36a3c06d155e5cee",
      "translation_progress_sha256": "efc6ef7478cee47fc08b06644d8e5d70343a7157a0ed4eefd391c7039e3f6f0c",
      "published_rows_canonical_sha256": "fdcf2dd477a3a19cc3222354311743e03bde269b0f16433392bbd8c4fb287887",
      "direct_hard_set": {
        "accepted": 14,
        "rejected": 7,
        "clean": 6,
        "minor": 4,
        "hard": 4,
        "hard_rate_among_accepted": 0.2857142857142857
      }
    },
    "openai_gpt_5_5": {
      "provider": "OpenAI",
      "model_snapshot": "gpt-5.5-2026-04-23",
      "endpoint": "https://api.openai.com/v1/responses",
      "service_tier": "default",
      "store": false,
      "strict_structured_output": true,
      "sdk_or_client_retries": 0,
      "probe_rows_sha256": "948da4d9aaa5c08ff05e4503b6a3e8e20d0a0f7f8eec96c2f8983a1ff4193507",
      "direct_hard_set": {
        "completed": 21,
        "accepted": 20,
        "rejected": 1,
        "clean": 16,
        "minor": 4,
        "hard": 0,
        "hard_rate_among_accepted": 0
      },
      "usage": {
        "input_tokens": 10883,
        "output_tokens": 1809,
        "total_tokens": 12692
      },
      "calculated_public_list_price_estimate_usd": 0.108685,
      "billing_evidence": false,
      "pricing_snapshot": {
        "checked_date": "2026-07-13",
        "standard_input_usd_per_million": 5,
        "standard_cached_input_usd_per_million": 0.5,
        "standard_output_usd_per_million": 30,
        "source": "https://developers.openai.com/api/docs/pricing"
      },
      "flex_smoke": {
        "run_id": "he-translate-openai-gpt55-flex-v9-v11-smoke-1",
        "diagnostic_only": true,
        "summary_file": "hebrew-teacher-smoke-summary.json",
        "summary_sha256": "26147b86920fd43251216a4b90bfa24f4449f5858f4480d867692694ef49f350",
        "semantic_rows_file": "hebrew-teacher-smoke-semantic-review.jsonl",
        "semantic_rows_sha256": "6b5de38f7e833f0f719fb2765fa7d30bbebd4aa4fb981cdf7ff1bbb2b61c9bc5",
        "selected_rows": 140,
        "accepted_rows": 140,
        "retried_rows": 3,
        "dropped_rows": 0,
        "semantic_grades": {
          "clean": 127,
          "minor": 13,
          "hard": 0
        },
        "requests": 143,
        "usage": {
          "input_tokens": 73359,
          "cached_input_tokens": 0,
          "output_tokens": 11618,
          "reasoning_output_tokens": 0,
          "total_tokens": 84977
        },
        "calculated_public_list_price_estimate_usd": "0.357667500",
        "billing_evidence": false,
        "producer_working_tree_clean": false,
        "claim_boundary": "The smoke selected the provider runtime and teacher. Its dirty producer provenance and non-native model-assisted semantic review make it diagnostic, not publication-grade evidence."
      }
    }
  },
  "direct_comparison": {
    "rows_file": "hebrew-teacher-probe-results.jsonl",
    "rows_sha256": "42341bdc93518271dba52873ee6949cbfb5fed0cb99afa49747416c94db64379",
    "public_redactions": [
      "OpenAI request_id",
      "OpenAI response_id"
    ],
    "qwen_acceptance_rate": 0.6666666666666666,
    "openai_acceptance_rate": 0.9523809523809523,
    "acceptance_difference_percentage_points": 28.57142857142857,
    "qwen_hard_semantic_errors": 4,
    "openai_hard_semantic_errors": 0
  },
  "decision": {
    "selected_teacher": "openai:gpt-5.5-2026-04-23",
    "selected_production_service_tier": "flex",
    "rationale": "The dated GPT-5.5 snapshot had substantially higher hard-set acceptance and zero hard semantic errors; Qwen3-Next changed operational intent in four accepted rows. A separate 140-row Flex smoke accepted every row, required three audited retries, and found 127 clean, 13 minor, and zero hard semantic outcomes under a non-native model-assisted diagnostic review.",
    "sovereign_boundary": "The external model is a one-time dataset teacher. Sommelier v3 training, adapter weights, evaluation, and deployment remain on the pinned open Nemotron/Llama-derived stack. Translation spend is reported separately from QLoRA training and inference TCO."
  },
  "claim_boundary": "This bounded, deliberately adversarial 21-row diagnostic selects a teacher candidate. It is not the full Hebrew corpus, native-speaker validation, a provider-weight checksum, a v3 accuracy result, or evidence of byte-identical provider regeneration. Full publication still requires the preregistered 200-row semantic review with zero critical errors."
}
