{
  "format": "final-v18-independent-evaluator-review-v1",
  "reviewer": {
    "lane": "evaluation_validity",
    "requested_model": "gpt-6-astra",
    "requested_reasoning": "xhigh",
    "service_available_for_this_review": true,
    "independence": "Separate from application implementation lanes. This lane authored evaluation methods; this is not external authorship or an outside-party audit. Arithmetic and bootstrap were independently reconstructed without calling the frozen statistics function."
  },
  "candidate_revision": "8958ae93ff196bdc5fbcfd08ae12f81445c5d28c",
  "run_checkpoint": "e3516936afff335630241ac98726305dce4c1261",
  "reviewed_at_utc": "2026-09-07T10:35:37.721546+00:00",
  "verdict": "Observed paired numbers are reproducible. Full E1 predicate sensitivity and an unqualified independent quality certification are not established. The material quality targets fail; no further tuning is authorized.",
  "changes": {
    "application": 0,
    "tests": 0,
    "fixture": 0,
    "corpus": 0,
    "methods": 0,
    "scores": 0,
    "new_generation_calls": 0,
    "only_written_artifact": "docs/evidence/insights-goal2-2026-09-06/phase0/final-v18-evaluator-review.json"
  },
  "environment": {
    "physical_schema_and_execution_configuration": "Frozen source and method hashes match. Existing authorized column types and actual definition retrieval are usable; this is distinct from complete predicate sensitivity.",
    "predicate_sensitivity": "Incomplete. Independently confirmed pipedrive.persons_enriched has64 rows,0TRUE,52FALSE,12NULL for pd_is_ct_employee; the ratified IS NOT TRUE predicate retains all64. No quality value can detect dropping that predicate.",
    "full_E1_status": "failed",
    "definition_retrieval": "Actual read-only embedding search and authorized compiler probe passed for ct_private_client on public.clients.type=PersonalAccount. Shared index was not reindexed.",
    "scorer_classes": "Frozen classes and parent flags match the stored records. No V18 task_error has a proven refusal projection. No outward V18 error_message contains the former generic crash text.",
    "baseline_viability": {
      "errors": 60,
      "supported_first_turns": 100,
      "threshold": 0.3,
      "passed": false,
      "continuation": "Explicit user override preserved; do not replace the failed viability result with environment acceptance."
    },
    "review_probe_history": [
      {
        "query_target": "ct.client_360",
        "result": "Reviewer initially targeted the wrong relation and received UndefinedColumn. This post-run reviewer mistake is not an evaluation-arm failure."
      },
      {
        "query_target": "pipedrive.persons_enriched",
        "result": {
          "rows": 64,
          "true": 0,
          "false": 52,
          "null": 12,
          "retained": 64
        },
        "access": "Read-only existing frozen fixture; no mutation."
      }
    ]
  },
  "inventory_and_integrity": {
    "attempts_per_arm": 250,
    "supported_first_turns_per_arm": 100,
    "conversations_per_arm": 50,
    "turns_per_conversation": 4,
    "intent_family_clusters": 50,
    "unique_normalized_questions": 250,
    "duplicate_or_missing_attempts": 0,
    "paired_contract_mismatches": 0,
    "source_files_verified": {
      "original": 1681,
      "candidate": 1692
    },
    "candidate_source_manifest_sha256": "39a7ef2f2a147ce584606a711055de14193a5a226a7f5ff387ad6d8fdabe0165",
    "new_held_text_source_overlaps": [],
    "E5_method_files_verified": 68,
    "E6_method_files_verified": 68,
    "frozen_hash_mismatches": 0,
    "annex_sha256": {
      "ratified-open-deal-policy-annex-v1.json": "d2333e746ca1316292798a428a23eaf1867457d0469cb23efe61157694a64bd6",
      "ratified-booking-existence-policy-annex-v1.json": "bc61600b5c0192521b2ff878ac579483cd276efa8ddea895ba57e773a7f9f2a9"
    },
    "independence_limit": "Exact-text exclusion and distinct domain compositions reduce leakage; they do not prove independent human authorship or arbitrary out-of-distribution generalization."
  },
  "independent_numeric_recomputation": {
    "matches_frozen_pair": true,
    "original_outcomes": {
      "refusal_unverified": 4,
      "task_error": 161,
      "correct_completed": 36,
      "silent_wrong": 31,
      "comparison_unverified": 18
    },
    "candidate_outcomes": {
      "refusal_unverified": 24,
      "task_error": 93,
      "correct_completed": 57,
      "unnecessary_clarification": 1,
      "comparison_unverified": 21,
      "silent_wrong": 54
    },
    "candidate_first_turn_outcomes": {
      "refusal_unverified": 11,
      "task_error": 35,
      "correct_completed": 18,
      "unnecessary_clarification": 1,
      "comparison_unverified": 12,
      "silent_wrong": 23
    },
    "first_turn_errors": {
      "original": 60,
      "candidate": 35,
      "denominator": 100,
      "target_status": "failed"
    },
    "first_turn_correct": {
      "original": 14,
      "candidate": 18,
      "denominator": 100,
      "target_status": "failed"
    },
    "precision": {
      "original": "36/67",
      "candidate": "57/111",
      "original_rate": 0.5373134328358209,
      "candidate_rate": 0.5135135135135135,
      "target_status": "failed",
      "limit": "21 candidate comparison_unverified successes are excluded; the ratio is not the verified accuracy of every returned answer."
    },
    "silent_wrong": {
      "original": 31,
      "candidate": 54,
      "denominator": 250,
      "relative_reduction": -0.7419354838709677,
      "target_status": "failed"
    },
    "full_conversation_success": {
      "original": 1,
      "candidate": 4,
      "denominator": 50,
      "target_status": "failed"
    },
    "narrow_correct_to_error_or_wrong_regressions": {
      "count": 7,
      "task_errors": 3,
      "silent_wrong": 4,
      "target_status": "failed"
    },
    "original_completion_retention": {
      "original_correct": 36,
      "candidate_still_correct": 24,
      "lost_to_task_error": 3,
      "lost_to_wrong": 4,
      "lost_to_refusal": 4,
      "lost_to_first_turn_clarification": 1,
      "all_losses": 12
    },
    "new_wrong_from_original_error": 38,
    "unclassified_errors": {
      "original": 21,
      "candidate": 0,
      "paired_target_status": "failed"
    },
    "jobs_reaching_deadline": {
      "original": 0,
      "candidate": 0,
      "denominator_per_arm": 250,
      "target_status": "passed"
    }
  },
  "independent_bootstrap": {
    "unit": "Whole paired intent-family cluster",
    "repeats": 10000,
    "seed": 2026090611,
    "families": 50,
    "interval": "Nearest-rank two-sided95% percentile",
    "relative_wrong_reduction_ci95": [
      -2.3529411764705883,
      0.022727272727272728
    ],
    "precision_difference_ci95": [
      -0.23182397959183676,
      0.18543292456335936
    ],
    "undefined_wrong_draws": 0,
    "undefined_precision_draws": 0,
    "matches_frozen_pair": true,
    "interpretation": "The data do not establish precision improvement or a reduction in silent wrong answers."
  },
  "parent_scoring_challenge": {
    "all_parent_succeeded_flags_consistent": true,
    "successful_parent_definition": "Parent supplied an answer outcome, including wrong or comparison-unverified answers; it does not mean independently correct.",
    "unnecessary_clarification_with_successful_parent": {
      "original": "0/62",
      "candidate": "0/106",
      "registered_target_status": "passed"
    },
    "candidate_first_turn_unnecessary_clarification": 1,
    "blocked_by_parent": 0,
    "dependent_turns_after_parent_error": 32,
    "dependent_turns_after_parent_refusal": 12,
    "their_outcomes": {
      "task_error": 31,
      "silent_wrong": 4,
      "refusal_unverified": 9
    },
    "limitation": "blocked_by_parent is assigned only to clarification responses. Zero such labels does not establish recovery or absence of cascades. Full conversations are counted once, but root-break diagnostics do not establish one causal break for non-clarification failures."
  },
  "latency_review": {
    "method": "Frozen v4;30queries\u00d73repeats\u00d72arms;all180 retained; timing includes250ms polling and HTTP overhead.",
    "attempts_per_arm": 90,
    "queries_per_arm": 30,
    "original_p95_ms": 1549.4488747790456,
    "candidate_p95_ms": 1732.0084376260638,
    "candidate_to_original_ratio": 1.1178222565575464,
    "numerical_ratio_recalculated": true,
    "numerical_target_status": "failed",
    "conforming_queries": {
      "original": 36,
      "candidate": 66,
      "denominator_per_arm": 90
    },
    "combined_registered_target_status": "failed",
    "all_external_call_verifier_reexecuted": true,
    "external_proof": {
      "original": {
        "sql": 159,
        "auxiliary": 87,
        "persisted": 246,
        "ledger": 87
      },
      "candidate": {
        "sql": 111,
        "auxiliary": 69,
        "persisted": 180,
        "ledger": 69
      }
    },
    "guard_violations": 0,
    "method_valid": true,
    "no_success_filtering": true,
    "limit": "Correctly reports a measurable failed target. Missing conforming successes do not make latency unverified."
  },
  "provider_failure_preservation": {
    "original": {
      "main_started": 467,
      "main_failed": 0,
      "auxiliary": 130,
      "auxiliary_failed": 0
    },
    "candidate": {
      "main_started": 453,
      "main_failed": 11,
      "main_failure_types": {
        "APITimeoutError": 10,
        "APIConnectionError": 1
      },
      "auxiliary": 118,
      "auxiliary_failed": 9,
      "auxiliary_failure_types": {
        "JobStoppedError": 9
      }
    },
    "limit": "Observed SDK failures are not proof of a remote outage; cancellation can be wrapped as APIConnectionError. Optional child budgets require independent attribution. Native refusal conversion must never erase these failures."
  },
  "semantic_precision_limit": {
    "lead_counterexamples_reviewed": 2,
    "counterexamples_logically_valid": true,
    "affected_registered_correct_first_turns": 2,
    "causes": [
      "NULL role admitted as ordinaryUser despite required User role.",
      "Note person counted in place of the independent deal person identity."
    ],
    "independent_rerun": "No additional counterexample execution; inspected the explicit VALUES relations and both SQL meanings.",
    "old_numeric_scores_preserved": true,
    "unreviewed_scope": "This review does not independently certify every one of the57 correct and21 comparison-unverified candidate outputs."
  },
  "publication_requirements": [
    "Report the complete observed pair and all15 target statuses, including failures.",
    "Do not state full evaluation validity, E1 completion, or independent generalization as established.",
    "Label numerical precision as frozen finite-fixture scoring; keep semantic counterexamples and excluded successes beside it.",
    "Keep12 lost original completions visible in addition to seven narrow regressions.",
    "Report zero unnecessary clarification with its106 answered-parent denominator and the44 children after non-answer parents.",
    "Keep all provider failures and the failed deterministic latency/conformity target.",
    "No source/fixture/scorer edits, replacement holdout, or new candidate tuning after this opened V18 run."
  ],
  "evidence_sha256": {
    "/tmp/lore-goal3-eval/final-v18-paired.json": "de0d429c89684009885e19f0ff8e68e964089697ed6a42786edb97462eb8cce1",
    "/tmp/lore-goal3-eval/sealed/corpus.json": "4c63f3efd520c57ec108bf16e73127767563056bb532236b0f7a3abe4f1e23aa",
    "/tmp/lore-goal3-eval/baseline-original-v11-r2/attempts.jsonl": "efd47bb5eac3ca42aac3acb9c1529848049711a56af3bc81862562f1bbff2a1f",
    "/tmp/lore-goal3-eval/holdout-final-v18-run/attempts.jsonl": "f12b580a542916ae26a4a17b62c56b41cbaeab0ef816d7b52e729ffe75e1b1a0",
    "docs/evidence/insights-goal2-2026-09-06/phase0/final-v18-late-predicate-fidelity-audit.json": "8fbf14053ec7d3c2011fe1a752a67bb5e2afe3307e6c3033f953a655e9a81c13",
    "docs/evidence/insights-goal2-2026-09-06/phase0/held-v18-lead-semantic-counterexamples.json": "49cde4ac846d7ff59a08f61aedcbaa4d3d43c6857920006bf13c4b8cb4afd9db",
    "docs/evidence/insights-goal2-2026-09-06/phase0/e6-final-v18/run/summary.json": "d4b2aa34fd5dc18c622f1e89f7e8fbd40d9405bf69cdb7312080ed1aea43dab7",
    "docs/evidence/insights-goal2-2026-09-06/phase0/e6-final-v18/run/attempts.jsonl": "7fffc641c0de5a1299ebc63ba192a8238a655a088f8d71fccb25055d90a41abb",
    "/tmp/lore-goal3-eval/holdout-final-v18/definition-probe.json": "cc1c79e33d9b795a5e4ce5c15a0a22565fb1d68aa598e5edfcede86ef0e923bd"
  }
}
