{
  "schemaVersion": "agent-lab.ui-harness.arm-analysis/1",
  "activeFamily": "omni-ui-harness-v0",
  "runId": "ui-reporting-repairmode-k1-20260628T012257Z",
  "task": {
    "suiteId": "omni-builder",
    "id": "reporting-dashboard",
    "shape": "dense-dashboard",
    "route": "/reporting",
    "buildFile": "src/app/(app)/reporting/page.tsx",
    "brief": "Build a compact clinic reporting dashboard for the owner/operator. It should answer 'is the clinic healthy today?': revenue pace, patient pipeline, schedule utilization, sales pipeline, marketing attribution, and recent operational activity — the six data objects already fetched on the page (revenue, patients, scheduling, sales, attribution, feed). Prioritize fast scanning over presentation; it should feel like a working clinic cockpit, not an investor dashboard. Keep the auth + data fetching exactly as-is and replace only the render body. Use ONLY @/components/ui primitives + lucide icons + cn — do NOT import or read the existing @/components/reporting/* widgets. Compact zone, tabular-nums for money, round to decision-relevant precision. Handle: no revenue today, refunds exceeding payments, 0 leads, empty activity feed, long names."
  },
  "framing": "same model/effort; arm is raw_ui_agent vs full_ui_harness; final lab gates and judges are symmetric for both arms",
  "scoringFrame": "Report code judge, visual judge, judgeBlend, strict recorded effective, calibrated effective, and gate pass separately. Calibrated effective floors to 50 on hard failures and caps at 60 on rendered severe geometry (overlap, viewport overflow, clipping). Soft aesthetic/density flags remain diagnostics. judgeBlend is the unfloored 65% code + 35% visual score.",
  "threshold": {
    "medianLiftPoints": 10,
    "medianJudgeBlendLiftPoints": 10,
    "maxAcceptedHarnessLossPoints": 5,
    "requiredValidSeedsPerArmForPromotion": 3,
    "requireLabPassFull": true,
    "requireSymmetricPostRunGates": true,
    "effectiveFloorRule": "floor to 50 on hard failures; cap at 60 on rendered severe geometry (overlap, viewport overflow, clipping); soft flags remain diagnostics"
  },
  "lowConfidence": true,
  "verdict": "revise_full_harness",
  "recommendedRead": "Treat this primary analysis as the candidate-generation and visual/style shakedown. The contract conclusion now comes from the task-specific replay/rescore pass, which drives browser proof states for both arms.",
  "delta": {
    "fullMinusRawMedianEffective": 10,
    "fullMinusRawMedianJudgeBlend": 2,
    "fullMinusRawMedianCodeJudge": 0,
    "fullMinusRawMedianVisualJudge": 3,
    "rawMinusFullMeanBlocking": 15,
    "fullMinusRawMeanFlags": -2
  },
  "caveats": [
    "n=1 per arm; this is a methods shakedown, not a promotion-grade estimate.",
    "The primary run score is useful for builder process, static/rendered lint, screenshots, and visual judge output, not as the final product-contract verdict.",
    "Use the replay/rescore report for contract preservation because it drives task-specific browser proof states.",
    "The primary screenshots remain visual shakedown states; deeper contract proof lives in the rescore artifacts.",
    "Use this report to compare builder process, static/rendered lint, screenshots, and visual judge output; use the rescore report to compare contract preservation."
  ],
  "recommendedNext": [
    "Run harness/arena/rescore-ui-arm-run.mjs on this runId before drawing contract conclusions.",
    "Treat rendered overlap, viewport overflow, and clipping as severe geometry defects; keep only soft aesthetic/density flags as diagnostics or score-shading.",
    "Promote only if the rescore shows contract preservation while the visual/style lane also improves."
  ],
  "arms": [
    {
      "arm": "raw_ui_agent",
      "n": 1,
      "validN": 1,
      "infraErrors": 0,
      "timedOutN": 0,
      "leakedN": 0,
      "strictLabPassRate": 0,
      "labPassRate": 0,
      "screenshotPassRate": 100,
      "medianEffective": 50,
      "meanEffective": 50,
      "medianRecordedEffective": 50,
      "meanRecordedEffective": 50,
      "medianJudgeBlend": 74,
      "meanJudgeBlend": 74,
      "meanFloorPenalty": 24,
      "meanRecordedFloorPenalty": 24,
      "medianCodeJudge": 79,
      "medianVisualJudge": 66,
      "meanBlocking": 20,
      "meanFlags": 25,
      "medianCorrectionLoops": 0,
      "medianSelfGateRuns": 0,
      "medianRenderedSelfGateRuns": 0,
      "meanCostUsd": 0.3604,
      "medianTurns": 8,
      "records": [
        {
          "seed": 1,
          "effectiveScore": 50,
          "recordedEffectiveScore": 50,
          "judgeBlend": 74,
          "floorPenalty": 24,
          "recordedFloorPenalty": 24,
          "labPass": false,
          "strictLabPass": false,
          "codeJudge": 79,
          "visualJudge": 66,
          "blocking": 20,
          "flags": 25,
          "correctionLoopCount": 0,
          "selfGateRuns": 0,
          "costUsd": 0.3604373,
          "timedOut": false
        }
      ]
    },
    {
      "arm": "full_ui_harness",
      "n": 1,
      "validN": 1,
      "infraErrors": 0,
      "timedOutN": 0,
      "leakedN": 0,
      "strictLabPassRate": 100,
      "labPassRate": 0,
      "screenshotPassRate": 100,
      "medianEffective": 60,
      "meanEffective": 60,
      "medianRecordedEffective": 76,
      "meanRecordedEffective": 76,
      "medianJudgeBlend": 76,
      "meanJudgeBlend": 76,
      "meanFloorPenalty": 16,
      "meanRecordedFloorPenalty": 0,
      "medianCodeJudge": 79,
      "medianVisualJudge": 69,
      "meanBlocking": 5,
      "meanFlags": 23,
      "medianCorrectionLoops": 2,
      "medianSelfGateRuns": 5,
      "medianRenderedSelfGateRuns": 2,
      "meanCostUsd": 0.578,
      "medianTurns": 19,
      "records": [
        {
          "seed": 1,
          "effectiveScore": 60,
          "recordedEffectiveScore": 76,
          "judgeBlend": 76,
          "floorPenalty": 16,
          "recordedFloorPenalty": 0,
          "labPass": false,
          "strictLabPass": true,
          "codeJudge": 79,
          "visualJudge": 69,
          "blocking": 5,
          "flags": 23,
          "correctionLoopCount": 2,
          "selfGateRuns": 5,
          "costUsd": 0.5779569,
          "timedOut": false
        }
      ]
    }
  ],
  "generatedAt": "2026-06-28T02:13:50.069Z"
}
