{
  "_meta": {
    "testId": "0.4",
    "jiraTicket": "SAIL-153",
    "question": "Does a curve-fit residual on the model's OWN drawn points (no ground truth needed) predict whether that line is actually correct — usable as a free per-line quality gate?",
    "inferenceUrl": "https://inference-production-208d.up.railway.app",
    "modelVersion": "sailscan-v26",
    "holdoutFile": "test-holdout-ids-canonical.json",
    "holdoutFrozen": "2026-08-25",
    "nHoldout": 332,
    "nPhotosScored": 332,
    "nDetectionsTotal": 1286,
    "nDetectionsWithValidFit": 1286,
    "nDetectionsDegenerateOrTooFewPoints": 0,
    "minPointsForFit": 5,
    "nMatchedForCorrelation": 1091,
    "correlationNote": "Correlation computed ONLY over detections that won their greedy confidence-first match to a label stripe (distNormExclusive != null) — same matching philosophy as gate0_confidence_stratification.mjs, so both units (residualNorm, distNormExclusive) are directly comparable and both normalized by image diagonal. Spearman is the primary read (heavy-tailed distances); Pearson reported alongside.",
    "battenGrabDefinition": "residualNorm <= smoothThreshold (bottom quartile of all valid-fit detections, computed value 0.0007269529233424296) AND distNormAny >= farThreshold (90th percentile of distance-to-nearest-real-label-stripe among valid-fit detections, computed value 0.03020775697624586). distNormAny ignores greedy-match exclusivity so a detection that lost its match to a higher-confidence rival is still checked against the real stripe nearest to it.",
    "rawDetectionsFile": "gate0-curvefit-detections.jsonl"
  },
  "correlation": {
    "spearman": 0.6197368710190955,
    "pearson": 0.056940507173106863,
    "pearsonPValue": 0.060090236131110794,
    "pearsonAfterDroppingTop1PctError": 0.1651,
    "n": 1091,
    "residualMedian": 0.0015417171557013303,
    "errorMedian": 0.0016162760738454372,
    "verificationNote": "spearman/pearson independently recomputed with scipy.stats.spearmanr/pearsonr over gate0-curvefit-detections.jsonl (matched rows only) and matched the hand-rolled implementation above to 4 decimals. Pearson is not significant on its own (p=0.06) and is depressed by a small number of extreme-error lines: dropping the worst 1% of matched lines by error raises Pearson to 0.1651, consistent with Spearman being the more honest read of this relationship."
  },
  "battenGrab": {
    "smoothThreshold": 0.0007269529233424296,
    "farThreshold": 0.03020775697624586,
    "count": 15,
    "pctOfValidFitDetections": 0.01166407465007776,
    "examples": [
      {
        "id": "45bc65b1-8b0b-44e1-81f5-ea9ccdc0191e",
        "stripeIdx": 0,
        "confidence": 0.352,
        "residualNorm": 0.000672195943045521,
        "distNormAny": 0.15210251002569478
      },
      {
        "id": "68fd2c599e3418f5653c0e11",
        "stripeIdx": 0,
        "confidence": 0.617,
        "residualNorm": 0.0004748000738647659,
        "distNormAny": 0.03852686360105985
      },
      {
        "id": "68fd2c599e3418f5653c0e11",
        "stripeIdx": 1,
        "confidence": 0.786,
        "residualNorm": 0.0005381823351158079,
        "distNormAny": 0.03779601929396351
      },
      {
        "id": "692ee2dd1b550c3bcb690cfd",
        "stripeIdx": 0,
        "confidence": 0.802,
        "residualNorm": 0.0005538848833839842,
        "distNormAny": 0.09378756481707769
      },
      {
        "id": "6983c7303b15d0ca7a335979",
        "stripeIdx": 0,
        "confidence": 0.777,
        "residualNorm": 0.0002470944992937227,
        "distNormAny": 0.042872938194777395
      },
      {
        "id": "6983c7303b15d0ca7a335979",
        "stripeIdx": 1,
        "confidence": 0.688,
        "residualNorm": 0.0002524809942118372,
        "distNormAny": 0.03128701894962245
      },
      {
        "id": "69aeb1c431726f412729f849",
        "stripeIdx": 1,
        "confidence": 0.673,
        "residualNorm": 0.000678344599743437,
        "distNormAny": 0.2586132697525245
      },
      {
        "id": "69aeb1c431726f412729f849",
        "stripeIdx": 2,
        "confidence": 0.578,
        "residualNorm": 0.00018267634134426977,
        "distNormAny": 0.3031106994728625
      },
      {
        "id": "69aeb1c431726f412729f849",
        "stripeIdx": 3,
        "confidence": 0.386,
        "residualNorm": 0.00033432152985916723,
        "distNormAny": 0.30452270008931676
      },
      {
        "id": "69aeb1c431726f412729f849",
        "stripeIdx": 4,
        "confidence": 0.327,
        "residualNorm": 0.00006453036762355982,
        "distNormAny": 0.3316686494844508
      },
      {
        "id": "69bca9fdd26d710e284a87e7",
        "stripeIdx": 0,
        "confidence": 0.49,
        "residualNorm": 0.0005064944358724088,
        "distNormAny": 0.11177718247447396
      },
      {
        "id": "69c44d5c8c718d5d543a2010",
        "stripeIdx": 9,
        "confidence": 0.117,
        "residualNorm": 0.00023134866723609305,
        "distNormAny": 0.14846030682925535
      },
      {
        "id": "69c44d5c8c718d5d543a2010",
        "stripeIdx": 10,
        "confidence": 0.714,
        "residualNorm": 0.000709142111820156,
        "distNormAny": 0.14826002682142667
      },
      {
        "id": "69d8c28fb4793b96e2e1f51f",
        "stripeIdx": 3,
        "confidence": 0.61,
        "residualNorm": 0.0001252220128043606,
        "distNormAny": 0.03862712813565971
      },
      {
        "id": "69f09238270a835330087b8c",
        "stripeIdx": 0,
        "confidence": 0.704,
        "residualNorm": 0.00019521042762854398,
        "distNormAny": 0.043356829033377155
      }
    ]
  }
}