{
  "schema_version": "rendero.public-segmentation-replay.v1",
  "benchmark_id": "historical-segmentation-refiner-replay-50-20260903",
  "title": "Historical segmentation refiner replay: 50 COCO images",
  "experiment_date": "2026-09-03",
  "experiment_date_provenance": "Dated recovery REPORT.md and HISTORY.md; exact run time was not recorded in summary.json.",
  "public_export_date": "2026-09-04",
  "environment": "Offline local replay of recorded AWS Test parent masks; no new model inference.",
  "deployment_context": "The Test baseline at the time was 5ceddcfee6bc0826085c81e1884e7aef378a2f0e. The evaluated opt-in recovery branch was not deployed.",
  "scope": "50 existing development images; historical refiner replay on frozen parent masks, not a fresh automatic or VLM end-to-end benchmark",
  "promotion_eligible": false,
  "sample_construction": {
    "cases": 50,
    "unique_coco_image_ids": 50,
    "tracks": {
      "object": 30,
      "surface": 20
    },
    "categories": {
      "bed": 3,
      "bench": 3,
      "chair": 3,
      "couch": 3,
      "dining table": 3,
      "oven": 3,
      "potted plant": 3,
      "refrigerator": 3,
      "sink": 3,
      "toilet": 3,
      "wall-brick": 3,
      "wall-concrete": 3,
      "wall-other": 3,
      "wall-panel": 3,
      "wall-stone": 3,
      "wall-tile": 3,
      "wall-wood": 2
    },
    "original_split_counts": {
      "train": 30,
      "holdout": 20
    },
    "current_split_status": "All 50 had already informed development. Original train/holdout labels are retained for provenance only; no fresh independent confirmation set was tested.",
    "selection": "Fifty image-disjoint COCO validation scenes: thirty furniture or architectural fixture instances and twenty COCO-Stuff wall surfaces. Cases cover occlusion, texture, thin structure and clutter; clear category mistakes and unusable annotations were excluded during fixture construction.",
    "original_selection_note": "The fixture originally described 30 diagnosis cases and 20 sealed holdouts. That was its construction-time designation; subsequent reuse means this replay has no untouched holdout.",
    "domain": "whole-object furniture completeness and architectural wall continuity",
    "resolution": "Native image and mask dimensions; widths 320-640px and heights 240-640px. No resizing during this replay.",
    "photo_licenses_recorded": {
      "Attribution License": 34,
      "Attribution-ShareAlike License": 16
    },
    "photo_publication": "No photographs, mask bitmaps or annotations are distributed by this export. Recorded photo licenses are provenance, not a claim that attribution obligations have been completed for a future image publication."
  },
  "frozen_upstream": {
    "model": {
      "name": "rsam-mode-a-sam2-hiera-large",
      "provider": "rsam",
      "mode": "sam2_compat",
      "encoder_source_revision": "2b90b9f5ceec907a1c18123530e92e794ad901a4",
      "decoder_source_revision": "c8ba6c487049b4dda894f03e0eb4b89fbfdd771e"
    },
    "operation": "parent_masks_v1",
    "selection_policy": "low_confidence_multimask_v1",
    "raw_parent_mask_source": "Immutable parent-mask responses from the recorded AWS Test benchmark run.",
    "local_integrity_check": "All 50 complete response-file SHA256 values match the hashes recorded by the replay.",
    "box_construction": "Official reference mask/instance bounds plus deterministic 5% padding isolate segmentation mask quality from localization.",
    "reference_pixel_masks_passed_to_inference": false,
    "prediction_selection": "Use the saved parent mask whose object_id matches the registered case.",
    "missing_metadata": "Complete checkpoint-weight digest, instance/hardware type and full upstream inference configuration are not included in the replay record. No latency, provider-superiority or end-to-end benchmark claim is supported."
  },
  "stage_labels": {
    "raw": "Saved raw parent mask",
    "monday": "Historical refiner 9317364",
    "rectangle_guard": "Rectangle guard 91b8f00",
    "current": "Test refiner 5ceddcf (2026-09-03)",
    "candidate": "Opt-in line recovery candidate (unpromoted)"
  },
  "refiner_sources": {
    "monday": {
      "revision": "9317364",
      "sha256": "fe7410da2eded778683008ad0af0b4fd3ee9737a5fd0d91e220ab8d617bad3f8"
    },
    "rectangle_guard": {
      "revision": "91b8f00",
      "sha256": "57b13fb13faf64e5ab97987ec10fe04c857bd4b2bc87454996127e8f6c25c01a"
    },
    "current": {
      "revision": "5ceddcf",
      "sha256": "d4aa69aad7a7be1b24e8614bf7e76f04e08a7e49eb660add055cacea9c56697d"
    },
    "candidate": {
      "sha256": "134ce4e20463a136f8d89518b1d0bb9f5b2db4f98851b87cc11df2f499df117c"
    }
  },
  "refiner_execution": "Each version receives the same native RGB array, saved raw segmentation and saved predicted_iou. postprocess_masks defaults are defined by the exact source revision/hash. The candidate additionally runs pixel-based semantic annotation and the opt-in architectural line pass. It changed zero cases.",
  "reference": {
    "object_cases": "Official COCO 2017 validation instance polygons, rasterized by repository benchmark_sam3_shadow.reference_mask.",
    "surface_cases": "Official COCO-Stuff 2017 validation stuffthingmap references, decoded without resizing.",
    "independent_of_prediction": true,
    "limitations": "COCO labels are not detailed architectural CAD truth or fine alpha/leaf-gap ground truth. Reference-derived localization boxes make this a controlled mask/refiner test."
  },
  "metrics": {
    "binarization": "Prediction mask >127. Reference is a binary mask at the same native dimensions.",
    "iou": "Intersection / union. Empty union denominator is clamped to 1. Higher is better.",
    "leakage": "False-positive pixels / predicted foreground pixels. Denominator clamped to 1. Lower is better.",
    "missed": "False-negative pixels / reference foreground pixels. Denominator clamped to 1. Lower is better.",
    "boundary_f1": "Harmonic mean of boundary precision and recall. Boundaries use four-connected erosion with the outer image border treated as boundary. Matches use a square 5x5 max filter (Chebyshev distance <=2 native pixels). If both boundaries are empty, score=1; otherwise absent boundaries give zero match credit.",
    "aggregation": "Unweighted arithmetic mean of 50 per-case scores; each case has equal weight. Mean F1 is the average per-case F1, not F1 computed from averaged precision/recall.",
    "paired_counts": "Same 50 cases compared directly. Differences within 1e-12 are treated as ties; improvement direction is reversed for leakage and missed rate. Percentage-point delta = score delta *100.",
    "uncertainty": "Descriptive development-set results. No confidence interval, statistical significance or population generalization claim is supplied."
  },
  "registered_quality_bar": {
    "maximum_case_leakage_rate": 0.25,
    "maximum_case_missed_rate": 0.3,
    "maximum_overall_mean_leakage_rate": 0.12,
    "maximum_overall_mean_missed_rate": 0.18,
    "minimum_case_boundary_f1": 0.5,
    "minimum_case_iou": 0.55,
    "minimum_object_mean_iou": 0.78,
    "minimum_overall_mean_boundary_f1": 0.68,
    "minimum_overall_mean_iou": 0.75,
    "minimum_surface_mean_iou": 0.7,
    "required_passing_cases": 50
  },
  "gate_result": "Not a passing promotion gate. Test refiner mean boundary F1 0.51467 is below the registered 0.68 bar; mean missed rate 0.20469 exceeds the 0.18 bar. New candidate is identical on all 50 cases.",
  "limitations": [
    "Historical Test development evidence, not current production quality.",
    "Frozen parents isolate postprocessing; automatic detection, VLM inventory, object localization, UI selection and model-generated material rendering are outside scope.",
    "Refinements can increase boundary F1 while increasing leakage; show both changes.",
    "All 50 cases were reused in development; original holdout labels do not make this an untouched holdout.",
    "The opt-in line candidate changes zero cases and supports no improvement claim.",
    "No competitor comparison, rendering-speed measurement, photorealism score or cost claim follows from this dataset.",
    "No new inference or source-image redistribution occurred during public export."
  ],
  "source_artifact_sha256": {
    "replay-50-final/summary.json": "804ae3e53205851a93a3b7ed1f8b0c86667c13ee61f6df075f909f14c865bea6",
    "segmentation-coco-mixed-50-v2.json": "ac669f9edeb97f87468fcb10809bc1654b43e17b7692408802980ec6284e1bfd",
    "replay-segmentation-recovery.py": "fd8a9cdb96657e0fd2a19ff1aaa1b4337bb1fbac0c5f581767bc64bc283c10f8",
    "evaluate-rsam-supported-lines.py": "fa2e692709112ddb0dda51a5f53e79fdd145456e0b5bced8512b996b3b28bddf",
    "benchmark_sam3_shadow.py": "45225ff897cc7097ea8c90c1bbba6eab9d8b66be85e10021ed6cf1a57c266da6"
  },
  "export_verification": {
    "raw_response_hashes_matched": 50,
    "source_case_count": 50,
    "independently_recomputed_aggregate_values": 20,
    "maximum_difference_from_recorded_aggregate": 5.551115123125783e-17,
    "paired_comparisons_recomputed": 12,
    "candidate_changed_cases": 0
  },
  "public_references": [
    {
      "title": "COCO dataset",
      "url": "https://cocodataset.org/"
    },
    {
      "title": "COCO-Stuff dataset",
      "url": "https://github.com/nightrome/cocostuff"
    },
    {
      "title": "COCO-Stuff label definitions",
      "url": "https://github.com/nightrome/cocostuff/blob/master/labels.txt"
    },
    {
      "title": "SAM 2 upstream project",
      "url": "https://github.com/facebookresearch/sam2"
    }
  ]
}
