diff --git a/artifacts/evidence/accuracy/model-training/20260809-ai-assisted-visual-review.json b/artifacts/evidence/accuracy/model-training/20260809-ai-assisted-visual-review.json new file mode 100644 index 00000000..fe3da58d --- /dev/null +++ b/artifacts/evidence/accuracy/model-training/20260809-ai-assisted-visual-review.json @@ -0,0 +1,62 @@ +{ + "schema_version": 1, + "generated_at": "2026-08-09T09:57:02.8274719Z", + "review_kind": "ai_assisted_visual_review", + "reviewer_id": "openai-codex-gpt-5", + "human_review": false, + "repository_commit": "0209167cfd37c4f7865813a06e60bdd7ba5d918d", + "scope": "Bounded visual inspection of deterministic YOLO contact sheets; this is not a human-signed operational training release.", + "decision": { + "status": "experimental_training_only", + "promotion_allowed": false, + "protected_test_access_allowed": false, + "reason": "The Kempen background-aware corpus is materially better aligned than the rejected national candidates, but only a bounded contact-sheet sample was visually inspected and the mandatory human review gate remains open." + }, + "accepted_experimental_input": { + "dataset_yaml": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8/dataset.yaml", + "dataset_yaml_sha256": "6995843d63735bc248fb573d6ab3ebedb1cd1c86f1bd6dbb864068c5fb8f816d", + "dataset_summary": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8/yolo_tile_dataset_summary.json", + "dataset_summary_sha256": "c0c94295e52de5cc4d4cdcaaabb670220f15d2139c0051fff63d05730d33bdce", + "contact_sheet": "/app/storage/operator-evidence/accuracy-review/20260809-bgaware512r8-r1/contact_sheet_001.png", + "contact_sheet_sha256": "c5112d5c76d84153f8d219699071e7f42c8b188b0624c2bf5f9406a2146e66b6", + "rendered_tile_count": 32, + "dataset_tile_count": 207, + "observations": [ + "Most sampled labels visually overlap roof structures in the Kempen positive AOIs.", + "The training corpus contains repeated pure-empty background tiles and the validation split contains independent named positive and pure-empty AOIs.", + "Some dense scenes still contain footprint-versus-visible-roof ambiguity; this prevents operational acceptance." + ] + }, + "base_model": { + "path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt", + "sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1" + }, + "rejected_inputs": [ + { + "family": "building-be-v56 national corpus", + "decision": "exclude_from_new_training", + "reason": "Representative sheets show frequent ground-footprint versus visible-roof displacement and dense-scene ambiguity; 0/180 samples have formal human acceptance." + }, + { + "family": "building-be-v65 main/SAM candidates", + "decision": "quarantine", + "reason": "Visual samples include boxes on trees, paving and unrelated compounds after automatic refinement." + }, + { + "family": "building-be-v66 source/alignment/SAM candidates", + "decision": "quarantine", + "reason": "Systematic misalignment remains; the swapped-axis trial uses unrelated imagery and the detector-alignment sheet contains invalid labels." + } + ], + "training_constraints": { + "device": "cuda:0", + "allowed_evaluation_split": "val", + "forbidden_splits": [ + "test", + "challenge", + "background-test" + ], + "automatic_promotion": false, + "formal_release_manifest": false + } +} diff --git a/artifacts/evidence/accuracy/model-training/20260809-v67-training-decision.json b/artifacts/evidence/accuracy/model-training/20260809-v67-training-decision.json new file mode 100644 index 00000000..76ff0379 --- /dev/null +++ b/artifacts/evidence/accuracy/model-training/20260809-v67-training-decision.json @@ -0,0 +1,51 @@ +{ + "schema_version": 1, + "generated_at": "2026-08-09T10:01:00Z", + "run_id": "building-kempen-v67-ai-assisted-bgaware-r1", + "status": "completed_rejected", + "device": { + "requested": "cuda:0", + "observed": "NVIDIA GeForce RTX 4080 SUPER", + "torch": "2.11.0+cu128", + "ultralytics": "8.4.99" + }, + "training": { + "epochs": 12, + "image_size": 640, + "batch_size": 8, + "workers": 0, + "seed": 20260809, + "base_model_sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "candidate_best_sha256": "1337eb9fdd6bb73c525bd616e2ba961f20d585387db940576055afd38a71ca91", + "results_csv_sha256": "73bae1cc3e377fda8b1004a801c881f8313d6dffbbb5986555e2c5226212177d", + "args_yaml_sha256": "374613f927c57f9e9832cf7e0131c987dc978cbe1ea787a827030a8378385122" + }, + "validation": { + "split": "val", + "image_count": 36, + "positive_image_count": 18, + "pure_background_image_count": 18, + "instance_count": 5686, + "baseline": { + "precision": 0.542, + "recall": 0.430, + "map50": 0.345, + "map50_95": 0.141, + "pure_background_detections_at_confidence_0_15": 0 + }, + "candidate_best": { + "precision": 0.530, + "recall": 0.435, + "map50": 0.337, + "map50_95": 0.135, + "pure_background_detections_at_confidence_0_15": 3 + } + }, + "decision": { + "promote": false, + "production_model_changed": false, + "reason": "The small recall gain does not compensate for lower precision and mAP plus a regression from zero to three detections on pure-background validation tiles.", + "active_model_sha256_after_decision": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1" + }, + "claim_boundary": "Experimental Kempen validation only. No protected test or challenge data was opened, no national accuracy claim is made, and formal human review remains required for any future promotion." +}