Automate failure-driven training continuation
This commit is contained in:
@@ -62,6 +62,20 @@ def test_training_command_supports_conservative_aerial_finetuning(tmp_path: Path
|
|||||||
assert f"data={tmp_path / 'dataset.yaml'}" in command
|
assert f"data={tmp_path / 'dataset.yaml'}" in command
|
||||||
|
|
||||||
|
|
||||||
|
def test_failed_iteration_builds_train_only_sampling_for_next_checkpoint(tmp_path: Path) -> None:
|
||||||
|
command = MODULE.failure_sampling_command(
|
||||||
|
scripts_dir=tmp_path / "scripts",
|
||||||
|
train_summary=tmp_path / "train-summary.json",
|
||||||
|
corpus_manifest=tmp_path / "manifest.json",
|
||||||
|
assessment=tmp_path / "assessment.json",
|
||||||
|
output_dir=tmp_path / "iteration-001" / "failure-driven-training",
|
||||||
|
)
|
||||||
|
assert command[1].endswith("build_failure_driven_yolo_sampling.py")
|
||||||
|
assert command[command.index("--summary") + 1].endswith("train-summary.json")
|
||||||
|
assert command[command.index("--assessment") + 1].endswith("assessment.json")
|
||||||
|
assert command[command.index("--output-dir") + 1].endswith("failure-driven-training")
|
||||||
|
|
||||||
|
|
||||||
def test_loop_refuses_failed_dataset_audit(tmp_path: Path) -> None:
|
def test_loop_refuses_failed_dataset_audit(tmp_path: Path) -> None:
|
||||||
audit = tmp_path / "audit.json"
|
audit = tmp_path / "audit.json"
|
||||||
audit.write_text(json.dumps({"status": "needs_attention", "low_variance_positive_tile_count": 4}))
|
audit.write_text(json.dumps({"status": "needs_attention", "low_variance_positive_tile_count": 4}))
|
||||||
@@ -73,6 +87,8 @@ def test_loop_refuses_failed_dataset_audit(tmp_path: Path) -> None:
|
|||||||
str(tmp_path / "base.pt"),
|
str(tmp_path / "base.pt"),
|
||||||
"--train-yaml",
|
"--train-yaml",
|
||||||
str(tmp_path / "dataset.yaml"),
|
str(tmp_path / "dataset.yaml"),
|
||||||
|
"--train-summary",
|
||||||
|
str(tmp_path / "train-summary.json"),
|
||||||
"--dataset-audit",
|
"--dataset-audit",
|
||||||
str(audit),
|
str(audit),
|
||||||
"--calibration-summary",
|
"--calibration-summary",
|
||||||
|
|||||||
@@ -92,6 +92,10 @@ recall are repeated, while true negative train tiles are repeated when a
|
|||||||
regional precision gate or the pure-background gate fails. Calibration, test,
|
regional precision gate or the pure-background gate fails. Calibration, test,
|
||||||
background-test and validation AOIs are excluded by their frozen corpus split;
|
background-test and validation AOIs are excluded by their frozen corpus split;
|
||||||
the generated evidence records that no protected sample entered training.
|
the generated evidence records that no protected sample entered training.
|
||||||
|
The checkpointed orchestrator invokes this builder after every rejected
|
||||||
|
iteration, stores its checksum in `training-loop-state.json`, and uses the
|
||||||
|
resulting dataset YAML for the next checkpoint. A restart resumes both the
|
||||||
|
candidate weights and that exact failure-driven training input.
|
||||||
The orchestrator refuses to start unless every automated frozen-dataset gate
|
The orchestrator refuses to start unless every automated frozen-dataset gate
|
||||||
passes and the corpus contains zero blank/low-variance positive tiles. The
|
passes and the corpus contains zero blank/low-variance positive tiles. The
|
||||||
audit status may remain `needs_human_review` while training and objective
|
audit status may remain `needs_human_review` while training and objective
|
||||||
|
|||||||
@@ -11773,11 +11773,20 @@ Deployment evidence:
|
|||||||
immutable, automated failures are empty, spatial leakage is `ok`, and blank
|
immutable, automated failures are empty, spatial leakage is `ok`, and blank
|
||||||
positive-tile count is zero. Human sign-off remains a separate mandatory
|
positive-tile count is zero. Human sign-off remains a separate mandatory
|
||||||
final promotion gate.
|
final promotion gate.
|
||||||
|
- Closed the next orchestration gap: a rejected iteration now invokes the
|
||||||
|
leak-free failure-driven sampler automatically, records its evidence
|
||||||
|
checksum and next dataset YAML in `training-loop-state.json`, and resumes
|
||||||
|
both the candidate weights and exact sampling input after interruption.
|
||||||
|
- Confirmed v37 epoch 1 completed on CUDA with validation precision `0.601`,
|
||||||
|
recall `0.455`, mAP50 `0.474` and mAP50-95 `0.205`; the run remains inactive
|
||||||
|
and these internal-validation metrics are not release evidence.
|
||||||
|
|
||||||
Verified in this pass:
|
Verified in this pass:
|
||||||
|
|
||||||
- `py -3 -m pytest -q backend/tests/test_belgium_training_loop.py backend/tests/test_belgium_training_iteration_assessment.py backend/tests/test_belgium_training_portfolio.py`
|
- `py -3 -m pytest -q backend/tests/test_belgium_training_loop.py backend/tests/test_belgium_training_iteration_assessment.py backend/tests/test_belgium_training_portfolio.py`
|
||||||
(`12 passed`).
|
(`12 passed`).
|
||||||
|
- `py -3 -m pytest -q backend/tests/test_belgium_training_loop.py backend/tests/test_failure_driven_yolo_sampling.py backend/tests/test_belgium_training_iteration_assessment.py`
|
||||||
|
(`15 passed`).
|
||||||
|
|
||||||
Open:
|
Open:
|
||||||
|
|
||||||
|
|||||||
@@ -972,6 +972,7 @@ This file now starts with the current implementation status. Older preparation/b
|
|||||||
- [x] Reject positive labels over blank/no-data imagery and replace partial SPW 2024 coverage with the complete dated SPW 2023 campaign.
|
- [x] Reject positive labels over blank/no-data imagery and replace partial SPW 2024 coverage with the complete dated SPW 2023 campaign.
|
||||||
- [x] Exclude GRB/PICC features created after the corresponding dated imagery period while retaining auditable rejection evidence.
|
- [x] Exclude GRB/PICC features created after the corresponding dated imagery period while retaining auditable rejection evidence.
|
||||||
- [x] Allow the objective CUDA loop to consume an automatically clean `needs_human_review` corpus while keeping final human sign-off as a separate, mandatory promotion gate.
|
- [x] Allow the objective CUDA loop to consume an automatically clean `needs_human_review` corpus while keeping final human sign-off as a separate, mandatory promotion gate.
|
||||||
|
- [x] Persist checksummed train-only failure-driven sampling after every rejected loop iteration and resume the next checkpoint from that exact dataset YAML.
|
||||||
- [x] Evaluate the completed v36 YOLO11x checkpoint calibration-first on the rotated v30 holdouts; reject it before opening test/background because the regional calibration gate failed.
|
- [x] Evaluate the completed v36 YOLO11x checkpoint calibration-first on the rotated v30 holdouts; reject it before opening test/background because the regional calibration gate failed.
|
||||||
- [ ] Finish and assess the leak-free v37 YOLO11x failure-driven CUDA iteration; open test/background evidence only if every calibration gate passes.
|
- [ ] Finish and assess the leak-free v37 YOLO11x failure-driven CUDA iteration; open test/background evidence only if every calibration gate passes.
|
||||||
|
|
||||||
|
|||||||
@@ -131,6 +131,24 @@ def training_command(
|
|||||||
return command
|
return command
|
||||||
|
|
||||||
|
|
||||||
|
def failure_sampling_command(
|
||||||
|
*,
|
||||||
|
scripts_dir: Path,
|
||||||
|
train_summary: Path,
|
||||||
|
corpus_manifest: Path,
|
||||||
|
assessment: Path,
|
||||||
|
output_dir: Path,
|
||||||
|
) -> list[str]:
|
||||||
|
return [
|
||||||
|
sys.executable,
|
||||||
|
str(scripts_dir / "build_failure_driven_yolo_sampling.py"),
|
||||||
|
"--summary", str(train_summary),
|
||||||
|
"--corpus-manifest", str(corpus_manifest),
|
||||||
|
"--assessment", str(assessment),
|
||||||
|
"--output-dir", str(output_dir),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def run(command: list[str], log_path: Path | None = None, *, allowed: set[int] = {0}) -> int:
|
def run(command: list[str], log_path: Path | None = None, *, allowed: set[int] = {0}) -> int:
|
||||||
if log_path:
|
if log_path:
|
||||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
@@ -147,6 +165,7 @@ def main() -> int:
|
|||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("--initial-model", type=Path, required=True)
|
parser.add_argument("--initial-model", type=Path, required=True)
|
||||||
parser.add_argument("--train-yaml", type=Path, required=True)
|
parser.add_argument("--train-yaml", type=Path, required=True)
|
||||||
|
parser.add_argument("--train-summary", type=Path, required=True)
|
||||||
parser.add_argument("--dataset-audit", type=Path, required=True)
|
parser.add_argument("--dataset-audit", type=Path, required=True)
|
||||||
parser.add_argument("--calibration-summary", type=Path, required=True)
|
parser.add_argument("--calibration-summary", type=Path, required=True)
|
||||||
parser.add_argument("--test-summary", type=Path, required=True)
|
parser.add_argument("--test-summary", type=Path, required=True)
|
||||||
@@ -195,6 +214,7 @@ def main() -> int:
|
|||||||
state = json.loads(state_path.read_text(encoding="utf-8"))
|
state = json.loads(state_path.read_text(encoding="utf-8"))
|
||||||
state["status"] = "running"
|
state["status"] = "running"
|
||||||
model = Path(state.get("next_model") or args.initial_model)
|
model = Path(state.get("next_model") or args.initial_model)
|
||||||
|
train_yaml = Path(state.get("next_train_yaml") or args.train_yaml)
|
||||||
first_index = len(state["iterations"]) + 1
|
first_index = len(state["iterations"]) + 1
|
||||||
scripts_dir = Path(__file__).resolve().parent
|
scripts_dir = Path(__file__).resolve().parent
|
||||||
|
|
||||||
@@ -206,7 +226,7 @@ def main() -> int:
|
|||||||
command = training_command(
|
command = training_command(
|
||||||
args.yolo,
|
args.yolo,
|
||||||
model=model,
|
model=model,
|
||||||
data=args.train_yaml,
|
data=train_yaml,
|
||||||
project=args.output_dir / "runs",
|
project=args.output_dir / "runs",
|
||||||
name=name,
|
name=name,
|
||||||
epochs=args.epochs,
|
epochs=args.epochs,
|
||||||
@@ -339,7 +359,27 @@ def main() -> int:
|
|||||||
write_json(state_path, state)
|
write_json(state_path, state)
|
||||||
print(json.dumps(state, indent=2))
|
print(json.dumps(state, indent=2))
|
||||||
return 0
|
return 0
|
||||||
|
sampling_dir = iteration_dir / "failure-driven-training"
|
||||||
|
run(
|
||||||
|
failure_sampling_command(
|
||||||
|
scripts_dir=scripts_dir,
|
||||||
|
train_summary=args.train_summary,
|
||||||
|
corpus_manifest=args.corpus_manifest,
|
||||||
|
assessment=assessment,
|
||||||
|
output_dir=sampling_dir,
|
||||||
|
),
|
||||||
|
iteration_dir / "failure-driven-sampling.log",
|
||||||
|
)
|
||||||
|
sampling_evidence = sampling_dir / "failure-driven-sampling.json"
|
||||||
|
next_train_yaml = sampling_dir / "dataset.yaml"
|
||||||
|
if not sampling_evidence.is_file() or not next_train_yaml.is_file():
|
||||||
|
raise RuntimeError("Failure-driven sampling produced incomplete evidence")
|
||||||
|
record["failure_driven_sampling"] = str(sampling_evidence)
|
||||||
|
record["failure_driven_sampling_sha256"] = sha256(sampling_evidence)
|
||||||
|
record["next_train_yaml"] = str(next_train_yaml)
|
||||||
|
state["next_train_yaml"] = str(next_train_yaml)
|
||||||
model = candidate
|
model = candidate
|
||||||
|
train_yaml = next_train_yaml
|
||||||
write_json(state_path, state)
|
write_json(state_path, state)
|
||||||
|
|
||||||
state["status"] = "continue_training_loop"
|
state["status"] = "continue_training_loop"
|
||||||
|
|||||||
Reference in New Issue
Block a user