agrobot_base/Python/OAK/datasets/oak-fcc-3/audit_radiometric_normaliza...

539 lines
19 KiB
Python

#!/usr/bin/env python3
"""Audita a normalizacao radiometrica do OAK-FFC-3 usando os meta.json.
O calculo replica o contrato do RawProcessorCore:
factor_actual = exposure_time_us * (sensitivity_iso / iso_base)
factor_ref = ref_exposure_time_us * (ref_iso / iso_base)
raw_scale = factor_ref / factor_actual
applied_scale = clip(raw_scale, scale_min, scale_max)
O script nao abre os .bin e, portanto, mede o risco de clamp do fator, nao a
saturacao real dos pixels. Ele gera um CSV por frame e um resumo JSON.
"""
from __future__ import annotations
import argparse
import csv
import json
import math
import sys
from collections import defaultdict
from pathlib import Path
from typing import Any, Iterable
ROLES = ("rgb", "re", "nir")
ROLE_FALLBACK = {"CAM_A": "rgb", "CAM_B": "re", "CAM_C": "nir"}
PERCENTILES = (1, 5, 10, 25, 50, 75, 90, 95, 99)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Audita exposure/ISO e fatores da radiometric_normalization."
)
parser.add_argument(
"--meta-root",
type=Path,
default=Path("dataset/brutas/group"),
help="Raiz contendo os grupos e suas pastas metas (padrao: dataset/brutas/group).",
)
parser.add_argument(
"--module-params",
type=Path,
default=Path("calibration/module_params.json"),
help="JSON com radiometric_normalization.",
)
parser.add_argument(
"--pattern",
default="*.json",
help="Padrao dos metadados dentro das pastas metas (padrao: *.json).",
)
parser.add_argument(
"--out-dir",
type=Path,
default=Path("radiometric_audit"),
help="Pasta dos relatorios.",
)
parser.add_argument(
"--recursive-anywhere",
action="store_true",
help="Procura JSON em qualquer subpasta; por padrao prioriza <grupo>/metas/.",
)
return parser.parse_args()
def load_json(path: Path) -> dict[str, Any]:
with path.open("r", encoding="utf-8-sig") as handle:
data = json.load(handle)
if not isinstance(data, dict):
raise ValueError("raiz JSON nao e objeto")
return data
def finite_float(value: Any, default: float | None = None) -> float | None:
try:
result = float(value)
except (TypeError, ValueError):
return default
return result if math.isfinite(result) else default
def percentile(values: list[float], q: float) -> float | None:
if not values:
return None
ordered = sorted(values)
if len(ordered) == 1:
return float(ordered[0])
position = (len(ordered) - 1) * float(q) / 100.0
lo = int(math.floor(position))
hi = int(math.ceil(position))
if lo == hi:
return float(ordered[lo])
fraction = position - lo
return float(ordered[lo] * (1.0 - fraction) + ordered[hi] * fraction)
def describe(values: Iterable[float]) -> dict[str, Any]:
vals = [float(v) for v in values if math.isfinite(float(v))]
if not vals:
return {"count": 0}
mean = sum(vals) / len(vals)
variance = sum((v - mean) ** 2 for v in vals) / len(vals)
result: dict[str, Any] = {
"count": len(vals),
"min": min(vals),
"max": max(vals),
"mean": mean,
"std": math.sqrt(max(variance, 0.0)),
}
result.update({f"p{q:02d}": percentile(vals, q) for q in PERCENTILES})
return result
def get_nested_stream_meta(meta: dict[str, Any]) -> dict[str, Any]:
candidates = [
meta.get("stream_meta"),
meta.get("meta"),
meta,
]
for candidate in candidates:
if not isinstance(candidate, dict):
continue
nested = candidate.get("stream_meta")
if isinstance(nested, dict) and isinstance(nested.get("frame_controls"), dict):
return nested
if isinstance(candidate.get("frame_controls"), dict):
return candidate
return {}
def role_map_from_stream(stream_meta: dict[str, Any]) -> dict[str, str]:
mapping: dict[str, str] = {}
camera_info = stream_meta.get("camera_info", {})
if isinstance(camera_info, dict):
for cam_id, info in camera_info.items():
if not isinstance(info, dict):
continue
role = str(info.get("role", "")).strip().lower()
if role in ROLES:
mapping[str(cam_id).upper()] = role
for cam_id, role in ROLE_FALLBACK.items():
mapping.setdefault(cam_id, role)
return mapping
def infer_group(meta_path: Path, root: Path) -> str:
try:
relative = meta_path.relative_to(root)
except ValueError:
return "unknown"
parts = relative.parts
if "metas" in parts:
index = parts.index("metas")
if index > 0:
return parts[index - 1]
return parts[0] if len(parts) > 1 else "root"
def discover_meta_files(root: Path, pattern: str, anywhere: bool) -> list[Path]:
if root.is_file():
return [root]
if not root.is_dir():
raise FileNotFoundError(f"Raiz de metadados nao encontrada: {root}")
if anywhere:
return sorted(p for p in root.rglob(pattern) if p.is_file())
files = sorted(p for p in root.glob(f"*/metas/{pattern}") if p.is_file())
if files:
return files
return sorted(p for p in root.rglob(pattern) if p.is_file())
def read_contract(module_params: dict[str, Any]) -> dict[str, Any]:
cfg = module_params.get("radiometric_normalization", {})
if not isinstance(cfg, dict):
raise ValueError("radiometric_normalization ausente ou invalido")
iso_base = finite_float(cfg.get("iso_base"), 100.0) or 100.0
references = cfg.get("reference_controls", {})
limits = cfg.get("scale_limits", {})
if not isinstance(references, dict):
references = {}
if not isinstance(limits, dict):
limits = {}
roles: dict[str, Any] = {}
for role in ROLES:
ref = references.get(role, {})
if not isinstance(ref, dict):
ref = {}
exp = finite_float(ref.get("exposure_time_us"), 0.0) or 0.0
iso = finite_float(ref.get("sensitivity_iso"), iso_base) or iso_base
reference_factor = exp * (iso / iso_base)
role_limits = limits.get(role)
if not isinstance(role_limits, dict):
role_limits = limits.get("default", {})
if not isinstance(role_limits, dict):
role_limits = {}
scale_min = finite_float(role_limits.get("min"), 0.15) or 0.15
scale_max = finite_float(role_limits.get("max"), 6.0) or 6.0
if scale_min <= 0:
scale_min = 0.001
if scale_max < scale_min:
scale_max = scale_min
roles[role] = {
"reference_exposure_time_us": exp,
"reference_sensitivity_iso": iso,
"reference_factor": reference_factor,
"scale_min": scale_min,
"scale_max": scale_max,
}
return {
"enabled": bool(cfg.get("enabled", False)),
"method": cfg.get("method"),
"iso_base": iso_base,
"roles": roles,
}
def extract_role_controls(meta: dict[str, Any]) -> tuple[dict[str, dict[str, Any]], dict[str, Any]]:
stream = get_nested_stream_meta(meta)
controls = stream.get("frame_controls", {})
if not isinstance(controls, dict):
controls = {}
mapping = role_map_from_stream(stream)
by_role: dict[str, dict[str, Any]] = {}
for cam_id, control in controls.items():
if not isinstance(control, dict):
continue
role = mapping.get(str(cam_id).upper(), "")
if role in ROLES:
by_role[role] = control
info = {
"frame_id": stream.get("frame_id", meta.get("frame_id")),
"sync_ok": stream.get("sync_ok", meta.get("sync_ok")),
"sync_dt_ms": stream.get("sync_dt_ms", meta.get("sync_dt_ms")),
"timestamp": meta.get("ts", stream.get("ts")),
}
return by_role, info
def compute_role_row(control: dict[str, Any], contract: dict[str, Any]) -> dict[str, Any] | None:
exp = finite_float(control.get("exposure_time_us"))
iso = finite_float(control.get("sensitivity_iso"))
gain_est = finite_float(control.get("analogue_gain_est"))
gain = finite_float(control.get("analogue_gain"))
if exp is None or exp <= 0:
return None
iso_base = float(contract["iso_base"])
if iso is not None:
gain_factor = iso / iso_base
gain_source = "sensitivity_iso"
elif gain_est is not None:
gain_factor = gain_est
gain_source = "analogue_gain_est"
elif gain is not None:
gain_factor = gain
gain_source = "analogue_gain"
else:
gain_factor = 1.0
gain_source = "unity_fallback"
if not math.isfinite(gain_factor) or gain_factor <= 0:
gain_factor = 1.0
gain_source = "unity_invalid_fallback"
actual_factor = exp * gain_factor
reference_factor = float(contract["reference_factor"])
raw_scale = reference_factor / actual_factor if actual_factor > 0 else math.nan
scale_min = float(contract["scale_min"])
scale_max = float(contract["scale_max"])
applied_scale = min(max(raw_scale, scale_min), scale_max)
tolerance = max(1e-9, 1e-9 * max(abs(raw_scale), abs(scale_max), 1.0))
return {
"exposure_time_us": exp,
"sensitivity_iso": iso,
"analogue_gain_est": gain_est,
"gain_factor_used": gain_factor,
"gain_source": gain_source,
"actual_factor": actual_factor,
"reference_factor": reference_factor,
"raw_scale": raw_scale,
"applied_scale": applied_scale,
"scale_min": scale_min,
"scale_max": scale_max,
"clipped_low": raw_scale < scale_min - tolerance,
"clipped_high": raw_scale > scale_max + tolerance,
"at_low_limit": abs(applied_scale - scale_min) <= tolerance,
"at_high_limit": abs(applied_scale - scale_max) <= tolerance,
}
def empty_accumulator() -> dict[str, list[float] | int]:
return {
"exposure_time_us": [],
"sensitivity_iso": [],
"gain_factor_used": [],
"actual_factor": [],
"raw_scale": [],
"applied_scale": [],
"clipped_low": 0,
"clipped_high": 0,
}
def add_to_accumulator(acc: dict[str, Any], row: dict[str, Any]) -> None:
for key in (
"exposure_time_us",
"sensitivity_iso",
"gain_factor_used",
"actual_factor",
"raw_scale",
"applied_scale",
):
value = row.get(key)
if value is not None and math.isfinite(float(value)):
acc[key].append(float(value))
acc["clipped_low"] += int(bool(row.get("clipped_low")))
acc["clipped_high"] += int(bool(row.get("clipped_high")))
def summarize_accumulator(acc: dict[str, Any], contract: dict[str, Any]) -> dict[str, Any]:
actual = list(acc["actual_factor"])
count = len(actual)
clipped_low = int(acc["clipped_low"])
clipped_high = int(acc["clipped_high"])
scale_min = float(contract["scale_min"])
scale_max = float(contract["scale_max"])
p01 = percentile(actual, 1)
p50 = percentile(actual, 50)
p99 = percentile(actual, 99)
feasible_low = scale_min * p99 if p99 is not None else None
feasible_high = scale_max * p01 if p01 is not None else None
return {
"count": count,
"contract": contract,
"exposure_time_us": describe(acc["exposure_time_us"]),
"sensitivity_iso": describe(acc["sensitivity_iso"]),
"gain_factor_used": describe(acc["gain_factor_used"]),
"actual_factor": describe(actual),
"raw_scale": describe(acc["raw_scale"]),
"applied_scale": describe(acc["applied_scale"]),
"clipping": {
"low_count": clipped_low,
"low_pct": 100.0 * clipped_low / count if count else 0.0,
"high_count": clipped_high,
"high_pct": 100.0 * clipped_high / count if count else 0.0,
"any_count": clipped_low + clipped_high,
"any_pct": 100.0 * (clipped_low + clipped_high) / count if count else 0.0,
},
"recommendation": {
"median_actual_factor": p50,
"suggested_reference_exposure_us_at_iso100": p50,
"central_98pct_feasible_reference_factor_min": feasible_low,
"central_98pct_feasible_reference_factor_max": feasible_high,
"central_98pct_has_feasible_interval": (
feasible_low is not None
and feasible_high is not None
and feasible_low <= feasible_high
),
"note": (
"A mediana coloca a escala mediana em 1.0. Nao aplique automaticamente: "
"valide tambem a saturacao dos pixels e preserve a calibracao cruzada entre bandas."
),
},
}
def json_safe(value: Any) -> Any:
if isinstance(value, dict):
return {str(k): json_safe(v) for k, v in value.items()}
if isinstance(value, list):
return [json_safe(v) for v in value]
if isinstance(value, float) and not math.isfinite(value):
return None
return value
def main() -> int:
args = parse_args()
module_params = load_json(args.module_params)
contract = read_contract(module_params)
meta_files = discover_meta_files(args.meta_root, args.pattern, args.recursive_anywhere)
if not meta_files:
print(f"[ERRO] Nenhum meta encontrado em {args.meta_root}", file=sys.stderr)
return 2
args.out_dir.mkdir(parents=True, exist_ok=True)
rows: list[dict[str, Any]] = []
skipped: list[dict[str, Any]] = []
global_acc = {role: empty_accumulator() for role in ROLES}
group_acc: dict[str, dict[str, dict[str, Any]]] = defaultdict(
lambda: {role: empty_accumulator() for role in ROLES}
)
samples_complete = 0
samples_with_valid_controls = 0
samples_any_clipped = 0
for meta_path in meta_files:
group = infer_group(meta_path, args.meta_root)
try:
meta = load_json(meta_path)
controls_by_role, info = extract_role_controls(meta)
except Exception as exc:
skipped.append({"path": str(meta_path), "reason": f"json_error:{exc}"})
continue
if not controls_by_role:
skipped.append({"path": str(meta_path), "reason": "missing_frame_controls"})
continue
samples_with_valid_controls += 1
complete = all(role in controls_by_role for role in ROLES)
samples_complete += int(complete)
frame_any_clipped = False
for role in ROLES:
control = controls_by_role.get(role)
if control is None:
skipped.append({"path": str(meta_path), "reason": f"missing_role:{role}"})
continue
role_row = compute_role_row(control, {"iso_base": contract["iso_base"], **contract["roles"][role]})
if role_row is None:
skipped.append({"path": str(meta_path), "reason": f"invalid_controls:{role}"})
continue
frame_any_clipped = frame_any_clipped or bool(
role_row["clipped_low"] or role_row["clipped_high"]
)
add_to_accumulator(global_acc[role], role_row)
add_to_accumulator(group_acc[group][role], role_row)
rows.append(
{
"group": group,
"meta_path": str(meta_path),
"timestamp": info.get("timestamp"),
"frame_id": info.get("frame_id"),
"sync_ok": info.get("sync_ok"),
"sync_dt_ms": info.get("sync_dt_ms"),
"role": role,
**role_row,
}
)
samples_any_clipped += int(frame_any_clipped)
report = {
"meta_root": str(args.meta_root.resolve()),
"module_params": str(args.module_params.resolve()),
"radiometric_normalization_enabled": contract["enabled"],
"method": contract["method"],
"iso_base": contract["iso_base"],
"files_discovered": len(meta_files),
"files_with_any_valid_controls": len({row["meta_path"] for row in rows}),
"complete_rgb_re_nir_samples": samples_complete,
"samples_with_valid_controls": samples_with_valid_controls,
"samples_with_any_clipped_role": samples_any_clipped,
"samples_with_any_clipped_role_pct": (
100.0 * samples_any_clipped / samples_with_valid_controls
if samples_with_valid_controls else 0.0
),
"skipped_records": len(skipped),
"global": {
role: summarize_accumulator(global_acc[role], contract["roles"][role])
for role in ROLES
},
"groups": {
group: {
role: summarize_accumulator(acc_by_role[role], contract["roles"][role])
for role in ROLES
}
for group, acc_by_role in sorted(group_acc.items())
},
}
rows_path = args.out_dir / "radiometric_audit_rows.csv"
summary_path = args.out_dir / "radiometric_audit_summary.json"
skipped_path = args.out_dir / "radiometric_audit_skipped.csv"
fieldnames = [
"group", "meta_path", "timestamp", "frame_id", "sync_ok", "sync_dt_ms", "role",
"exposure_time_us", "sensitivity_iso", "analogue_gain_est", "gain_factor_used",
"gain_source", "actual_factor", "reference_factor", "raw_scale", "applied_scale",
"scale_min", "scale_max", "clipped_low", "clipped_high", "at_low_limit", "at_high_limit",
]
with rows_path.open("w", newline="", encoding="utf-8-sig") as handle:
writer = csv.DictWriter(handle, fieldnames=fieldnames, extrasaction="ignore")
writer.writeheader()
writer.writerows(rows)
with skipped_path.open("w", newline="", encoding="utf-8-sig") as handle:
writer = csv.DictWriter(handle, fieldnames=["path", "reason"])
writer.writeheader()
writer.writerows(skipped)
with summary_path.open("w", encoding="utf-8") as handle:
json.dump(json_safe(report), handle, ensure_ascii=False, indent=2)
print("=" * 64)
print("Auditoria da normalizacao radiometrica")
print(f"Metas encontrados : {len(meta_files)}")
print(f"Amostras completas: {samples_complete}")
print(f"Com algum clamp : {samples_any_clipped}")
print("-" * 64)
for role in ROLES:
summary = report["global"][role]
scale = summary["applied_scale"]
clipping = summary["clipping"]
rec = summary["recommendation"]
print(
f"{role.upper():>3} | n={summary['count']:5d} "
f"scale_med={scale.get('p50', 0):7.3f} "
f"scale_p05={scale.get('p05', 0):7.3f} "
f"scale_p95={scale.get('p95', 0):7.3f} "
f"clamp_low={clipping['low_pct']:6.2f}% "
f"clamp_high={clipping['high_pct']:6.2f}% "
f"ref_atual={summary['contract']['reference_factor']:.1f} "
f"ref_mediana={float(rec['median_actual_factor'] or 0):.1f}"
)
print("-" * 64)
print(f"Rows : {rows_path}")
print(f"Summary : {summary_path}")
print(f"Skipped : {skipped_path}")
print("=" * 64)
return 0
if __name__ == "__main__":
raise SystemExit(main())