diff --git a/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_eepromToJson.json b/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_eepromToJson.json new file mode 100644 index 000000000..06ace2607 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_eepromToJson.json @@ -0,0 +1,247 @@ +{ + "batchName": "", + "batchTime": 1756192307, + "boardConf": "nIR-C00M05-00", + "boardCustom": "", + "boardName": "DM1090", + "boardOptions": 0, + "boardRev": "R3M0E3", + "cameraData": [ + [ + 1, + { + "cameraType": 0, + "distortionCoeff": [ + -0.2667059302330017, + 0.24688485264778137, + -0.000222075788769871, + 0.00012030570360366255, + -0.039670947939157486, + -0.26931697130203247, + 0.24513530731201172, + -0.039285946637392044, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "extrinsics": { + "rotationMatrix": [ + [ + 0.9999990463256836, + -0.00016664451686665416, + -0.0013813182013109326 + ], + [ + 0.00011326302046654746, + 0.999256432056427, + -0.038555704057216644 + ], + [ + 0.0013867162633687258, + 0.038555510342121124, + 0.9992554783821106 + ] + ], + "specTranslation": { + "x": 2.5, + "y": 0.0, + "z": 0.0 + }, + "toCameraSocket": 2, + "translation": { + "x": 2.3664393424987793, + "y": 0.2041260451078415, + "z": 8.34726333618164 + } + }, + "height": 800, + "intrinsicMatrix": [ + [ + 329.8727111816406, + 0.0, + 655.0603637695312 + ], + [ + 0.0, + 329.14971923828125, + 353.3602600097656 + ], + [ + 0.0, + 0.0, + 1.0 + ] + ], + "lensPosition": 0, + "specHfovDeg": 70.9000015258789, + "width": 1280 + } + ], + [ + 2, + { + "cameraType": 0, + "distortionCoeff": [ + 2.599353075027466, + 1.7201688289642334, + -0.0002882175613194704, + 0.0005083332071080804, + -2.265909433364868, + 2.5841195583343506, + 1.6573377847671509, + -2.2209253311157227, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "extrinsics": { + "rotationMatrix": [ + [ + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0 + ] + ], + "specTranslation": { + "x": -0.0, + "y": -0.0, + "z": -0.0 + }, + "toCameraSocket": -1, + "translation": { + "x": 0.0, + "y": 0.0, + "z": 0.0 + } + }, + "height": 800, + "intrinsicMatrix": [ + [ + 577.9281005859375, + 0.0, + 644.1484375 + ], + [ + 0.0, + 577.5459594726562, + 379.2917175292969 + ], + [ + 0.0, + 0.0, + 1.0 + ] + ], + "lensPosition": 0, + "specHfovDeg": 70.9000015258789, + "width": 1280 + } + ] + ], + "deviceName": "", + "hardwareConf": "F0-FV00-BC000", + "housingExtrinsics": { + "rotationMatrix": [], + "specTranslation": { + "x": 0.0, + "y": 0.0, + "z": 0.0 + }, + "toCameraSocket": -1, + "translation": { + "x": 0.0, + "y": 0.0, + "z": 0.0 + } + }, + "imuExtrinsics": { + "rotationMatrix": [ + [ + 1.0, + 0.0, + 0.0 + ], + [ + 0.0, + 1.0, + 0.0 + ], + [ + 0.0, + 0.0, + 1.0 + ] + ], + "specTranslation": { + "x": 0.0, + "y": 0.0, + "z": 0.0 + }, + "toCameraSocket": -1, + "translation": { + "x": 0.0, + "y": 0.0, + "z": 0.0 + } + }, + "miscellaneousData": [], + "productName": "OAK-FFC-3P", + "stereoEnableDistortionCorrection": false, + "stereoRectificationData": { + "leftCameraSocket": 1, + "rectifiedRotationLeft": [ + [ + 0.27401065826416016, + 0.06054104119539261, + 0.9598191976547241 + ], + [ + -0.0419992059469223, + 0.9978178143501282, + -0.050947822630405426 + ], + [ + -0.9608091711997986, + -0.02635139599442482, + 0.2759554088115692 + ] + ], + "rectifiedRotationRight": [ + [ + 0.2726745009422302, + 0.02352055534720421, + 0.9618188142776489 + ], + [ + -0.04209506884217262, + 0.9990354776382446, + -0.012496757321059704 + ], + [ + -0.9611850380897522, + -0.03708028048276901, + 0.2734015882015228 + ] + ], + "rightCameraSocket": 2 + }, + "stereoUseSpecTranslation": true, + "version": 7, + "verticalCameraSocket": -1 +} \ No newline at end of file diff --git a/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_probe_report.json b/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_probe_report.json new file mode 100644 index 000000000..6ab3da466 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/calibration/calibration_probe_out/calibration_probe_report.json @@ -0,0 +1,282 @@ +{ + "created_at": "2026-05-25 13:57:51", + "device_id": "194430108133AC2F00", + "calibration_read_method": "readCalibration", + "connected_cameras": [ + { + "socket": "CAM_A", + "sensorName": "OV9782", + "width": 1280, + "height": 800, + "orientation": "CameraImageOrientation.AUTO", + "supportedTypes": [ + "CameraSensorType.COLOR", + "CameraSensorType.MONO" + ] + }, + { + "socket": "CAM_B", + "sensorName": "OV9282", + "width": 1280, + "height": 800, + "orientation": "CameraImageOrientation.AUTO", + "supportedTypes": [ + "CameraSensorType.MONO", + "CameraSensorType.COLOR" + ] + }, + { + "socket": "CAM_C", + "sensorName": "OV9282", + "width": 1280, + "height": 800, + "orientation": "CameraImageOrientation.AUTO", + "supportedTypes": [ + "CameraSensorType.MONO", + "CameraSensorType.COLOR" + ] + } + ], + "sockets_requested": [ + "CAM_A", + "CAM_B", + "CAM_C" + ], + "width": 1280, + "height": 800, + "calibration_dump": { + "available": true, + "method": "eepromToJson", + "path": "calibration\\calibration_probe_out\\calibration_eepromToJson.json", + "error": null + }, + "sockets": [ + { + "socket": "CAM_A", + "intrinsics": null, + "intrinsics_method": "failed:getCameraIntrinsics", + "intrinsics_ok": false, + "distortion": null, + "distortion_method": "missing:distortion", + "distortion_ok": false, + "fov_deg": null, + "fov_method": "missing:getFov" + }, + { + "socket": "CAM_B", + "intrinsics": [ + [ + 329.8727111816406, + 0.0, + 655.0603637695312 + ], + [ + 0.0, + 329.14971923828125, + 353.3602600097656 + ], + [ + 0.0, + 0.0, + 1.0 + ] + ], + "intrinsics_method": "getCameraIntrinsics3args", + "intrinsics_ok": true, + "distortion": [ + -0.2667059302330017, + 0.24688485264778137, + -0.000222075788769871, + 0.00012030570360366255, + -0.039670947939157486, + -0.26931697130203247, + 0.24513530731201172, + -0.039285946637392044, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "distortion_method": "getDistortionCoefficients", + "distortion_ok": true, + "fov_deg": 70.9000015258789, + "fov_method": "getFov" + }, + { + "socket": "CAM_C", + "intrinsics": [ + [ + 577.9281005859375, + 0.0, + 644.1484375 + ], + [ + 0.0, + 577.5459594726562, + 379.2917175292969 + ], + [ + 0.0, + 0.0, + 1.0 + ] + ], + "intrinsics_method": "getCameraIntrinsics3args", + "intrinsics_ok": true, + "distortion": [ + 2.599353075027466, + 1.7201688289642334, + -0.0002882175613194704, + 0.0005083332071080804, + -2.265909433364868, + 2.5841195583343506, + 1.6573377847671509, + -2.2209253311157227, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "distortion_method": "getDistortionCoefficients", + "distortion_ok": true, + "fov_deg": 70.9000015258789, + "fov_method": "getFov" + } + ], + "pairs": [ + { + "pair": "CAM_A->CAM_B", + "src": "CAM_A", + "dst": "CAM_B", + "extrinsics": null, + "extrinsics_method": "failed:getCameraExtrinsics", + "extrinsics_ok": false, + "translation": null, + "baseline": null, + "baseline_method": "missing:getBaselineDistance", + "baseline_ok": false + }, + { + "pair": "CAM_A->CAM_C", + "src": "CAM_A", + "dst": "CAM_C", + "extrinsics": null, + "extrinsics_method": "failed:getCameraExtrinsics", + "extrinsics_ok": false, + "translation": null, + "baseline": null, + "baseline_method": "missing:getBaselineDistance", + "baseline_ok": false + }, + { + "pair": "CAM_B->CAM_A", + "src": "CAM_B", + "dst": "CAM_A", + "extrinsics": null, + "extrinsics_method": "failed:getCameraExtrinsics", + "extrinsics_ok": false, + "translation": null, + "baseline": null, + "baseline_method": "missing:getBaselineDistance", + "baseline_ok": false + }, + { + "pair": "CAM_B->CAM_C", + "src": "CAM_B", + "dst": "CAM_C", + "extrinsics": [ + [ + 0.9999990463256836, + -0.00016664451686665416, + -0.0013813182013109326, + 2.3664393424987793 + ], + [ + 0.00011326302046654746, + 0.999256432056427, + -0.038555704057216644, + 0.2041260451078415 + ], + [ + 0.0013867162633687258, + 0.038555510342121124, + 0.9992554783821106, + 8.34726333618164 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ], + "extrinsics_method": "getCameraExtrinsics2args", + "extrinsics_ok": true, + "translation": [ + 2.3664393424987793, + 0.2041260451078415, + 8.34726333618164 + ], + "baseline": 2.5, + "baseline_method": "getBaselineDistance2args", + "baseline_ok": true + }, + { + "pair": "CAM_C->CAM_A", + "src": "CAM_C", + "dst": "CAM_A", + "extrinsics": null, + "extrinsics_method": "failed:getCameraExtrinsics", + "extrinsics_ok": false, + "translation": null, + "baseline": null, + "baseline_method": "missing:getBaselineDistance", + "baseline_ok": false + }, + { + "pair": "CAM_C->CAM_B", + "src": "CAM_C", + "dst": "CAM_B", + "extrinsics": [ + [ + 0.9999990463256836, + 0.00011326302046654746, + 0.0013867162633687258, + -2.378035545349121 + ], + [ + -0.00016664451686665416, + 0.999256432056427, + 0.038555510342121124, + -0.525412917137146 + ], + [ + -0.0013813182013109326, + -0.038555704057216644, + 0.9992554783821106, + -8.329909324645996 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ], + "extrinsics_method": "getCameraExtrinsics2args", + "extrinsics_ok": true, + "translation": [ + -2.378035545349121, + -0.525412917137146, + -8.329909324645996 + ], + "baseline": 2.5, + "baseline_method": "getBaselineDistance2args", + "baseline_ok": true + } + ] +} \ No newline at end of file diff --git a/Python/OAK/datasets/oak-fcc-3/calibration/module_params.json b/Python/OAK/datasets/oak-fcc-3/calibration/module_params.json index 860780c90..42a2f14dd 100644 --- a/Python/OAK/datasets/oak-fcc-3/calibration/module_params.json +++ b/Python/OAK/datasets/oak-fcc-3/calibration/module_params.json @@ -588,15 +588,15 @@ "reference_mode": "fixed", "reference_controls": { "rgb": { - "exposure_time_us": 2200, + "exposure_time_us": 3000, "sensitivity_iso": 100 }, "re": { - "exposure_time_us": 2500, + "exposure_time_us": 4000, "sensitivity_iso": 100 }, "nir": { - "exposure_time_us": 2500, + "exposure_time_us": 4000, "sensitivity_iso": 100 } }, @@ -612,11 +612,11 @@ }, "re": { "min": 0.15, - "max": 2.5 + "max": 3.0 }, "nir": { "min": 0.15, - "max": 2.5 + "max": 3.0 } }, @@ -679,7 +679,7 @@ "rgb_saturation_threshold": 0.97 }, "rgb_calibration": { - "enabled": true, + "enabled": false, "gains": { "R": 1.061, "G": 1.0, diff --git a/Python/OAK/datasets/oak-fcc-3/config.json b/Python/OAK/datasets/oak-fcc-3/config.json index 8c2bd90ad..d08fd5444 100644 --- a/Python/OAK/datasets/oak-fcc-3/config.json +++ b/Python/OAK/datasets/oak-fcc-3/config.json @@ -1,7 +1,7 @@ { "camera": "oak-fcc-3", "modelo": "segformer_b1", - "model_name": "target_aug", + "model_name": "target_fixed", "main_class_name": "cana", "es_classes": "", "model_to_use": "geral", diff --git a/Python/OAK/datasets/oak-fcc-3/core/oak_fcc3_manager.py b/Python/OAK/datasets/oak-fcc-3/core/oak_fcc3_manager.py index 4827b9006..536be8425 100644 --- a/Python/OAK/datasets/oak-fcc-3/core/oak_fcc3_manager.py +++ b/Python/OAK/datasets/oak-fcc-3/core/oak_fcc3_manager.py @@ -878,6 +878,19 @@ class OakFcc3Manager: self.has_imu_pipeline = False self.running = False + def _is_fatal_depthai_error(self, erro): + txt = str(erro) + + sinais = [ + "X_LINK_ERROR", + "Communication exception", + "Couldn't read data from stream", + "Device already closed", + "device has been closed", + ] + + return any(s in txt for s in sinais) + # ============================================================ # Status # ============================================================ @@ -924,7 +937,15 @@ class OakFcc3Manager: def get_next_frame(self, timeout=1.0): if not self.running: - raise RuntimeError("OakFcc3Manager não está rodando. Chame start() primeiro.") + last_error = None + try: + last_error = self._capture_thread_stats.get("last_error") + except Exception: + pass + + raise RuntimeError( + f"OakFcc3Manager não está rodando. Último erro: {last_error}" + ) # Fallback síncrono se desligar async. if not bool(getattr(self, "async_capture_enabled", True)): @@ -1785,10 +1806,34 @@ class OakFcc3Manager: self._capture_cond.notify_all() except Exception as e: + erro = f"{type(e).__name__}: {e}" + try: - self._capture_thread_stats["last_error"] = f"{type(e).__name__}: {e}" + self._capture_thread_stats["last_error"] = erro except Exception: pass + + if self._is_fatal_depthai_error(e): + try: + self._capture_thread_stats["fatal_error"] = True + except Exception: + pass + + self.running = False + + try: + self._capture_stop_event.set() + except Exception: + pass + + try: + with self._capture_cond: + self._capture_cond.notify_all() + except Exception: + pass + + break + time.sleep(0.005) try: diff --git a/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset.py b/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset.py new file mode 100644 index 000000000..c13746b49 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset.py @@ -0,0 +1,1836 @@ +import argparse +import csv +import json +import math +import unicodedata +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, List, Optional, Tuple, Any + +import cv2 +import numpy as np +import shutil + +try: + from core.raw_processor_core import RawProcessorCore +except Exception: + RawProcessorCore = None + + +# ============================================================ +# Auditoria final de dataset multiespectral OAK-FCC-3P +# ------------------------------------------------------------ +# Foco: tensor final pronto para o modelo, CHW float32: +# [R, G, B, RE, NIR] +# +# Gera: +# - audit_summary.json +# - audit_samples.csv +# - audit_by_class_channel.csv +# - audit_by_sample_class_channel.csv +# - audit_warnings.json +# - visuals/*.png com painéis de sanidade +# +# Layout esperado: +# dataset_root/ +# metas/*.json +# bins/*.raw ou *.bin +# masks/*.png/.tif/.npy +# previews/*.png opcional +# +# Também aceita apontar diretamente para dataset_root/metas etc. +# +# python -m utils.audit_dataset --input_path .\dataset\original\group\ --out_dir audit_multispec_out --save-visuals --visual-every 10 --build-fixed-dataset --save-rejected-previews --rejected-preview-source auto --rejected-preview-max-width 640 --clean-reject-low-corr-only-if-shift-bad --clean-max-abs-shift-px 22 --clean-max-dev-shift-px 7 --clean-min-edge-corr 0.045 --clean-min-target-pct 0.0025 +# +# ============================================================ + + +CHANNELS = ["R", "G", "B", "RE", "NIR"] +DERIVED = [ + "NDVI", + "NDRE", + "NIR_minus_RE", + "NIR_over_RE", + "NIR_over_R", + "RE_over_R", + "NIR_over_G", + "RE_over_G", + "G_minus_R", +] +ALL_FEATURES = CHANNELS + DERIVED + +DEFAULT_CLASS_MAP = { + 0: "chao", + 1: "cana", + 2: "erva", +} + +# Labelmap das mascaras coloridas, informado pelo projeto. +# Valores em RGB, como normalmente aparecem em ferramentas de anotacao. +# Como o OpenCV le PNG em BGR, a funcao decode_color_mask converte internamente. +DEFAULT_MASK_COLOR_MAP_RGB = { + (128, 0, 0): 0, # chao + (0, 0, 128): 1, # cana + (0, 128, 0): 2, # erva +} + +# Paleta de visualizacao em BGR para o painel GT mask. +# Mantem visual equivalente ao labelmap RGB acima. +DEFAULT_VIS_PALETTE_BGR = { + 0: (0, 0, 128), # chao: RGB 128,0,0 + 1: (128, 0, 0), # cana: RGB 0,0,128 + 2: (0, 128, 0), # erva: RGB 0,128,0 + 255: (0, 0, 0), +} + +EPS = 1e-6 + + +# ============================================================ +# Utilidades básicas +# ============================================================ + + +def ensure_dir(path: Path | str) -> Path: + p = Path(path) + p.mkdir(parents=True, exist_ok=True) + return p + + +def load_json(path: Path) -> dict: + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + + +def write_json(path: Path, data: Any): + with open(path, "w", encoding="utf-8") as f: + json.dump(data, f, ensure_ascii=False, indent=2) + + +def safe_float(x: Any, default: float = 0.0) -> float: + try: + if x is None: + return default + v = float(x) + if math.isnan(v) or math.isinf(v): + return default + return v + except Exception: + return default + + +def parse_class_map(text: Optional[str]) -> Dict[int, str]: + if not text: + return dict(DEFAULT_CLASS_MAP) + + out = {} + # Formatos aceitos: + # "0:chao,1:cana,2:erva" + # "0=chao,1=cana,2=erva" + for item in text.split(","): + item = item.strip() + if not item: + continue + if ":" in item: + k, v = item.split(":", 1) + elif "=" in item: + k, v = item.split("=", 1) + else: + raise ValueError(f"Classe inválida em --class-map: {item}") + out[int(k.strip())] = v.strip() + + return out + + +def normalize_to_u8(x: np.ndarray, p_low: float = 1.0, p_high: float = 99.0) -> np.ndarray: + arr = x.astype(np.float32, copy=False) + finite = np.isfinite(arr) + if not np.any(finite): + return np.zeros(arr.shape, dtype=np.uint8) + + vals = arr[finite] + lo = np.percentile(vals, p_low) + hi = np.percentile(vals, p_high) + if hi <= lo + EPS: + hi = lo + 1.0 + + y = (arr - lo) / (hi - lo) + y = np.clip(y, 0, 1) + return (y * 255.0).astype(np.uint8) + + +def float01_to_u8(x: np.ndarray) -> np.ndarray: + return np.clip(x.astype(np.float32) * 255.0, 0, 255).astype(np.uint8) + + +def rgb_from_tensor(tensor: np.ndarray, stretch: bool = False) -> np.ndarray: + rgb = np.transpose(tensor[:3], (1, 2, 0)).astype(np.float32) + if stretch: + chans = [normalize_to_u8(rgb[:, :, i]) for i in range(3)] + rgb_u8 = np.dstack(chans) + else: + rgb_u8 = float01_to_u8(rgb) + return cv2.cvtColor(rgb_u8, cv2.COLOR_RGB2BGR) + + +def apply_colormap_gray(x: np.ndarray, stretch: bool = True) -> np.ndarray: + if stretch: + u8 = normalize_to_u8(x) + else: + u8 = float01_to_u8(x) + return cv2.applyColorMap(u8, cv2.COLORMAP_VIRIDIS) + + +def cv_text(text: Any) -> str: + """ + OpenCV putText nao lida bem com acentos/cedilha em muitos ambientes. + Converte qualquer texto para ASCII seguro. + """ + s = str(text) + s = unicodedata.normalize("NFKD", s) + s = s.encode("ascii", "ignore").decode("ascii") + return s + + +def put_label(img: np.ndarray, title: str, subtitle: str = "") -> np.ndarray: + out = img.copy() + title = cv_text(title) + subtitle = cv_text(subtitle) + cv2.rectangle(out, (0, 0), (out.shape[1], 58 if subtitle else 36), (0, 0, 0), -1) + cv2.putText(out, title, (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.65, (0, 255, 255), 2, cv2.LINE_AA) + if subtitle: + cv2.putText(out, subtitle[:120], (10, 48), cv2.FONT_HERSHEY_SIMPLEX, 0.45, (255, 255, 255), 1, cv2.LINE_AA) + return out + + +def resize_keep(img: np.ndarray, size: Tuple[int, int]) -> np.ndarray: + return cv2.resize(img, size, interpolation=cv2.INTER_AREA) + + +def make_grid(panels: List[Tuple[str, np.ndarray, str]], panel_w: int = 360) -> np.ndarray: + rendered = [] + for title, img, subtitle in panels: + scale = panel_w / img.shape[1] + panel_h = max(1, int(img.shape[0] * scale)) + small = cv2.resize(img, (panel_w, panel_h), interpolation=cv2.INTER_AREA) + rendered.append(put_label(small, title, subtitle)) + + if not rendered: + return np.zeros((200, 400, 3), dtype=np.uint8) + + max_h = max(x.shape[0] for x in rendered) + padded = [] + for img in rendered: + if img.shape[0] < max_h: + pad = np.zeros((max_h - img.shape[0], img.shape[1], 3), dtype=np.uint8) + img = np.vstack([img, pad]) + padded.append(img) + + cols = 3 + rows = [] + gap_w = np.full((max_h, 12, 3), 25, dtype=np.uint8) + gap_h = np.full((12, cols * panel_w + (cols - 1) * 12, 3), 25, dtype=np.uint8) + + for i in range(0, len(padded), cols): + row_imgs = padded[i:i + cols] + while len(row_imgs) < cols: + row_imgs.append(np.zeros_like(padded[0])) + row = np.hstack([row_imgs[0], gap_w, row_imgs[1], gap_w, row_imgs[2]]) + rows.append(row) + + canvas = rows[0] + for r in rows[1:]: + canvas = np.vstack([canvas, gap_h, r]) + return canvas + + +# ============================================================ +# Detecção de layout e leitura dos tensores/máscaras +# ============================================================ + + +def is_dataset_root(path: Path) -> bool: + """ + Um dataset final válido tem, no mínimo: + root/metas + root/bins + + masks/ e previews/ são opcionais para a auditoria, embora masks/ seja + necessário para estatísticas por classe. + """ + return (path / "metas").is_dir() and (path / "bins").is_dir() + + +def find_dataset_roots(path: Path) -> List[Path]: + """ + Detecta um ou vários datasets. + + Suporta: + 1) dataset direto: + root/metas + root/bins + root/masks + + 2) subpasta do dataset: + root/metas, root/bins etc, mas input_path=root/metas ou root/bins + + 3) super-root com grupos dentro: + group/chao/metas + group/chao/bins + group/chao_cana/metas + group/chao_cana/bins + ... + + Isso resolve o caso: + --input_path dataset/1024x640/group + quando group contém vários datasets filhos. + """ + p = path.resolve() + + candidates: List[Path] = [] + if p.is_file(): + candidates.extend([p.parent, p.parent.parent]) + else: + candidates.extend([p, p.parent]) + + roots: List[Path] = [] + + # Primeiro tenta o próprio caminho ou pai direto. + for c in candidates: + if c.name.lower() in ("metas", "bins", "masks", "previews"): + root = c.parent + else: + root = c + + if is_dataset_root(root): + roots.append(root) + + # Se não achou, procura recursivamente datasets filhos. + search_base = p if p.is_dir() else p.parent + if not roots and search_base.exists(): + for metas_dir in search_base.rglob("metas"): + root = metas_dir.parent + if is_dataset_root(root): + roots.append(root) + + # Remove duplicados preservando ordem. + unique: List[Path] = [] + seen = set() + for r in roots: + rr = r.resolve() + if rr not in seen: + unique.append(rr) + seen.add(rr) + + if not unique: + raise FileNotFoundError( + f"Não consegui detectar dataset_root a partir de {path}. " + "Esperado root/metas e root/bins, ou um super-root contendo grupos com metas/bins." + ) + + return unique + + +def find_dataset_root(path: Path) -> Path: + """ + Compatibilidade com chamadas antigas: retorna o primeiro root encontrado. + Para a auditoria principal, use find_dataset_roots(). + """ + return find_dataset_roots(path)[0] + + +def list_meta_files(dataset_root: Path) -> List[Path]: + metas = sorted((dataset_root / "metas").glob("*.json")) + if not metas: + raise RuntimeError(f"Nenhum .json encontrado em {dataset_root / 'metas'}") + return metas + + +def find_sibling(dataset_root: Path, subdir: str, stem: str, exts: Tuple[str, ...]) -> Optional[Path]: + folder = dataset_root / subdir + if not folder.is_dir(): + return None + for ext in exts: + p = folder / f"{stem}{ext}" + if p.exists(): + return p + return None + + +def resolve_mask_path(dataset_root: Path, meta_path: Path) -> Optional[Path]: + """ + Resolve mascara da amostra. + + Importante: o sample_name usado no relatorio pode receber prefixo do grupo + tipo 'chao__2026...', mas o arquivo real da mascara continua usando + meta_path.stem, sem prefixo. Esse foi o bug da rodada anterior. + """ + real_stem = meta_path.stem + masks_dir = dataset_root / "masks" + + candidates: List[Path] = [] + for ext in (".png", ".tif", ".tiff", ".npy"): + candidates.append(masks_dir / f"{real_stem}{ext}") + + for suffix in ("_mask", "_gt", "_label", "_labels", "_seg"): + for ext in (".png", ".tif", ".tiff", ".npy"): + candidates.append(masks_dir / f"{real_stem}{suffix}{ext}") + + if masks_dir.is_dir(): + candidates.extend(sorted(masks_dir.glob(f"{real_stem}*.*"))) + + for p in candidates: + if p.exists() and p.suffix.lower() in (".png", ".tif", ".tiff", ".npy"): + return p + + return None + + +def resolve_tensor_payload(meta_path: Path, dataset_root: Path, meta: dict) -> Optional[Path]: + """ + Resolve payload final já pronto, quando o dataset já contém MULTISPEC salvo. + Para RAW_BRUTO multi por câmera, use build_multispec_from_raw_native_multi(). + """ + stem = meta_path.stem + bins = dataset_root / "bins" + + candidates = [] + saved_payload_path = meta.get("saved_payload_path") + if saved_payload_path: + sp = Path(saved_payload_path) + candidates.extend([ + bins / sp.name, + meta_path.parent / sp, + dataset_root / sp, + ]) + + candidates.extend([ + bins / f"{stem}.raw", + bins / f"{stem}.bin", + bins / f"{stem}_multispec.raw", + bins / f"{stem}_multispec.bin", + bins / f"{stem}_offline_multispec.raw", + bins / f"{stem}_offline_multispec.bin", + ]) + + for c in candidates: + if c.exists(): + return c + return None + + +def resolve_camera_payloads(meta_path: Path, dataset_root: Path, meta: dict) -> Dict[str, Path]: + """ + Resolve os .bin/.raw por câmera quando saved_payload_type='raw_native_multi'. + """ + stem = meta_path.stem + bins = dataset_root / "bins" + paths = {} + + saved_payload_paths = meta.get("saved_payload_paths", {}) or {} + for cam_id, fname in saved_payload_paths.items(): + fp = Path(fname) + candidates = [ + bins / fp.name, + meta_path.parent / fname, + dataset_root / fname, + bins / f"{stem}_{cam_id}.bin", + bins / f"{stem}_{cam_id}.raw", + bins / f"{stem}_{cam_id.lower()}.bin", + bins / f"{stem}_{cam_id.lower()}.raw", + ] + + found = None + for c in candidates: + if Path(c).exists(): + found = Path(c) + break + + if found is None: + raise FileNotFoundError(f"Payload bruto não encontrado para {cam_id} em {meta_path.name}: {fname}") + + paths[cam_id] = found + + return paths + + +def resolve_module_params_path(meta_path: Path, dataset_root: Path, meta: dict) -> str: + """ + Tenta encontrar o module_params.json usado para reconstruir o tensor. + A ideia é usar exatamente o mesmo caminho salvo no meta quando existir. + """ + calib_path = meta.get("camera_params_json") or meta.get("module_params_json") or "calibration/module_params.json" + p = Path(calib_path) + + candidates = [] + if p.is_absolute(): + candidates.append(p) + else: + candidates.extend([ + Path.cwd() / p, + meta_path.parent / p, + dataset_root / p, + dataset_root.parent / p, + dataset_root.parent.parent / p, + Path.cwd() / "calibration" / "module_params.json", + ]) + + for c in candidates: + if c.exists(): + return str(c) + + raise FileNotFoundError( + f"module_params.json não encontrado. Valor no meta={calib_path}. " + "Sem ele eu não consigo aplicar homografia/flat/radiometria igual ao pipeline final." + ) + + +def get_raw_processor_core( + core_cache: Dict[Tuple[int, int, str, str], Any], + sensor_width: int, + sensor_height: int, + bayer: str, + calib_path: str, +): + """ + Reaproveita RawProcessorCore entre amostras. + + Isso evita recarregar flat-field e recriar caches de gain a cada imagem. + A chave considera tamanho, bayer e module_params.json. + """ + key = (int(sensor_width), int(sensor_height), str(bayer), str(Path(calib_path).resolve())) + if key not in core_cache: + if RawProcessorCore is None: + raise RuntimeError( + "Não consegui importar RawProcessorCore. Rode com python -m utils.audit_dataset " + "a partir da raiz do projeto, igual você fez, e confirme se core/raw_processor_core.py existe." + ) + core_cache[key] = RawProcessorCore( + sensor_width=int(sensor_width), + sensor_height=int(sensor_height), + bayer_pattern=str(bayer), + calibration_json_path=str(calib_path), + ) + return core_cache[key] + + +def build_multispec_from_raw_native_multi(meta_path: Path, dataset_root: Path, meta: dict, core_cache: Dict[Tuple[int, int, str, str], Any]) -> Tuple[np.ndarray, Path]: + """ + Reconstrói o tensor final [R,G,B,RE,NIR] a partir do RAW_BRUTO multi por câmera. + + Usa o mesmo princípio do script visual antigo: + - saved_payload_paths + - saved_payload_shapes + - saved_payload_dtypes + - stream_meta.camera_info + - actual_camera_controls/startup_camera_controls + - module_params.json com homografia/flat/radiometria + """ + saved_dtypes = meta.get("saved_payload_dtypes", {}) or {} + saved_shapes = meta.get("saved_payload_shapes", {}) or {} + cam_paths = resolve_camera_payloads(meta_path, dataset_root, meta) + + frame = {} + for cam_id, payload_path in cam_paths.items(): + saved_dtype = saved_dtypes.get(cam_id) + saved_shape = saved_shapes.get(cam_id) + if saved_dtype is None or saved_shape is None: + raise RuntimeError(f"Faltam dtype/shape para {cam_id} em {meta_path.name}") + + arr = np.fromfile(str(payload_path), dtype=np.dtype(saved_dtype)).reshape(tuple(saved_shape)) + frame[cam_id] = arr + + sensor_width = int(meta.get("sensor_width", 1280)) + sensor_height = int(meta.get("sensor_height", 800)) + bayer = meta.get("bayer_pattern", "RGGB") + calib_path = resolve_module_params_path(meta_path, dataset_root, meta) + + core = get_raw_processor_core( + core_cache=core_cache, + sensor_width=sensor_width, + sensor_height=sensor_height, + bayer=bayer, + calib_path=calib_path, + ) + + stream_meta = meta.get("stream_meta", {}) or {} + processing_meta = dict(stream_meta) + + if meta.get("actual_camera_controls") is not None: + processing_meta["actual_camera_controls"] = meta.get("actual_camera_controls") + if meta.get("startup_camera_controls") is not None: + processing_meta["startup_camera_controls"] = meta.get("startup_camera_controls") + if meta.get("radiometric_last_result") is not None: + processing_meta["radiometric_last_result"] = meta.get("radiometric_last_result") + + tensor = core.build_infer_tensor_from_stream(frame, processing_meta, 5) + if tensor is None: + raise RuntimeError(f"RawProcessorCore retornou tensor None para {meta_path.name}") + + tensor = np.asarray(tensor, dtype=np.float32) + if tensor.ndim != 3: + raise RuntimeError(f"Tensor reconstruído inválido em {meta_path.name}: shape={tensor.shape}") + if tensor.shape[0] != 5 and tensor.shape[-1] == 5: + tensor = np.transpose(tensor, (2, 0, 1)) + if tensor.shape[0] < 5: + raise RuntimeError(f"Tensor reconstruído precisa ter 5 canais, recebido shape={tensor.shape}") + + # Retorna o primeiro payload bruto só como referência de origem no CSV. + first_payload = next(iter(cam_paths.values())) + return np.ascontiguousarray(tensor[:5]), first_payload + + +def load_multispec_tensor(meta_path: Path, dataset_root: Path, core_cache: Dict[Tuple[int, int, str, str], Any]) -> Tuple[np.ndarray, dict, Path]: + meta = load_json(meta_path) + saved_type = meta.get("saved_payload_type") + + # Caso 1: dataset original bruto por câmera. Aqui o script monta o tensor final. + if saved_type == "raw_native_multi" or "saved_payload_paths" in meta: + tensor, source_payload = build_multispec_from_raw_native_multi(meta_path, dataset_root, meta, core_cache) + return tensor, meta, source_payload + + # Caso 2: dataset já normalizado/MULTISPEC salvo como payload único. + payload_path = resolve_tensor_payload(meta_path, dataset_root, meta) + if payload_path is None: + raise FileNotFoundError(f"Payload tensor final não encontrado para {meta_path.name}") + + dtype = meta.get("saved_payload_dtype", "float32") + shape = meta.get("saved_payload_shape") + + if shape is None: + # Alguns JSONs podem guardar isso em outro campo. + shape = meta.get("tensor_shape") or meta.get("shape") + + if shape is None: + raise RuntimeError( + f"Shape do tensor não encontrado em {meta_path.name}. " + "Esperado saved_payload_shape=[5,H,W]." + ) + + arr = np.fromfile(str(payload_path), dtype=np.dtype(dtype)).reshape(tuple(shape)) + arr = arr.astype(np.float32, copy=False) + + if arr.ndim != 3: + raise RuntimeError(f"Tensor inválido em {payload_path}: shape={arr.shape}") + + if arr.shape[0] != 5 and arr.shape[-1] == 5: + arr = np.transpose(arr, (2, 0, 1)) + + if arr.shape[0] < 5: + raise RuntimeError(f"Tensor precisa ter 5 canais [R,G,B,RE,NIR], recebido shape={arr.shape}") + + return arr[:5], meta, payload_path + + +def decode_color_mask(mask_img: np.ndarray, ignore_index: int) -> np.ndarray: + """ + Converte mascara colorida do labelmap para indices de classe. + + Seu labelmap atual em RGB: + chao = 128,0,0 -> classe 0 + cana = 0,0,128 -> classe 1 + erva = 0,128,0 -> classe 2 + + O OpenCV le PNG como BGR, entao fazemos a conversao RGB->BGR antes de comparar. + """ + if mask_img.ndim == 2: + return mask_img.astype(np.int32, copy=False) + + if mask_img.shape[2] == 4: + bgr = mask_img[:, :, :3] + else: + bgr = mask_img[:, :, :3] + + out = np.full(mask_img.shape[:2], ignore_index, dtype=np.int32) + + for rgb_color, cls_id in DEFAULT_MASK_COLOR_MAP_RGB.items(): + r, g, b = rgb_color + bgr_color = np.array([b, g, r], dtype=np.uint8) + hit = np.all(bgr == bgr_color, axis=2) + out[hit] = int(cls_id) + + # Fallback: se por algum motivo a imagem ja veio em ordem RGB, tenta tambem RGB direto. + # Isso evita quebrar caso alguma leitura futura nao use cv2. + for rgb_color, cls_id in DEFAULT_MASK_COLOR_MAP_RGB.items(): + rgb_arr = np.array(rgb_color, dtype=np.uint8) + hit = np.all(bgr == rgb_arr, axis=2) + # So preenche pixels ainda nao identificados para evitar troca em cores ambiguas. + out[(out == ignore_index) & hit] = int(cls_id) + + return out + + +def load_mask(mask_path: Optional[Path], target_hw: Tuple[int, int], ignore_index: int) -> Optional[np.ndarray]: + if mask_path is None: + return None + + if mask_path.suffix.lower() == ".npy": + mask = np.load(str(mask_path)) + if mask.ndim == 3: + mask = decode_color_mask(mask.astype(np.uint8), ignore_index) + else: + mask_img = cv2.imread(str(mask_path), cv2.IMREAD_UNCHANGED) + if mask_img is None: + raise RuntimeError(f"Falha ao ler mascara: {mask_path}") + mask = decode_color_mask(mask_img, ignore_index) + + mask = mask.astype(np.int32, copy=False) + h, w = target_hw + if mask.shape[:2] != (h, w): + mask = cv2.resize(mask, (w, h), interpolation=cv2.INTER_NEAREST) + return mask + + +def mask_unique_summary(mask: Optional[np.ndarray], max_items: int = 20) -> str: + if mask is None: + return "none" + vals, counts = np.unique(mask, return_counts=True) + parts = [] + for v, c in zip(vals[:max_items], counts[:max_items]): + cls_name = DEFAULT_CLASS_MAP.get(int(v), "ignore" if int(v) == 255 else "unk") + parts.append(f"{int(v)}:{cls_name}:{int(c)}") + if len(vals) > max_items: + parts.append("...") + return ",".join(parts) + + +# ============================================================ +# Features espectrais e estatísticas +# ============================================================ + + +def compute_feature_maps(tensor: np.ndarray) -> Dict[str, np.ndarray]: + r, g, b, re, nir = [tensor[i].astype(np.float32, copy=False) for i in range(5)] + + features = { + "R": r, + "G": g, + "B": b, + "RE": re, + "NIR": nir, + "NDVI": (nir - r) / (nir + r + EPS), + "NDRE": (nir - re) / (nir + re + EPS), + "NIR_minus_RE": nir - re, + "NIR_over_RE": nir / (re + EPS), + "NIR_over_R": nir / (r + EPS), + "RE_over_R": re / (r + EPS), + "NIR_over_G": nir / (g + EPS), + "RE_over_G": re / (g + EPS), + "G_minus_R": g - r, + } + + # Evita explosões absurdas em razão quando denominador está quase zero. + for k in list(features.keys()): + features[k] = np.nan_to_num(features[k], nan=0.0, posinf=0.0, neginf=0.0).astype(np.float32) + return features + + +def calc_stats(values: np.ndarray, raw01: bool = False) -> Dict[str, float]: + v = values.astype(np.float32, copy=False) + v = v[np.isfinite(v)] + if v.size == 0: + return { + "count": 0, + "mean": 0.0, + "std": 0.0, + "min": 0.0, + "p01": 0.0, + "p05": 0.0, + "p25": 0.0, + "p50": 0.0, + "p75": 0.0, + "p95": 0.0, + "p99": 0.0, + "max": 0.0, + "iqr": 0.0, + "p95_p05": 0.0, + "dark_pct": 0.0, + "sat_pct": 0.0, + } + + p = np.percentile(v, [1, 5, 25, 50, 75, 95, 99]) + out = { + "count": int(v.size), + "mean": float(np.mean(v)), + "std": float(np.std(v)), + "min": float(np.min(v)), + "p01": float(p[0]), + "p05": float(p[1]), + "p25": float(p[2]), + "p50": float(p[3]), + "p75": float(p[4]), + "p95": float(p[5]), + "p99": float(p[6]), + "max": float(np.max(v)), + "iqr": float(p[4] - p[2]), + "p95_p05": float(p[5] - p[1]), + "dark_pct": 0.0, + "sat_pct": 0.0, + } + + if raw01: + out["dark_pct"] = float(np.mean(v <= 0.01) * 100.0) + out["sat_pct"] = float(np.mean(v >= 0.99) * 100.0) + + return out + + +@dataclass +class RunningFeatureStats: + values: Dict[str, List[float]] = field(default_factory=lambda: {f: [] for f in ALL_FEATURES}) + counts: Dict[str, int] = field(default_factory=lambda: {f: 0 for f in ALL_FEATURES}) + pixel_count: int = 0 + + def add(self, feature_name: str, values: np.ndarray, max_samples: int = 25000): + v = values.astype(np.float32, copy=False) + v = v[np.isfinite(v)] + if v.size == 0: + return + self.pixel_count += int(v.size) + self.counts[feature_name] += int(v.size) + + if v.size > max_samples: + idx = np.random.choice(v.size, size=max_samples, replace=False) + v = v[idx] + self.values[feature_name].extend(v.tolist()) + + def summarize(self) -> Dict[str, Dict[str, float]]: + out = {} + for f, vals in self.values.items(): + arr = np.asarray(vals, dtype=np.float32) + raw01 = f in CHANNELS + s = calc_stats(arr, raw01=raw01) + s["total_pixels_seen"] = int(self.counts.get(f, 0)) + out[f] = s + return out + + +# ============================================================ +# Alinhamento por borda/correlação +# ============================================================ + + +def gradient_mag(x: np.ndarray) -> np.ndarray: + u8 = normalize_to_u8(x) + gx = cv2.Sobel(u8, cv2.CV_32F, 1, 0, ksize=3) + gy = cv2.Sobel(u8, cv2.CV_32F, 0, 1, ksize=3) + mag = cv2.magnitude(gx, gy) + return mag.astype(np.float32) + + +def estimate_shift_phase(a: np.ndarray, b: np.ndarray) -> Tuple[float, float, float]: + # dx, dy estimados entre mapas. Valores grandes sugerem desalinhamento residual. + aa = normalize_to_u8(a).astype(np.float32) + bb = normalize_to_u8(b).astype(np.float32) + try: + (dx, dy), response = cv2.phaseCorrelate(aa, bb) + return float(dx), float(dy), float(response) + except Exception: + return 0.0, 0.0, 0.0 + + +def edge_agreement(a: np.ndarray, b: np.ndarray) -> Dict[str, float]: + ga = gradient_mag(a) + gb = gradient_mag(b) + + va = ga.reshape(-1) + vb = gb.reshape(-1) + if np.std(va) < EPS or np.std(vb) < EPS: + corr = 0.0 + else: + corr = float(np.corrcoef(va, vb)[0, 1]) + + dx, dy, resp = estimate_shift_phase(ga, gb) + return { + "edge_corr": corr, + "phase_dx": dx, + "phase_dy": dy, + "phase_response": resp, + } + + +def make_edge_overlay(tensor: np.ndarray) -> np.ndarray: + r, g, b, re, nir = [tensor[i] for i in range(5)] + rgb_gray = (0.299 * r + 0.587 * g + 0.114 * b).astype(np.float32) + + e_rgb = normalize_to_u8(gradient_mag(rgb_gray), 5, 99) + e_re = normalize_to_u8(gradient_mag(re), 5, 99) + e_nir = normalize_to_u8(gradient_mag(nir), 5, 99) + + # BGR: RGB edge em verde, RE em vermelho, NIR em azul. + overlay = np.zeros((tensor.shape[1], tensor.shape[2], 3), dtype=np.uint8) + overlay[:, :, 1] = e_rgb + overlay[:, :, 2] = e_re + overlay[:, :, 0] = e_nir + return overlay + + +# ============================================================ +# Visualizações por amostra +# ============================================================ + + +def colorize_mask(mask: Optional[np.ndarray], class_map: Dict[int, str], target_hw: Tuple[int, int]) -> np.ndarray: + h, w = target_hw + out = np.zeros((h, w, 3), dtype=np.uint8) + if mask is None: + return out + + palette = dict(DEFAULT_VIS_PALETTE_BGR) + for cls_id in np.unique(mask): + cls_id = int(cls_id) + if cls_id in palette: + out[mask == cls_id] = palette[cls_id] + elif cls_id in class_map: + rng = np.random.default_rng(cls_id) + out[mask == cls_id] = rng.integers(40, 220, size=3) + return out + + +def mask_edges_on_rgb(rgb_bgr: np.ndarray, mask: Optional[np.ndarray]) -> np.ndarray: + out = rgb_bgr.copy() + if mask is None: + return out + m = mask.astype(np.uint8) + edges = cv2.Canny(m, 0, 1) + out[edges > 0] = (0, 255, 255) + return out + + +def make_sample_visual( + sample_name: str, + tensor: np.ndarray, + mask: Optional[np.ndarray], + class_map: Dict[int, str], + sample_stats: Dict[str, Any], +) -> np.ndarray: + features = compute_feature_maps(tensor) + h, w = tensor.shape[1], tensor.shape[2] + + rgb = rgb_from_tensor(tensor, stretch=False) + rgb_stretch = rgb_from_tensor(tensor, stretch=True) + re = apply_colormap_gray(features["RE"], stretch=True) + nir = apply_colormap_gray(features["NIR"], stretch=True) + ndvi = apply_colormap_gray(features["NDVI"], stretch=True) + ndre = apply_colormap_gray(features["NDRE"], stretch=True) + diff = apply_colormap_gray(features["NIR_minus_RE"], stretch=True) + ratio = apply_colormap_gray(features["NIR_over_RE"], stretch=True) + edge = make_edge_overlay(tensor) + mask_bgr = colorize_mask(mask, class_map, (h, w)) + + overlay_mask = rgb.copy() + if mask is not None: + overlay_mask = cv2.addWeighted(rgb, 0.65, mask_bgr, 0.35, 0) + + mask_edges = mask_edges_on_rgb(rgb, mask) + + align = sample_stats.get("alignment", {}) + re_align = align.get("RGBgray_vs_RE", {}) + nir_align = align.get("RGBgray_vs_NIR", {}) + + panels = [ + ("RGB tensor", rgb, sample_name), + ("RGB stretch", rgb_stretch, "visual apenas para contraste"), + ("GT mask", mask_bgr, f"unique={mask_unique_summary(mask)}"), + ("Mask overlay", overlay_mask, "GT over RGB"), + ("Mask edges", mask_edges, "GT edges over RGB"), + ("RE", re, "canal 3 | stretch p1-p99"), + ("NIR", nir, "canal 4 | stretch p1-p99"), + ("NDVI", ndvi, "(NIR-R)/(NIR+R)"), + ("NDRE", ndre, "(NIR-RE)/(NIR+RE)"), + ("NIR - RE", diff, "diferença direta"), + ("NIR / RE", ratio, "razão com eps"), + ( + "Edge overlay", + edge, + f"G=RGB | R=RE | B=NIR | REcorr={safe_float(re_align.get('edge_corr')):.3f} NIRcorr={safe_float(nir_align.get('edge_corr')):.3f}", + ), + ] + return make_grid(panels, panel_w=380) + + +# ============================================================ +# Auditoria principal +# ============================================================ + + +def resolve_rejected_preview_root(fixed_root: Path) -> Path: + # fixed_root normalmente é dataset/fixed/group + # queremos dataset/fixed/rejected_previews + if fixed_root.name == "group": + return fixed_root.parent / "rejected_previews" + return fixed_root / "_rejected_previews" + + +def find_original_preview(dataset_root: Path, real_stem: str) -> Optional[Path]: + previews_dir = dataset_root / "previews" + if not previews_dir.is_dir(): + return None + + for ext in (".png", ".jpg", ".jpeg", ".webp"): + p = previews_dir / f"{real_stem}{ext}" + if p.exists(): + return p + + # fallback caso tenha sufixo no nome + matches = [] + for ext in (".png", ".jpg", ".jpeg", ".webp"): + matches.extend(previews_dir.glob(f"{real_stem}*{ext}")) + + return matches[0] if matches else None + + +def resize_preview_max_width(img: np.ndarray, max_width: int) -> np.ndarray: + if img is None or img.size == 0: + return img + if max_width <= 0 or img.shape[1] <= max_width: + return img + + scale = max_width / img.shape[1] + new_w = int(img.shape[1] * scale) + new_h = int(img.shape[0] * scale) + return cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA) + + +def make_rejected_filename(row: Dict[str, Any]) -> str: + sample = str(row.get("sample", "sample")) + reasons = str(row.get("reject_reasons", "")) + + # Nome curto com principais motivos. + tags = [] + if "abs_shift" in reasons: + tags.append("absShift") + if "outlier" in reasons: + tags.append("outlier") + if "low_edge_corr" in reasons: + tags.append("lowCorr") + if "too_small" in reasons: + tags.append("smallTarget") + if "missing_mask" in reasons: + tags.append("noMask") + + tag = "_".join(tags) if tags else "rejected" + return f"{sample}__{tag}.png" + + +def draw_rejected_header(img: np.ndarray, row: Dict[str, Any]) -> np.ndarray: + if img is None or img.size == 0: + return img + + out = img.copy() + h, w = out.shape[:2] + + header_h = 92 + canvas = np.zeros((h + header_h, w, 3), dtype=np.uint8) + canvas[:header_h, :] = (20, 20, 20) + canvas[header_h:, :] = out + + sample = cv_text(row.get("sample", "")) + reasons = cv_text(row.get("reject_reasons", "")) + + re_mag = safe_float(row.get("re_shift_mag")) + nir_mag = safe_float(row.get("nir_shift_mag")) + max_dev = safe_float(row.get("max_shift_dev_px")) + re_corr = safe_float(row.get("re_edge_corr")) + nir_corr = safe_float(row.get("nir_edge_corr")) + + line1 = sample[:120] + line2 = f"RE={re_mag:.1f}px NIR={nir_mag:.1f}px dev={max_dev:.1f}px corrRE={re_corr:.3f} corrNIR={nir_corr:.3f}" + line3 = reasons[:150] + + cv2.putText(canvas, line1, (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.58, (0, 255, 255), 2, cv2.LINE_AA) + cv2.putText(canvas, line2, (10, 52), cv2.FONT_HERSHEY_SIMPLEX, 0.50, (255, 255, 255), 1, cv2.LINE_AA) + cv2.putText(canvas, line3, (10, 78), cv2.FONT_HERSHEY_SIMPLEX, 0.43, (120, 220, 255), 1, cv2.LINE_AA) + + return canvas + + +def save_rejected_preview( + row: Dict[str, Any], + fixed_root: Path, + args, + tensor: Optional[np.ndarray] = None, + mask: Optional[np.ndarray] = None, + class_map: Optional[Dict[int, str]] = None, +): + rejected_root = resolve_rejected_preview_root(fixed_root) + group = str(row.get("group", "unknown")) + dataset_root = Path(str(row["dataset_root"])) + real_stem = str(row["real_stem"]) + + dst_dir = ensure_dir(rejected_root / group) + dst_path = dst_dir / make_rejected_filename(row) + + source_mode = str(args.rejected_preview_source) + + img = None + + # 1) preview original, se existir. + if source_mode in ("auto", "preview"): + preview_path = find_original_preview(dataset_root, real_stem) + if preview_path is not None: + img = cv2.imread(str(preview_path), cv2.IMREAD_COLOR) + + # 2) painel completo do audit, se já existir e o usuário pediu. + # Aqui é útil quando --save-visuals também estiver ligado. + if img is None and source_mode in ("auto", "audit_panel"): + # Procura por qualquer painel visual que contenha o sample no nome. + # Fica em out_dir/visuals, mas não temos out_dir aqui. Então esse modo + # fica mais útil se você passar painel diretamente no futuro. + pass + + # 3) RGB reconstruído do tensor, se disponível. + if img is None and tensor is not None and source_mode in ("auto", "rgb_tensor", "audit_panel"): + img = rgb_from_tensor(tensor, stretch=False) + + if mask is not None and class_map is not None: + mask_bgr = colorize_mask(mask, class_map, (tensor.shape[1], tensor.shape[2])) + img = cv2.addWeighted(img, 0.70, mask_bgr, 0.30, 0) + + if img is None: + # Último fallback: imagem preta com cabeçalho, para não perder rastreabilidade. + img = np.zeros((360, 640, 3), dtype=np.uint8) + + img = resize_preview_max_width(img, int(args.rejected_preview_max_width)) + img = draw_rejected_header(img, row) + cv2.imwrite(str(dst_path), img) + + +def robust_median_mad(values: List[float]) -> Tuple[float, float]: + arr = np.asarray([v for v in values if np.isfinite(v)], dtype=np.float32) + if arr.size == 0: + return 0.0, 1.0 + + med = float(np.median(arr)) + mad = float(np.median(np.abs(arr - med))) + + # Evita divisão por zero em grupos muito estáveis. + if mad < 0.5: + mad = 0.5 + + return med, mad + + +def safe_copy_file(src: Path, dst: Path): + if not src.exists(): + return + ensure_dir(dst.parent) + shutil.copy2(str(src), str(dst)) + + +def resolve_fixed_out_root(dataset_roots: List[Path], args) -> Path: + if args.fixed_out_root: + return Path(args.fixed_out_root).resolve() + + # Esperado: + # dataset/original/group/chao + # dataset/original/group/chao_cana + # + # Queremos: + # dataset/fixed/group + first = dataset_roots[0].resolve() + + # Se o root é .../original/group/chao, sobe 3: chao -> group -> original -> dataset + if first.parent.name == "group" and first.parent.parent.name == "original": + dataset_base = first.parent.parent.parent + return dataset_base / "fixed" / "group" + + # Fallback seguro. + return first.parent / "fixed" / "group" + + +def sample_shift_metrics(row: Dict[str, Any]) -> Dict[str, float]: + re_dx = safe_float(row.get("re_phase_dx")) + re_dy = safe_float(row.get("re_phase_dy")) + nir_dx = safe_float(row.get("nir_phase_dx")) + nir_dy = safe_float(row.get("nir_phase_dy")) + + return { + "re_dx": re_dx, + "re_dy": re_dy, + "nir_dx": nir_dx, + "nir_dy": nir_dy, + "re_mag": math.hypot(re_dx, re_dy), + "nir_mag": math.hypot(nir_dx, nir_dy), + } + + +def build_group_shift_baselines(sample_rows: List[Dict[str, Any]]) -> Dict[str, Dict[str, Tuple[float, float]]]: + grouped: Dict[str, Dict[str, List[float]]] = {} + + for row in sample_rows: + group = str(row.get("group", "unknown")) + grouped.setdefault(group, { + "re_dx": [], + "re_dy": [], + "nir_dx": [], + "nir_dy": [], + "re_mag": [], + "nir_mag": [], + }) + + m = sample_shift_metrics(row) + for k, v in m.items(): + grouped[group][k].append(v) + + baselines: Dict[str, Dict[str, Tuple[float, float]]] = {} + for group, vals in grouped.items(): + baselines[group] = {} + for k, arr in vals.items(): + baselines[group][k] = robust_median_mad(arr) + + return baselines + + +def classify_sample_for_training( + row: Dict[str, Any], + baselines: Dict[str, Dict[str, Tuple[float, float]]], + class_map: Dict[int, str], + args, +) -> Tuple[bool, List[str], Dict[str, float]]: + reasons: List[str] = [] + metrics = sample_shift_metrics(row) + + group = str(row.get("group", "unknown")) + base = baselines.get(group, {}) + + if not bool(row.get("mask_found")): + reasons.append("missing_mask") + + # Shift absoluto extremo. + if metrics["re_mag"] > args.clean_max_abs_shift_px: + reasons.append(f"re_abs_shift_high:{metrics['re_mag']:.2f}") + if metrics["nir_mag"] > args.clean_max_abs_shift_px: + reasons.append(f"nir_abs_shift_high:{metrics['nir_mag']:.2f}") + + # Desvio robusto por grupo. + max_dev = 0.0 + for k in ("re_dx", "re_dy", "nir_dx", "nir_dy", "re_mag", "nir_mag"): + med, mad = base.get(k, (0.0, 1.0)) + dev_abs = abs(metrics[k] - med) + max_dev = max(max_dev, dev_abs) + + if dev_abs > args.clean_max_dev_shift_px: + reasons.append(f"{k}_outlier:val={metrics[k]:.2f},med={med:.2f},dev={dev_abs:.2f}") + + re_corr = safe_float(row.get("re_edge_corr")) + nir_corr = safe_float(row.get("nir_edge_corr")) + low_corr = re_corr < args.clean_min_edge_corr or nir_corr < args.clean_min_edge_corr + shift_bad = max_dev > args.clean_max_dev_shift_px * 0.75 + + if low_corr: + if args.clean_reject_low_corr_only_if_shift_bad: + if shift_bad: + reasons.append(f"low_edge_corr_with_shift:re={re_corr:.3f},nir={nir_corr:.3f}") + else: + reasons.append(f"low_edge_corr:re={re_corr:.3f},nir={nir_corr:.3f}") + + # Filtro semântico opcional por presença mínima da classe alvo. + # Mantém chão puro sem exigir cana/erva. + min_target_pct = float(args.clean_min_target_pct) + if min_target_pct > 0: + h = int(row.get("H", 0) or 0) + w = int(row.get("W", 0) or 0) + total_px = max(1, h * w) + + group_lower = group.lower() + if "cana" in group_lower: + pct_cana = safe_float(row.get("pixels_cana")) / total_px + if pct_cana < min_target_pct: + reasons.append(f"cana_too_small:{pct_cana:.5f}") + + if "erva" in group_lower: + pct_erva = safe_float(row.get("pixels_erva")) / total_px + if pct_erva < min_target_pct: + reasons.append(f"erva_too_small:{pct_erva:.5f}") + + approved = len(reasons) == 0 + + extra = { + "re_shift_mag": metrics["re_mag"], + "nir_shift_mag": metrics["nir_mag"], + "max_shift_dev_px": max_dev, + "re_edge_corr": re_corr, + "nir_edge_corr": nir_corr, + } + + return approved, reasons, extra + + +def copy_sample_to_fixed(row: Dict[str, Any], fixed_root: Path, copy_previews: bool): + dataset_root = Path(str(row["dataset_root"])) + group = str(row["group"]) + real_stem = str(row["real_stem"]) + meta_path = Path(str(row["meta_path"])) + + dst_group = fixed_root / group + + # Copia meta. + safe_copy_file(meta_path, dst_group / "metas" / meta_path.name) + + # Copia máscara. + mask_path = Path(str(row.get("mask_path", ""))) + if str(mask_path) and mask_path.exists(): + safe_copy_file(mask_path, dst_group / "masks" / mask_path.name) + + # Copia payloads do meta. + meta = load_json(meta_path) + + # Caso raw_native_multi: vários payloads. + saved_payload_paths = meta.get("saved_payload_paths", {}) or {} + if saved_payload_paths: + for _, fname in saved_payload_paths.items(): + src = dataset_root / "bins" / Path(fname).name + safe_copy_file(src, dst_group / "bins" / src.name) + + # Caso payload único. + saved_payload_path = meta.get("saved_payload_path") + if saved_payload_path: + src = dataset_root / "bins" / Path(saved_payload_path).name + safe_copy_file(src, dst_group / "bins" / src.name) + + # Fallback usando payload_path do CSV. + payload_path = Path(str(row.get("payload_path", ""))) + if str(payload_path) and payload_path.exists(): + safe_copy_file(payload_path, dst_group / "bins" / payload_path.name) + + # Preview opcional. + if copy_previews: + previews_dir = dataset_root / "previews" + if previews_dir.is_dir(): + for ext in (".png", ".jpg", ".jpeg", ".webp"): + src = previews_dir / f"{real_stem}{ext}" + if src.exists(): + safe_copy_file(src, dst_group / "previews" / src.name) + + +def build_fixed_dataset_from_audit( + sample_rows: List[Dict[str, Any]], + dataset_roots: List[Path], + class_map: Dict[int, str], + args, +): + fixed_root = resolve_fixed_out_root(dataset_roots, args) + ensure_dir(fixed_root) + + baselines = build_group_shift_baselines(sample_rows) + + manifest_rows = [] + approved_rows = [] + rejected_rows = [] + + for row in sample_rows: + approved, reasons, extra = classify_sample_for_training(row, baselines, class_map, args) + + out_row = dict(row) + out_row.update(extra) + out_row["approved_for_training"] = approved + out_row["reject_reasons"] = " ; ".join(reasons) + + manifest_rows.append(out_row) + + if approved: + approved_rows.append(out_row) + copy_sample_to_fixed(row, fixed_root, copy_previews=args.clean_copy_previews) + else: + rejected_rows.append(out_row) + if args.save_rejected_previews: + save_rejected_preview( + row=out_row, + fixed_root=fixed_root, + args=args, + tensor=None, + mask=None, + class_map=class_map, + ) + + summary = { + "schema": "multispec_fixed_dataset_v1", + "fixed_root": str(fixed_root), + "samples_total": len(sample_rows), + "samples_approved": len(approved_rows), + "samples_rejected": len(rejected_rows), + "approval_pct": 100.0 * len(approved_rows) / max(1, len(sample_rows)), + "filters": { + "clean_max_abs_shift_px": args.clean_max_abs_shift_px, + "clean_max_dev_shift_px": args.clean_max_dev_shift_px, + "clean_min_edge_corr": args.clean_min_edge_corr, + "clean_min_target_pct": args.clean_min_target_pct, + "clean_reject_low_corr_only_if_shift_bad": args.clean_reject_low_corr_only_if_shift_bad, + }, + "group_shift_baselines": { + group: { + k: { + "median": float(v[0]), + "mad": float(v[1]), + } + for k, v in vals.items() + } + for group, vals in baselines.items() + }, + "approved_by_group": {}, + "rejected_by_group": {}, + "rejected_previews": { + "enabled": bool(args.save_rejected_previews), + "root": str(resolve_rejected_preview_root(fixed_root)), + "source": args.rejected_preview_source, + "max_width": args.rejected_preview_max_width, + }, + } + + for row in approved_rows: + g = str(row.get("group", "unknown")) + summary["approved_by_group"][g] = summary["approved_by_group"].get(g, 0) + 1 + + for row in rejected_rows: + g = str(row.get("group", "unknown")) + summary["rejected_by_group"][g] = summary["rejected_by_group"].get(g, 0) + 1 + + write_csv(fixed_root / "fixed_manifest.csv", manifest_rows) + write_csv(fixed_root / "fixed_approved.csv", approved_rows) + write_csv(fixed_root / "fixed_rejected.csv", rejected_rows) + write_json(fixed_root / "fixed_summary.json", summary) + + print("\n========== FIXED DATASET ==========") + print(f"Saída fixed/group : {fixed_root}") + print(f"Aprovadas : {len(approved_rows)} / {len(sample_rows)}") + print(f"Rejeitadas : {len(rejected_rows)} / {len(sample_rows)}") + print(f"Manifest : {fixed_root / 'fixed_manifest.csv'}") + print("===================================\n") + + +def audit_dataset(args): + np.random.seed(args.seed) + + dataset_roots = find_dataset_roots(Path(args.input_path)) + out_dir = ensure_dir(args.out_dir) + visuals_dir = ensure_dir(out_dir / "visuals") + + class_map = parse_class_map(args.class_map) + + # Monta uma lista única de amostras: (dataset_root, meta_path) + entries: List[Tuple[Path, Path]] = [] + for root in dataset_roots: + for meta_path in list_meta_files(root): + entries.append((root, meta_path)) + + if args.limit and args.limit > 0: + entries = entries[:args.limit] + + global_acc = RunningFeatureStats() + class_acc: Dict[int, RunningFeatureStats] = {cls: RunningFeatureStats() for cls in class_map.keys()} + + sample_rows = [] + by_sample_class_channel_rows = [] + warnings = [] + core_cache: Dict[Tuple[int, int, str, str], Any] = {} + + print(f"[INFO] dataset_roots={len(dataset_roots)}") + for root in dataset_roots: + print(f" - {root}") + print(f"[INFO] amostras={len(entries)}") + print(f"[INFO] classes={class_map}") + print(f"[INFO] out_dir={out_dir}") + + for idx, (dataset_root, meta_path) in enumerate(entries, start=1): + # Prefixa com o nome do grupo para evitar colisão de nomes entre subdatasets. + group_name = dataset_root.name + stem = f"{group_name}__{meta_path.stem}" + try: + tensor, meta, payload_path = load_multispec_tensor(meta_path, dataset_root, core_cache) + h, w = tensor.shape[1], tensor.shape[2] + mask_path = resolve_mask_path(dataset_root, meta_path) + mask = load_mask(mask_path, (h, w), args.ignore_index) + + features = compute_feature_maps(tensor) + + # Estatísticas globais da amostra. + sample_feature_stats = {} + for fname, fmap in features.items(): + raw01 = fname in CHANNELS + sample_feature_stats[fname] = calc_stats(fmap.reshape(-1), raw01=raw01) + global_acc.add(fname, fmap.reshape(-1), max_samples=args.max_pixels_per_feature) + + # Estatísticas por classe da amostra. + class_pixel_counts = {} + if mask is not None: + valid_mask = mask != args.ignore_index + for cls_id, cls_name in class_map.items(): + cm = (mask == cls_id) & valid_mask + n = int(np.sum(cm)) + class_pixel_counts[cls_name] = n + if n < args.min_class_pixels: + continue + + for fname, fmap in features.items(): + vals = fmap[cm] + class_acc[cls_id].add(fname, vals, max_samples=args.max_pixels_per_feature) + s = calc_stats(vals, raw01=(fname in CHANNELS)) + by_sample_class_channel_rows.append({ + "sample": stem, + "class_id": cls_id, + "class_name": cls_name, + "feature": fname, + **s, + }) + else: + warnings.append({ + "sample": stem, + "type": "missing_mask", + "message": "Mascara nao encontrada; estatistica por classe ignorada.", + }) + + # Alinhamento por bordas. + r, g, b, re, nir = [tensor[i] for i in range(5)] + rgb_gray = (0.299 * r + 0.587 * g + 0.114 * b).astype(np.float32) + alignment = { + "RGBgray_vs_RE": edge_agreement(rgb_gray, re), + "RGBgray_vs_NIR": edge_agreement(rgb_gray, nir), + "RE_vs_NIR": edge_agreement(re, nir), + } + + # Heurísticas de alerta. + sample_warn = [] + for ch in CHANNELS: + st = sample_feature_stats[ch] + if st["sat_pct"] > args.warn_sat_pct: + sample_warn.append(f"{ch}: saturação alta {st['sat_pct']:.2f}%") + if st["dark_pct"] > args.warn_dark_pct: + sample_warn.append(f"{ch}: pixels escuros alto {st['dark_pct']:.2f}%") + if st["p95_p05"] < args.warn_low_dynamic: + sample_warn.append(f"{ch}: baixa dinâmica p95-p05={st['p95_p05']:.4f}") + + for pair_name, al in alignment.items(): + dx = abs(safe_float(al.get("phase_dx"))) + dy = abs(safe_float(al.get("phase_dy"))) + corr = safe_float(al.get("edge_corr")) + if dx > args.warn_shift_px or dy > args.warn_shift_px: + sample_warn.append(f"{pair_name}: possível shift residual dx={dx:.2f}, dy={dy:.2f}") + if corr < args.warn_edge_corr: + sample_warn.append(f"{pair_name}: baixa correlação de borda {corr:.3f}") + + if sample_warn: + warnings.append({ + "sample": stem, + "type": "sample_warning", + "messages": sample_warn, + }) + + row = { + "idx": idx, + "sample": stem, + "group": group_name, + "real_stem": meta_path.stem, + "dataset_root": str(dataset_root), + "meta_path": str(meta_path), + "payload_path": str(payload_path), + "mask_path": str(mask_path) if mask_path else "", + "mask_found": bool(mask_path is not None), + "mask_shape": str(mask.shape) if mask is not None else "", + "mask_unique": mask_unique_summary(mask), + "H": h, + "W": w, + "warnings_count": len(sample_warn), + "warning_text": " ; ".join(sample_warn), + "re_edge_corr": alignment["RGBgray_vs_RE"]["edge_corr"], + "nir_edge_corr": alignment["RGBgray_vs_NIR"]["edge_corr"], + "re_phase_dx": alignment["RGBgray_vs_RE"]["phase_dx"], + "re_phase_dy": alignment["RGBgray_vs_RE"]["phase_dy"], + "nir_phase_dx": alignment["RGBgray_vs_NIR"]["phase_dx"], + "nir_phase_dy": alignment["RGBgray_vs_NIR"]["phase_dy"], + } + + for fname in ALL_FEATURES: + st = sample_feature_stats[fname] + row[f"{fname}_mean"] = st["mean"] + row[f"{fname}_std"] = st["std"] + row[f"{fname}_p50"] = st["p50"] + row[f"{fname}_p95"] = st["p95"] + row[f"{fname}_p95_p05"] = st["p95_p05"] + if fname in CHANNELS: + row[f"{fname}_dark_pct"] = st["dark_pct"] + row[f"{fname}_sat_pct"] = st["sat_pct"] + + for cls_name, count in class_pixel_counts.items(): + row[f"pixels_{cls_name}"] = count + + sample_rows.append(row) + + sample_stats_for_visual = { + "alignment": alignment, + "features": sample_feature_stats, + } + + should_save_visual = args.save_visuals and ( + args.visual_every <= 1 or idx % args.visual_every == 0 or len(sample_warn) > 0 + ) + if should_save_visual: + canvas = make_sample_visual(stem, tensor, mask, class_map, sample_stats_for_visual) + cv2.imwrite(str(visuals_dir / f"{idx:05d}_{stem}.png"), canvas) + + if idx % args.print_every == 0 or idx == len(entries): + print(f"[OK] {idx}/{len(entries)} | {stem} | warnings={len(sample_warn)}") + + except Exception as e: + msg = str(e) + warnings.append({ + "sample": stem, + "type": "exception", + "message": msg, + }) + print(f"[ERRO] {idx}/{len(entries)} | {stem}: {msg}") + if args.stop_on_error: + raise + + # Resumos finais. + global_summary = global_acc.summarize() + class_summary = { + str(cls_id): { + "class_name": class_map[cls_id], + "features": acc.summarize(), + } + for cls_id, acc in class_acc.items() + } + + diagnosis = build_diagnosis(global_summary, class_summary, sample_rows, warnings, args) + + summary = { + "schema": "multispec_dataset_audit_v1", + "dataset_roots": [str(r) for r in dataset_roots], + "samples_processed": len(sample_rows), + "channels": CHANNELS, + "derived_features": DERIVED, + "class_map": {str(k): v for k, v in class_map.items()}, + "global_summary": global_summary, + "class_summary": class_summary, + "diagnosis": diagnosis, + } + + write_json(out_dir / "audit_summary.json", summary) + write_json(out_dir / "audit_warnings.json", warnings) + write_csv(out_dir / "audit_samples.csv", sample_rows) + write_class_channel_csv(out_dir / "audit_by_class_channel.csv", class_summary) + write_csv(out_dir / "audit_by_sample_class_channel.csv", by_sample_class_channel_rows) + + if args.build_fixed_dataset: + build_fixed_dataset_from_audit( + sample_rows=sample_rows, + dataset_roots=dataset_roots, + class_map=class_map, + args=args, + ) + + print("\n========== AUDITORIA FINALIZADA ==========") + print(f"Amostras processadas : {len(sample_rows)}") + print(f"Warnings : {len(warnings)}") + print(f"Resumo : {out_dir / 'audit_summary.json'}") + print(f"CSV amostras : {out_dir / 'audit_samples.csv'}") + print(f"CSV classes/canais : {out_dir / 'audit_by_class_channel.csv'}") + print(f"Visuais : {visuals_dir}") + print("==========================================\n") + + + + +def write_csv(path: Path, rows: List[Dict[str, Any]]): + if not rows: + with open(path, "w", encoding="utf-8", newline="") as f: + f.write("") + return + + keys = [] + for row in rows: + for k in row.keys(): + if k not in keys: + keys.append(k) + + with open(path, "w", encoding="utf-8", newline="") as f: + writer = csv.DictWriter(f, fieldnames=keys, extrasaction="ignore") + writer.writeheader() + writer.writerows(rows) + + +def write_class_channel_csv(path: Path, class_summary: Dict[str, Any]): + rows = [] + for cls_id, item in class_summary.items(): + cls_name = item["class_name"] + for feature, stats in item["features"].items(): + rows.append({ + "class_id": cls_id, + "class_name": cls_name, + "feature": feature, + **stats, + }) + write_csv(path, rows) + + +# ============================================================ +# Diagnóstico automático simples +# ============================================================ + + +def build_diagnosis( + global_summary: Dict[str, Dict[str, float]], + class_summary: Dict[str, Any], + sample_rows: List[Dict[str, Any]], + warnings: List[Dict[str, Any]], + args, +) -> Dict[str, Any]: + notes = [] + risks = [] + positives = [] + + # Sanidade global dos canais. + for ch in CHANNELS: + st = global_summary.get(ch, {}) + sat = safe_float(st.get("sat_pct")) + dark = safe_float(st.get("dark_pct")) + dyn = safe_float(st.get("p95_p05")) + mean = safe_float(st.get("mean")) + + if sat > args.warn_sat_pct: + risks.append(f"{ch}: saturação global alta ({sat:.2f}%).") + if dark > args.warn_dark_pct: + risks.append(f"{ch}: muitos pixels escuros globalmente ({dark:.2f}%).") + if dyn < args.warn_low_dynamic: + risks.append(f"{ch}: baixa dinâmica global p95-p05={dyn:.4f}.") + if 0.02 < mean < 0.98 and dyn >= args.warn_low_dynamic: + positives.append(f"{ch}: média/dinâmica globais parecem utilizáveis (mean={mean:.3f}, p95-p05={dyn:.3f}).") + + # Alinhamento médio. + if sample_rows: + re_corr = np.mean([safe_float(r.get("re_edge_corr")) for r in sample_rows]) + nir_corr = np.mean([safe_float(r.get("nir_edge_corr")) for r in sample_rows]) + re_shift = np.mean([math.hypot(safe_float(r.get("re_phase_dx")), safe_float(r.get("re_phase_dy"))) for r in sample_rows]) + nir_shift = np.mean([math.hypot(safe_float(r.get("nir_phase_dx")), safe_float(r.get("nir_phase_dy"))) for r in sample_rows]) + + notes.append(f"Correlação média de bordas RGB-RE={re_corr:.3f}, RGB-NIR={nir_corr:.3f}.") + notes.append(f"Shift médio estimado RGB-RE={re_shift:.2f}px, RGB-NIR={nir_shift:.2f}px.") + + if re_shift > args.warn_shift_px: + risks.append(f"RE: shift médio estimado alto ({re_shift:.2f}px). Verificar homografia/fusão.") + if nir_shift > args.warn_shift_px: + risks.append(f"NIR: shift médio estimado alto ({nir_shift:.2f}px). Verificar homografia/fusão.") + if re_corr < args.warn_edge_corr: + risks.append(f"RE: correlação média de borda baixa ({re_corr:.3f}). Pode indicar desalinhamento ou textura espectral muito diferente.") + if nir_corr < args.warn_edge_corr: + risks.append(f"NIR: correlação média de borda baixa ({nir_corr:.3f}). Pode indicar desalinhamento ou textura espectral muito diferente.") + + # Separabilidade simples por classes. + separability = {} + class_items = list(class_summary.items()) + for feature in ALL_FEATURES: + vals = [] + for cls_id, item in class_items: + st = item["features"].get(feature, {}) + vals.append((item["class_name"], safe_float(st.get("mean")), safe_float(st.get("std")))) + + sep_rows = [] + for i in range(len(vals)): + for j in range(i + 1, len(vals)): + a_name, a_mean, a_std = vals[i] + b_name, b_mean, b_std = vals[j] + pooled = math.sqrt((a_std * a_std + b_std * b_std) / 2.0) + EPS + d = abs(a_mean - b_mean) / pooled + sep_rows.append({ + "pair": f"{a_name}_vs_{b_name}", + "effect_size_d": d, + "mean_a": a_mean, + "mean_b": b_mean, + }) + separability[feature] = sep_rows + + # Destaca features com maior separação média. + ranking = [] + for feature, rows in separability.items(): + if rows: + avg_d = float(np.mean([r["effect_size_d"] for r in rows])) + ranking.append((feature, avg_d)) + ranking.sort(key=lambda x: x[1], reverse=True) + + notes.append("Ranking simples de separabilidade média por feature: " + ", ".join([f"{f}={d:.2f}" for f, d in ranking[:8]])) + + for f, d in ranking[:5]: + if d > 0.5: + positives.append(f"{f}: mostra separabilidade média interessante entre classes (d≈{d:.2f}).") + + if warnings: + risks.append(f"Foram gerados {len(warnings)} avisos/exceções. Ver audit_warnings.json.") + + return { + "positives": positives, + "risks": risks, + "notes": notes, + "feature_separability": separability, + "feature_separability_ranking": [{"feature": f, "avg_effect_size_d": d} for f, d in ranking], + } + + +# ============================================================ +# CLI +# ============================================================ + + +def main(): + parser = argparse.ArgumentParser( + description="Auditoria final do tensor multiespectral [R,G,B,RE,NIR] pronto para treinamento." + ) + parser.add_argument("--input_path", required=True, help="Caminho para dataset_root, metas, bins, masks ou arquivo dentro do dataset.") + parser.add_argument("--out_dir", default="audit_multispec_out", help="Pasta de saída da auditoria.") + parser.add_argument("--class-map", default="0:chao,1:cana,2:erva", help="Mapa de classes. Ex: 0:chao,1:cana,2:erva") + parser.add_argument("--ignore-index", type=int, default=255, help="Valor ignorado na máscara.") + parser.add_argument("--limit", type=int, default=0, help="Limita número de amostras para teste rápido. 0 = todas.") + parser.add_argument("--save-visuals", action="store_true", help="Salva painéis visuais por amostra.") + parser.add_argument("--visual-every", type=int, default=10, help="Salva visual a cada N amostras. Amostras com warning sempre são salvas.") + parser.add_argument("--print-every", type=int, default=10, help="Mostra progresso a cada N amostras.") + parser.add_argument("--min-class-pixels", type=int, default=50, help="Mínimo de pixels por classe para estatística por amostra.") + parser.add_argument("--max-pixels-per-feature", type=int, default=25000, help="Amostragem máxima de pixels por feature/amostra para acumuladores.") + parser.add_argument("--warn-sat-pct", type=float, default=1.0, help="Alerta se canal tiver saturação acima deste percentual.") + parser.add_argument("--warn-dark-pct", type=float, default=35.0, help="Alerta se canal tiver pixels <=0.01 acima deste percentual.") + parser.add_argument("--warn-low-dynamic", type=float, default=0.03, help="Alerta se p95-p05 do canal for menor que isso.") + parser.add_argument("--warn-shift-px", type=float, default=3.0, help="Alerta se shift estimado por phase correlation passar disso.") + parser.add_argument("--warn-edge-corr", type=float, default=0.08, help="Alerta se correlação de borda for menor que isso.") + parser.add_argument("--seed", type=int, default=42) + parser.add_argument("--stop-on-error", action="store_true") + + parser.add_argument("--build-fixed-dataset", action="store_true", + help="Cria uma cópia filtrada do dataset em fixed/group, mantendo apenas amostras aprovadas para treino.") + + parser.add_argument("--fixed-out-root", default="", + help="Raiz de saída do dataset filtrado. Se vazio, usa dataset_root/../../fixed/group.") + + parser.add_argument("--clean-max-abs-shift-px", type=float, default=22.0, + help="Reprova amostras com shift absoluto extremo em RE ou NIR.") + + parser.add_argument("--clean-max-dev-shift-px", type=float, default=7.0, + help="Reprova amostras cujo shift foge muito da mediana robusta do grupo.") + + parser.add_argument("--clean-min-edge-corr", type=float, default=0.045, + help="Correlação mínima de borda para aprovar. Valor baixo porque canais espectrais naturalmente diferem do RGB.") + + parser.add_argument("--clean-reject-low-corr-only-if-shift-bad", action="store_true", + help="Só reprova baixa correlação se também houver desvio geométrico alto.") + + parser.add_argument("--clean-min-target-pct", type=float, default=0.0025, + help="Percentual mínimo da classe alvo em grupos com cana/erva. 0.0025 = 0.25%%. Use 0 para desativar.") + + parser.add_argument("--clean-copy-previews", action="store_true", + help="Copia previews quando existirem.") + + parser.add_argument("--save-rejected-previews", action="store_true", + help="Salva PNGs leves das amostras rejeitadas em fixed/rejected_previews para inspeção visual.") + + parser.add_argument("--rejected-preview-source", default="auto", + choices=["auto", "preview", "audit_panel", "rgb_tensor"], + help="Fonte do preview dos rejeitados: preview original, painel audit, RGB reconstruído ou automático.") + + parser.add_argument("--rejected-preview-max-width", type=int, default=640, + help="Largura máxima dos previews rejeitados salvos.") + + args = parser.parse_args() + + audit_dataset(args) + + +if __name__ == "__main__": + main() diff --git a/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset_manual.py b/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset_manual.py new file mode 100644 index 000000000..a132a17e7 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/audit_dataset_manual.py @@ -0,0 +1,2085 @@ +import argparse +import csv +import json +import math +import unicodedata +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, List, Optional, Tuple, Any + +import cv2 +import numpy as np +import shutil + +try: + from core.raw_processor_core import RawProcessorCore +except Exception: + RawProcessorCore = None + + +# ============================================================ +# Auditoria final de dataset multiespectral OAK-FCC-3P +# ------------------------------------------------------------ +# Foco: tensor final pronto para o modelo, CHW float32: +# [R, G, B, RE, NIR] +# +# Gera: +# - audit_summary.json +# - audit_samples.csv +# - audit_by_class_channel.csv +# - audit_by_sample_class_channel.csv +# - audit_warnings.json +# - visuals/*.png com painéis de sanidade +# +# Layout esperado: +# dataset_root/ +# metas/*.json +# bins/*.raw ou *.bin +# masks/*.png/.tif/.npy +# previews/*.png opcional +# +# Também aceita apontar diretamente para dataset_root/metas etc. +# +# python -m utils.audit_dataset --input_path .\dataset\original\group\ --out_dir audit_multispec_out --save-visuals --visual-every 10 --build-fixed-dataset --save-rejected-previews --rejected-preview-source auto --rejected-preview-max-width 640 --clean-reject-low-corr-only-if-shift-bad --clean-max-abs-shift-px 22 --clean-max-dev-shift-px 7 --clean-min-edge-corr 0.045 --clean-min-target-pct 0.0025 +# +# ============================================================ + + +CHANNELS = ["R", "G", "B", "RE", "NIR"] +DERIVED = [ + "NDVI", + "NDRE", + "NIR_minus_RE", + "NIR_over_RE", + "NIR_over_R", + "RE_over_R", + "NIR_over_G", + "RE_over_G", + "G_minus_R", +] +ALL_FEATURES = CHANNELS + DERIVED + +DEFAULT_CLASS_MAP = { + 0: "chao", + 1: "cana", + 2: "erva", +} + +# Labelmap das mascaras coloridas, informado pelo projeto. +# Valores em RGB, como normalmente aparecem em ferramentas de anotacao. +# Como o OpenCV le PNG em BGR, a funcao decode_color_mask converte internamente. +DEFAULT_MASK_COLOR_MAP_RGB = { + (128, 0, 0): 0, # chao + (0, 0, 128): 1, # cana + (0, 128, 0): 2, # erva +} + +# Paleta de visualizacao em BGR para o painel GT mask. +# Mantem visual equivalente ao labelmap RGB acima. +DEFAULT_VIS_PALETTE_BGR = { + 0: (0, 0, 128), # chao: RGB 128,0,0 + 1: (128, 0, 0), # cana: RGB 0,0,128 + 2: (0, 128, 0), # erva: RGB 0,128,0 + 255: (0, 0, 0), +} + +EPS = 1e-6 + + +# ============================================================ +# Utilidades básicas +# ============================================================ + + +def ensure_dir(path: Path | str) -> Path: + p = Path(path) + p.mkdir(parents=True, exist_ok=True) + return p + + +def load_json(path: Path) -> dict: + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + + +def write_json(path: Path, data: Any): + with open(path, "w", encoding="utf-8") as f: + json.dump(data, f, ensure_ascii=False, indent=2) + + +def safe_float(x: Any, default: float = 0.0) -> float: + try: + if x is None: + return default + v = float(x) + if math.isnan(v) or math.isinf(v): + return default + return v + except Exception: + return default + + +def parse_class_map(text: Optional[str]) -> Dict[int, str]: + if not text: + return dict(DEFAULT_CLASS_MAP) + + out = {} + # Formatos aceitos: + # "0:chao,1:cana,2:erva" + # "0=chao,1=cana,2=erva" + for item in text.split(","): + item = item.strip() + if not item: + continue + if ":" in item: + k, v = item.split(":", 1) + elif "=" in item: + k, v = item.split("=", 1) + else: + raise ValueError(f"Classe inválida em --class-map: {item}") + out[int(k.strip())] = v.strip() + + return out + + +def parse_csv_set(text: Optional[str]) -> set: + """ + Converte lista simples separada por vírgula em set normalizado. + + Ex: + "chao" -> {"chao"} + "chao, chao_cana" -> {"chao", "chao_cana"} + """ + if not text: + return set() + + out = set() + for item in str(text).split(","): + item = item.strip() + if item: + out.add(item.lower()) + return out + + +def normalize_to_u8(x: np.ndarray, p_low: float = 1.0, p_high: float = 99.0) -> np.ndarray: + arr = x.astype(np.float32, copy=False) + finite = np.isfinite(arr) + if not np.any(finite): + return np.zeros(arr.shape, dtype=np.uint8) + + vals = arr[finite] + lo = np.percentile(vals, p_low) + hi = np.percentile(vals, p_high) + if hi <= lo + EPS: + hi = lo + 1.0 + + y = (arr - lo) / (hi - lo) + y = np.clip(y, 0, 1) + return (y * 255.0).astype(np.uint8) + + +def float01_to_u8(x: np.ndarray) -> np.ndarray: + return np.clip(x.astype(np.float32) * 255.0, 0, 255).astype(np.uint8) + + +def rgb_from_tensor(tensor: np.ndarray, stretch: bool = False) -> np.ndarray: + rgb = np.transpose(tensor[:3], (1, 2, 0)).astype(np.float32) + if stretch: + chans = [normalize_to_u8(rgb[:, :, i]) for i in range(3)] + rgb_u8 = np.dstack(chans) + else: + rgb_u8 = float01_to_u8(rgb) + return cv2.cvtColor(rgb_u8, cv2.COLOR_RGB2BGR) + + +def apply_colormap_gray(x: np.ndarray, stretch: bool = True) -> np.ndarray: + if stretch: + u8 = normalize_to_u8(x) + else: + u8 = float01_to_u8(x) + return cv2.applyColorMap(u8, cv2.COLORMAP_VIRIDIS) + + +def cv_text(text: Any) -> str: + """ + OpenCV putText nao lida bem com acentos/cedilha em muitos ambientes. + Converte qualquer texto para ASCII seguro. + """ + s = str(text) + s = unicodedata.normalize("NFKD", s) + s = s.encode("ascii", "ignore").decode("ascii") + return s + + +def put_label(img: np.ndarray, title: str, subtitle: str = "") -> np.ndarray: + out = img.copy() + title = cv_text(title) + subtitle = cv_text(subtitle) + cv2.rectangle(out, (0, 0), (out.shape[1], 58 if subtitle else 36), (0, 0, 0), -1) + cv2.putText(out, title, (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.65, (0, 255, 255), 2, cv2.LINE_AA) + if subtitle: + cv2.putText(out, subtitle[:120], (10, 48), cv2.FONT_HERSHEY_SIMPLEX, 0.45, (255, 255, 255), 1, cv2.LINE_AA) + return out + + +def resize_keep(img: np.ndarray, size: Tuple[int, int]) -> np.ndarray: + return cv2.resize(img, size, interpolation=cv2.INTER_AREA) + + +def make_grid(panels: List[Tuple[str, np.ndarray, str]], panel_w: int = 360) -> np.ndarray: + rendered = [] + for title, img, subtitle in panels: + scale = panel_w / img.shape[1] + panel_h = max(1, int(img.shape[0] * scale)) + small = cv2.resize(img, (panel_w, panel_h), interpolation=cv2.INTER_AREA) + rendered.append(put_label(small, title, subtitle)) + + if not rendered: + return np.zeros((200, 400, 3), dtype=np.uint8) + + max_h = max(x.shape[0] for x in rendered) + padded = [] + for img in rendered: + if img.shape[0] < max_h: + pad = np.zeros((max_h - img.shape[0], img.shape[1], 3), dtype=np.uint8) + img = np.vstack([img, pad]) + padded.append(img) + + cols = 3 + rows = [] + gap_w = np.full((max_h, 12, 3), 25, dtype=np.uint8) + gap_h = np.full((12, cols * panel_w + (cols - 1) * 12, 3), 25, dtype=np.uint8) + + for i in range(0, len(padded), cols): + row_imgs = padded[i:i + cols] + while len(row_imgs) < cols: + row_imgs.append(np.zeros_like(padded[0])) + row = np.hstack([row_imgs[0], gap_w, row_imgs[1], gap_w, row_imgs[2]]) + rows.append(row) + + canvas = rows[0] + for r in rows[1:]: + canvas = np.vstack([canvas, gap_h, r]) + return canvas + + +# ============================================================ +# Detecção de layout e leitura dos tensores/máscaras +# ============================================================ + + +def is_dataset_root(path: Path) -> bool: + """ + Um dataset final válido tem, no mínimo: + root/metas + root/bins + + masks/ e previews/ são opcionais para a auditoria, embora masks/ seja + necessário para estatísticas por classe. + """ + return (path / "metas").is_dir() and (path / "bins").is_dir() + + +def find_dataset_roots(path: Path) -> List[Path]: + """ + Detecta um ou vários datasets. + + Suporta: + 1) dataset direto: + root/metas + root/bins + root/masks + + 2) subpasta do dataset: + root/metas, root/bins etc, mas input_path=root/metas ou root/bins + + 3) super-root com grupos dentro: + group/chao/metas + group/chao/bins + group/chao_cana/metas + group/chao_cana/bins + ... + + Isso resolve o caso: + --input_path dataset/1024x640/group + quando group contém vários datasets filhos. + """ + p = path.resolve() + + candidates: List[Path] = [] + if p.is_file(): + candidates.extend([p.parent, p.parent.parent]) + else: + candidates.extend([p, p.parent]) + + roots: List[Path] = [] + + # Primeiro tenta o próprio caminho ou pai direto. + for c in candidates: + if c.name.lower() in ("metas", "bins", "masks", "previews"): + root = c.parent + else: + root = c + + if is_dataset_root(root): + roots.append(root) + + # Se não achou, procura recursivamente datasets filhos. + search_base = p if p.is_dir() else p.parent + if not roots and search_base.exists(): + for metas_dir in search_base.rglob("metas"): + root = metas_dir.parent + if is_dataset_root(root): + roots.append(root) + + # Remove duplicados preservando ordem. + unique: List[Path] = [] + seen = set() + for r in roots: + rr = r.resolve() + if rr not in seen: + unique.append(rr) + seen.add(rr) + + if not unique: + raise FileNotFoundError( + f"Não consegui detectar dataset_root a partir de {path}. " + "Esperado root/metas e root/bins, ou um super-root contendo grupos com metas/bins." + ) + + return unique + + +def find_dataset_root(path: Path) -> Path: + """ + Compatibilidade com chamadas antigas: retorna o primeiro root encontrado. + Para a auditoria principal, use find_dataset_roots(). + """ + return find_dataset_roots(path)[0] + + +def list_meta_files(dataset_root: Path) -> List[Path]: + metas = sorted((dataset_root / "metas").glob("*.json")) + if not metas: + raise RuntimeError(f"Nenhum .json encontrado em {dataset_root / 'metas'}") + return metas + + +def find_sibling(dataset_root: Path, subdir: str, stem: str, exts: Tuple[str, ...]) -> Optional[Path]: + folder = dataset_root / subdir + if not folder.is_dir(): + return None + for ext in exts: + p = folder / f"{stem}{ext}" + if p.exists(): + return p + return None + + +def resolve_mask_path(dataset_root: Path, meta_path: Path) -> Optional[Path]: + """ + Resolve mascara da amostra. + + Importante: o sample_name usado no relatorio pode receber prefixo do grupo + tipo 'chao__2026...', mas o arquivo real da mascara continua usando + meta_path.stem, sem prefixo. Esse foi o bug da rodada anterior. + """ + real_stem = meta_path.stem + masks_dir = dataset_root / "masks" + + candidates: List[Path] = [] + for ext in (".png", ".tif", ".tiff", ".npy"): + candidates.append(masks_dir / f"{real_stem}{ext}") + + for suffix in ("_mask", "_gt", "_label", "_labels", "_seg"): + for ext in (".png", ".tif", ".tiff", ".npy"): + candidates.append(masks_dir / f"{real_stem}{suffix}{ext}") + + if masks_dir.is_dir(): + candidates.extend(sorted(masks_dir.glob(f"{real_stem}*.*"))) + + for p in candidates: + if p.exists() and p.suffix.lower() in (".png", ".tif", ".tiff", ".npy"): + return p + + return None + + +def resolve_tensor_payload(meta_path: Path, dataset_root: Path, meta: dict) -> Optional[Path]: + """ + Resolve payload final já pronto, quando o dataset já contém MULTISPEC salvo. + Para RAW_BRUTO multi por câmera, use build_multispec_from_raw_native_multi(). + """ + stem = meta_path.stem + bins = dataset_root / "bins" + + candidates = [] + saved_payload_path = meta.get("saved_payload_path") + if saved_payload_path: + sp = Path(saved_payload_path) + candidates.extend([ + bins / sp.name, + meta_path.parent / sp, + dataset_root / sp, + ]) + + candidates.extend([ + bins / f"{stem}.raw", + bins / f"{stem}.bin", + bins / f"{stem}_multispec.raw", + bins / f"{stem}_multispec.bin", + bins / f"{stem}_offline_multispec.raw", + bins / f"{stem}_offline_multispec.bin", + ]) + + for c in candidates: + if c.exists(): + return c + return None + + +def resolve_camera_payloads(meta_path: Path, dataset_root: Path, meta: dict) -> Dict[str, Path]: + """ + Resolve os .bin/.raw por câmera quando saved_payload_type='raw_native_multi'. + """ + stem = meta_path.stem + bins = dataset_root / "bins" + paths = {} + + saved_payload_paths = meta.get("saved_payload_paths", {}) or {} + for cam_id, fname in saved_payload_paths.items(): + fp = Path(fname) + candidates = [ + bins / fp.name, + meta_path.parent / fname, + dataset_root / fname, + bins / f"{stem}_{cam_id}.bin", + bins / f"{stem}_{cam_id}.raw", + bins / f"{stem}_{cam_id.lower()}.bin", + bins / f"{stem}_{cam_id.lower()}.raw", + ] + + found = None + for c in candidates: + if Path(c).exists(): + found = Path(c) + break + + if found is None: + raise FileNotFoundError(f"Payload bruto não encontrado para {cam_id} em {meta_path.name}: {fname}") + + paths[cam_id] = found + + return paths + + +def resolve_module_params_path(meta_path: Path, dataset_root: Path, meta: dict) -> str: + """ + Tenta encontrar o module_params.json usado para reconstruir o tensor. + A ideia é usar exatamente o mesmo caminho salvo no meta quando existir. + """ + calib_path = meta.get("camera_params_json") or meta.get("module_params_json") or "calibration/module_params.json" + p = Path(calib_path) + + candidates = [] + if p.is_absolute(): + candidates.append(p) + else: + candidates.extend([ + Path.cwd() / p, + meta_path.parent / p, + dataset_root / p, + dataset_root.parent / p, + dataset_root.parent.parent / p, + Path.cwd() / "calibration" / "module_params.json", + ]) + + for c in candidates: + if c.exists(): + return str(c) + + raise FileNotFoundError( + f"module_params.json não encontrado. Valor no meta={calib_path}. " + "Sem ele eu não consigo aplicar homografia/flat/radiometria igual ao pipeline final." + ) + + +def get_raw_processor_core( + core_cache: Dict[Tuple[int, int, str, str], Any], + sensor_width: int, + sensor_height: int, + bayer: str, + calib_path: str, +): + """ + Reaproveita RawProcessorCore entre amostras. + + Isso evita recarregar flat-field e recriar caches de gain a cada imagem. + A chave considera tamanho, bayer e module_params.json. + """ + key = (int(sensor_width), int(sensor_height), str(bayer), str(Path(calib_path).resolve())) + if key not in core_cache: + if RawProcessorCore is None: + raise RuntimeError( + "Não consegui importar RawProcessorCore. Rode com python -m utils.audit_dataset " + "a partir da raiz do projeto, igual você fez, e confirme se core/raw_processor_core.py existe." + ) + core_cache[key] = RawProcessorCore( + sensor_width=int(sensor_width), + sensor_height=int(sensor_height), + bayer_pattern=str(bayer), + calibration_json_path=str(calib_path), + ) + return core_cache[key] + + +def build_multispec_from_raw_native_multi(meta_path: Path, dataset_root: Path, meta: dict, core_cache: Dict[Tuple[int, int, str, str], Any]) -> Tuple[np.ndarray, Path]: + """ + Reconstrói o tensor final [R,G,B,RE,NIR] a partir do RAW_BRUTO multi por câmera. + + Usa o mesmo princípio do script visual antigo: + - saved_payload_paths + - saved_payload_shapes + - saved_payload_dtypes + - stream_meta.camera_info + - actual_camera_controls/startup_camera_controls + - module_params.json com homografia/flat/radiometria + """ + saved_dtypes = meta.get("saved_payload_dtypes", {}) or {} + saved_shapes = meta.get("saved_payload_shapes", {}) or {} + cam_paths = resolve_camera_payloads(meta_path, dataset_root, meta) + + frame = {} + for cam_id, payload_path in cam_paths.items(): + saved_dtype = saved_dtypes.get(cam_id) + saved_shape = saved_shapes.get(cam_id) + if saved_dtype is None or saved_shape is None: + raise RuntimeError(f"Faltam dtype/shape para {cam_id} em {meta_path.name}") + + arr = np.fromfile(str(payload_path), dtype=np.dtype(saved_dtype)).reshape(tuple(saved_shape)) + frame[cam_id] = arr + + sensor_width = int(meta.get("sensor_width", 1280)) + sensor_height = int(meta.get("sensor_height", 800)) + bayer = meta.get("bayer_pattern", "RGGB") + calib_path = resolve_module_params_path(meta_path, dataset_root, meta) + + core = get_raw_processor_core( + core_cache=core_cache, + sensor_width=sensor_width, + sensor_height=sensor_height, + bayer=bayer, + calib_path=calib_path, + ) + + stream_meta = meta.get("stream_meta", {}) or {} + processing_meta = dict(stream_meta) + + if meta.get("actual_camera_controls") is not None: + processing_meta["actual_camera_controls"] = meta.get("actual_camera_controls") + if meta.get("startup_camera_controls") is not None: + processing_meta["startup_camera_controls"] = meta.get("startup_camera_controls") + if meta.get("radiometric_last_result") is not None: + processing_meta["radiometric_last_result"] = meta.get("radiometric_last_result") + + tensor = core.build_infer_tensor_from_stream(frame, processing_meta, 5) + if tensor is None: + raise RuntimeError(f"RawProcessorCore retornou tensor None para {meta_path.name}") + + tensor = np.asarray(tensor, dtype=np.float32) + if tensor.ndim != 3: + raise RuntimeError(f"Tensor reconstruído inválido em {meta_path.name}: shape={tensor.shape}") + if tensor.shape[0] != 5 and tensor.shape[-1] == 5: + tensor = np.transpose(tensor, (2, 0, 1)) + if tensor.shape[0] < 5: + raise RuntimeError(f"Tensor reconstruído precisa ter 5 canais, recebido shape={tensor.shape}") + + # Retorna o primeiro payload bruto só como referência de origem no CSV. + first_payload = next(iter(cam_paths.values())) + return np.ascontiguousarray(tensor[:5]), first_payload + + +def load_multispec_tensor(meta_path: Path, dataset_root: Path, core_cache: Dict[Tuple[int, int, str, str], Any]) -> Tuple[np.ndarray, dict, Path]: + meta = load_json(meta_path) + saved_type = meta.get("saved_payload_type") + + # Caso 1: dataset original bruto por câmera. Aqui o script monta o tensor final. + if saved_type == "raw_native_multi" or "saved_payload_paths" in meta: + tensor, source_payload = build_multispec_from_raw_native_multi(meta_path, dataset_root, meta, core_cache) + return tensor, meta, source_payload + + # Caso 2: dataset já normalizado/MULTISPEC salvo como payload único. + payload_path = resolve_tensor_payload(meta_path, dataset_root, meta) + if payload_path is None: + raise FileNotFoundError(f"Payload tensor final não encontrado para {meta_path.name}") + + dtype = meta.get("saved_payload_dtype", "float32") + shape = meta.get("saved_payload_shape") + + if shape is None: + # Alguns JSONs podem guardar isso em outro campo. + shape = meta.get("tensor_shape") or meta.get("shape") + + if shape is None: + raise RuntimeError( + f"Shape do tensor não encontrado em {meta_path.name}. " + "Esperado saved_payload_shape=[5,H,W]." + ) + + arr = np.fromfile(str(payload_path), dtype=np.dtype(dtype)).reshape(tuple(shape)) + arr = arr.astype(np.float32, copy=False) + + if arr.ndim != 3: + raise RuntimeError(f"Tensor inválido em {payload_path}: shape={arr.shape}") + + if arr.shape[0] != 5 and arr.shape[-1] == 5: + arr = np.transpose(arr, (2, 0, 1)) + + if arr.shape[0] < 5: + raise RuntimeError(f"Tensor precisa ter 5 canais [R,G,B,RE,NIR], recebido shape={arr.shape}") + + return arr[:5], meta, payload_path + + +def decode_color_mask(mask_img: np.ndarray, ignore_index: int) -> np.ndarray: + """ + Converte mascara colorida do labelmap para indices de classe. + + Seu labelmap atual em RGB: + chao = 128,0,0 -> classe 0 + cana = 0,0,128 -> classe 1 + erva = 0,128,0 -> classe 2 + + O OpenCV le PNG como BGR, entao fazemos a conversao RGB->BGR antes de comparar. + """ + if mask_img.ndim == 2: + return mask_img.astype(np.int32, copy=False) + + if mask_img.shape[2] == 4: + bgr = mask_img[:, :, :3] + else: + bgr = mask_img[:, :, :3] + + out = np.full(mask_img.shape[:2], ignore_index, dtype=np.int32) + + for rgb_color, cls_id in DEFAULT_MASK_COLOR_MAP_RGB.items(): + r, g, b = rgb_color + bgr_color = np.array([b, g, r], dtype=np.uint8) + hit = np.all(bgr == bgr_color, axis=2) + out[hit] = int(cls_id) + + # Fallback: se por algum motivo a imagem ja veio em ordem RGB, tenta tambem RGB direto. + # Isso evita quebrar caso alguma leitura futura nao use cv2. + for rgb_color, cls_id in DEFAULT_MASK_COLOR_MAP_RGB.items(): + rgb_arr = np.array(rgb_color, dtype=np.uint8) + hit = np.all(bgr == rgb_arr, axis=2) + # So preenche pixels ainda nao identificados para evitar troca em cores ambiguas. + out[(out == ignore_index) & hit] = int(cls_id) + + return out + + +def load_mask(mask_path: Optional[Path], target_hw: Tuple[int, int], ignore_index: int) -> Optional[np.ndarray]: + if mask_path is None: + return None + + if mask_path.suffix.lower() == ".npy": + mask = np.load(str(mask_path)) + if mask.ndim == 3: + mask = decode_color_mask(mask.astype(np.uint8), ignore_index) + else: + mask_img = cv2.imread(str(mask_path), cv2.IMREAD_UNCHANGED) + if mask_img is None: + raise RuntimeError(f"Falha ao ler mascara: {mask_path}") + mask = decode_color_mask(mask_img, ignore_index) + + mask = mask.astype(np.int32, copy=False) + h, w = target_hw + if mask.shape[:2] != (h, w): + mask = cv2.resize(mask, (w, h), interpolation=cv2.INTER_NEAREST) + return mask + + +def mask_unique_summary(mask: Optional[np.ndarray], max_items: int = 20) -> str: + if mask is None: + return "none" + vals, counts = np.unique(mask, return_counts=True) + parts = [] + for v, c in zip(vals[:max_items], counts[:max_items]): + cls_name = DEFAULT_CLASS_MAP.get(int(v), "ignore" if int(v) == 255 else "unk") + parts.append(f"{int(v)}:{cls_name}:{int(c)}") + if len(vals) > max_items: + parts.append("...") + return ",".join(parts) + + +# ============================================================ +# Features espectrais e estatísticas +# ============================================================ + + +def compute_feature_maps(tensor: np.ndarray) -> Dict[str, np.ndarray]: + r, g, b, re, nir = [tensor[i].astype(np.float32, copy=False) for i in range(5)] + + features = { + "R": r, + "G": g, + "B": b, + "RE": re, + "NIR": nir, + "NDVI": (nir - r) / (nir + r + EPS), + "NDRE": (nir - re) / (nir + re + EPS), + "NIR_minus_RE": nir - re, + "NIR_over_RE": nir / (re + EPS), + "NIR_over_R": nir / (r + EPS), + "RE_over_R": re / (r + EPS), + "NIR_over_G": nir / (g + EPS), + "RE_over_G": re / (g + EPS), + "G_minus_R": g - r, + } + + # Evita explosões absurdas em razão quando denominador está quase zero. + for k in list(features.keys()): + features[k] = np.nan_to_num(features[k], nan=0.0, posinf=0.0, neginf=0.0).astype(np.float32) + return features + + +def calc_stats(values: np.ndarray, raw01: bool = False) -> Dict[str, float]: + v = values.astype(np.float32, copy=False) + v = v[np.isfinite(v)] + if v.size == 0: + return { + "count": 0, + "mean": 0.0, + "std": 0.0, + "min": 0.0, + "p01": 0.0, + "p05": 0.0, + "p25": 0.0, + "p50": 0.0, + "p75": 0.0, + "p95": 0.0, + "p99": 0.0, + "max": 0.0, + "iqr": 0.0, + "p95_p05": 0.0, + "dark_pct": 0.0, + "sat_pct": 0.0, + } + + p = np.percentile(v, [1, 5, 25, 50, 75, 95, 99]) + out = { + "count": int(v.size), + "mean": float(np.mean(v)), + "std": float(np.std(v)), + "min": float(np.min(v)), + "p01": float(p[0]), + "p05": float(p[1]), + "p25": float(p[2]), + "p50": float(p[3]), + "p75": float(p[4]), + "p95": float(p[5]), + "p99": float(p[6]), + "max": float(np.max(v)), + "iqr": float(p[4] - p[2]), + "p95_p05": float(p[5] - p[1]), + "dark_pct": 0.0, + "sat_pct": 0.0, + } + + if raw01: + out["dark_pct"] = float(np.mean(v <= 0.01) * 100.0) + out["sat_pct"] = float(np.mean(v >= 0.99) * 100.0) + + return out + + +@dataclass +class RunningFeatureStats: + values: Dict[str, List[float]] = field(default_factory=lambda: {f: [] for f in ALL_FEATURES}) + counts: Dict[str, int] = field(default_factory=lambda: {f: 0 for f in ALL_FEATURES}) + pixel_count: int = 0 + + def add(self, feature_name: str, values: np.ndarray, max_samples: int = 25000): + v = values.astype(np.float32, copy=False) + v = v[np.isfinite(v)] + if v.size == 0: + return + self.pixel_count += int(v.size) + self.counts[feature_name] += int(v.size) + + if v.size > max_samples: + idx = np.random.choice(v.size, size=max_samples, replace=False) + v = v[idx] + self.values[feature_name].extend(v.tolist()) + + def summarize(self) -> Dict[str, Dict[str, float]]: + out = {} + for f, vals in self.values.items(): + arr = np.asarray(vals, dtype=np.float32) + raw01 = f in CHANNELS + s = calc_stats(arr, raw01=raw01) + s["total_pixels_seen"] = int(self.counts.get(f, 0)) + out[f] = s + return out + + +# ============================================================ +# Alinhamento por borda/correlação +# ============================================================ + + +def gradient_mag(x: np.ndarray) -> np.ndarray: + u8 = normalize_to_u8(x) + gx = cv2.Sobel(u8, cv2.CV_32F, 1, 0, ksize=3) + gy = cv2.Sobel(u8, cv2.CV_32F, 0, 1, ksize=3) + mag = cv2.magnitude(gx, gy) + return mag.astype(np.float32) + + +def estimate_shift_phase(a: np.ndarray, b: np.ndarray) -> Tuple[float, float, float]: + # dx, dy estimados entre mapas. Valores grandes sugerem desalinhamento residual. + aa = normalize_to_u8(a).astype(np.float32) + bb = normalize_to_u8(b).astype(np.float32) + try: + (dx, dy), response = cv2.phaseCorrelate(aa, bb) + return float(dx), float(dy), float(response) + except Exception: + return 0.0, 0.0, 0.0 + + +def edge_agreement(a: np.ndarray, b: np.ndarray) -> Dict[str, float]: + ga = gradient_mag(a) + gb = gradient_mag(b) + + va = ga.reshape(-1) + vb = gb.reshape(-1) + if np.std(va) < EPS or np.std(vb) < EPS: + corr = 0.0 + else: + corr = float(np.corrcoef(va, vb)[0, 1]) + + dx, dy, resp = estimate_shift_phase(ga, gb) + return { + "edge_corr": corr, + "phase_dx": dx, + "phase_dy": dy, + "phase_response": resp, + } + + +def make_edge_overlay(tensor: np.ndarray) -> np.ndarray: + r, g, b, re, nir = [tensor[i] for i in range(5)] + rgb_gray = (0.299 * r + 0.587 * g + 0.114 * b).astype(np.float32) + + e_rgb = normalize_to_u8(gradient_mag(rgb_gray), 5, 99) + e_re = normalize_to_u8(gradient_mag(re), 5, 99) + e_nir = normalize_to_u8(gradient_mag(nir), 5, 99) + + # BGR: RGB edge em verde, RE em vermelho, NIR em azul. + overlay = np.zeros((tensor.shape[1], tensor.shape[2], 3), dtype=np.uint8) + overlay[:, :, 1] = e_rgb + overlay[:, :, 2] = e_re + overlay[:, :, 0] = e_nir + return overlay + + +# ============================================================ +# Visualizações por amostra +# ============================================================ + + +def colorize_mask(mask: Optional[np.ndarray], class_map: Dict[int, str], target_hw: Tuple[int, int]) -> np.ndarray: + h, w = target_hw + out = np.zeros((h, w, 3), dtype=np.uint8) + if mask is None: + return out + + palette = dict(DEFAULT_VIS_PALETTE_BGR) + for cls_id in np.unique(mask): + cls_id = int(cls_id) + if cls_id in palette: + out[mask == cls_id] = palette[cls_id] + elif cls_id in class_map: + rng = np.random.default_rng(cls_id) + out[mask == cls_id] = rng.integers(40, 220, size=3) + return out + + +def mask_edges_on_rgb(rgb_bgr: np.ndarray, mask: Optional[np.ndarray]) -> np.ndarray: + out = rgb_bgr.copy() + if mask is None: + return out + m = mask.astype(np.uint8) + edges = cv2.Canny(m, 0, 1) + out[edges > 0] = (0, 255, 255) + return out + + +def make_sample_visual( + sample_name: str, + tensor: np.ndarray, + mask: Optional[np.ndarray], + class_map: Dict[int, str], + sample_stats: Dict[str, Any], +) -> np.ndarray: + features = compute_feature_maps(tensor) + h, w = tensor.shape[1], tensor.shape[2] + + rgb = rgb_from_tensor(tensor, stretch=False) + rgb_stretch = rgb_from_tensor(tensor, stretch=True) + re = apply_colormap_gray(features["RE"], stretch=True) + nir = apply_colormap_gray(features["NIR"], stretch=True) + ndvi = apply_colormap_gray(features["NDVI"], stretch=True) + ndre = apply_colormap_gray(features["NDRE"], stretch=True) + diff = apply_colormap_gray(features["NIR_minus_RE"], stretch=True) + ratio = apply_colormap_gray(features["NIR_over_RE"], stretch=True) + edge = make_edge_overlay(tensor) + mask_bgr = colorize_mask(mask, class_map, (h, w)) + + overlay_mask = rgb.copy() + if mask is not None: + overlay_mask = cv2.addWeighted(rgb, 0.65, mask_bgr, 0.35, 0) + + mask_edges = mask_edges_on_rgb(rgb, mask) + + align = sample_stats.get("alignment", {}) + re_align = align.get("RGBgray_vs_RE", {}) + nir_align = align.get("RGBgray_vs_NIR", {}) + + panels = [ + ("RGB tensor", rgb, sample_name), + ("RGB stretch", rgb_stretch, "visual apenas para contraste"), + ("GT mask", mask_bgr, f"unique={mask_unique_summary(mask)}"), + ("Mask overlay", overlay_mask, "GT over RGB"), + ("Mask edges", mask_edges, "GT edges over RGB"), + ("RE", re, "canal 3 | stretch p1-p99"), + ("NIR", nir, "canal 4 | stretch p1-p99"), + ("NDVI", ndvi, "(NIR-R)/(NIR+R)"), + ("NDRE", ndre, "(NIR-RE)/(NIR+RE)"), + ("NIR - RE", diff, "diferença direta"), + ("NIR / RE", ratio, "razão com eps"), + ( + "Edge overlay", + edge, + f"G=RGB | R=RE | B=NIR | REcorr={safe_float(re_align.get('edge_corr')):.3f} NIRcorr={safe_float(nir_align.get('edge_corr')):.3f}", + ), + ] + return make_grid(panels, panel_w=380) + + +# ============================================================ +# Auditoria principal +# ============================================================ + + +def resolve_rejected_preview_root(fixed_root: Path) -> Path: + # fixed_root normalmente é dataset/fixed/group + # queremos dataset/fixed/rejected_previews + if fixed_root.name == "group": + return fixed_root.parent / "rejected_previews" + return fixed_root / "_rejected_previews" + + +def find_original_preview(dataset_root: Path, real_stem: str) -> Optional[Path]: + previews_dir = dataset_root / "previews" + if not previews_dir.is_dir(): + return None + + for ext in (".png", ".jpg", ".jpeg", ".webp"): + p = previews_dir / f"{real_stem}{ext}" + if p.exists(): + return p + + # fallback caso tenha sufixo no nome + matches = [] + for ext in (".png", ".jpg", ".jpeg", ".webp"): + matches.extend(previews_dir.glob(f"{real_stem}*{ext}")) + + return matches[0] if matches else None + + +def resize_preview_max_width(img: np.ndarray, max_width: int) -> np.ndarray: + if img is None or img.size == 0: + return img + if max_width <= 0 or img.shape[1] <= max_width: + return img + + scale = max_width / img.shape[1] + new_w = int(img.shape[1] * scale) + new_h = int(img.shape[0] * scale) + return cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA) + + +def make_rejected_filename(row: Dict[str, Any]) -> str: + sample = str(row.get("sample", "sample")) + reasons = str(row.get("reject_reasons", "")) + + # Nome curto com principais motivos. + tags = [] + if "abs_shift" in reasons: + tags.append("absShift") + if "outlier" in reasons: + tags.append("outlier") + if "low_edge_corr" in reasons: + tags.append("lowCorr") + if "too_small" in reasons: + tags.append("smallTarget") + if "missing_mask" in reasons: + tags.append("noMask") + + tag = "_".join(tags) if tags else "rejected" + return f"{sample}__{tag}.png" + + +def draw_rejected_header(img: np.ndarray, row: Dict[str, Any]) -> np.ndarray: + if img is None or img.size == 0: + return img + + out = img.copy() + h, w = out.shape[:2] + + header_h = 92 + canvas = np.zeros((h + header_h, w, 3), dtype=np.uint8) + canvas[:header_h, :] = (20, 20, 20) + canvas[header_h:, :] = out + + sample = cv_text(row.get("sample", "")) + reasons = cv_text(row.get("reject_reasons", "")) + + re_mag = safe_float(row.get("re_shift_mag")) + nir_mag = safe_float(row.get("nir_shift_mag")) + max_dev = safe_float(row.get("max_shift_dev_px")) + re_corr = safe_float(row.get("re_edge_corr")) + nir_corr = safe_float(row.get("nir_edge_corr")) + + line1 = sample[:120] + line2 = f"RE={re_mag:.1f}px NIR={nir_mag:.1f}px dev={max_dev:.1f}px corrRE={re_corr:.3f} corrNIR={nir_corr:.3f}" + line3 = reasons[:150] + + cv2.putText(canvas, line1, (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.58, (0, 255, 255), 2, cv2.LINE_AA) + cv2.putText(canvas, line2, (10, 52), cv2.FONT_HERSHEY_SIMPLEX, 0.50, (255, 255, 255), 1, cv2.LINE_AA) + cv2.putText(canvas, line3, (10, 78), cv2.FONT_HERSHEY_SIMPLEX, 0.43, (120, 220, 255), 1, cv2.LINE_AA) + + return canvas + + +def save_rejected_preview( + row: Dict[str, Any], + fixed_root: Path, + args, + tensor: Optional[np.ndarray] = None, + mask: Optional[np.ndarray] = None, + class_map: Optional[Dict[int, str]] = None, +): + rejected_root = resolve_rejected_preview_root(fixed_root) + group = str(row.get("group", "unknown")) + dataset_root = Path(str(row["dataset_root"])) + real_stem = str(row["real_stem"]) + + dst_dir = ensure_dir(rejected_root / group) + dst_path = dst_dir / make_rejected_filename(row) + + source_mode = str(args.rejected_preview_source) + + img = None + + # 1) preview original, se existir. + if source_mode in ("auto", "preview"): + preview_path = find_original_preview(dataset_root, real_stem) + if preview_path is not None: + img = cv2.imread(str(preview_path), cv2.IMREAD_COLOR) + + # 2) painel completo do audit, se já existir e o usuário pediu. + # Aqui é útil quando --save-visuals também estiver ligado. + if img is None and source_mode in ("auto", "audit_panel"): + # Procura por qualquer painel visual que contenha o sample no nome. + # Fica em out_dir/visuals, mas não temos out_dir aqui. Então esse modo + # fica mais útil se você passar painel diretamente no futuro. + pass + + # 3) RGB reconstruído do tensor, se disponível. + if img is None and tensor is not None and source_mode in ("auto", "rgb_tensor", "audit_panel"): + img = rgb_from_tensor(tensor, stretch=False) + + if mask is not None and class_map is not None: + mask_bgr = colorize_mask(mask, class_map, (tensor.shape[1], tensor.shape[2])) + img = cv2.addWeighted(img, 0.70, mask_bgr, 0.30, 0) + + if img is None: + # Último fallback: imagem preta com cabeçalho, para não perder rastreabilidade. + img = np.zeros((360, 640, 3), dtype=np.uint8) + + img = resize_preview_max_width(img, int(args.rejected_preview_max_width)) + img = draw_rejected_header(img, row) + cv2.imwrite(str(dst_path), img) + + +def robust_median_mad(values: List[float]) -> Tuple[float, float]: + arr = np.asarray([v for v in values if np.isfinite(v)], dtype=np.float32) + if arr.size == 0: + return 0.0, 1.0 + + med = float(np.median(arr)) + mad = float(np.median(np.abs(arr - med))) + + # Evita divisão por zero em grupos muito estáveis. + if mad < 0.5: + mad = 0.5 + + return med, mad + + +def safe_copy_file(src: Path, dst: Path): + if not src.exists(): + return + ensure_dir(dst.parent) + shutil.copy2(str(src), str(dst)) + + +def resolve_fixed_out_root(dataset_roots: List[Path], args) -> Path: + if args.fixed_out_root: + return Path(args.fixed_out_root).resolve() + + # Esperado: + # dataset/original/group/chao + # dataset/original/group/chao_cana + # + # Queremos: + # dataset/fixed/group + first = dataset_roots[0].resolve() + + # Se o root é .../original/group/chao, sobe 3: chao -> group -> original -> dataset + if first.parent.name == "group" and first.parent.parent.name == "original": + dataset_base = first.parent.parent.parent + return dataset_base / "fixed" / "group" + + # Fallback seguro. + return first.parent / "fixed" / "group" + + +def sample_shift_metrics(row: Dict[str, Any]) -> Dict[str, float]: + re_dx = safe_float(row.get("re_phase_dx")) + re_dy = safe_float(row.get("re_phase_dy")) + nir_dx = safe_float(row.get("nir_phase_dx")) + nir_dy = safe_float(row.get("nir_phase_dy")) + + return { + "re_dx": re_dx, + "re_dy": re_dy, + "nir_dx": nir_dx, + "nir_dy": nir_dy, + "re_mag": math.hypot(re_dx, re_dy), + "nir_mag": math.hypot(nir_dx, nir_dy), + } + + +def build_group_shift_baselines(sample_rows: List[Dict[str, Any]]) -> Dict[str, Dict[str, Tuple[float, float]]]: + grouped: Dict[str, Dict[str, List[float]]] = {} + + for row in sample_rows: + group = str(row.get("group", "unknown")) + grouped.setdefault(group, { + "re_dx": [], + "re_dy": [], + "nir_dx": [], + "nir_dy": [], + "re_mag": [], + "nir_mag": [], + }) + + m = sample_shift_metrics(row) + for k, v in m.items(): + grouped[group][k].append(v) + + baselines: Dict[str, Dict[str, Tuple[float, float]]] = {} + for group, vals in grouped.items(): + baselines[group] = {} + for k, arr in vals.items(): + baselines[group][k] = robust_median_mad(arr) + + return baselines + + +def classify_sample_for_training( + row: Dict[str, Any], + baselines: Dict[str, Dict[str, Tuple[float, float]]], + class_map: Dict[int, str], + args, +) -> Tuple[bool, List[str], Dict[str, float]]: + reasons: List[str] = [] + metrics = sample_shift_metrics(row) + + group = str(row.get("group", "unknown")) + base = baselines.get(group, {}) + + if not bool(row.get("mask_found")): + reasons.append("missing_mask") + + # Shift absoluto extremo. + if metrics["re_mag"] > args.clean_max_abs_shift_px: + reasons.append(f"re_abs_shift_high:{metrics['re_mag']:.2f}") + if metrics["nir_mag"] > args.clean_max_abs_shift_px: + reasons.append(f"nir_abs_shift_high:{metrics['nir_mag']:.2f}") + + # Desvio robusto por grupo. + max_dev = 0.0 + for k in ("re_dx", "re_dy", "nir_dx", "nir_dy", "re_mag", "nir_mag"): + med, mad = base.get(k, (0.0, 1.0)) + dev_abs = abs(metrics[k] - med) + max_dev = max(max_dev, dev_abs) + + if dev_abs > args.clean_max_dev_shift_px: + reasons.append(f"{k}_outlier:val={metrics[k]:.2f},med={med:.2f},dev={dev_abs:.2f}") + + re_corr = safe_float(row.get("re_edge_corr")) + nir_corr = safe_float(row.get("nir_edge_corr")) + low_corr = re_corr < args.clean_min_edge_corr or nir_corr < args.clean_min_edge_corr + shift_bad = max_dev > args.clean_max_dev_shift_px * 0.75 + + if low_corr: + if args.clean_reject_low_corr_only_if_shift_bad: + if shift_bad: + reasons.append(f"low_edge_corr_with_shift:re={re_corr:.3f},nir={nir_corr:.3f}") + else: + reasons.append(f"low_edge_corr:re={re_corr:.3f},nir={nir_corr:.3f}") + + # Filtro semântico opcional por presença mínima da classe alvo. + # Mantém chão puro sem exigir cana/erva. + min_target_pct = float(args.clean_min_target_pct) + if min_target_pct > 0: + h = int(row.get("H", 0) or 0) + w = int(row.get("W", 0) or 0) + total_px = max(1, h * w) + + group_lower = group.lower() + if "cana" in group_lower: + pct_cana = safe_float(row.get("pixels_cana")) / total_px + if pct_cana < min_target_pct: + reasons.append(f"cana_too_small:{pct_cana:.5f}") + + if "erva" in group_lower: + pct_erva = safe_float(row.get("pixels_erva")) / total_px + if pct_erva < min_target_pct: + reasons.append(f"erva_too_small:{pct_erva:.5f}") + + approved = len(reasons) == 0 + + extra = { + "re_shift_mag": metrics["re_mag"], + "nir_shift_mag": metrics["nir_mag"], + "max_shift_dev_px": max_dev, + "re_edge_corr": re_corr, + "nir_edge_corr": nir_corr, + } + + return approved, reasons, extra + + +def copy_sample_to_fixed(row: Dict[str, Any], fixed_root: Path, copy_previews: bool): + dataset_root = Path(str(row["dataset_root"])) + group = str(row["group"]) + real_stem = str(row["real_stem"]) + meta_path = Path(str(row["meta_path"])) + + dst_group = fixed_root / group + + # Copia meta. + safe_copy_file(meta_path, dst_group / "metas" / meta_path.name) + + # Copia máscara. + mask_path = Path(str(row.get("mask_path", ""))) + if str(mask_path) and mask_path.exists(): + safe_copy_file(mask_path, dst_group / "masks" / mask_path.name) + + # Copia payloads do meta. + meta = load_json(meta_path) + + # Caso raw_native_multi: vários payloads. + saved_payload_paths = meta.get("saved_payload_paths", {}) or {} + if saved_payload_paths: + for _, fname in saved_payload_paths.items(): + src = dataset_root / "bins" / Path(fname).name + safe_copy_file(src, dst_group / "bins" / src.name) + + # Caso payload único. + saved_payload_path = meta.get("saved_payload_path") + if saved_payload_path: + src = dataset_root / "bins" / Path(saved_payload_path).name + safe_copy_file(src, dst_group / "bins" / src.name) + + # Fallback usando payload_path do CSV. + payload_path = Path(str(row.get("payload_path", ""))) + if str(payload_path) and payload_path.exists(): + safe_copy_file(payload_path, dst_group / "bins" / payload_path.name) + + # Preview opcional. + if copy_previews: + previews_dir = dataset_root / "previews" + if previews_dir.is_dir(): + for ext in (".png", ".jpg", ".jpeg", ".webp"): + src = previews_dir / f"{real_stem}{ext}" + if src.exists(): + safe_copy_file(src, dst_group / "previews" / src.name) + + +def manual_fit_to_width(img: np.ndarray, max_width: int) -> np.ndarray: + if max_width <= 0 or img.shape[1] <= max_width: + return img + scale = max_width / img.shape[1] + return cv2.resize( + img, + (int(img.shape[1] * scale), int(img.shape[0] * scale)), + interpolation=cv2.INTER_AREA, + ) + + +def draw_manual_review_bar(canvas: np.ndarray, row: Dict[str, Any], idx: int, total: int) -> np.ndarray: + """ + Adiciona uma faixa superior com comandos do modo manual. + """ + h, w = canvas.shape[:2] + bar_h = 86 + out = np.zeros((h + bar_h, w, 3), dtype=np.uint8) + out[:bar_h, :] = (18, 18, 18) + out[bar_h:, :] = canvas + + sample = cv_text(row.get("sample", "")) + group = cv_text(row.get("group", "unknown")) + warnings_count = int(row.get("warnings_count", 0) or 0) + + line1 = f"[{idx}/{total}] group={group} | warnings={warnings_count} | {sample}" + line2 = "A/ENTER = aprovar | R/DEL/BACKSPACE = rejeitar | S = pular | Q/ESC = finalizar parcial" + line3 = cv_text(str(row.get("warning_text", "")))[:170] + + cv2.putText(out, line1[:170], (10, 25), cv2.FONT_HERSHEY_SIMPLEX, 0.58, (0, 255, 255), 2, cv2.LINE_AA) + cv2.putText(out, line2, (10, 53), cv2.FONT_HERSHEY_SIMPLEX, 0.52, (255, 255, 255), 1, cv2.LINE_AA) + if line3: + cv2.putText(out, line3, (10, 78), cv2.FONT_HERSHEY_SIMPLEX, 0.42, (120, 220, 255), 1, cv2.LINE_AA) + + return out + + +def manual_review_decision( + canvas: np.ndarray, + row: Dict[str, Any], + idx: int, + total: int, + args, +) -> str: + """ + Mostra o painel da amostra e espera a decisão humana. + + Retorna: + approved + rejected + skipped + quit + """ + window_name = str(args.manual_window_name) + view = draw_manual_review_bar(canvas, row, idx, total) + view = manual_fit_to_width(view, int(args.manual_window_width)) + + cv2.namedWindow(window_name, cv2.WINDOW_NORMAL) + cv2.imshow(window_name, view) + + while True: + key = cv2.waitKey(0) & 0xFF + + # A, Enter, Espaço: aprova + if key in (ord("a"), ord("A"), 13, 32): + return "approved" + + # R, Delete, Backspace: rejeita + if key in (ord("r"), ord("R"), 8, 127): + return "rejected" + + # S: pula sem copiar para fixed. + if key in (ord("s"), ord("S")): + return "skipped" + + # Q ou ESC: para o loop e gera relatórios parciais. + if key in (ord("q"), ord("Q"), 27): + return "quit" + + +def build_fixed_dataset_from_manual_review( + sample_rows: List[Dict[str, Any]], + dataset_roots: List[Path], + args, +): + """ + Cria fixed/group usando decisões humanas coletadas no modo manual. + """ + fixed_root = resolve_fixed_out_root(dataset_roots, args) + ensure_dir(fixed_root) + + manifest_rows = [] + approved_rows = [] + rejected_rows = [] + skipped_rows = [] + + for row in sample_rows: + decision = str(row.get("manual_decision", "")) + + out_row = dict(row) + out_row["approved_for_training"] = decision == "approved" + out_row["reject_reasons"] = "manual_rejected" if decision == "rejected" else "" + out_row["manual_review"] = True + + manifest_rows.append(out_row) + + if decision == "approved": + approved_rows.append(out_row) + copy_sample_to_fixed(out_row, fixed_root, copy_previews=args.clean_copy_previews) + elif decision == "rejected": + rejected_rows.append(out_row) + else: + skipped_rows.append(out_row) + + summary = { + "schema": "multispec_fixed_dataset_manual_v1", + "fixed_root": str(fixed_root), + "samples_total": len(sample_rows), + "samples_approved": len(approved_rows), + "samples_rejected": len(rejected_rows), + "samples_skipped": len(skipped_rows), + "approval_pct": 100.0 * len(approved_rows) / max(1, len(sample_rows)), + "approved_by_group": {}, + "rejected_by_group": {}, + "skipped_by_group": {}, + "manual_review": { + "enabled": True, + "window_width": args.manual_window_width, + "groups_except": args.groups_except, + }, + "rejected_previews": { + "enabled": bool(args.save_rejected_previews), + "root": str(resolve_rejected_preview_root(fixed_root)), + "source": args.rejected_preview_source, + "max_width": args.rejected_preview_max_width, + }, + } + + for name, rows in ( + ("approved_by_group", approved_rows), + ("rejected_by_group", rejected_rows), + ("skipped_by_group", skipped_rows), + ): + for row in rows: + g = str(row.get("group", "unknown")) + summary[name][g] = summary[name].get(g, 0) + 1 + + write_csv(fixed_root / "fixed_manifest.csv", manifest_rows) + write_csv(fixed_root / "fixed_approved.csv", approved_rows) + write_csv(fixed_root / "fixed_rejected.csv", rejected_rows) + write_csv(fixed_root / "fixed_skipped.csv", skipped_rows) + write_json(fixed_root / "fixed_summary.json", summary) + + print("\n========== FIXED DATASET MANUAL ==========") + print(f"Saída fixed/group : {fixed_root}") + print(f"Aprovadas : {len(approved_rows)} / {len(sample_rows)}") + print(f"Rejeitadas : {len(rejected_rows)} / {len(sample_rows)}") + print(f"Puladas : {len(skipped_rows)} / {len(sample_rows)}") + print(f"Manifest : {fixed_root / 'fixed_manifest.csv'}") + print("==========================================\n") + + +def build_fixed_dataset_from_audit( + sample_rows: List[Dict[str, Any]], + dataset_roots: List[Path], + class_map: Dict[int, str], + args, +): + fixed_root = resolve_fixed_out_root(dataset_roots, args) + ensure_dir(fixed_root) + + baselines = build_group_shift_baselines(sample_rows) + + manifest_rows = [] + approved_rows = [] + rejected_rows = [] + + for row in sample_rows: + approved, reasons, extra = classify_sample_for_training(row, baselines, class_map, args) + + out_row = dict(row) + out_row.update(extra) + out_row["approved_for_training"] = approved + out_row["reject_reasons"] = " ; ".join(reasons) + + manifest_rows.append(out_row) + + if approved: + approved_rows.append(out_row) + copy_sample_to_fixed(row, fixed_root, copy_previews=args.clean_copy_previews) + else: + rejected_rows.append(out_row) + if args.save_rejected_previews: + save_rejected_preview( + row=out_row, + fixed_root=fixed_root, + args=args, + tensor=None, + mask=None, + class_map=class_map, + ) + + summary = { + "schema": "multispec_fixed_dataset_v1", + "fixed_root": str(fixed_root), + "samples_total": len(sample_rows), + "samples_approved": len(approved_rows), + "samples_rejected": len(rejected_rows), + "approval_pct": 100.0 * len(approved_rows) / max(1, len(sample_rows)), + "filters": { + "clean_max_abs_shift_px": args.clean_max_abs_shift_px, + "clean_max_dev_shift_px": args.clean_max_dev_shift_px, + "clean_min_edge_corr": args.clean_min_edge_corr, + "clean_min_target_pct": args.clean_min_target_pct, + "clean_reject_low_corr_only_if_shift_bad": args.clean_reject_low_corr_only_if_shift_bad, + }, + "group_shift_baselines": { + group: { + k: { + "median": float(v[0]), + "mad": float(v[1]), + } + for k, v in vals.items() + } + for group, vals in baselines.items() + }, + "approved_by_group": {}, + "rejected_by_group": {}, + "rejected_previews": { + "enabled": bool(args.save_rejected_previews), + "root": str(resolve_rejected_preview_root(fixed_root)), + "source": args.rejected_preview_source, + "max_width": args.rejected_preview_max_width, + }, + } + + for row in approved_rows: + g = str(row.get("group", "unknown")) + summary["approved_by_group"][g] = summary["approved_by_group"].get(g, 0) + 1 + + for row in rejected_rows: + g = str(row.get("group", "unknown")) + summary["rejected_by_group"][g] = summary["rejected_by_group"].get(g, 0) + 1 + + write_csv(fixed_root / "fixed_manifest.csv", manifest_rows) + write_csv(fixed_root / "fixed_approved.csv", approved_rows) + write_csv(fixed_root / "fixed_rejected.csv", rejected_rows) + write_json(fixed_root / "fixed_summary.json", summary) + + print("\n========== FIXED DATASET ==========") + print(f"Saída fixed/group : {fixed_root}") + print(f"Aprovadas : {len(approved_rows)} / {len(sample_rows)}") + print(f"Rejeitadas : {len(rejected_rows)} / {len(sample_rows)}") + print(f"Manifest : {fixed_root / 'fixed_manifest.csv'}") + print("===================================\n") + + +def audit_dataset(args): + np.random.seed(args.seed) + + dataset_roots = find_dataset_roots(Path(args.input_path)) + out_dir = ensure_dir(args.out_dir) + visuals_dir = ensure_dir(out_dir / "visuals") + + class_map = parse_class_map(args.class_map) + + # Monta uma lista única de amostras: (dataset_root, meta_path) + groups_except = parse_csv_set(getattr(args, "groups_except", "")) + if groups_except: + dataset_roots = [r for r in dataset_roots if r.name.lower() not in groups_except] + + entries: List[Tuple[Path, Path]] = [] + for root in dataset_roots: + for meta_path in list_meta_files(root): + entries.append((root, meta_path)) + + if args.manual_start_index and args.manual_start_index > 1: + entries = entries[int(args.manual_start_index) - 1:] + + if args.limit and args.limit > 0: + entries = entries[:args.limit] + + global_acc = RunningFeatureStats() + class_acc: Dict[int, RunningFeatureStats] = {cls: RunningFeatureStats() for cls in class_map.keys()} + + sample_rows = [] + by_sample_class_channel_rows = [] + warnings = [] + core_cache: Dict[Tuple[int, int, str, str], Any] = {} + + print(f"[INFO] dataset_roots={len(dataset_roots)}") + for root in dataset_roots: + print(f" - {root}") + print(f"[INFO] amostras={len(entries)}") + print(f"[INFO] classes={class_map}") + print(f"[INFO] out_dir={out_dir}") + if groups_except: + print(f"[INFO] groups_except={sorted(groups_except)}") + if args.manual_review: + print("[MANUAL] Teclas: A/ENTER/ESPAÇO aprova | R/DEL/BACKSPACE rejeita | S pula | Q/ESC finaliza parcial") + + manual_stop_requested = False + + for idx, (dataset_root, meta_path) in enumerate(entries, start=1): + # Prefixa com o nome do grupo para evitar colisão de nomes entre subdatasets. + group_name = dataset_root.name + stem = f"{group_name}__{meta_path.stem}" + try: + tensor, meta, payload_path = load_multispec_tensor(meta_path, dataset_root, core_cache) + h, w = tensor.shape[1], tensor.shape[2] + mask_path = resolve_mask_path(dataset_root, meta_path) + mask = load_mask(mask_path, (h, w), args.ignore_index) + + features = compute_feature_maps(tensor) + + # Estatísticas globais da amostra. + sample_feature_stats = {} + for fname, fmap in features.items(): + raw01 = fname in CHANNELS + sample_feature_stats[fname] = calc_stats(fmap.reshape(-1), raw01=raw01) + global_acc.add(fname, fmap.reshape(-1), max_samples=args.max_pixels_per_feature) + + # Estatísticas por classe da amostra. + class_pixel_counts = {} + if mask is not None: + valid_mask = mask != args.ignore_index + for cls_id, cls_name in class_map.items(): + cm = (mask == cls_id) & valid_mask + n = int(np.sum(cm)) + class_pixel_counts[cls_name] = n + if n < args.min_class_pixels: + continue + + for fname, fmap in features.items(): + vals = fmap[cm] + class_acc[cls_id].add(fname, vals, max_samples=args.max_pixels_per_feature) + s = calc_stats(vals, raw01=(fname in CHANNELS)) + by_sample_class_channel_rows.append({ + "sample": stem, + "class_id": cls_id, + "class_name": cls_name, + "feature": fname, + **s, + }) + else: + warnings.append({ + "sample": stem, + "type": "missing_mask", + "message": "Mascara nao encontrada; estatistica por classe ignorada.", + }) + + # Alinhamento por bordas. + r, g, b, re, nir = [tensor[i] for i in range(5)] + rgb_gray = (0.299 * r + 0.587 * g + 0.114 * b).astype(np.float32) + alignment = { + "RGBgray_vs_RE": edge_agreement(rgb_gray, re), + "RGBgray_vs_NIR": edge_agreement(rgb_gray, nir), + "RE_vs_NIR": edge_agreement(re, nir), + } + + # Heurísticas de alerta. + sample_warn = [] + for ch in CHANNELS: + st = sample_feature_stats[ch] + if st["sat_pct"] > args.warn_sat_pct: + sample_warn.append(f"{ch}: saturação alta {st['sat_pct']:.2f}%") + if st["dark_pct"] > args.warn_dark_pct: + sample_warn.append(f"{ch}: pixels escuros alto {st['dark_pct']:.2f}%") + if st["p95_p05"] < args.warn_low_dynamic: + sample_warn.append(f"{ch}: baixa dinâmica p95-p05={st['p95_p05']:.4f}") + + for pair_name, al in alignment.items(): + dx = abs(safe_float(al.get("phase_dx"))) + dy = abs(safe_float(al.get("phase_dy"))) + corr = safe_float(al.get("edge_corr")) + if dx > args.warn_shift_px or dy > args.warn_shift_px: + sample_warn.append(f"{pair_name}: possível shift residual dx={dx:.2f}, dy={dy:.2f}") + if corr < args.warn_edge_corr: + sample_warn.append(f"{pair_name}: baixa correlação de borda {corr:.3f}") + + if sample_warn: + warnings.append({ + "sample": stem, + "type": "sample_warning", + "messages": sample_warn, + }) + + row = { + "idx": idx, + "sample": stem, + "group": group_name, + "real_stem": meta_path.stem, + "dataset_root": str(dataset_root), + "meta_path": str(meta_path), + "payload_path": str(payload_path), + "mask_path": str(mask_path) if mask_path else "", + "mask_found": bool(mask_path is not None), + "mask_shape": str(mask.shape) if mask is not None else "", + "mask_unique": mask_unique_summary(mask), + "H": h, + "W": w, + "warnings_count": len(sample_warn), + "warning_text": " ; ".join(sample_warn), + "re_edge_corr": alignment["RGBgray_vs_RE"]["edge_corr"], + "nir_edge_corr": alignment["RGBgray_vs_NIR"]["edge_corr"], + "re_phase_dx": alignment["RGBgray_vs_RE"]["phase_dx"], + "re_phase_dy": alignment["RGBgray_vs_RE"]["phase_dy"], + "nir_phase_dx": alignment["RGBgray_vs_NIR"]["phase_dx"], + "nir_phase_dy": alignment["RGBgray_vs_NIR"]["phase_dy"], + } + + for fname in ALL_FEATURES: + st = sample_feature_stats[fname] + row[f"{fname}_mean"] = st["mean"] + row[f"{fname}_std"] = st["std"] + row[f"{fname}_p50"] = st["p50"] + row[f"{fname}_p95"] = st["p95"] + row[f"{fname}_p95_p05"] = st["p95_p05"] + if fname in CHANNELS: + row[f"{fname}_dark_pct"] = st["dark_pct"] + row[f"{fname}_sat_pct"] = st["sat_pct"] + + for cls_name, count in class_pixel_counts.items(): + row[f"pixels_{cls_name}"] = count + + sample_rows.append(row) + + sample_stats_for_visual = { + "alignment": alignment, + "features": sample_feature_stats, + } + + should_save_visual = args.save_visuals and ( + args.visual_every <= 1 or idx % args.visual_every == 0 or len(sample_warn) > 0 + ) + canvas = None + if should_save_visual or args.manual_review: + canvas = make_sample_visual(stem, tensor, mask, class_map, sample_stats_for_visual) + + if should_save_visual and canvas is not None: + cv2.imwrite(str(visuals_dir / f"{idx:05d}_{stem}.png"), canvas) + + if args.manual_review and canvas is not None: + decision = manual_review_decision(canvas, row, idx, len(entries), args) + if decision == "quit": + manual_stop_requested = True + row["manual_decision"] = "skipped" + row["manual_note"] = "manual_quit_here" + else: + row["manual_decision"] = decision + row["manual_note"] = "" + + if decision == "rejected" and args.save_rejected_previews: + fixed_root = resolve_fixed_out_root(dataset_roots, args) + rejected_row = dict(row) + rejected_row["reject_reasons"] = "manual_rejected" + save_rejected_preview( + row=rejected_row, + fixed_root=fixed_root, + args=args, + tensor=tensor, + mask=mask, + class_map=class_map, + ) + + print(f"[MANUAL] {idx}/{len(entries)} | {stem} | decisão={row.get('manual_decision')}") + if manual_stop_requested: + break + + if idx % args.print_every == 0 or idx == len(entries): + print(f"[OK] {idx}/{len(entries)} | {stem} | warnings={len(sample_warn)}") + + except Exception as e: + msg = str(e) + warnings.append({ + "sample": stem, + "type": "exception", + "message": msg, + }) + print(f"[ERRO] {idx}/{len(entries)} | {stem}: {msg}") + if args.stop_on_error: + raise + + if args.manual_review: + cv2.destroyAllWindows() + + # Resumos finais. + global_summary = global_acc.summarize() + class_summary = { + str(cls_id): { + "class_name": class_map[cls_id], + "features": acc.summarize(), + } + for cls_id, acc in class_acc.items() + } + + diagnosis = build_diagnosis(global_summary, class_summary, sample_rows, warnings, args) + + summary = { + "schema": "multispec_dataset_audit_v1", + "dataset_roots": [str(r) for r in dataset_roots], + "samples_processed": len(sample_rows), + "channels": CHANNELS, + "derived_features": DERIVED, + "class_map": {str(k): v for k, v in class_map.items()}, + "global_summary": global_summary, + "class_summary": class_summary, + "diagnosis": diagnosis, + } + + write_json(out_dir / "audit_summary.json", summary) + write_json(out_dir / "audit_warnings.json", warnings) + write_csv(out_dir / "audit_samples.csv", sample_rows) + write_class_channel_csv(out_dir / "audit_by_class_channel.csv", class_summary) + write_csv(out_dir / "audit_by_sample_class_channel.csv", by_sample_class_channel_rows) + + if args.build_fixed_dataset: + if args.manual_review: + build_fixed_dataset_from_manual_review( + sample_rows=sample_rows, + dataset_roots=dataset_roots, + args=args, + ) + else: + build_fixed_dataset_from_audit( + sample_rows=sample_rows, + dataset_roots=dataset_roots, + class_map=class_map, + args=args, + ) + + print("\n========== AUDITORIA FINALIZADA ==========") + print(f"Amostras processadas : {len(sample_rows)}") + print(f"Warnings : {len(warnings)}") + print(f"Resumo : {out_dir / 'audit_summary.json'}") + print(f"CSV amostras : {out_dir / 'audit_samples.csv'}") + print(f"CSV classes/canais : {out_dir / 'audit_by_class_channel.csv'}") + print(f"Visuais : {visuals_dir}") + print("==========================================\n") + + + + +def write_csv(path: Path, rows: List[Dict[str, Any]]): + if not rows: + with open(path, "w", encoding="utf-8", newline="") as f: + f.write("") + return + + keys = [] + for row in rows: + for k in row.keys(): + if k not in keys: + keys.append(k) + + with open(path, "w", encoding="utf-8", newline="") as f: + writer = csv.DictWriter(f, fieldnames=keys, extrasaction="ignore") + writer.writeheader() + writer.writerows(rows) + + +def write_class_channel_csv(path: Path, class_summary: Dict[str, Any]): + rows = [] + for cls_id, item in class_summary.items(): + cls_name = item["class_name"] + for feature, stats in item["features"].items(): + rows.append({ + "class_id": cls_id, + "class_name": cls_name, + "feature": feature, + **stats, + }) + write_csv(path, rows) + + +# ============================================================ +# Diagnóstico automático simples +# ============================================================ + + +def build_diagnosis( + global_summary: Dict[str, Dict[str, float]], + class_summary: Dict[str, Any], + sample_rows: List[Dict[str, Any]], + warnings: List[Dict[str, Any]], + args, +) -> Dict[str, Any]: + notes = [] + risks = [] + positives = [] + + # Sanidade global dos canais. + for ch in CHANNELS: + st = global_summary.get(ch, {}) + sat = safe_float(st.get("sat_pct")) + dark = safe_float(st.get("dark_pct")) + dyn = safe_float(st.get("p95_p05")) + mean = safe_float(st.get("mean")) + + if sat > args.warn_sat_pct: + risks.append(f"{ch}: saturação global alta ({sat:.2f}%).") + if dark > args.warn_dark_pct: + risks.append(f"{ch}: muitos pixels escuros globalmente ({dark:.2f}%).") + if dyn < args.warn_low_dynamic: + risks.append(f"{ch}: baixa dinâmica global p95-p05={dyn:.4f}.") + if 0.02 < mean < 0.98 and dyn >= args.warn_low_dynamic: + positives.append(f"{ch}: média/dinâmica globais parecem utilizáveis (mean={mean:.3f}, p95-p05={dyn:.3f}).") + + # Alinhamento médio. + if sample_rows: + re_corr = np.mean([safe_float(r.get("re_edge_corr")) for r in sample_rows]) + nir_corr = np.mean([safe_float(r.get("nir_edge_corr")) for r in sample_rows]) + re_shift = np.mean([math.hypot(safe_float(r.get("re_phase_dx")), safe_float(r.get("re_phase_dy"))) for r in sample_rows]) + nir_shift = np.mean([math.hypot(safe_float(r.get("nir_phase_dx")), safe_float(r.get("nir_phase_dy"))) for r in sample_rows]) + + notes.append(f"Correlação média de bordas RGB-RE={re_corr:.3f}, RGB-NIR={nir_corr:.3f}.") + notes.append(f"Shift médio estimado RGB-RE={re_shift:.2f}px, RGB-NIR={nir_shift:.2f}px.") + + if re_shift > args.warn_shift_px: + risks.append(f"RE: shift médio estimado alto ({re_shift:.2f}px). Verificar homografia/fusão.") + if nir_shift > args.warn_shift_px: + risks.append(f"NIR: shift médio estimado alto ({nir_shift:.2f}px). Verificar homografia/fusão.") + if re_corr < args.warn_edge_corr: + risks.append(f"RE: correlação média de borda baixa ({re_corr:.3f}). Pode indicar desalinhamento ou textura espectral muito diferente.") + if nir_corr < args.warn_edge_corr: + risks.append(f"NIR: correlação média de borda baixa ({nir_corr:.3f}). Pode indicar desalinhamento ou textura espectral muito diferente.") + + # Separabilidade simples por classes. + separability = {} + class_items = list(class_summary.items()) + for feature in ALL_FEATURES: + vals = [] + for cls_id, item in class_items: + st = item["features"].get(feature, {}) + vals.append((item["class_name"], safe_float(st.get("mean")), safe_float(st.get("std")))) + + sep_rows = [] + for i in range(len(vals)): + for j in range(i + 1, len(vals)): + a_name, a_mean, a_std = vals[i] + b_name, b_mean, b_std = vals[j] + pooled = math.sqrt((a_std * a_std + b_std * b_std) / 2.0) + EPS + d = abs(a_mean - b_mean) / pooled + sep_rows.append({ + "pair": f"{a_name}_vs_{b_name}", + "effect_size_d": d, + "mean_a": a_mean, + "mean_b": b_mean, + }) + separability[feature] = sep_rows + + # Destaca features com maior separação média. + ranking = [] + for feature, rows in separability.items(): + if rows: + avg_d = float(np.mean([r["effect_size_d"] for r in rows])) + ranking.append((feature, avg_d)) + ranking.sort(key=lambda x: x[1], reverse=True) + + notes.append("Ranking simples de separabilidade média por feature: " + ", ".join([f"{f}={d:.2f}" for f, d in ranking[:8]])) + + for f, d in ranking[:5]: + if d > 0.5: + positives.append(f"{f}: mostra separabilidade média interessante entre classes (d≈{d:.2f}).") + + if warnings: + risks.append(f"Foram gerados {len(warnings)} avisos/exceções. Ver audit_warnings.json.") + + return { + "positives": positives, + "risks": risks, + "notes": notes, + "feature_separability": separability, + "feature_separability_ranking": [{"feature": f, "avg_effect_size_d": d} for f, d in ranking], + } + + +# ============================================================ +# CLI +# ============================================================ + + +def main(): + parser = argparse.ArgumentParser( + description="Auditoria final do tensor multiespectral [R,G,B,RE,NIR] pronto para treinamento." + ) + parser.add_argument("--input_path", required=True, help="Caminho para dataset_root, metas, bins, masks ou arquivo dentro do dataset.") + parser.add_argument("--out_dir", default="audit_multispec_out", help="Pasta de saída da auditoria.") + parser.add_argument("--class-map", default="0:chao,1:cana,2:erva", help="Mapa de classes. Ex: 0:chao,1:cana,2:erva") + parser.add_argument("--ignore-index", type=int, default=255, help="Valor ignorado na máscara.") + parser.add_argument("--limit", type=int, default=0, help="Limita número de amostras para teste rápido. 0 = todas.") + parser.add_argument("--save-visuals", action="store_true", help="Salva painéis visuais por amostra.") + parser.add_argument("--visual-every", type=int, default=10, help="Salva visual a cada N amostras. Amostras com warning sempre são salvas.") + parser.add_argument("--print-every", type=int, default=10, help="Mostra progresso a cada N amostras.") + parser.add_argument("--min-class-pixels", type=int, default=50, help="Mínimo de pixels por classe para estatística por amostra.") + parser.add_argument("--max-pixels-per-feature", type=int, default=25000, help="Amostragem máxima de pixels por feature/amostra para acumuladores.") + parser.add_argument("--warn-sat-pct", type=float, default=1.0, help="Alerta se canal tiver saturação acima deste percentual.") + parser.add_argument("--warn-dark-pct", type=float, default=35.0, help="Alerta se canal tiver pixels <=0.01 acima deste percentual.") + parser.add_argument("--warn-low-dynamic", type=float, default=0.03, help="Alerta se p95-p05 do canal for menor que isso.") + parser.add_argument("--warn-shift-px", type=float, default=3.0, help="Alerta se shift estimado por phase correlation passar disso.") + parser.add_argument("--warn-edge-corr", type=float, default=0.08, help="Alerta se correlação de borda for menor que isso.") + parser.add_argument("--seed", type=int, default=42) + parser.add_argument("--stop-on-error", action="store_true") + + parser.add_argument("--build-fixed-dataset", action="store_true", + help="Cria uma cópia filtrada do dataset em fixed/group, mantendo apenas amostras aprovadas para treino.") + + parser.add_argument("--fixed-out-root", default="", + help="Raiz de saída do dataset filtrado. Se vazio, usa dataset_root/../../fixed/group.") + + parser.add_argument("--clean-max-abs-shift-px", type=float, default=22.0, + help="Reprova amostras com shift absoluto extremo em RE ou NIR.") + + parser.add_argument("--clean-max-dev-shift-px", type=float, default=7.0, + help="Reprova amostras cujo shift foge muito da mediana robusta do grupo.") + + parser.add_argument("--clean-min-edge-corr", type=float, default=0.045, + help="Correlação mínima de borda para aprovar. Valor baixo porque canais espectrais naturalmente diferem do RGB.") + + parser.add_argument("--clean-reject-low-corr-only-if-shift-bad", action="store_true", + help="Só reprova baixa correlação se também houver desvio geométrico alto.") + + parser.add_argument("--clean-min-target-pct", type=float, default=0.0025, + help="Percentual mínimo da classe alvo em grupos com cana/erva. 0.0025 = 0.25%%. Use 0 para desativar.") + + parser.add_argument("--clean-copy-previews", action="store_true", + help="Copia previews quando existirem.") + + parser.add_argument("--save-rejected-previews", action="store_true", + help="Salva PNGs leves das amostras rejeitadas em fixed/rejected_previews para inspeção visual.") + + parser.add_argument("--rejected-preview-source", default="auto", + choices=["auto", "preview", "audit_panel", "rgb_tensor"], + help="Fonte do preview dos rejeitados: preview original, painel audit, RGB reconstruído ou automático.") + + parser.add_argument("--rejected-preview-max-width", type=int, default=640, + help="Largura máxima dos previews rejeitados salvos.") + + parser.add_argument("--manual-review", action="store_true", + help="Ativa revisão manual interativa. Mostra cada painel visual e espera aprovar/rejeitar.") + + parser.add_argument("--manual-window-width", type=int, default=1500, + help="Largura máxima da janela de revisão manual. Use 0 para não redimensionar.") + + parser.add_argument("--manual-window-name", default="Audit Dataset - Revisao Manual", + help="Nome da janela OpenCV usada na revisão manual.") + + parser.add_argument("--manual-start-index", type=int, default=1, + help="Índice inicial da fila manual. Útil para retomar uma revisão interrompida.") + + parser.add_argument("--groups-except", default="", + help="Lista de grupos para ignorar, separados por vírgula. Ex: chao,chao_cana") + + args = parser.parse_args() + + audit_dataset(args) + + +if __name__ == "__main__": + main() diff --git a/Python/OAK/datasets/oak-fcc-3/utils/charuco_preview_probe.py b/Python/OAK/datasets/oak-fcc-3/utils/charuco_preview_probe.py new file mode 100644 index 000000000..487bfbb4b --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/charuco_preview_probe.py @@ -0,0 +1,420 @@ +import argparse +import time +from collections import deque +from typing import Optional +from pathlib import Path + +import cv2 +import numpy as np + +try: + import depthai as dai +except Exception as e: + raise RuntimeError( + "Nao consegui importar depthai. Ative o venv correto e instale depthai antes de rodar. " + f"Erro original: {e}" + ) + + +# ============================================================ +# OAK-FCC-3P Charuco Preview Probe +# ------------------------------------------------------------ +# Objetivo: +# Abrir CAM_A/CAM_B/CAM_C ao vivo para verificar se o Charuco no monitor +# ou impresso aparece bem nas tres cameras, principalmente nas mono RE/NIR. +# +# Exemplo: +# python -m utils.charuco_preview_probe --fps 10 +# +# Teclas: +# Q/ESC = sair +# S = salvar snapshot +# E = alterna detector de bordas +# C = alterna contraste auto/normal nas mono +# ============================================================ + + +# ============================================================ +# DepthAI helpers +# ============================================================ + + +def socket_from_name(name: str): + name = str(name).strip().upper() + aliases = { + "A": "CAM_A", + "B": "CAM_B", + "C": "CAM_C", + "RGB": "CAM_A", + "RE": "CAM_B", + "NIR": "CAM_C", + } + name = aliases.get(name, name) + + if hasattr(dai.CameraBoardSocket, name): + return getattr(dai.CameraBoardSocket, name) + + legacy = { + "CAM_A": getattr(dai.CameraBoardSocket, "RGB", None), + "CAM_B": getattr(dai.CameraBoardSocket, "LEFT", None), + "CAM_C": getattr(dai.CameraBoardSocket, "RIGHT", None), + } + if legacy.get(name) is not None: + return legacy[name] + + raise ValueError(f"Socket invalido: {name}. Use CAM_A, CAM_B ou CAM_C.") + + + +def mono_resolution_from_name(name: str): + name = str(name).strip().lower() + r = dai.MonoCameraProperties.SensorResolution + table = { + "400p": getattr(r, "THE_400_P", None), + "480p": getattr(r, "THE_480_P", None), + "720p": getattr(r, "THE_720_P", None), + "800p": getattr(r, "THE_800_P", None), + } + if name not in table or table[name] is None: + valid = ", ".join(k for k, v in table.items() if v is not None) + raise ValueError(f"Resolucao mono invalida: {name}. Valid={valid}") + return table[name] + + + +def create_output_queue(output, name: str, max_size: int = 4, blocking: bool = False): + fn = getattr(output, "createOutputQueue", None) + if callable(fn): + return fn(maxSize=max_size, blocking=blocking) + raise RuntimeError(f"A saida '{name}' nao possui createOutputQueue().") + + + +def get_frame(q) -> Optional[np.ndarray]: + if q is None: + return None + try: + msg = q.tryGet() + except Exception: + return None + if msg is None: + return None + + try: + return msg.getCvFrame() + except Exception: + pass + try: + return msg.getFrame() + except Exception: + return None + + +# ============================================================ +# Visual helpers +# ============================================================ + + +def normalize_u8(img: np.ndarray, auto: bool = True) -> np.ndarray: + if img is None: + return np.zeros((300, 400), dtype=np.uint8) + + arr = np.asarray(img) + if arr.ndim == 3: + return arr.astype(np.uint8) + + arr = arr.astype(np.float32) + if not auto: + if arr.max() <= 1.5: + return np.clip(arr * 255.0, 0, 255).astype(np.uint8) + return np.clip(arr, 0, 255).astype(np.uint8) + + finite = np.isfinite(arr) + if not np.any(finite): + return np.zeros(arr.shape[:2], dtype=np.uint8) + vals = arr[finite] + lo = float(np.percentile(vals, 1)) + hi = float(np.percentile(vals, 99)) + if hi <= lo + 1e-6: + hi = lo + 1.0 + out = np.clip((arr - lo) / (hi - lo), 0, 1) + return (out * 255).astype(np.uint8) + + + +def edge_view(gray_u8: np.ndarray) -> np.ndarray: + if gray_u8.ndim == 3: + gray_u8 = cv2.cvtColor(gray_u8, cv2.COLOR_BGR2GRAY) + edges = cv2.Canny(gray_u8, 60, 140) + return edges + + + +def put_label(img: np.ndarray, title: str, subtitle: str = "") -> np.ndarray: + if img.ndim == 2: + img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR) + out = img.copy() + hbox = 58 if subtitle else 36 + cv2.rectangle(out, (0, 0), (out.shape[1], hbox), (0, 0, 0), -1) + cv2.putText(out, str(title)[:80], (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.65, (0, 255, 255), 2, cv2.LINE_AA) + if subtitle: + cv2.putText(out, str(subtitle)[:115], (10, 48), cv2.FONT_HERSHEY_SIMPLEX, 0.43, (255, 255, 255), 1, cv2.LINE_AA) + return out + + + +def resize_keep(img: np.ndarray, width: int) -> np.ndarray: + scale = width / img.shape[1] + height = max(1, int(img.shape[0] * scale)) + return cv2.resize(img, (width, height), interpolation=cv2.INTER_AREA) + + + +def make_grid(panels, panel_w: int = 520, cols: int = 3) -> np.ndarray: + rendered = [] + for title, img, subtitle in panels: + if img.ndim == 2: + img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR) + small = resize_keep(img, panel_w) + rendered.append(put_label(small, title, subtitle)) + + max_h = max(x.shape[0] for x in rendered) + padded = [] + for im in rendered: + if im.shape[0] < max_h: + pad = np.zeros((max_h - im.shape[0], im.shape[1], 3), dtype=np.uint8) + im = np.vstack([im, pad]) + padded.append(im) + + gap = 10 + gap_w = np.full((max_h, gap, 3), 25, dtype=np.uint8) + rows = [] + for i in range(0, len(padded), cols): + items = padded[i:i + cols] + while len(items) < cols: + items.append(np.zeros_like(padded[0])) + row = items[0] + for j in range(1, cols): + row = np.hstack([row, gap_w, items[j]]) + rows.append(row) + + gap_h = np.full((gap, rows[0].shape[1], 3), 25, dtype=np.uint8) + canvas = rows[0] + for row in rows[1:]: + canvas = np.vstack([canvas, gap_h, row]) + return canvas + + + +def stats_line(img: np.ndarray) -> str: + if img is None: + return "sem frame" + arr = np.asarray(img, dtype=np.float32) + if arr.ndim == 3: + gray = cv2.cvtColor(arr.astype(np.uint8), cv2.COLOR_BGR2GRAY).astype(np.float32) + else: + gray = arr + return f"mean={gray.mean():.1f} p05={np.percentile(gray,5):.1f} p95={np.percentile(gray,95):.1f}" + + +# ============================================================ +# Pipeline +# ============================================================ + + +def create_pipeline_and_outputs(args): + pipeline = dai.Pipeline() + + # CAM_A color + rgb = pipeline.create(dai.node.ColorCamera) + rgb.setBoardSocket(socket_from_name(args.rgb)) + rgb.setResolution(dai.ColorCameraProperties.SensorResolution.THE_800_P) + rgb.setFps(float(args.fps)) + rgb.setInterleaved(False) + rgb.setColorOrder(dai.ColorCameraProperties.ColorOrder.BGR) + rgb.setPreviewSize(int(args.preview_w), int(args.preview_h)) + + # CAM_B/C mono + mono_b = pipeline.create(dai.node.MonoCamera) + mono_c = pipeline.create(dai.node.MonoCamera) + mono_b.setBoardSocket(socket_from_name(args.cam_b)) + mono_c.setBoardSocket(socket_from_name(args.cam_c)) + mono_b.setResolution(mono_resolution_from_name(args.mono_resolution)) + mono_c.setResolution(mono_resolution_from_name(args.mono_resolution)) + mono_b.setFps(float(args.fps)) + mono_c.setFps(float(args.fps)) + + outputs = { + "rgb": rgb.preview, + "cam_b": mono_b.out, + "cam_c": mono_c.out, + } + return pipeline, outputs + + +# ============================================================ +# Main +# ============================================================ + + +def start_pipeline(pipeline): + fn = getattr(pipeline, "start", None) + if not callable(fn): + raise RuntimeError("pipeline.start() nao existe nesta versao do DepthAI.") + fn() + + + +def stop_pipeline(pipeline): + try: + fn = getattr(pipeline, "stop", None) + if callable(fn): + fn() + except Exception: + pass + + + +def save_snapshot(out_dir: str, rgb, cam_b, cam_c, canvas): + folder = Path(out_dir) + folder.mkdir(parents=True, exist_ok=True) + ts = time.strftime("%Y%m%d_%H%M%S") + if rgb is not None: + cv2.imwrite(str(folder / f"{ts}_CAM_A_rgb.png"), rgb) + if cam_b is not None: + cv2.imwrite(str(folder / f"{ts}_CAM_B_mono.png"), normalize_u8(cam_b, auto=True)) + if cam_c is not None: + cv2.imwrite(str(folder / f"{ts}_CAM_C_mono.png"), normalize_u8(cam_c, auto=True)) + if canvas is not None: + cv2.imwrite(str(folder / f"{ts}_canvas.png"), canvas) + print(f"[OK] snapshot salvo em {folder}") + + + +def main(args): + pipeline, outputs = create_pipeline_and_outputs(args) + + queues = { + name: create_output_queue(output, name, max_size=4, blocking=False) + for name, output in outputs.items() + } + + print("[INFO] Pipeline preview criado sem StereoDepth.") + print("[INFO] Abra o PDF Charuco em tela cheia no monitor e aponte a camera para ele.") + print("[INFO] O objetivo e ver se CAM_B e CAM_C enxergam marcadores/cantos com contraste.") + + start_pipeline(pipeline) + + cv2.namedWindow("OAK-FCC-3P Charuco Preview Probe", cv2.WINDOW_NORMAL) + cv2.resizeWindow("OAK-FCC-3P Charuco Preview Probe", 1600, 900) + + show_edges = False + auto_contrast = True + frame_times = deque(maxlen=40) + last_canvas = None + + rgb_frame = None + b_frame = None + c_frame = None + + try: + while True: + updated = False + for name, q in queues.items(): + frame = get_frame(q) + if frame is None: + continue + updated = True + if name == "rgb": + rgb_frame = frame + elif name == "cam_b": + b_frame = frame + elif name == "cam_c": + c_frame = frame + + if updated: + frame_times.append(time.time()) + + if len(frame_times) >= 2: + fps = (len(frame_times) - 1) / max(1e-6, frame_times[-1] - frame_times[0]) + else: + fps = 0.0 + + if rgb_frame is None or b_frame is None or c_frame is None: + key = cv2.waitKey(1) & 0xFF + if key in (27, ord("q"), ord("Q")): + break + continue + + rgb_vis = rgb_frame.copy() + b_vis = normalize_u8(b_frame, auto=auto_contrast) + c_vis = normalize_u8(c_frame, auto=auto_contrast) + + if show_edges: + rgb_gray = cv2.cvtColor(rgb_vis, cv2.COLOR_BGR2GRAY) + rgb_panel = edge_view(rgb_gray) + b_panel = edge_view(b_vis) + c_panel = edge_view(c_vis) + mode = "edges" + else: + rgb_panel = rgb_vis + b_panel = b_vis + c_panel = c_vis + mode = "preview" + + panels = [ + ("CAM_A RGB", rgb_panel, f"{stats_line(rgb_frame)} | fps={fps:.1f}"), + ("CAM_B mono / RE", b_panel, stats_line(b_frame)), + ("CAM_C mono / NIR", c_panel, stats_line(c_frame)), + ("CAM_B edges" if not show_edges else "CAM_B preview", edge_view(b_vis) if not show_edges else b_vis, "bordas para ver marcador"), + ("CAM_C edges" if not show_edges else "CAM_C preview", edge_view(c_vis) if not show_edges else c_vis, "bordas para ver marcador"), + ("Info", np.zeros((300, 600, 3), dtype=np.uint8), f"mode={mode} auto_contrast={auto_contrast} | E edges | C contraste | S save | Q sair"), + ] + + canvas = make_grid(panels, panel_w=args.panel_w, cols=3) + + # Escreve texto grande no painel Info vazio, ultimo quadrante. + info_y0 = canvas.shape[0] - resize_keep(np.zeros((300, 600, 3), dtype=np.uint8), args.panel_w).shape[0] + cv2.putText(canvas, "Charuco visibility test", (2 * (args.panel_w + 10) + 15, info_y0 + 95), cv2.FONT_HERSHEY_SIMPLEX, 0.75, (0, 255, 255), 2, cv2.LINE_AA) + cv2.putText(canvas, "Olhe CAM_B/C: marcadores precisam aparecer nitidos", (2 * (args.panel_w + 10) + 15, info_y0 + 135), cv2.FONT_HERSHEY_SIMPLEX, 0.52, (255, 255, 255), 1, cv2.LINE_AA) + cv2.putText(canvas, "E=edges C=auto contrast S=snapshot Q=sair", (2 * (args.panel_w + 10) + 15, info_y0 + 170), cv2.FONT_HERSHEY_SIMPLEX, 0.52, (255, 255, 255), 1, cv2.LINE_AA) + + last_canvas = canvas + cv2.imshow("OAK-FCC-3P Charuco Preview Probe", canvas) + + key = cv2.waitKey(1) & 0xFF + if key in (27, ord("q"), ord("Q")): + break + elif key in (ord("e"), ord("E")): + show_edges = not show_edges + elif key in (ord("c"), ord("C")): + auto_contrast = not auto_contrast + elif key in (ord("s"), ord("S")): + save_snapshot(args.out_dir, rgb_frame, b_frame, c_frame, last_canvas) + + finally: + stop_pipeline(pipeline) + cv2.destroyAllWindows() + + +# ============================================================ +# CLI +# ============================================================ + + +def build_argparser(): + ap = argparse.ArgumentParser(description="Preview rapido CAM_A/CAM_B/CAM_C para testar visibilidade do Charuco.") + ap.add_argument("--rgb", type=str, default="CAM_A") + ap.add_argument("--cam-b", type=str, default="CAM_B") + ap.add_argument("--cam-c", type=str, default="CAM_C") + ap.add_argument("--mono-resolution", type=str, default="800p", choices=["400p", "480p", "720p", "800p"]) + ap.add_argument("--fps", type=float, default=10.0) + ap.add_argument("--preview-w", type=int, default=640) + ap.add_argument("--preview-h", type=int, default=400) + ap.add_argument("--panel-w", type=int, default=500) + ap.add_argument("--out-dir", type=str, default="charuco_preview_out") + return ap + + +if __name__ == "__main__": + main(build_argparser().parse_args()) diff --git a/Python/OAK/datasets/oak-fcc-3/utils/dataset_alignment_browser.py b/Python/OAK/datasets/oak-fcc-3/utils/dataset_alignment_browser.py new file mode 100644 index 000000000..533e0f052 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/dataset_alignment_browser.py @@ -0,0 +1,1199 @@ +import argparse +import json +import math +import unicodedata +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +import cv2 +import numpy as np + +# ============================================================ +# Dataset Alignment Browser V2 +# ------------------------------------------------------------ +# Objetivo: +# Navegar no dataset multiespectral e comparar: +# 1) space=final -> tensor final usado no treino/inferencia +# 2) space=native -> dados decodificados nativos, antes da homografia/crop/fusao +# +# Tambem testa alinhamento dinamico por bordas com prioridades: +# - global : usa a imagem toda +# - largest_blob : usa o maior blob de bordas fortes +# - gt_target : usa mascara GT de cana/erva, se existir +# +# Requisitos: +# - Colocar este arquivo no mesmo projeto onde existe utils/audit_dataset_manual.py +# - Rodar a partir da raiz do projeto, por exemplo: +# python -m utils.dataset_alignment_browser_v2 --input_path .\dataset\original\group\ --groups-except chao --space native +# ============================================================ + +_AUDIT_IMPORT_ERROR = None +try: + from utils.audit_dataset_manual import ( + find_dataset_roots, + list_meta_files, + load_multispec_tensor, + resolve_mask_path, + load_mask, + parse_csv_set, + load_json, + resolve_camera_payloads, + resolve_module_params_path, + get_raw_processor_core, + ) +except Exception as e1: + try: + from audit_dataset_manual import ( + find_dataset_roots, + list_meta_files, + load_multispec_tensor, + resolve_mask_path, + load_mask, + parse_csv_set, + load_json, + resolve_camera_payloads, + resolve_module_params_path, + get_raw_processor_core, + ) + except Exception as e2: + _AUDIT_IMPORT_ERROR = (e1, e2) + find_dataset_roots = None + list_meta_files = None + load_multispec_tensor = None + resolve_mask_path = None + load_mask = None + parse_csv_set = None + load_json = None + resolve_camera_payloads = None + resolve_module_params_path = None + get_raw_processor_core = None + +EPS = 1e-6 +IGNORE_INDEX = 255 + + +# ============================================================ +# Utilidades gerais +# ============================================================ + + +def ensure_imports_ok(): + if find_dataset_roots is None: + msg = ( + "Nao consegui importar funcoes do audit_dataset_manual.py.\n" + "Coloque este script no mesmo projeto do auditor e rode a partir da raiz do projeto.\n" + ) + if _AUDIT_IMPORT_ERROR: + msg += f"\nImport error 1: {_AUDIT_IMPORT_ERROR[0]}\nImport error 2: {_AUDIT_IMPORT_ERROR[1]}" + raise RuntimeError(msg) + + +def cv_text(text: Any) -> str: + s = str(text) + s = unicodedata.normalize("NFKD", s) + s = s.encode("ascii", "ignore").decode("ascii") + return s + + +def normalize_to_u8(x: np.ndarray, p_low: float = 1.0, p_high: float = 99.0) -> np.ndarray: + arr = x.astype(np.float32, copy=False) + finite = np.isfinite(arr) + if not np.any(finite): + return np.zeros(arr.shape[:2], dtype=np.uint8) + + vals = arr[finite] + lo = float(np.percentile(vals, p_low)) + hi = float(np.percentile(vals, p_high)) + if hi <= lo + EPS: + hi = lo + 1.0 + + y = (arr - lo) / (hi - lo) + y = np.clip(y, 0.0, 1.0) + return (y * 255.0).astype(np.uint8) + + +def float01_to_u8(x: np.ndarray) -> np.ndarray: + return np.clip(x.astype(np.float32) * 255.0, 0, 255).astype(np.uint8) + + +def rgb_hwc_to_bgr(rgb: np.ndarray, stretch: bool = False) -> np.ndarray: + rgb = np.asarray(rgb, dtype=np.float32) + if stretch: + chans = [normalize_to_u8(rgb[:, :, i]) for i in range(3)] + rgb_u8 = np.dstack(chans) + else: + rgb_u8 = float01_to_u8(rgb) + return cv2.cvtColor(rgb_u8, cv2.COLOR_RGB2BGR) + + +def rgb_from_tensor(tensor: np.ndarray, stretch: bool = False) -> np.ndarray: + rgb = np.transpose(tensor[:3], (1, 2, 0)).astype(np.float32) + return rgb_hwc_to_bgr(rgb, stretch=stretch) + + +def gray_from_rgb_hwc(rgb: np.ndarray) -> np.ndarray: + rgb = np.asarray(rgb, dtype=np.float32) + return (0.299 * rgb[:, :, 0] + 0.587 * rgb[:, :, 1] + 0.114 * rgb[:, :, 2]).astype(np.float32) + + +def gray_from_tensor_rgb(tensor: np.ndarray) -> np.ndarray: + r, g, b = [tensor[i].astype(np.float32, copy=False) for i in range(3)] + return (0.299 * r + 0.587 * g + 0.114 * b).astype(np.float32) + + +def resize_to(img: np.ndarray, hw: Tuple[int, int], interp: int = cv2.INTER_LINEAR) -> np.ndarray: + h, w = int(hw[0]), int(hw[1]) + if img.shape[:2] == (h, w): + return img + return cv2.resize(img, (w, h), interpolation=interp) + + +def gradient_mag(x: np.ndarray) -> np.ndarray: + u8 = normalize_to_u8(x) + gx = cv2.Sobel(u8, cv2.CV_32F, 1, 0, ksize=3) + gy = cv2.Sobel(u8, cv2.CV_32F, 0, 1, ksize=3) + return cv2.magnitude(gx, gy).astype(np.float32) + + +def edge_binary(x: np.ndarray, low: int = 60, high: int = 140) -> np.ndarray: + return cv2.Canny(normalize_to_u8(x), low, high) + + +def colorize_gray(x: np.ndarray, cmap: int = cv2.COLORMAP_VIRIDIS) -> np.ndarray: + return cv2.applyColorMap(normalize_to_u8(x), cmap) + + +def put_label(img: np.ndarray, title: str, subtitle: str = "") -> np.ndarray: + out = img.copy() + title = cv_text(title) + subtitle = cv_text(subtitle) + hbox = 58 if subtitle else 36 + cv2.rectangle(out, (0, 0), (out.shape[1], hbox), (0, 0, 0), -1) + cv2.putText(out, title, (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.62, (0, 255, 255), 2, cv2.LINE_AA) + if subtitle: + cv2.putText(out, subtitle[:165], (10, 48), cv2.FONT_HERSHEY_SIMPLEX, 0.43, (255, 255, 255), 1, cv2.LINE_AA) + return out + + +def resize_keep(img: np.ndarray, target_w: int) -> np.ndarray: + scale = float(target_w) / float(img.shape[1]) + target_h = max(1, int(img.shape[0] * scale)) + return cv2.resize(img, (target_w, target_h), interpolation=cv2.INTER_AREA) + + +def make_grid(panels: List[Tuple[str, np.ndarray, str]], panel_w: int = 410, cols: int = 3) -> np.ndarray: + rendered: List[np.ndarray] = [] + for title, img, subtitle in panels: + if img.ndim == 2: + img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR) + small = resize_keep(img, panel_w) + rendered.append(put_label(small, title, subtitle)) + + if not rendered: + return np.zeros((300, 600, 3), dtype=np.uint8) + + max_h = max(x.shape[0] for x in rendered) + padded: List[np.ndarray] = [] + for im in rendered: + if im.shape[0] < max_h: + pad = np.zeros((max_h - im.shape[0], im.shape[1], 3), dtype=np.uint8) + im = np.vstack([im, pad]) + padded.append(im) + + gap = 10 + gap_w = np.full((max_h, gap, 3), 24, dtype=np.uint8) + rows: List[np.ndarray] = [] + filler = np.zeros_like(padded[0]) + + for i in range(0, len(padded), cols): + items = padded[i:i + cols] + while len(items) < cols: + items.append(filler.copy()) + row = items[0] + for j in range(1, cols): + row = np.hstack([row, gap_w, items[j]]) + rows.append(row) + + gap_h = np.full((gap, rows[0].shape[1], 3), 24, dtype=np.uint8) + canvas = rows[0] + for r in rows[1:]: + canvas = np.vstack([canvas, gap_h, r]) + return canvas + + +def make_info_panel(lines: List[str], size: Tuple[int, int] = (900, 280)) -> np.ndarray: + w, h = size + img = np.zeros((h, w, 3), dtype=np.uint8) + cv2.rectangle(img, (0, 0), (w - 1, h - 1), (70, 70, 70), 1) + y = 28 + for i, line in enumerate(lines): + color = (0, 255, 255) if i == 0 else (235, 235, 235) + cv2.putText(img, cv_text(line[:145]), (12, y), cv2.FONT_HERSHEY_SIMPLEX, 0.52, color, 1, cv2.LINE_AA) + y += 23 + if y > h - 12: + break + return img + + +def falsecolor_overlay(a: np.ndarray, b: np.ndarray) -> np.ndarray: + """ + Verde=A, Magenta=B. + Onde casa, tende a ficar claro/cinza/branco. Onde desalinha, aparecem franjas. + """ + au8 = normalize_to_u8(a) + bu8 = normalize_to_u8(b) + out = np.zeros((au8.shape[0], au8.shape[1], 3), dtype=np.uint8) + out[:, :, 1] = au8 + out[:, :, 0] = bu8 + out[:, :, 2] = bu8 + return out + + +def alpha_blend(base_bgr: np.ndarray, overlay_gray: np.ndarray, alpha: float = 0.38, + cmap: int = cv2.COLORMAP_TURBO) -> np.ndarray: + cm = colorize_gray(overlay_gray, cmap) + cm = resize_to(cm, base_bgr.shape[:2]) + return cv2.addWeighted(base_bgr, 1.0 - alpha, cm, alpha, 0.0) + + +def edge_overlay(rgb_gray: np.ndarray, re: np.ndarray, nir: np.ndarray) -> np.ndarray: + e_rgb = normalize_to_u8(gradient_mag(rgb_gray), 5, 99) + e_re = normalize_to_u8(gradient_mag(re), 5, 99) + e_nir = normalize_to_u8(gradient_mag(nir), 5, 99) + out = np.zeros((e_rgb.shape[0], e_rgb.shape[1], 3), dtype=np.uint8) + out[:, :, 1] = e_rgb + out[:, :, 2] = e_re + out[:, :, 0] = e_nir + return out + + +def draw_edges_on_rgb(rgb_bgr: np.ndarray, img: np.ndarray, color: Tuple[int, int, int]) -> np.ndarray: + out = rgb_bgr.copy() + img = resize_to(img, rgb_bgr.shape[:2]) + ed = edge_binary(img) + out[ed > 0] = color + return out + + +def mask_overlay(rgb_bgr: np.ndarray, mask: Optional[np.ndarray]) -> np.ndarray: + if mask is None: + return rgb_bgr.copy() + + mask = resize_to(mask.astype(np.int32), rgb_bgr.shape[:2], interp=cv2.INTER_NEAREST) + out = rgb_bgr.copy() + color_mask = np.zeros_like(out) + palette = { + 0: (0, 0, 128), + 1: (128, 0, 0), + 2: (0, 128, 0), + 255: (0, 0, 0), + } + for cls_id in np.unique(mask): + color_mask[mask == int(cls_id)] = palette.get(int(cls_id), (100, 100, 100)) + return cv2.addWeighted(out, 0.70, color_mask, 0.30, 0) + + +def draw_roi(img: np.ndarray, roi_mask: Optional[np.ndarray], bbox: Optional[Tuple[int, int, int, int]], title: str = "") -> np.ndarray: + out = img.copy() + if roi_mask is not None: + mask = resize_to(roi_mask.astype(np.uint8), out.shape[:2], interp=cv2.INTER_NEAREST) + tint = np.zeros_like(out) + tint[:, :, 1] = 255 + out = np.where(mask[:, :, None] > 0, cv2.addWeighted(out, 0.55, tint, 0.45, 0), out) + if bbox is not None: + x0, y0, x1, y1 = bbox + cv2.rectangle(out, (x0, y0), (x1, y1), (0, 255, 255), 2) + if title: + cv2.putText(out, cv_text(title), (10, out.shape[0] - 12), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (0, 255, 255), 1, cv2.LINE_AA) + return out + + +def draw_crop_box(img: np.ndarray, crop_box: Optional[Tuple[int, int, int, int]], label: str = "crop") -> np.ndarray: + out = img.copy() + if crop_box is None: + return out + x0, y0, x1, y1 = [int(v) for v in crop_box] + cv2.rectangle(out, (x0, y0), (x1, y1), (0, 255, 255), 2) + cv2.putText(out, cv_text(label), (x0 + 6, max(22, y0 + 22)), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (0, 255, 255), 1, cv2.LINE_AA) + return out + + +# ============================================================ +# Registro por bordas / ROI +# ============================================================ + + +@dataclass +class AlignResult: + method: str + priority: str + accepted: bool + warp: np.ndarray + used_inverse_map: bool + roi_mask: Optional[np.ndarray] + roi_bbox: Optional[Tuple[int, int, int, int]] + phase_dx: float = 0.0 + phase_dy: float = 0.0 + phase_response: float = 0.0 + edge_corr_before: float = 0.0 + edge_corr_after: float = 0.0 + roi_corr_before: float = 0.0 + roi_corr_after: float = 0.0 + translation_px: float = 0.0 + rotation_deg: float = 0.0 + note: str = "" + + + +def edge_corr(a: np.ndarray, b: np.ndarray, roi_mask: Optional[np.ndarray] = None) -> float: + ga = gradient_mag(a) + gb = gradient_mag(b) + + if roi_mask is not None: + m = resize_to(roi_mask.astype(np.uint8), ga.shape[:2], interp=cv2.INTER_NEAREST) > 0 + if np.count_nonzero(m) < 32: + return 0.0 + va = ga[m].reshape(-1) + vb = gb[m].reshape(-1) + else: + va = ga.reshape(-1) + vb = gb.reshape(-1) + + if va.size == 0 or vb.size == 0 or np.std(va) < EPS or np.std(vb) < EPS: + return 0.0 + return float(np.corrcoef(va, vb)[0, 1]) + + +def bbox_from_mask(mask: np.ndarray, pad: int = 8) -> Optional[Tuple[int, int, int, int]]: + ys, xs = np.where(mask > 0) + if xs.size == 0 or ys.size == 0: + return None + h, w = mask.shape[:2] + x0 = max(0, int(xs.min()) - pad) + y0 = max(0, int(ys.min()) - pad) + x1 = min(w, int(xs.max()) + 1 + pad) + y1 = min(h, int(ys.max()) + 1 + pad) + if x1 <= x0 or y1 <= y0: + return None + return x0, y0, x1, y1 + + +def largest_edge_blob_mask(ref: np.ndarray, tgt: np.ndarray, min_area_frac: float = 0.003, + dilate_iter: int = 5) -> Tuple[Optional[np.ndarray], Optional[Tuple[int, int, int, int]], str]: + ref_g = gradient_mag(ref) + tgt_g = gradient_mag(tgt) + combo = np.maximum(normalize_to_u8(ref_g, 70, 99.5), normalize_to_u8(tgt_g, 70, 99.5)) + + # threshold robusto por percentil: pega bordas fortes, nao toda a palhada fina. + th = max(25, int(np.percentile(combo, 88))) + strong = (combo >= th).astype(np.uint8) * 255 + + k = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5)) + strong = cv2.morphologyEx(strong, cv2.MORPH_CLOSE, k, iterations=2) + strong = cv2.dilate(strong, k, iterations=max(1, int(dilate_iter))) + + n, labels, stats, _cent = cv2.connectedComponentsWithStats(strong, connectivity=8) + if n <= 1: + return None, None, "sem componentes" + + h, w = strong.shape[:2] + min_area = int(float(min_area_frac) * h * w) + + best_id = None + best_score = -1.0 + for cid in range(1, n): + x, y, bw, bh, area = stats[cid] + if area < min_area: + continue + # Favorece area, mas tambem energia de borda dentro do blob. + m = labels == cid + energy = float(np.mean(combo[m])) if np.any(m) else 0.0 + score = float(area) * (1.0 + energy / 255.0) + if score > best_score: + best_score = score + best_id = cid + + if best_id is None: + return None, None, f"sem blob >= {min_area}px" + + mask = (labels == best_id).astype(np.uint8) + bbox = bbox_from_mask(mask, pad=10) + area = int(stats[best_id, cv2.CC_STAT_AREA]) + return mask, bbox, f"largest_blob id={best_id} area={area} score={best_score:.1f}" + + +def gt_target_mask(mask: Optional[np.ndarray], target_classes: List[int], hw: Tuple[int, int]) -> Tuple[Optional[np.ndarray], Optional[Tuple[int, int, int, int]], str]: + if mask is None: + return None, None, "sem GT mask" + m = resize_to(mask.astype(np.int32), hw, interp=cv2.INTER_NEAREST) + out = np.zeros(hw, dtype=np.uint8) + for cls_id in target_classes: + out[m == int(cls_id)] = 1 + if np.count_nonzero(out) < 32: + return None, None, f"GT target vazio classes={target_classes}" + k = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (7, 7)) + out = cv2.dilate(out, k, iterations=3) + return out, bbox_from_mask(out, pad=10), f"GT target classes={target_classes} area={int(np.count_nonzero(out))}" + + +def make_alignment_roi(ref: np.ndarray, tgt: np.ndarray, priority: str, + mask: Optional[np.ndarray], target_classes: List[int]) -> Tuple[Optional[np.ndarray], Optional[Tuple[int, int, int, int]], str]: + priority = str(priority).lower() + hw = ref.shape[:2] + if priority == "global": + return None, None, "global/full frame" + if priority == "largest_blob": + return largest_edge_blob_mask(ref, tgt) + if priority == "gt_target": + return gt_target_mask(mask, target_classes, hw) + return None, None, f"priority desconhecida: {priority}" + + +def crop_by_bbox(a: np.ndarray, bbox: Optional[Tuple[int, int, int, int]]) -> np.ndarray: + if bbox is None: + return a + x0, y0, x1, y1 = bbox + return a[y0:y1, x0:x1] + + +def estimate_phase_shift( + ref: np.ndarray, + tgt: np.ndarray, + bbox: Optional[Tuple[int, int, int, int]] = None, + roi_mask: Optional[np.ndarray] = None, +) -> Tuple[float, float, float]: + rr = crop_by_bbox(gradient_mag(ref), bbox) + tt = crop_by_bbox(gradient_mag(tgt), bbox) + + if roi_mask is not None: + m = resize_to(roi_mask.astype(np.uint8), ref.shape[:2], interp=cv2.INTER_NEAREST) + m = crop_by_bbox(m, bbox) + if m.shape[:2] == rr.shape[:2] and np.count_nonzero(m) >= 32: + # Usa a mascara real da ROI, nao apenas o bbox. Isso faz gt_target/largest_blob + # puxarem a estimativa para o objeto dominante em vez da textura do retangulo inteiro. + mf = (m > 0).astype(np.float32) + rr = rr * mf + tt = tt * mf + + rr = normalize_to_u8(rr).astype(np.float32) + tt = normalize_to_u8(tt).astype(np.float32) + try: + (dx, dy), resp = cv2.phaseCorrelate(rr, tt) + return float(dx), float(dy), float(resp) + except Exception: + return 0.0, 0.0, 0.0 + + +def warp_from_phase(dx: float, dy: float) -> np.ndarray: + # Para alinhar tgt ao ref, desloca pelo negativo do shift estimado. + return np.array([[1.0, 0.0, -dx], [0.0, 1.0, -dy]], dtype=np.float32) + + +def warp_2x3_to_3x3(W: np.ndarray) -> np.ndarray: + H = np.eye(3, dtype=np.float32) + H[:2, :] = W.astype(np.float32) + return H + + +def warp_3x3_to_2x3(H: np.ndarray) -> np.ndarray: + return H[:2, :].astype(np.float32) + + +def local_warp_to_full(W_local: np.ndarray, bbox: Optional[Tuple[int, int, int, int]]) -> np.ndarray: + if bbox is None: + return W_local.astype(np.float32) + x0, y0, _x1, _y1 = bbox + T_full_to_local = np.array([[1, 0, -x0], [0, 1, -y0], [0, 0, 1]], dtype=np.float32) + T_local_to_full = np.array([[1, 0, x0], [0, 1, y0], [0, 0, 1]], dtype=np.float32) + H_local = warp_2x3_to_3x3(W_local) + H_full = T_local_to_full @ H_local @ T_full_to_local + return warp_3x3_to_2x3(H_full) + + +def try_ecc_alignment(ref: np.ndarray, tgt: np.ndarray, method: str, + bbox: Optional[Tuple[int, int, int, int]] = None, + max_iter: int = 60, eps: float = 1e-5) -> Tuple[np.ndarray, bool, str]: + ref_crop = crop_by_bbox(ref, bbox) + tgt_crop = crop_by_bbox(tgt, bbox) + + ref_img = normalize_to_u8(gradient_mag(ref_crop)).astype(np.float32) / 255.0 + tgt_img = normalize_to_u8(gradient_mag(tgt_crop)).astype(np.float32) / 255.0 + + if ref_img.shape[0] < 30 or ref_img.shape[1] < 30: + return np.eye(2, 3, dtype=np.float32), False, "ECC ROI pequena" + + if method == "ecc_translation": + motion = cv2.MOTION_TRANSLATION + elif method == "ecc_euclidean": + motion = cv2.MOTION_EUCLIDEAN + elif method == "ecc_affine": + motion = cv2.MOTION_AFFINE + else: + raise ValueError(f"Metodo ECC invalido: {method}") + + warp = np.eye(2, 3, dtype=np.float32) + criteria = (cv2.TERM_CRITERIA_EPS | cv2.TERM_CRITERIA_COUNT, max_iter, eps) + try: + cc, W_local = cv2.findTransformECC(ref_img, tgt_img, warp, motion, criteria) + W_full = local_warp_to_full(W_local.astype(np.float32), bbox) + return W_full, True, f"ECC cc={cc:.4f}" + except cv2.error as e: + return np.eye(2, 3, dtype=np.float32), False, f"ECC falhou: {str(e)[:90]}" + + +def extract_warp_metrics(warp: np.ndarray) -> Tuple[float, float]: + tx = float(warp[0, 2]) + ty = float(warp[1, 2]) + translation = math.hypot(tx, ty) + a = float(warp[0, 0]) + b = float(warp[0, 1]) + rot = -math.degrees(math.atan2(b, a)) + return translation, rot + + +def apply_warp(img: np.ndarray, warp: np.ndarray, inverse_map: bool = False, + border_mode: int = cv2.BORDER_REFLECT101) -> np.ndarray: + flags = cv2.INTER_LINEAR + if inverse_map: + flags |= cv2.WARP_INVERSE_MAP + return cv2.warpAffine( + img.astype(np.float32), + warp.astype(np.float32), + (img.shape[1], img.shape[0]), + flags=flags, + borderMode=border_mode, + borderValue=0.0, + ) + + +def estimate_alignment(ref: np.ndarray, tgt: np.ndarray, method: str, priority: str, + mask: Optional[np.ndarray], target_classes: List[int], + max_shift_px: float, max_rotation_deg: float, + min_improve_corr: float = -0.005) -> AlignResult: + roi_mask, roi_bbox, roi_note = make_alignment_roi(ref, tgt, priority, mask, target_classes) + + corr_before = edge_corr(ref, tgt) + roi_corr_before = edge_corr(ref, tgt, roi_mask) + pdx, pdy, presp = estimate_phase_shift(ref, tgt, roi_bbox, roi_mask) + + if method == "none": + return AlignResult( + method=method, + priority=priority, + accepted=False, + warp=np.eye(2, 3, dtype=np.float32), + used_inverse_map=False, + roi_mask=roi_mask, + roi_bbox=roi_bbox, + phase_dx=pdx, + phase_dy=pdy, + phase_response=presp, + edge_corr_before=corr_before, + edge_corr_after=corr_before, + roi_corr_before=roi_corr_before, + roi_corr_after=roi_corr_before, + note=f"sem correcao | {roi_note}", + ) + + if method == "phase": + warp = warp_from_phase(pdx, pdy) + corrected = apply_warp(tgt, warp, inverse_map=False) + corr_after = edge_corr(ref, corrected) + roi_corr_after = edge_corr(ref, corrected, roi_mask) + trans, rot = extract_warp_metrics(warp) + accepted = trans <= max_shift_px and (corr_after >= corr_before + min_improve_corr or roi_corr_after >= roi_corr_before + min_improve_corr) + note = f"phase resp={presp:.3f} | {roi_note}" + if not accepted: + # Mantem o warp proposto mesmo rejeitado. Assim o browser consegue mostrar + # o modo PROPOSTO/forcado para diagnostico visual. + note += " | rejeitado" + return AlignResult( + method=method, + priority=priority, + accepted=accepted, + warp=warp, + used_inverse_map=False, + roi_mask=roi_mask, + roi_bbox=roi_bbox, + phase_dx=pdx, + phase_dy=pdy, + phase_response=presp, + edge_corr_before=corr_before, + edge_corr_after=corr_after, + roi_corr_before=roi_corr_before, + roi_corr_after=roi_corr_after, + translation_px=trans, + rotation_deg=rot, + note=note, + ) + + warp, ok, note = try_ecc_alignment(ref, tgt, method, bbox=roi_bbox) + if not ok: + return AlignResult( + method=method, + priority=priority, + accepted=False, + warp=np.eye(2, 3, dtype=np.float32), + used_inverse_map=True, + roi_mask=roi_mask, + roi_bbox=roi_bbox, + phase_dx=pdx, + phase_dy=pdy, + phase_response=presp, + edge_corr_before=corr_before, + edge_corr_after=corr_before, + roi_corr_before=roi_corr_before, + roi_corr_after=roi_corr_before, + note=f"{note} | {roi_note}", + ) + + corrected = apply_warp(tgt, warp, inverse_map=True) + corr_after = edge_corr(ref, corrected) + roi_corr_after = edge_corr(ref, corrected, roi_mask) + trans, rot = extract_warp_metrics(warp) + accepted = ( + trans <= max_shift_px + and abs(rot) <= max_rotation_deg + and (corr_after >= corr_before + min_improve_corr or roi_corr_after >= roi_corr_before + min_improve_corr) + ) + + if not accepted: + # Mantem o warp proposto mesmo rejeitado. Assim o browser consegue mostrar + # o modo PROPOSTO/forcado para diagnostico visual. + note += " | rejeitado por limite/correlacao" + + return AlignResult( + method=method, + priority=priority, + accepted=accepted, + warp=warp, + used_inverse_map=True, + roi_mask=roi_mask, + roi_bbox=roi_bbox, + phase_dx=pdx, + phase_dy=pdy, + phase_response=presp, + edge_corr_before=corr_before, + edge_corr_after=corr_after, + roi_corr_before=roi_corr_before, + roi_corr_after=roi_corr_after, + translation_px=trans, + rotation_deg=rot, + note=f"{note} | {roi_note}", + ) + + +# ============================================================ +# Leitura dos dados: final e native +# ============================================================ + + +@dataclass +class SampleEntry: + dataset_root: Path + meta_path: Path + sample_name: str + + +@dataclass +class FusionDebug: + available: bool + warped_re: Optional[np.ndarray] = None + warped_nir: Optional[np.ndarray] = None + valid_re: Optional[np.ndarray] = None + valid_nir: Optional[np.ndarray] = None + common_mask: Optional[np.ndarray] = None + crop_box: Optional[Tuple[int, int, int, int]] = None + note: str = "" + + +class NativeDecoder: + def __init__(self): + self.core_cache: Dict[Any, Any] = {} + + def load_native(self, entry: SampleEntry) -> Dict[str, Any]: + meta = load_json(entry.meta_path) + saved_dtypes = meta.get("saved_payload_dtypes", {}) or {} + saved_shapes = meta.get("saved_payload_shapes", {}) or {} + cam_paths = resolve_camera_payloads(entry.meta_path, entry.dataset_root, meta) + + frame: Dict[str, np.ndarray] = {} + for cam_id, payload_path in cam_paths.items(): + saved_dtype = saved_dtypes.get(cam_id) + saved_shape = saved_shapes.get(cam_id) + if saved_dtype is None or saved_shape is None: + raise RuntimeError(f"Faltam dtype/shape para {cam_id} em {entry.meta_path.name}") + arr = np.fromfile(str(payload_path), dtype=np.dtype(saved_dtype)).reshape(tuple(saved_shape)) + frame[cam_id] = arr + + sensor_width = int(meta.get("sensor_width", 1280)) + sensor_height = int(meta.get("sensor_height", 800)) + bayer = meta.get("bayer_pattern", "RGGB") + calib_path = resolve_module_params_path(entry.meta_path, entry.dataset_root, meta) + + core = get_raw_processor_core( + core_cache=self.core_cache, + sensor_width=sensor_width, + sensor_height=sensor_height, + bayer=bayer, + calib_path=calib_path, + ) + + stream_meta = meta.get("stream_meta", {}) or {} + processing_meta = dict(stream_meta) + + if meta.get("actual_camera_controls") is not None: + processing_meta["actual_camera_controls"] = meta.get("actual_camera_controls") + if meta.get("startup_camera_controls") is not None: + processing_meta["startup_camera_controls"] = meta.get("startup_camera_controls") + if meta.get("radiometric_last_result") is not None: + processing_meta["radiometric_last_result"] = meta.get("radiometric_last_result") + + decoded = core.decode_stream_cameras(frame, processing_meta) + return { + "meta": meta, + "processing_meta": processing_meta, + "decoded": decoded, + "core": core, + "calib_path": calib_path, + "source": "native_decoded_before_fusion", + } + + +def role_to_images(decoded: Dict[str, Any]) -> Tuple[np.ndarray, np.ndarray, np.ndarray, Dict[str, str]]: + rgb = None + re = None + nir = None + role_cam = {} + for cam_id, item in decoded.items(): + role = str(item.get("role") or item.get("meta", {}).get("role") or "").lower() + if role: + role_cam[role] = str(cam_id) + img = item.get("image") + if role == "rgb": + rgb = img + elif role == "re": + re = img + elif role == "nir": + nir = img + if rgb is None or re is None or nir is None: + raise RuntimeError(f"Decoded sem rgb/re/nir completos. roles={role_cam}") + if rgb.ndim != 3 or rgb.shape[2] != 3: + raise RuntimeError(f"RGB nativo invalido: shape={rgb.shape}") + if re.ndim != 2 or nir.ndim != 2: + raise RuntimeError(f"RE/NIR nativos invalidos: re={re.shape} nir={nir.shape}") + return rgb.astype(np.float32), re.astype(np.float32), nir.astype(np.float32), role_cam + + +def compute_current_fusion_debug(core: Any, decoded: Dict[str, Any], processing_meta: Dict[str, Any]) -> FusionDebug: + try: + rgb, re, nir, _role_cam = role_to_images(decoded) + ref_h, ref_w = rgb.shape[:2] + ref_shape = (ref_h, ref_w) + + warped_re, valid_re = core._warp_with_valid_mask(re, "re", ref_shape, processing_meta) + warped_nir, valid_nir = core._warp_with_valid_mask(nir, "nir", ref_shape, processing_meta) + + common = np.ones((ref_h, ref_w), dtype=np.uint8) + common = np.logical_and(common > 0, valid_re > 0) + common = np.logical_and(common > 0, valid_nir > 0).astype(np.uint8) + + crop_box = core._compute_common_crop_box([np.ones((ref_h, ref_w), dtype=np.uint8), valid_re, valid_nir]) + if crop_box is not None: + crop_box = tuple(int(v) for v in crop_box) + + return FusionDebug( + available=True, + warped_re=warped_re.astype(np.float32), + warped_nir=warped_nir.astype(np.float32), + valid_re=valid_re.astype(np.uint8), + valid_nir=valid_nir.astype(np.uint8), + common_mask=common.astype(np.uint8), + crop_box=crop_box, + note="homografia atual aplicada so para debug", + ) + except Exception as e: + return FusionDebug(available=False, note=f"fusion_debug indisponivel: {str(e)[:160]}") + + +# ============================================================ +# Browser +# ============================================================ + + +@dataclass +class BrowserState: + space: str = "final" + method: str = "phase" + priority: str = "global" + show_corrected: bool = False + force_apply: bool = False + show_edges: bool = True + show_mask: bool = True + show_crop_debug: bool = True + panel_w: int = 410 + + +class DatasetAlignmentBrowserV2: + def __init__(self, args: argparse.Namespace): + ensure_imports_ok() + self.args = args + self.state = BrowserState( + space=args.space, + method=args.method, + priority=args.priority, + panel_w=args.panel_w, + show_crop_debug=not args.hide_crop_debug, + ) + self.final_core_cache: Dict[Any, Any] = {} + self.native_decoder = NativeDecoder() + self.entries = self._build_entries(args.input_path, args.groups_except) + if not self.entries: + raise RuntimeError("Nenhuma amostra encontrada para navegar.") + self.index = max(0, min(args.start_index, len(self.entries) - 1)) + self.window_name = "Dataset Alignment Browser V2" + self.save_dir = Path(args.save_dir) if args.save_dir else Path("alignment_browser_v2_out") + self.save_dir.mkdir(parents=True, exist_ok=True) + self.target_classes = [int(x) for x in str(args.target_classes).split(",") if x.strip()] + + def _build_entries(self, input_path: str, groups_except: str) -> List[SampleEntry]: + roots = find_dataset_roots(Path(input_path)) + skip = parse_csv_set(groups_except) if groups_except and parse_csv_set else set() + entries: List[SampleEntry] = [] + for root in roots: + if root.name.lower() in skip: + continue + for meta_path in list_meta_files(root): + entries.append(SampleEntry(root, meta_path, f"{root.name}__{meta_path.stem}")) + return entries + + def _load_mask_for_entry(self, entry: SampleEntry, hw: Tuple[int, int]) -> Optional[np.ndarray]: + try: + mask_path = resolve_mask_path(entry.dataset_root, entry.meta_path) + if mask_path is None: + return None + return load_mask(mask_path, hw, IGNORE_INDEX) + except Exception: + return None + + def load_sample_final(self, entry: SampleEntry) -> Dict[str, Any]: + tensor, meta, _source_payload = load_multispec_tensor(entry.meta_path, entry.dataset_root, self.final_core_cache) + h, w = tensor.shape[1], tensor.shape[2] + mask = self._load_mask_for_entry(entry, (h, w)) + + rgb_bgr = rgb_from_tensor(tensor, stretch=False) + rgb_stretch_bgr = rgb_from_tensor(tensor, stretch=True) + rgb_gray = gray_from_tensor_rgb(tensor) + re = tensor[3].astype(np.float32, copy=False) + nir = tensor[4].astype(np.float32, copy=False) + + return { + "space": "final", + "entry": entry, + "meta": meta, + "mask": mask, + "rgb_bgr": rgb_bgr, + "rgb_stretch_bgr": rgb_stretch_bgr, + "rgb_gray": rgb_gray, + "re": re, + "nir": nir, + "native_note": "tensor final: pos homografia/crop/resize/flat/radnorm conforme pipeline", + "fusion_debug": None, + } + + def load_sample_native(self, entry: SampleEntry) -> Dict[str, Any]: + native = self.native_decoder.load_native(entry) + rgb_hwc, re_native, nir_native, role_cam = role_to_images(native["decoded"]) + + # Para comparacao visual sem homografia, redimensiona RE/NIR para shape do RGB. + # Isso e apenas resize escalar, nao corrige paralaxe nem homografia. + ref_h, ref_w = rgb_hwc.shape[:2] + re_cmp = resize_to(re_native, (ref_h, ref_w)) + nir_cmp = resize_to(nir_native, (ref_h, ref_w)) + + rgb_bgr = rgb_hwc_to_bgr(rgb_hwc, stretch=False) + rgb_stretch_bgr = rgb_hwc_to_bgr(rgb_hwc, stretch=True) + rgb_gray = gray_from_rgb_hwc(rgb_hwc) + + mask = self._load_mask_for_entry(entry, (ref_h, ref_w)) + fusion_debug = compute_current_fusion_debug(native["core"], native["decoded"], native["processing_meta"]) + + return { + "space": "native", + "entry": entry, + "meta": native["meta"], + "mask": mask, + "rgb_bgr": rgb_bgr, + "rgb_stretch_bgr": rgb_stretch_bgr, + "rgb_gray": rgb_gray, + "re": re_cmp.astype(np.float32), + "nir": nir_cmp.astype(np.float32), + "re_native_original": re_native.astype(np.float32), + "nir_native_original": nir_native.astype(np.float32), + "rgb_native_original": rgb_hwc.astype(np.float32), + "role_cam": role_cam, + "native_note": f"native decoded antes da fusao | RGB={rgb_hwc.shape[:2]} RE={re_native.shape} NIR={nir_native.shape} | overlay usa resize simples para RGB", + "fusion_debug": fusion_debug, + } + + def load_current_sample(self) -> Dict[str, Any]: + entry = self.entries[self.index] + if self.state.space == "native": + sample = self.load_sample_native(entry) + elif self.state.space == "final": + sample = self.load_sample_final(entry) + else: + # fallback defensivo + sample = self.load_sample_final(entry) + + re_align = estimate_alignment( + sample["rgb_gray"], sample["re"], self.state.method, self.state.priority, + sample["mask"] if self.state.show_mask else None, self.target_classes, + max_shift_px=self.args.max_shift_px, + max_rotation_deg=self.args.max_rotation_deg, + min_improve_corr=self.args.min_improve_corr, + ) + nir_align = estimate_alignment( + sample["rgb_gray"], sample["nir"], self.state.method, self.state.priority, + sample["mask"] if self.state.show_mask else None, self.target_classes, + max_shift_px=self.args.max_shift_px, + max_rotation_deg=self.args.max_rotation_deg, + min_improve_corr=self.args.min_improve_corr, + ) + + apply_re = bool(re_align.accepted or self.state.force_apply) + apply_nir = bool(nir_align.accepted or self.state.force_apply) + re_corr = apply_warp(sample["re"], re_align.warp, inverse_map=re_align.used_inverse_map) if apply_re else sample["re"] + nir_corr = apply_warp(sample["nir"], nir_align.warp, inverse_map=nir_align.used_inverse_map) if apply_nir else sample["nir"] + + sample["re_align"] = re_align + sample["nir_align"] = nir_align + sample["re_corr"] = re_corr + sample["nir_corr"] = nir_corr + return sample + + def build_panels(self, sample: Dict[str, Any]) -> np.ndarray: + entry = sample["entry"] + idx_txt = f"[{self.index + 1}/{len(self.entries)}] {entry.sample_name}" + space_txt = self.state.space.upper() + if self.state.show_corrected and self.state.force_apply: + mode_txt = "PROPOSTO" + elif self.state.show_corrected: + mode_txt = "CORRIGIDO" + else: + mode_txt = "BRUTO" + + rgb_bgr = sample["rgb_bgr"] + rgb_stretch = sample["rgb_stretch_bgr"] + rgb_gray = sample["rgb_gray"] + re_raw = sample["re"] + nir_raw = sample["nir"] + re = sample["re_corr"] if self.state.show_corrected else re_raw + nir = sample["nir_corr"] if self.state.show_corrected else nir_raw + mask = sample["mask"] if self.state.show_mask else None + re_align: AlignResult = sample["re_align"] + nir_align: AlignResult = sample["nir_align"] + + panels: List[Tuple[str, np.ndarray, str]] = [] + + info_lines = [ + idx_txt, + f"space={space_txt} | exibicao={mode_txt} | metodo={self.state.method} | prioridade={self.state.priority}", + "teclas: A/D prev/next | Up/Down +/-10 | X space | C raw/corr | F force/proposto | M metodo | P prioridade | V crop | E edges | K mask | S save | Q sair", + sample.get("native_note", ""), + f"RE global before/after={re_align.edge_corr_before:.4f}/{re_align.edge_corr_after:.4f} | ROI before/after={re_align.roi_corr_before:.4f}/{re_align.roi_corr_after:.4f}", + f"RE phase=({re_align.phase_dx:.2f},{re_align.phase_dy:.2f}) resp={re_align.phase_response:.3f} | accepted={re_align.accepted} | force={self.state.force_apply} | trans={re_align.translation_px:.2f}px rot={re_align.rotation_deg:.2f}deg", + f"NIR global before/after={nir_align.edge_corr_before:.4f}/{nir_align.edge_corr_after:.4f} | ROI before/after={nir_align.roi_corr_before:.4f}/{nir_align.roi_corr_after:.4f}", + f"NIR phase=({nir_align.phase_dx:.2f},{nir_align.phase_dy:.2f}) resp={nir_align.phase_response:.3f} | accepted={nir_align.accepted} | force={self.state.force_apply} | trans={nir_align.translation_px:.2f}px rot={nir_align.rotation_deg:.2f}deg", + f"RE note: {re_align.note}", + f"NIR note: {nir_align.note}", + ] + info_panel = make_info_panel(info_lines, size=(950, 305)) + + mask_on_rgb = mask_overlay(rgb_bgr, mask) + roi_re_panel = draw_roi(rgb_bgr, re_align.roi_mask, re_align.roi_bbox, "ROI RE") + roi_nir_panel = draw_roi(rgb_bgr, nir_align.roi_mask, nir_align.roi_bbox, "ROI NIR") + + panels.extend([ + ("RGB", rgb_bgr, idx_txt), + ("RGB stretch", rgb_stretch, "visual somente"), + ("Mask overlay", mask_on_rgb, "GT over RGB" if mask is not None else "sem mascara"), + ("RE bruto", colorize_gray(re_raw), f"space={space_txt}"), + ("NIR bruto", colorize_gray(nir_raw), f"space={space_txt}"), + ("Info", info_panel, "metricas de alinhamento"), + ("ROI usado RE", roi_re_panel, f"priority={self.state.priority}"), + ("ROI usado NIR", roi_nir_panel, f"priority={self.state.priority}"), + (f"RE exibido {mode_txt}", colorize_gray(re), "bruto ou corrigido"), + ]) + + panels.extend([ + (f"RGBgray vs RE ({mode_txt})", falsecolor_overlay(rgb_gray, re), "verde=RGBgray magenta=RE"), + (f"RGBgray vs NIR ({mode_txt})", falsecolor_overlay(rgb_gray, nir), "verde=RGBgray magenta=NIR"), + (f"RE vs NIR ({mode_txt})", falsecolor_overlay(re, nir), "verde=RE magenta=NIR"), + (f"RGB + RE tint ({mode_txt})", alpha_blend(rgb_bgr, re, alpha=0.38, cmap=cv2.COLORMAP_INFERNO), "RGB com RE"), + (f"RGB + NIR tint ({mode_txt})", alpha_blend(rgb_bgr, nir, alpha=0.38, cmap=cv2.COLORMAP_VIRIDIS), "RGB com NIR"), + (f"RGB + RE edges ({mode_txt})", draw_edges_on_rgb(rgb_bgr, re, (0, 0, 255)), "bordas RE sobre RGB"), + ]) + + if self.state.show_edges: + panels.extend([ + ("Edge overlay bruto", edge_overlay(rgb_gray, re_raw, nir_raw), "G=RGB | R=RE | B=NIR"), + (f"Edge overlay {mode_txt}", edge_overlay(rgb_gray, re, nir), "G=RGB | R=RE | B=NIR"), + (f"RGB + NIR edges ({mode_txt})", draw_edges_on_rgb(rgb_bgr, nir, (255, 255, 0)), "bordas NIR sobre RGB"), + ]) + + # Debug de homografia/crop atual no modo native. + fusion_debug: Optional[FusionDebug] = sample.get("fusion_debug") + if self.state.show_crop_debug and sample.get("space") == "native": + if fusion_debug and fusion_debug.available: + warped_re = resize_to(fusion_debug.warped_re, rgb_bgr.shape[:2]) + warped_nir = resize_to(fusion_debug.warped_nir, rgb_bgr.shape[:2]) + common = fusion_debug.common_mask.astype(np.uint8) * 255 + common_bgr = cv2.cvtColor(common, cv2.COLOR_GRAY2BGR) + crop_rgb = draw_crop_box(rgb_bgr, fusion_debug.crop_box, "crop comum atual") + panels.extend([ + ("Debug homografia atual RE", falsecolor_overlay(rgb_gray, warped_re), "verde=RGBgray magenta=RE warp atual"), + ("Debug homografia atual NIR", falsecolor_overlay(rgb_gray, warped_nir), "verde=RGBgray magenta=NIR warp atual"), + ("Mascara valida comum", common_bgr, f"crop={fusion_debug.crop_box}"), + ("Crop comum atual", crop_rgb, "area que vira tensor final"), + ("Warp atual RE vs NIR", falsecolor_overlay(warped_re, warped_nir), "verde=RE magenta=NIR apos H atual"), + ("Edge H atual", edge_overlay(rgb_gray, warped_re, warped_nir), "G=RGB | R=RE | B=NIR"), + ]) + else: + note = fusion_debug.note if fusion_debug else "sem fusion debug" + panels.append(("Crop debug indisponivel", make_info_panel([note], size=(900, 260)), "")) + + canvas = make_grid(panels, panel_w=self.state.panel_w, cols=3) + footer_h = 34 + footer = np.full((footer_h, canvas.shape[1], 3), 18, dtype=np.uint8) + footer_text = ( + f"sample {self.index+1}/{len(self.entries)} | space={space_txt} | exibicao={mode_txt} | force={self.state.force_apply} | metodo={self.state.method} | prioridade={self.state.priority} | " + f"RE g={re_align.edge_corr_before:.3f}->{re_align.edge_corr_after:.3f} roi={re_align.roi_corr_before:.3f}->{re_align.roi_corr_after:.3f} | " + f"NIR g={nir_align.edge_corr_before:.3f}->{nir_align.edge_corr_after:.3f} roi={nir_align.roi_corr_before:.3f}->{nir_align.roi_corr_after:.3f}" + ) + cv2.putText(footer, cv_text(footer_text[:260]), (10, 22), cv2.FONT_HERSHEY_SIMPLEX, 0.52, (220, 220, 220), 1, cv2.LINE_AA) + return np.vstack([canvas, footer]) + + def save_current(self, canvas: np.ndarray, entry: SampleEntry): + mode_name = 'proposto' if (self.state.show_corrected and self.state.force_apply) else ('corr' if self.state.show_corrected else 'raw') + fname = f"{entry.sample_name}__space-{self.state.space}__{self.state.method}__{self.state.priority}__{mode_name}.png" + out_path = self.save_dir / fname + cv2.imwrite(str(out_path), canvas) + print(f"[OK] painel salvo: {out_path}") + + def next_method(self): + methods = ["phase", "ecc_translation", "ecc_euclidean", "ecc_affine", "none"] + cur = methods.index(self.state.method) if self.state.method in methods else 0 + self.state.method = methods[(cur + 1) % len(methods)] + + def next_priority(self): + priorities = ["global", "largest_blob", "gt_target"] + cur = priorities.index(self.state.priority) if self.state.priority in priorities else 0 + self.state.priority = priorities[(cur + 1) % len(priorities)] + + def next_space(self): + spaces = ["final", "native"] + cur = spaces.index(self.state.space) if self.state.space in spaces else 0 + self.state.space = spaces[(cur + 1) % len(spaces)] + + def run(self): + cv2.namedWindow(self.window_name, cv2.WINDOW_NORMAL) + cv2.resizeWindow(self.window_name, 1640, 980) + + while True: + entry = self.entries[self.index] + try: + sample = self.load_current_sample() + canvas = self.build_panels(sample) + except Exception as e: + canvas = make_info_panel([ + f"Erro ao carregar sample {self.index+1}/{len(self.entries)}", + str(entry.meta_path), + str(e), + "Use A/D para navegar, Q para sair.", + ], size=(1100, 420)) + print(f"[ERRO] {entry.sample_name}: {e}") + + cv2.imshow(self.window_name, canvas) + key = cv2.waitKeyEx(0) + + if key in (27, ord('q'), ord('Q')): + break + elif key in (ord('d'), ord('D'), 2555904): + self.index = min(self.index + 1, len(self.entries) - 1) + elif key in (ord('a'), ord('A'), 2424832): + self.index = max(self.index - 1, 0) + elif key == 2490368: + self.index = max(self.index - 10, 0) + elif key == 2621440: + self.index = min(self.index + 10, len(self.entries) - 1) + elif key in (ord('c'), ord('C')): + self.state.show_corrected = not self.state.show_corrected + elif key in (ord('f'), ord('F')): + self.state.force_apply = not self.state.force_apply + if self.state.force_apply: + self.state.show_corrected = True + elif key in (ord('x'), ord('X')): + self.next_space() + elif key in (ord('m'), ord('M')): + self.next_method() + elif key in (ord('p'), ord('P')): + self.next_priority() + elif key in (ord('e'), ord('E')): + self.state.show_edges = not self.state.show_edges + elif key in (ord('k'), ord('K')): + self.state.show_mask = not self.state.show_mask + elif key in (ord('v'), ord('V')): + self.state.show_crop_debug = not self.state.show_crop_debug + elif key in (ord('s'), ord('S')): + self.save_current(canvas, entry) + elif key in (ord('h'), ord('H')): + print("\n=== HELP V2 ===") + print("A / Left : amostra anterior") + print("D / Right : proxima amostra") + print("Up / Down : pula -10 / +10") + print("X : alterna space final/native") + print("C : alterna bruto/corrigido") + print("F : forca aplicar warp proposto mesmo se rejeitado") + print("M : alterna metodo phase/ecc_translation/ecc_euclidean/ecc_affine/none") + print("P : alterna prioridade global/largest_blob/gt_target") + print("V : mostra/esconde debug de homografia/crop atual") + print("E : alterna paineis de borda") + print("K : mostra/esconde mascara") + print("S : salva painel atual") + print("Q / Esc : sair") + print("==============\n") + + cv2.destroyAllWindows() + + +# ============================================================ +# CLI +# ============================================================ + + +def build_argparser() -> argparse.ArgumentParser: + ap = argparse.ArgumentParser( + description="Navegador visual V2 para comparar alinhamento nativo/final RGB/RE/NIR e testar correcao por bordas." + ) + ap.add_argument("--input_path", type=str, required=True, help="Raiz do dataset ou super-root com grupos.") + ap.add_argument("--groups-except", type=str, default="", help="Ex: chao ou chao,chao_cana") + ap.add_argument("--start-index", type=int, default=0, help="Indice inicial para navegacao.") + ap.add_argument("--space", type=str, default="native", choices=["native", "final"], help="Espaco inicial de visualizacao.") + ap.add_argument("--method", type=str, default="phase", choices=["phase", "ecc_translation", "ecc_euclidean", "ecc_affine", "none"], help="Metodo inicial de correcao dinamica.") + ap.add_argument("--priority", type=str, default="global", choices=["global", "largest_blob", "gt_target"], help="Prioridade inicial para ROI do alinhamento.") + ap.add_argument("--target-classes", type=str, default="1,2", help="Classes usadas no modo gt_target. Padrao: 1,2 = cana,erva") + ap.add_argument("--panel-w", type=int, default=410, help="Largura de cada painel no grid.") + ap.add_argument("--max-shift-px", type=float, default=35.0, help="Limite de translacao aceito para a correcao dinamica.") + ap.add_argument("--max-rotation-deg", type=float, default=3.0, help="Limite de rotacao aceito para ECC euclidean/affine.") + ap.add_argument("--min-improve-corr", type=float, default=-0.003, help="Melhoria minima aceitavel na correlacao. Negativo leve permite correcao equivalente.") + ap.add_argument("--hide-crop-debug", action="store_true", help="Esconde paineis de debug da homografia/crop atual no modo native.") + ap.add_argument("--save-dir", type=str, default="alignment_browser_v2_out", help="Pasta para salvar paineis com tecla S.") + return ap + + +if __name__ == "__main__": + args = build_argparser().parse_args() + browser = DatasetAlignmentBrowserV2(args) + browser.run() diff --git a/Python/OAK/datasets/oak-fcc-3/utils/depth_probe.py b/Python/OAK/datasets/oak-fcc-3/utils/depth_probe.py new file mode 100644 index 000000000..22561cbd2 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/depth_probe.py @@ -0,0 +1,690 @@ +import argparse +import json +import time +from pathlib import Path +from collections import deque +from typing import Dict, Optional, Tuple + +import cv2 +import numpy as np + +try: + import depthai as dai +except Exception as e: + raise RuntimeError( + "Nao consegui importar depthai. Ative o venv correto e instale depthai antes de rodar. " + f"Erro original: {e}" + ) + + +# ============================================================ +# OAK-FCC-3P Depth Probe - API v3 style +# ------------------------------------------------------------ +# Este script evita XLinkOut/getOutputQueue, porque seu ambiente DepthAI +# nao expoe dai.node.XLinkOut. Ele usa createOutputQueue() direto nas saidas. +# +# Exemplo: +# python -m utils.depth_probe --left CAM_B --right CAM_C --rgb CAM_A --enable-rgb --lrcheck --extended --subpixel --confidence 200 --median 7 +# +# Se depth/disparity parecer invertido ou muito ruim: +# python -m utils.depth_probe --left CAM_C --right CAM_B --rgb CAM_A --enable-rgb --lrcheck --extended --subpixel +# ============================================================ + + +# ============================================================ +# DepthAI helpers +# ============================================================ + + +def socket_from_name(name: str): + name = str(name).strip().upper() + aliases = { + "A": "CAM_A", + "B": "CAM_B", + "C": "CAM_C", + "LEFT": "CAM_B", + "RIGHT": "CAM_C", + "RGB": "CAM_A", + } + name = aliases.get(name, name) + + if hasattr(dai.CameraBoardSocket, name): + return getattr(dai.CameraBoardSocket, name) + + legacy = { + "CAM_A": getattr(dai.CameraBoardSocket, "RGB", None), + "CAM_B": getattr(dai.CameraBoardSocket, "LEFT", None), + "CAM_C": getattr(dai.CameraBoardSocket, "RIGHT", None), + } + if legacy.get(name) is not None: + return legacy[name] + + raise ValueError(f"Socket invalido: {name}. Use CAM_A, CAM_B ou CAM_C.") + + + +def create_node(pipeline: dai.Pipeline, node_type): + """Wrapper pequeno para manter o codigo legivel.""" + return pipeline.create(node_type) + + + +def mono_resolution_from_name(name: str): + name = str(name).strip().lower() + r = dai.MonoCameraProperties.SensorResolution + table = { + "400p": getattr(r, "THE_400_P", None), + "480p": getattr(r, "THE_480_P", None), + "720p": getattr(r, "THE_720_P", None), + "800p": getattr(r, "THE_800_P", None), + } + if name not in table or table[name] is None: + valid = ", ".join(k for k, v in table.items() if v is not None) + raise ValueError(f"Resolucao mono invalida: {name}. Valid={valid}") + return table[name] + + + +def median_filter_from_name(name: str): + name = str(name).strip().upper() + + enum_candidates = [] + if hasattr(dai, "MedianFilter"): + enum_candidates.append(dai.MedianFilter) + if hasattr(dai, "StereoDepthProperties") and hasattr(dai.StereoDepthProperties, "MedianFilter"): + enum_candidates.append(dai.StereoDepthProperties.MedianFilter) + + key_map = { + "OFF": ("MEDIAN_OFF", "KERNEL_NONE", "OFF"), + "3": ("KERNEL_3x3", "MEDIAN_3x3"), + "5": ("KERNEL_5x5", "MEDIAN_5x5"), + "7": ("KERNEL_7x7", "MEDIAN_7x7"), + } + + if name not in key_map: + raise ValueError("--median deve ser OFF, 3, 5 ou 7") + + for enum in enum_candidates: + for attr in key_map[name]: + value = getattr(enum, attr, None) + if value is not None: + return value + + print("[WARN] Esta versao do DepthAI nao expos enum de MedianFilter; seguindo sem aplicar median filter.") + return None + + + +def set_if_exists(obj, method_name: str, *args) -> bool: + fn = getattr(obj, method_name, None) + if callable(fn): + try: + fn(*args) + return True + except Exception as e: + print(f"[WARN] {method_name} falhou: {e}") + return False + + + +def apply_stereo_config(stereo, args): + # Preset: tenta alguns nomes comuns. + try: + preset = getattr(dai.node.StereoDepth.PresetMode, "HIGH_DENSITY", None) + if preset is None: + preset = getattr(dai.node.StereoDepth.PresetMode, "FAST_DENSITY", None) + if preset is not None: + stereo.setDefaultProfilePreset(preset) + except Exception as e: + print(f"[WARN] preset StereoDepth nao aplicado: {e}") + + set_if_exists(stereo, "setLeftRightCheck", bool(args.lrcheck)) + set_if_exists(stereo, "setExtendedDisparity", bool(args.extended)) + set_if_exists(stereo, "setSubpixel", bool(args.subpixel)) + + # Confidence threshold: mudou bastante entre versoes. + applied_conf = False + applied_conf = set_if_exists(stereo, "setConfidenceThreshold", int(args.confidence)) or applied_conf + + if not applied_conf: + try: + applied_conf = set_if_exists(stereo.initialConfig, "setConfidenceThreshold", int(args.confidence)) or applied_conf + except Exception: + pass + + # Algumas APIs v3 nao tem initialConfig.get(); tentamos manipular config direto se existir. + try: + cfg = stereo.initialConfig + if hasattr(cfg, "costMatching") and hasattr(cfg.costMatching, "confidenceThreshold"): + cfg.costMatching.confidenceThreshold = int(args.confidence) + applied_conf = True + except Exception: + pass + + if not applied_conf: + print("[WARN] Nao consegui aplicar confidenceThreshold nesta versao. Seguindo com default.") + + median_value = median_filter_from_name(args.median) + if median_value is not None: + applied_median = False + try: + applied_median = set_if_exists(stereo.initialConfig, "setMedianFilter", median_value) + except Exception: + pass + if not applied_median: + try: + cfg = stereo.initialConfig + if hasattr(cfg, "postProcessing") and hasattr(cfg.postProcessing, "median"): + cfg.postProcessing.median = median_value + applied_median = True + except Exception: + pass + if not applied_median: + print("[WARN] Nao consegui aplicar median filter nesta versao. Seguindo com default.") + + # Pos-processamento opcional. Tudo defensivo. + try: + cfg = stereo.initialConfig + pp = getattr(cfg, "postProcessing", None) + if pp is not None: + if hasattr(pp, "speckleFilter"): + pp.speckleFilter.enable = bool(args.speckle) + pp.speckleFilter.speckleRange = int(args.speckle_range) + if hasattr(pp, "temporalFilter"): + pp.temporalFilter.enable = bool(args.temporal) + if hasattr(pp, "spatialFilter"): + pp.spatialFilter.enable = bool(args.spatial) + if hasattr(pp.spatialFilter, "holeFillingRadius"): + pp.spatialFilter.holeFillingRadius = int(args.hole_filling_radius) + if hasattr(pp.spatialFilter, "numIterations"): + pp.spatialFilter.numIterations = int(args.spatial_iterations) + except Exception as e: + print(f"[WARN] Nao consegui aplicar filtros de pos-processamento: {e}") + + +# ============================================================ +# Visual helpers +# ============================================================ + + +def normalize_u8(arr: np.ndarray, p_low: float = 1.0, p_high: float = 99.0) -> np.ndarray: + x = np.asarray(arr, dtype=np.float32) + finite = np.isfinite(x) + if not np.any(finite): + return np.zeros(x.shape[:2], dtype=np.uint8) + vals = x[finite] + lo = float(np.percentile(vals, p_low)) + hi = float(np.percentile(vals, p_high)) + if hi <= lo + 1e-6: + hi = lo + 1.0 + y = np.clip((x - lo) / (hi - lo), 0.0, 1.0) + return (y * 255).astype(np.uint8) + + + +def heatmap(arr: np.ndarray, p_low: float = 1.0, p_high: float = 99.0, cmap=cv2.COLORMAP_TURBO) -> np.ndarray: + return cv2.applyColorMap(normalize_u8(arr, p_low, p_high), cmap) + + + +def put_label(img: np.ndarray, title: str, subtitle: str = "") -> np.ndarray: + if img is None: + img = np.zeros((300, 400, 3), dtype=np.uint8) + if img.ndim == 2: + img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR) + out = img.copy() + hbox = 58 if subtitle else 36 + cv2.rectangle(out, (0, 0), (out.shape[1], hbox), (0, 0, 0), -1) + cv2.putText(out, str(title)[:90], (10, 24), cv2.FONT_HERSHEY_SIMPLEX, 0.65, (0, 255, 255), 2, cv2.LINE_AA) + if subtitle: + cv2.putText(out, str(subtitle)[:120], (10, 48), cv2.FONT_HERSHEY_SIMPLEX, 0.44, (255, 255, 255), 1, cv2.LINE_AA) + return out + + + +def resize_keep(img: np.ndarray, width: int) -> np.ndarray: + scale = width / img.shape[1] + height = max(1, int(img.shape[0] * scale)) + return cv2.resize(img, (width, height), interpolation=cv2.INTER_AREA) + + + +def make_grid(panels, panel_w: int = 430, cols: int = 3) -> np.ndarray: + rendered = [] + for title, img, subtitle in panels: + if img is None: + img = np.zeros((300, 400, 3), dtype=np.uint8) + if img.ndim == 2: + img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR) + small = resize_keep(img, panel_w) + rendered.append(put_label(small, title, subtitle)) + + if not rendered: + return np.zeros((300, 600, 3), dtype=np.uint8) + + max_h = max(x.shape[0] for x in rendered) + padded = [] + for im in rendered: + if im.shape[0] < max_h: + im = np.vstack([im, np.zeros((max_h - im.shape[0], im.shape[1], 3), dtype=np.uint8)]) + padded.append(im) + + gap = 10 + gap_w = np.full((max_h, gap, 3), 22, dtype=np.uint8) + filler = np.zeros_like(padded[0]) + rows = [] + for i in range(0, len(padded), cols): + items = padded[i:i + cols] + while len(items) < cols: + items.append(filler.copy()) + row = items[0] + for j in range(1, cols): + row = np.hstack([row, gap_w, items[j]]) + rows.append(row) + + gap_h = np.full((gap, rows[0].shape[1], 3), 22, dtype=np.uint8) + canvas = rows[0] + for row in rows[1:]: + canvas = np.vstack([canvas, gap_h, row]) + return canvas + + + +def safe_stats_depth_mm(depth: np.ndarray, min_mm: int, max_mm: int) -> Dict[str, float]: + d = np.asarray(depth, dtype=np.float32) + valid = np.isfinite(d) & (d > min_mm) & (d < max_mm) + total = int(d.size) + count = int(np.count_nonzero(valid)) + if count <= 0: + return { + "valid_pct": 0.0, + "count": 0, + "mean_mm": 0.0, + "median_mm": 0.0, + "p10_mm": 0.0, + "p90_mm": 0.0, + "std_mm": 0.0, + } + vals = d[valid] + return { + "valid_pct": float(count * 100.0 / max(1, total)), + "count": count, + "mean_mm": float(np.mean(vals)), + "median_mm": float(np.median(vals)), + "p10_mm": float(np.percentile(vals, 10)), + "p90_mm": float(np.percentile(vals, 90)), + "std_mm": float(np.std(vals)), + } + + + +def stats_disparity(disp: np.ndarray) -> Dict[str, float]: + d = np.asarray(disp, dtype=np.float32) + valid = np.isfinite(d) & (d > 0) + total = int(d.size) + count = int(np.count_nonzero(valid)) + if count <= 0: + return {"valid_pct": 0.0, "mean": 0.0, "median": 0.0, "p90": 0.0, "std": 0.0} + vals = d[valid] + return { + "valid_pct": float(count * 100.0 / max(1, total)), + "mean": float(np.mean(vals)), + "median": float(np.median(vals)), + "p90": float(np.percentile(vals, 90)), + "std": float(np.std(vals)), + } + + + +def draw_metrics_panel(metrics: Dict[str, float], disp_stats: Dict[str, float], fps: float, args: argparse.Namespace, + size: Tuple[int, int] = (900, 260)) -> np.ndarray: + w, h = size + img = np.zeros((h, w, 3), dtype=np.uint8) + lines = [ + "OAK-FCC-3P depth probe - API v3 queues", + f"left={args.left} right={args.right} rgb={args.rgb} | fps={fps:.1f}", + f"lrcheck={args.lrcheck} extended={args.extended} subpixel={args.subpixel} median={args.median} confidence={args.confidence}", + f"depth valid={metrics['valid_pct']:.1f}% | median={metrics['median_mm']:.0f}mm mean={metrics['mean_mm']:.0f}mm p10={metrics['p10_mm']:.0f} p90={metrics['p90_mm']:.0f} std={metrics['std_mm']:.0f}", + f"disp valid={disp_stats['valid_pct']:.1f}% | median={disp_stats['median']:.2f} mean={disp_stats['mean']:.2f} p90={disp_stats['p90']:.2f} std={disp_stats['std']:.2f}", + "teclas: Q/ESC sair | S salvar snapshot | H ajuda", + "Leitura: heatmap coerente + valid% alto = vale investigar depth. Ruido/sopa = descartar depth metrico.", + ] + y = 28 + for i, line in enumerate(lines): + color = (0, 255, 255) if i == 0 else (235, 235, 235) + cv2.putText(img, line[:145], (12, y), cv2.FONT_HERSHEY_SIMPLEX, 0.55, color, 1, cv2.LINE_AA) + y += 26 + return img + + +# ============================================================ +# Queue helpers +# ============================================================ + + +def create_output_queue(output, name: str, max_size: int = 4, blocking: bool = False): + if output is None: + return None + + fn = getattr(output, "createOutputQueue", None) + if callable(fn): + return fn(maxSize=max_size, blocking=blocking) + + raise RuntimeError( + f"A saida '{name}' nao possui createOutputQueue(). " + "Seu DepthAI parece nao ter XLinkOut, mas tambem nao expos queues v3 nessa saida." + ) + + + +def get_frame(q) -> Optional[np.ndarray]: + if q is None: + return None + try: + msg = q.tryGet() + except Exception: + return None + if msg is None: + return None + + # ImgFrame normalmente tem getFrame(). Alguns previews coloridos podem ter getCvFrame(). + try: + return msg.getFrame() + except Exception: + pass + try: + return msg.getCvFrame() + except Exception: + return None + + +# ============================================================ +# Pipeline +# ============================================================ + + +def create_pipeline_and_outputs(args: argparse.Namespace): + pipeline = dai.Pipeline() + + left = create_node(pipeline, dai.node.MonoCamera) + right = create_node(pipeline, dai.node.MonoCamera) + + left.setBoardSocket(socket_from_name(args.left)) + right.setBoardSocket(socket_from_name(args.right)) + left.setResolution(mono_resolution_from_name(args.mono_resolution)) + right.setResolution(mono_resolution_from_name(args.mono_resolution)) + left.setFps(float(args.fps)) + right.setFps(float(args.fps)) + + stereo = create_node(pipeline, dai.node.StereoDepth) + apply_stereo_config(stereo, args) + + left.out.link(stereo.left) + right.out.link(stereo.right) + + outputs = { + "left": left.out, + "right": right.out, + "disparity": stereo.disparity, + "depth": stereo.depth, + "rectified_left": stereo.rectifiedLeft, + "rectified_right": stereo.rectifiedRight, + } + + nodes = { + "left": left, + "right": right, + "stereo": stereo, + } + + if args.enable_rgb: + rgb = create_node(pipeline, dai.node.ColorCamera) + rgb.setBoardSocket(socket_from_name(args.rgb)) + rgb.setResolution(dai.ColorCameraProperties.SensorResolution.THE_800_P) + rgb.setFps(float(args.fps)) + rgb.setInterleaved(False) + rgb.setColorOrder(dai.ColorCameraProperties.ColorOrder.BGR) + rgb.setPreviewSize(int(args.rgb_preview_w), int(args.rgb_preview_h)) + outputs["rgb"] = rgb.preview + nodes["rgb"] = rgb + + return pipeline, outputs, nodes + + +# ============================================================ +# Runtime +# ============================================================ + + +def save_snapshot(out_dir: Path, frames: Dict[str, np.ndarray], metrics: Dict[str, float], disp_stats: Dict[str, float], args: argparse.Namespace): + ts = time.strftime("%Y%m%d_%H%M%S") + folder = out_dir / f"depth_probe_{ts}" + folder.mkdir(parents=True, exist_ok=True) + + for name, frame in frames.items(): + if frame is None: + continue + if frame.ndim == 2: + if frame.dtype == np.uint16: + np.save(str(folder / f"{name}.npy"), frame) + cv2.imwrite(str(folder / f"{name}_preview.png"), normalize_u8(frame)) + else: + cv2.imwrite(str(folder / f"{name}.png"), normalize_u8(frame)) + else: + cv2.imwrite(str(folder / f"{name}.png"), frame) + + meta = { + "created_at": ts, + "args": vars(args), + "depth_metrics": metrics, + "disparity_metrics": disp_stats, + } + with open(folder / "metrics.json", "w", encoding="utf-8") as f: + json.dump(meta, f, ensure_ascii=False, indent=2) + + print(f"[OK] snapshot salvo em: {folder}") + + + +def start_pipeline_v3(pipeline): + fn = getattr(pipeline, "start", None) + if not callable(fn): + raise RuntimeError( + "Este ambiente nao tem pipeline.start(). " + "Tambem nao tinha XLinkOut. Pode ser uma build DepthAI intermediaria/incompleta." + ) + fn() + + + +def stop_pipeline_v3(pipeline): + try: + fn = getattr(pipeline, "stop", None) + if callable(fn): + fn() + except Exception: + pass + + + +def pipeline_running(pipeline) -> bool: + fn = getattr(pipeline, "isRunning", None) + if callable(fn): + try: + return bool(fn()) + except Exception: + return True + return True + + + +def main(args: argparse.Namespace): + out_dir = Path(args.out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + + pipeline, outputs, _nodes = create_pipeline_and_outputs(args) + + print("[INFO] Pipeline criado em modo API v3/sem XLinkOut.") + print(f"[INFO] left={args.left} right={args.right} rgb={args.rgb} enable_rgb={args.enable_rgb}") + print("[INFO] Se depth vier ruim, teste invertendo --left/--right.") + + queues = { + name: create_output_queue(output, name, max_size=4, blocking=False) + for name, output in outputs.items() + } + + start_pipeline_v3(pipeline) + + cv2.namedWindow("OAK-FCC-3P Depth Probe", cv2.WINDOW_NORMAL) + cv2.resizeWindow("OAK-FCC-3P Depth Probe", 1500, 900) + + last_frames: Dict[str, Optional[np.ndarray]] = { + "left": None, + "right": None, + "rectified_left": None, + "rectified_right": None, + "disparity": None, + "depth": None, + "rgb": None, + "canvas": None, + } + + frame_times = deque(maxlen=40) + last_metrics = safe_stats_depth_mm(np.zeros((1, 1), dtype=np.uint16), args.min_depth_mm, args.max_depth_mm) + last_disp_stats = stats_disparity(np.zeros((1, 1), dtype=np.float32)) + + try: + while pipeline_running(pipeline): + updated = False + + for name, queue in queues.items(): + frame = get_frame(queue) + if frame is not None: + last_frames[name] = frame + updated = True + + if not updated: + key = cv2.waitKey(1) & 0xFF + if key in (27, ord("q"), ord("Q")): + break + continue + + if last_frames["disparity"] is not None: + frame_times.append(time.time()) + + if len(frame_times) >= 2: + fps = (len(frame_times) - 1) / max(1e-6, frame_times[-1] - frame_times[0]) + else: + fps = 0.0 + + left = last_frames["left"] + right = last_frames["right"] + rect_left = last_frames["rectified_left"] + rect_right = last_frames["rectified_right"] + disp = last_frames["disparity"] + depth = last_frames["depth"] + rgb = last_frames["rgb"] + + if disp is None or depth is None or left is None or right is None: + continue + + metrics = safe_stats_depth_mm(depth, args.min_depth_mm, args.max_depth_mm) + disp_s = stats_disparity(disp) + last_metrics = metrics + last_disp_stats = disp_s + + depth_f = depth.astype(np.float32) + depth_valid = np.where( + (depth_f > args.min_depth_mm) & (depth_f < args.max_depth_mm), + depth_f, + np.nan, + ) + + disp_hm = heatmap(disp, 1, 99, cv2.COLORMAP_TURBO) + depth_hm = heatmap(depth_valid, 1, 99, cv2.COLORMAP_TURBO) + valid_mask = np.where(np.isfinite(depth_valid), 255, 0).astype(np.uint8) + valid_bgr = cv2.cvtColor(valid_mask, cv2.COLOR_GRAY2BGR) + + base_for_overlay = rect_left if rect_left is not None else left + base_bgr = cv2.cvtColor(normalize_u8(base_for_overlay), cv2.COLOR_GRAY2BGR) + depth_hm_res = cv2.resize(depth_hm, (base_bgr.shape[1], base_bgr.shape[0]), interpolation=cv2.INTER_AREA) + overlay = cv2.addWeighted(base_bgr, 0.55, depth_hm_res, 0.45, 0) + + panels = [ + ("Left mono", normalize_u8(left), f"{args.left}"), + ("Right mono", normalize_u8(right), f"{args.right}"), + ("Metrics", draw_metrics_panel(metrics, disp_s, fps, args), ""), + ("Rectified left", normalize_u8(rect_left), "stereo.rectifiedLeft"), + ("Rectified right", normalize_u8(rect_right), "stereo.rectifiedRight"), + ("Disparity heatmap", disp_hm, f"valid={disp_s['valid_pct']:.1f}%"), + ("Depth heatmap", depth_hm, f"valid={metrics['valid_pct']:.1f}% median={metrics['median_mm']:.0f}mm"), + ("Valid depth mask", valid_bgr, f"range={args.min_depth_mm}-{args.max_depth_mm}mm"), + ("Depth overlay", overlay, "heatmap sobre rectified left"), + ] + + if rgb is not None: + panels.append(("RGB preview", rgb, f"{args.rgb}")) + + canvas = make_grid(panels, panel_w=args.panel_w, cols=3) + last_frames["canvas"] = canvas + cv2.imshow("OAK-FCC-3P Depth Probe", canvas) + + key = cv2.waitKey(1) & 0xFF + if key in (27, ord("q"), ord("Q")): + break + if key in (ord("s"), ord("S")): + frames_to_save = {k: v for k, v in last_frames.items() if v is not None} + save_snapshot(out_dir, frames_to_save, last_metrics, last_disp_stats, args) + if key in (ord("h"), ord("H")): + print("\n=== HELP ===") + print("Q/ESC : sair") + print("S : salvar snapshot") + print("Teste tambem invertendo --left/--right se disparity/depth parecer quebrado.") + print("===========\n") + + finally: + stop_pipeline_v3(pipeline) + cv2.destroyAllWindows() + + +# ============================================================ +# CLI +# ============================================================ + + +def build_argparser() -> argparse.ArgumentParser: + ap = argparse.ArgumentParser(description="Teste de depth/disparity na OAK-FFC-3P usando par mono RE/NIR, sem XLinkOut.") + + ap.add_argument("--left", type=str, default="CAM_B", help="Socket mono esquerda. Ex: CAM_B ou CAM_C") + ap.add_argument("--right", type=str, default="CAM_C", help="Socket mono direita. Ex: CAM_C ou CAM_B") + ap.add_argument("--rgb", type=str, default="CAM_A", help="Socket RGB opcional.") + ap.add_argument("--enable-rgb", action="store_true", help="Tambem mostra preview RGB.") + + ap.add_argument("--mono-resolution", type=str, default="800p", choices=["400p", "480p", "720p", "800p"]) + ap.add_argument("--fps", type=float, default=10.0) + ap.add_argument("--rgb-preview-w", type=int, default=640) + ap.add_argument("--rgb-preview-h", type=int, default=400) + + ap.add_argument("--lrcheck", action="store_true", help="Ativa left-right check para remover matches ruins/oclusoes.") + ap.add_argument("--extended", action="store_true", help="Ativa extended disparity, util para curto alcance.") + ap.add_argument("--subpixel", action="store_true", help="Ativa subpixel disparity, util para suavidade/maior precisao.") + ap.add_argument("--confidence", type=int, default=200, help="Confidence threshold do StereoDepth. Tente 180-245.") + ap.add_argument("--median", type=str, default="7", choices=["OFF", "3", "5", "7"], help="Filtro de mediana.") + + ap.add_argument("--speckle", action="store_true", help="Ativa speckle filter no post-processing.") + ap.add_argument("--speckle-range", type=int, default=50) + ap.add_argument("--temporal", action="store_true", help="Ativa temporal filter, se suportado pela versao.") + ap.add_argument("--spatial", action="store_true", help="Ativa spatial filter, se suportado pela versao.") + ap.add_argument("--hole-filling-radius", type=int, default=2) + ap.add_argument("--spatial-iterations", type=int, default=1) + + ap.add_argument("--min-depth-mm", type=int, default=150) + ap.add_argument("--max-depth-mm", type=int, default=5000) + ap.add_argument("--panel-w", type=int, default=430) + ap.add_argument("--out-dir", type=str, default="depth_probe_out") + + return ap + + +if __name__ == "__main__": + main(build_argparser().parse_args()) diff --git a/Python/OAK/datasets/oak-fcc-3/utils/depth_probe_calibration.py b/Python/OAK/datasets/oak-fcc-3/utils/depth_probe_calibration.py new file mode 100644 index 000000000..12e746664 --- /dev/null +++ b/Python/OAK/datasets/oak-fcc-3/utils/depth_probe_calibration.py @@ -0,0 +1,546 @@ +import argparse +import json +import math +import time +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +import numpy as np + +try: + import depthai as dai +except Exception as e: + raise RuntimeError( + "Nao consegui importar depthai. Ative o venv correto e instale depthai antes de rodar. " + f"Erro original: {e}" + ) + + +# ============================================================ +# OAK-FCC-3P Calibration Probe +# ------------------------------------------------------------ +# Objetivo: +# Ler o que existe de calibracao no device OAK/DepthAI: +# - cameras conectadas +# - sockets / sensores +# - intrinsecos por camera, quando disponivel +# - distorcao por camera, quando disponivel +# - extrinsecos entre pares CAM_A/CAM_B/CAM_C +# - baseline estimado, quando a API permitir +# - dump JSON bruto da calibracao, quando disponivel +# +# Uso: +# python -m utils.calibration_probe --out_dir calibration_probe_out +# +# Para escolher device por MXID: +# python -m utils.calibration_probe --mx_id 194430108133AC2F00 +# ============================================================ + + +# ============================================================ +# Helpers gerais +# ============================================================ + + +def to_jsonable(x: Any): + if x is None: + return None + if isinstance(x, (str, int, float, bool)): + if isinstance(x, float) and (math.isnan(x) or math.isinf(x)): + return None + return x + if isinstance(x, np.ndarray): + return x.tolist() + if isinstance(x, (list, tuple)): + return [to_jsonable(v) for v in x] + if isinstance(x, dict): + return {str(k): to_jsonable(v) for k, v in x.items()} + try: + return str(x) + except Exception: + return repr(x) + + + +def safe_call(label: str, fn, *args, default=None, verbose: bool = False): + try: + return fn(*args) + except Exception as e: + if verbose: + print(f"[WARN] {label} falhou: {type(e).__name__}: {e}") + return default + + + +def get_device_id_from_info(dev_info) -> Optional[str]: + for name in ("getMxId", "getDeviceId"): + try: + fn = getattr(dev_info, name, None) + if callable(fn): + value = fn() + if value: + return str(value) + except Exception: + pass + + for attr in ("mxid", "deviceId", "name"): + try: + value = getattr(dev_info, attr, None) + if value: + return str(value) + except Exception: + pass + + return None + + + +def resolve_device_info(mx_id: Optional[str] = None): + devices = dai.Device.getAllAvailableDevices() + if not devices: + raise RuntimeError("Nenhum dispositivo DepthAI/OAK encontrado.") + + if not mx_id: + return devices[0] + + target = str(mx_id).strip() + for dev_info in devices: + dev_id = get_device_id_from_info(dev_info) + if dev_id == target: + return dev_info + + available = [get_device_id_from_info(d) or str(d) for d in devices] + raise RuntimeError(f"Device mx_id='{target}' nao encontrado. Disponiveis={available}") + + + +def socket_from_name(name: str): + name = str(name).strip().upper() + aliases = { + "A": "CAM_A", + "B": "CAM_B", + "C": "CAM_C", + "D": "CAM_D", + "RGB": "CAM_A", + "LEFT": "CAM_B", + "RIGHT": "CAM_C", + } + name = aliases.get(name, name) + + if hasattr(dai.CameraBoardSocket, name): + return getattr(dai.CameraBoardSocket, name) + + legacy = { + "CAM_A": getattr(dai.CameraBoardSocket, "RGB", None), + "CAM_B": getattr(dai.CameraBoardSocket, "LEFT", None), + "CAM_C": getattr(dai.CameraBoardSocket, "RIGHT", None), + "CAM_D": getattr(dai.CameraBoardSocket, "CAM_D", None), + } + if legacy.get(name) is not None: + return legacy[name] + + raise ValueError(f"Socket invalido: {name}") + + + +def socket_name(socket_obj) -> str: + try: + return str(socket_obj.name) + except Exception: + return str(socket_obj) + + + +def matrix_shape_ok(m, rows: int, cols: int) -> bool: + try: + arr = np.asarray(m, dtype=np.float64) + return arr.shape == (rows, cols) and np.all(np.isfinite(arr)) + except Exception: + return False + + + +def flatten_matrix(m): + try: + return np.asarray(m, dtype=np.float64).tolist() + except Exception: + return to_jsonable(m) + + +# ============================================================ +# Calibration read helpers +# ============================================================ + + +def read_calibration(device, verbose: bool = False): + # Contratos comuns: readCalibration(), readCalibration2(). + for method in ("readCalibration", "readCalibration2"): + fn = getattr(device, method, None) + if callable(fn): + calib = safe_call(method, fn, default=None, verbose=verbose) + if calib is not None: + print(f"[OK] Calibracao lida via device.{method}()") + return calib, method + + raise RuntimeError("Nao encontrei device.readCalibration/readCalibration2 nesta versao do DepthAI.") + + + +def dump_calibration_json(calib, out_dir: Path, verbose: bool = False) -> Dict[str, Any]: + """Tenta extrair dump bruto da calibracao por varios contratos de API.""" + result = { + "available": False, + "method": None, + "path": None, + "data": None, + "error": None, + } + + # 1) eepromToJson() costuma devolver dict/json. + for method in ("eepromToJson", "toJson"): + fn = getattr(calib, method, None) + if callable(fn): + try: + data = fn() + if isinstance(data, str): + try: + data_obj = json.loads(data) + except Exception: + data_obj = data + else: + data_obj = data + path = out_dir / f"calibration_{method}.json" + with open(path, "w", encoding="utf-8") as f: + json.dump(to_jsonable(data_obj), f, ensure_ascii=False, indent=2) + result.update({"available": True, "method": method, "path": str(path), "data": to_jsonable(data_obj)}) + print(f"[OK] Dump bruto salvo via calib.{method}(): {path}") + return result + except Exception as e: + result["error"] = f"{method}: {type(e).__name__}: {e}" + if verbose: + print(f"[WARN] dump {method} falhou: {e}") + + # 2) Alguns handlers escrevem direto em arquivo. + for method in ("saveToJsonFile", "saveCalibrationFile", "saveToFile"): + fn = getattr(calib, method, None) + if callable(fn): + path = out_dir / f"calibration_{method}.json" + try: + fn(str(path)) + result.update({"available": True, "method": method, "path": str(path), "data": None}) + print(f"[OK] Dump bruto salvo via calib.{method}(): {path}") + return result + except Exception as e: + result["error"] = f"{method}: {type(e).__name__}: {e}" + if verbose: + print(f"[WARN] dump {method} falhou: {e}") + + print("[WARN] Nao consegui gerar dump JSON bruto da calibracao por API conhecida.") + return result + + + +def get_connected_cameras(device) -> List[Dict[str, Any]]: + features = safe_call("getConnectedCameraFeatures", device.getConnectedCameraFeatures, default=[], verbose=False) + out = [] + for f in features: + item = {} + try: + item["socket"] = socket_name(f.socket) + except Exception: + item["socket"] = None + for attr in ("sensorName", "width", "height", "orientation", "supportedTypes"): + try: + v = getattr(f, attr, None) + item[attr] = to_jsonable(v) + except Exception: + pass + out.append(item) + return out + + + +def get_intrinsics(calib, socket, width: int, height: int, verbose: bool = False): + # Contratos comuns: + # getCameraIntrinsics(socket) + # getCameraIntrinsics(socket, width, height) + fn = getattr(calib, "getCameraIntrinsics", None) + if not callable(fn): + return None, "missing:getCameraIntrinsics" + + for args in ((socket, width, height), (socket,)): + try: + value = fn(*args) + if value is not None: + return flatten_matrix(value), f"getCameraIntrinsics{len(args)}args" + except Exception as e: + if verbose: + print(f"[WARN] intrinsics {socket_name(socket)} args={len(args)} falhou: {e}") + return None, "failed:getCameraIntrinsics" + + + +def get_distortion(calib, socket, verbose: bool = False): + for method in ("getDistortionCoefficients", "getDistortionCoeff"): + fn = getattr(calib, method, None) + if callable(fn): + try: + value = fn(socket) + return to_jsonable(value), method + except Exception as e: + if verbose: + print(f"[WARN] distortion {socket_name(socket)} {method} falhou: {e}") + return None, "missing:distortion" + + + +def get_fov(calib, socket, verbose: bool = False): + fn = getattr(calib, "getFov", None) + if callable(fn): + try: + return float(fn(socket)), "getFov" + except Exception as e: + if verbose: + print(f"[WARN] fov {socket_name(socket)} falhou: {e}") + return None, "missing:getFov" + + + +def get_extrinsics(calib, src_socket, dst_socket, verbose: bool = False): + fn = getattr(calib, "getCameraExtrinsics", None) + if not callable(fn): + return None, "missing:getCameraExtrinsics" + + # Contratos comuns: + # getCameraExtrinsics(src, dst) + # getCameraExtrinsics(src, dst, useSpecTranslation) + for args in ((src_socket, dst_socket), (src_socket, dst_socket, False), (src_socket, dst_socket, True)): + try: + value = fn(*args) + if value is not None: + return flatten_matrix(value), f"getCameraExtrinsics{len(args)}args" + except Exception as e: + if verbose: + print(f"[WARN] extrinsics {socket_name(src_socket)}->{socket_name(dst_socket)} args={len(args)} falhou: {e}") + return None, "failed:getCameraExtrinsics" + + + +def get_baseline(calib, src_socket, dst_socket, verbose: bool = False): + # Varia entre versoes; em algumas, getBaselineDistance(cam1, cam2, useSpecTranslation) + fn = getattr(calib, "getBaselineDistance", None) + if callable(fn): + for args in ((src_socket, dst_socket), (src_socket, dst_socket, False), (src_socket, dst_socket, True)): + try: + value = fn(*args) + if value is not None: + return float(value), f"getBaselineDistance{len(args)}args" + except Exception as e: + if verbose: + print(f"[WARN] baseline {socket_name(src_socket)}-{socket_name(dst_socket)} args={len(args)} falhou: {e}") + + # Fallback: calcula norma da translacao da matriz 4x4, se existir. + ext, method = get_extrinsics(calib, src_socket, dst_socket, verbose=False) + if ext is not None: + try: + arr = np.asarray(ext, dtype=np.float64) + if arr.shape == (4, 4): + t = arr[:3, 3] + return float(np.linalg.norm(t)), f"norm_translation_from_{method}" + except Exception: + pass + return None, "missing:getBaselineDistance" + + + +def inspect_socket(calib, socket_name_str: str, width: int, height: int, verbose: bool = False) -> Dict[str, Any]: + socket = socket_from_name(socket_name_str) + intr, intr_method = get_intrinsics(calib, socket, width, height, verbose=verbose) + dist, dist_method = get_distortion(calib, socket, verbose=verbose) + fov, fov_method = get_fov(calib, socket, verbose=verbose) + + intr_ok = matrix_shape_ok(intr, 3, 3) + dist_ok = dist is not None + + return { + "socket": socket_name_str, + "intrinsics": intr, + "intrinsics_method": intr_method, + "intrinsics_ok": bool(intr_ok), + "distortion": dist, + "distortion_method": dist_method, + "distortion_ok": bool(dist_ok), + "fov_deg": fov, + "fov_method": fov_method, + } + + + +def inspect_pair(calib, src_name: str, dst_name: str, verbose: bool = False) -> Dict[str, Any]: + src = socket_from_name(src_name) + dst = socket_from_name(dst_name) + ext, ext_method = get_extrinsics(calib, src, dst, verbose=verbose) + base, base_method = get_baseline(calib, src, dst, verbose=verbose) + + ext_ok = matrix_shape_ok(ext, 4, 4) + translation = None + if ext_ok: + try: + arr = np.asarray(ext, dtype=np.float64) + translation = arr[:3, 3].tolist() + except Exception: + translation = None + + return { + "pair": f"{src_name}->{dst_name}", + "src": src_name, + "dst": dst_name, + "extrinsics": ext, + "extrinsics_method": ext_method, + "extrinsics_ok": bool(ext_ok), + "translation": translation, + "baseline": base, + "baseline_method": base_method, + "baseline_ok": bool(base is not None), + } + + +# ============================================================ +# Report +# ============================================================ + + +def print_summary(report: Dict[str, Any]): + print("\n================ CALIBRATION PROBE SUMMARY ================") + print(f"device_id : {report.get('device_id')}") + print(f"read_method : {report.get('calibration_read_method')}") + print(f"dump_json : {report.get('calibration_dump', {}).get('path')}") + + print("\n[Cameras conectadas]") + for cam in report.get("connected_cameras", []): + print(f" - socket={cam.get('socket')} sensor={cam.get('sensorName')} size={cam.get('width')}x{cam.get('height')}") + + print("\n[Intrinsecos por socket]") + for s in report.get("sockets", []): + print( + f" - {s['socket']}: intrinsics_ok={s['intrinsics_ok']} " + f"distortion_ok={s['distortion_ok']} fov={s.get('fov_deg')}" + ) + + print("\n[Extrinsecos entre pares]") + for p in report.get("pairs", []): + flag = "OK" if p.get("extrinsics_ok") else "MISSING" + print( + f" - {p['pair']}: {flag} | baseline={p.get('baseline')} " + f"| method={p.get('extrinsics_method')}" + ) + + # Diagnostico direto para o caso de depth RE/NIR. + bc = next((p for p in report.get("pairs", []) if p.get("pair") == "CAM_B->CAM_C"), None) + cb = next((p for p in report.get("pairs", []) if p.get("pair") == "CAM_C->CAM_B"), None) + + print("\n[Diagnostico CAM_B/CAM_C para StereoDepth]") + if (bc and bc.get("extrinsics_ok")) or (cb and cb.get("extrinsics_ok")): + print(" ✅ Existe extrinseco entre CAM_B e CAM_C. O StereoDepth deve ter chance de iniciar.") + else: + print(" ❌ Nao existe extrinseco CAM_B<->CAM_C legivel pela API.") + print(" Isso explica erro: 'There is no available extrinsic calibration between camera ID: 1 and 2'.") + print(" Proximo passo: calibrar o par CAM_B/CAM_C ou carregar um calibration.json valido.") + + print("===========================================================\n") + + +# ============================================================ +# Main +# ============================================================ + + +def main(args: argparse.Namespace): + out_dir = Path(args.out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + + dev_info = resolve_device_info(args.mx_id) + device_id = get_device_id_from_info(dev_info) + + print(f"[INFO] Abrindo device: {device_id}") + + with dai.Device(dev_info) as device: + connected = get_connected_cameras(device) + calib, read_method = read_calibration(device, verbose=args.verbose) + dump = dump_calibration_json(calib, out_dir=out_dir, verbose=args.verbose) + + sockets = [s.strip().upper() for s in args.sockets.split(",") if s.strip()] + pairs = [] + socket_reports = [] + + for s in sockets: + try: + socket_reports.append(inspect_socket(calib, s, args.width, args.height, verbose=args.verbose)) + except Exception as e: + socket_reports.append({ + "socket": s, + "error": f"{type(e).__name__}: {e}", + "intrinsics_ok": False, + "distortion_ok": False, + }) + + for src in sockets: + for dst in sockets: + if src == dst: + continue + try: + pairs.append(inspect_pair(calib, src, dst, verbose=args.verbose)) + except Exception as e: + pairs.append({ + "pair": f"{src}->{dst}", + "src": src, + "dst": dst, + "error": f"{type(e).__name__}: {e}", + "extrinsics_ok": False, + "baseline_ok": False, + }) + + report = { + "created_at": time.strftime("%Y-%m-%d %H:%M:%S"), + "device_id": device_id, + "calibration_read_method": read_method, + "connected_cameras": connected, + "sockets_requested": sockets, + "width": int(args.width), + "height": int(args.height), + "calibration_dump": { + "available": dump.get("available"), + "method": dump.get("method"), + "path": dump.get("path"), + "error": dump.get("error"), + }, + "sockets": socket_reports, + "pairs": pairs, + } + + out_path = out_dir / "calibration_probe_report.json" + with open(out_path, "w", encoding="utf-8") as f: + json.dump(to_jsonable(report), f, ensure_ascii=False, indent=2) + + print(f"[OK] Relatorio salvo em: {out_path}") + print_summary(report) + + +# ============================================================ +# CLI +# ============================================================ + + +def build_argparser() -> argparse.ArgumentParser: + ap = argparse.ArgumentParser(description="Inspeciona calibracao EEPROM/JSON do device OAK/DepthAI.") + ap.add_argument("--mx_id", type=str, default=None, help="MXID opcional do device.") + ap.add_argument("--out_dir", type=str, default="calibration_probe_out", help="Pasta de saida.") + ap.add_argument("--sockets", type=str, default="CAM_A,CAM_B,CAM_C", help="Sockets para testar. Ex: CAM_A,CAM_B,CAM_C") + ap.add_argument("--width", type=int, default=1280, help="Largura usada ao pedir intrinsecos escalados.") + ap.add_argument("--height", type=int, default=800, help="Altura usada ao pedir intrinsecos escalados.") + ap.add_argument("--verbose", action="store_true", help="Mostra warnings detalhados de APIs que falharam.") + return ap + + +if __name__ == "__main__": + main(build_argparser().parse_args()) diff --git a/Python/OAK/datasets/oak-fcc-3/utils/manual_fusion_calibrator.py b/Python/OAK/datasets/oak-fcc-3/utils/manual_fusion_calibrator.py index 10bb99124..734c084ab 100644 --- a/Python/OAK/datasets/oak-fcc-3/utils/manual_fusion_calibrator.py +++ b/Python/OAK/datasets/oak-fcc-3/utils/manual_fusion_calibrator.py @@ -10,6 +10,10 @@ import numpy as np from core.oak_fcc3_client import OakFcc3Client as MultiSpectralClient +# ============================================================ +# Utilidades gerais +# ============================================================ + def now_str() -> str: return datetime.now().strftime("%Y-%m-%d %H:%M:%S") @@ -100,7 +104,7 @@ def build_overlay_fuse( if spec01 is None: return base_bgr - if calibration_mode == "homography": + if calibration_mode in ("homography", "charuco_auto"): warped = apply_homography(spec01, H) else: warped = apply_affine(spec01, dx, dy, theta_deg) @@ -166,9 +170,13 @@ def validate_module_ready(status, frame_type, raw_policy): raise RuntimeError(f"frame_type desconhecido para validação: {frame_type}") +# ============================================================ +# JSON de calibração +# ============================================================ + def default_offsets_payload(args, effective_capture_mode): return { - "schema": "manual_multispec_offsets_v2", + "schema": "manual_multispec_offsets_v3", "saved_at": now_str(), "frame_type": "RAW_BRUTO", "capture_mode_requested": args.capture_mode, @@ -188,6 +196,8 @@ def default_offsets_payload(args, effective_capture_mode): "re_to_rgb": None, "nir_to_rgb": None, }, + "homography_metrics": {}, + "charuco": {}, "notes": args.notes or "", } @@ -199,12 +209,14 @@ def load_offsets_json(path, args, effective_capture_mode): with open(path, "r", encoding="utf-8") as f: data = json.load(f) - data.setdefault("schema", "manual_multispec_offsets_v2") + data.setdefault("schema", "manual_multispec_offsets_v3") data.setdefault("reference_camera", "rgb") data.setdefault("baseline_mm", args.baseline_mm) data.setdefault("alignment_mode", "manual_affine") data.setdefault("manual_offsets", {}) data.setdefault("homographies", {}) + data.setdefault("homography_metrics", {}) + data.setdefault("charuco", {}) data["manual_offsets"].setdefault("re", {"dx": 0, "dy": 0, "theta_deg": 0.0}) data["manual_offsets"].setdefault("nir", {"dx": 0, "dy": 0, "theta_deg": 0.0}) @@ -224,9 +236,216 @@ def save_offsets_json(path, data): json.dump(data, f, ensure_ascii=False, indent=2) +# ============================================================ +# ChArUco automático para homografia planar +# ============================================================ + +def require_aruco(): + if not hasattr(cv2, "aruco"): + raise RuntimeError("cv2.aruco não disponível. Instale opencv-contrib-python no ambiente.") + + +def get_aruco_dictionary(dict_name: str): + require_aruco() + if not hasattr(cv2.aruco, dict_name): + available = sorted([x for x in dir(cv2.aruco) if x.startswith("DICT_")]) + raise ValueError(f"Dicionário ArUco inválido: {dict_name}. Disponíveis: {available}") + + dict_id = getattr(cv2.aruco, dict_name) + if hasattr(cv2.aruco, "getPredefinedDictionary"): + return cv2.aruco.getPredefinedDictionary(dict_id) + return cv2.aruco.Dictionary_get(dict_id) + + +def create_charuco_board(squares_x, squares_y, square_length, marker_length, dictionary): + require_aruco() + # OpenCV novo: cv2.aruco.CharucoBoard((x, y), squareLength, markerLength, dictionary) + if hasattr(cv2.aruco, "CharucoBoard"): + try: + return cv2.aruco.CharucoBoard((squares_x, squares_y), square_length, marker_length, dictionary) + except TypeError: + pass + + # OpenCV legado: cv2.aruco.CharucoBoard_create(x, y, squareLength, markerLength, dictionary) + if hasattr(cv2.aruco, "CharucoBoard_create"): + return cv2.aruco.CharucoBoard_create(squares_x, squares_y, square_length, marker_length, dictionary) + + raise RuntimeError("API ChArUco não encontrada no cv2.aruco deste ambiente.") + + +def create_detector_params(): + require_aruco() + if hasattr(cv2.aruco, "DetectorParameters"): + return cv2.aruco.DetectorParameters() + return cv2.aruco.DetectorParameters_create() + + +def to_gray_u8_for_charuco(img01, equalize=True, invert=False): + if img01 is None: + return None + + img = np.asarray(img01) + + if img.ndim == 3: + if img.shape[2] == 3: + gray = cv2.cvtColor(np.clip(img * 255.0, 0, 255).astype(np.uint8), cv2.COLOR_RGB2GRAY) + else: + gray = img[..., 0] + else: + gray = img + + if gray.dtype != np.uint8: + gray = np.asarray(gray, dtype=np.float32) + finite = np.isfinite(gray) + if not finite.any(): + return np.zeros(gray.shape[:2], dtype=np.uint8) + + lo = float(np.percentile(gray[finite], 1.0)) + hi = float(np.percentile(gray[finite], 99.5)) + if hi <= lo + 1e-9: + hi = lo + 1.0 + gray = np.clip((gray - lo) / (hi - lo) * 255.0, 0, 255).astype(np.uint8) + + if invert: + gray = 255 - gray + + if equalize: + try: + clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8)) + gray = clahe.apply(gray) + except Exception: + gray = cv2.equalizeHist(gray) + + return gray + + +def detect_charuco_points( + img01, + board, + dictionary, + detector_params, + equalize=True, + invert=False, + min_markers=4, +): + """ + Retorna dict: charuco_id -> (x, y), além de resumo de detecção. + Compatível com APIs nova/legada do OpenCV. + """ + gray = to_gray_u8_for_charuco(img01, equalize=equalize, invert=invert) + if gray is None: + return {}, {"markers": 0, "corners": 0, "ok": False} + + # API nova pode ter ArucoDetector, mas detectMarkers continua existindo em quase todos. + corners, ids, rejected = cv2.aruco.detectMarkers(gray, dictionary, parameters=detector_params) + + n_markers = 0 if ids is None else int(len(ids)) + if ids is None or n_markers < min_markers: + return {}, {"markers": n_markers, "corners": 0, "ok": False} + + try: + cv2.aruco.refineDetectedMarkers(gray, board, corners, ids, rejected) + except Exception: + pass + + ret, charuco_corners, charuco_ids = cv2.aruco.interpolateCornersCharuco( + markerCorners=corners, + markerIds=ids, + image=gray, + board=board, + ) + + if charuco_corners is None or charuco_ids is None: + return {}, {"markers": n_markers, "corners": 0, "ok": False} + + point_by_id = {} + ids_flat = charuco_ids.reshape(-1) + pts = charuco_corners.reshape(-1, 2) + + for cid, pt in zip(ids_flat, pts): + point_by_id[int(cid)] = (float(pt[0]), float(pt[1])) + + return point_by_id, { + "markers": n_markers, + "corners": int(len(point_by_id)), + "ok": len(point_by_id) >= 4, + } + + +def append_charuco_pairs(accum, role, rgb_points, spec_points, sample_id): + common_ids = sorted(set(rgb_points.keys()) & set(spec_points.keys())) + added = 0 + + for cid in common_ids: + spec_pt = spec_points[cid] + rgb_pt = rgb_points[cid] + accum[role]["spec"].append(spec_pt) + accum[role]["rgb"].append(rgb_pt) + accum[role]["ids"].append(int(cid)) + accum[role]["sample_ids"].append(int(sample_id)) + added += 1 + + return added, common_ids + + +def compute_homography_from_points(src_pts, dst_pts, ransac_reproj_threshold=3.0): + if len(src_pts) < 4 or len(dst_pts) < 4 or len(src_pts) != len(dst_pts): + return None, None, None + + src = np.array(src_pts, dtype=np.float32) + dst = np.array(dst_pts, dtype=np.float32) + + H, status = cv2.findHomography(src, dst, method=cv2.RANSAC, ransacReprojThreshold=float(ransac_reproj_threshold)) + if H is None: + return None, status, None + + projected = cv2.perspectiveTransform(src.reshape(-1, 1, 2), H).reshape(-1, 2) + err = np.linalg.norm(projected - dst, axis=1) + + if status is not None: + inlier_mask = status.reshape(-1).astype(bool) + else: + inlier_mask = np.ones(len(err), dtype=bool) + + if inlier_mask.any(): + err_in = err[inlier_mask] + else: + err_in = err + + metrics = { + "points": int(len(src_pts)), + "inliers": int(inlier_mask.sum()), + "outliers": int(len(src_pts) - inlier_mask.sum()), + "inlier_ratio": float(inlier_mask.sum() / max(len(src_pts), 1)), + "mean_error_px": float(np.mean(err_in)) if len(err_in) else None, + "median_error_px": float(np.median(err_in)) if len(err_in) else None, + "max_error_px": float(np.max(err_in)) if len(err_in) else None, + "ransac_reproj_threshold_px": float(ransac_reproj_threshold), + } + + return H, status, metrics + + +def draw_charuco_points_on_panel(board_img, rect, point_by_id, color, max_labels=80): + if rect is None or not point_by_id: + return + + x0, y0, _, _ = rect + for idx, (cid, pt) in enumerate(point_by_id.items()): + px = int(x0 + pt[0]) + py = int(y0 + pt[1]) + cv2.circle(board_img, (px, py), 4, color, -1) + if idx < max_labels: + cv2.putText(board_img, str(cid), (px + 5, py - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.38, color, 1, cv2.LINE_AA) + + +# ============================================================ +# Main +# ============================================================ + def main(): parser = argparse.ArgumentParser( - description="Calibrador manual de offsets para fusão RGB/RE/NIR a partir do stream RAW_BRUTO.", + description="Calibrador de offsets/homografia para fusão RGB/RE/NIR a partir do stream RAW_BRUTO.", formatter_class=argparse.ArgumentDefaultsHelpFormatter, ) @@ -244,6 +463,24 @@ def main(): parser.add_argument("--out_json", default="calibration/manual_offsets.json") parser.add_argument("--load_json", default="") parser.add_argument("--notes", default="") + parser.add_argument("--module_calibration_json", default="calibration/module_params.json") + + # ChArUco automático + parser.add_argument("--charuco_auto", action="store_true", help="Inicia direto no modo de homografia automática por ChArUco.") + parser.add_argument("--charuco_dictionary", default="DICT_5X5_100") + parser.add_argument("--charuco_squares_x", type=int, default=7) + parser.add_argument("--charuco_squares_y", type=int, default=5) + parser.add_argument("--charuco_square_length", type=float, default=1.0) + parser.add_argument("--charuco_marker_length", type=float, default=0.70) + parser.add_argument("--charuco_min_markers", type=int, default=4) + parser.add_argument("--charuco_min_common_corners", type=int, default=8) + parser.add_argument("--charuco_ransac_px", type=float, default=3.0) + parser.add_argument("--charuco_equalize", action="store_true", default=True) + parser.add_argument("--charuco_no_equalize", action="store_false", dest="charuco_equalize") + parser.add_argument("--charuco_invert_rgb", action="store_true") + parser.add_argument("--charuco_invert_re", action="store_true") + parser.add_argument("--charuco_invert_nir", action="store_true") + parser.add_argument("--charuco_autocalc", action="store_true", help="Calcula H automaticamente após cada captura ChArUco válida.") args = parser.parse_args() @@ -253,11 +490,38 @@ def main(): offsets = offsets_data["manual_offsets"] selected_role = "re" - calibration_mode = offsets_data.get("alignment_mode", "manual_affine") + calibration_mode = "charuco_auto" if args.charuco_auto else offsets_data.get("alignment_mode", "manual_affine") selected_points_spec = {"re": [], "nir": []} selected_points_rgb = {"re": [], "nir": []} + charuco_accum = { + "re": {"spec": [], "rgb": [], "ids": [], "sample_ids": []}, + "nir": {"spec": [], "rgb": [], "ids": [], "sample_ids": []}, + } + charuco_last = { + "rgb": {}, + "re": {}, + "nir": {}, + "summary": {}, + } + charuco_sample_id = 0 + + dictionary = None + board_charuco = None + detector_params = None + + if args.charuco_auto: + dictionary = get_aruco_dictionary(args.charuco_dictionary) + board_charuco = create_charuco_board( + args.charuco_squares_x, + args.charuco_squares_y, + args.charuco_square_length, + args.charuco_marker_length, + dictionary, + ) + detector_params = create_detector_params() + panel_rects = { "fuse": None, "rgb": None, @@ -278,7 +542,7 @@ def main(): stream_frames_accum = 0 last_stream_frame_id = None - window_name = "Manual Fusion Calibrator" + window_name = "Fusion Calibrator - Manual/Homography/ChArUco" def inside(rect, px, py): if rect is None: @@ -290,6 +554,21 @@ def main(): x0, y0, _, _ = rect return float(px - x0), float(py - y0) + def ensure_charuco_runtime(): + nonlocal dictionary, board_charuco, detector_params + if dictionary is None: + dictionary = get_aruco_dictionary(args.charuco_dictionary) + if board_charuco is None: + board_charuco = create_charuco_board( + args.charuco_squares_x, + args.charuco_squares_y, + args.charuco_square_length, + args.charuco_marker_length, + dictionary, + ) + if detector_params is None: + detector_params = create_detector_params() + def on_mouse(event, x, y, flags, param): nonlocal last_msg, last_msg_t, selected_role, calibration_mode @@ -329,6 +608,108 @@ def main(): return None + def calculate_and_store_h_for_role(role): + pts = charuco_accum[role] + if len(pts["spec"]) < 4 or len(pts["rgb"]) < 4: + return False, f"{role.upper()}: pontos acumulados insuficientes" + + H, status, metrics = compute_homography_from_points( + pts["spec"], + pts["rgb"], + ransac_reproj_threshold=args.charuco_ransac_px, + ) + + if H is None: + return False, f"{role.upper()}: falha ao calcular H" + + offsets_data.setdefault("homographies", {}) + offsets_data.setdefault("homography_metrics", {}) + offsets_data["homographies"][f"{role}_to_rgb"] = H.tolist() + offsets_data["homography_metrics"][f"{role}_to_rgb"] = metrics + + return True, ( + f"{role.upper()}: H OK | pts={metrics['points']} | " + f"inliers={metrics['inliers']} | err_med={metrics['mean_error_px']:.2f}px" + ) + + def calculate_charuco_homographies(all_roles=False): + roles = ("re", "nir") if all_roles else (selected_role,) + messages = [] + ok_any = False + + for role in roles: + ok, msg = calculate_and_store_h_for_role(role) + ok_any = ok_any or ok + messages.append(msg) + + return ok_any, " | ".join(messages) + + def grab_charuco_sample(rgb01, re01, nir01): + nonlocal charuco_sample_id + ensure_charuco_runtime() + + rgb_pts, rgb_sum = detect_charuco_points( + rgb01, + board_charuco, + dictionary, + detector_params, + equalize=args.charuco_equalize, + invert=args.charuco_invert_rgb, + min_markers=args.charuco_min_markers, + ) + + re_pts, re_sum = detect_charuco_points( + re01, + board_charuco, + dictionary, + detector_params, + equalize=args.charuco_equalize, + invert=args.charuco_invert_re, + min_markers=args.charuco_min_markers, + ) if re01 is not None else ({}, {"markers": 0, "corners": 0, "ok": False}) + + nir_pts, nir_sum = detect_charuco_points( + nir01, + board_charuco, + dictionary, + detector_params, + equalize=args.charuco_equalize, + invert=args.charuco_invert_nir, + min_markers=args.charuco_min_markers, + ) if nir01 is not None else ({}, {"markers": 0, "corners": 0, "ok": False}) + + charuco_last["rgb"] = rgb_pts + charuco_last["re"] = re_pts + charuco_last["nir"] = nir_pts + charuco_last["summary"] = {"rgb": rgb_sum, "re": re_sum, "nir": nir_sum} + + if not rgb_pts: + return False, "ChArUco: RGB não detectou cantos válidos" + + charuco_sample_id += 1 + messages = [] + added_total = 0 + + for role, spec_pts in (("re", re_pts), ("nir", nir_pts)): + if not spec_pts: + messages.append(f"{role.upper()}: sem detecção") + continue + + common = sorted(set(rgb_pts.keys()) & set(spec_pts.keys())) + if len(common) < args.charuco_min_common_corners: + messages.append(f"{role.upper()}: comum={len(common)} < {args.charuco_min_common_corners}") + continue + + added, _ = append_charuco_pairs(charuco_accum, role, rgb_pts, spec_pts, charuco_sample_id) + added_total += added + messages.append(f"{role.upper()}: +{added} pares") + + if args.charuco_autocalc and added_total > 0: + _, calc_msg = calculate_charuco_homographies(all_roles=True) + messages.append(calc_msg) + + return added_total > 0, "ChArUco sample #{:03d}: {}".format(charuco_sample_id, " | ".join(messages)) + cv2.namedWindow(window_name, cv2.WINDOW_NORMAL) cv2.setMouseCallback(window_name, on_mouse) @@ -342,7 +723,7 @@ def main(): output_dtype="uint8", capture_mode=effective_capture_mode, raw_policy=args.raw_policy, - module_calibration_json="calibration/module_params.json" + module_calibration_json=args.module_calibration_json, ) as cam: validate_module_ready(cam.get_status(), "RAW_BRUTO", args.raw_policy) @@ -437,12 +818,19 @@ def main(): spec_pts = len(selected_points_spec[selected_role]) rgb_pts = len(selected_points_rgb[selected_role]) + ch_re_pts = len(charuco_accum["re"]["spec"]) + ch_nir_pts = len(charuco_accum["nir"]["spec"]) + ch_sum = charuco_last.get("summary", {}) or {} + ch_rgb = ch_sum.get("rgb", {}).get("corners", 0) + ch_re = ch_sum.get("re", {}).get("corners", 0) + ch_nir = ch_sum.get("nir", {}).get("corners", 0) lines_fuse = [ f"FUSE: RGB + {active_spec_name}", f"mode={calibration_mode} | selecionada={selected_role.upper()}", f"dx={dx} | dy={dy} | theta={theta_deg:.2f}g | step={args.step} | ang_step={args.angle_step:.2f}g", - f"pts_spec={spec_pts} | pts_rgb={rgb_pts} | min=4 | fps_stream={fps_stream:.1f} | fps_view={fps_view:.1f}", + f"manual pts_spec={spec_pts} pts_rgb={rgb_pts} | charuco RE={ch_re_pts} NIR={ch_nir_pts}", + f"last corners RGB={ch_rgb} RE={ch_re} NIR={ch_nir} | fps_stream={fps_stream:.1f} fps_view={fps_view:.1f}", ] overlay_hud(fuse_panel, lines_fuse) @@ -490,18 +878,18 @@ def main(): board = np.vstack([top, bottom]) help_lines = [ - "M=manual_affine | H=homography | clique pares | >=4 pares | SPACE=salva | C=limpa pts | Z=zera sel | X=zera tudo", - "A/W/S/D movem | J/L rotacionam | O/P ang_step | I/U remove ponto | ENTER calcula H | TAB alterna RE/NIR | Q/Esc sai", + "M=manual | H=homog clique | K=charuco | G=captura charuco | ENTER=calcula H | SPACE=salva | C=limpa pts | V=limpa charuco", + "2/3 seleciona | TAB alterna | A/W/S/D movem | J/L rotacionam | I/U remove clique | Z=zera sel | X=zera tudo | Q/Esc sai", ] overlay_hud(board, help_lines, x=16, y=board.shape[0] - 44, font_scale=0.55, line_step=20) - if last_msg and (time.time() - last_msg_t) < 2.5: + if last_msg and (time.time() - last_msg_t) < 3.5: cv2.putText( board, last_msg, (16, board.shape[0] - 72), cv2.FONT_HERSHEY_SIMPLEX, - 0.7, + 0.62, (0, 255, 0), 2, cv2.LINE_AA, @@ -529,6 +917,11 @@ def main(): cv2.circle(board, (px, py), 5, color_rgb, -1) cv2.putText(board, str(idx + 1), (px + 6, py - 6), cv2.FONT_HERSHEY_SIMPLEX, 0.5, color_rgb, 1, cv2.LINE_AA) + if calibration_mode == "charuco_auto": + draw_charuco_points_on_panel(board, panel_rects["rgb"], charuco_last.get("rgb", {}), (0, 255, 0)) + draw_charuco_points_on_panel(board, panel_rects["re"], charuco_last.get("re", {}), (0, 255, 255)) + draw_charuco_points_on_panel(board, panel_rects["nir"], charuco_last.get("nir", {}), (255, 255, 0)) + if args.preview_scale != 1.0: board = cv2.resize( board, @@ -557,34 +950,72 @@ def main(): elif k in (ord("h"), ord("H")): calibration_mode = "homography" offsets_data["alignment_mode"] = calibration_mode - last_msg = "Modo: homography" + last_msg = "Modo: homography manual por cliques" + last_msg_t = time.time() + + elif k in (ord("k"), ord("K")): + ensure_charuco_runtime() + calibration_mode = "charuco_auto" + offsets_data["alignment_mode"] = calibration_mode + last_msg = "Modo: charuco_auto" + last_msg_t = time.time() + + elif k in (ord("g"), ord("G")): + if decoded_last: + rgb_id, rgb01 = get_image_by_role(decoded_last, "rgb") + _, re01 = get_image_by_role(decoded_last, "re") + _, nir01 = get_image_by_role(decoded_last, "nir") + if rgb01 is not None: + base_h, base_w = rgb01.shape[:2] + re01 = resize_if_needed(re01, (base_h, base_w)) + nir01 = resize_if_needed(nir01, (base_h, base_w)) + ok, msg = grab_charuco_sample(rgb01, re01, nir01) + last_msg = msg + else: + last_msg = "ChArUco: RGB indisponível" + else: + last_msg = "ChArUco: sem decoded_last" last_msg_t = time.time() elif k in (ord("c"), ord("C")): selected_points_spec[selected_role] = [] selected_points_rgb[selected_role] = [] - last_msg = f"Pontos limpos: {selected_role.upper()}" + last_msg = f"Pontos manuais limpos: {selected_role.upper()}" + last_msg_t = time.time() + + elif k in (ord("v"), ord("V")): + charuco_accum[selected_role] = {"spec": [], "rgb": [], "ids": [], "sample_ids": []} + last_msg = f"Pontos ChArUco limpos: {selected_role.upper()}" last_msg_t = time.time() elif k == 13: - spec_pts = selected_points_spec[selected_role] - rgb_pts = selected_points_rgb[selected_role] - - if len(spec_pts) >= 4 and len(rgb_pts) >= 4 and len(spec_pts) == len(rgb_pts): - src = np.array(spec_pts, dtype=np.float32) - dst = np.array(rgb_pts, dtype=np.float32) - - H, status = cv2.findHomography(src, dst, method=cv2.RANSAC) - if H is not None: - offsets_data.setdefault("homographies", {}) - offsets_data["homographies"][f"{selected_role}_to_rgb"] = H.tolist() - inliers = int(status.sum()) if status is not None else len(spec_pts) - last_msg = f"H calculada para {selected_role.upper()} | pts={len(spec_pts)} | inliers={inliers}" - else: - last_msg = f"Falha ao calcular H para {selected_role.upper()}" + if calibration_mode == "charuco_auto": + ok, msg = calculate_charuco_homographies(all_roles=False) + last_msg = msg else: - last_msg = f"{selected_role.upper()}: precisa de >=4 pares e mesmo numero de pontos" + spec_pts = selected_points_spec[selected_role] + rgb_pts = selected_points_rgb[selected_role] + if len(spec_pts) >= 4 and len(rgb_pts) >= 4 and len(spec_pts) == len(rgb_pts): + H, status, metrics = compute_homography_from_points( + spec_pts, + rgb_pts, + ransac_reproj_threshold=args.charuco_ransac_px, + ) + if H is not None: + offsets_data.setdefault("homographies", {}) + offsets_data.setdefault("homography_metrics", {}) + offsets_data["homographies"][f"{selected_role}_to_rgb"] = H.tolist() + offsets_data["homography_metrics"][f"{selected_role}_to_rgb"] = metrics + last_msg = ( + f"H manual calculada para {selected_role.upper()} | " + f"pts={metrics['points']} | inliers={metrics['inliers']} | " + f"err_med={metrics['mean_error_px']:.2f}px" + ) + else: + last_msg = f"Falha ao calcular H para {selected_role.upper()}" + else: + last_msg = f"{selected_role.upper()}: precisa de >=4 pares e mesmo numero de pontos" last_msg_t = time.time() elif k == ord("2"): @@ -633,12 +1064,29 @@ def main(): offsets_data["manual_offsets"] = offsets offsets_data.setdefault("homographies", {}) - offsets_data["schema"] = "manual_multispec_offsets_v2" + offsets_data.setdefault("homography_metrics", {}) + offsets_data["schema"] = "manual_multispec_offsets_v3" offsets_data["reference_camera"] = "rgb" + offsets_data["alignment_mode"] = calibration_mode offsets_data["homography_calibration_size"] = [int(base_w), int(base_h)] + offsets_data["charuco"] = { + "enabled": calibration_mode == "charuco_auto", + "dictionary": args.charuco_dictionary, + "squares_x": int(args.charuco_squares_x), + "squares_y": int(args.charuco_squares_y), + "square_length": float(args.charuco_square_length), + "marker_length": float(args.charuco_marker_length), + "samples": int(charuco_sample_id), + "accumulated_pairs": { + "re": int(len(charuco_accum["re"]["spec"])), + "nir": int(len(charuco_accum["nir"]["spec"])), + }, + "same_physical_plane_required": True, + "notes": "Homografias ChArUco estimadas para um único plano físico. Mova o tabuleiro no plano do solo; não misture inclinações/alturas para uma H única.", + } save_offsets_json(args.out_json, offsets_data) - last_msg = f"Offsets salvos em: {args.out_json}" + last_msg = f"Calibração salva em: {args.out_json}" last_msg_t = time.time() elif k in (ord("+"), ord("=")): @@ -709,8 +1157,8 @@ def main(): finally: cv2.destroyAllWindows() - print("Fim da calibração manual.") + print("Fim da calibração de fusão.") if __name__ == "__main__": - main() \ No newline at end of file + main()