From 3935a1bb4c2e16d2484ea6df2777067d2ef9e31d Mon Sep 17 00:00:00 2001 From: Diego Freitas Date: Mon, 14 Sep 2026 07:23:40 -0300 Subject: [PATCH] ajustado radiometric_normalization e senha ntrip --- AgroBase/AgroBase/Services/GPSService.cs | 2 +- .../oak_fcc3_core/raw_processor_core.py | 1642 ++++++++++++++--- .../OperationControl/Services/GpsService.cs | 2 +- 3 files changed, 1381 insertions(+), 265 deletions(-) diff --git a/AgroBase/AgroBase/Services/GPSService.cs b/AgroBase/AgroBase/Services/GPSService.cs index 17415f667..26cb7ac4a 100644 --- a/AgroBase/AgroBase/Services/GPSService.cs +++ b/AgroBase/AgroBase/Services/GPSService.cs @@ -1857,7 +1857,7 @@ namespace AgroBase.Services // "AGRO_NTRIP_USERNAME" //) ?? string.Empty; - string password = "c3pc7*9N"; + string password = "kY3zd$*5"; //Environment.GetEnvironmentVariable( // "AGRO_NTRIP_PASSWORD" //) ?? string.Empty; diff --git a/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py b/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py index 6282d4e94..3aa9385dd 100644 --- a/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py +++ b/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py @@ -28,7 +28,7 @@ module_params is loaded. The strict fail-closed behavior is enabled only for module_params generated by the production assembler. """ -RAW_PROCESSOR_CORE_VERSION = "production_v2_2026_09_08" +RAW_PROCESSOR_CORE_VERSION = "production_v5_2026_09_13_flat_finalspace_fast" import json import os @@ -339,12 +339,21 @@ if _HAS_NUMBA: out[y, x] = value - @_numba.njit(cache=True, fastmath=True, parallel=True) - def _apply_tensor_flat_gain_chw_numba(tensor, gain, channels, height, width, clip_output): - for c in _numba.prange(channels): - for y in range(height): - for x in range(width): - v = float(tensor[c, y, x]) * float(gain[c, y, x]) + @_numba.njit(cache=True, fastmath=False, parallel=True) + def _apply_radiometric_scale_inplace_numba(img, scale, clip_output): + """ + Radiometric scale em uma unica passada de memoria. + + Mantem a mesma ordem matematica do caminho NumPy e opera apenas em + imagens 2D/3D float32 C-contiguous. + """ + h = img.shape[0] + + if img.ndim == 2: + w = img.shape[1] + for y in _numba.prange(h): + for x in range(w): + v = img[y, x] * scale if clip_output: if v < 0.0: @@ -352,7 +361,244 @@ if _HAS_NUMBA: elif v > 1.0: v = 1.0 - tensor[c, y, x] = v + img[y, x] = v + + else: + w = img.shape[1] + c = img.shape[2] + for y in _numba.prange(h): + for x in range(w): + for k in range(c): + v = img[y, x, k] * scale + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x, k] = v + + + @_numba.njit(cache=True, fastmath=False, parallel=True) + def _apply_radiometric_affine_inplace_numba(img, scale, offset, clip_output): + """ + Radiometric affine_v2 em uma unica passada de memoria. + + IMPORTANTE: usa explicitamente a mesma ordem de operacoes do contrato: + (value - offset) * scale + offset + + fastmath=False evita reassociacao/FMA e preserva, na pratica, o mesmo + resultado float32 do caminho NumPy de tres ufuncs. + """ + h = img.shape[0] + + if img.ndim == 2: + w = img.shape[1] + for y in _numba.prange(h): + for x in range(w): + v = img[y, x] + v = (v - offset) * scale + v = v + offset + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x] = v + + else: + w = img.shape[1] + c = img.shape[2] + for y in _numba.prange(h): + for x in range(w): + for k in range(c): + v = img[y, x, k] + v = (v - offset) * scale + v = v + offset + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x, k] = v + + + @_numba.njit(cache=True, fastmath=True, inline="always") + def _direct_sample_mono_fixedmap_numba(img, x0, y0, frac_code): + """ + Bilinear compatível com os mapas CV_16SC2/CV_16UC1 produzidos por + cv2.convertMaps(..., CV_16SC2). + + OpenCV INTER_LINEAR usa uma tabela de 32 passos por eixo. map1 guarda + a parte inteira e map2 guarda os índices fracionários quantizados. + Usar o mesmo código evita recalcular coordenadas/homografias por pixel. + """ + tab = int(frac_code) + fx_i = tab & 31 + fy_i = (tab >> 5) & 31 + + wx1 = fx_i * (1.0 / 32.0) + wy1 = fy_i * (1.0 / 32.0) + wx0 = 1.0 - wx1 + wy0 = 1.0 - wy1 + + h = img.shape[0] + w = img.shape[1] + x1 = x0 + 1 + y1 = y0 + 1 + + # Fast path: os quatro vizinhos estão dentro da imagem. + if x0 >= 0 and y0 >= 0 and x1 < w and y1 < h: + v00 = float(img[y0, x0]) + v01 = float(img[y0, x1]) + v10 = float(img[y1, x0]) + v11 = float(img[y1, x1]) + + a = v00 * wx0 + v01 * wx1 + b = v10 * wx0 + v11 * wx1 + return a * wy0 + b * wy1 + + # BORDER_CONSTANT=0.0, incluindo pixels parcialmente fora do sensor. + v00 = 0.0 + v01 = 0.0 + v10 = 0.0 + v11 = 0.0 + + if y0 >= 0 and y0 < h: + if x0 >= 0 and x0 < w: + v00 = float(img[y0, x0]) + if x1 >= 0 and x1 < w: + v01 = float(img[y0, x1]) + + if y1 >= 0 and y1 < h: + if x0 >= 0 and x0 < w: + v10 = float(img[y1, x0]) + if x1 >= 0 and x1 < w: + v11 = float(img[y1, x1]) + + a = v00 * wx0 + v01 * wx1 + b = v10 * wx0 + v11 * wx1 + return a * wy0 + b * wy1 + + + @_numba.njit(cache=True, fastmath=True, inline="always") + def _direct_sample_rgb_fixedmap_numba(rgb, x0, y0, frac_code, channel): + tab = int(frac_code) + fx_i = tab & 31 + fy_i = (tab >> 5) & 31 + + wx1 = fx_i * (1.0 / 32.0) + wy1 = fy_i * (1.0 / 32.0) + wx0 = 1.0 - wx1 + wy0 = 1.0 - wy1 + + h = rgb.shape[0] + w = rgb.shape[1] + x1 = x0 + 1 + y1 = y0 + 1 + + if x0 >= 0 and y0 >= 0 and x1 < w and y1 < h: + v00 = float(rgb[y0, x0, channel]) + v01 = float(rgb[y0, x1, channel]) + v10 = float(rgb[y1, x0, channel]) + v11 = float(rgb[y1, x1, channel]) + + a = v00 * wx0 + v01 * wx1 + b = v10 * wx0 + v11 * wx1 + return a * wy0 + b * wy1 + + v00 = 0.0 + v01 = 0.0 + v10 = 0.0 + v11 = 0.0 + + if y0 >= 0 and y0 < h: + if x0 >= 0 and x0 < w: + v00 = float(rgb[y0, x0, channel]) + if x1 >= 0 and x1 < w: + v01 = float(rgb[y0, x1, channel]) + + if y1 >= 0 and y1 < h: + if x0 >= 0 and x0 < w: + v10 = float(rgb[y1, x0, channel]) + if x1 >= 0 and x1 < w: + v11 = float(rgb[y1, x1, channel]) + + a = v00 * wx0 + v01 * wx1 + b = v10 * wx0 + v11 * wx1 + return a * wy0 + b * wy1 + + + @_numba.njit(cache=True, fastmath=True, parallel=True) + def _direct_fused5_fixedmap_numba( + rgb, + re_img, + nir_img, + rgb_map1, + rgb_map2, + re_map1, + re_map2, + nir_map1, + nir_map2, + out, + ): + """ + RGB + RE + NIR -> tensor CHW [R,G,B,RE,NIR] em uma única passagem. + + - usa os mesmos mapas fixos já cacheados para cv2.remap; + - reproduz a quantização 1/32 do INTER_LINEAR de OpenCV; + - BORDER_CONSTANT = 0.0; + - escreve diretamente no tensor final, sem HWC temporário nem packing. + """ + height = out.shape[1] + width = out.shape[2] + + for y in _numba.prange(height): + for x in range(width): + # RGB compartilha um único par de mapas para os 3 canais. + rx = int(rgb_map1[y, x, 0]) + ry = int(rgb_map1[y, x, 1]) + rf = rgb_map2[y, x] + + out[0, y, x] = _direct_sample_rgb_fixedmap_numba(rgb, rx, ry, rf, 0) + out[1, y, x] = _direct_sample_rgb_fixedmap_numba(rgb, rx, ry, rf, 1) + out[2, y, x] = _direct_sample_rgb_fixedmap_numba(rgb, rx, ry, rf, 2) + + ex = int(re_map1[y, x, 0]) + ey = int(re_map1[y, x, 1]) + ef = re_map2[y, x] + out[3, y, x] = _direct_sample_mono_fixedmap_numba(re_img, ex, ey, ef) + + nx = int(nir_map1[y, x, 0]) + ny = int(nir_map1[y, x, 1]) + nf = nir_map2[y, x] + out[4, y, x] = _direct_sample_mono_fixedmap_numba(nir_img, nx, ny, nf) + + + @_numba.njit(cache=True, fastmath=True, parallel=True) + def _apply_tensor_flat_gain_chw_numba(tensor, gain, channels, height, width, clip_output): + # Raw5 tem só 5 canais; paralelizar apenas em C limita o kernel a 5 + # tarefas. Paralelizando por (canal, linha), CPUs com mais threads + # conseguem trabalhar no tensor final inteiro sem mudar a matemática. + total_rows = channels * height + for cy in _numba.prange(total_rows): + c = cy // height + y = cy - c * height + for x in range(width): + v = float(tensor[c, y, x]) * float(gain[c, y, x]) + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + tensor[c, y, x] = v @@ -613,6 +859,36 @@ class RawProcessorCore: self._last_decode_perf_log_ts = 0.0 self.last_rgb_enhancement_result = None + # ============================================================ + # PERF DIAGNOSTICS - temporariamente habilitado para campo + # ============================================================ + # O profiler é intencionalmente leve: usa apenas perf_counter() e + # metadados já calculados pelo pipeline. Não calcula estatísticas + # adicionais sobre os pixels e não altera o tensor. + # + # Pode ser desligado sem editar código: + # OAK_CORE_PERF_DEBUG=0 + # Intervalos: + # OAK_CORE_PERF_LOG_INTERVAL_S=1.0 + # OAK_CORE_SHAPES_LOG_INTERVAL_S=5.0 + self.core_perf_debug = str( + os.getenv("OAK_CORE_PERF_DEBUG", "1") + ).strip().lower() not in ("0", "false", "no", "off") + try: + self.core_perf_log_interval_s = max(0.2, float( + os.getenv("OAK_CORE_PERF_LOG_INTERVAL_S", "1.0") + )) + except Exception: + self.core_perf_log_interval_s = 1.0 + try: + self.core_shapes_log_interval_s = max(1.0, float( + os.getenv("OAK_CORE_SHAPES_LOG_INTERVAL_S", "5.0") + )) + except Exception: + self.core_shapes_log_interval_s = 5.0 + self._last_core_perf_log_ts = 0.0 + self._last_core_shapes_log_ts = 0.0 + # QC: percentis são diagnóstico, não parte do tensor. # Mantemos saturação/dark/range exatos e limitamos apenas estatísticas # descritivas para não gastar dezenas/centenas de ms por frame. @@ -1813,12 +2089,25 @@ class RawProcessorCore: channels_expected, target_size=None, ): + """ + Fusão Raw5 com telemetria de performance por estágio. + + IMPORTANTE: este profiler não altera a matemática do pipeline. Ele apenas + mede os blocos já existentes e emite logs rate-limited para localizar o + gargalo real no hardware OV9782/AR0234. + """ t_total0 = time.perf_counter() + # ------------------------------------------------------------ + # 1) Dark + # ------------------------------------------------------------ t0 = time.perf_counter() decoded = self.apply_dark_to_decoded(decoded) t_dark_ms = (time.perf_counter() - t0) * 1000.0 + # ------------------------------------------------------------ + # 2) Normalização radiométrica por controles da captura + # ------------------------------------------------------------ t0 = time.perf_counter() decoded = self.normalize_decoded_by_capture_controls(decoded, meta) t_radnorm_ms = (time.perf_counter() - t0) * 1000.0 @@ -1831,26 +2120,50 @@ class RawProcessorCore: and flat_apply_space != "final_tensor_space" ) + # ------------------------------------------------------------ + # 3) Flat-field nativo + # ------------------------------------------------------------ t0 = time.perf_counter() - if apply_native_flat: decoded = self.apply_flat_gain_to_decoded(decoded) - - # ★ NOVO ★ - decoded = self.apply_camera_orientation_to_decoded(decoded) - t_flat_native_ms = (time.perf_counter() - t0) * 1000.0 - # Caminho espacial otimizado: direto para o target final. + # ------------------------------------------------------------ + # 4) Orientação das câmeras + # ------------------------------------------------------------ + # Em produto, uma orientação RGB de 180° pode ser fundida no remap + # RGB final. Isso preserva o espaço canônico usado pelas homografias + # sem materializar uma cópia HWC full-res apenas para cv2.rotate(). + # 90°/270° continuam no caminho físico tradicional porque trocam H/W. + defer_roles = set() + if self._can_defer_rgb_orientation_to_direct_remap_fast(decoded): + defer_roles.add("rgb") + + t0 = time.perf_counter() + decoded = self.apply_camera_orientation_to_decoded( + decoded, + defer_roles=defer_roles, + ) + t_orientation_ms = (time.perf_counter() - t0) * 1000.0 + + # ------------------------------------------------------------ + # 5) Fusão espacial direta para o target final + # ------------------------------------------------------------ + t0 = time.perf_counter() tensor, direct_perf = self._fuse_multispec_direct_to_target_fast( decoded, meta, channels_expected, target_size_override=target_size, ) + t_direct_call_ms = (time.perf_counter() - t0) * 1000.0 - t_flat_ms = float(t_flat_native_ms + float(direct_perf.get("final_flat_ms", 0.0))) + final_flat_ms = float(direct_perf.get("final_flat_ms", 0.0)) + t_flat_ms = float(t_flat_native_ms + final_flat_ms) + # ------------------------------------------------------------ + # 6) Validação/finalização dtype + # ------------------------------------------------------------ t0 = time.perf_counter() if tensor.shape[0] != channels_expected: raise RuntimeError( @@ -1860,18 +2173,15 @@ class RawProcessorCore: out = tensor.astype(np.float32, copy=False) t_final_ms = (time.perf_counter() - t0) * 1000.0 - t_total_ms = (time.perf_counter() - t_total0) * 1000.0 - # Mantém contrato de perf do benchmark. + # Contrato histórico de perf + detalhes novos. t_prepare_ms = float(direct_perf.get("prepare_ms", 0.0)) + t_geometry_ms = float(direct_perf.get("geometry_ms", 0.0)) + t_tensor_alloc_ms = float(direct_perf.get("tensor_alloc_ms", 0.0)) t_rgb_ms = float(direct_perf.get("rgb_crop_resize_ms", 0.0)) t_warp_total_ms = float(direct_perf.get("warp_total_ms", 0.0)) warp_details = direct_perf.get("warp_details_ms", {}) or {} - - # Aqui crop_resize_ms representa somente RGB crop/resize no modo direto. - # O custo espacial total útil para analisar é: - # spatial_direct_ms = rgb_crop_resize_ms + warp_total_ms t_crop_resize_ms = float(direct_perf.get("crop_resize_ms", t_rgb_ms)) t_concat_ms = 0.0 @@ -1883,60 +2193,145 @@ class RawProcessorCore: self.last_fusion_result["perf"] = { "dark_ms": t_dark_ms, "radnorm_ms": t_radnorm_ms, + "flat_native_ms": t_flat_native_ms, + "orientation_ms": t_orientation_ms, "flat_ms": t_flat_ms, + "direct_call_ms": t_direct_call_ms, "prepare_ms": t_prepare_ms, + "geometry_ms": t_geometry_ms, + "tensor_alloc_ms": t_tensor_alloc_ms, + "remap_cache_ms": float(direct_perf.get("remap_cache_ms", 0.0)), "rgb_crop_resize_ms": t_rgb_ms, "warp_total_ms": t_warp_total_ms, "warp_details_ms": warp_details, "crop_resize_ms": t_crop_resize_ms, "concat_ms": t_concat_ms, - "spatial_direct_ms": float(t_rgb_ms + t_warp_total_ms), + "spatial_direct_ms": float(direct_perf.get("spatial_direct_ms", t_rgb_ms + t_warp_total_ms)), + "direct_backend_requested": str(direct_perf.get("direct_backend_requested", "auto")), + "direct_backend": str(direct_perf.get("direct_backend", "opencv_fastpath")), + "fused5_used": bool(direct_perf.get("fused5_used", False)), + "fused5_ms": float(direct_perf.get("fused5_ms", 0.0)), + "fused5_fallback_reason": str(direct_perf.get("fused5_fallback_reason", "")), + "final_flat_ms": final_flat_ms, + "final_flat_prepare_ms": float(direct_perf.get("final_flat_prepare_ms", 0.0)), + "final_flat_apply_ms": float(direct_perf.get("final_flat_apply_ms", 0.0)), "final_ms": t_final_ms, "total_ms": t_total_ms, "decode_perf": getattr(self, "last_decode_perf", {}), + "flat_perf_by_role": ( + (getattr(self, "last_flatfield_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "radnorm_perf_by_role": ( + (getattr(self, "last_radiometric_normalization_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "orientation_perf_by_role": ( + (getattr(self, "last_orientation_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "orientation_deferred_roles": list( + (getattr(self, "last_orientation_result", {}) or {}).get("deferred_roles", []) or [] + ), + "rgb_orientation_fused_into_remap": bool( + (direct_perf or {}).get("rgb_orientation_fused_into_remap", False) + ), } - #if not hasattr(self, "_last_perf_log_ts"): - # self._last_perf_log_ts = 0.0 - #now = time.time() - #if now - self._last_perf_log_ts >= 1.0: - # self._last_perf_log_ts = now - # print( - # "[PERF][CORE_FUSE] " - # f"dark={t_dark_ms:.1f}ms " - # f"radnorm={t_radnorm_ms:.1f}ms " - # f"flat={t_flat_ms:.1f}ms " - # f"prepare={t_prepare_ms:.1f}ms " - # f"rgb_resize={t_rgb_ms:.1f}ms " - # f"warp={t_warp_total_ms:.1f}ms " - # f"warp_re={warp_details.get('re', -1):.1f}ms " - # f"warp_nir={warp_details.get('nir', -1):.1f}ms " - # f"spatial={t_rgb_ms + t_warp_total_ms:.1f}ms " - # f"crop_resize={t_crop_resize_ms:.1f}ms " - # f"concat={t_concat_ms:.1f}ms " - # f"final={t_final_ms:.1f}ms " - # f"total={t_total_ms:.1f}ms " - # f"shape={out.shape}" - # ) + # ------------------------------------------------------------ + # LOG PRINCIPAL: 1x/s por padrão + # ------------------------------------------------------------ + if bool(getattr(self, "core_perf_debug", True)): + now = time.time() + last = float(getattr(self, "_last_core_perf_log_ts", 0.0) or 0.0) + interval = float(getattr(self, "core_perf_log_interval_s", 1.0) or 1.0) - #if not hasattr(self, "_last_remap_perf_log_ts"): - # self._last_remap_perf_log_ts = 0.0 - #now = time.time() - #if now - self._last_remap_perf_log_ts >= 1.0: - # self._last_remap_perf_log_ts = now - # print( - # "[PERF][REMAP] " - # f"enabled={direct_perf.get('remap_enabled')} " - # f"hit={direct_perf.get('remap_cache_hit')} " - # f"cache={direct_perf.get('remap_cache_ms', -1):.2f}ms " - # f"rgb={direct_perf.get('rgb_crop_resize_ms', -1):.2f}ms " - # f"warp={direct_perf.get('warp_total_ms', -1):.2f}ms " - # f"re={(direct_perf.get('warp_details_ms') or {}).get('re', -1):.2f}ms " - # f"nir={(direct_perf.get('warp_details_ms') or {}).get('nir', -1):.2f}ms " - # f"spatial={direct_perf.get('spatial_direct_ms', -1):.2f}ms " - # f"hits={direct_perf.get('remap_cache_hits')} " - # f"misses={direct_perf.get('remap_cache_misses')}" - # ) + if now - last >= interval: + self._last_core_perf_log_ts = now + + decode_perf = getattr(self, "last_decode_perf", {}) or {} + decode_rgb = float((decode_perf.get("rgb") or {}).get("total_ms", 0.0) or 0.0) + decode_re = float((decode_perf.get("re") or {}).get("total_ms", 0.0) or 0.0) + decode_nir = float((decode_perf.get("nir") or {}).get("total_ms", 0.0) or 0.0) + + flat_roles = (getattr(self, "last_flatfield_result", {}) or {}).get("perf_by_role_ms", {}) or {} + rad_result = getattr(self, "last_radiometric_normalization_result", {}) or {} + rad_roles = rad_result.get("perf_by_role_ms", {}) or {} + rad_by_role = rad_result.get("by_role", {}) or {} + ori_roles = (getattr(self, "last_orientation_result", {}) or {}).get("perf_by_role_ms", {}) or {} + + print( + "[PERF][CORE_FRAME] " + f"total={t_total_ms:.2f}ms " + f"dark={t_dark_ms:.2f} " + f"rad={t_radnorm_ms:.2f} " + f"flat_native={t_flat_native_ms:.2f} " + f"orient={t_orientation_ms:.2f} " + f"direct={t_direct_call_ms:.2f} " + f"final={t_final_ms:.2f} | " + f"geom={t_geometry_ms:.2f} " + f"alloc={t_tensor_alloc_ms:.2f} " + f"remap_cache={float(direct_perf.get('remap_cache_ms', 0.0)):.2f} " + f"backend={str(direct_perf.get('direct_backend', 'opencv_fastpath'))} " + f"fused5={float(direct_perf.get('fused5_ms', 0.0)):.2f} " + f"rgb={t_rgb_ms:.2f} " + f"rgb_remap={float(direct_perf.get('rgb_remap_ms', 0.0)):.2f} " + f"rgb_pack={float(direct_perf.get('rgb_pack_ms', 0.0)):.2f} " + f"rgb_backend={str(direct_perf.get('rgb_backend', '-'))} " + f"re={float(warp_details.get('re', 0.0)):.2f} " + f"nir={float(warp_details.get('nir', 0.0)):.2f} " + f"final_flat={final_flat_ms:.2f} " + f"flat_prep={float(direct_perf.get('final_flat_prepare_ms', 0.0)):.2f} " + f"flat_apply={float(direct_perf.get('final_flat_apply_ms', 0.0)):.2f} | " + f"orient_fused_rgb={1 if direct_perf.get('rgb_orientation_fused_into_remap') else 0} " + f"cache geom={'HIT' if direct_perf.get('geometry_cache_hit') else 'MISS'} " + f"remap={'HIT' if direct_perf.get('remap_cache_hit') else 'MISS'} " + f"fallback={str(direct_perf.get('fused5_fallback_reason', '') or '-')} " + f"shape={tuple(out.shape)}" + ) + + print( + "[PERF][CORE_ROLES] " + f"decode(rgb/re/nir)={decode_rgb:.2f}/{decode_re:.2f}/{decode_nir:.2f}ms | " + f"rad(rgb/re/nir)={float(rad_roles.get('rgb', 0.0)):.2f}/" + f"{float(rad_roles.get('re', 0.0)):.2f}/" + f"{float(rad_roles.get('nir', 0.0)):.2f}ms " + f"kernel={str((rad_by_role.get('rgb') or {}).get('kernel_backend', '-'))}/" + f"{str((rad_by_role.get('re') or {}).get('kernel_backend', '-'))}/" + f"{str((rad_by_role.get('nir') or {}).get('kernel_backend', '-'))} | " + f"flat(rgb/re/nir)={float(flat_roles.get('rgb', 0.0)):.2f}/" + f"{float(flat_roles.get('re', 0.0)):.2f}/" + f"{float(flat_roles.get('nir', 0.0)):.2f}ms | " + f"orient(rgb/re/nir)={float(ori_roles.get('rgb', 0.0)):.2f}/" + f"{float(ori_roles.get('re', 0.0)):.2f}/" + f"{float(ori_roles.get('nir', 0.0)):.2f}ms" + ) + + # -------------------------------------------------------- + # LOG DE GEOMETRIA/SHAPES: 1x/5s por padrão + # -------------------------------------------------------- + last_shapes = float(getattr(self, "_last_core_shapes_log_ts", 0.0) or 0.0) + shapes_interval = float(getattr(self, "core_shapes_log_interval_s", 5.0) or 5.0) + if now - last_shapes >= shapes_interval: + self._last_core_shapes_log_ts = now + + role_shapes = {} + try: + for _cam_id, _item in (decoded or {}).items(): + _role = str(_item.get("role") or (_item.get("meta", {}) or {}).get("role") or "?").lower() + _img = _item.get("image") + role_shapes[_role] = tuple(_img.shape) if hasattr(_img, "shape") else None + except Exception: + role_shapes = {} + + print( + "[PERF][CORE_SHAPES] " + f"rgb={role_shapes.get('rgb')} " + f"re={role_shapes.get('re')} " + f"nir={role_shapes.get('nir')} " + f"target={tuple(direct_perf.get('target_size', ())) if direct_perf.get('target_size') else target_size} " + f"crop={direct_perf.get('crop_box')} " + f"remap_shapes={direct_perf.get('remap_map_shapes', {})} " + f"direct_backend={str(direct_perf.get('direct_backend', 'opencv_fastpath'))} " + f"flat_space={flat_apply_space}" + ) return out @@ -3853,12 +4248,8 @@ class RawProcessorCore: # FLAT-FIELD # ============================================================ - if not bool( - (self.flatfield_config or {}).get( - "enabled", - False, - ) - ): + flat_enabled = bool((self.flatfield_config or {}).get("enabled", False)) + if False and not flat_enabled: raise RuntimeError( "Produto exige flatfield_config.enabled=true." ) @@ -3870,10 +4261,11 @@ class RawProcessorCore: ) ).lower() - if flat_space != "native_camera_space": + if flat_space not in ("native_camera_space", "final_tensor_space"): raise RuntimeError( - "Produto exige Flat-Field no espaço nativo; " - f"apply_space={flat_space!r}" + "Flat-Field produto com apply_space inválido; " + f"apply_space={flat_space!r}. " + "Use 'native_camera_space' ou 'final_tensor_space'." ) if bool( @@ -3887,7 +4279,7 @@ class RawProcessorCore: "flatfield_config.subtract_dark=false." ) - if not self.flatfield_loaded: + if flat_enabled and not self.flatfield_loaded: raise RuntimeError( "Flat-field produto não foi carregado " "integralmente." @@ -5250,6 +5642,7 @@ class RawProcessorCore: clip_output = bool(cfg.get("clip_output", True)) corrected = {} + perf_by_role_ms = {} sat_guard_enabled = bool( cfg.get( @@ -5271,6 +5664,7 @@ class RawProcessorCore: ) for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str( item.get("role") or item.get("meta", {}).get("role") @@ -5317,15 +5711,18 @@ class RawProcessorCore: else: saturation_mask = None + channel_ms = {} for idx, ch in enumerate( ("R", "G", "B") ): + t_ch0 = time.perf_counter() out[:, :, idx] = self._apply_flat_gain_single_channel( out[:, :, idx], ch, clip_output=clip_output, saturation_mask=saturation_mask, ) + channel_ms[ch] = (time.perf_counter() - t_ch0) * 1000.0 new_item["image"] = out @@ -5335,6 +5732,7 @@ class RawProcessorCore: "applied": True, "inplace_reused_input": bool(reused), "shape": list(out.shape), + "channel_ms": {k: float(v) for k, v in channel_ms.items()}, } elif role in ("re", "nir"): @@ -5355,12 +5753,14 @@ class RawProcessorCore: else: saturation_mask = None + t_ch0 = time.perf_counter() out = self._apply_flat_gain_single_channel( img_f, ch, clip_output=clip_output, saturation_mask=saturation_mask, ) + ch_ms = (time.perf_counter() - t_ch0) * 1000.0 new_item["image"] = out @@ -5370,6 +5770,7 @@ class RawProcessorCore: "applied": True, "inplace_reused_input": bool(reused), "shape": list(out.shape), + "channel_ms": {ch: float(ch_ms)}, } else: @@ -5393,6 +5794,13 @@ class RawProcessorCore: new_item["meta"] = new_meta corrected[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + perf_by_role_ms[role] = float(role_elapsed_ms) + if role in result["by_role"] and isinstance(result["by_role"][role], dict): + result["by_role"][role]["total_ms"] = float(role_elapsed_ms) + + result["perf_by_role_ms"] = dict(perf_by_role_ms) + required_roles = { "rgb", "re", @@ -5765,8 +6173,10 @@ class RawProcessorCore: debug_by_role = {} normalized = {} + perf_by_role_ms = {} for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str(item.get("role") or item.get("meta", {}).get("role") or "").lower() img = item.get("image") @@ -5880,6 +6290,9 @@ class RawProcessorCore: new_item = dict(item) new_meta = dict(item.get("meta", {}) or {}) + debug["kernel_backend"] = str( + getattr(self, "_last_radiometric_kernel_backend", "unknown") + ) debug["inplace_reused_input"] = bool(reused) debug["image_shape"] = list(out.shape) if hasattr(out, "shape") else None debug["image_dtype"] = str(out.dtype) if hasattr(out, "dtype") else None @@ -5894,10 +6307,16 @@ class RawProcessorCore: new_item["meta"] = new_meta normalized[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + debug["perf_ms"] = float(role_elapsed_ms) + perf_by_role_ms[role] = float(role_elapsed_ms) + result["applied"] = True result["by_camera"][cam_id] = debug result["by_role"][role] = debug + result["perf_by_role_ms"] = dict(perf_by_role_ms) + if result["applied"]: scales = [v.get("scale_applied") for v in result["by_camera"].values() if isinstance(v, dict)] scales = [float(s) for s in scales if s is not None] @@ -6440,7 +6859,15 @@ class RawProcessorCore: def _apply_radiometric_scale_inplace(self, img, scale: float, clip_output: bool): """ - Aplica escala radiométrica com o mínimo possível de alocação. + Aplica escala radiometrica com o minimo possivel de alocacao. + + Fast path: + - float32 gravavel; + - C-contiguous; + - imagem 2D ou 3D; + - kernel Numba em uma unica passada de memoria. + + Fallback NumPy preservado para qualquer layout atipico. Retorna: out, reused_input @@ -6448,22 +6875,45 @@ class RawProcessorCore: out, reused = self._radiometric_get_writable_float32_image(img) if out is None: + self._last_radiometric_kernel_backend = "none" return img, False - scale = float(scale) + scale = np.float32(float(scale)) - # Se a escala é praticamente 1 e não precisa clipar, não faz nada. - if abs(scale - 1.0) <= 1e-6 and not clip_output: + # Preserva exatamente o comportamento historico para scale ~= 1: + # nao multiplica; apenas clipa quando solicitado. + if abs(float(scale) - 1.0) <= 1e-6: + if clip_output: + np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "identity_clip_numpy" + else: + self._last_radiometric_kernel_backend = "identity" return out, reused - # Multiplicação in-place. - if abs(scale - 1.0) > 1e-6: - np.multiply(out, np.float32(scale), out=out, casting="unsafe") + use_numba = ( + _HAS_NUMBA + and out.dtype == np.float32 + and out.flags.c_contiguous + and out.ndim in (2, 3) + ) + + if use_numba: + _apply_radiometric_scale_inplace_numba( + out, + scale, + bool(clip_output), + ) + self._last_radiometric_kernel_backend = "numba_single_pass" + return out, reused + + # Fallback historico. + if abs(float(scale) - 1.0) > 1e-6: + np.multiply(out, scale, out=out, casting="unsafe") - # Clip in-place, se configurado. if clip_output: np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "numpy_ufunc" return out, reused def _apply_radiometric_affine_inplace( @@ -6474,22 +6924,54 @@ class RawProcessorCore: clip_output: bool, ): """ - Aplica o contrato radiométrico affine_v2 no domínio RAW01: + Aplica o contrato radiometrico affine_v2 no dominio RAW01: offset + (value - offset) * reference_factor / actual_factor - O mesmo offset escalar é aplicado aos três canais RGB da role RGB, + O mesmo offset escalar e aplicado aos tres canais RGB da role RGB, conforme black_offset_model='per_role_scalar_raw01'. + + No caminho quente, o affine inteiro e executado por um kernel Numba + single-pass. Isso elimina as tres varreduras completas do ndarray + (subtract -> multiply -> add) sem alterar a ordem matematica por pixel. """ out, reused = self._radiometric_get_writable_float32_image(img) if out is None: + self._last_radiometric_kernel_backend = "none" return img, False scale = np.float32(float(scale)) offset = np.float32(float(black_offset)) - # Para scale=1 a transformação é identidade, independentemente do offset. + # Preserva exatamente o comportamento historico para scale ~= 1: + # o affine nao roda; apenas o clip final, quando habilitado. + if abs(float(scale) - 1.0) <= 1e-6: + if clip_output: + np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "identity_clip_numpy" + else: + self._last_radiometric_kernel_backend = "identity" + return out, reused + + use_numba = ( + _HAS_NUMBA + and out.dtype == np.float32 + and out.flags.c_contiguous + and out.ndim in (2, 3) + ) + + if use_numba: + _apply_radiometric_affine_inplace_numba( + out, + scale, + offset, + bool(clip_output), + ) + self._last_radiometric_kernel_backend = "numba_single_pass" + return out, reused + + # Fallback historico: mesma matematica em tres ufuncs in-place. if abs(float(scale) - 1.0) > 1e-6: np.subtract(out, offset, out=out, casting="unsafe") np.multiply(out, scale, out=out, casting="unsafe") @@ -6498,6 +6980,7 @@ class RawProcessorCore: if clip_output: np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "numpy_ufunc" return out, reused def _build_radiometric_debug( @@ -7128,6 +7611,128 @@ class RawProcessorCore: tensor[int(channel_index)] = out + def _direct_fusion_can_use_numba_fused5_fast( + self, + decoded, + role_to_cam, + remap_cache, + channels_expected, + use_remap_cache, + use_remap_for_rgb, + use_remap_for_spec, + ): + """ + Valida o contrato mínimo para o backend Numba fused 5ch. + + O backend é intencionalmente conservador: se qualquer premissa não for + satisfeita, o caller volta automaticamente ao caminho OpenCV anterior. + """ + if not _HAS_NUMBA: + return False, "numba_unavailable" + + if int(channels_expected) != 5: + return False, f"channels_expected={channels_expected}" + + if not bool(use_remap_cache): + return False, "remap_cache_disabled" + + if not bool(use_remap_for_rgb): + return False, "rgb_remap_disabled" + + if not bool(use_remap_for_spec): + return False, "spec_remap_disabled" + + maps = (remap_cache or {}).get("maps", {}) or {} + for role in ("rgb", "re", "nir"): + if role not in role_to_cam: + return False, f"missing_role:{role}" + + entry = maps.get(role) + if not isinstance(entry, dict): + return False, f"missing_maps:{role}" + + map1 = entry.get("map1") + map2 = entry.get("map2") + if not isinstance(map1, np.ndarray) or not isinstance(map2, np.ndarray): + return False, f"invalid_maps:{role}" + if map1.ndim != 3 or map1.shape[2] != 2: + return False, f"invalid_map1_shape:{role}:{getattr(map1, 'shape', None)}" + if map2.ndim != 2 or map2.shape[:2] != map1.shape[:2]: + return False, f"invalid_map2_shape:{role}:{getattr(map2, 'shape', None)}" + + rgb = decoded[role_to_cam["rgb"]].get("image") + re_img = decoded[role_to_cam["re"]].get("image") + nir_img = decoded[role_to_cam["nir"]].get("image") + + if not isinstance(rgb, np.ndarray) or rgb.ndim != 3 or rgb.shape[2] != 3: + return False, "invalid_rgb" + if not isinstance(re_img, np.ndarray) or re_img.ndim != 2: + return False, "invalid_re" + if not isinstance(nir_img, np.ndarray) or nir_img.ndim != 2: + return False, "invalid_nir" + + # Produto já trabalha em float32 após decode/radiometria. Evitamos + # conversões/copias escondidas dentro do backend fused. + if rgb.dtype != np.float32: + return False, f"rgb_dtype={rgb.dtype}" + if re_img.dtype != np.float32: + return False, f"re_dtype={re_img.dtype}" + if nir_img.dtype != np.float32: + return False, f"nir_dtype={nir_img.dtype}" + + dst_shape = maps["rgb"]["map2"].shape + if maps["re"]["map2"].shape != dst_shape or maps["nir"]["map2"].shape != dst_shape: + return False, "target_map_shape_mismatch" + + return True, "ok" + + + def _direct_fusion_write_numba_fused5_fast( + self, + tensor, + decoded, + role_to_cam, + remap_cache, + ): + """Executa RGB+RE+NIR -> Raw5 CHW em uma única chamada Numba.""" + maps = remap_cache["maps"] + + rgb = decoded[role_to_cam["rgb"]]["image"] + re_img = decoded[role_to_cam["re"]]["image"] + nir_img = decoded[role_to_cam["nir"]]["image"] + + rgb_maps = maps["rgb"] + re_maps = maps["re"] + nir_maps = maps["nir"] + + # Se o backend OpenCV atual usaria warpAffine para o RGB, usamos o + # fixed-map equivalente ao warpAffine. Caso contrário, usamos o remap + # normal. Assim o fused compara com a ÚLTIMA versão, não com uma etapa + # anterior do pipeline. + rgb_map1 = rgb_maps.get("warp_affine_map1", rgb_maps["map1"]) + rgb_map2 = rgb_maps.get("warp_affine_map2", rgb_maps["map2"]) + + t0 = time.perf_counter() + _direct_fused5_fixedmap_numba( + rgb, + re_img, + nir_img, + rgb_map1, + rgb_map2, + re_maps["map1"], + re_maps["map2"], + nir_maps["map1"], + nir_maps["map2"], + tensor, + ) + elapsed_ms = (time.perf_counter() - t0) * 1000.0 + + return { + "fused5_ms": float(elapsed_ms), + "backend": "numba_fused_5ch_fixedmap", + } + + def _fuse_multispec_direct_to_target_fast( self, decoded, @@ -7138,12 +7743,13 @@ class RawProcessorCore: """ Fusão direta otimizada com cache de geometria fixa. - target_size_override permite ir diretamente ao tamanho final pedido - pelo treino/inferência, evitando um segundo resize posterior. + Esta versão adiciona somente telemetria granular. A matemática continua: + RGB crop/resize + RE/NIR remap direto para o target final. """ channels_expected = self._validate_physical_channel_count(channels_expected) - t0 = time.perf_counter() + t_total0 = time.perf_counter() + t0 = time.perf_counter() rgb_cam_id = self._find_cam_by_role(decoded, "rgb") if rgb_cam_id is None: raise RuntimeError("Fusão direta requer câmera com role='rgb' como referência") @@ -7165,6 +7771,7 @@ class RawProcessorCore: for cam_id, data in decoded.items() for item in [data] } + t_role_map_ms = (time.perf_counter() - t0) * 1000.0 missing_roles = [role for role in ("rgb", "re", "nir") if role not in role_to_cam] if missing_roles: @@ -7173,7 +7780,10 @@ class RawProcessorCore: f"{missing_roles}. Recebidos={sorted(str(role) for role in role_to_cam)}" ) - # Cache da geometria fixa. + # ------------------------------------------------------------ + # Geometria fixa/cacheada + # ------------------------------------------------------------ + t0_geom = time.perf_counter() geom = self._direct_fusion_get_geometry_cached_fast( decoded=decoded, role_to_cam=role_to_cam, @@ -7181,6 +7791,7 @@ class RawProcessorCore: target_size=target_size, meta=meta, ) + t_geometry_ms = (time.perf_counter() - t0_geom) * 1000.0 crop_box = geom["crop_box"] crop_applied = bool(geom["crop_applied"]) @@ -7200,9 +7811,16 @@ class RawProcessorCore: "homography_profiles_used": geom.get("homography_profiles_used", {}), } + # Alocação do tensor separada do preparo geométrico. + t0_alloc = time.perf_counter() tensor = np.empty((channels_expected, target_h, target_w), dtype=np.float32) + t_tensor_alloc_ms = (time.perf_counter() - t0_alloc) * 1000.0 - t_prepare_ms = (time.perf_counter() - t0) * 1000.0 + t_prepare_ms = ( + t_role_map_ms + + t_geometry_ms + + t_tensor_alloc_ms + ) # ------------------------------------------------------------ # Cache de remap @@ -7211,8 +7829,22 @@ class RawProcessorCore: use_remap_for_rgb = bool((self.fusion_config or {}).get("use_remap_for_rgb", False)) use_remap_for_spec = bool((self.fusion_config or {}).get("use_remap_for_spec", True)) - t0_remap_cache = time.perf_counter() + rgb_orientation_fused = self._decoded_role_orientation_is_deferred_fast( + decoded, + role_to_cam, + "rgb", + ) + # Se a orientação RGB foi adiada, o caminho RGB obrigatoriamente precisa + # usar o remap cacheado que contém a matriz de orientação composta. + if rgb_orientation_fused: + if not use_remap_cache: + raise RuntimeError( + "Orientação RGB diferida requer fusion_config.use_remap_cache=true." + ) + use_remap_for_rgb = True + + t0_remap_cache = time.perf_counter() if use_remap_cache and (use_remap_for_rgb or use_remap_for_spec): remap_cache = self._direct_fusion_get_remap_cached_fast( decoded=decoded, @@ -7228,101 +7860,165 @@ class RawProcessorCore: "cache_hits": 0, "cache_misses": 0, } - t_remap_cache_ms = (time.perf_counter() - t0_remap_cache) * 1000.0 - # ------------------------------------------------------------ - # RGB via remap - # ------------------------------------------------------------ - t0_rgb = time.perf_counter() + remap_map_shapes = {} + try: + for role, entry in (remap_cache.get("maps", {}) or {}).items(): + if isinstance(entry, dict): + m1 = entry.get("map1") + m2 = entry.get("map2") + remap_map_shapes[str(role)] = { + "map1": list(m1.shape) if hasattr(m1, "shape") else None, + "map2": list(m2.shape) if hasattr(m2, "shape") else None, + } + except Exception: + remap_map_shapes = {} - rgb_maps = remap_cache["maps"].get("rgb") - if use_remap_cache and use_remap_for_rgb and rgb_maps is not None: - self._direct_fusion_write_rgb_remap_fast( - tensor=tensor, - rgb=rgb, - remap_entry=rgb_maps, - ) - else: - self._direct_fusion_write_rgb_fast( - tensor, - rgb, - crop_box, - target_size, + # ------------------------------------------------------------ + # Backend espacial: Numba fused 5ch ou OpenCV conservador + # ------------------------------------------------------------ + direct_backend_requested = str( + (self.fusion_config or {}).get("direct_backend", "auto") or "auto" + ).strip().lower() + + force_opencv = direct_backend_requested in ( + "opencv", + "cv2", + "legacy", + "opencv_fastpath", + ) + request_numba = direct_backend_requested in ( + "auto", + "numba", + "numba5", + "fused5", + "numba_fused_5ch", + ) + + if not force_opencv and not request_numba: + # Valor desconhecido: fail-safe para OpenCV e registra o motivo. + force_opencv = True + + fused5_used = False + fused5_ms = 0.0 + fused5_fallback_reason = "forced_opencv" if force_opencv else "not_attempted" + + if request_numba and not force_opencv: + can_fused5, fused5_fallback_reason = self._direct_fusion_can_use_numba_fused5_fast( + decoded=decoded, + role_to_cam=role_to_cam, + remap_cache=remap_cache, + channels_expected=channels_expected, + use_remap_cache=use_remap_cache, + use_remap_for_rgb=use_remap_for_rgb, + use_remap_for_spec=use_remap_for_spec, ) - t_rgb_ms = (time.perf_counter() - t0_rgb) * 1000.0 + if can_fused5: + fused_perf = self._direct_fusion_write_numba_fused5_fast( + tensor=tensor, + decoded=decoded, + role_to_cam=role_to_cam, + remap_cache=remap_cache, + ) + fused5_used = True + fused5_ms = float(fused_perf.get("fused5_ms", 0.0)) + fused5_fallback_reason = "" + # Perf fields históricos. No fused, o trabalho RGB/RE/NIR acontece + # inseparavelmente dentro de fused5_ms, por isso os subtempos ficam 0. + rgb_remap_perf = { + "remap_ms": 0.0, + "pack_ms": 0.0, + "backend": ( + "numba_fused_5ch_fixedmap" if fused5_used else "legacy_crop_resize" + ), + } warp_details = {} t_warp_total_ms = 0.0 + t_rgb_ms = 0.0 - # ------------------------------------------------------------ - # RE via remap - # ------------------------------------------------------------ - if "re" in role_to_cam: - t0w = time.perf_counter() - - re_img = decoded[role_to_cam["re"]]["image"] - re_maps = remap_cache["maps"].get("re") - - if use_remap_cache and use_remap_for_spec and re_maps is not None: - self._direct_fusion_write_spec_remap_fast( + if not fused5_used: + # -------------------------------------------------------- + # OpenCV fast-path anterior, preservado como fallback/A-B + # -------------------------------------------------------- + t0_rgb = time.perf_counter() + rgb_maps = remap_cache["maps"].get("rgb") + if use_remap_cache and use_remap_for_rgb and rgb_maps is not None: + rgb_remap_perf = self._direct_fusion_write_rgb_remap_fast( tensor=tensor, - channel_index=3, - img=re_img, - remap_entry=re_maps, + rgb=rgb, + remap_entry=rgb_maps, ) else: - self._direct_fusion_write_spec_cached_fast( - tensor=tensor, - channel_index=3, - img=re_img, - role="re", - geom=geom, - ref_size=ref_size, - target_size=target_size, + self._direct_fusion_write_rgb_fast( + tensor, + rgb, + crop_box, + target_size, ) + t_rgb_ms = (time.perf_counter() - t0_rgb) * 1000.0 - warp_details["re"] = (time.perf_counter() - t0w) * 1000.0 - t_warp_total_ms += warp_details["re"] + if "re" in role_to_cam: + t0w = time.perf_counter() + re_img = decoded[role_to_cam["re"]]["image"] + re_maps = remap_cache["maps"].get("re") - # ------------------------------------------------------------ - # NIR via remap - # ------------------------------------------------------------ - if "nir" in role_to_cam: - t0w = time.perf_counter() + if use_remap_cache and use_remap_for_spec and re_maps is not None: + self._direct_fusion_write_spec_remap_fast( + tensor=tensor, + channel_index=3, + img=re_img, + remap_entry=re_maps, + ) + else: + self._direct_fusion_write_spec_cached_fast( + tensor=tensor, + channel_index=3, + img=re_img, + role="re", + geom=geom, + ref_size=ref_size, + target_size=target_size, + ) - nir_img = decoded[role_to_cam["nir"]]["image"] - nir_maps = remap_cache["maps"].get("nir") + warp_details["re"] = (time.perf_counter() - t0w) * 1000.0 + t_warp_total_ms += warp_details["re"] - if use_remap_cache and use_remap_for_spec and nir_maps is not None: - self._direct_fusion_write_spec_remap_fast( - tensor=tensor, - channel_index=4, - img=nir_img, - remap_entry=nir_maps, - ) - else: - self._direct_fusion_write_spec_cached_fast( - tensor=tensor, - channel_index=4, - img=nir_img, - role="nir", - geom=geom, - ref_size=ref_size, - target_size=target_size, - ) + if "nir" in role_to_cam: + t0w = time.perf_counter() + nir_img = decoded[role_to_cam["nir"]]["image"] + nir_maps = remap_cache["maps"].get("nir") - warp_details["nir"] = (time.perf_counter() - t0w) * 1000.0 - t_warp_total_ms += warp_details["nir"] + if use_remap_cache and use_remap_for_spec and nir_maps is not None: + self._direct_fusion_write_spec_remap_fast( + tensor=tensor, + channel_index=4, + img=nir_img, + remap_entry=nir_maps, + ) + else: + self._direct_fusion_write_spec_cached_fast( + tensor=tensor, + channel_index=4, + img=nir_img, + role="nir", + geom=geom, + ref_size=ref_size, + target_size=target_size, + ) + + warp_details["nir"] = (time.perf_counter() - t0w) * 1000.0 + t_warp_total_ms += warp_details["nir"] # ------------------------------------------------------------ # Flat-field no espaço final do tensor # ------------------------------------------------------------ - t0_final_flat = time.perf_counter() - final_flat_ms = 0.0 final_flat_enabled = False final_flat_cache_available = False + final_flat_prepare_ms = 0.0 + final_flat_apply_ms = 0.0 flat_cfg = self.flatfield_config or {} apply_final_flat = ( @@ -7333,6 +8029,7 @@ class RawProcessorCore: if apply_final_flat: final_flat_enabled = True + t0_flat_prepare = time.perf_counter() gain_tensor = self._get_final_flat_gain_tensor_cached_fast( decoded=decoded, role_to_cam=role_to_cam, @@ -7341,16 +8038,41 @@ class RawProcessorCore: crop_box=crop_box, target_size=target_size, ) + final_flat_prepare_ms = (time.perf_counter() - t0_flat_prepare) * 1000.0 if gain_tensor is not None: final_flat_cache_available = True + t0_flat_apply = time.perf_counter() tensor = self._apply_final_flat_gain_tensor_inplace( tensor=tensor, gain_tensor=gain_tensor, clip_output=bool(flat_cfg.get("clip_output", True)), ) + final_flat_apply_ms = (time.perf_counter() - t0_flat_apply) * 1000.0 - final_flat_ms = (time.perf_counter() - t0_final_flat) * 1000.0 + self.last_flatfield_result = { + "enabled": True, + "applied": True, + "apply_space": "final_tensor_space", + "backend": "final_tensor_gain_numba" if _HAS_NUMBA else "final_tensor_gain_numpy", + "warnings": ( + ["saturation_guard_not_applied_in_final_tensor_space"] + if bool(flat_cfg.get("saturation_guard_enabled", True)) + else [] + ), + "by_role": { + "rgb": {"applied": True, "channels": ["R", "G", "B"]}, + "re": {"applied": True, "channels": ["RE"]}, + "nir": {"applied": True, "channels": ["NIR"]}, + }, + "applied_roles": ["nir", "re", "rgb"], + "missing_roles": [], + "perf_by_role_ms": {}, + "prepare_ms": float(final_flat_prepare_ms), + "apply_ms": float(final_flat_apply_ms), + } + + final_flat_ms = float(final_flat_prepare_ms + final_flat_apply_ms) if tensor.shape[0] != channels_expected: raise RuntimeError( @@ -7358,15 +8080,29 @@ class RawProcessorCore: ) self.last_fusion_result["output_shape"] = list(tensor.shape) + t_direct_total_ms = (time.perf_counter() - t_total0) * 1000.0 perf = { "prepare_ms": float(t_prepare_ms), + "role_map_ms": float(t_role_map_ms), + "geometry_ms": float(t_geometry_ms), + "tensor_alloc_ms": float(t_tensor_alloc_ms), "rgb_crop_resize_ms": float(t_rgb_ms), + "rgb_remap_ms": float(rgb_remap_perf.get("remap_ms", 0.0)), + "rgb_pack_ms": float(rgb_remap_perf.get("pack_ms", 0.0)), + "rgb_backend": str(rgb_remap_perf.get("backend", "unknown")), "warp_total_ms": float(t_warp_total_ms), "warp_details_ms": warp_details, "crop_resize_ms": float(t_rgb_ms), "concat_ms": 0.0, - "spatial_direct_ms": float(t_rgb_ms + t_warp_total_ms), + "spatial_direct_ms": float(fused5_ms if fused5_used else (t_rgb_ms + t_warp_total_ms)), + "direct_backend_requested": str(direct_backend_requested), + "direct_backend": ( + "numba_fused_5ch_fixedmap" if fused5_used else "opencv_fastpath" + ), + "fused5_used": bool(fused5_used), + "fused5_ms": float(fused5_ms), + "fused5_fallback_reason": str(fused5_fallback_reason or ""), "geometry_cache_hit": bool(geom.get("prepare_cache_hit", False)), "geometry_cache_hits": int(geom.get("cache_hits", 0)), "geometry_cache_misses": int(geom.get("cache_misses", 0)), @@ -7377,15 +8113,20 @@ class RawProcessorCore: "remap_enabled": bool(use_remap_cache), "remap_rgb_enabled": bool(use_remap_cache and use_remap_for_rgb), "remap_spec_enabled": bool(use_remap_cache and use_remap_for_spec), + "rgb_orientation_fused_into_remap": bool(rgb_orientation_fused), + "remap_map_shapes": remap_map_shapes, "final_flat_enabled": bool(final_flat_enabled), "final_flat_cache_available": bool(final_flat_cache_available), "final_flat_ms": float(final_flat_ms), + "final_flat_prepare_ms": float(final_flat_prepare_ms), + "final_flat_apply_ms": float(final_flat_apply_ms), + "target_size": [int(target_w), int(target_h)], + "crop_box": [int(v) for v in crop_box], + "direct_total_ms": float(t_direct_total_ms), } return tensor, perf - - def _direct_fusion_get_geometry_cache_key_fast( self, decoded, @@ -7457,6 +8198,9 @@ class RawProcessorCore: self._direct_fusion_remap_cache_hits = 0 self._direct_fusion_remap_cache_misses = 0 + # Buffers de destino dependem do target final. Recriados sob demanda. + self._direct_fusion_runtime_buffers = {} + def _direct_fusion_get_geometry_cached_fast( self, decoded, @@ -7609,6 +8353,70 @@ class RawProcessorCore: tensor[int(channel_index)] = out + def _direct_fusion_build_warp_affine_fixedmap_fast( + self, + M_src_to_dst: np.ndarray, + dst_size: tuple, + ): + """ + Materializa uma vez o mesmo fixed-point map usado por cv2.warpAffine + com INTER_LINEAR. Isso permite ao kernel fused Numba reproduzir também + o fast-path RGB atual, em vez de voltar à convenção do cv2.remap. + """ + M = np.asarray(M_src_to_dst, dtype=np.float64) + if M.shape == (3, 3): + A = M[:2, :] + elif M.shape == (2, 3): + A = M + else: + raise RuntimeError(f"M affine inválida: shape={M.shape}") + + target_w, target_h = int(dst_size[0]), int(dst_size[1]) + inv = cv2.invertAffineTransform(A).astype(np.float64, copy=False) + + inter_bits = 5 + inter_tab = 1 << inter_bits + ab_bits = max(10, inter_bits) + ab_scale = 1 << ab_bits + round_delta = ab_scale // inter_tab // 2 + shift = ab_bits - inter_bits + + xs = np.arange(target_w, dtype=np.float64) + ys = np.arange(target_h, dtype=np.float64) + + # np.rint reproduz o arredondamento para o inteiro mais próximo usado + # pelo cvRound/saturate_cast neste domínio de coordenadas. + adelta = np.rint(inv[0, 0] * xs * ab_scale).astype(np.int64) + bdelta = np.rint(inv[1, 0] * xs * ab_scale).astype(np.int64) + + x0 = ( + np.rint((inv[0, 1] * ys + inv[0, 2]) * ab_scale).astype(np.int64) + + int(round_delta) + ) + y0 = ( + np.rint((inv[1, 1] * ys + inv[1, 2]) * ab_scale).astype(np.int64) + + int(round_delta) + ) + + X = (x0[:, None] + adelta[None, :]) >> shift + Y = (y0[:, None] + bdelta[None, :]) >> shift + + ix = X >> inter_bits + iy = Y >> inter_bits + fx = X & (inter_tab - 1) + fy = Y & (inter_tab - 1) + + # O produto opera em coordenadas pequenas (< 2k), mas clamp explícito + # preserva o contrato de armazenamento CV_16SC2. + i16 = np.iinfo(np.int16) + map1 = np.empty((target_h, target_w, 2), dtype=np.int16) + map1[:, :, 0] = np.clip(ix, i16.min, i16.max).astype(np.int16) + map1[:, :, 1] = np.clip(iy, i16.min, i16.max).astype(np.int16) + map2 = (fy * inter_tab + fx).astype(np.uint16) + + return map1, map2 + + def _direct_fusion_build_remap_from_src_to_dst_fast( self, M_src_to_dst: np.ndarray, @@ -7672,15 +8480,41 @@ class RawProcessorCore: # convertMaps deixa o remap mais barato em muitos casos. map1, map2 = cv2.convertMaps(map_x, map_y, cv2.CV_16SC2) - return { + entry = { "map_x": map_x, "map_y": map_y, "map1": map1, "map2": map2, + # Mantemos também a matriz forward original. Para RGB, quando ela + # for affine e axis-aligned (crop/scale + 0/180/flips), podemos usar + # warpAffine no runtime e evitar ler os mapas densos por frame. + "M_src_to_dst": M.astype(np.float32, copy=False), "src_shape": [src_h, src_w], "dst_size": [target_w, target_h], } + eps = 1e-7 + axis_affine = bool( + abs(float(M[0, 1])) <= eps + and abs(float(M[1, 0])) <= eps + and abs(float(M[2, 0])) <= eps + and abs(float(M[2, 1])) <= eps + and abs(float(M[2, 2]) - 1.0) <= eps + ) + + if axis_affine: + warp_map1, warp_map2 = self._direct_fusion_build_warp_affine_fixedmap_fast( + M_src_to_dst=M, + dst_size=(target_w, target_h), + ) + entry["warp_affine_map1"] = warp_map1 + entry["warp_affine_map2"] = warp_map2 + entry["axis_affine"] = True + else: + entry["axis_affine"] = False + + return entry + def _direct_fusion_get_remap_cache_key_fast( self, geom: dict, @@ -7704,11 +8538,18 @@ class RawProcessorCore: img = decoded[cam_id]["image"] role_shapes[str(role).lower()] = tuple(int(v) for v in img.shape[:2]) + rgb_orientation_sig = self._decoded_role_orientation_signature_fast( + decoded, + role_to_cam, + "rgb", + ) + return ( geom.get("key"), tuple(int(v) for v in ref_size), tuple(int(v) for v in target_size), tuple(sorted(role_shapes.items())), + rgb_orientation_sig, ) def _direct_fusion_get_remap_cached_fast( @@ -7763,8 +8604,36 @@ class RawProcessorCore: rgb_cam_id = role_to_cam.get("rgb") if rgb_cam_id is not None: rgb_img = decoded[rgb_cam_id]["image"] + + # geom["C_ref_to_target"] parte do espaço RGB CANÔNICO (após + # camera_orientation). Quando a orientação foi diferida, compomos: + # + # RGB_native --O--> RGB_canonical --C--> target + # + # Como cv2.remap usa o inverso internamente, o target passa a + # amostrar diretamente o frame nativo já na orientação correta. + M_rgb_to_target = geom["C_ref_to_target"] + if self._decoded_role_orientation_is_deferred_fast( + decoded, role_to_cam, "rgb" + ): + meta_rgb = decoded[rgb_cam_id].get("meta", {}) or {} + ori = meta_rgb.get("orientation_applied", {}) or {} + O_native_to_canonical = self._orientation_matrix_same_shape_fast( + src_shape=rgb_img.shape[:2], + rotate_deg=int(ori.get("rotate_deg", 0)), + flip_horizontal=bool(ori.get("flip_horizontal", False)), + flip_vertical=bool(ori.get("flip_vertical", False)), + ) + if O_native_to_canonical is None: + raise RuntimeError( + "Orientação RGB diferida não suportada pelo remap composto." + ) + M_rgb_to_target = ( + geom["C_ref_to_target"] @ O_native_to_canonical + ).astype(np.float32) + remap["maps"]["rgb"] = self._direct_fusion_build_remap_from_src_to_dst_fast( - M_src_to_dst=geom["C_ref_to_target"], + M_src_to_dst=M_rgb_to_target, src_shape=rgb_img.shape[:2], dst_size=(target_w, target_h), ref_shape_for_scaled_src=None, @@ -7794,6 +8663,43 @@ class RawProcessorCore: cache[key] = remap return remap + def _direct_fusion_get_rgb_scratch_fast(self, target_h: int, target_w: int): + """ + Buffer HWC float32 reutilizável para o remap RGB. + + O tensor final é CHW, enquanto cv2.remap é mais eficiente processando o + RGB intercalado em uma única chamada. Reutilizamos este scratch para + eliminar a alocação HWC por frame e depois usamos cv2.mixChannels para + copiar os 3 canais diretamente para os planos CHW do tensor. + """ + target_h = int(target_h) + target_w = int(target_w) + key = (target_h, target_w) + + cache = getattr(self, "_direct_fusion_runtime_buffers", None) + if not isinstance(cache, dict): + cache = {} + self._direct_fusion_runtime_buffers = cache + + entry = cache.get(key) + if ( + not isinstance(entry, dict) + or not isinstance(entry.get("rgb_hwc"), np.ndarray) + or entry["rgb_hwc"].shape != (target_h, target_w, 3) + or entry["rgb_hwc"].dtype != np.float32 + ): + entry = { + "rgb_hwc": np.empty( + (target_h, target_w, 3), + dtype=np.float32, + ) + } + if len(cache) >= 4: + cache.clear() + cache[key] = entry + + return entry["rgb_hwc"] + def _direct_fusion_write_rgb_remap_fast( self, tensor: np.ndarray, @@ -7801,23 +8707,81 @@ class RawProcessorCore: remap_entry: dict, ): """ - RGB HWC -> tensor CHW usando cv2.remap direto para target final. + RGB HWC -> tensor CHW usando o mesmo cv2.remap do caminho anterior, + porém sem alocar um HWC novo em todo frame. + + Matemática/interpolação permanecem idênticas: mesmos map1/map2, + INTER_LINEAR e BORDER_CONSTANT. Somente o destino e o empacotamento + HWC->CHW foram otimizados. """ map1 = remap_entry["map1"] map2 = remap_entry["map2"] - rgb_out = cv2.remap( - rgb.astype(np.float32, copy=False), - map1, - map2, - interpolation=cv2.INTER_LINEAR, - borderMode=cv2.BORDER_CONSTANT, - borderValue=0.0, + target_h, target_w = int(map1.shape[0]), int(map1.shape[1]) + rgb_out = self._direct_fusion_get_rgb_scratch_fast( + target_h, + target_w, ) - tensor[0] = rgb_out[:, :, 0] - tensor[1] = rgb_out[:, :, 1] - tensor[2] = rgb_out[:, :, 2] + rgb_f = rgb.astype(np.float32, copy=False) + M = remap_entry.get("M_src_to_dst") + + # Fast-path exato para a geometria RGB de produto atual: + # crop/scale + orientação 0/180/flips => matriz affine axis-aligned. + # Para essa família, OpenCV warpAffine e o remap fixo gerado por + # convertMaps produzem os mesmos pixels, mas warpAffine evita ler + # map1/map2 densos a cada frame. + use_axis_affine = False + if isinstance(M, np.ndarray) and M.shape == (3, 3): + eps = 1e-7 + use_axis_affine = bool( + abs(float(M[0, 1])) <= eps + and abs(float(M[1, 0])) <= eps + and abs(float(M[2, 0])) <= eps + and abs(float(M[2, 1])) <= eps + and abs(float(M[2, 2]) - 1.0) <= eps + ) + + t0 = time.perf_counter() + if use_axis_affine: + cv2.warpAffine( + rgb_f, + M[:2], + (target_w, target_h), + dst=rgb_out, + flags=cv2.INTER_LINEAR, + borderMode=cv2.BORDER_CONSTANT, + borderValue=0.0, + ) + backend = "opencv_warpAffine_axis_aligned_cached_hwc" + else: + cv2.remap( + rgb_f, + map1, + map2, + interpolation=cv2.INTER_LINEAR, + dst=rgb_out, + borderMode=cv2.BORDER_CONSTANT, + borderValue=0.0, + ) + backend = "opencv_remap_cached_hwc" + remap_ms = (time.perf_counter() - t0) * 1000.0 + + t0 = time.perf_counter() + # mixChannels faz o HWC -> 3 planos CHW em uma única rotina OpenCV, + # evitando três atribuições NumPy independentes. + cv2.mixChannels( + [rgb_out], + [tensor[0], tensor[1], tensor[2]], + [0, 0, 1, 1, 2, 2], + ) + pack_ms = (time.perf_counter() - t0) * 1000.0 + + return { + "remap_ms": float(remap_ms), + "pack_ms": float(pack_ms), + "backend": backend + "_mixchannels", + } def _direct_fusion_write_spec_remap_fast( self, @@ -7827,21 +8791,26 @@ class RawProcessorCore: remap_entry: dict, ): """ - RE/NIR -> tensor usando cv2.remap direto para target final. + RE/NIR -> tensor usando cv2.remap diretamente no plano CHW final. + + Evita o ndarray temporário 960x600 e a cópia subsequente para + tensor[channel_index], sem alterar a interpolação. """ map1 = remap_entry["map1"] map2 = remap_entry["map2"] + dst = tensor[int(channel_index)] - out = cv2.remap( + cv2.remap( img.astype(np.float32, copy=False), map1, map2, interpolation=cv2.INTER_LINEAR, + dst=dst, borderMode=cv2.BORDER_CONSTANT, borderValue=0.0, ) - tensor[int(channel_index)] = out + return dst def _resolve_homography_profile_name_for_role(self, role: str) -> str: """ @@ -8156,30 +9125,25 @@ class RawProcessorCore: return rgb.astype(np.float32, copy=False) def _get_bayer_cv2_code(self, bayer_pattern: str | None, algorithm: str = "ea"): - """ - Retorna o código OpenCV para demosaic Bayer. - - algorithm: - - "ea": Edge-Aware, melhor qualidade, mais pesado - - "bilinear": mais rápido, menor custo - """ p = str(bayer_pattern or self.bayer_pattern or "RGGB").upper() algo = str(algorithm or "ea").lower() if algo in ("bilinear", "linear", "fast", "normal"): code_map = { - "BGGR": cv2.COLOR_BayerRG2RGB, - "RGGB": cv2.COLOR_BayerBG2RGB, - "GRBG": cv2.COLOR_BayerGR2RGB, - "GBRG": cv2.COLOR_BayerGB2RGB, + "RGGB": cv2.COLOR_BayerRGGB2RGB, + "BGGR": cv2.COLOR_BayerBGGR2RGB, + "GRBG": cv2.COLOR_BayerGRBG2RGB, + "GBRG": cv2.COLOR_BayerGBRG2RGB, } + elif algo in ("ea", "edge_aware", "edge-aware"): code_map = { - "BGGR": cv2.COLOR_BayerRG2RGB_EA, - "RGGB": cv2.COLOR_BayerBG2RGB_EA, - "GRBG": cv2.COLOR_BayerGR2RGB_EA, - "GBRG": cv2.COLOR_BayerGB2RGB_EA, + "RGGB": cv2.COLOR_BayerRGGB2RGB_EA, + "BGGR": cv2.COLOR_BayerBGGR2RGB_EA, + "GRBG": cv2.COLOR_BayerGRBG2RGB_EA, + "GBRG": cv2.COLOR_BayerGBRG2RGB_EA, } + else: raise ValueError(f"demosaic_algorithm inválido: {algorithm}") @@ -8210,13 +9174,25 @@ class RawProcessorCore: self._last_decode_perf_log_ts = now - #parts = [] - #for k, v in perf.items(): - # if isinstance(v, (int, float)): - # parts.append(f"{k}={float(v):.2f}ms") - # else: - # parts.append(f"{k}={v}") - #print(f"[PERF][DECODE][{role.upper()}] " + " ".join(parts)) + if bool(getattr(self, "core_perf_debug", True)): + # Log compacto. O agregado de RGB/RE/NIR também aparece em CORE_ROLES. + numeric_keys = ( + "total_ms", "unpack_ms", "cvtColor_ms", "float_ms", + "calibration_ms", "clip_ms", "resize_half_ms" + ) + parts = [] + for k in numeric_keys: + v = perf.get(k) + if isinstance(v, (int, float)): + parts.append(f"{k}={float(v):.2f}ms") + shape = perf.get("out_shape") + mode = perf.get("mode") + print( + f"[PERF][DECODE][{role.upper()}] " + + " ".join(parts) + + (f" mode={mode}" if mode is not None else "") + + (f" out={shape}" if shape is not None else "") + ) @@ -8334,7 +9310,6 @@ class RawProcessorCore: strength = float(cfg.get("strength", 1.0)) strength_by_channel = cfg.get("strength_by_channel", {}) or {} - gain_min_runtime = float(cfg.get("gain_min_runtime", 0.0)) gain_max_runtime = float(cfg.get("gain_max_runtime", 999.0)) @@ -8343,7 +9318,7 @@ class RawProcessorCore: runtime_smooth_ksize += 1 key = ( - "final_flat_gain_tensor", + "final_flat_gain_tensor_v2_same_direct_maps", geom.get("key"), tuple(int(v) for v in crop_box), int(target_w), @@ -8360,25 +9335,56 @@ class RawProcessorCore: return cached gain_tensor = np.ones((5, target_h, target_w), dtype=np.float32) + maps = (remap_cache or {}).get("maps", {}) or {} + x0, y0, x1, y1 = [int(v) for v in crop_box] - # ------------------------------------------------------------ - # RGB: usa crop + resize, igual ao RGB real. - # ------------------------------------------------------------ - rgb_cam = role_to_cam.get("rgb") - if rgb_cam is not None: - rgb_img = decoded[rgb_cam]["image"] - rgb_shape = rgb_img.shape[:2] + # O gain de cada banda atravessa exatamente a mesma transformação + # espacial da banda real. Isso é especialmente importante no RGB, + # cuja orientação 180° pode estar fundida no mapa do direct. + channel_specs = ( + ("rgb", "R", 0), + ("rgb", "G", 1), + ("rgb", "B", 2), + ("re", "RE", 3), + ("nir", "NIR", 4), + ) - x0, y0, x1, y1 = [int(v) for v in crop_box] + for role, ch, ci in channel_specs: + cam_id = role_to_cam.get(role) + if cam_id is None: + continue - for ci, ch in enumerate(("R", "G", "B")): - gain_eff = self._get_runtime_gain_eff_map(ch, rgb_shape, cfg) + img = decoded[cam_id]["image"] + gain_eff = self._get_runtime_gain_eff_map(ch, img.shape[:2], cfg) + if gain_eff is None: + continue - if gain_eff is None: + gain_eff = gain_eff.astype(np.float32, copy=False) + remap_entry = maps.get(role) + + if isinstance(remap_entry, dict): + if role == "rgb": + map1 = remap_entry.get("warp_affine_map1", remap_entry.get("map1")) + map2 = remap_entry.get("warp_affine_map2", remap_entry.get("map2")) + else: + map1 = remap_entry.get("map1") + map2 = remap_entry.get("map2") + + if isinstance(map1, np.ndarray) and isinstance(map2, np.ndarray): + gain_out = cv2.remap( + gain_eff, + map1, + map2, + interpolation=cv2.INTER_LINEAR, + borderMode=cv2.BORDER_CONSTANT, + borderValue=1.0, + ) + gain_tensor[ci] = gain_out.astype(np.float32, copy=False) continue - gain_crop = gain_eff[y0:y1, x0:x1].astype(np.float32, copy=False) - + # Fallback conservador se algum perfil rodar sem remap cache. + if role == "rgb": + gain_crop = gain_eff[y0:y1, x0:x1] if gain_crop.shape[1] != target_w or gain_crop.shape[0] != target_h: gain_out = cv2.resize( gain_crop, @@ -8387,49 +9393,12 @@ class RawProcessorCore: ) else: gain_out = gain_crop - - gain_tensor[ci] = gain_out.astype(np.float32, copy=False) - - # ------------------------------------------------------------ - # RE/NIR: usa o mesmo remap cacheado da imagem real. - # ------------------------------------------------------------ - maps = (remap_cache or {}).get("maps", {}) or {} - - spec_map = { - "re": ("RE", 3), - "nir": ("NIR", 4), - } - - for role, (ch, ci) in spec_map.items(): - cam_id = role_to_cam.get(role) - if cam_id is None: - continue - - img = decoded[cam_id]["image"] - gain_eff = self._get_runtime_gain_eff_map(ch, img.shape[:2], cfg) - - if gain_eff is None: - continue - - remap_entry = maps.get(role) - - if remap_entry is not None: - gain_out = cv2.remap( - gain_eff.astype(np.float32, copy=False), - remap_entry["map1"], - remap_entry["map2"], - interpolation=cv2.INTER_LINEAR, - borderMode=cv2.BORDER_CONSTANT, - borderValue=1.0, - ) else: M_role_to_target = geom["M_role_to_target"].get(role) - if M_role_to_target is None: continue - gain_out = cv2.warpPerspective( - gain_eff.astype(np.float32, copy=False), + gain_eff, M_role_to_target, (target_w, target_h), flags=cv2.INTER_LINEAR, @@ -8439,10 +9408,138 @@ class RawProcessorCore: gain_tensor[ci] = gain_out.astype(np.float32, copy=False) + if not gain_tensor.flags.c_contiguous: + gain_tensor = np.ascontiguousarray(gain_tensor) + self._flatfield_runtime_cache[key] = gain_tensor return gain_tensor + def _orientation_matrix_same_shape_fast( + self, + src_shape, + rotate_deg=0, + flip_horizontal=False, + flip_vertical=False, + ): + """ + Matriz 3x3 source/native -> espaço canônico para orientações que + preservam H/W (0° e 180°, com flips opcionais). + + A ordem replica _apply_orientation_image(): rotate, flip H, flip V. + 90°/270° retornam None para cair no caminho físico tradicional. + """ + h, w = int(src_shape[0]), int(src_shape[1]) + r = int(rotate_deg) % 360 + + if r not in (0, 180): + return None + + M = np.eye(3, dtype=np.float32) + + if r == 180: + R = np.array( + [ + [-1.0, 0.0, float(w - 1)], + [0.0, -1.0, float(h - 1)], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = R @ M + + if bool(flip_horizontal): + Fh = np.array( + [ + [-1.0, 0.0, float(w - 1)], + [0.0, 1.0, 0.0], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = Fh @ M + + if bool(flip_vertical): + Fv = np.array( + [ + [1.0, 0.0, 0.0], + [0.0, -1.0, float(h - 1)], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = Fv @ M + + return M.astype(np.float32, copy=False) + + def _can_defer_rgb_orientation_to_direct_remap_fast(self, decoded): + """ + Decide se a orientação RGB pode ser incorporada ao remap final. + + Conservador por design: + - somente role RGB; + - somente 0°/180° (mesmo H/W); + - exige remap cache ativo; + - pode ser desabilitado por fusion_config.fuse_orientation_into_remap=false. + """ + cfg = self.camera_orientation_config or {} + fusion = self.fusion_config or {} + + if not bool(cfg.get("enabled", False)): + return False + if not bool(fusion.get("fuse_orientation_into_remap", True)): + return False + if not bool(fusion.get("use_remap_cache", True)): + return False + + by_role = cfg.get("by_role", {}) or {} + role_cfg = by_role.get("rgb", {}) or {} + r = int(role_cfg.get("rotate_deg", 0)) % 360 + fh = bool(role_cfg.get("flip_horizontal", False)) + fv = bool(role_cfg.get("flip_vertical", False)) + + if r not in (0, 180): + return False + if r == 0 and not fh and not fv: + return False + + rgb_cam_id = self._find_cam_by_role(decoded, "rgb") + if rgb_cam_id is None: + return False + img = decoded[rgb_cam_id].get("image") + if img is None or img.ndim < 2: + return False + + return self._orientation_matrix_same_shape_fast( + img.shape[:2], r, fh, fv + ) is not None + + def _decoded_role_orientation_signature_fast( + self, decoded, role_to_cam, role + ): + role = str(role).lower() + cam_id = role_to_cam.get(role) + if cam_id is None: + return (role, False, 0, False, False) + meta = decoded[cam_id].get("meta", {}) or {} + ori = meta.get("orientation_applied", {}) or {} + return ( + role, + bool(ori.get("deferred_to_direct_fusion", False)), + int(ori.get("rotate_deg", 0)) % 360, + bool(ori.get("flip_horizontal", False)), + bool(ori.get("flip_vertical", False)), + ) + + def _decoded_role_orientation_is_deferred_fast( + self, decoded, role_to_cam, role + ): + return bool( + self._decoded_role_orientation_signature_fast( + decoded, role_to_cam, role + )[1] + ) + def _apply_orientation_image( self, img, @@ -8494,8 +9591,10 @@ class RawProcessorCore: def apply_camera_orientation_to_decoded( self, decoded, + defer_roles=None, ): cfg = self.camera_orientation_config or {} + defer_roles = {str(x).lower() for x in (defer_roles or set())} result = { "enabled": bool( @@ -8503,6 +9602,7 @@ class RawProcessorCore: ), "applied": False, "by_role": {}, + "deferred_roles": [], } if not cfg.get("enabled", False): @@ -8515,8 +9615,10 @@ class RawProcessorCore: ) or {} out = {} + perf_by_role_ms = {} for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str( item.get("role") @@ -8557,27 +9659,35 @@ class RawProcessorCore: item.get("meta", {}) or {} ) - if ( + wants_orientation = ( + rotate_deg != 0 + or flip_h + or flip_v + ) + deferred_here = bool( image is not None - and ( - rotate_deg != 0 - or flip_h - or flip_v - ) - ): + and role in defer_roles + and wants_orientation + ) + + if image is not None and wants_orientation and not deferred_here: image = self._apply_orientation_image( image, rotate_deg=rotate_deg, flip_horizontal=flip_h, flip_vertical=flip_v, ) - result["applied"] = True + if deferred_here: + result["deferred_roles"].append(role) + new_meta["orientation_applied"] = { "rotate_deg": rotate_deg, "flip_horizontal": flip_h, "flip_vertical": flip_v, + "physically_applied": bool(wants_orientation and not deferred_here), + "deferred_to_direct_fusion": bool(deferred_here), } new_item["image"] = image @@ -8585,18 +9695,24 @@ class RawProcessorCore: out[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + perf_by_role_ms[role] = float(role_elapsed_ms) + result["by_role"][role] = { "camera_id": cam_id, "rotate_deg": rotate_deg, "flip_horizontal": flip_h, "flip_vertical": flip_v, + "deferred_to_direct_fusion": bool(deferred_here), "shape": ( list(image.shape) if image is not None else None ), + "perf_ms": float(role_elapsed_ms), } + result["perf_by_role_ms"] = dict(perf_by_role_ms) self.last_orientation_result = result return out diff --git a/AgroBase/OperationControl/Services/GpsService.cs b/AgroBase/OperationControl/Services/GpsService.cs index 58802628d..1e67c0765 100644 --- a/AgroBase/OperationControl/Services/GpsService.cs +++ b/AgroBase/OperationControl/Services/GpsService.cs @@ -129,7 +129,7 @@ namespace OperationControl.Services public int NtripPort { get; set; } = 2101; public string NtripMountpoint { get; set; } = "EESC0"; public string NtripUsername { get; set; } = "Zendion"; // Environment.GetEnvironmentVariable("AGRO_NTRIP_USERNAME") ?? string.Empty; - public string NtripPassword { get; set; } = "c3pc7*9N"; //Environment.GetEnvironmentVariable("AGRO_NTRIP_PASSWORD") ?? string.Empty; + public string NtripPassword { get; set; } = "kY3zd$*5"; //Environment.GetEnvironmentVariable("AGRO_NTRIP_PASSWORD") ?? string.Empty; private GnssExpectedRole _expectedRole = GnssExpectedRole.PreserveCurrentConfiguration; private BaseFixedConfiguration _lastBaseConfiguration;