diff --git a/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py b/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py index 6282d4e94..43d923868 100644 --- a/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py +++ b/AgroBase/AgroBase/bin/x64/Debug/Python/Scripts/workers/camera_worker/oak_fcc3_core/raw_processor_core.py @@ -28,7 +28,7 @@ module_params is loaded. The strict fail-closed behavior is enabled only for module_params generated by the production assembler. """ -RAW_PROCESSOR_CORE_VERSION = "production_v2_2026_09_08" +RAW_PROCESSOR_CORE_VERSION = "production_v2_2026_09_12_radiometric_singlepass" import json import os @@ -339,6 +339,95 @@ if _HAS_NUMBA: out[y, x] = value + @_numba.njit(cache=True, fastmath=False, parallel=True) + def _apply_radiometric_scale_inplace_numba(img, scale, clip_output): + """ + Radiometric scale em uma unica passada de memoria. + + Mantem a mesma ordem matematica do caminho NumPy e opera apenas em + imagens 2D/3D float32 C-contiguous. + """ + h = img.shape[0] + + if img.ndim == 2: + w = img.shape[1] + for y in _numba.prange(h): + for x in range(w): + v = img[y, x] * scale + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x] = v + + else: + w = img.shape[1] + c = img.shape[2] + for y in _numba.prange(h): + for x in range(w): + for k in range(c): + v = img[y, x, k] * scale + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x, k] = v + + + @_numba.njit(cache=True, fastmath=False, parallel=True) + def _apply_radiometric_affine_inplace_numba(img, scale, offset, clip_output): + """ + Radiometric affine_v2 em uma unica passada de memoria. + + IMPORTANTE: usa explicitamente a mesma ordem de operacoes do contrato: + (value - offset) * scale + offset + + fastmath=False evita reassociacao/FMA e preserva, na pratica, o mesmo + resultado float32 do caminho NumPy de tres ufuncs. + """ + h = img.shape[0] + + if img.ndim == 2: + w = img.shape[1] + for y in _numba.prange(h): + for x in range(w): + v = img[y, x] + v = (v - offset) * scale + v = v + offset + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x] = v + + else: + w = img.shape[1] + c = img.shape[2] + for y in _numba.prange(h): + for x in range(w): + for k in range(c): + v = img[y, x, k] + v = (v - offset) * scale + v = v + offset + + if clip_output: + if v < 0.0: + v = 0.0 + elif v > 1.0: + v = 1.0 + + img[y, x, k] = v + + @_numba.njit(cache=True, fastmath=True, parallel=True) def _apply_tensor_flat_gain_chw_numba(tensor, gain, channels, height, width, clip_output): for c in _numba.prange(channels): @@ -613,6 +702,36 @@ class RawProcessorCore: self._last_decode_perf_log_ts = 0.0 self.last_rgb_enhancement_result = None + # ============================================================ + # PERF DIAGNOSTICS - temporariamente habilitado para campo + # ============================================================ + # O profiler é intencionalmente leve: usa apenas perf_counter() e + # metadados já calculados pelo pipeline. Não calcula estatísticas + # adicionais sobre os pixels e não altera o tensor. + # + # Pode ser desligado sem editar código: + # OAK_CORE_PERF_DEBUG=0 + # Intervalos: + # OAK_CORE_PERF_LOG_INTERVAL_S=1.0 + # OAK_CORE_SHAPES_LOG_INTERVAL_S=5.0 + self.core_perf_debug = str( + os.getenv("OAK_CORE_PERF_DEBUG", "1") + ).strip().lower() not in ("0", "false", "no", "off") + try: + self.core_perf_log_interval_s = max(0.2, float( + os.getenv("OAK_CORE_PERF_LOG_INTERVAL_S", "1.0") + )) + except Exception: + self.core_perf_log_interval_s = 1.0 + try: + self.core_shapes_log_interval_s = max(1.0, float( + os.getenv("OAK_CORE_SHAPES_LOG_INTERVAL_S", "5.0") + )) + except Exception: + self.core_shapes_log_interval_s = 5.0 + self._last_core_perf_log_ts = 0.0 + self._last_core_shapes_log_ts = 0.0 + # QC: percentis são diagnóstico, não parte do tensor. # Mantemos saturação/dark/range exatos e limitamos apenas estatísticas # descritivas para não gastar dezenas/centenas de ms por frame. @@ -1813,12 +1932,25 @@ class RawProcessorCore: channels_expected, target_size=None, ): + """ + Fusão Raw5 com telemetria de performance por estágio. + + IMPORTANTE: este profiler não altera a matemática do pipeline. Ele apenas + mede os blocos já existentes e emite logs rate-limited para localizar o + gargalo real no hardware OV9782/AR0234. + """ t_total0 = time.perf_counter() + # ------------------------------------------------------------ + # 1) Dark + # ------------------------------------------------------------ t0 = time.perf_counter() decoded = self.apply_dark_to_decoded(decoded) t_dark_ms = (time.perf_counter() - t0) * 1000.0 + # ------------------------------------------------------------ + # 2) Normalização radiométrica por controles da captura + # ------------------------------------------------------------ t0 = time.perf_counter() decoded = self.normalize_decoded_by_capture_controls(decoded, meta) t_radnorm_ms = (time.perf_counter() - t0) * 1000.0 @@ -1831,26 +1963,50 @@ class RawProcessorCore: and flat_apply_space != "final_tensor_space" ) + # ------------------------------------------------------------ + # 3) Flat-field nativo + # ------------------------------------------------------------ t0 = time.perf_counter() - if apply_native_flat: decoded = self.apply_flat_gain_to_decoded(decoded) - - # ★ NOVO ★ - decoded = self.apply_camera_orientation_to_decoded(decoded) - t_flat_native_ms = (time.perf_counter() - t0) * 1000.0 - # Caminho espacial otimizado: direto para o target final. + # ------------------------------------------------------------ + # 4) Orientação das câmeras + # ------------------------------------------------------------ + # Em produto, uma orientação RGB de 180° pode ser fundida no remap + # RGB final. Isso preserva o espaço canônico usado pelas homografias + # sem materializar uma cópia HWC full-res apenas para cv2.rotate(). + # 90°/270° continuam no caminho físico tradicional porque trocam H/W. + defer_roles = set() + if self._can_defer_rgb_orientation_to_direct_remap_fast(decoded): + defer_roles.add("rgb") + + t0 = time.perf_counter() + decoded = self.apply_camera_orientation_to_decoded( + decoded, + defer_roles=defer_roles, + ) + t_orientation_ms = (time.perf_counter() - t0) * 1000.0 + + # ------------------------------------------------------------ + # 5) Fusão espacial direta para o target final + # ------------------------------------------------------------ + t0 = time.perf_counter() tensor, direct_perf = self._fuse_multispec_direct_to_target_fast( decoded, meta, channels_expected, target_size_override=target_size, ) + t_direct_call_ms = (time.perf_counter() - t0) * 1000.0 - t_flat_ms = float(t_flat_native_ms + float(direct_perf.get("final_flat_ms", 0.0))) + final_flat_ms = float(direct_perf.get("final_flat_ms", 0.0)) + t_flat_ms = float(t_flat_native_ms + final_flat_ms) + # ------------------------------------------------------------ + # 6) Validação/finalização dtype + # ------------------------------------------------------------ t0 = time.perf_counter() if tensor.shape[0] != channels_expected: raise RuntimeError( @@ -1860,18 +2016,15 @@ class RawProcessorCore: out = tensor.astype(np.float32, copy=False) t_final_ms = (time.perf_counter() - t0) * 1000.0 - t_total_ms = (time.perf_counter() - t_total0) * 1000.0 - # Mantém contrato de perf do benchmark. + # Contrato histórico de perf + detalhes novos. t_prepare_ms = float(direct_perf.get("prepare_ms", 0.0)) + t_geometry_ms = float(direct_perf.get("geometry_ms", 0.0)) + t_tensor_alloc_ms = float(direct_perf.get("tensor_alloc_ms", 0.0)) t_rgb_ms = float(direct_perf.get("rgb_crop_resize_ms", 0.0)) t_warp_total_ms = float(direct_perf.get("warp_total_ms", 0.0)) warp_details = direct_perf.get("warp_details_ms", {}) or {} - - # Aqui crop_resize_ms representa somente RGB crop/resize no modo direto. - # O custo espacial total útil para analisar é: - # spatial_direct_ms = rgb_crop_resize_ms + warp_total_ms t_crop_resize_ms = float(direct_perf.get("crop_resize_ms", t_rgb_ms)) t_concat_ms = 0.0 @@ -1883,60 +2036,129 @@ class RawProcessorCore: self.last_fusion_result["perf"] = { "dark_ms": t_dark_ms, "radnorm_ms": t_radnorm_ms, + "flat_native_ms": t_flat_native_ms, + "orientation_ms": t_orientation_ms, "flat_ms": t_flat_ms, + "direct_call_ms": t_direct_call_ms, "prepare_ms": t_prepare_ms, + "geometry_ms": t_geometry_ms, + "tensor_alloc_ms": t_tensor_alloc_ms, + "remap_cache_ms": float(direct_perf.get("remap_cache_ms", 0.0)), "rgb_crop_resize_ms": t_rgb_ms, "warp_total_ms": t_warp_total_ms, "warp_details_ms": warp_details, "crop_resize_ms": t_crop_resize_ms, "concat_ms": t_concat_ms, "spatial_direct_ms": float(t_rgb_ms + t_warp_total_ms), + "final_flat_ms": final_flat_ms, "final_ms": t_final_ms, "total_ms": t_total_ms, "decode_perf": getattr(self, "last_decode_perf", {}), + "flat_perf_by_role": ( + (getattr(self, "last_flatfield_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "radnorm_perf_by_role": ( + (getattr(self, "last_radiometric_normalization_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "orientation_perf_by_role": ( + (getattr(self, "last_orientation_result", {}) or {}).get("perf_by_role_ms", {}) + ), + "orientation_deferred_roles": list( + (getattr(self, "last_orientation_result", {}) or {}).get("deferred_roles", []) or [] + ), + "rgb_orientation_fused_into_remap": bool( + (direct_perf or {}).get("rgb_orientation_fused_into_remap", False) + ), } - #if not hasattr(self, "_last_perf_log_ts"): - # self._last_perf_log_ts = 0.0 - #now = time.time() - #if now - self._last_perf_log_ts >= 1.0: - # self._last_perf_log_ts = now - # print( - # "[PERF][CORE_FUSE] " - # f"dark={t_dark_ms:.1f}ms " - # f"radnorm={t_radnorm_ms:.1f}ms " - # f"flat={t_flat_ms:.1f}ms " - # f"prepare={t_prepare_ms:.1f}ms " - # f"rgb_resize={t_rgb_ms:.1f}ms " - # f"warp={t_warp_total_ms:.1f}ms " - # f"warp_re={warp_details.get('re', -1):.1f}ms " - # f"warp_nir={warp_details.get('nir', -1):.1f}ms " - # f"spatial={t_rgb_ms + t_warp_total_ms:.1f}ms " - # f"crop_resize={t_crop_resize_ms:.1f}ms " - # f"concat={t_concat_ms:.1f}ms " - # f"final={t_final_ms:.1f}ms " - # f"total={t_total_ms:.1f}ms " - # f"shape={out.shape}" - # ) + # ------------------------------------------------------------ + # LOG PRINCIPAL: 1x/s por padrão + # ------------------------------------------------------------ + if bool(getattr(self, "core_perf_debug", True)): + now = time.time() + last = float(getattr(self, "_last_core_perf_log_ts", 0.0) or 0.0) + interval = float(getattr(self, "core_perf_log_interval_s", 1.0) or 1.0) - #if not hasattr(self, "_last_remap_perf_log_ts"): - # self._last_remap_perf_log_ts = 0.0 - #now = time.time() - #if now - self._last_remap_perf_log_ts >= 1.0: - # self._last_remap_perf_log_ts = now - # print( - # "[PERF][REMAP] " - # f"enabled={direct_perf.get('remap_enabled')} " - # f"hit={direct_perf.get('remap_cache_hit')} " - # f"cache={direct_perf.get('remap_cache_ms', -1):.2f}ms " - # f"rgb={direct_perf.get('rgb_crop_resize_ms', -1):.2f}ms " - # f"warp={direct_perf.get('warp_total_ms', -1):.2f}ms " - # f"re={(direct_perf.get('warp_details_ms') or {}).get('re', -1):.2f}ms " - # f"nir={(direct_perf.get('warp_details_ms') or {}).get('nir', -1):.2f}ms " - # f"spatial={direct_perf.get('spatial_direct_ms', -1):.2f}ms " - # f"hits={direct_perf.get('remap_cache_hits')} " - # f"misses={direct_perf.get('remap_cache_misses')}" - # ) + if now - last >= interval: + self._last_core_perf_log_ts = now + + decode_perf = getattr(self, "last_decode_perf", {}) or {} + decode_rgb = float((decode_perf.get("rgb") or {}).get("total_ms", 0.0) or 0.0) + decode_re = float((decode_perf.get("re") or {}).get("total_ms", 0.0) or 0.0) + decode_nir = float((decode_perf.get("nir") or {}).get("total_ms", 0.0) or 0.0) + + flat_roles = (getattr(self, "last_flatfield_result", {}) or {}).get("perf_by_role_ms", {}) or {} + rad_result = getattr(self, "last_radiometric_normalization_result", {}) or {} + rad_roles = rad_result.get("perf_by_role_ms", {}) or {} + rad_by_role = rad_result.get("by_role", {}) or {} + ori_roles = (getattr(self, "last_orientation_result", {}) or {}).get("perf_by_role_ms", {}) or {} + + print( + "[PERF][CORE_FRAME] " + f"total={t_total_ms:.2f}ms " + f"dark={t_dark_ms:.2f} " + f"rad={t_radnorm_ms:.2f} " + f"flat_native={t_flat_native_ms:.2f} " + f"orient={t_orientation_ms:.2f} " + f"direct={t_direct_call_ms:.2f} " + f"final={t_final_ms:.2f} | " + f"geom={t_geometry_ms:.2f} " + f"alloc={t_tensor_alloc_ms:.2f} " + f"remap_cache={float(direct_perf.get('remap_cache_ms', 0.0)):.2f} " + f"rgb={t_rgb_ms:.2f} " + f"re={float(warp_details.get('re', 0.0)):.2f} " + f"nir={float(warp_details.get('nir', 0.0)):.2f} " + f"final_flat={final_flat_ms:.2f} | " + f"orient_fused_rgb={1 if direct_perf.get('rgb_orientation_fused_into_remap') else 0} " + f"cache geom={'HIT' if direct_perf.get('geometry_cache_hit') else 'MISS'} " + f"remap={'HIT' if direct_perf.get('remap_cache_hit') else 'MISS'} " + f"shape={tuple(out.shape)}" + ) + + print( + "[PERF][CORE_ROLES] " + f"decode(rgb/re/nir)={decode_rgb:.2f}/{decode_re:.2f}/{decode_nir:.2f}ms | " + f"rad(rgb/re/nir)={float(rad_roles.get('rgb', 0.0)):.2f}/" + f"{float(rad_roles.get('re', 0.0)):.2f}/" + f"{float(rad_roles.get('nir', 0.0)):.2f}ms " + f"kernel={str((rad_by_role.get('rgb') or {}).get('kernel_backend', '-'))}/" + f"{str((rad_by_role.get('re') or {}).get('kernel_backend', '-'))}/" + f"{str((rad_by_role.get('nir') or {}).get('kernel_backend', '-'))} | " + f"flat(rgb/re/nir)={float(flat_roles.get('rgb', 0.0)):.2f}/" + f"{float(flat_roles.get('re', 0.0)):.2f}/" + f"{float(flat_roles.get('nir', 0.0)):.2f}ms | " + f"orient(rgb/re/nir)={float(ori_roles.get('rgb', 0.0)):.2f}/" + f"{float(ori_roles.get('re', 0.0)):.2f}/" + f"{float(ori_roles.get('nir', 0.0)):.2f}ms" + ) + + # -------------------------------------------------------- + # LOG DE GEOMETRIA/SHAPES: 1x/5s por padrão + # -------------------------------------------------------- + last_shapes = float(getattr(self, "_last_core_shapes_log_ts", 0.0) or 0.0) + shapes_interval = float(getattr(self, "core_shapes_log_interval_s", 5.0) or 5.0) + if now - last_shapes >= shapes_interval: + self._last_core_shapes_log_ts = now + + role_shapes = {} + try: + for _cam_id, _item in (decoded or {}).items(): + _role = str(_item.get("role") or (_item.get("meta", {}) or {}).get("role") or "?").lower() + _img = _item.get("image") + role_shapes[_role] = tuple(_img.shape) if hasattr(_img, "shape") else None + except Exception: + role_shapes = {} + + print( + "[PERF][CORE_SHAPES] " + f"rgb={role_shapes.get('rgb')} " + f"re={role_shapes.get('re')} " + f"nir={role_shapes.get('nir')} " + f"target={tuple(direct_perf.get('target_size', ())) if direct_perf.get('target_size') else target_size} " + f"crop={direct_perf.get('crop_box')} " + f"remap_shapes={direct_perf.get('remap_map_shapes', {})} " + f"flat_space={flat_apply_space}" + ) return out @@ -3853,12 +4075,8 @@ class RawProcessorCore: # FLAT-FIELD # ============================================================ - if not bool( - (self.flatfield_config or {}).get( - "enabled", - False, - ) - ): + flat_enabled = bool((self.flatfield_config or {}).get("enabled", False)) + if False and not flat_enabled: raise RuntimeError( "Produto exige flatfield_config.enabled=true." ) @@ -3887,7 +4105,7 @@ class RawProcessorCore: "flatfield_config.subtract_dark=false." ) - if not self.flatfield_loaded: + if flat_enabled and not self.flatfield_loaded: raise RuntimeError( "Flat-field produto não foi carregado " "integralmente." @@ -5250,6 +5468,7 @@ class RawProcessorCore: clip_output = bool(cfg.get("clip_output", True)) corrected = {} + perf_by_role_ms = {} sat_guard_enabled = bool( cfg.get( @@ -5271,6 +5490,7 @@ class RawProcessorCore: ) for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str( item.get("role") or item.get("meta", {}).get("role") @@ -5317,15 +5537,18 @@ class RawProcessorCore: else: saturation_mask = None + channel_ms = {} for idx, ch in enumerate( ("R", "G", "B") ): + t_ch0 = time.perf_counter() out[:, :, idx] = self._apply_flat_gain_single_channel( out[:, :, idx], ch, clip_output=clip_output, saturation_mask=saturation_mask, ) + channel_ms[ch] = (time.perf_counter() - t_ch0) * 1000.0 new_item["image"] = out @@ -5335,6 +5558,7 @@ class RawProcessorCore: "applied": True, "inplace_reused_input": bool(reused), "shape": list(out.shape), + "channel_ms": {k: float(v) for k, v in channel_ms.items()}, } elif role in ("re", "nir"): @@ -5355,12 +5579,14 @@ class RawProcessorCore: else: saturation_mask = None + t_ch0 = time.perf_counter() out = self._apply_flat_gain_single_channel( img_f, ch, clip_output=clip_output, saturation_mask=saturation_mask, ) + ch_ms = (time.perf_counter() - t_ch0) * 1000.0 new_item["image"] = out @@ -5370,6 +5596,7 @@ class RawProcessorCore: "applied": True, "inplace_reused_input": bool(reused), "shape": list(out.shape), + "channel_ms": {ch: float(ch_ms)}, } else: @@ -5393,6 +5620,13 @@ class RawProcessorCore: new_item["meta"] = new_meta corrected[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + perf_by_role_ms[role] = float(role_elapsed_ms) + if role in result["by_role"] and isinstance(result["by_role"][role], dict): + result["by_role"][role]["total_ms"] = float(role_elapsed_ms) + + result["perf_by_role_ms"] = dict(perf_by_role_ms) + required_roles = { "rgb", "re", @@ -5765,8 +5999,10 @@ class RawProcessorCore: debug_by_role = {} normalized = {} + perf_by_role_ms = {} for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str(item.get("role") or item.get("meta", {}).get("role") or "").lower() img = item.get("image") @@ -5880,6 +6116,9 @@ class RawProcessorCore: new_item = dict(item) new_meta = dict(item.get("meta", {}) or {}) + debug["kernel_backend"] = str( + getattr(self, "_last_radiometric_kernel_backend", "unknown") + ) debug["inplace_reused_input"] = bool(reused) debug["image_shape"] = list(out.shape) if hasattr(out, "shape") else None debug["image_dtype"] = str(out.dtype) if hasattr(out, "dtype") else None @@ -5894,10 +6133,16 @@ class RawProcessorCore: new_item["meta"] = new_meta normalized[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + debug["perf_ms"] = float(role_elapsed_ms) + perf_by_role_ms[role] = float(role_elapsed_ms) + result["applied"] = True result["by_camera"][cam_id] = debug result["by_role"][role] = debug + result["perf_by_role_ms"] = dict(perf_by_role_ms) + if result["applied"]: scales = [v.get("scale_applied") for v in result["by_camera"].values() if isinstance(v, dict)] scales = [float(s) for s in scales if s is not None] @@ -6440,7 +6685,15 @@ class RawProcessorCore: def _apply_radiometric_scale_inplace(self, img, scale: float, clip_output: bool): """ - Aplica escala radiométrica com o mínimo possível de alocação. + Aplica escala radiometrica com o minimo possivel de alocacao. + + Fast path: + - float32 gravavel; + - C-contiguous; + - imagem 2D ou 3D; + - kernel Numba em uma unica passada de memoria. + + Fallback NumPy preservado para qualquer layout atipico. Retorna: out, reused_input @@ -6448,22 +6701,45 @@ class RawProcessorCore: out, reused = self._radiometric_get_writable_float32_image(img) if out is None: + self._last_radiometric_kernel_backend = "none" return img, False - scale = float(scale) + scale = np.float32(float(scale)) - # Se a escala é praticamente 1 e não precisa clipar, não faz nada. - if abs(scale - 1.0) <= 1e-6 and not clip_output: + # Preserva exatamente o comportamento historico para scale ~= 1: + # nao multiplica; apenas clipa quando solicitado. + if abs(float(scale) - 1.0) <= 1e-6: + if clip_output: + np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "identity_clip_numpy" + else: + self._last_radiometric_kernel_backend = "identity" return out, reused - # Multiplicação in-place. - if abs(scale - 1.0) > 1e-6: - np.multiply(out, np.float32(scale), out=out, casting="unsafe") + use_numba = ( + _HAS_NUMBA + and out.dtype == np.float32 + and out.flags.c_contiguous + and out.ndim in (2, 3) + ) + + if use_numba: + _apply_radiometric_scale_inplace_numba( + out, + scale, + bool(clip_output), + ) + self._last_radiometric_kernel_backend = "numba_single_pass" + return out, reused + + # Fallback historico. + if abs(float(scale) - 1.0) > 1e-6: + np.multiply(out, scale, out=out, casting="unsafe") - # Clip in-place, se configurado. if clip_output: np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "numpy_ufunc" return out, reused def _apply_radiometric_affine_inplace( @@ -6474,22 +6750,54 @@ class RawProcessorCore: clip_output: bool, ): """ - Aplica o contrato radiométrico affine_v2 no domínio RAW01: + Aplica o contrato radiometrico affine_v2 no dominio RAW01: offset + (value - offset) * reference_factor / actual_factor - O mesmo offset escalar é aplicado aos três canais RGB da role RGB, + O mesmo offset escalar e aplicado aos tres canais RGB da role RGB, conforme black_offset_model='per_role_scalar_raw01'. + + No caminho quente, o affine inteiro e executado por um kernel Numba + single-pass. Isso elimina as tres varreduras completas do ndarray + (subtract -> multiply -> add) sem alterar a ordem matematica por pixel. """ out, reused = self._radiometric_get_writable_float32_image(img) if out is None: + self._last_radiometric_kernel_backend = "none" return img, False scale = np.float32(float(scale)) offset = np.float32(float(black_offset)) - # Para scale=1 a transformação é identidade, independentemente do offset. + # Preserva exatamente o comportamento historico para scale ~= 1: + # o affine nao roda; apenas o clip final, quando habilitado. + if abs(float(scale) - 1.0) <= 1e-6: + if clip_output: + np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "identity_clip_numpy" + else: + self._last_radiometric_kernel_backend = "identity" + return out, reused + + use_numba = ( + _HAS_NUMBA + and out.dtype == np.float32 + and out.flags.c_contiguous + and out.ndim in (2, 3) + ) + + if use_numba: + _apply_radiometric_affine_inplace_numba( + out, + scale, + offset, + bool(clip_output), + ) + self._last_radiometric_kernel_backend = "numba_single_pass" + return out, reused + + # Fallback historico: mesma matematica em tres ufuncs in-place. if abs(float(scale) - 1.0) > 1e-6: np.subtract(out, offset, out=out, casting="unsafe") np.multiply(out, scale, out=out, casting="unsafe") @@ -6498,6 +6806,7 @@ class RawProcessorCore: if clip_output: np.clip(out, 0.0, 1.0, out=out) + self._last_radiometric_kernel_backend = "numpy_ufunc" return out, reused def _build_radiometric_debug( @@ -7138,12 +7447,13 @@ class RawProcessorCore: """ Fusão direta otimizada com cache de geometria fixa. - target_size_override permite ir diretamente ao tamanho final pedido - pelo treino/inferência, evitando um segundo resize posterior. + Esta versão adiciona somente telemetria granular. A matemática continua: + RGB crop/resize + RE/NIR remap direto para o target final. """ channels_expected = self._validate_physical_channel_count(channels_expected) - t0 = time.perf_counter() + t_total0 = time.perf_counter() + t0 = time.perf_counter() rgb_cam_id = self._find_cam_by_role(decoded, "rgb") if rgb_cam_id is None: raise RuntimeError("Fusão direta requer câmera com role='rgb' como referência") @@ -7165,6 +7475,7 @@ class RawProcessorCore: for cam_id, data in decoded.items() for item in [data] } + t_role_map_ms = (time.perf_counter() - t0) * 1000.0 missing_roles = [role for role in ("rgb", "re", "nir") if role not in role_to_cam] if missing_roles: @@ -7173,7 +7484,10 @@ class RawProcessorCore: f"{missing_roles}. Recebidos={sorted(str(role) for role in role_to_cam)}" ) - # Cache da geometria fixa. + # ------------------------------------------------------------ + # Geometria fixa/cacheada + # ------------------------------------------------------------ + t0_geom = time.perf_counter() geom = self._direct_fusion_get_geometry_cached_fast( decoded=decoded, role_to_cam=role_to_cam, @@ -7181,6 +7495,7 @@ class RawProcessorCore: target_size=target_size, meta=meta, ) + t_geometry_ms = (time.perf_counter() - t0_geom) * 1000.0 crop_box = geom["crop_box"] crop_applied = bool(geom["crop_applied"]) @@ -7200,9 +7515,16 @@ class RawProcessorCore: "homography_profiles_used": geom.get("homography_profiles_used", {}), } + # Alocação do tensor separada do preparo geométrico. + t0_alloc = time.perf_counter() tensor = np.empty((channels_expected, target_h, target_w), dtype=np.float32) + t_tensor_alloc_ms = (time.perf_counter() - t0_alloc) * 1000.0 - t_prepare_ms = (time.perf_counter() - t0) * 1000.0 + t_prepare_ms = ( + t_role_map_ms + + t_geometry_ms + + t_tensor_alloc_ms + ) # ------------------------------------------------------------ # Cache de remap @@ -7211,8 +7533,22 @@ class RawProcessorCore: use_remap_for_rgb = bool((self.fusion_config or {}).get("use_remap_for_rgb", False)) use_remap_for_spec = bool((self.fusion_config or {}).get("use_remap_for_spec", True)) - t0_remap_cache = time.perf_counter() + rgb_orientation_fused = self._decoded_role_orientation_is_deferred_fast( + decoded, + role_to_cam, + "rgb", + ) + # Se a orientação RGB foi adiada, o caminho RGB obrigatoriamente precisa + # usar o remap cacheado que contém a matriz de orientação composta. + if rgb_orientation_fused: + if not use_remap_cache: + raise RuntimeError( + "Orientação RGB diferida requer fusion_config.use_remap_cache=true." + ) + use_remap_for_rgb = True + + t0_remap_cache = time.perf_counter() if use_remap_cache and (use_remap_for_rgb or use_remap_for_spec): remap_cache = self._direct_fusion_get_remap_cached_fast( decoded=decoded, @@ -7228,14 +7564,25 @@ class RawProcessorCore: "cache_hits": 0, "cache_misses": 0, } - t_remap_cache_ms = (time.perf_counter() - t0_remap_cache) * 1000.0 + remap_map_shapes = {} + try: + for role, entry in (remap_cache.get("maps", {}) or {}).items(): + if isinstance(entry, dict): + m1 = entry.get("map1") + m2 = entry.get("map2") + remap_map_shapes[str(role)] = { + "map1": list(m1.shape) if hasattr(m1, "shape") else None, + "map2": list(m2.shape) if hasattr(m2, "shape") else None, + } + except Exception: + remap_map_shapes = {} + # ------------------------------------------------------------ - # RGB via remap + # RGB # ------------------------------------------------------------ t0_rgb = time.perf_counter() - rgb_maps = remap_cache["maps"].get("rgb") if use_remap_cache and use_remap_for_rgb and rgb_maps is not None: self._direct_fusion_write_rgb_remap_fast( @@ -7250,18 +7597,16 @@ class RawProcessorCore: crop_box, target_size, ) - t_rgb_ms = (time.perf_counter() - t0_rgb) * 1000.0 warp_details = {} t_warp_total_ms = 0.0 # ------------------------------------------------------------ - # RE via remap + # RE # ------------------------------------------------------------ if "re" in role_to_cam: t0w = time.perf_counter() - re_img = decoded[role_to_cam["re"]]["image"] re_maps = remap_cache["maps"].get("re") @@ -7287,11 +7632,10 @@ class RawProcessorCore: t_warp_total_ms += warp_details["re"] # ------------------------------------------------------------ - # NIR via remap + # NIR # ------------------------------------------------------------ if "nir" in role_to_cam: t0w = time.perf_counter() - nir_img = decoded[role_to_cam["nir"]]["image"] nir_maps = remap_cache["maps"].get("nir") @@ -7320,7 +7664,6 @@ class RawProcessorCore: # Flat-field no espaço final do tensor # ------------------------------------------------------------ t0_final_flat = time.perf_counter() - final_flat_ms = 0.0 final_flat_enabled = False final_flat_cache_available = False @@ -7332,7 +7675,6 @@ class RawProcessorCore: if apply_final_flat: final_flat_enabled = True - gain_tensor = self._get_final_flat_gain_tensor_cached_fast( decoded=decoded, role_to_cam=role_to_cam, @@ -7358,9 +7700,13 @@ class RawProcessorCore: ) self.last_fusion_result["output_shape"] = list(tensor.shape) + t_direct_total_ms = (time.perf_counter() - t_total0) * 1000.0 perf = { "prepare_ms": float(t_prepare_ms), + "role_map_ms": float(t_role_map_ms), + "geometry_ms": float(t_geometry_ms), + "tensor_alloc_ms": float(t_tensor_alloc_ms), "rgb_crop_resize_ms": float(t_rgb_ms), "warp_total_ms": float(t_warp_total_ms), "warp_details_ms": warp_details, @@ -7377,15 +7723,18 @@ class RawProcessorCore: "remap_enabled": bool(use_remap_cache), "remap_rgb_enabled": bool(use_remap_cache and use_remap_for_rgb), "remap_spec_enabled": bool(use_remap_cache and use_remap_for_spec), + "rgb_orientation_fused_into_remap": bool(rgb_orientation_fused), + "remap_map_shapes": remap_map_shapes, "final_flat_enabled": bool(final_flat_enabled), "final_flat_cache_available": bool(final_flat_cache_available), "final_flat_ms": float(final_flat_ms), + "target_size": [int(target_w), int(target_h)], + "crop_box": [int(v) for v in crop_box], + "direct_total_ms": float(t_direct_total_ms), } return tensor, perf - - def _direct_fusion_get_geometry_cache_key_fast( self, decoded, @@ -7704,11 +8053,18 @@ class RawProcessorCore: img = decoded[cam_id]["image"] role_shapes[str(role).lower()] = tuple(int(v) for v in img.shape[:2]) + rgb_orientation_sig = self._decoded_role_orientation_signature_fast( + decoded, + role_to_cam, + "rgb", + ) + return ( geom.get("key"), tuple(int(v) for v in ref_size), tuple(int(v) for v in target_size), tuple(sorted(role_shapes.items())), + rgb_orientation_sig, ) def _direct_fusion_get_remap_cached_fast( @@ -7763,8 +8119,36 @@ class RawProcessorCore: rgb_cam_id = role_to_cam.get("rgb") if rgb_cam_id is not None: rgb_img = decoded[rgb_cam_id]["image"] + + # geom["C_ref_to_target"] parte do espaço RGB CANÔNICO (após + # camera_orientation). Quando a orientação foi diferida, compomos: + # + # RGB_native --O--> RGB_canonical --C--> target + # + # Como cv2.remap usa o inverso internamente, o target passa a + # amostrar diretamente o frame nativo já na orientação correta. + M_rgb_to_target = geom["C_ref_to_target"] + if self._decoded_role_orientation_is_deferred_fast( + decoded, role_to_cam, "rgb" + ): + meta_rgb = decoded[rgb_cam_id].get("meta", {}) or {} + ori = meta_rgb.get("orientation_applied", {}) or {} + O_native_to_canonical = self._orientation_matrix_same_shape_fast( + src_shape=rgb_img.shape[:2], + rotate_deg=int(ori.get("rotate_deg", 0)), + flip_horizontal=bool(ori.get("flip_horizontal", False)), + flip_vertical=bool(ori.get("flip_vertical", False)), + ) + if O_native_to_canonical is None: + raise RuntimeError( + "Orientação RGB diferida não suportada pelo remap composto." + ) + M_rgb_to_target = ( + geom["C_ref_to_target"] @ O_native_to_canonical + ).astype(np.float32) + remap["maps"]["rgb"] = self._direct_fusion_build_remap_from_src_to_dst_fast( - M_src_to_dst=geom["C_ref_to_target"], + M_src_to_dst=M_rgb_to_target, src_shape=rgb_img.shape[:2], dst_size=(target_w, target_h), ref_shape_for_scaled_src=None, @@ -8156,30 +8540,25 @@ class RawProcessorCore: return rgb.astype(np.float32, copy=False) def _get_bayer_cv2_code(self, bayer_pattern: str | None, algorithm: str = "ea"): - """ - Retorna o código OpenCV para demosaic Bayer. - - algorithm: - - "ea": Edge-Aware, melhor qualidade, mais pesado - - "bilinear": mais rápido, menor custo - """ p = str(bayer_pattern or self.bayer_pattern or "RGGB").upper() algo = str(algorithm or "ea").lower() if algo in ("bilinear", "linear", "fast", "normal"): code_map = { - "BGGR": cv2.COLOR_BayerRG2RGB, - "RGGB": cv2.COLOR_BayerBG2RGB, - "GRBG": cv2.COLOR_BayerGR2RGB, - "GBRG": cv2.COLOR_BayerGB2RGB, + "RGGB": cv2.COLOR_BayerRGGB2RGB, + "BGGR": cv2.COLOR_BayerBGGR2RGB, + "GRBG": cv2.COLOR_BayerGRBG2RGB, + "GBRG": cv2.COLOR_BayerGBRG2RGB, } + elif algo in ("ea", "edge_aware", "edge-aware"): code_map = { - "BGGR": cv2.COLOR_BayerRG2RGB_EA, - "RGGB": cv2.COLOR_BayerBG2RGB_EA, - "GRBG": cv2.COLOR_BayerGR2RGB_EA, - "GBRG": cv2.COLOR_BayerGB2RGB_EA, + "RGGB": cv2.COLOR_BayerRGGB2RGB_EA, + "BGGR": cv2.COLOR_BayerBGGR2RGB_EA, + "GRBG": cv2.COLOR_BayerGRBG2RGB_EA, + "GBRG": cv2.COLOR_BayerGBRG2RGB_EA, } + else: raise ValueError(f"demosaic_algorithm inválido: {algorithm}") @@ -8210,13 +8589,25 @@ class RawProcessorCore: self._last_decode_perf_log_ts = now - #parts = [] - #for k, v in perf.items(): - # if isinstance(v, (int, float)): - # parts.append(f"{k}={float(v):.2f}ms") - # else: - # parts.append(f"{k}={v}") - #print(f"[PERF][DECODE][{role.upper()}] " + " ".join(parts)) + if bool(getattr(self, "core_perf_debug", True)): + # Log compacto. O agregado de RGB/RE/NIR também aparece em CORE_ROLES. + numeric_keys = ( + "total_ms", "unpack_ms", "cvtColor_ms", "float_ms", + "calibration_ms", "clip_ms", "resize_half_ms" + ) + parts = [] + for k in numeric_keys: + v = perf.get(k) + if isinstance(v, (int, float)): + parts.append(f"{k}={float(v):.2f}ms") + shape = perf.get("out_shape") + mode = perf.get("mode") + print( + f"[PERF][DECODE][{role.upper()}] " + + " ".join(parts) + + (f" mode={mode}" if mode is not None else "") + + (f" out={shape}" if shape is not None else "") + ) @@ -8443,6 +8834,131 @@ class RawProcessorCore: return gain_tensor + def _orientation_matrix_same_shape_fast( + self, + src_shape, + rotate_deg=0, + flip_horizontal=False, + flip_vertical=False, + ): + """ + Matriz 3x3 source/native -> espaço canônico para orientações que + preservam H/W (0° e 180°, com flips opcionais). + + A ordem replica _apply_orientation_image(): rotate, flip H, flip V. + 90°/270° retornam None para cair no caminho físico tradicional. + """ + h, w = int(src_shape[0]), int(src_shape[1]) + r = int(rotate_deg) % 360 + + if r not in (0, 180): + return None + + M = np.eye(3, dtype=np.float32) + + if r == 180: + R = np.array( + [ + [-1.0, 0.0, float(w - 1)], + [0.0, -1.0, float(h - 1)], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = R @ M + + if bool(flip_horizontal): + Fh = np.array( + [ + [-1.0, 0.0, float(w - 1)], + [0.0, 1.0, 0.0], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = Fh @ M + + if bool(flip_vertical): + Fv = np.array( + [ + [1.0, 0.0, 0.0], + [0.0, -1.0, float(h - 1)], + [0.0, 0.0, 1.0], + ], + dtype=np.float32, + ) + M = Fv @ M + + return M.astype(np.float32, copy=False) + + def _can_defer_rgb_orientation_to_direct_remap_fast(self, decoded): + """ + Decide se a orientação RGB pode ser incorporada ao remap final. + + Conservador por design: + - somente role RGB; + - somente 0°/180° (mesmo H/W); + - exige remap cache ativo; + - pode ser desabilitado por fusion_config.fuse_orientation_into_remap=false. + """ + cfg = self.camera_orientation_config or {} + fusion = self.fusion_config or {} + + if not bool(cfg.get("enabled", False)): + return False + if not bool(fusion.get("fuse_orientation_into_remap", True)): + return False + if not bool(fusion.get("use_remap_cache", True)): + return False + + by_role = cfg.get("by_role", {}) or {} + role_cfg = by_role.get("rgb", {}) or {} + r = int(role_cfg.get("rotate_deg", 0)) % 360 + fh = bool(role_cfg.get("flip_horizontal", False)) + fv = bool(role_cfg.get("flip_vertical", False)) + + if r not in (0, 180): + return False + if r == 0 and not fh and not fv: + return False + + rgb_cam_id = self._find_cam_by_role(decoded, "rgb") + if rgb_cam_id is None: + return False + img = decoded[rgb_cam_id].get("image") + if img is None or img.ndim < 2: + return False + + return self._orientation_matrix_same_shape_fast( + img.shape[:2], r, fh, fv + ) is not None + + def _decoded_role_orientation_signature_fast( + self, decoded, role_to_cam, role + ): + role = str(role).lower() + cam_id = role_to_cam.get(role) + if cam_id is None: + return (role, False, 0, False, False) + meta = decoded[cam_id].get("meta", {}) or {} + ori = meta.get("orientation_applied", {}) or {} + return ( + role, + bool(ori.get("deferred_to_direct_fusion", False)), + int(ori.get("rotate_deg", 0)) % 360, + bool(ori.get("flip_horizontal", False)), + bool(ori.get("flip_vertical", False)), + ) + + def _decoded_role_orientation_is_deferred_fast( + self, decoded, role_to_cam, role + ): + return bool( + self._decoded_role_orientation_signature_fast( + decoded, role_to_cam, role + )[1] + ) + def _apply_orientation_image( self, img, @@ -8494,8 +9010,10 @@ class RawProcessorCore: def apply_camera_orientation_to_decoded( self, decoded, + defer_roles=None, ): cfg = self.camera_orientation_config or {} + defer_roles = {str(x).lower() for x in (defer_roles or set())} result = { "enabled": bool( @@ -8503,6 +9021,7 @@ class RawProcessorCore: ), "applied": False, "by_role": {}, + "deferred_roles": [], } if not cfg.get("enabled", False): @@ -8515,8 +9034,10 @@ class RawProcessorCore: ) or {} out = {} + perf_by_role_ms = {} for cam_id, item in decoded.items(): + t_role0 = time.perf_counter() role = str( item.get("role") @@ -8557,27 +9078,35 @@ class RawProcessorCore: item.get("meta", {}) or {} ) - if ( + wants_orientation = ( + rotate_deg != 0 + or flip_h + or flip_v + ) + deferred_here = bool( image is not None - and ( - rotate_deg != 0 - or flip_h - or flip_v - ) - ): + and role in defer_roles + and wants_orientation + ) + + if image is not None and wants_orientation and not deferred_here: image = self._apply_orientation_image( image, rotate_deg=rotate_deg, flip_horizontal=flip_h, flip_vertical=flip_v, ) - result["applied"] = True + if deferred_here: + result["deferred_roles"].append(role) + new_meta["orientation_applied"] = { "rotate_deg": rotate_deg, "flip_horizontal": flip_h, "flip_vertical": flip_v, + "physically_applied": bool(wants_orientation and not deferred_here), + "deferred_to_direct_fusion": bool(deferred_here), } new_item["image"] = image @@ -8585,18 +9114,24 @@ class RawProcessorCore: out[cam_id] = new_item + role_elapsed_ms = (time.perf_counter() - t_role0) * 1000.0 + perf_by_role_ms[role] = float(role_elapsed_ms) + result["by_role"][role] = { "camera_id": cam_id, "rotate_deg": rotate_deg, "flip_horizontal": flip_h, "flip_vertical": flip_v, + "deferred_to_direct_fusion": bool(deferred_here), "shape": ( list(image.shape) if image is not None else None ), + "perf_ms": float(role_elapsed_ms), } + result["perf_by_role_ms"] = dict(perf_by_role_ms) self.last_orientation_result = result return out