427 lines
16 KiB
Python
427 lines
16 KiB
Python
import glob
|
|
import multiprocessing
|
|
import os
|
|
import torch
|
|
import torchvision.transforms as T
|
|
import numpy as np
|
|
from PIL import Image
|
|
import cv2
|
|
import depthai as dai
|
|
from utils import carregar_labelmap_completo
|
|
|
|
# === CONFIGURAÇÕES ===
|
|
MODELO = "oak-1"
|
|
MODEL_NAME = "ervasModel"
|
|
NUM_CLASSES = 3
|
|
RESOLUCAO = (512, 512)
|
|
model_ext = '.onnx'
|
|
model_path = os.path.join(MODELO, "backup")
|
|
model_name = MODEL_NAME + "_best"
|
|
labelmap_path = os.path.join(MODELO, "dataset", "labelmap.txt")
|
|
teste_pasta_operacao = False
|
|
if not teste_pasta_operacao:
|
|
image_path = os.path.join(MODELO, "dataset", "split", "test", "images")
|
|
else:
|
|
image_path = 'C:\\ZendionInc\\agrobot_base\\AgroBase\\AgroBase\\bin\\x64\\Debug\\Operacoes\\25_07_2025_14_39_14\\Cam0\\'
|
|
USE_CAMERA = False
|
|
mostrar_dados_visuais = False
|
|
|
|
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
|
|
|
# === FUNÇÕES AUXILIARES ===
|
|
|
|
def segment_image_pth(image_path, model):
|
|
image = Image.open(image_path).convert("RGB")
|
|
transform = T.Compose([
|
|
T.Resize(RESOLUCAO),
|
|
T.ToTensor(),
|
|
T.Normalize(mean=[0.485, 0.456, 0.406],
|
|
std=[0.229, 0.224, 0.225])
|
|
])
|
|
input_tensor = transform(image).unsqueeze(0).to(device)
|
|
|
|
model.eval()
|
|
with torch.no_grad():
|
|
output = model(input_tensor)['out']
|
|
prediction = torch.argmax(output.squeeze(), dim=0).cpu().numpy()
|
|
return prediction
|
|
|
|
def segment_image_onnx(image_path, session):
|
|
image = Image.open(image_path).convert("RGB")
|
|
transform = T.Compose([
|
|
T.Resize(RESOLUCAO),
|
|
T.ToTensor(),
|
|
T.Normalize(mean=[0.485, 0.456, 0.406],
|
|
std=[0.229, 0.224, 0.225])
|
|
])
|
|
input_tensor = transform(image).unsqueeze(0).numpy() # já vira numpy direto
|
|
input_name = session.get_inputs()[0].name
|
|
|
|
# Executa a inferência
|
|
outputs = session.run(None, {input_name: input_tensor})
|
|
prediction = outputs[0] # shape: (1, num_classes, H, W)
|
|
|
|
# Seleciona a classe com maior score
|
|
prediction = prediction.squeeze(0).argmax(axis=0)
|
|
return prediction
|
|
|
|
def display_segmentation(original_path, prediction):
|
|
original = cv2.imread(original_path)
|
|
original = cv2.resize(original, RESOLUCAO)
|
|
|
|
cor_para_id, cores_bgr, id_para_nome, ignore_bgr = carregar_labelmap_completo(labelmap_path)
|
|
|
|
seg_color = np.zeros_like(original)
|
|
for class_id, color in enumerate(cores_bgr):
|
|
seg_color[prediction == class_id] = color
|
|
|
|
overlay = cv2.addWeighted(original, 0.5, seg_color, 0.5, 0)
|
|
|
|
# Legenda
|
|
legenda_inicio_y = 20
|
|
for i, cor in enumerate(cores_bgr):
|
|
nome = id_para_nome.get(i, f"Classe {i}")
|
|
pos_y = legenda_inicio_y + i * 30
|
|
cv2.rectangle(overlay, (10, pos_y - 15), (30, pos_y + 5), cor, -1)
|
|
cv2.putText(overlay, nome, (40, pos_y + 2), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255,255,255), 1, cv2.LINE_AA)
|
|
|
|
cv2.imshow("Segmentacao - Overlay", overlay)
|
|
# Remove: cv2.waitKey() e destroyAllWindows
|
|
|
|
|
|
def extrair_corredor_principal(mask_classes, classe_rua):
|
|
altura, largura = mask_classes.shape
|
|
centro_img = largura // 2
|
|
|
|
mask_rua = np.uint8(mask_classes == classe_rua)
|
|
num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(mask_rua, connectivity=8)
|
|
|
|
melhor_blob_idx = -1
|
|
melhor_score = float('-inf')
|
|
for i in range(1, num_labels): # ignora fundo
|
|
x, y, w, h, area = stats[i]
|
|
centro_blob = x + w // 2
|
|
dist_centro = abs(centro_blob - centro_img)
|
|
score = area - (dist_centro * 2) # prioriza blobs grandes e centrais
|
|
if score > melhor_score:
|
|
melhor_score = score
|
|
melhor_blob_idx = i
|
|
|
|
return np.uint8(labels == melhor_blob_idx)
|
|
|
|
def analisar_corredor_visual(mask_classes, depth_frame=None):
|
|
CLASSE_RUA = 0
|
|
CLASSE_CANA = 1
|
|
ALTURA, LARGURA = mask_classes.shape
|
|
centro_x = LARGURA // 2
|
|
erro_lateral = None
|
|
erro_angular = None
|
|
status_corredor = "desconhecido"
|
|
|
|
mask_corredor_principal = extrair_corredor_principal(mask_classes, CLASSE_RUA)
|
|
scanlines = [int(ALTURA * f) for f in [0.999, 0.85, 0.7, 0.55, 0.4, 0.25, 0.1]]
|
|
centros_corredor = []
|
|
for y in scanlines:
|
|
linha = mask_corredor_principal[y]
|
|
mask_bin = np.uint8(linha > 0)
|
|
if np.count_nonzero(mask_bin) == 0:
|
|
centros_corredor.append((-1, y))
|
|
continue
|
|
# Agora garantido que só tem um blob principal
|
|
indices = np.where(mask_bin > 0)[0]
|
|
centro_x = int(np.mean(indices))
|
|
centros_corredor.append((centro_x, y))
|
|
|
|
# 2. Calcular erro angular (reta entre os centros)
|
|
if len(centros_corredor) >= 2:
|
|
# Angular curto
|
|
(x1c, y1c), (x2c, y2c) = centros_corredor[0], centros_corredor[1]
|
|
erro_angular_curto = np.arctan2(x2c - x1c, y1c - y2c)
|
|
# Angular longo
|
|
(x1l, y1l), (x2l, y2l) = centros_corredor[0], centros_corredor[-1]
|
|
erro_angular_longo = np.arctan2(x2l - x1l, y1l - y2l)
|
|
# Peso para suavizar ou escolher dinamicamente
|
|
erro_angular = 0.2 * erro_angular_curto + 0.8 * erro_angular_longo
|
|
|
|
# 3. Calcular erro lateral (deslocamento da base)
|
|
erro_lateral_px = None
|
|
erro_lateral_pct = 0
|
|
largura_corredor_px = None
|
|
if len(centros_corredor) > 0:
|
|
x_base, y_base = centros_corredor[0]
|
|
if x_base != -1:
|
|
erro_lateral_px = centro_x - x_base
|
|
# Recalcular largura do maior blob na linha base
|
|
linha_base = mask_classes[y_base]
|
|
mask_bin_base = np.uint8(linha_base == CLASSE_RUA)
|
|
num_labels, _, stats, _ = cv2.connectedComponentsWithStats(mask_bin_base.reshape(1, -1), connectivity=8)
|
|
max_area = 0
|
|
largura_corredor_px = None
|
|
for i in range(1, num_labels):
|
|
w = stats[i, cv2.CC_STAT_WIDTH]
|
|
area = stats[i, cv2.CC_STAT_AREA]
|
|
if area > max_area:
|
|
max_area = area
|
|
largura_corredor_px = w
|
|
if largura_corredor_px and largura_corredor_px > 0:
|
|
erro_lateral_pct = (erro_lateral_px / largura_corredor_px) * 100
|
|
|
|
# 5. Novo status de corredor
|
|
LIMIAR_CANA = 1000
|
|
parte_cima = mask_classes[:int(ALTURA * 0.3), :]
|
|
parte_baixo = mask_classes[int(ALTURA * 0.7):, :]
|
|
|
|
cana_cima = np.sum(parte_cima == CLASSE_CANA)
|
|
cana_baixo = np.sum(parte_baixo == CLASSE_CANA)
|
|
cana_total = np.sum(mask_classes == CLASSE_CANA)
|
|
|
|
if cana_total < 500:
|
|
status_corredor = "fora"
|
|
elif cana_baixo > LIMIAR_CANA and cana_cima > LIMIAR_CANA:
|
|
status_corredor = "dentro"
|
|
elif cana_baixo < LIMIAR_CANA and cana_cima > LIMIAR_CANA:
|
|
status_corredor = "entrando"
|
|
elif cana_baixo > LIMIAR_CANA and cana_cima < LIMIAR_CANA:
|
|
status_corredor = "saindo"
|
|
|
|
return {
|
|
"erro_angular_rad": round(erro_angular, 3) if erro_angular is not None else None,
|
|
"erro_lateral_pct": erro_lateral_pct,
|
|
"status_corredor": status_corredor,
|
|
"centros_corredor": centros_corredor
|
|
}
|
|
|
|
def display_segmentation_debug(original_path, prediction, dados_visuais=None, centros_corredor=None, largura_robo_px=60):
|
|
original = cv2.imread(original_path)
|
|
original = cv2.resize(original, RESOLUCAO)
|
|
|
|
cor_para_id, cores_bgr, id_para_nome, ignore_bgr = carregar_labelmap_completo(labelmap_path)
|
|
|
|
seg_color = np.zeros_like(original)
|
|
for class_id, color in enumerate(cores_bgr):
|
|
seg_color[prediction == class_id] = color
|
|
|
|
overlay = cv2.addWeighted(original, 0.5, seg_color, 0.5, 0)
|
|
altura, largura = prediction.shape
|
|
centro_x = largura // 2
|
|
|
|
# 1. Linha do centro do corredor + largura do robô
|
|
if centros_corredor and len(centros_corredor) >= 2:
|
|
for ponto in centros_corredor:
|
|
cv2.circle(overlay, ponto, 4, (0, 255, 255), -1)
|
|
for i in range(len(centros_corredor) - 1):
|
|
cv2.line(overlay, centros_corredor[i], centros_corredor[i + 1], (0, 255, 255), 2)
|
|
for ponto in centros_corredor:
|
|
x, y = ponto
|
|
cv2.line(overlay, (x - largura_robo_px // 2, y), (x + largura_robo_px // 2, y), (255, 0, 255), 1)
|
|
|
|
# 4. Texto de métricas
|
|
if dados_visuais:
|
|
erro_lateral_pct = dados_visuais["erro_lateral_pct"]
|
|
erro_angular_deg = np.degrees(dados_visuais["erro_angular_rad"]) if dados_visuais["erro_angular_rad"] else 0
|
|
texto = [
|
|
f"Erro angular: {erro_angular_deg:.2f} graus",
|
|
f"Erro lateral: {erro_lateral_pct:.2f} %",
|
|
f"Status: {dados_visuais['status_corredor']}"
|
|
]
|
|
for i, t in enumerate(texto):
|
|
cv2.putText(overlay, t, (10, 25 + i * 25), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255, 255, 255), 2)
|
|
|
|
# 5. Legenda das classes
|
|
legenda_inicio_y = 140
|
|
for i, cor in enumerate(cores_bgr):
|
|
nome = id_para_nome[i]
|
|
pos_y = legenda_inicio_y + i * 30
|
|
cv2.rectangle(overlay, (10, pos_y - 15), (30, pos_y + 5), cor, -1)
|
|
cv2.putText(overlay, nome, (40, pos_y + 2), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255,255,255), 1, cv2.LINE_AA)
|
|
|
|
# Mostrar
|
|
cv2.imshow("Segmentacao - Overlay", overlay)
|
|
#cv2.waitKey(0)
|
|
#cv2.destroyAllWindows()
|
|
|
|
|
|
def visualizar_corredor_principal(mask_classes, original_path):
|
|
CLASSE_RUA = 0
|
|
|
|
imagem_rgb = cv2.imread(original_path)
|
|
imagem_rgb = cv2.resize(imagem_rgb, RESOLUCAO)
|
|
|
|
altura, largura = mask_classes.shape
|
|
centro_img = largura // 2
|
|
# Máscara binária da classe "rua"
|
|
mask_rua = np.uint8(mask_classes == CLASSE_RUA)
|
|
# Encontrar blobs conectados
|
|
num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(mask_rua, connectivity=8)
|
|
melhor_blob_idx = -1
|
|
melhor_score = float('-inf')
|
|
for i in range(1, num_labels): # ignorar o fundo (label 0)
|
|
x, y, w, h, area = stats[i]
|
|
centro_blob = x + w // 2
|
|
dist_centro = abs(centro_blob - centro_img)
|
|
# Score baseado em tamanho e centralização
|
|
score = area - (dist_centro * 2) # ajuste o peso conforme necessário
|
|
if score > melhor_score:
|
|
melhor_score = score
|
|
melhor_blob_idx = i
|
|
# Criar nova máscara com apenas o melhor blob
|
|
mascara_corredor = np.uint8(labels == melhor_blob_idx)
|
|
|
|
# Overlay para visualização
|
|
overlay = imagem_rgb.copy()
|
|
overlay[mascara_corredor == 1] = [0, 255, 255] # amarelo para corredor principal
|
|
|
|
blend = cv2.addWeighted(imagem_rgb, 0.6, overlay, 0.4, 0)
|
|
|
|
cv2.imshow("Corredor Principal", blend)
|
|
|
|
|
|
# === CARREGA O MODELO ===
|
|
if model_ext == ".pth":
|
|
# === CARREGA MODELO PTH ===
|
|
import torchvision.models.segmentation as models
|
|
model = models.deeplabv3_resnet50(weights=None, num_classes=NUM_CLASSES)
|
|
model.load_state_dict(torch.load(os.path.join(model_path, model_name + model_ext), map_location=device))
|
|
model.to(device)
|
|
if model_ext == ".onnx":
|
|
# === CARREGA MODELO ONNX ===
|
|
import onnxruntime as ort
|
|
onnx_model_path = os.path.join(model_path, model_name + model_ext) # deve terminar com .onnx
|
|
sess_options = ort.SessionOptions()
|
|
sess_options.intra_op_num_threads = multiprocessing.cpu_count() # usa todos os núcleos disponíveis
|
|
sess_options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL # otimiza o grafo
|
|
onnx_session = ort.InferenceSession(
|
|
onnx_model_path,
|
|
sess_options,
|
|
providers=["CUDAExecutionProvider", "CPUExecutionProvider"]
|
|
)
|
|
print("[💡] Provedores ativos:", onnx_session.get_providers())
|
|
|
|
# === EXECUÇÃO EM TEMPO REAL COM OAK-D ===
|
|
if USE_CAMERA:
|
|
print("[📷] Iniciando segmentação em tempo real com a OAK-D Lite...")
|
|
|
|
# Setup da câmera
|
|
pipeline = dai.Pipeline()
|
|
cam_rgb = pipeline.createColorCamera()
|
|
cam_rgb.setPreviewSize(640, 640)
|
|
cam_rgb.setResolution(dai.ColorCameraProperties.SensorResolution.THE_1080_P)
|
|
cam_rgb.setInterleaved(False)
|
|
cam_rgb.setColorOrder(dai.ColorCameraProperties.ColorOrder.BGR)
|
|
|
|
xout = pipeline.createXLinkOut()
|
|
xout.setStreamName("rgb")
|
|
cam_rgb.preview.link(xout.input)
|
|
|
|
cor_para_id, cores_bgr, id_para_nome, ignore_bgr = carregar_labelmap_completo(labelmap_path)
|
|
|
|
with dai.Device(pipeline) as oak_device:
|
|
queue = oak_device.getOutputQueue(name="rgb", maxSize=1, blocking=False)
|
|
|
|
while True:
|
|
in_rgb = queue.get()
|
|
frame = in_rgb.getCvFrame()
|
|
frame_resized = cv2.resize(frame, (512, 512))
|
|
|
|
# Pré-processa o frame
|
|
image_pil = Image.fromarray(cv2.cvtColor(frame_resized, cv2.COLOR_BGR2RGB))
|
|
transform = T.Compose([
|
|
T.ToTensor(),
|
|
T.Normalize(mean=[0.485, 0.456, 0.406],
|
|
std=[0.229, 0.224, 0.225])
|
|
])
|
|
input_tensor = transform(image_pil).unsqueeze(0).to(device)
|
|
|
|
# Segmentação
|
|
model.eval()
|
|
with torch.no_grad():
|
|
output = model(input_tensor)['out']
|
|
prediction = torch.argmax(output.squeeze(), dim=0).cpu().numpy()
|
|
|
|
# === ANÁLISE DO CORREDOR VISUAL ===
|
|
dados_visuais = analisar_corredor_visual(prediction)
|
|
print("📊 Análise Visual:")
|
|
print(f" ↔️ Erro lateral (px): {dados_visuais['erro_lateral_px']}")
|
|
print(f" 📐 Erro angular (rad): {dados_visuais['erro_angular_rad']}")
|
|
print(f" ⛔ Distância frente (m): {dados_visuais['distancia_frente_m']}")
|
|
print(f" 🛣️ Status corredor: {dados_visuais['status_corredor']}")
|
|
|
|
# Cria imagem colorida da segmentação
|
|
seg_color = np.zeros_like(frame_resized)
|
|
for class_id, color in enumerate(cores_bgr):
|
|
seg_color[prediction == class_id] = color
|
|
|
|
overlay = cv2.addWeighted(frame_resized, 0.5, seg_color, 0.5, 0)
|
|
|
|
# Legenda
|
|
legenda_inicio_y = 20
|
|
for i, cor in enumerate(cores_bgr):
|
|
nome = id_para_nome.get(i, f"Classe {i}")
|
|
pos_y = legenda_inicio_y + i * 30
|
|
cv2.rectangle(overlay, (10, pos_y - 15), (30, pos_y + 5), cor, -1)
|
|
cv2.putText(overlay, nome, (40, pos_y + 2), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255,255,255), 1, cv2.LINE_AA)
|
|
|
|
# Mostra
|
|
cv2.imshow("Segmentacao em Tempo Real", overlay)
|
|
if cv2.waitKey(1) & 0xFF == ord('q'):
|
|
break
|
|
|
|
cv2.destroyAllWindows()
|
|
else:
|
|
# === SEGMENTAÇÃO POR NAVEGAÇÃO ENTRE IMAGENS ===
|
|
if image_path and os.path.isdir(image_path):
|
|
print(f"[📁] Navegando imagens em: {image_path}")
|
|
|
|
# Lista de imagens
|
|
extensoes = ("*.jpg", "*.png", "*.jpeg")
|
|
arquivos = []
|
|
for ext in extensoes:
|
|
arquivos.extend(glob.glob(os.path.join(image_path, ext)))
|
|
arquivos.sort()
|
|
|
|
if teste_pasta_operacao:
|
|
arquivos = glob.glob(os.path.join(image_path, "*_rgb.jpeg"))
|
|
arquivos.sort(key=lambda x: int(os.path.basename(x).split('_')[0]))
|
|
|
|
if not arquivos:
|
|
print("[!] Nenhuma imagem encontrada na pasta de teste.")
|
|
exit()
|
|
|
|
indice = 0
|
|
while True:
|
|
caminho_img = arquivos[indice]
|
|
if model_ext == ".pth":
|
|
output_predictions = segment_image_pth(caminho_img, model)
|
|
if model_ext == ".onnx":
|
|
output_predictions = segment_image_onnx(caminho_img, onnx_session)
|
|
|
|
# === ANÁLISE DO CORREDOR VISUAL ===
|
|
dados_visuais = analisar_corredor_visual(output_predictions) if mostrar_dados_visuais else None
|
|
centros = dados_visuais["centros_corredor"] if mostrar_dados_visuais else None
|
|
display_segmentation_debug(caminho_img, output_predictions, dados_visuais, centros, 84)
|
|
|
|
print(f"[{indice+1}/{len(arquivos)}] {os.path.basename(caminho_img)}")
|
|
|
|
key = cv2.waitKey(0) & 0xFF
|
|
print(f"key: {key}")
|
|
cv2.destroyAllWindows() # ← fecha imagem ao mudar
|
|
|
|
if key == ord('d'):
|
|
if indice < len(arquivos) - 1:
|
|
indice += 1
|
|
else:
|
|
indice = 0
|
|
elif key == ord('a'):
|
|
if indice > 0:
|
|
indice -= 1
|
|
else:
|
|
indice = len(arquivos) - 1
|
|
elif key == ord('v'):
|
|
visualizar_corredor_principal(output_predictions, caminho_img)
|
|
elif key == ord('q'):
|
|
print("[👋] Saindo da visualização.")
|
|
break
|
|
else:
|
|
print(f"[!] Tecla inválida ({key}). Use A, D ou Q.")
|
|
|