Spaces:
Running
Running
Isaac Quarenta
fix(vision+reply+tokens): HF router free Qwen2-VL-7B, LSTM reply lock, max token largo 8192
f0c12a5 Download modules/computervision.py from akra35567/AKIRA-SOFTEDGE: direct link, hf CLI and curl.
- Browser
- Download file 38 kB
-
https://huggingface.co/spaces/akra35567/AKIRA-SOFTEDGE/resolve/main/modules/computervision.py
- Command line
-
hf download hf://spaces/akra35567/AKIRA-SOFTEDGE/modules/computervision.py
-
curl -L -o computervision.py https://huggingface.co/spaces/akra35567/AKIRA-SOFTEDGE/resolve/main/modules/computervision.py
38 kB
| # type: ignore | |
| """ | |
| modules/computervision.py | |
| ================================================================================ | |
| VISION AI MÓDULO - MULTIMODAL GEMINI + QR CODE + fallback OCR | |
| ================================================================================ | |
| Versão 3.0 - AKIRA "The Seer" | |
| Este módulo evoluiu de detecção de bordas para entendimento semântico. | |
| Pipeline de Processamento: | |
| 1. Vision AI (Multimodal): Descrição de cena, objetos, cores e contexto. | |
| Cadeia de fallbacks: | |
| 1a. HuggingFace Inference Vision (Phi-3-vision / Qwen2-VL / LLaVA - gratuito) | |
| 1b. Gemini Vision (Multimodal) | |
| 1c. ToRouter Vision (gpt-5.4-nano / gpt-4o-mini - $1 free/account) | |
| 1d. Groq Vision (Llama 3.2 - gratuito) | |
| 1e. OpenRouter Vision (Llama 3.2 / Qwen-VL - free tier) | |
| 1f. Pollinations.ai (OpenAI Vision - gratuito) | |
| 2. QR Code Scanner: Extração de dados de códigos QR. | |
| 3. OCR (Tesseract): Extração de texto (fallback para técnica/precisão). | |
| 4. CV2 Analytics: Contagem de formas e objetos (Haar Cascades). | |
| 5. RAG Visual: Armazena hashes de imagens conhecidas para lembrança rápida. | |
| Diferente da V2, este módulo não apenas "vê" pixels, ele "entende" a imagem. | |
| ================================================================================ | |
| """ | |
| import os | |
| import io | |
| import json | |
| import time | |
| import base64 | |
| import hashlib | |
| from datetime import datetime | |
| from typing import Dict, Any, List, Optional, Tuple, Union | |
| from dataclasses import dataclass | |
| from loguru import logger | |
| try: | |
| from modules.hf_inference_rotation import get_hf_inference_rotation | |
| except ImportError: | |
| from .hf_inference_rotation import get_hf_inference_rotation | |
| try: | |
| from .config import DB_PATH, GROQ_API_KEY, OPENROUTER_API_KEY, TOROUTER_MODEL, TOROUTER_VISION_MODEL, TOROUTER_BASE_URL | |
| except (ImportError, ValueError): | |
| try: | |
| from modules.config import DB_PATH, GROQ_API_KEY, OPENROUTER_API_KEY, TOROUTER_MODEL, TOROUTER_VISION_MODEL, TOROUTER_BASE_URL | |
| except ImportError: | |
| DB_PATH = "akira.db" | |
| GROQ_API_KEY = "" | |
| OPENROUTER_API_KEY = "" | |
| TOROUTER_MODEL = "openai/gpt-5.5" | |
| TOROUTER_VISION_MODEL = "openai/gpt-5.4-nano" | |
| TOROUTER_BASE_URL = "https://torouter.ai/v1" | |
| # ============================================================ | |
| # Imports Lazy para Performance | |
| # ============================================================ | |
| _cv2 = None | |
| _np = None | |
| _pytesseract = None | |
| _PIL_Image = None | |
| _genai = None | |
| _groq_client = None | |
| _openai_client = None | |
| def _check_core_deps(): | |
| global _cv2, _np, _pytesseract, _PIL_Image, _genai, _groq_client, _openai_client | |
| try: | |
| import cv2 as cv | |
| import numpy as np | |
| import pytesseract as pt | |
| from PIL import Image as PILImg | |
| _cv2, _np, _pytesseract, _PIL_Image = cv, np, pt, PILImg | |
| # Google GenAI (nova API) | |
| try: | |
| import google.genai as genai_new | |
| _genai = genai_new | |
| except ImportError: | |
| try: | |
| import google.generativeai as genai_old | |
| _genai = genai_old | |
| except ImportError: | |
| _genai = None | |
| return True | |
| except Exception as e: | |
| logger.warning(f"Visão parcial: {e}") | |
| return False | |
| _DEPS_OK = _check_core_deps() | |
| # ============================================================ | |
| # HF DNS circuit breaker — evita loop api-inference.huggingface.co que trava backend >120s | |
| # ============================================================ | |
| _HF_DNS_FAILED_AT: float | None = None | |
| HF_DNS_COOLDOWN_S = 5*60 # 5 min | |
| # ============================================================ | |
| # CONFIGURAÇÕES | |
| # ============================================================ | |
| class VisionConfig: | |
| ocr_lang: str = "por+eng" | |
| similarity_threshold: float = 0.88 | |
| max_image_res: int = 1200 | |
| enable_gemini: bool = True | |
| enable_qr: bool = True | |
| db_path: str = DB_PATH | |
| # ============================================================ | |
| # CLASSE PRINCIPAL | |
| # ============================================================ | |
| class ComputerVision: | |
| """ | |
| Controlador de Visão Computacional de Nova Geração. | |
| """ | |
| def __init__(self, config: Optional[VisionConfig] = None): | |
| self.config = config or VisionConfig() | |
| self.db_path = self.config.db_path | |
| self._setup_db() | |
| self._init_cascades() | |
| # API Key do Gemini (preferencialmente injetada via config) | |
| self.api_key = os.getenv("GEMINI_API_KEY") or os.getenv("GOOGLE_API_KEY") or "" | |
| # API Keys Groq + OpenRouter + ToRouter para fallback de visão | |
| self.groq_api_key = os.getenv("GROQ_API_KEY") or GROQ_API_KEY or "" | |
| self.openrouter_api_key = os.getenv("OPENROUTER_API_KEY") or OPENROUTER_API_KEY or "" | |
| self.torouter_api_key = os.getenv("TOROUTER_API_KEY") or os.getenv("GITAKIRA_TOROUTER_API") or "" | |
| self.torouter_base_url = os.getenv("TOROUTER_BASE_URL") or TOROUTER_BASE_URL or "https://torouter.ai/v1" | |
| self.torouter_vision_model = os.getenv("TOROUTER_VISION_MODEL") or TOROUTER_VISION_MODEL or "openai/gpt-5.4-nano" | |
| self.torouter_model = os.getenv("TOROUTER_MODEL") or TOROUTER_MODEL or "openai/gpt-5.5" | |
| # HF_TOKEN para Vision (Qwen2-VL via HF Inference) | |
| self.hf_token = os.getenv("HF_TOKEN") or os.getenv("ANN_HF_TOKEN") or os.getenv("ISAAC_HF_TOKEN") or "" | |
| if not self.hf_token: | |
| logger.warning("⚠️ HF_TOKEN não configurado — HF Vision (Qwen2-VL) desabilitado") | |
| def _setup_db(self): | |
| """Garante tabela de memória visual.""" | |
| try: | |
| from .database import Database | |
| db = Database(self.db_path) | |
| db._execute_with_retry(""" | |
| CREATE TABLE IF NOT EXISTS image_memory ( | |
| hash TEXT PRIMARY KEY, | |
| user_id TEXT, | |
| description TEXT, | |
| ocr_text TEXT, | |
| qr_data TEXT, | |
| metadata TEXT, | |
| timestamp TIMESTAMP | |
| ) | |
| """, commit=True) | |
| except Exception as e: | |
| logger.error(f"Erro DB Visão: {e}") | |
| def _init_cascades(self): | |
| """Carrega modelos Haar Cascades para detecção básica.""" | |
| if not _cv2: return | |
| try: | |
| self._face_cascade = _cv2.CascadeClassifier(_cv2.data.haarcascades + 'haarcascade_frontalface_default.xml') | |
| except: | |
| self._face_cascade = None | |
| # ================================================================== | |
| # 🎯 PIPELINE PRINCIPAL | |
| # ================================================================== | |
| # ================================================================== | |
| # PROCESSAMENTO | |
| # ================================================================== | |
| def analyze_image(self, input_data: Union[str, bytes], user_id: str = "anon") -> Dict[str, Any]: | |
| """ | |
| Processa imagem através de todo o pipeline. | |
| Aceita: Caminho de arquivo (str), Base64 (str) ou Bytes brutos (bytes). | |
| """ | |
| if not input_data: return {"success": False, "error": "Entrada vazia"} | |
| img_bytes = None | |
| try: | |
| # 1. Detecção e Normalização da Entrada | |
| if isinstance(input_data, bytes): | |
| img_bytes = input_data | |
| elif isinstance(input_data, str): | |
| # Caso A: Caminho de arquivo local | |
| if os.path.isfile(input_data): | |
| with open(input_data, "rb") as f: | |
| img_bytes = f.read() | |
| # Caso B: Base64 | |
| else: | |
| try: | |
| b64_str = input_data | |
| if "," in b64_str: b64_str = b64_str.split(",")[1] | |
| img_bytes = base64.b64decode(b64_str) | |
| except Exception: | |
| return {"success": False, "error": "String informada não é um caminho válido nem Base64 válido"} | |
| if not img_bytes: | |
| return {"success": False, "error": "Falha ao extrair bytes da imagem"} | |
| img_hash = hashlib.md5(img_bytes).hexdigest() | |
| # 2. Check Memória Visual (Cache BD) | |
| cached = self._get_from_memory(img_hash) | |
| if cached: | |
| logger.info(f"🧠 Memória Visual recordada: {img_hash}") | |
| cached["cached"] = True | |
| return cached | |
| # 2.1 Validação de magic bytes antes de PIL (corrige PIL.UnidentifiedImageError para HTML/corrupt) | |
| # PNG: 89 50 4E 47, JPEG: FF D8 FF, GIF: 47 49 46 (GIF87a/GIF89a) | |
| _is_valid_magic = False | |
| if img_bytes.startswith(b"\x89PNG"): | |
| _is_valid_magic = True | |
| elif img_bytes.startswith(b"\xff\xd8\xff"): | |
| _is_valid_magic = True | |
| elif img_bytes.startswith(b"GIF87a") or img_bytes.startswith(b"GIF89a"): | |
| _is_valid_magic = True | |
| elif img_bytes.startswith(b"BM") or (img_bytes.startswith(b"RIFF") and b"WEBP" in img_bytes[:12]) or img_bytes.startswith(b"\x00\x00\x01\x00"): | |
| # BMP, WEBP, ICO — formatos válidos adicionais para não quebrar retrocompatibilidade | |
| _is_valid_magic = True | |
| if not _is_valid_magic: | |
| stripped = img_bytes.lstrip()[:32].lower() | |
| if stripped.startswith(b"<") or stripped.startswith(b"<!doctype") or b"<html" in stripped: | |
| logger.warning(f"⚠️ [VISION] Bytes parecem HTML, não imagem (magic bytes inválidos, {len(img_bytes)}B)") | |
| else: | |
| logger.warning(f"⚠️ [VISION] Magic bytes inválidos: {img_bytes[:8]!r} ({len(img_bytes)}B)") | |
| return {"success": False, "error": "invalid_image", "description": "Imagem não pôde ser identificada"} | |
| # 3. Preparação para OCR e CV2 | |
| nparr = _np.frombuffer(img_bytes, _np.uint8) | |
| img_cv = _cv2.imdecode(nparr, _cv2.IMREAD_COLOR) | |
| try: | |
| pil_img = _PIL_Image.open(io.BytesIO(img_bytes)) | |
| # Força load para detectar imagens truncadas que só falham no decode | |
| pil_img.load() | |
| # Rewind após load() pois load() pode ter consumido o stream em algumas versões | |
| pil_img = _PIL_Image.open(io.BytesIO(img_bytes)) | |
| except Exception as e: | |
| # Trata PIL.UnidentifiedImageError (subclasse de OSError) e outros erros de decode | |
| try: | |
| from PIL.UnidentifiedImageError import UnidentifiedImageError # type: ignore | |
| except ImportError: | |
| try: | |
| from PIL import UnidentifiedImageError # type: ignore | |
| except ImportError: | |
| UnidentifiedImageError = OSError # fallback | |
| if isinstance(e, UnidentifiedImageError) or isinstance(e, (OSError, ValueError)): | |
| logger.warning(f"⚠️ [VISION] PIL não pôde identificar imagem ({len(img_bytes)}B): {e}") | |
| else: | |
| logger.warning(f"⚠️ [VISION] Erro ao abrir imagem com PIL: {e}") | |
| return {"success": False, "error": "invalid_image", "description": "Imagem não pôde ser identificada"} | |
| # 🔧 AUTO-RESIZE: Redimensiona se excede max_image_res (padrão 1200px) | |
| # Economiza tokens/tempo nas APIs de visão e evita erros de payload grande | |
| max_res = self.config.max_image_res | |
| if img_cv is not None: | |
| h, w = img_cv.shape[:2] | |
| if max(h, w) > max_res: | |
| scale = max_res / max(h, w) | |
| new_w, new_h = int(w * scale), int(h * scale) | |
| logger.info(f"🔧 [VISION] Redimensionando {w}x{h} → {new_w}x{new_h} (max_res={max_res})") | |
| img_cv = _cv2.resize(img_cv, (new_w, new_h), interpolation=_cv2.INTER_AREA) | |
| # Re-encode para bytes para as APIs de visão | |
| _, img_bytes = _cv2.imencode('.jpg', img_cv, [_cv2.IMWRITE_JPEG_QUALITY, 85]) | |
| img_bytes = img_bytes.tobytes() | |
| # Recria PIL para OCR (protegido contra UnidentifiedImageError) | |
| try: | |
| pil_img = _PIL_Image.open(io.BytesIO(img_bytes)) | |
| pil_img.load() | |
| pil_img = _PIL_Image.open(io.BytesIO(img_bytes)) | |
| except Exception as e: | |
| logger.warning(f"⚠️ [VISION] Falha ao recriar PIL após resize: {e}") | |
| # Mantém pil_img anterior; se não houver, tenta fallback via cv2 | |
| try: | |
| if pil_img is None: | |
| raise ValueError("pil_img anterior nulo") | |
| except Exception: | |
| # Fallback: tenta criar PIL a partir do img_cv | |
| try: | |
| from PIL import Image as _FallbackPIL | |
| import cv2 as _cv2_fallback | |
| # Converte BGR -> RGB | |
| rgb = _cv2_fallback.cvtColor(img_cv, _cv2_fallback.COLOR_BGR2RGB) | |
| pil_img = _FallbackPIL.fromarray(rgb) | |
| except Exception: | |
| pass | |
| # --- EXECUÇÃO DO PIPELINE --- | |
| # A. QR Code (Rápido) | |
| qr_data = self._scan_qr(img_cv) if self.config.enable_qr else None | |
| # B. Vision AI (Semântico - O Coração) | |
| descricao = "" | |
| # Cadeia de fallbacks: HF (free, Qwen2-VL) → Pollinations → OpenRouter → Groq → Gemini → ToRouter | |
| descricao = self._hf_visual_analyze(img_bytes) | |
| if not descricao: | |
| descricao = self._pollinations_visual_analyze(img_bytes) | |
| if not descricao: | |
| descricao = self._openrouter_visual_analyze(img_bytes) | |
| if not descricao: | |
| descricao = self._groq_visual_analyze(img_bytes) | |
| if not descricao and self.config.enable_gemini and self.api_key: | |
| descricao = self._gemini_visual_analyze(img_bytes) | |
| if not descricao: | |
| descricao = self._torouter_visual_analyze(img_bytes) | |
| # C. OCR (Fallback/Técnico) | |
| ocr_text = self._run_ocr(pil_img) | |
| # D. CV2 Analytics (Estatístico/Objetos) | |
| analytics = self._run_cv2_analytics(img_cv) | |
| # 4. Consolidação | |
| result = { | |
| "success": True, | |
| "hash": img_hash, | |
| "description": descricao or "Não foi possível descrever a imagem semanticamente.", | |
| "ocr": ocr_text, | |
| "qr": qr_data, | |
| "objects": analytics.get("objects", []), | |
| "details": { | |
| "faces": analytics.get("faces", 0), | |
| "resolution": f"{img_cv.shape[1]}x{img_cv.shape[0]}" if img_cv is not None else "N/A" | |
| }, | |
| "timestamp": datetime.now().isoformat() | |
| } | |
| # 5. Salva na Memória | |
| self._save_to_memory(result, user_id) | |
| return result | |
| except Exception as e: | |
| logger.exception("Falha no pipeline de visão") | |
| return {"success": False, "error": str(e)} | |
| # ================================================================== | |
| # 👁️ MOTORES ESPECÍFICOS | |
| # ================================================================== | |
| def _gemini_visual_analyze(self, img_bytes: bytes) -> str: | |
| """Usa Google Gemini Multimodal para descrever a imagem.""" | |
| if not _genai or not self.api_key: return "" | |
| try: | |
| # Detecta se é a API nova ou antiga | |
| if hasattr(_genai, 'Client'): # Nova API google.genai | |
| client = _genai.Client(api_key=self.api_key) | |
| # Otimizado: Tenta os modelos mais novos (Série 3 e 3.1) primeiro | |
| # FIX 2026-08-28: gemini-3.5-flash-lite NÃO EXISTE. Usar modelos reais. | |
| model_priority = [ | |
| "gemini-2.5-flash", | |
| "gemini-2.5-flash-lite", | |
| "gemini-2.0-flash", | |
| ] | |
| # Se houver modelo configurado no ENV, coloca no topo da lista | |
| env_model = os.getenv("GEMINI_MODEL", "") | |
| if env_model and env_model not in model_priority: | |
| model_priority.insert(0, env_model) | |
| # Detetar MimeType dinâmico | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| last_err = None | |
| for model_id in model_priority: | |
| try: | |
| logger.info(f"👁️ Tentando Gemini Vision com modelo: {model_id}") | |
| response = client.models.generate_content( | |
| model=model_id, | |
| contents=[ | |
| "Analise esta imagem com extrema precisão para uma IA assistente autônoma. Descreva tudo: objetos, textos, contexto, ambiente, cores e expressões. Se houver códigos, links ou dados sensíveis, extraia-os. Seja assertivo.", | |
| _genai.types.Part.from_bytes(data=img_bytes, mime_type=mime_type), | |
| ] | |
| ) | |
| if response and response.text: | |
| logger.success(f"✅ Gemini Vision ({model_id}) sucesso!") | |
| return response.text | |
| except Exception as e: | |
| last_err = e | |
| if "404" in str(e) or "not found" in str(e).lower() or "permission" in str(e).lower(): | |
| logger.warning(f"⚠️ Modelo {model_id} indisponível ou sem permissão. Tentando próximo...") | |
| continue | |
| logger.error(f"❌ Erro crítico no modelo {model_id}: {e}") | |
| break | |
| if last_err: raise last_err | |
| return "" | |
| else: | |
| # API antiga google.generativeai | |
| _genai.configure(api_key=self.api_key) | |
| # Tenta 1.5 Flash que é mais estável na API antiga | |
| model_name = 'gemini-1.5-flash' | |
| model = _genai.GenerativeModel(model_name) | |
| response = model.generate_content([ | |
| "Descreva esta imagem detalhadamente. Seja direto e informativo.", | |
| _PIL_Image.open(io.BytesIO(img_bytes)) | |
| ]) | |
| return response.text if response else "" | |
| except Exception as e: | |
| logger.warning(f"Gemini Vision falhou: {e}") | |
| return "" | |
| def _groq_visual_analyze(self, img_bytes: bytes) -> str: | |
| """Fallback usando Groq (Llama 3.2 Vision - gratuito).""" | |
| if not self.groq_api_key: | |
| return "" | |
| try: | |
| from groq import Groq | |
| import base64 | |
| client = Groq(api_key=self.groq_api_key) | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| b64_data = base64.b64encode(img_bytes).decode('utf-8') | |
| logger.info("👁️ Usando Groq Vision (Llama 3.2) como fallback...") | |
| for model_id in ["llama-3.2-11b-vision-preview", "llama-3.2-90b-vision-preview"]: | |
| try: | |
| response = client.chat.completions.create( | |
| model=model_id, | |
| messages=[{ | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": "Descreva esta imagem em detalhes para uma IA assistente. O que você vê?"}, | |
| {"type": "image_url", "image_url": {"url": f"data:{mime_type};base64,{b64_data}"}} | |
| ] | |
| }], | |
| max_tokens=500 | |
| ) | |
| if response and response.choices: | |
| text = response.choices[0].message.content or "" | |
| if text.strip(): | |
| logger.success(f"✅ Groq Vision ({model_id}) sucesso!") | |
| return text | |
| except Exception as e: | |
| logger.warning(f"Groq Vision {model_id} falhou: {e}") | |
| continue | |
| return "" | |
| except Exception as e: | |
| logger.warning(f"Groq Vision falhou: {e}") | |
| return "" | |
| def _openrouter_visual_analyze(self, img_bytes: bytes) -> str: | |
| """Fallback usando OpenRouter (multi-modelo visão - free tier).""" | |
| if not self.openrouter_api_key: | |
| return "" | |
| try: | |
| from openai import OpenAI | |
| import base64 | |
| client = OpenAI(api_key=self.openrouter_api_key, base_url="https://openrouter.ai/api/v1") | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| b64_data = base64.b64encode(img_bytes).decode('utf-8') | |
| models = [ | |
| "meta-llama/llama-3.2-11b-vision:free", | |
| "qwen/qwen2-vl-72b-instruct", | |
| "meta-llama/llama-3.2-90b-vision:free" | |
| ] | |
| logger.info("👁️ Usando OpenRouter Vision como fallback...") | |
| for model_id in models: | |
| try: | |
| response = client.chat.completions.create( | |
| model=model_id, | |
| messages=[{ | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": "Descreva esta imagem em detalhes para uma IA assistente. O que você vê?"}, | |
| {"type": "image_url", "image_url": {"url": f"data:{mime_type};base64,{b64_data}"}} | |
| ] | |
| }], | |
| max_tokens=500 | |
| ) | |
| if response and response.choices: | |
| text = response.choices[0].message.content or "" | |
| if text.strip(): | |
| logger.success(f"✅ OpenRouter Vision ({model_id}) sucesso!") | |
| return text | |
| except Exception as e: | |
| logger.warning(f"OpenRouter Vision {model_id} falhou: {e}") | |
| continue | |
| return "" | |
| except Exception as e: | |
| logger.warning(f"OpenRouter Vision falhou: {e}") | |
| return "" | |
| def _torouter_visual_analyze(self, img_bytes: bytes) -> str: | |
| """Fallback usando ToRouter com modelos baratos (gpt-5.4-nano / gpt-4o-mini).""" | |
| if not self.torouter_api_key: | |
| return "" | |
| try: | |
| from openai import OpenAI | |
| import base64 | |
| client = OpenAI(api_key=self.torouter_api_key, base_url=self.torouter_base_url) | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| b64_data = base64.b64encode(img_bytes).decode('utf-8') | |
| models = [ | |
| self.torouter_vision_model, | |
| "openai/gpt-4o-mini", | |
| "xiaomi/mimo-v2.5" | |
| ] | |
| logger.info("👁️ Usando ToRouter Vision como fallback...") | |
| for model_id in models: | |
| try: | |
| response = client.chat.completions.create( | |
| model=model_id, | |
| messages=[{ | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": "Descreva esta imagem em detalhes para uma IA assistente. O que você vê?"}, | |
| {"type": "image_url", "image_url": {"url": f"data:{mime_type};base64,{b64_data}"}} | |
| ] | |
| }], | |
| max_tokens=500 | |
| ) | |
| if response and response.choices: | |
| text = response.choices[0].message.content or "" | |
| if text.strip(): | |
| logger.success(f"✅ ToRouter Vision ({model_id}) sucesso!") | |
| return text | |
| except Exception as e: | |
| logger.warning(f"ToRouter Vision {model_id} falhou: {e}") | |
| continue | |
| return "" | |
| except Exception as e: | |
| logger.warning(f"ToRouter Vision falhou: {e}") | |
| return "" | |
| def _hf_visual_analyze(self, img_bytes: Union[bytes, str, None] = None, img_url: str = "") -> str: | |
| """HF Inference Vision API — aceita bytes OU URL direta (o servidor HF consegue baixar de catbox).""" | |
| global _HF_DNS_FAILED_AT | |
| if _HF_DNS_FAILED_AT is not None and (time.time() - _HF_DNS_FAILED_AT) < HF_DNS_COOLDOWN_S: | |
| logger.info("HF vision em cooldown DNS, pulando") | |
| return "" | |
| try: | |
| import requests as req | |
| import base64 | |
| except ImportError: | |
| return "" | |
| hf_rotation = get_hf_inference_rotation() | |
| prompt = "Descreva esta imagem em detalhes para uma IA assistente. O que você vê? Responda em português." | |
| # Construir image_url content block | |
| if img_url: | |
| image_url_content = {"type": "image_url", "image_url": {"url": img_url}} | |
| elif isinstance(img_bytes, bytes) and img_bytes: | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| b64_data = base64.b64encode(img_bytes).decode('utf-8') | |
| image_url_content = {"type": "image_url", "image_url": {"url": f"data:{mime_type};base64,{b64_data}"}} | |
| else: | |
| return "" | |
| # FIX 2026-09-04: multimodal FREE - Qwen2-VL-7B priorizado para PT, router.huggingface.co (api-inference deprecated entra em cooldown) | |
| models = [ | |
| "Qwen/Qwen2-VL-7B-Instruct", | |
| "microsoft/Phi-3-vision-128k-instruct", | |
| "llava-hf/llava-v1.6-mistral-7b-hf", | |
| "llava-hf/llava-1.5-7b-hf", | |
| ] | |
| # Prompt profundo para análise detalhada (HF free) | |
| prompt = "Descreva esta imagem em PORTUGUÊS com análise profunda: 1) cena/iluminação/composição/cores, 2) objetos com posição/cor/material, 3) TODO texto visível (OCR) entre aspas, 4) faces/emoções/idade/vestuário, 5) contexto/inferência cultural (se Angola/Luanda priorize), 6) QR/links se visível. Seja assertivo, não alucine." | |
| logger.info(f"🤗 HF Inference Vision (FREE multimodal): {'URL' if img_url else f'{len(img_bytes) if isinstance(img_bytes, bytes) else 0}B'} -> Qwen2-VL-7B primeiro") | |
| dns_failed = False | |
| max_retries = len(hf_rotation.account_order) * 2 | |
| # Endpoints: router novo primeiro, legacy apenas fallback | |
| hf_endpoints = [ | |
| "https://huggingface.co/static-proxy/router.huggingface.co/hf-inference/models/{model}/v1/chat/completions", | |
| "https://huggingface.co/static-proxy/api-inference.huggingface.co/models/{model}/v1/chat/completions", | |
| ] | |
| for retry in range(max_retries): | |
| token = hf_rotation.get_current_api_token() | |
| if not token: | |
| logger.warning("⚠️ HF Inference: nenhum token disponível") | |
| return "" | |
| _should_retry_token = False | |
| for model_id in models: | |
| for api_url_template in hf_endpoints: | |
| try: | |
| api_url = api_url_template.format(model=model_id) | |
| # router/hf-inference requer token; se for legacy e DNS falhou, pular | |
| if "api-inference.huggingface.co" in api_url and _HF_DNS_FAILED_AT is not None and (time.time() - _HF_DNS_FAILED_AT) < HF_DNS_COOLDOWN_S: | |
| continue | |
| headers = { | |
| "Authorization": f"Bearer {token}", | |
| "Content-Type": "application/json", | |
| } | |
| payload = { | |
| "model": model_id, | |
| "messages": [ | |
| { | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": prompt}, | |
| image_url_content, | |
| ], | |
| } | |
| ], | |
| "max_tokens": 800, | |
| } | |
| response = req.post(api_url, headers=headers, json=payload, timeout=30) | |
| if response.status_code == 200: | |
| data = response.json() | |
| text = data.get("choices", [{}])[0].get("message", {}).get("content", "") | |
| if text.strip(): | |
| logger.success(f"✅ HF Vision FREE ({model_id} via {api_url.split('/')[2]}) OK") | |
| return text | |
| elif response.status_code == 429: | |
| logger.warning(f"⚠️ HF 429 em {model_id} ({api_url.split('/')[2]})") | |
| hf_rotation.handle_rate_limit_error("429") | |
| _should_retry_token = True | |
| break | |
| elif response.status_code == 503: | |
| logger.info(f"⏳ HF {model_id} loading (503) — tentando próximo endpoint/modelo") | |
| continue | |
| else: | |
| logger.warning(f"⚠️ HF {model_id} HTTP {response.status_code} em {api_url.split('/')[2]}: {response.text[:120]}") | |
| except Exception as e: | |
| err_str = str(e) | |
| if "NameResolutionError" in err_str or "No address associated with hostname" in err_str or "api-inference.huggingface.co" in err_str: | |
| if not dns_failed: | |
| logger.warning(f"HF DNS fail (api-inference.huggingface.co deprecated) - marcando cooldown e tentando router: {e}") | |
| dns_failed = True | |
| _HF_DNS_FAILED_AT = time.time() | |
| # não retorna, tenta router na próxima iteração | |
| continue | |
| logger.warning(f"HF {model_id}: {e}") | |
| continue | |
| if _should_retry_token: | |
| break | |
| # se chegou aqui sem retry, tenta próximo modelo | |
| continue | |
| return "" | |
| def _pollinations_visual_analyze(self, img_bytes: bytes) -> str: | |
| """Fallback Gratuito usando Pollinations.ai (Modelo OpenAI Vision).""" | |
| try: | |
| import requests | |
| import base64 | |
| logger.info("🎙️ Usando Pollinations (Poly) para visão gratuita...") | |
| # Detetar MimeType dinâmico | |
| mime_type = "image/png" if img_bytes.startswith(b"\x89PNG") else "image/jpeg" | |
| b64_data = base64.b64encode(img_bytes).decode('utf-8') | |
| payload = { | |
| "model": "openai", | |
| "messages": [ | |
| { | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": "Descreva esta imagem em detalhes para uma IA assistente. O que você vê?"}, | |
| { | |
| "type": "image_url", | |
| "image_url": {"url": f"data:{mime_type};base64,{b64_data}"} | |
| } | |
| ] | |
| } | |
| ] | |
| } | |
| response = requests.post( | |
| "https://gen.pollinations.ai/v1/chat/completions", | |
| json=payload, | |
| timeout=30 | |
| ) | |
| if response.status_code == 200: | |
| data = response.json() | |
| res_text = data['choices'][0]['message']['content'] | |
| logger.info(f"✅ Pollinations Vision OK: {res_text[:50]}...") | |
| return res_text | |
| return "" | |
| except Exception as e: | |
| logger.warning(f"Pollinations Vision falhou: {e}") | |
| return "" | |
| def _scan_qr(self, img_cv) -> Optional[str]: | |
| """Detecta e decodifica QR Code.""" | |
| if not _cv2 or img_cv is None: return None | |
| try: | |
| detector = _cv2.QRCodeDetector() | |
| data, _, _ = detector.detectAndDecode(img_cv) | |
| return data if data else None | |
| except: | |
| return None | |
| def _run_ocr(self, pil_img) -> str: | |
| """Extrai texto da imagem via Tesseract.""" | |
| if not _pytesseract: return "" | |
| try: | |
| return _pytesseract.image_to_string(pil_img, lang=self.config.ocr_lang).strip() | |
| except: | |
| return "" | |
| def _run_cv2_analytics(self, img_cv) -> Dict[str, Any]: | |
| """Detecta faces e extrai metadados visuais básicos.""" | |
| res = {"faces": 0, "objects": []} | |
| if not _cv2 or img_cv is None: return res | |
| try: | |
| gray = _cv2.cvtColor(img_cv, _cv2.COLOR_BGR2GRAY) | |
| # Faces | |
| if self._face_cascade: | |
| faces = self._face_cascade.detectMultiScale(gray, 1.1, 4) | |
| res["faces"] = len(faces) | |
| if len(faces) > 0: res["objects"].append("pessoa/rosto") | |
| # Brilho médio | |
| avg_color = _np.mean(img_cv, axis=(0, 1)) | |
| res["avg_color_bgr"] = avg_color.tolist() | |
| except: pass | |
| return res | |
| # ================================================================== | |
| # 🗄️ PERSISTÊNCIA (MEMÓRIA VISUAL) | |
| # ================================================================== | |
| def _get_from_memory(self, img_hash: str) -> Optional[Dict]: | |
| try: | |
| from .database import Database | |
| db = Database(self.db_path) | |
| rows = db._execute_with_retry("SELECT * FROM image_memory WHERE hash = %s", (img_hash,)) | |
| if rows: | |
| res = dict(rows[0]) | |
| return { | |
| "success": True, | |
| "hash": res["hash"], | |
| "description": res["description"], | |
| "ocr": res["ocr_text"], | |
| "qr": res["qr_data"], | |
| "timestamp": res["timestamp"], | |
| "from_memory": True | |
| } | |
| except: pass | |
| return None | |
| def _save_to_memory(self, result: Dict, user_id: str): | |
| try: | |
| from .database import Database | |
| db = Database(self.db_path) | |
| db._execute_with_retry(""" | |
| INSERT INTO image_memory | |
| (hash, user_id, description, ocr_text, qr_data, metadata, timestamp) | |
| VALUES (%s, %s, %s, %s, %s, %s, %s) | |
| ON CONFLICT (hash) DO UPDATE SET | |
| description=EXCLUDED.description, ocr_text=EXCLUDED.ocr_text, | |
| qr_data=EXCLUDED.qr_data, metadata=EXCLUDED.metadata, timestamp=EXCLUDED.timestamp | |
| """, ( | |
| result["hash"], | |
| user_id, | |
| result["description"], | |
| result["ocr"], | |
| result["qr"], | |
| json.dumps(result.get("details", {})), | |
| result["timestamp"] | |
| ), commit=True) | |
| except Exception as e: | |
| logger.debug(f"Erro ao salvar memória visual: {e}") | |
| # ============================================================ | |
| # SINGLETON EXPORT | |
| # ============================================================ | |
| _vision_instance = None | |
| def get_computer_vision(config=None) -> ComputerVision: | |
| global _vision_instance | |
| if _vision_instance is None: | |
| _vision_instance = ComputerVision(config) | |
| return _vision_instance | |
| def analyze_image_base64(b64_str: str, user_id: str = "anon") -> Dict[str, Any]: | |
| return get_computer_vision().analyze_image(b64_str, user_id) | |
| __all__ = ["ComputerVision", "get_computer_vision", "analyze_image_base64", | |
| "ImageFeature", "analyze_image_from_base64", "analyze_image_file"] | |
| # ============================================================ | |
| # COMPATIBILIDADE — aliases para imports legados | |
| # ============================================================ | |
| class ImageFeature: | |
| """Representação simplificada de features de uma imagem.""" | |
| description: str = "" | |
| ocr_text: str = "" | |
| qr_data: Optional[str] = None | |
| objects: List[str] = None # type: ignore | |
| def __post_init__(self): | |
| if self.objects is None: | |
| self.objects = [] | |
| def analyze_image_from_base64(b64_str: str, user_id: str = "anon") -> Dict[str, Any]: | |
| """Alias legado para analyze_image_base64.""" | |
| return analyze_image_base64(b64_str, user_id) | |
| def analyze_image_file(filepath: str, user_id: str = "anon") -> Dict[str, Any]: | |
| """Analisa imagem a partir de caminho de arquivo.""" | |
| return get_computer_vision().analyze_image(filepath, user_id) | |