OCR de imagenes con Mistral: ImageProcessor (png/jpg/webp/gif/bmp/tiff), modelo mistral-ocr-latest y recuperacion de fuentes unknown

This commit is contained in:
urieljareth
2026-09-13 22:39:39 -06:00
parent 239a7510c6
commit 15cf00cc42
12 changed files with 79 additions and 14 deletions
+1 -1
View File
@@ -8,7 +8,7 @@ class Settings(BaseSettings):
app_data_dir: Path = Path("./data")
database_path: Path = Path("./data/app.db")
mistral_api_key: str = ""
mistral_ocr_model: str = "mistral-ocr-4-0"
mistral_ocr_model: str = "mistral-ocr-latest"
mistral_ocr_batch_threshold: int = 3
mistral_ocr_batch_poll_seconds: int = 10
mistral_ocr_batch_timeout_seconds: int = 1800
+1
View File
@@ -34,6 +34,7 @@ def health() -> dict[str, str]:
def settings_status() -> dict:
return {
"mistral_configured": bool(settings.mistral_api_key),
"mistral_ocr_model": settings.mistral_ocr_model,
"deepgram_configured": bool(settings.deepgram_api_key),
"data_dir": str(settings.app_data_dir),
"database_path": str(settings.database_path),
+14
View File
@@ -378,6 +378,18 @@ class KnowledgeManager:
raise ValueError("Fuente no encontrada")
return dict(row)
def refresh_source_type(self, source_id: int) -> str:
"""Recalcula y persiste el tipo de una fuente guardada como ``unknown``.
Permite reprocesar archivos subidos antes de soportar su formato
(por ejemplo, imagenes que quedaron como ``unknown`` y fallaban).
"""
source = self.get_source(source_id)
source_type = self.detect_source_type(Path(source["stored_path"]))
with self._connect() as conn:
conn.execute("update sources set source_type = ? where id = ?", (source_type, source_id))
return source_type
def list_sources(self, subject_id: int, week_number: int) -> list[dict[str, Any]]:
week = self.get_week(subject_id, week_number)
with self._connect() as conn:
@@ -552,6 +564,8 @@ class KnowledgeManager:
return "docx"
if ext == "pdf":
return "pdf"
if ext in {"png", "jpg", "jpeg", "webp", "gif", "bmp", "tif", "tiff"}:
return "image"
if ext in {"mp3", "wav", "m4a", "ogg", "flac", "webm"}:
return "audio"
if ext in {"mp4", "mov", "mkv", "avi"}:
+14 -3
View File
@@ -6,6 +6,7 @@ from backend.app.managers.knowledge_manager import KnowledgeManager
from backend.app.services.markdown_builder import build_markdown
from backend.app.services.processors.audio_processor import AudioProcessor
from backend.app.services.processors.docx_processor import DocxProcessor
from backend.app.services.processors.image_processor import ImageProcessor
from backend.app.services.processors.pdf_processor import PdfProcessor
from backend.app.services.processors.text_processor import TextProcessor
from backend.app.services.processors.video_processor import VideoProcessor
@@ -19,7 +20,8 @@ class IngestionService:
source = self.manager.get_source(source_id)
context = self.manager.get_week_context(source["week_id"])
path = Path(source["stored_path"])
processor = self._processor_for(source["source_type"], use_ocr, page_ranges)
source_type = self._resolve_source_type(source, path)
processor = self._processor_for(source_type, use_ocr, page_ranges)
processed = processor.process(path)
markdown = build_markdown(
title=processed.title,
@@ -29,14 +31,21 @@ class IngestionService:
"subject_slug": context["subject_slug"],
"week": context["week_number"],
"source_file": source["original_name"],
"source_type": source["source_type"],
"source_type": source_type,
"processor": processed.processor,
"page_ranges": page_ranges if source["source_type"] == "pdf" and page_ranges else None,
"page_ranges": page_ranges if source_type == "pdf" and page_ranges else None,
"language": "es",
},
)
return self.manager.create_document(source_id, processed.title, markdown, processor=processed.processor, page_ranges=page_ranges)
def _resolve_source_type(self, source: dict, path: Path) -> str:
if source["source_type"] != "unknown":
return source["source_type"]
# Archivos subidos antes de soportar su formato quedaron como
# ``unknown``; al reprocesarlos se recalcula el tipo por extension.
return self.manager.refresh_source_type(source["id"])
def _processor_for(self, source_type: str, use_ocr: bool, page_ranges: str | None):
if source_type == "text":
return TextProcessor()
@@ -44,6 +53,8 @@ class IngestionService:
return DocxProcessor()
if source_type == "pdf":
return PdfProcessor(use_ocr=use_ocr, page_ranges=page_ranges)
if source_type == "image":
return ImageProcessor()
if source_type == "audio":
return AudioProcessor()
if source_type == "video":
@@ -0,0 +1,19 @@
from __future__ import annotations
from pathlib import Path
from backend.app.services.markdown_builder import title_from_path
from backend.app.services.processors.base import ProcessedContent
from backend.app.services.providers.mistral_client import MistralClient
class ImageProcessor:
"""Procesa imagenes sueltas (png, jpg, webp, gif, bmp, tiff) con Mistral OCR.
No hay extraccion de texto alternativa para imagenes: siempre pasan por
OCR, independiente de la opcion ``use_ocr``.
"""
def process(self, path: Path) -> ProcessedContent:
markdown = MistralClient().ocr_image(path)
return ProcessedContent(title=title_from_path(path), body=markdown, processor="mistral-ocr-image")
@@ -36,6 +36,17 @@ class MistralClient:
files_url = f"{base_url}/files"
batch_url = f"{base_url}/batch/jobs"
_IMAGE_MIME_BY_EXT = {
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".webp": "image/webp",
".gif": "image/gif",
".bmp": "image/bmp",
".tif": "image/tiff",
".tiff": "image/tiff",
}
def __init__(self) -> None:
self._api_key = settings.mistral_api_key
self._model = settings.mistral_ocr_model
@@ -285,4 +296,5 @@ class MistralClient:
def _image_data_url(self, image_path: Path) -> str:
encoded = base64.b64encode(image_path.read_bytes()).decode("ascii")
return f"data:image/jpeg;base64,{encoded}"
mime = self._IMAGE_MIME_BY_EXT.get(image_path.suffix.lower(), "image/jpeg")
return f"data:{mime};base64,{encoded}"