OCR de imagenes con Mistral: ImageProcessor (png/jpg/webp/gif/bmp/tiff), modelo mistral-ocr-latest y recuperacion de fuentes unknown
This commit is contained in:
@@ -6,6 +6,7 @@ from backend.app.managers.knowledge_manager import KnowledgeManager
|
||||
from backend.app.services.markdown_builder import build_markdown
|
||||
from backend.app.services.processors.audio_processor import AudioProcessor
|
||||
from backend.app.services.processors.docx_processor import DocxProcessor
|
||||
from backend.app.services.processors.image_processor import ImageProcessor
|
||||
from backend.app.services.processors.pdf_processor import PdfProcessor
|
||||
from backend.app.services.processors.text_processor import TextProcessor
|
||||
from backend.app.services.processors.video_processor import VideoProcessor
|
||||
@@ -19,7 +20,8 @@ class IngestionService:
|
||||
source = self.manager.get_source(source_id)
|
||||
context = self.manager.get_week_context(source["week_id"])
|
||||
path = Path(source["stored_path"])
|
||||
processor = self._processor_for(source["source_type"], use_ocr, page_ranges)
|
||||
source_type = self._resolve_source_type(source, path)
|
||||
processor = self._processor_for(source_type, use_ocr, page_ranges)
|
||||
processed = processor.process(path)
|
||||
markdown = build_markdown(
|
||||
title=processed.title,
|
||||
@@ -29,14 +31,21 @@ class IngestionService:
|
||||
"subject_slug": context["subject_slug"],
|
||||
"week": context["week_number"],
|
||||
"source_file": source["original_name"],
|
||||
"source_type": source["source_type"],
|
||||
"source_type": source_type,
|
||||
"processor": processed.processor,
|
||||
"page_ranges": page_ranges if source["source_type"] == "pdf" and page_ranges else None,
|
||||
"page_ranges": page_ranges if source_type == "pdf" and page_ranges else None,
|
||||
"language": "es",
|
||||
},
|
||||
)
|
||||
return self.manager.create_document(source_id, processed.title, markdown, processor=processed.processor, page_ranges=page_ranges)
|
||||
|
||||
def _resolve_source_type(self, source: dict, path: Path) -> str:
|
||||
if source["source_type"] != "unknown":
|
||||
return source["source_type"]
|
||||
# Archivos subidos antes de soportar su formato quedaron como
|
||||
# ``unknown``; al reprocesarlos se recalcula el tipo por extension.
|
||||
return self.manager.refresh_source_type(source["id"])
|
||||
|
||||
def _processor_for(self, source_type: str, use_ocr: bool, page_ranges: str | None):
|
||||
if source_type == "text":
|
||||
return TextProcessor()
|
||||
@@ -44,6 +53,8 @@ class IngestionService:
|
||||
return DocxProcessor()
|
||||
if source_type == "pdf":
|
||||
return PdfProcessor(use_ocr=use_ocr, page_ranges=page_ranges)
|
||||
if source_type == "image":
|
||||
return ImageProcessor()
|
||||
if source_type == "audio":
|
||||
return AudioProcessor()
|
||||
if source_type == "video":
|
||||
|
||||
Reference in New Issue
Block a user