OCR de imagenes con Mistral: ImageProcessor (png/jpg/webp/gif/bmp/tiff), modelo mistral-ocr-latest y recuperacion de fuentes unknown
This commit is contained in:
+1
-1
@@ -2,7 +2,7 @@ APP_ENV=development
|
|||||||
APP_DATA_DIR=./data
|
APP_DATA_DIR=./data
|
||||||
DATABASE_PATH=./data/app.db
|
DATABASE_PATH=./data/app.db
|
||||||
MISTRAL_API_KEY=
|
MISTRAL_API_KEY=
|
||||||
MISTRAL_OCR_MODEL=mistral-ocr-4-0
|
MISTRAL_OCR_MODEL=mistral-ocr-latest
|
||||||
MISTRAL_OCR_BATCH_THRESHOLD=3
|
MISTRAL_OCR_BATCH_THRESHOLD=3
|
||||||
MISTRAL_OCR_BATCH_POLL_SECONDS=10
|
MISTRAL_OCR_BATCH_POLL_SECONDS=10
|
||||||
MISTRAL_OCR_BATCH_TIMEOUT_SECONDS=1800
|
MISTRAL_OCR_BATCH_TIMEOUT_SECONDS=1800
|
||||||
|
|||||||
@@ -18,6 +18,7 @@
|
|||||||
- `KnowledgeManager` owns filesystem layout and SQLite metadata; processors should not write final Markdown directly.
|
- `KnowledgeManager` owns filesystem layout and SQLite metadata; processors should not write final Markdown directly.
|
||||||
- `IngestionService` selects processors by `source_type` and wraps output with Markdown frontmatter for LLM ingestion.
|
- `IngestionService` selects processors by `source_type` and wraps output with Markdown frontmatter for LLM ingestion.
|
||||||
- Text and Markdown are read directly; `.docx` uses `python-docx`; PDFs use PyMuPDF text extraction unless `use_ocr=true` is passed. PDF uploads can also pass `page_ranges` like `1-5,8,10-12`; ranges apply to both standard extraction and Mistral OCR.
|
- Text and Markdown are read directly; `.docx` uses `python-docx`; PDFs use PyMuPDF text extraction unless `use_ocr=true` is passed. PDF uploads can also pass `page_ranges` like `1-5,8,10-12`; ranges apply to both standard extraction and Mistral OCR.
|
||||||
|
- Images (png, jpg, jpeg, webp, gif, bmp, tiff) are always processed with Mistral OCR (`ImageProcessor`), no `use_ocr` flag needed. Sources stored with `source_type="unknown"` (uploaded before a format was supported) are re-detected by extension on reprocess.
|
||||||
- OCR uses Mistral `https://api.mistral.ai/v1/ocr` with `MISTRAL_OCR_MODEL`, defaulting to `mistral-ocr-latest`.
|
- OCR uses Mistral `https://api.mistral.ai/v1/ocr` with `MISTRAL_OCR_MODEL`, defaulting to `mistral-ocr-latest`.
|
||||||
- Audio uses Deepgram `https://api.deepgram.com/v1/listen?model=nova-3&smart_format=true&language=es`.
|
- Audio uses Deepgram `https://api.deepgram.com/v1/listen?model=nova-3&smart_format=true&language=es`.
|
||||||
- Video processing requires `ffmpeg`; audio is extracted to `data/tmp/`, transcribed with Deepgram, then deleted.
|
- Video processing requires `ffmpeg`; audio is extracted to `data/tmp/`, transcribed with Deepgram, then deleted.
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ class Settings(BaseSettings):
|
|||||||
app_data_dir: Path = Path("./data")
|
app_data_dir: Path = Path("./data")
|
||||||
database_path: Path = Path("./data/app.db")
|
database_path: Path = Path("./data/app.db")
|
||||||
mistral_api_key: str = ""
|
mistral_api_key: str = ""
|
||||||
mistral_ocr_model: str = "mistral-ocr-4-0"
|
mistral_ocr_model: str = "mistral-ocr-latest"
|
||||||
mistral_ocr_batch_threshold: int = 3
|
mistral_ocr_batch_threshold: int = 3
|
||||||
mistral_ocr_batch_poll_seconds: int = 10
|
mistral_ocr_batch_poll_seconds: int = 10
|
||||||
mistral_ocr_batch_timeout_seconds: int = 1800
|
mistral_ocr_batch_timeout_seconds: int = 1800
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ def health() -> dict[str, str]:
|
|||||||
def settings_status() -> dict:
|
def settings_status() -> dict:
|
||||||
return {
|
return {
|
||||||
"mistral_configured": bool(settings.mistral_api_key),
|
"mistral_configured": bool(settings.mistral_api_key),
|
||||||
|
"mistral_ocr_model": settings.mistral_ocr_model,
|
||||||
"deepgram_configured": bool(settings.deepgram_api_key),
|
"deepgram_configured": bool(settings.deepgram_api_key),
|
||||||
"data_dir": str(settings.app_data_dir),
|
"data_dir": str(settings.app_data_dir),
|
||||||
"database_path": str(settings.database_path),
|
"database_path": str(settings.database_path),
|
||||||
|
|||||||
@@ -378,6 +378,18 @@ class KnowledgeManager:
|
|||||||
raise ValueError("Fuente no encontrada")
|
raise ValueError("Fuente no encontrada")
|
||||||
return dict(row)
|
return dict(row)
|
||||||
|
|
||||||
|
def refresh_source_type(self, source_id: int) -> str:
|
||||||
|
"""Recalcula y persiste el tipo de una fuente guardada como ``unknown``.
|
||||||
|
|
||||||
|
Permite reprocesar archivos subidos antes de soportar su formato
|
||||||
|
(por ejemplo, imagenes que quedaron como ``unknown`` y fallaban).
|
||||||
|
"""
|
||||||
|
source = self.get_source(source_id)
|
||||||
|
source_type = self.detect_source_type(Path(source["stored_path"]))
|
||||||
|
with self._connect() as conn:
|
||||||
|
conn.execute("update sources set source_type = ? where id = ?", (source_type, source_id))
|
||||||
|
return source_type
|
||||||
|
|
||||||
def list_sources(self, subject_id: int, week_number: int) -> list[dict[str, Any]]:
|
def list_sources(self, subject_id: int, week_number: int) -> list[dict[str, Any]]:
|
||||||
week = self.get_week(subject_id, week_number)
|
week = self.get_week(subject_id, week_number)
|
||||||
with self._connect() as conn:
|
with self._connect() as conn:
|
||||||
@@ -552,6 +564,8 @@ class KnowledgeManager:
|
|||||||
return "docx"
|
return "docx"
|
||||||
if ext == "pdf":
|
if ext == "pdf":
|
||||||
return "pdf"
|
return "pdf"
|
||||||
|
if ext in {"png", "jpg", "jpeg", "webp", "gif", "bmp", "tif", "tiff"}:
|
||||||
|
return "image"
|
||||||
if ext in {"mp3", "wav", "m4a", "ogg", "flac", "webm"}:
|
if ext in {"mp3", "wav", "m4a", "ogg", "flac", "webm"}:
|
||||||
return "audio"
|
return "audio"
|
||||||
if ext in {"mp4", "mov", "mkv", "avi"}:
|
if ext in {"mp4", "mov", "mkv", "avi"}:
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ from backend.app.managers.knowledge_manager import KnowledgeManager
|
|||||||
from backend.app.services.markdown_builder import build_markdown
|
from backend.app.services.markdown_builder import build_markdown
|
||||||
from backend.app.services.processors.audio_processor import AudioProcessor
|
from backend.app.services.processors.audio_processor import AudioProcessor
|
||||||
from backend.app.services.processors.docx_processor import DocxProcessor
|
from backend.app.services.processors.docx_processor import DocxProcessor
|
||||||
|
from backend.app.services.processors.image_processor import ImageProcessor
|
||||||
from backend.app.services.processors.pdf_processor import PdfProcessor
|
from backend.app.services.processors.pdf_processor import PdfProcessor
|
||||||
from backend.app.services.processors.text_processor import TextProcessor
|
from backend.app.services.processors.text_processor import TextProcessor
|
||||||
from backend.app.services.processors.video_processor import VideoProcessor
|
from backend.app.services.processors.video_processor import VideoProcessor
|
||||||
@@ -19,7 +20,8 @@ class IngestionService:
|
|||||||
source = self.manager.get_source(source_id)
|
source = self.manager.get_source(source_id)
|
||||||
context = self.manager.get_week_context(source["week_id"])
|
context = self.manager.get_week_context(source["week_id"])
|
||||||
path = Path(source["stored_path"])
|
path = Path(source["stored_path"])
|
||||||
processor = self._processor_for(source["source_type"], use_ocr, page_ranges)
|
source_type = self._resolve_source_type(source, path)
|
||||||
|
processor = self._processor_for(source_type, use_ocr, page_ranges)
|
||||||
processed = processor.process(path)
|
processed = processor.process(path)
|
||||||
markdown = build_markdown(
|
markdown = build_markdown(
|
||||||
title=processed.title,
|
title=processed.title,
|
||||||
@@ -29,14 +31,21 @@ class IngestionService:
|
|||||||
"subject_slug": context["subject_slug"],
|
"subject_slug": context["subject_slug"],
|
||||||
"week": context["week_number"],
|
"week": context["week_number"],
|
||||||
"source_file": source["original_name"],
|
"source_file": source["original_name"],
|
||||||
"source_type": source["source_type"],
|
"source_type": source_type,
|
||||||
"processor": processed.processor,
|
"processor": processed.processor,
|
||||||
"page_ranges": page_ranges if source["source_type"] == "pdf" and page_ranges else None,
|
"page_ranges": page_ranges if source_type == "pdf" and page_ranges else None,
|
||||||
"language": "es",
|
"language": "es",
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
return self.manager.create_document(source_id, processed.title, markdown, processor=processed.processor, page_ranges=page_ranges)
|
return self.manager.create_document(source_id, processed.title, markdown, processor=processed.processor, page_ranges=page_ranges)
|
||||||
|
|
||||||
|
def _resolve_source_type(self, source: dict, path: Path) -> str:
|
||||||
|
if source["source_type"] != "unknown":
|
||||||
|
return source["source_type"]
|
||||||
|
# Archivos subidos antes de soportar su formato quedaron como
|
||||||
|
# ``unknown``; al reprocesarlos se recalcula el tipo por extension.
|
||||||
|
return self.manager.refresh_source_type(source["id"])
|
||||||
|
|
||||||
def _processor_for(self, source_type: str, use_ocr: bool, page_ranges: str | None):
|
def _processor_for(self, source_type: str, use_ocr: bool, page_ranges: str | None):
|
||||||
if source_type == "text":
|
if source_type == "text":
|
||||||
return TextProcessor()
|
return TextProcessor()
|
||||||
@@ -44,6 +53,8 @@ class IngestionService:
|
|||||||
return DocxProcessor()
|
return DocxProcessor()
|
||||||
if source_type == "pdf":
|
if source_type == "pdf":
|
||||||
return PdfProcessor(use_ocr=use_ocr, page_ranges=page_ranges)
|
return PdfProcessor(use_ocr=use_ocr, page_ranges=page_ranges)
|
||||||
|
if source_type == "image":
|
||||||
|
return ImageProcessor()
|
||||||
if source_type == "audio":
|
if source_type == "audio":
|
||||||
return AudioProcessor()
|
return AudioProcessor()
|
||||||
if source_type == "video":
|
if source_type == "video":
|
||||||
|
|||||||
@@ -0,0 +1,19 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from backend.app.services.markdown_builder import title_from_path
|
||||||
|
from backend.app.services.processors.base import ProcessedContent
|
||||||
|
from backend.app.services.providers.mistral_client import MistralClient
|
||||||
|
|
||||||
|
|
||||||
|
class ImageProcessor:
|
||||||
|
"""Procesa imagenes sueltas (png, jpg, webp, gif, bmp, tiff) con Mistral OCR.
|
||||||
|
|
||||||
|
No hay extraccion de texto alternativa para imagenes: siempre pasan por
|
||||||
|
OCR, independiente de la opcion ``use_ocr``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def process(self, path: Path) -> ProcessedContent:
|
||||||
|
markdown = MistralClient().ocr_image(path)
|
||||||
|
return ProcessedContent(title=title_from_path(path), body=markdown, processor="mistral-ocr-image")
|
||||||
@@ -36,6 +36,17 @@ class MistralClient:
|
|||||||
files_url = f"{base_url}/files"
|
files_url = f"{base_url}/files"
|
||||||
batch_url = f"{base_url}/batch/jobs"
|
batch_url = f"{base_url}/batch/jobs"
|
||||||
|
|
||||||
|
_IMAGE_MIME_BY_EXT = {
|
||||||
|
".png": "image/png",
|
||||||
|
".jpg": "image/jpeg",
|
||||||
|
".jpeg": "image/jpeg",
|
||||||
|
".webp": "image/webp",
|
||||||
|
".gif": "image/gif",
|
||||||
|
".bmp": "image/bmp",
|
||||||
|
".tif": "image/tiff",
|
||||||
|
".tiff": "image/tiff",
|
||||||
|
}
|
||||||
|
|
||||||
def __init__(self) -> None:
|
def __init__(self) -> None:
|
||||||
self._api_key = settings.mistral_api_key
|
self._api_key = settings.mistral_api_key
|
||||||
self._model = settings.mistral_ocr_model
|
self._model = settings.mistral_ocr_model
|
||||||
@@ -285,4 +296,5 @@ class MistralClient:
|
|||||||
|
|
||||||
def _image_data_url(self, image_path: Path) -> str:
|
def _image_data_url(self, image_path: Path) -> str:
|
||||||
encoded = base64.b64encode(image_path.read_bytes()).decode("ascii")
|
encoded = base64.b64encode(image_path.read_bytes()).decode("ascii")
|
||||||
return f"data:image/jpeg;base64,{encoded}"
|
mime = self._IMAGE_MIME_BY_EXT.get(image_path.suffix.lower(), "image/jpeg")
|
||||||
|
return f"data:{mime};base64,{encoded}"
|
||||||
|
|||||||
+1
-1
@@ -11,7 +11,7 @@ services:
|
|||||||
APP_DATA_DIR: /app/data
|
APP_DATA_DIR: /app/data
|
||||||
DATABASE_PATH: /app/data/app.db
|
DATABASE_PATH: /app/data/app.db
|
||||||
MISTRAL_API_KEY: ${MISTRAL_API_KEY:-}
|
MISTRAL_API_KEY: ${MISTRAL_API_KEY:-}
|
||||||
MISTRAL_OCR_MODEL: ${MISTRAL_OCR_MODEL:-mistral-ocr-4-0}
|
MISTRAL_OCR_MODEL: ${MISTRAL_OCR_MODEL:-mistral-ocr-latest}
|
||||||
MISTRAL_OCR_BATCH_THRESHOLD: ${MISTRAL_OCR_BATCH_THRESHOLD:-3}
|
MISTRAL_OCR_BATCH_THRESHOLD: ${MISTRAL_OCR_BATCH_THRESHOLD:-3}
|
||||||
MISTRAL_OCR_BATCH_POLL_SECONDS: ${MISTRAL_OCR_BATCH_POLL_SECONDS:-10}
|
MISTRAL_OCR_BATCH_POLL_SECONDS: ${MISTRAL_OCR_BATCH_POLL_SECONDS:-10}
|
||||||
MISTRAL_OCR_BATCH_TIMEOUT_SECONDS: ${MISTRAL_OCR_BATCH_TIMEOUT_SECONDS:-1800}
|
MISTRAL_OCR_BATCH_TIMEOUT_SECONDS: ${MISTRAL_OCR_BATCH_TIMEOUT_SECONDS:-1800}
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ export default function SettingsPage() {
|
|||||||
<Row label="API" value={apiOk ? "Activa" : "No disponible"} ok={apiOk} />
|
<Row label="API" value={apiOk ? "Activa" : "No disponible"} ok={apiOk} />
|
||||||
<Row label="Deepgram" value={settings?.deepgram_configured ? "Configurado" : "Falta DEEPGRAM_API_KEY"} ok={!!settings?.deepgram_configured} />
|
<Row label="Deepgram" value={settings?.deepgram_configured ? "Configurado" : "Falta DEEPGRAM_API_KEY"} ok={!!settings?.deepgram_configured} />
|
||||||
<Row label="Mistral" value={settings?.mistral_configured ? "Configurado" : "Falta MISTRAL_API_KEY"} ok={!!settings?.mistral_configured} />
|
<Row label="Mistral" value={settings?.mistral_configured ? "Configurado" : "Falta MISTRAL_API_KEY"} ok={!!settings?.mistral_configured} />
|
||||||
|
{settings?.mistral_ocr_model && <Row label="Modelo OCR" value={settings.mistral_ocr_model} ok={true} />}
|
||||||
</div>
|
</div>
|
||||||
{error && <p className="mt-4 text-sm text-red-300">{error}</p>}
|
{error && <p className="mt-4 text-sm text-red-300">{error}</p>}
|
||||||
</Card>
|
</Card>
|
||||||
|
|||||||
@@ -51,7 +51,7 @@ export default function WeekPage({ params }: { params: { subjectId: string; week
|
|||||||
setUploading(true); setError("");
|
setUploading(true); setError("");
|
||||||
try {
|
try {
|
||||||
const ranges = pageMode === "ranges" ? pageRanges : undefined;
|
const ranges = pageMode === "ranges" ? pageRanges : undefined;
|
||||||
for (const file of files) await api.upload(subjectId, weekNumber, file, pdfMethod === "mistral_ocr", ranges);
|
for (const file of files) await api.upload(subjectId, weekNumber, file, isImageFile(file) || pdfMethod === "mistral_ocr", ranges);
|
||||||
await load();
|
await load();
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
setError(err instanceof Error ? err.message : "No se pudo subir archivo");
|
setError(err instanceof Error ? err.message : "No se pudo subir archivo");
|
||||||
@@ -65,6 +65,10 @@ export default function WeekPage({ params }: { params: { subjectId: string; week
|
|||||||
return file.type === "application/pdf" || file.name.toLowerCase().endsWith(".pdf");
|
return file.type === "application/pdf" || file.name.toLowerCase().endsWith(".pdf");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function isImageFile(file: File) {
|
||||||
|
return file.type.startsWith("image/") || /\.(png|jpe?g|webp|gif|bmp|tiff?)$/i.test(file.name);
|
||||||
|
}
|
||||||
|
|
||||||
async function prepareFiles(files: File[]) {
|
async function prepareFiles(files: File[]) {
|
||||||
if (!files.length) return;
|
if (!files.length) return;
|
||||||
setError("");
|
setError("");
|
||||||
@@ -136,8 +140,10 @@ export default function WeekPage({ params }: { params: { subjectId: string; week
|
|||||||
async function reprocessSource(source: Source) {
|
async function reprocessSource(source: Source) {
|
||||||
setReprocessingId(source.id); setError("");
|
setReprocessingId(source.id); setError("");
|
||||||
try {
|
try {
|
||||||
const ranges = source.source_type === "pdf" && pageMode === "ranges" ? pageRanges : undefined;
|
const isPdf = source.source_type === "pdf";
|
||||||
await api.reprocessSource(source.id, source.source_type === "pdf" && pdfMethod === "mistral_ocr", ranges);
|
const ranges = isPdf && pageMode === "ranges" ? pageRanges : undefined;
|
||||||
|
const useOcr = isPdf ? pdfMethod === "mistral_ocr" : source.source_type === "image";
|
||||||
|
await api.reprocessSource(source.id, useOcr, ranges);
|
||||||
await load();
|
await load();
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
setError(err instanceof Error ? err.message : "No se pudo reprocesar la fuente");
|
setError(err instanceof Error ? err.message : "No se pudo reprocesar la fuente");
|
||||||
@@ -187,11 +193,11 @@ export default function WeekPage({ params }: { params: { subjectId: string; week
|
|||||||
<div className="flex flex-wrap items-center justify-between gap-4">
|
<div className="flex flex-wrap items-center justify-between gap-4">
|
||||||
<div>
|
<div>
|
||||||
<h2 className="text-xl font-semibold">Subir archivos</h2>
|
<h2 className="text-xl font-semibold">Subir archivos</h2>
|
||||||
<p className="mt-1 text-sm text-slate-400">PDF, DOCX, TXT, Markdown, audio y video.</p>
|
<p className="mt-1 text-sm text-slate-400">PDF, DOCX, TXT, Markdown, imagenes, audio y video.</p>
|
||||||
</div>
|
</div>
|
||||||
<label className="cursor-pointer rounded-xl bg-blue-600 px-4 py-2 font-medium hover:bg-blue-500">
|
<label className="cursor-pointer rounded-xl bg-blue-600 px-4 py-2 font-medium hover:bg-blue-500">
|
||||||
{uploading ? "Subiendo..." : "Seleccionar archivos"}
|
{uploading ? "Subiendo..." : "Seleccionar archivos"}
|
||||||
<input className="hidden" type="file" multiple onChange={uploadFiles} disabled={uploading} />
|
<input className="hidden" type="file" multiple onChange={uploadFiles} disabled={uploading} accept=".pdf,.docx,.txt,.md,.png,.jpg,.jpeg,.webp,.gif,.bmp,.tif,.tiff,.mp3,.wav,.m4a,.ogg,.flac,.webm,.mp4,.mov,.mkv,.avi" />
|
||||||
</label>
|
</label>
|
||||||
</div>
|
</div>
|
||||||
<div
|
<div
|
||||||
@@ -204,7 +210,7 @@ export default function WeekPage({ params }: { params: { subjectId: string; week
|
|||||||
<div className="mx-auto grid h-14 w-14 place-items-center rounded-2xl bg-blue-500/15 text-2xl">↑</div>
|
<div className="mx-auto grid h-14 w-14 place-items-center rounded-2xl bg-blue-500/15 text-2xl">↑</div>
|
||||||
<p className="mt-4 font-medium">Arrastra archivos aqui</p>
|
<p className="mt-4 font-medium">Arrastra archivos aqui</p>
|
||||||
<p className="mt-2 text-sm text-slate-400">Puedes soltar varios archivos a la vez. Se procesaran en esta semana.</p>
|
<p className="mt-2 text-sm text-slate-400">Puedes soltar varios archivos a la vez. Se procesaran en esta semana.</p>
|
||||||
<p className="mt-3 text-xs text-slate-500">Soporta PDF, DOCX, TXT, Markdown, audio y video.</p>
|
<p className="mt-3 text-xs text-slate-500">Soporta PDF, DOCX, TXT, Markdown, imagenes (OCR automatico), audio y video.</p>
|
||||||
{uploading && <p className="mt-4 text-sm text-blue-300">Subiendo archivos...</p>}
|
{uploading && <p className="mt-4 text-sm text-blue-300">Subiendo archivos...</p>}
|
||||||
</div>
|
</div>
|
||||||
{pendingFiles.length > 0 && (
|
{pendingFiles.length > 0 && (
|
||||||
|
|||||||
+1
-1
@@ -35,7 +35,7 @@ export type DocumentItem = {
|
|||||||
source_page_ranges?: string | null;
|
source_page_ranges?: string | null;
|
||||||
created_at: string;
|
created_at: string;
|
||||||
};
|
};
|
||||||
export type SettingsStatus = { mistral_configured: boolean; deepgram_configured: boolean; data_dir: string; database_path: string };
|
export type SettingsStatus = { mistral_configured: boolean; mistral_ocr_model?: string; deepgram_configured: boolean; data_dir: string; database_path: string };
|
||||||
|
|
||||||
async function parseResponse<T>(response: Response): Promise<T> {
|
async function parseResponse<T>(response: Response): Promise<T> {
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
|
|||||||
Reference in New Issue
Block a user