Files
Estacion-de-Documentos/backend/app/services/ingestion_service.py
T

63 lines
2.9 KiB
Python

from __future__ import annotations
from pathlib import Path
from backend.app.managers.knowledge_manager import KnowledgeManager
from backend.app.services.markdown_builder import build_markdown
from backend.app.services.processors.audio_processor import AudioProcessor
from backend.app.services.processors.docx_processor import DocxProcessor
from backend.app.services.processors.image_processor import ImageProcessor
from backend.app.services.processors.pdf_processor import PdfProcessor
from backend.app.services.processors.text_processor import TextProcessor
from backend.app.services.processors.video_processor import VideoProcessor
class IngestionService:
def __init__(self, manager: KnowledgeManager):
self.manager = manager
def process_source(self, source_id: int, use_ocr: bool = False, page_ranges: str | None = None) -> dict:
source = self.manager.get_source(source_id)
context = self.manager.get_week_context(source["week_id"])
path = Path(source["stored_path"])
source_type = self._resolve_source_type(source, path)
processor = self._processor_for(source_type, use_ocr, page_ranges)
processed = processor.process(path)
markdown = build_markdown(
title=processed.title,
body=processed.body,
metadata={
"subject": context["subject_name"],
"subject_slug": context["subject_slug"],
"week": context["week_number"],
"source_file": source["original_name"],
"source_type": source_type,
"processor": processed.processor,
"page_ranges": page_ranges if source_type == "pdf" and page_ranges else None,
"language": "es",
},
)
return self.manager.create_document(source_id, processed.title, markdown, processor=processed.processor, page_ranges=page_ranges)
def _resolve_source_type(self, source: dict, path: Path) -> str:
if source["source_type"] != "unknown":
return source["source_type"]
# Archivos subidos antes de soportar su formato quedaron como
# ``unknown``; al reprocesarlos se recalcula el tipo por extension.
return self.manager.refresh_source_type(source["id"])
def _processor_for(self, source_type: str, use_ocr: bool, page_ranges: str | None):
if source_type == "text":
return TextProcessor()
if source_type == "docx":
return DocxProcessor()
if source_type == "pdf":
return PdfProcessor(use_ocr=use_ocr, page_ranges=page_ranges)
if source_type == "image":
return ImageProcessor()
if source_type == "audio":
return AudioProcessor()
if source_type == "video":
return VideoProcessor(self.manager.tmp_dir)
raise RuntimeError(f"Tipo de fuente no soportado: {source_type}")