feat: image parser

2026-02-04 21:30:35 +00:00 · 2024-11-19 19:06:53 +00:00
parent e0a3b8004c
commit 312cb9ae70
6 changed files with 53 additions and 1 deletions
--- a/application/parser/file/docs_parser.py
+++ b/application/parser/file/docs_parser.py
@@ -7,7 +7,8 @@ from pathlib import Path
 from typing import Dict

 from application.parser.file.base_parser import BaseParser
-
+from application.core.settings import settings
+import requests

 class PDFParser(BaseParser):
    """PDF parser."""
@@ -18,6 +19,15 @@ class PDFParser(BaseParser):

    def parse_file(self, file: Path, errors: str = "ignore") -> str:
        """Parse file."""
+        if settings.PARSE_PDF_AS_IMAGE:
+            doc2md_service = "https://llm.arc53.com/doc2md"
+            # alternatively you can use local vision capable LLM
+            with open(file, "rb") as file_loaded:
+                files = {'file': file_loaded}
+                response = requests.post(doc2md_service, files=files)   
+                data = response.json()["markdown"] 
+            return data
+
        try:
            import PyPDF2
        except ImportError: