diff --git a/swagger_server/controllers/extract_controller.py b/swagger_server/controllers/extract_controller.py index 5c25423..a526727 100644 --- a/swagger_server/controllers/extract_controller.py +++ b/swagger_server/controllers/extract_controller.py @@ -12,11 +12,11 @@ from swagger_server.util import get_random_filename, create_random_file_in_tmp_f import requests from werkzeug.utils import secure_filename -# ATEapi_endpoint = "http://localhost:5000/predict" - - ATEapi_endpoint = "http://ate-api:5000/predict" +# endpoint below to be used only for development purposes (don't need to run docker) +# ATEapi_endpoint = "http://localhost:5000/predict" + def do_izlusci(conllus, prepovedane_besede): tmp_file_path = "" diff --git a/swagger_server/controllers/jobs_controller.py b/swagger_server/controllers/jobs_controller.py index 2607fa4..278d3a1 100644 --- a/swagger_server/controllers/jobs_controller.py +++ b/swagger_server/controllers/jobs_controller.py @@ -1,6 +1,7 @@ import datetime import json import os.path +import traceback import peewee import asyncio @@ -92,6 +93,7 @@ def try_do_jobs_ateapi(): [ex.submit(execute_ateapi_job, job) for job in unfinished_jobs] except Exception as e: print(f"Exception in try_do_jobs_ateapi") + traceback.print_exc() finally: time.sleep(3) @@ -120,6 +122,7 @@ def try_do_jobs_classla(): except Exception as e: print(f"Exception in try_do_jobs_classla") + traceback.print_exc() finally: time.sleep(3) @@ -136,6 +139,7 @@ def try_do_jobs_doc2text(): [ex.submit(execute_doc2text_job, job) for job in unfinished_jobs] except Exception as e: print(f"Exception in try_do_jobs_doc2text") + traceback.print_exc() finally: time.sleep(3) @@ -163,9 +167,9 @@ def execute_doc2text_job(job: Job): jtype = job.job_type text = "" if jtype in [1, 12]: - text = txt_utils.extract_text_prepResp(file) + text, _ = txt_utils.extract_text_prepResp(file) elif jtype in [3, 32]: - text = txt_utils.ocr_text_prepResp(file) + text, _ = txt_utils.ocr_text_prepResp(file) if jtype in [1, 3]: job.job_output = text diff --git a/swagger_server/controllers/oss_controller.py b/swagger_server/controllers/oss_controller.py index c95e49e..ffabf77 100644 --- a/swagger_server/controllers/oss_controller.py +++ b/swagger_server/controllers/oss_controller.py @@ -1,4 +1,4 @@ -from swagger_server import db_utils +from swagger_server.utils import db_utils from swagger_server import util from flask import send_file diff --git a/swagger_server/utils/db_utils.py b/swagger_server/utils/db_utils.py index 44b3bfa..241a178 100644 --- a/swagger_server/utils/db_utils.py +++ b/swagger_server/utils/db_utils.py @@ -2,8 +2,6 @@ import mariadb import os import sys - - database_info = { 'database': os.environ.get("MDB_DATABASE", default="true"), 'host': os.environ.get("MDB_HOST", default="true"), @@ -14,21 +12,21 @@ database_info = { cur = None + # Connect to MariaDB Platform def get_files_by_udc(udc): - ret = [] + ret = [] - try: + try: conn = mariadb.connect(**database_info) cur = conn.cursor() - #cur.execute(f'SELECT * from os2022_ngrams WHERE file_id = {file_id}') - #cur.execute(f'SELECT COUNT(*) FROM os2022_ngrams') - #ret = list(cur) + # cur.execute(f'SELECT * from os2022_ngrams WHERE file_id = {file_id}') + # cur.execute(f'SELECT COUNT(*) FROM os2022_ngrams') + # ret = list(cur) except mariadb.Error as e: print(f"Error connecting to MariaDB Platform: {e}") - return ret diff --git a/swagger_server/utils/txt_utils.py b/swagger_server/utils/txt_utils.py index cca18df..3df799a 100644 --- a/swagger_server/utils/txt_utils.py +++ b/swagger_server/utils/txt_utils.py @@ -12,54 +12,74 @@ import magic tika_server = "http://tika2:9999/tika" - # endpoint below to be used only for development purposes (don't need to run docker) # tika_server = "http://rsdo.lhrs.feri.um.si:9998/tika" + def extract_text_prepResp(file, content_type=""): content_type = file.content_type if content_type is None: content_type = magic.from_file(file.stream.name, mime=True) + content = "" if tika_responding(): - response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"}) - return response.text, 200 - if "openxmlformats-officedocument.wordprocessingml.document" in content_type: - content = '\n'.join([p.text for p in docx.Document(file).paragraphs]) - elif "application/pdf" in content_type: - reader = PdfReader(file) - content = '\n'.join([p.extract_text() for p in reader.pages]) - content = content - elif "text/xml" in content_type: - root = ET.parse(file).getroot() - plainText = root.findall('PlainText') - if len(plainText) == 0: - return "Didn't find anything in PlainText", 400 - content = '\n'.join([pt.text for pt in plainText]) - # elif "text/plain" in file.content_type: - else: - content = file.read().decode('utf-8') + try: + response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"}) + content = response.text + except: + content = "ERROR - something went wrong when reading file with tika" + + if content == "": + if "openxmlformats-officedocument.wordprocessingml.document" in content_type: + content = '\n'.join([p.text for p in docx.Document(file).paragraphs]) + elif "application/pdf" in content_type: + reader = PdfReader(file) + content = '\n'.join([p.extract_text() for p in reader.pages]) + content = content + elif "text/xml" in content_type: + root = ET.parse(file).getroot() + plainText = root.findall('PlainText') + if len(plainText) == 0: + return "Didn't find anything in PlainText", 400 + content = '\n'.join([pt.text for pt in plainText]) + # elif "text/plain" in file.content_type: + else: + try: + content = file.read().decode('utf-8') + except: + content = "ERROR - something went wrong when reading file with not-tika method!" return content, 200 def ocr_text_prepResp(file): + content = "" if tika_responding(): - response = requests.put(tika_server, data=file, - headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng","Accept": "text/plain; charset=UTF-8"}) - return response.text, 200 + try: + response = requests.put(tika_server, data=file, + headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng", + "Accept": "text/plain; charset=UTF-8"}) + content = response.text + except: + content = "ERROR - something went wrong when reading file with tika (OCR)" - win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe" - if os.path.exists(win_p): - pytesseract.pytesseract.tesseract_cmd = win_p + if content == "": + try: + win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe" + if os.path.exists(win_p): + pytesseract.pytesseract.tesseract_cmd = win_p - # convert string data to numpy array - file_bytes = np.fromstring(file.read(), np.uint8) - # convert numpy array to image - img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR) + # convert string data to numpy array + file_bytes = np.fromstring(file.read(), np.uint8) + # convert numpy array to image + img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR) - conf = '-l eng+slv' - return pytesseract.image_to_string(img, config=conf), 200 + conf = '-l eng+slv' + content = pytesseract.image_to_string(img, config=conf) + except: + content = "ERROR - something went wrong when reading file with not-tika method! (OCR)" + + return content, 200 def tika_responding(): @@ -67,4 +87,4 @@ def tika_responding(): ret = requests.get(tika_server) return ret.status_code == 200 except: - return False \ No newline at end of file + return False