Added additional error handling to job related functionalities. (Formatted file in db_utils, which might cause a merge conflict, just override my file later)

This commit is contained in:
Kikimanox
2022-10-18 22:48:51 +02:00
parent 8ffa85cd76
commit acb028073a
5 changed files with 67 additions and 45 deletions
@@ -12,11 +12,11 @@ from swagger_server.util import get_random_filename, create_random_file_in_tmp_f
import requests import requests
from werkzeug.utils import secure_filename from werkzeug.utils import secure_filename
# ATEapi_endpoint = "http://localhost:5000/predict"
ATEapi_endpoint = "http://ate-api:5000/predict" ATEapi_endpoint = "http://ate-api:5000/predict"
# endpoint below to be used only for development purposes (don't need to run docker)
# ATEapi_endpoint = "http://localhost:5000/predict"
def do_izlusci(conllus, prepovedane_besede): def do_izlusci(conllus, prepovedane_besede):
tmp_file_path = "" tmp_file_path = ""
@@ -1,6 +1,7 @@
import datetime import datetime
import json import json
import os.path import os.path
import traceback
import peewee import peewee
import asyncio import asyncio
@@ -92,6 +93,7 @@ def try_do_jobs_ateapi():
[ex.submit(execute_ateapi_job, job) for job in unfinished_jobs] [ex.submit(execute_ateapi_job, job) for job in unfinished_jobs]
except Exception as e: except Exception as e:
print(f"Exception in try_do_jobs_ateapi") print(f"Exception in try_do_jobs_ateapi")
traceback.print_exc()
finally: finally:
time.sleep(3) time.sleep(3)
@@ -120,6 +122,7 @@ def try_do_jobs_classla():
except Exception as e: except Exception as e:
print(f"Exception in try_do_jobs_classla") print(f"Exception in try_do_jobs_classla")
traceback.print_exc()
finally: finally:
time.sleep(3) time.sleep(3)
@@ -136,6 +139,7 @@ def try_do_jobs_doc2text():
[ex.submit(execute_doc2text_job, job) for job in unfinished_jobs] [ex.submit(execute_doc2text_job, job) for job in unfinished_jobs]
except Exception as e: except Exception as e:
print(f"Exception in try_do_jobs_doc2text") print(f"Exception in try_do_jobs_doc2text")
traceback.print_exc()
finally: finally:
time.sleep(3) time.sleep(3)
@@ -163,9 +167,9 @@ def execute_doc2text_job(job: Job):
jtype = job.job_type jtype = job.job_type
text = "" text = ""
if jtype in [1, 12]: if jtype in [1, 12]:
text = txt_utils.extract_text_prepResp(file) text, _ = txt_utils.extract_text_prepResp(file)
elif jtype in [3, 32]: elif jtype in [3, 32]:
text = txt_utils.ocr_text_prepResp(file) text, _ = txt_utils.ocr_text_prepResp(file)
if jtype in [1, 3]: if jtype in [1, 3]:
job.job_output = text job.job_output = text
+1 -1
View File
@@ -1,4 +1,4 @@
from swagger_server import db_utils from swagger_server.utils import db_utils
from swagger_server import util from swagger_server import util
from flask import send_file from flask import send_file
+6 -8
View File
@@ -2,8 +2,6 @@ import mariadb
import os import os
import sys import sys
database_info = { database_info = {
'database': os.environ.get("MDB_DATABASE", default="true"), 'database': os.environ.get("MDB_DATABASE", default="true"),
'host': os.environ.get("MDB_HOST", default="true"), 'host': os.environ.get("MDB_HOST", default="true"),
@@ -14,21 +12,21 @@ database_info = {
cur = None cur = None
# Connect to MariaDB Platform # Connect to MariaDB Platform
def get_files_by_udc(udc): def get_files_by_udc(udc):
ret = [] ret = []
try: try:
conn = mariadb.connect(**database_info) conn = mariadb.connect(**database_info)
cur = conn.cursor() cur = conn.cursor()
#cur.execute(f'SELECT * from os2022_ngrams WHERE file_id = {file_id}') # cur.execute(f'SELECT * from os2022_ngrams WHERE file_id = {file_id}')
#cur.execute(f'SELECT COUNT(*) FROM os2022_ngrams') # cur.execute(f'SELECT COUNT(*) FROM os2022_ngrams')
#ret = list(cur) # ret = list(cur)
except mariadb.Error as e: except mariadb.Error as e:
print(f"Error connecting to MariaDB Platform: {e}") print(f"Error connecting to MariaDB Platform: {e}")
return ret return ret
+51 -31
View File
@@ -12,54 +12,74 @@ import magic
tika_server = "http://tika2:9999/tika" tika_server = "http://tika2:9999/tika"
# endpoint below to be used only for development purposes (don't need to run docker) # endpoint below to be used only for development purposes (don't need to run docker)
# tika_server = "http://rsdo.lhrs.feri.um.si:9998/tika" # tika_server = "http://rsdo.lhrs.feri.um.si:9998/tika"
def extract_text_prepResp(file, content_type=""): def extract_text_prepResp(file, content_type=""):
content_type = file.content_type content_type = file.content_type
if content_type is None: if content_type is None:
content_type = magic.from_file(file.stream.name, mime=True) content_type = magic.from_file(file.stream.name, mime=True)
content = ""
if tika_responding(): if tika_responding():
response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"}) try:
return response.text, 200 response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"})
if "openxmlformats-officedocument.wordprocessingml.document" in content_type: content = response.text
content = '\n'.join([p.text for p in docx.Document(file).paragraphs]) except:
elif "application/pdf" in content_type: content = "ERROR - something went wrong when reading file with tika"
reader = PdfReader(file)
content = '\n'.join([p.extract_text() for p in reader.pages]) if content == "":
content = content if "openxmlformats-officedocument.wordprocessingml.document" in content_type:
elif "text/xml" in content_type: content = '\n'.join([p.text for p in docx.Document(file).paragraphs])
root = ET.parse(file).getroot() elif "application/pdf" in content_type:
plainText = root.findall('PlainText') reader = PdfReader(file)
if len(plainText) == 0: content = '\n'.join([p.extract_text() for p in reader.pages])
return "Didn't find anything in PlainText", 400 content = content
content = '\n'.join([pt.text for pt in plainText]) elif "text/xml" in content_type:
# elif "text/plain" in file.content_type: root = ET.parse(file).getroot()
else: plainText = root.findall('PlainText')
content = file.read().decode('utf-8') if len(plainText) == 0:
return "Didn't find anything in PlainText", 400
content = '\n'.join([pt.text for pt in plainText])
# elif "text/plain" in file.content_type:
else:
try:
content = file.read().decode('utf-8')
except:
content = "ERROR - something went wrong when reading file with not-tika method!"
return content, 200 return content, 200
def ocr_text_prepResp(file): def ocr_text_prepResp(file):
content = ""
if tika_responding(): if tika_responding():
response = requests.put(tika_server, data=file, try:
headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng","Accept": "text/plain; charset=UTF-8"}) response = requests.put(tika_server, data=file,
return response.text, 200 headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng",
"Accept": "text/plain; charset=UTF-8"})
content = response.text
except:
content = "ERROR - something went wrong when reading file with tika (OCR)"
win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe" if content == "":
if os.path.exists(win_p): try:
pytesseract.pytesseract.tesseract_cmd = win_p win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe"
if os.path.exists(win_p):
pytesseract.pytesseract.tesseract_cmd = win_p
# convert string data to numpy array # convert string data to numpy array
file_bytes = np.fromstring(file.read(), np.uint8) file_bytes = np.fromstring(file.read(), np.uint8)
# convert numpy array to image # convert numpy array to image
img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR) img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR)
conf = '-l eng+slv' conf = '-l eng+slv'
return pytesseract.image_to_string(img, config=conf), 200 content = pytesseract.image_to_string(img, config=conf)
except:
content = "ERROR - something went wrong when reading file with not-tika method! (OCR)"
return content, 200
def tika_responding(): def tika_responding():
@@ -67,4 +87,4 @@ def tika_responding():
ret = requests.get(tika_server) ret = requests.get(tika_server)
return ret.status_code == 200 return ret.status_code == 200
except: except:
return False return False