Added additional error handling to job related functionalities. (Formatted file in db_utils, which might cause a merge conflict, just override my file later)

This commit is contained in:
Kikimanox
2022-10-18 22:48:51 +02:00
parent 8ffa85cd76
commit acb028073a
5 changed files with 67 additions and 45 deletions
+51 -31
View File
@@ -12,54 +12,74 @@ import magic
tika_server = "http://tika2:9999/tika"
# endpoint below to be used only for development purposes (don't need to run docker)
# tika_server = "http://rsdo.lhrs.feri.um.si:9998/tika"
def extract_text_prepResp(file, content_type=""):
content_type = file.content_type
if content_type is None:
content_type = magic.from_file(file.stream.name, mime=True)
content = ""
if tika_responding():
response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"})
return response.text, 200
if "openxmlformats-officedocument.wordprocessingml.document" in content_type:
content = '\n'.join([p.text for p in docx.Document(file).paragraphs])
elif "application/pdf" in content_type:
reader = PdfReader(file)
content = '\n'.join([p.extract_text() for p in reader.pages])
content = content
elif "text/xml" in content_type:
root = ET.parse(file).getroot()
plainText = root.findall('PlainText')
if len(plainText) == 0:
return "Didn't find anything in PlainText", 400
content = '\n'.join([pt.text for pt in plainText])
# elif "text/plain" in file.content_type:
else:
content = file.read().decode('utf-8')
try:
response = requests.put(tika_server, data=file, headers={"Accept": "text/plain; charset=UTF-8"})
content = response.text
except:
content = "ERROR - something went wrong when reading file with tika"
if content == "":
if "openxmlformats-officedocument.wordprocessingml.document" in content_type:
content = '\n'.join([p.text for p in docx.Document(file).paragraphs])
elif "application/pdf" in content_type:
reader = PdfReader(file)
content = '\n'.join([p.extract_text() for p in reader.pages])
content = content
elif "text/xml" in content_type:
root = ET.parse(file).getroot()
plainText = root.findall('PlainText')
if len(plainText) == 0:
return "Didn't find anything in PlainText", 400
content = '\n'.join([pt.text for pt in plainText])
# elif "text/plain" in file.content_type:
else:
try:
content = file.read().decode('utf-8')
except:
content = "ERROR - something went wrong when reading file with not-tika method!"
return content, 200
def ocr_text_prepResp(file):
content = ""
if tika_responding():
response = requests.put(tika_server, data=file,
headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng","Accept": "text/plain; charset=UTF-8"})
return response.text, 200
try:
response = requests.put(tika_server, data=file,
headers={"X-Tika-PDFOcrStrategy": "ocr_only", "X-Tika-OCRLanguage": "slv+eng",
"Accept": "text/plain; charset=UTF-8"})
content = response.text
except:
content = "ERROR - something went wrong when reading file with tika (OCR)"
win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe"
if os.path.exists(win_p):
pytesseract.pytesseract.tesseract_cmd = win_p
if content == "":
try:
win_p = "C:/Program Files/Tesseract-OCR/tesseract.exe"
if os.path.exists(win_p):
pytesseract.pytesseract.tesseract_cmd = win_p
# convert string data to numpy array
file_bytes = np.fromstring(file.read(), np.uint8)
# convert numpy array to image
img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR)
# convert string data to numpy array
file_bytes = np.fromstring(file.read(), np.uint8)
# convert numpy array to image
img = cv2.imdecode(file_bytes, cv2.IMREAD_COLOR)
conf = '-l eng+slv'
return pytesseract.image_to_string(img, config=conf), 200
conf = '-l eng+slv'
content = pytesseract.image_to_string(img, config=conf)
except:
content = "ERROR - something went wrong when reading file with not-tika method! (OCR)"
return content, 200
def tika_responding():
@@ -67,4 +87,4 @@ def tika_responding():
ret = requests.get(tika_server)
return ret.status_code == 200
except:
return False
return False