cv-analysis-service/vidocp/utils/preprocessing.py

23 lines
705 B
Python

from numpy import array
import pdf2image
import cv2
def open_pdf(pdf, first_page=0, last_page=None):
first_page += 1
last_page = None if last_page is None else last_page + 1
if type(pdf) == str:
pages = pdf2image.convert_from_path(pdf, first_page=first_page, last_page=last_page)
elif type(pdf) == bytes:
pages = pdf2image.convert_from_bytes(pdf, first_page=first_page, last_page=last_page)
elif type(pdf) == list:
return pdf
pages = [array(p) for p in pages]
return pages
def preprocess_pdf_image(page):
if len(page.shape) > 2:
page = cv2.cvtColor(page, cv2.COLOR_BGR2GRAY)
page = cv2.fastNlMeansDenoising(page, h=3)
return page