Initial implementation of OCR pluggable backends

This commit is contained in:
Roberto Rosario
2014-06-24 22:48:16 -04:00
parent 198402385e
commit e0347785b7
5 changed files with 85 additions and 46 deletions
+26
View File
@@ -0,0 +1,26 @@
from __future__ import absolute_import
from django.utils.importlib import import_module
from ..conf.settings import BACKEND
class BackendBase(object):
def execute(input_filename, language=None):
raise NotImplemented
def get_ocr_backend():
"""
Return the OCR backend using the path specified in the configuration
settings
"""
try:
module = import_module(BACKEND)
except ImportError:
sys.stderr.write(u'\nWarning: No OCR backend named: %s\n\n' % BACKEND)
raise
else:
return module
ocr_backend = get_ocr_backend()
+45
View File
@@ -0,0 +1,45 @@
from __future__ import absolute_import
import codecs
import os
import subprocess
import tempfile
import sys
from . import BackendBase
from ..conf.settings import TESSERACT_PATH
def Tesseract(BackendBase):
def execute(input_filename, language=None):
"""
Execute the command line binary of tesseract
"""
fd, filepath = tempfile.mkstemp()
os.close(fd)
ocr_output = os.extsep.join([filepath, u'txt'])
command = [unicode(TESSERACT_PATH), unicode(input_filename), unicode(filepath)]
if lang is not None:
command.extend([u'-l', language])
proc = subprocess.Popen(command, close_fds=True, stderr=subprocess.PIPE, stdout=subprocess.PIPE)
return_code = proc.wait()
if return_code != 0:
error_text = proc.stderr.read()
cleanup(filepath)
cleanup(ocr_output)
if lang:
# If tesseract gives an error with a language parameter
# re-run it with no parameter again
return run_tesseract(input_filename, language=None)
else:
raise TesseractError(error_text)
fd = codecs.open(ocr_output, 'r', 'utf-8')
text = fd.read().strip()
fd.close()
os.unlink(filepath)
return text