Add new app to handle all dependencies
Signed-off-by: Roberto Rosario <roberto.rosario.gonzalez@gmail.com>
This commit is contained in:
@@ -19,6 +19,7 @@ from mayan.apps.documents.signals import post_version_upload
|
||||
from mayan.apps.navigation.classes import SourceColumn
|
||||
from mayan.celery import app
|
||||
|
||||
from .dependencies import * # NOQA
|
||||
from .handlers import (
|
||||
handler_index_document, handler_initialize_new_ocr_settings,
|
||||
handler_ocr_document_version,
|
||||
|
||||
@@ -26,47 +26,11 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
class Tesseract(OCRBackendBase):
|
||||
def __init__(self, *args, **kwargs):
|
||||
auto_initialize = kwargs.pop('auto_initialize', True)
|
||||
super(Tesseract, self).__init__(*args, **kwargs)
|
||||
self.languages = ()
|
||||
|
||||
backend_arguments = yaml.load(
|
||||
Loader=SafeLoader,
|
||||
stream=setting_ocr_backend_arguments.value or '{}',
|
||||
)
|
||||
|
||||
tesseract_binary_path = backend_arguments.get(
|
||||
'tesseract_path', DEFAULT_TESSERACT_BINARY_PATH
|
||||
)
|
||||
self.command_timeout = backend_arguments.get(
|
||||
'timeout', DEFAULT_TESSERACT_TIMEOUT
|
||||
)
|
||||
|
||||
try:
|
||||
self.command_tesseract = sh.Command(path=tesseract_binary_path)
|
||||
except sh.CommandNotError:
|
||||
self.command_tesseract = None
|
||||
raise OCRError(
|
||||
_('Tesseract not found.')
|
||||
)
|
||||
else:
|
||||
# Get version
|
||||
result = self.command_tesseract(v=True)
|
||||
logger.debug('Tesseract version: %s', result.stdout)
|
||||
|
||||
# Get languages
|
||||
result = self.command_tesseract(list_langs=True)
|
||||
# Sample output format
|
||||
# List of available languages (3):
|
||||
# deu
|
||||
# eng
|
||||
# osd
|
||||
# <- empty line
|
||||
|
||||
# Extaction: strip last line, split by newline, discard the first
|
||||
# line
|
||||
self.languages = force_text(result.stdout).strip().split('\n')[1:]
|
||||
|
||||
logger.debug('Available languages: %s', ', '.join(self.languages))
|
||||
if auto_initialize:
|
||||
self.initialize()
|
||||
|
||||
def execute(self, *args, **kwargs):
|
||||
"""
|
||||
@@ -117,3 +81,49 @@ class Tesseract(OCRBackendBase):
|
||||
return result
|
||||
finally:
|
||||
temporary_image_file.close()
|
||||
|
||||
def initialize(self):
|
||||
self.languages = ()
|
||||
|
||||
self.read_settings()
|
||||
|
||||
try:
|
||||
self.command_tesseract = sh.Command(path=self.tesseract_binary_path)
|
||||
except sh.CommandNotFound:
|
||||
self.command_tesseract = None
|
||||
raise OCRError(
|
||||
_('Tesseract OCR not found.')
|
||||
)
|
||||
else:
|
||||
# Get version
|
||||
result = self.command_tesseract(v=True)
|
||||
logger.debug('Tesseract version: %s', result.stdout)
|
||||
|
||||
# Get languages
|
||||
result = self.command_tesseract(list_langs=True)
|
||||
# Sample output format
|
||||
# List of available languages (3):
|
||||
# deu
|
||||
# eng
|
||||
# osd
|
||||
# <- empty line
|
||||
|
||||
# Extaction: strip last line, split by newline, discard the first
|
||||
# line
|
||||
self.languages = force_text(result.stdout).strip().split('\n')[1:]
|
||||
|
||||
logger.debug('Available languages: %s', ', '.join(self.languages))
|
||||
|
||||
def read_settings(self):
|
||||
backend_arguments = yaml.load(
|
||||
Loader=SafeLoader,
|
||||
stream=setting_ocr_backend_arguments.value or '{}',
|
||||
)
|
||||
|
||||
self.tesseract_binary_path = backend_arguments.get(
|
||||
'tesseract_path', DEFAULT_TESSERACT_BINARY_PATH
|
||||
)
|
||||
|
||||
self.command_timeout = backend_arguments.get(
|
||||
'timeout', DEFAULT_TESSERACT_TIMEOUT
|
||||
)
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
from __future__ import unicode_literals
|
||||
|
||||
from django.utils.translation import ugettext_lazy as _
|
||||
|
||||
from mayan.apps.dependencies.classes import BinaryDependency, PythonDependency
|
||||
|
||||
from .backends.tesseract import Tesseract
|
||||
|
||||
tesseract = Tesseract(auto_initialize=False)
|
||||
tesseract.read_settings()
|
||||
|
||||
BinaryDependency(
|
||||
copyright_text='''
|
||||
The code in this repository is licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
''', help_text=_('Free Open Source OCR Engine'), label='Tesseract',
|
||||
module=__name__, name='tesseract', path=tesseract.tesseract_binary_path
|
||||
)
|
||||
|
||||
PythonDependency(
|
||||
copyright_text='''
|
||||
PyOCR is released under the GPL v3+.
|
||||
Copyright belongs to the authors of each piece of code
|
||||
(see the file AUTHORS for the contributors list, and
|
||||
git blame to know which lines belong to which author).
|
||||
''', help_text=_(
|
||||
'PyOCR is a Python library simplifying the use of OCR tools like '
|
||||
'Tesseract or Cuneiform.'
|
||||
), module=__name__, name='pyocr', version_string='==0.6'
|
||||
)
|
||||
Reference in New Issue
Block a user