Add new app to handle all dependencies

Signed-off-by: Roberto Rosario <roberto.rosario.gonzalez@gmail.com>
This commit is contained in:
Roberto Rosario
2019-05-03 01:12:20 -04:00
parent 11e13cea1d
commit ea3b513ae3
208 changed files with 7847 additions and 45933 deletions
+1
View File
@@ -19,6 +19,7 @@ from mayan.apps.documents.signals import post_version_upload
from mayan.apps.navigation.classes import SourceColumn
from mayan.celery import app
from .dependencies import * # NOQA
from .handlers import (
handler_index_document, handler_initialize_new_ocr_settings,
handler_ocr_document_version,
+49 -39
View File
@@ -26,47 +26,11 @@ logger = logging.getLogger(__name__)
class Tesseract(OCRBackendBase):
def __init__(self, *args, **kwargs):
auto_initialize = kwargs.pop('auto_initialize', True)
super(Tesseract, self).__init__(*args, **kwargs)
self.languages = ()
backend_arguments = yaml.load(
Loader=SafeLoader,
stream=setting_ocr_backend_arguments.value or '{}',
)
tesseract_binary_path = backend_arguments.get(
'tesseract_path', DEFAULT_TESSERACT_BINARY_PATH
)
self.command_timeout = backend_arguments.get(
'timeout', DEFAULT_TESSERACT_TIMEOUT
)
try:
self.command_tesseract = sh.Command(path=tesseract_binary_path)
except sh.CommandNotError:
self.command_tesseract = None
raise OCRError(
_('Tesseract not found.')
)
else:
# Get version
result = self.command_tesseract(v=True)
logger.debug('Tesseract version: %s', result.stdout)
# Get languages
result = self.command_tesseract(list_langs=True)
# Sample output format
# List of available languages (3):
# deu
# eng
# osd
# <- empty line
# Extaction: strip last line, split by newline, discard the first
# line
self.languages = force_text(result.stdout).strip().split('\n')[1:]
logger.debug('Available languages: %s', ', '.join(self.languages))
if auto_initialize:
self.initialize()
def execute(self, *args, **kwargs):
"""
@@ -117,3 +81,49 @@ class Tesseract(OCRBackendBase):
return result
finally:
temporary_image_file.close()
def initialize(self):
self.languages = ()
self.read_settings()
try:
self.command_tesseract = sh.Command(path=self.tesseract_binary_path)
except sh.CommandNotFound:
self.command_tesseract = None
raise OCRError(
_('Tesseract OCR not found.')
)
else:
# Get version
result = self.command_tesseract(v=True)
logger.debug('Tesseract version: %s', result.stdout)
# Get languages
result = self.command_tesseract(list_langs=True)
# Sample output format
# List of available languages (3):
# deu
# eng
# osd
# <- empty line
# Extaction: strip last line, split by newline, discard the first
# line
self.languages = force_text(result.stdout).strip().split('\n')[1:]
logger.debug('Available languages: %s', ', '.join(self.languages))
def read_settings(self):
backend_arguments = yaml.load(
Loader=SafeLoader,
stream=setting_ocr_backend_arguments.value or '{}',
)
self.tesseract_binary_path = backend_arguments.get(
'tesseract_path', DEFAULT_TESSERACT_BINARY_PATH
)
self.command_timeout = backend_arguments.get(
'timeout', DEFAULT_TESSERACT_TIMEOUT
)
+39
View File
@@ -0,0 +1,39 @@
from __future__ import unicode_literals
from django.utils.translation import ugettext_lazy as _
from mayan.apps.dependencies.classes import BinaryDependency, PythonDependency
from .backends.tesseract import Tesseract
tesseract = Tesseract(auto_initialize=False)
tesseract.read_settings()
BinaryDependency(
copyright_text='''
The code in this repository is licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
''', help_text=_('Free Open Source OCR Engine'), label='Tesseract',
module=__name__, name='tesseract', path=tesseract.tesseract_binary_path
)
PythonDependency(
copyright_text='''
PyOCR is released under the GPL v3+.
Copyright belongs to the authors of each piece of code
(see the file AUTHORS for the contributors list, and
git blame to know which lines belong to which author).
''', help_text=_(
'PyOCR is a Python library simplifying the use of OCR tools like '
'Tesseract or Cuneiform.'
), module=__name__, name='pyocr', version_string='==0.6'
)