Added support for the command line program pdftotext from the poppler-utils packages to extract text from PDF documents without doing OCR

This commit is contained in:
Roberto Rosario
2011-04-15 23:59:52 -04:00
parent 73a52293e8
commit eaaaa5b645
5 changed files with 49 additions and 13 deletions
+1
View File
@@ -94,6 +94,7 @@ def check_settings(request):
{'name': 'OCR_TESSERACT_LANGUAGE', 'value': ocr_settings.TESSERACT_LANGUAGE},
{'name': 'OCR_NODE_CONCURRENT_EXECUTION', 'value': ocr_settings.NODE_CONCURRENT_EXECUTION},
{'name': 'OCR_REPLICATION_DELAY', 'value': ocr_settings.REPLICATION_DELAY},
{'name': 'OCR_PDFTOTEXT_PATH', 'value': ocr_settings.PDFTOTEXT_PATH, 'exists': True},
# Search
{'name': 'SEARCH_LIMIT', 'value': search_settings.LIMIT},