fts (3)
This commit is contained in:
@@ -33,6 +33,9 @@ ACME_HTTP_PORT=80
|
|||||||
# DOCUMENT_SCAN_SECONDS=30
|
# DOCUMENT_SCAN_SECONDS=30
|
||||||
# DOCUMENT_OCR_LANGUAGE=deu+eng
|
# DOCUMENT_OCR_LANGUAGE=deu+eng
|
||||||
# DOCUMENT_OCR_TIMEOUT_SECONDS=600
|
# DOCUMENT_OCR_TIMEOUT_SECONDS=600
|
||||||
|
# DOCUMENT_OCR_DPI=300
|
||||||
|
# DOCUMENT_OCR_MAX_DIMENSION=3500
|
||||||
|
# DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS=30
|
||||||
# DOCUMENT_MAX_FILE_MB=512
|
# DOCUMENT_MAX_FILE_MB=512
|
||||||
# DOCUMENT_MAX_PDF_PAGES=500
|
# DOCUMENT_MAX_PDF_PAGES=500
|
||||||
# TRASH_RETENTION_DAYS=7
|
# TRASH_RETENTION_DAYS=7
|
||||||
|
|||||||
@@ -79,7 +79,7 @@ Linux filesystem events update the filename catalog as files change, with period
|
|||||||
|
|
||||||
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
||||||
|
|
||||||
PDFs containing raster images and pages with little extracted text are queued for local OCR using OCRmyPDF and Tesseract, with German and English enabled by default. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
|
Only PDF pages that themselves contain raster images and little extracted text are queued for local OCR using Poppler and Tesseract, with German and English enabled by default. Search OCR renders those pages at 300 DPI, capped at 3500 pixels on the longest side, instead of inheriting high DPI from embedded logos or rebuilding an OCR PDF. Each render and recognition step has a 30-second timeout. Photos can still require a recognition attempt to establish whether they contain text; text-free pages finish without adding search content. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
|
||||||
|
|
||||||
Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities.
|
Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities.
|
||||||
|
|
||||||
@@ -92,6 +92,9 @@ Domain Admins monitor the worker at `/admin/documents`: current file and phase,
|
|||||||
| `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes |
|
| `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes |
|
||||||
| `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR |
|
| `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR |
|
||||||
| `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job |
|
| `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job |
|
||||||
|
| `DOCUMENT_OCR_DPI` | `300` | OCR render resolution (150–400 DPI), subject to the pixel cap |
|
||||||
|
| `DOCUMENT_OCR_MAX_DIMENSION` | `3500` | Maximum OCR image width/height (1500–5000 pixels); originals are unchanged |
|
||||||
|
| `DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS` | `30` | Maximum runtime per page render or recognition step (5–120 seconds) |
|
||||||
| `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search |
|
| `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search |
|
||||||
| `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction |
|
| `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction |
|
||||||
|
|
||||||
|
|||||||
@@ -211,7 +211,10 @@ def run_job(row, source, phase, cancelled=None):
|
|||||||
os.chown(workspace, user.pw_uid, user.pw_gid)
|
os.chown(workspace, user.pw_uid, user.pw_gid)
|
||||||
config = {'extension': row['extension'], 'phase': phase,
|
config = {'extension': row['extension'], 'phase': phase,
|
||||||
'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'),
|
'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'),
|
||||||
'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout}
|
'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout,
|
||||||
|
'ocr_dpi': setting('DOCUMENT_OCR_DPI', 300, 150, 400),
|
||||||
|
'ocr_max_dimension': setting('DOCUMENT_OCR_MAX_DIMENSION', 3500, 1500, 5000),
|
||||||
|
'ocr_page_timeout': setting('DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS', 30, 5, 120)}
|
||||||
for name, content in [('config.json', json.dumps(config))]:
|
for name, content in [('config.json', json.dumps(config))]:
|
||||||
path = os.path.join(workspace, name)
|
path = os.path.join(workspace, name)
|
||||||
with open(path, 'w') as handle:
|
with open(path, 'w') as handle:
|
||||||
|
|||||||
+49
-9
@@ -4,6 +4,7 @@
|
|||||||
import ctypes
|
import ctypes
|
||||||
import errno
|
import errno
|
||||||
import json
|
import json
|
||||||
|
import math
|
||||||
import os
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import platform
|
import platform
|
||||||
@@ -114,6 +115,47 @@ def office_text(path):
|
|||||||
return '\n'.join(text)[:MAX_TEXT]
|
return '\n'.join(text)[:MAX_TEXT]
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_candidates(body, image_list, pages):
|
||||||
|
"""Only text-poor pages that themselves contain raster images need OCR."""
|
||||||
|
image_pages = set()
|
||||||
|
for line in image_list.splitlines():
|
||||||
|
fields = line.split()
|
||||||
|
if len(fields) >= 5 and fields[0].isdigit() and fields[2] == 'image':
|
||||||
|
image_pages.add(int(fields[0]))
|
||||||
|
return [number for number, text in enumerate(body.split('\f')[:pages], 1)
|
||||||
|
if number in image_pages and len(re.sub(r'\s+', '', text)) < 80]
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_search_text(source, candidates, config):
|
||||||
|
# Search needs text only: do not rebuild a PDF or render at embedded-image
|
||||||
|
# DPI, which can enlarge an entire page because of a small high-DPI logo.
|
||||||
|
if not candidates:
|
||||||
|
return ''
|
||||||
|
sizes = command(['pdfinfo', '-f', '1', '-l', str(max(candidates)), str(source)])
|
||||||
|
dimensions = {int(number): max(float(width), float(height)) for number,width,height in
|
||||||
|
re.findall(r'Page\s+(\d+)\s+size:\s+([0-9.]+)\s+x\s+([0-9.]+)\s+pts', sizes)}
|
||||||
|
text = []
|
||||||
|
length = 0
|
||||||
|
dpi = config.get('ocr_dpi', 300)
|
||||||
|
maximum = config.get('ocr_max_dimension', 3500)
|
||||||
|
for page in candidates:
|
||||||
|
# The scale option overrides Poppler's DPI option. Compute the target
|
||||||
|
# from the physical page size so small pages are not upscaled to the cap.
|
||||||
|
scale = max(1, min(maximum, math.ceil(dimensions[page] * dpi / 72))) if page in dimensions else maximum
|
||||||
|
command(['pdftoppm', '-f', str(page), '-l', str(page), '-singlefile',
|
||||||
|
'-r', str(dpi), '-scale-to', str(scale),
|
||||||
|
'-gray', '-png', str(source), 'ocr-page'], timeout=config.get('ocr_page_timeout', 30))
|
||||||
|
fragment = command(['tesseract', 'ocr-page.png', 'stdout', '-l', config['language'],
|
||||||
|
'--dpi', str(config.get('ocr_dpi', 300))], timeout=config.get('ocr_page_timeout', 30))
|
||||||
|
fragment = fragment.strip()
|
||||||
|
if fragment:
|
||||||
|
text.append(fragment)
|
||||||
|
length += len(fragment) + 1
|
||||||
|
if length >= MAX_TEXT:
|
||||||
|
break
|
||||||
|
return '\n'.join(text)[:MAX_TEXT]
|
||||||
|
|
||||||
|
|
||||||
def extract(config):
|
def extract(config):
|
||||||
suffix = config['extension']
|
suffix = config['extension']
|
||||||
source = Path('input')
|
source = Path('input')
|
||||||
@@ -124,18 +166,16 @@ def extract(config):
|
|||||||
pages = int(found.group(1)) if found else 0
|
pages = int(found.group(1)) if found else 0
|
||||||
if pages > config['max_pages']:
|
if pages > config['max_pages']:
|
||||||
raise RuntimeError('PDF exceeds page limit')
|
raise RuntimeError('PDF exceeds page limit')
|
||||||
if config['phase'] == 'ocr':
|
|
||||||
command(['ocrmypdf', '--redo-ocr', '--output-type', 'pdf', '--optimize', '0', '--jobs', '1',
|
|
||||||
'--language', config['language'], '--tesseract-timeout', '120', '--skip-big', '50',
|
|
||||||
str(source), 'ocr.pdf'], timeout=config['timeout'] - 15)
|
|
||||||
source = Path('ocr.pdf')
|
|
||||||
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
|
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
|
||||||
body = read_text('text.txt')
|
body = read_text('text.txt')
|
||||||
if config['phase'] != 'ocr':
|
|
||||||
image_list = command(['pdfimages', '-list', str(source)])
|
image_list = command(['pdfimages', '-list', str(source)])
|
||||||
has_images = bool(re.search(r'^\s*\d+\s+\d+\s+image\s', image_list, re.MULTILINE))
|
candidates = ocr_candidates(body, image_list, pages)
|
||||||
page_text = body.split('\f')[:pages]
|
if config['phase'] == 'ocr':
|
||||||
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text)
|
recognized = ocr_search_text(source, candidates, config)
|
||||||
|
if recognized:
|
||||||
|
body = (body.rstrip() + '\n' + recognized)[:MAX_TEXT]
|
||||||
|
else:
|
||||||
|
needs_ocr = bool(candidates)
|
||||||
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
||||||
elif suffix in OFFICE_SUFFIXES:
|
elif suffix in OFFICE_SUFFIXES:
|
||||||
body = office_text(source)
|
body = office_text(source)
|
||||||
|
|||||||
@@ -105,6 +105,35 @@ class DocumentTests(DocumentFixture):
|
|||||||
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
|
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
|
||||||
self.find({'q':[query]})
|
self.find({'q':[query]})
|
||||||
|
|
||||||
|
def test_ocr_candidates_match_images_to_their_page_and_ignore_native_text_and_masks(self):
|
||||||
|
body='Short cover\f'+'Native text '*30+'\fTiny footer\fNo text\f'
|
||||||
|
images='page num type width height\n2 0 image 300 300\n3 1 image 500 500\n4 2 smask 500 500\n'
|
||||||
|
self.assertEqual(extract_document.ocr_candidates(body,images,4),[3])
|
||||||
|
self.assertEqual(extract_document.ocr_candidates(body,images,2),[])
|
||||||
|
self.assertEqual(extract_document.ocr_candidates('Header\fNative text '+('word '*30)+'\f','2 0 image 100 100',2),[])
|
||||||
|
|
||||||
|
def test_search_ocr_limits_render_size_and_time_and_does_not_rebuild_the_pdf(self):
|
||||||
|
calls=[]
|
||||||
|
def parser(args,timeout=120):
|
||||||
|
calls.append((args,timeout))
|
||||||
|
return ('Page 3 size: 595 x 842 pts\nPage 7 size: 144 x 288 pts' if args[0]=='pdfinfo' else
|
||||||
|
'Recognized page text' if args[0]=='tesseract' else '')
|
||||||
|
config={'language':'deu+eng','ocr_dpi':300,'ocr_max_dimension':3500,'ocr_page_timeout':30}
|
||||||
|
with mock.patch.object(extract_document,'command',side_effect=parser):
|
||||||
|
body=extract_document.ocr_search_text(Path('input'),[3,7],config)
|
||||||
|
self.assertEqual(body,'Recognized page text\nRecognized page text')
|
||||||
|
self.assertEqual([args[0] for args,_ in calls],['pdfinfo','pdftoppm','tesseract','pdftoppm','tesseract'])
|
||||||
|
for args,timeout in calls[1:]:
|
||||||
|
self.assertEqual(timeout,30)
|
||||||
|
if args[0]=='pdftoppm':
|
||||||
|
self.assertLessEqual(int(args[args.index('-scale-to')+1]),3500)
|
||||||
|
self.assertEqual(args[args.index('-r')+1],'300')
|
||||||
|
self.assertEqual(args[args.index('-f')+1],args[args.index('-l')+1])
|
||||||
|
self.assertEqual(calls[1][0][calls[1][0].index('-f')+1],'3')
|
||||||
|
self.assertEqual(calls[3][0][calls[3][0].index('-f')+1],'7')
|
||||||
|
self.assertEqual(calls[1][0][calls[1][0].index('-scale-to')+1],'3500')
|
||||||
|
self.assertEqual(calls[3][0][calls[3][0].index('-scale-to')+1],'1200')
|
||||||
|
|
||||||
def artifact_files(self):
|
def artifact_files(self):
|
||||||
files=[]
|
files=[]
|
||||||
for root,name in ((self.data/'Finance','Thumbs.db'),(self.data/'Finance'/'Child','THUMBS.DB'),
|
for root,name in ((self.data/'Finance','Thumbs.db'),(self.data/'Finance'/'Child','THUMBS.DB'),
|
||||||
|
|||||||
Reference in New Issue
Block a user