Files
ad-ds-simple-file-server/app/extract_document.py
T
2026-10-03 17:03:57 +00:00

213 lines
9.0 KiB
Python

#!/usr/bin/env python3
"""Run document parsers in a restricted, disposable workspace, never on originals."""
import ctypes
import errno
import json
import math
import os
from pathlib import Path
import platform
import re
import resource
import subprocess
import sys
import warnings
import zipfile
from xml.etree import ElementTree
try:
from .document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
except ImportError:
from document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
MAX_TEXT = 2 * 1024 * 1024
def sandbox(workspace):
"""Landlock filesystem rules and seccomp prevent parser access to live data/network."""
libc = ctypes.CDLL(None, use_errno=True)
abi = libc.syscall(444, 0, 0, 1)
if abi < 1:
raise RuntimeError('Document isolation requires Linux Landlock')
class Ruleset(ctypes.Structure):
_fields_ = [('fs', ctypes.c_uint64)]
class PathRule(ctypes.Structure):
_pack_ = 1
_fields_ = [('access', ctypes.c_uint64), ('fd', ctypes.c_int32)]
handled = (1 << 13) - 1
if abi >= 2:
handled |= 1 << 13
if abi >= 3:
handled |= 1 << 14
rules = Ruleset(handled)
rules_fd = libc.syscall(444, ctypes.byref(rules), ctypes.sizeof(rules), 0)
if rules_fd < 0:
raise OSError(ctypes.get_errno(), 'Cannot create document isolation')
read = (1 << 0) | (1 << 2) | (1 << 3)
try:
for path, access in [(workspace, handled), ('/usr', read), ('/lib', read), ('/lib64', read), ('/bin', read),
('/etc/fonts', read), ('/etc/ghostscript', read), ('/etc/ld.so.cache', 1 << 2),
('/etc/papersize', 1 << 2), ('/dev/null', (1 << 1) | (1 << 2)),
('/dev/urandom', 1 << 2), ('/proc/cpuinfo', 1 << 2), ('/proc/meminfo', 1 << 2)]:
if not os.path.exists(path):
continue
fd = os.open(path, os.O_PATH | os.O_CLOEXEC)
try:
rule = PathRule(access, fd)
if libc.syscall(445, rules_fd, 1, ctypes.byref(rule), 0) != 0:
raise OSError(ctypes.get_errno(), 'Cannot restrict document path')
finally:
os.close(fd)
if libc.prctl(38, 1, 0, 0, 0) != 0 or libc.syscall(446, rules_fd, 0) != 0:
raise OSError(ctypes.get_errno(), 'Cannot enter document isolation')
finally:
os.close(rules_fd)
# Permit local IPC used by OCR workers, deny creation of internet sockets.
socket_number = {'aarch64': 198, 'x86_64': 41}.get(platform.machine())
if socket_number is None:
raise RuntimeError('Unsupported document sandbox architecture')
class Filter(ctypes.Structure):
_fields_ = [('code', ctypes.c_ushort), ('jt', ctypes.c_ubyte), ('jf', ctypes.c_ubyte), ('k', ctypes.c_uint32)]
class Program(ctypes.Structure):
_fields_ = [('length', ctypes.c_ushort), ('filter', ctypes.POINTER(Filter))]
filters = (Filter * 6)(Filter(0x20,0,0,0), Filter(0x15,0,3,socket_number),
Filter(0x20,0,0,16), Filter(0x15,1,0,1),
Filter(0x06,0,0,0x50000 | errno.EPERM), Filter(0x06,0,0,0x7fff0000))
program = Program(len(filters), filters)
if libc.prctl(22, 2, ctypes.byref(program), 0, 0) != 0:
raise OSError(ctypes.get_errno(), 'Cannot isolate document networking')
def command(args, timeout=120):
result = subprocess.run(args, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout, check=False)
if result.returncode:
raise RuntimeError(f'{args[0]} failed ({result.returncode})')
return result.stdout.decode('utf-8', errors='replace')
def read_text(path):
with open(path, 'rb') as handle:
raw = handle.read(MAX_TEXT)
if raw.startswith((b'\xff\xfe', b'\xfe\xff')):
return raw.decode('utf-16', errors='replace')
return raw.decode('utf-8-sig', errors='replace')
def office_text(path):
text = []
total = 0
with zipfile.ZipFile(path) as archive:
entries = archive.infolist()
if len(entries) > 10000 or sum(entry.file_size for entry in entries) > 64 * 1024 * 1024:
raise RuntimeError('Office document exceeds extraction limits')
for entry in sorted(entries, key=lambda item: item.filename):
name = entry.filename
wanted = (name.startswith(('word/', 'ppt/slides/')) or name == 'content.xml') and name.endswith('.xml')
if not wanted or entry.file_size > 16 * 1024 * 1024:
continue
node = ElementTree.fromstring(archive.read(entry))
fragment = ' '.join(node.itertext())
text.append(fragment)
total += len(fragment)
if total > MAX_TEXT:
break
return '\n'.join(text)[:MAX_TEXT]
def ocr_candidates(body, image_list, pages):
"""Only text-poor pages that themselves contain raster images need OCR."""
image_pages = set()
for line in image_list.splitlines():
fields = line.split()
if len(fields) >= 5 and fields[0].isdigit() and fields[2] == 'image':
image_pages.add(int(fields[0]))
return [number for number, text in enumerate(body.split('\f')[:pages], 1)
if number in image_pages and len(re.sub(r'\s+', '', text)) < 80]
def ocr_search_text(source, candidates, config):
# Search needs text only: do not rebuild a PDF or render at embedded-image
# DPI, which can enlarge an entire page because of a small high-DPI logo.
if not candidates:
return ''
sizes = command(['pdfinfo', '-f', '1', '-l', str(max(candidates)), str(source)])
dimensions = {int(number): max(float(width), float(height)) for number,width,height in
re.findall(r'Page\s+(\d+)\s+size:\s+([0-9.]+)\s+x\s+([0-9.]+)\s+pts', sizes)}
text = []
length = 0
dpi = config.get('ocr_dpi', 300)
maximum = config.get('ocr_max_dimension', 3500)
for page in candidates:
# The scale option overrides Poppler's DPI option. Compute the target
# from the physical page size so small pages are not upscaled to the cap.
scale = max(1, min(maximum, math.ceil(dimensions[page] * dpi / 72))) if page in dimensions else maximum
command(['pdftoppm', '-f', str(page), '-l', str(page), '-singlefile',
'-r', str(dpi), '-scale-to', str(scale),
'-gray', '-png', str(source), 'ocr-page'], timeout=config.get('ocr_page_timeout', 30))
fragment = command(['tesseract', 'ocr-page.png', 'stdout', '-l', config['language'],
'--dpi', str(config.get('ocr_dpi', 300))], timeout=config.get('ocr_page_timeout', 30))
fragment = fragment.strip()
if fragment:
text.append(fragment)
length += len(fragment) + 1
if length >= MAX_TEXT:
break
return '\n'.join(text)[:MAX_TEXT]
def extract(config):
suffix = config['extension']
source = Path('input')
body, pages, needs_ocr = '', 0, False
if suffix == '.pdf':
info = command(['pdfinfo', str(source)])
found = re.search(r'^Pages:\s*(\d+)', info, flags=re.MULTILINE)
pages = int(found.group(1)) if found else 0
if pages > config['max_pages']:
raise RuntimeError('PDF exceeds page limit')
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
body = read_text('text.txt')
image_list = command(['pdfimages', '-list', str(source)])
candidates = ocr_candidates(body, image_list, pages)
if config['phase'] == 'ocr':
recognized = ocr_search_text(source, candidates, config)
if recognized:
body = (body.rstrip() + '\n' + recognized)[:MAX_TEXT]
else:
needs_ocr = bool(candidates)
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
elif suffix in OFFICE_SUFFIXES:
body = office_text(source)
elif suffix in IMAGE_SUFFIXES:
from PIL import Image
Image.MAX_IMAGE_PIXELS = 25_000_000
warnings.simplefilter('error', Image.DecompressionBombWarning)
with Image.open(source) as image:
image.thumbnail((1400,1400))
image.convert('RGB').save('preview.jpg', 'JPEG', quality=85)
else:
raise ValueError('File type is not eligible for content extraction')
return {'body': body[:MAX_TEXT], 'pages': pages, 'needsOcr': needs_ocr}
def main():
workspace = os.path.abspath(sys.argv[1])
os.chdir(workspace)
config = json.loads(Path('config.json').read_text())
resource.setrlimit(resource.RLIMIT_CORE, (0,0))
resource.setrlimit(resource.RLIMIT_AS, (2 * 1024**3, 2 * 1024**3))
resource.setrlimit(resource.RLIMIT_FSIZE, (512 * 1024**2, 512 * 1024**2))
resource.setrlimit(resource.RLIMIT_CPU, (config['timeout'], config['timeout']))
try:
sandbox(workspace)
result = extract(config)
except Exception as exc:
result = {'error': str(exc)[:200]}
Path('result.json').write_text(json.dumps(result))
return 0 if 'error' not in result else 1
if __name__ == '__main__':
sys.exit(main())