Compare commits
4
Commits
9944f2be1f
..
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2682658507 | ||
|
|
6ddc9ebcab | ||
|
|
fc05509ac7 | ||
|
|
2b23422fa3 |
@@ -33,6 +33,9 @@ ACME_HTTP_PORT=80
|
|||||||
# DOCUMENT_SCAN_SECONDS=30
|
# DOCUMENT_SCAN_SECONDS=30
|
||||||
# DOCUMENT_OCR_LANGUAGE=deu+eng
|
# DOCUMENT_OCR_LANGUAGE=deu+eng
|
||||||
# DOCUMENT_OCR_TIMEOUT_SECONDS=600
|
# DOCUMENT_OCR_TIMEOUT_SECONDS=600
|
||||||
|
# DOCUMENT_OCR_DPI=300
|
||||||
|
# DOCUMENT_OCR_MAX_DIMENSION=3500
|
||||||
|
# DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS=30
|
||||||
# DOCUMENT_MAX_FILE_MB=512
|
# DOCUMENT_MAX_FILE_MB=512
|
||||||
# DOCUMENT_MAX_PDF_PAGES=500
|
# DOCUMENT_MAX_PDF_PAGES=500
|
||||||
# TRASH_RETENTION_DAYS=7
|
# TRASH_RETENTION_DAYS=7
|
||||||
|
|||||||
@@ -71,15 +71,15 @@ The Data share uses Samba Windows ACL checks instead of POSIX ACLs. Keep its dat
|
|||||||
|
|
||||||
## File Search and PDF Recognition
|
## File Search and PDF Recognition
|
||||||
|
|
||||||
Users sign in at `/` with their existing AD credentials using `username`, `DOMAIN\username`, or `username@domain`. Domain-qualified names accept the configured AD DNS domain/realm; the short NetBIOS domain is also accepted after `@`. The file view provides live search by filename and content, folder and file-type filters, grid/list views, previews, and downloads of the original files. Only Domain Admins see the administration link and can use management pages and APIs under `/admin/...`; old administration URLs redirect there. Administrative APIs are under `/admin/api/...`, while sign-in, session, and document endpoints remain under `/api/...`. This upgrade expires existing web sessions.
|
Users sign in at `/` with their existing AD credentials using `username`, `DOMAIN\username`, or `username@domain`. Domain-qualified names accept the configured AD DNS domain/realm; the short NetBIOS domain is also accepted after `@`. The file view provides live search by filename and content, folder and file-type filters, grid/list views, previews, and downloads of the original files. Grid and list entries show file metadata without extracted content snippets. Only Domain Admins see the administration link and can use management pages and APIs under `/admin/...`; old administration URLs redirect there. Administrative APIs are under `/admin/api/...`, while sign-in, session, and document endpoints remain under `/api/...`. This upgrade expires existing web sessions.
|
||||||
|
|
||||||
The catalog includes active Data folders and each user's own Private folder. Every search, detail, preview, and download request checks current individual folder permissions. Revoking access or archiving a folder takes effect without waiting for reindexing. Domain Admins can see all active Data folders, but the user file view still shows only their own Private folder. Archived folders, recycle repositories, FSLogix profiles, symlinks, nested mounts, and special files are excluded. Private files additionally require ownership and read/traverse permissions for the signed-in user's mapped Unix identity.
|
The catalog includes active Data folders and each user's own Private folder. Every search, detail, preview, and download request checks current individual folder permissions. Revoking access or archiving a folder takes effect without waiting for reindexing. Domain Admins can see all active Data folders, but the user file view still shows only their own Private folder. Archived folders, recycle repositories, FSLogix profiles, symlinks, nested mounts, and special files are excluded. `Thumbs.db` (case-insensitive) and Office lock files whose names start with `~$` are omitted from results, counts and file-type filters in both Data and Private folders. Upgrades remove their existing catalog entries without deleting original files. Private files additionally require ownership and read/traverse permissions for the signed-in user's mapped Unix identity.
|
||||||
|
|
||||||
Linux filesystem events update the filename catalog as files change, with periodic scans recovering missed events. Text extraction and OCR run through a persistent background queue; files become searchable by filename before content extraction finishes. Search updates while typing, and open views refresh every two seconds to show completed extraction. Full-text queries match word prefixes and ignore accents; filename-only queries also match substrings.
|
Linux filesystem events update the filename catalog as files change, with periodic scans recovering missed events. Text extraction and OCR run through a persistent background queue; files become searchable by filename before content extraction finishes. Search updates while typing, and open views refresh every two seconds to show completed extraction. Full-text queries match word prefixes and ignore accents; filename-only queries also match substrings.
|
||||||
|
|
||||||
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
||||||
|
|
||||||
PDFs containing raster images and pages with little extracted text are queued for local OCR using OCRmyPDF and Tesseract, with German and English enabled by default. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
|
Only PDF pages that themselves contain raster images and little extracted text are queued for local OCR using Poppler and Tesseract, with German and English enabled by default. Search OCR renders those pages at 300 DPI, capped at 3500 pixels on the longest side, instead of inheriting high DPI from embedded logos or rebuilding an OCR PDF. Each render and recognition step has a 30-second timeout. Photos can still require a recognition attempt to establish whether they contain text; text-free pages finish without adding search content. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
|
||||||
|
|
||||||
Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities.
|
Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities.
|
||||||
|
|
||||||
@@ -92,6 +92,9 @@ Domain Admins monitor the worker at `/admin/documents`: current file and phase,
|
|||||||
| `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes |
|
| `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes |
|
||||||
| `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR |
|
| `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR |
|
||||||
| `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job |
|
| `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job |
|
||||||
|
| `DOCUMENT_OCR_DPI` | `300` | OCR render resolution (150–400 DPI), subject to the pixel cap |
|
||||||
|
| `DOCUMENT_OCR_MAX_DIMENSION` | `3500` | Maximum OCR image width/height (1500–5000 pixels); originals are unchanged |
|
||||||
|
| `DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS` | `30` | Maximum runtime per page render or recognition step (5–120 seconds) |
|
||||||
| `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search |
|
| `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search |
|
||||||
| `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction |
|
| `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction |
|
||||||
|
|
||||||
|
|||||||
@@ -211,7 +211,10 @@ def run_job(row, source, phase, cancelled=None):
|
|||||||
os.chown(workspace, user.pw_uid, user.pw_gid)
|
os.chown(workspace, user.pw_uid, user.pw_gid)
|
||||||
config = {'extension': row['extension'], 'phase': phase,
|
config = {'extension': row['extension'], 'phase': phase,
|
||||||
'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'),
|
'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'),
|
||||||
'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout}
|
'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout,
|
||||||
|
'ocr_dpi': setting('DOCUMENT_OCR_DPI', 300, 150, 400),
|
||||||
|
'ocr_max_dimension': setting('DOCUMENT_OCR_MAX_DIMENSION', 3500, 1500, 5000),
|
||||||
|
'ocr_page_timeout': setting('DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS', 30, 5, 120)}
|
||||||
for name, content in [('config.json', json.dumps(config))]:
|
for name, content in [('config.json', json.dumps(config))]:
|
||||||
path = os.path.join(workspace, name)
|
path = os.path.join(workspace, name)
|
||||||
with open(path, 'w') as handle:
|
with open(path, 'w') as handle:
|
||||||
@@ -300,6 +303,11 @@ def process_next(conn, prefer_ocr=False, stop=None):
|
|||||||
if row is None:
|
if row is None:
|
||||||
worker_state(conn, 'idle')
|
worker_state(conn, 'idle')
|
||||||
return False
|
return False
|
||||||
|
if documents.is_excluded_filename(row['name']):
|
||||||
|
conn.execute('DELETE FROM documents WHERE id=?', (row['id'],))
|
||||||
|
conn.commit()
|
||||||
|
worker_state(conn, 'idle')
|
||||||
|
return True
|
||||||
source = conn.execute('SELECT * FROM document_sources WHERE id=?', (row['source_id'],)).fetchone()
|
source = conn.execute('SELECT * FROM document_sources WHERE id=?', (row['source_id'],)).fetchone()
|
||||||
if source is None:
|
if source is None:
|
||||||
return False
|
return False
|
||||||
|
|||||||
+17
-6
@@ -25,6 +25,11 @@ except ImportError:
|
|||||||
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
|
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
|
||||||
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
|
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
|
||||||
MAX_TEXT = 2 * 1024 * 1024
|
MAX_TEXT = 2 * 1024 * 1024
|
||||||
|
SEARCHABLE_NAME_SQL = "lower({name}) <> 'thumbs.db' AND {name} NOT GLOB '~$*'"
|
||||||
|
|
||||||
|
|
||||||
|
def is_excluded_filename(name):
|
||||||
|
return name.casefold() == 'thumbs.db' or name.startswith('~$')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -83,17 +88,20 @@ def ensure_schema(conn):
|
|||||||
|
|
||||||
# Upgrade the derived index only: keep catalog IDs, permissions, previews
|
# Upgrade the derived index only: keep catalog IDs, permissions, previews
|
||||||
# and originals, but remove content and jobs for excluded file types.
|
# and originals, but remove content and jobs for excluded file types.
|
||||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
if conn.execute('PRAGMA user_version').fetchone()[0] < 2:
|
||||||
with conn:
|
with conn:
|
||||||
conn.execute('BEGIN IMMEDIATE')
|
conn.execute('BEGIN IMMEDIATE')
|
||||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
version = conn.execute('PRAGMA user_version').fetchone()[0]
|
||||||
|
if version < 1:
|
||||||
content = sorted(CONTENT_SUFFIXES)
|
content = sorted(CONTENT_SUFFIXES)
|
||||||
images = sorted(IMAGE_SUFFIXES)
|
images = sorted(IMAGE_SUFFIXES)
|
||||||
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
|
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
|
||||||
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
|
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
|
||||||
THEN 'pending' ELSE 'name-only' END
|
THEN 'pending' ELSE 'name-only' END
|
||||||
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
|
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
|
||||||
conn.execute('PRAGMA user_version=1')
|
if version < 2:
|
||||||
|
conn.execute(f"DELETE FROM documents WHERE NOT ({SEARCHABLE_NAME_SQL.format(name='name')})")
|
||||||
|
conn.execute('PRAGMA user_version=2')
|
||||||
|
|
||||||
|
|
||||||
def worker_paused(conn):
|
def worker_paused(conn):
|
||||||
@@ -243,6 +251,10 @@ def allowed_sources(conn, identity):
|
|||||||
|
|
||||||
|
|
||||||
def discover_file(conn, source, relative, seen=0):
|
def discover_file(conn, source, relative, seen=0):
|
||||||
|
name = relative.rsplit('/', 1)[-1]
|
||||||
|
if is_excluded_filename(name):
|
||||||
|
conn.execute('DELETE FROM documents WHERE source_id=? AND path=?', (source['id'], relative))
|
||||||
|
return
|
||||||
try:
|
try:
|
||||||
with open_file(source['root'], relative) as (_, info):
|
with open_file(source['root'], relative) as (_, info):
|
||||||
stamp = fingerprint(info)
|
stamp = fingerprint(info)
|
||||||
@@ -254,7 +266,6 @@ def discover_file(conn, source, relative, seen=0):
|
|||||||
if seen:
|
if seen:
|
||||||
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
|
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
|
||||||
return
|
return
|
||||||
name = relative.rsplit('/', 1)[-1]
|
|
||||||
extension = os.path.splitext(name)[1].lower()
|
extension = os.path.splitext(name)[1].lower()
|
||||||
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
|
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
|
||||||
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
|
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
|
||||||
@@ -348,7 +359,7 @@ def private_readable(source, relative, uid):
|
|||||||
def current_document(conn, document_id, identity):
|
def current_document(conn, document_id, identity):
|
||||||
sources = allowed_sources(conn, identity)
|
sources = allowed_sources(conn, identity)
|
||||||
row = conn.execute('SELECT * FROM documents WHERE id=?', (document_id,)).fetchone()
|
row = conn.execute('SELECT * FROM documents WHERE id=?', (document_id,)).fetchone()
|
||||||
if row is None or row['source_id'] not in sources:
|
if row is None or row['source_id'] not in sources or is_excluded_filename(row['name']):
|
||||||
raise FileNotFoundError('Document not found')
|
raise FileNotFoundError('Document not found')
|
||||||
source = sources[row['source_id']]
|
source = sources[row['source_id']]
|
||||||
if not private_readable(source, row['path'], identity.get('uid')):
|
if not private_readable(source, row['path'], identity.get('uid')):
|
||||||
@@ -397,7 +408,7 @@ def search(conn, identity, params):
|
|||||||
for row in conn.execute(f"SELECT id,source_id,path FROM documents WHERE source_id IN ({placeholders}) AND source_id LIKE 'private:%'", ids).fetchall():
|
for row in conn.execute(f"SELECT id,source_id,path FROM documents WHERE source_id IN ({placeholders}) AND source_id LIKE 'private:%'", ids).fetchall():
|
||||||
if private_readable(selected[row['source_id']], row['path'], identity.get('uid')):
|
if private_readable(selected[row['source_id']], row['path'], identity.get('uid')):
|
||||||
conn.execute('INSERT INTO readable_private VALUES(?)', (row['id'],))
|
conn.execute('INSERT INTO readable_private VALUES(?)', (row['id'],))
|
||||||
private_condition = "(d.source_id NOT LIKE 'private:%' OR EXISTS (SELECT 1 FROM readable_private r WHERE r.id=d.id))"
|
private_condition = SEARCHABLE_NAME_SQL.format(name='d.name') + " AND (d.source_id NOT LIKE 'private:%' OR EXISTS (SELECT 1 FROM readable_private r WHERE r.id=d.id))"
|
||||||
conditions = [f'd.source_id IN ({placeholders})', private_condition]
|
conditions = [f'd.source_id IN ({placeholders})', private_condition]
|
||||||
values = ids[:]
|
values = ids[:]
|
||||||
from_sql = 'documents d'
|
from_sql = 'documents d'
|
||||||
|
|||||||
+49
-9
@@ -4,6 +4,7 @@
|
|||||||
import ctypes
|
import ctypes
|
||||||
import errno
|
import errno
|
||||||
import json
|
import json
|
||||||
|
import math
|
||||||
import os
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import platform
|
import platform
|
||||||
@@ -114,6 +115,47 @@ def office_text(path):
|
|||||||
return '\n'.join(text)[:MAX_TEXT]
|
return '\n'.join(text)[:MAX_TEXT]
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_candidates(body, image_list, pages):
|
||||||
|
"""Only text-poor pages that themselves contain raster images need OCR."""
|
||||||
|
image_pages = set()
|
||||||
|
for line in image_list.splitlines():
|
||||||
|
fields = line.split()
|
||||||
|
if len(fields) >= 5 and fields[0].isdigit() and fields[2] == 'image':
|
||||||
|
image_pages.add(int(fields[0]))
|
||||||
|
return [number for number, text in enumerate(body.split('\f')[:pages], 1)
|
||||||
|
if number in image_pages and len(re.sub(r'\s+', '', text)) < 80]
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_search_text(source, candidates, config):
|
||||||
|
# Search needs text only: do not rebuild a PDF or render at embedded-image
|
||||||
|
# DPI, which can enlarge an entire page because of a small high-DPI logo.
|
||||||
|
if not candidates:
|
||||||
|
return ''
|
||||||
|
sizes = command(['pdfinfo', '-f', '1', '-l', str(max(candidates)), str(source)])
|
||||||
|
dimensions = {int(number): max(float(width), float(height)) for number,width,height in
|
||||||
|
re.findall(r'Page\s+(\d+)\s+size:\s+([0-9.]+)\s+x\s+([0-9.]+)\s+pts', sizes)}
|
||||||
|
text = []
|
||||||
|
length = 0
|
||||||
|
dpi = config.get('ocr_dpi', 300)
|
||||||
|
maximum = config.get('ocr_max_dimension', 3500)
|
||||||
|
for page in candidates:
|
||||||
|
# The scale option overrides Poppler's DPI option. Compute the target
|
||||||
|
# from the physical page size so small pages are not upscaled to the cap.
|
||||||
|
scale = max(1, min(maximum, math.ceil(dimensions[page] * dpi / 72))) if page in dimensions else maximum
|
||||||
|
command(['pdftoppm', '-f', str(page), '-l', str(page), '-singlefile',
|
||||||
|
'-r', str(dpi), '-scale-to', str(scale),
|
||||||
|
'-gray', '-png', str(source), 'ocr-page'], timeout=config.get('ocr_page_timeout', 30))
|
||||||
|
fragment = command(['tesseract', 'ocr-page.png', 'stdout', '-l', config['language'],
|
||||||
|
'--dpi', str(config.get('ocr_dpi', 300))], timeout=config.get('ocr_page_timeout', 30))
|
||||||
|
fragment = fragment.strip()
|
||||||
|
if fragment:
|
||||||
|
text.append(fragment)
|
||||||
|
length += len(fragment) + 1
|
||||||
|
if length >= MAX_TEXT:
|
||||||
|
break
|
||||||
|
return '\n'.join(text)[:MAX_TEXT]
|
||||||
|
|
||||||
|
|
||||||
def extract(config):
|
def extract(config):
|
||||||
suffix = config['extension']
|
suffix = config['extension']
|
||||||
source = Path('input')
|
source = Path('input')
|
||||||
@@ -124,18 +166,16 @@ def extract(config):
|
|||||||
pages = int(found.group(1)) if found else 0
|
pages = int(found.group(1)) if found else 0
|
||||||
if pages > config['max_pages']:
|
if pages > config['max_pages']:
|
||||||
raise RuntimeError('PDF exceeds page limit')
|
raise RuntimeError('PDF exceeds page limit')
|
||||||
if config['phase'] == 'ocr':
|
|
||||||
command(['ocrmypdf', '--redo-ocr', '--output-type', 'pdf', '--optimize', '0', '--jobs', '1',
|
|
||||||
'--language', config['language'], '--tesseract-timeout', '120', '--skip-big', '50',
|
|
||||||
str(source), 'ocr.pdf'], timeout=config['timeout'] - 15)
|
|
||||||
source = Path('ocr.pdf')
|
|
||||||
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
|
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
|
||||||
body = read_text('text.txt')
|
body = read_text('text.txt')
|
||||||
if config['phase'] != 'ocr':
|
|
||||||
image_list = command(['pdfimages', '-list', str(source)])
|
image_list = command(['pdfimages', '-list', str(source)])
|
||||||
has_images = bool(re.search(r'^\s*\d+\s+\d+\s+image\s', image_list, re.MULTILINE))
|
candidates = ocr_candidates(body, image_list, pages)
|
||||||
page_text = body.split('\f')[:pages]
|
if config['phase'] == 'ocr':
|
||||||
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text)
|
recognized = ocr_search_text(source, candidates, config)
|
||||||
|
if recognized:
|
||||||
|
body = (body.rstrip() + '\n' + recognized)[:MAX_TEXT]
|
||||||
|
else:
|
||||||
|
needs_ocr = bool(candidates)
|
||||||
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
||||||
elif suffix in OFFICE_SUFFIXES:
|
elif suffix in OFFICE_SUFFIXES:
|
||||||
body = office_text(source)
|
body = office_text(source)
|
||||||
|
|||||||
+1
-1
@@ -11,7 +11,7 @@ const esc = value => String(value ?? "").replace(/[&<>'"]/g, char => ({"&":"&
|
|||||||
function actionIcon(action) {
|
function actionIcon(action) {
|
||||||
const paths = {
|
const paths = {
|
||||||
download: '<path d="M12 3v12m-5-5 5 5 5-5"/><path d="M5 16v4h14v-4"/>',
|
download: '<path d="M12 3v12m-5-5 5 5 5-5"/><path d="M5 16v4h14v-4"/>',
|
||||||
restore: '<path d="M5 12a7 7 0 1 1 2.05 4.95"/><path d="m1 8 4 4 4-4"/>',
|
restore: '<path d="m9 4-5 5 5 5"/><path d="M4 9h10a6 6 0 0 1 0 12h-3"/>',
|
||||||
};
|
};
|
||||||
return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[action]}</svg>`;
|
return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[action]}</svg>`;
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-3
@@ -2,7 +2,6 @@
|
|||||||
|
|
||||||
const $ = selector => document.querySelector(selector);
|
const $ = selector => document.querySelector(selector);
|
||||||
const esc = value => String(value ?? "").replace(/[&<>'"]/g, char => ({"&":"&","<":"<",">":">","'":"'",'"':"""}[char]));
|
const esc = value => String(value ?? "").replace(/[&<>'"]/g, char => ({"&":"&","<":"<",">":">","'":"'",'"':"""}[char]));
|
||||||
const snippet = value => esc(value).replace(/\x01/g, "<mark>").replace(/\x02/g, "</mark>");
|
|
||||||
const sizeLabel = value => {
|
const sizeLabel = value => {
|
||||||
const units = ["B", "kB", "MB", "GB", "TB"];
|
const units = ["B", "kB", "MB", "GB", "TB"];
|
||||||
let amount = Number(value), index = 0;
|
let amount = Number(value), index = 0;
|
||||||
@@ -18,7 +17,6 @@ function icon(name) {
|
|||||||
};
|
};
|
||||||
return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[name]}</svg>`;
|
return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[name]}</svg>`;
|
||||||
}
|
}
|
||||||
const statusLabel = row => ({pending:"Text wird erfasst…",ocr:row.attempts ? "Texterkennung wird erneut versucht…" : "Texterkennung läuft…",failed:"Text konnte noch nicht erfasst werden.","name-only":"Suche nach Dateinamen verfügbar."}[row.state] || "");
|
|
||||||
let session = null, offset = 0, listMode = false, timer, debounce, request, serial = 0, lastResult = "", activeDocument = null, initialFilters = null;
|
let session = null, offset = 0, listMode = false, timer, debounce, request, serial = 0, lastResult = "", activeDocument = null, initialFilters = null;
|
||||||
const pageSize = 40;
|
const pageSize = 40;
|
||||||
|
|
||||||
@@ -99,7 +97,7 @@ async function refresh(userAction = true) {
|
|||||||
$("#documents").innerHTML = result.items.length ? result.items.map(row => `<article class="document-card">
|
$("#documents").innerHTML = result.items.length ? result.items.map(row => `<article class="document-card">
|
||||||
<button type="button" class="document-open" data-document="${esc(row.id)}" aria-label="Öffnen: ${esc(row.name)}">
|
<button type="button" class="document-open" data-document="${esc(row.id)}" aria-label="Öffnen: ${esc(row.name)}">
|
||||||
<div class="document-thumbnail">${row.hasPreview ? `<img src="/api/documents/${row.id}/preview?v=${row.version}" loading="lazy" alt="">` : `<div class="document-placeholder">${icon("file")}<span>${esc((row.extension || "Datei").toUpperCase())}</span></div>`}</div>
|
<div class="document-thumbnail">${row.hasPreview ? `<img src="/api/documents/${row.id}/preview?v=${row.version}" loading="lazy" alt="">` : `<div class="document-placeholder">${icon("file")}<span>${esc((row.extension || "Datei").toUpperCase())}</span></div>`}</div>
|
||||||
<div class="document-description"><h2 title="${esc(row.name)}">${esc(row.name)}</h2><p class="document-path" title="${esc(row.path)}">${esc(row.path)}</p><p class="document-excerpt">${snippet(row.snippet || statusLabel(row))}</p></div>
|
<div class="document-description"><h2 title="${esc(row.name)}">${esc(row.name)}</h2><p class="document-path" title="${esc(row.path)}">${esc(row.path)}</p></div>
|
||||||
</button><footer><span title="${esc(row.source)}">${esc(row.kind === "private" ? "Private" : row.source)} · ${sizeLabel(row.size)}<small>${modifiedLabel(row.modified)} UTC</small></span><a class="button icon-button" href="/api/documents/${row.id}/download" download="${esc(row.name)}" aria-label="Herunterladen: ${esc(row.name)}" title="Herunterladen">${icon("download")}</a></footer>
|
</button><footer><span title="${esc(row.source)}">${esc(row.kind === "private" ? "Private" : row.source)} · ${sizeLabel(row.size)}<small>${modifiedLabel(row.modified)} UTC</small></span><a class="button icon-button" href="/api/documents/${row.id}/download" download="${esc(row.name)}" aria-label="Herunterladen: ${esc(row.name)}" title="Herunterladen">${icon("download")}</a></footer>
|
||||||
</article>`).join("") : '<div class="portal-empty"><h2>Keine Dateien gefunden</h2><p>Suchbegriff oder Filter ändern.</p></div>';
|
</article>`).join("") : '<div class="portal-empty"><h2>Keine Dateien gefunden</h2><p>Suchbegriff oder Filter ändern.</p></div>';
|
||||||
}
|
}
|
||||||
|
|||||||
+8
-4
@@ -6,6 +6,7 @@
|
|||||||
* { box-sizing: border-box; }
|
* { box-sizing: border-box; }
|
||||||
[hidden] { display: none !important; }
|
[hidden] { display: none !important; }
|
||||||
body { margin: 0; min-height: 100vh; }
|
body { margin: 0; min-height: 100vh; }
|
||||||
|
html:has(#document-dialog[open]), body:has(#document-dialog[open]) { overflow: hidden; overscroll-behavior: none; }
|
||||||
a { color: #0645ad; }
|
a { color: #0645ad; }
|
||||||
button, input, select { font: inherit; }
|
button, input, select { font: inherit; }
|
||||||
button, .button { display: inline-block; padding: .4rem .7rem; border: 1px solid #777; color: #111; background: #eee; cursor: pointer; text-decoration: none; white-space: nowrap; }
|
button, .button { display: inline-block; padding: .4rem .7rem; border: 1px solid #777; color: #111; background: #eee; cursor: pointer; text-decoration: none; white-space: nowrap; }
|
||||||
@@ -192,7 +193,6 @@ progress { width: 100%; }
|
|||||||
.document-description { padding: .8rem; }
|
.document-description { padding: .8rem; }
|
||||||
.document-description h2 { font-size: 1rem; margin: 0 0 .3rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
.document-description h2 { font-size: 1rem; margin: 0 0 .3rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||||
.document-path { font-size: .8rem; color: #666; margin: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
.document-path { font-size: .8rem; color: #666; margin: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||||
.document-excerpt { height: 3.6em; overflow: hidden; font-size: .85rem; margin: .6rem 0 0; overflow-wrap: anywhere; }
|
|
||||||
mark { background: #fff0a0; color: #111; }
|
mark { background: #fff0a0; color: #111; }
|
||||||
.document-card footer { display: flex; justify-content: space-between; align-items: center; gap: .5rem; padding: .6rem .8rem; border-top: 1px solid #ccc; }
|
.document-card footer { display: flex; justify-content: space-between; align-items: center; gap: .5rem; padding: .6rem .8rem; border-top: 1px solid #ccc; }
|
||||||
.document-card footer > span { min-width: 0; font-size: .8rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
.document-card footer > span { min-width: 0; font-size: .8rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||||
@@ -203,7 +203,6 @@ mark { background: #fff0a0; color: #111; }
|
|||||||
.document-list .document-thumbnail { width: 80px; height: 100px; flex-shrink: 0; border: 0; border-right: 1px solid #ccc; }
|
.document-list .document-thumbnail { width: 80px; height: 100px; flex-shrink: 0; border: 0; border-right: 1px solid #ccc; }
|
||||||
.document-list .document-description { min-width: 0; }
|
.document-list .document-description { min-width: 0; }
|
||||||
.document-list footer { flex-shrink: 0; border: 0; }
|
.document-list footer { flex-shrink: 0; border: 0; }
|
||||||
.document-list .document-excerpt { height: 1.4em; }
|
|
||||||
.portal-empty { grid-column: 1/-1; padding: 2rem; border: 1px solid #aaa; text-align: center; color: #666; }
|
.portal-empty { grid-column: 1/-1; padding: 2rem; border: 1px solid #aaa; text-align: center; color: #666; }
|
||||||
.portal-empty h2 { color: #111; }
|
.portal-empty h2 { color: #111; }
|
||||||
.portal-pagination { display: flex; align-items: center; justify-content: center; gap: 1rem; margin-top: 1.5rem; }
|
.portal-pagination { display: flex; align-items: center; justify-content: center; gap: 1rem; margin-top: 1.5rem; }
|
||||||
@@ -217,8 +216,13 @@ mark { background: #fff0a0; color: #111; }
|
|||||||
.document-dialog-head p { font-size: .85rem; overflow-wrap: anywhere; margin-bottom: 0; }
|
.document-dialog-head p { font-size: .85rem; overflow-wrap: anywhere; margin-bottom: 0; }
|
||||||
.document-detail { height: calc(100% - 95px); min-height: 0; }
|
.document-detail { height: calc(100% - 95px); min-height: 0; }
|
||||||
.document-preview { padding: 1rem; overflow: auto; background: #f5f5f5; }
|
.document-preview { padding: 1rem; overflow: auto; background: #f5f5f5; }
|
||||||
.document-preview img { display: block; width: 100%; height: auto; }
|
.document-dialog:has(.document-preview img) { overflow: hidden; }
|
||||||
.document-dialog.pdf-dialog { width: 100vw; height: 100dvh; max-width: 100vw; max-height: 100dvh; margin: 0; border: 0; }
|
.document-dialog:has(.document-preview img)[open] { display: flex; flex-direction: column; }
|
||||||
|
.document-dialog:has(.document-preview img) .document-dialog-head { flex-shrink: 0; }
|
||||||
|
.document-dialog:has(.document-preview img) .document-detail { flex: 1; height: auto; overflow: hidden; }
|
||||||
|
.document-dialog:has(.document-preview img) .document-preview { display: flex; align-items: center; justify-content: center; height: 100%; overflow: hidden; }
|
||||||
|
.document-preview img { display: block; width: auto; height: auto; max-width: 100%; max-height: 100%; object-fit: contain; }
|
||||||
|
.document-dialog.pdf-dialog { width: 100vw; height: 100dvh; max-width: 100vw; max-height: 100dvh; margin: 0; border: 0; overflow: hidden; }
|
||||||
.document-dialog.pdf-dialog[open] { display: flex; flex-direction: column; }
|
.document-dialog.pdf-dialog[open] { display: flex; flex-direction: column; }
|
||||||
.pdf-dialog .document-dialog-head { flex-shrink: 0; padding: .6rem 1rem; }
|
.pdf-dialog .document-dialog-head { flex-shrink: 0; padding: .6rem 1rem; }
|
||||||
.pdf-dialog .document-detail { flex: 1; height: auto; overflow: hidden; }
|
.pdf-dialog .document-detail { flex: 1; height: auto; overflow: hidden; }
|
||||||
|
|||||||
+26
-9
@@ -332,19 +332,24 @@ def authenticate_user(username: str, password: str) -> Optional[Dict[str, str]]:
|
|||||||
|
|
||||||
workgroup = os.environ["WORKGROUP"]
|
workgroup = os.environ["WORKGROUP"]
|
||||||
realm = os.environ["REALM"]
|
realm = os.environ["REALM"]
|
||||||
|
is_upn = "\\" not in qualified
|
||||||
if "\\" in qualified:
|
if "\\" in qualified:
|
||||||
domain_name, account = qualified.split("\\", 1)
|
domain_name, account = qualified.split("\\", 1)
|
||||||
if domain_name.casefold() != workgroup.casefold():
|
if domain_name.casefold() != workgroup.casefold():
|
||||||
return None
|
return None
|
||||||
|
principal = f"{account}@{realm}"
|
||||||
else:
|
else:
|
||||||
account, principal_realm = qualified.rsplit("@", 1)
|
if qualified.count("@") != 1:
|
||||||
domain_names = {realm.casefold(), workgroup.casefold(), os.getenv("DOMAIN", realm).casefold()}
|
|
||||||
if principal_realm.casefold() not in domain_names:
|
|
||||||
return None
|
return None
|
||||||
if not account or is_excluded_user(account):
|
account, principal_realm = qualified.rsplit("@", 1)
|
||||||
|
# A UPN suffix belongs to AD, including alternate DNS suffixes. DOMAIN
|
||||||
|
# names the domain controller; WORKGROUP is only for DOMAIN\user logins.
|
||||||
|
if not principal_realm or principal_realm.casefold() == workgroup.casefold() or any(char.isspace() for char in principal_realm):
|
||||||
|
return None
|
||||||
|
principal = qualified
|
||||||
|
if not account or any(char in account for char in "@\\/") or is_excluded_user(account):
|
||||||
return None
|
return None
|
||||||
canonical_name = f"{workgroup}\\{account}"
|
canonical_name = f"{workgroup}\\{account}"
|
||||||
principal = f"{account}@{realm}"
|
|
||||||
|
|
||||||
cache_fd = -1
|
cache_fd = -1
|
||||||
cache_path = ""
|
cache_path = ""
|
||||||
@@ -356,8 +361,9 @@ def authenticate_user(username: str, password: str) -> Optional[Dict[str, str]]:
|
|||||||
cache_fd = -1
|
cache_fd = -1
|
||||||
command_env = os.environ.copy()
|
command_env = os.environ.copy()
|
||||||
command_env["KRB5CCNAME"] = f"FILE:{cache_path}"
|
command_env["KRB5CCNAME"] = f"FILE:{cache_path}"
|
||||||
|
command_env["LC_ALL"] = "C"
|
||||||
auth_result = subprocess.run(
|
auth_result = subprocess.run(
|
||||||
["kinit", principal],
|
["kinit", *(["-C", "-E"] if is_upn else []), "--", principal],
|
||||||
input=f"{password}\n",
|
input=f"{password}\n",
|
||||||
capture_output=True,
|
capture_output=True,
|
||||||
text=True,
|
text=True,
|
||||||
@@ -365,6 +371,20 @@ def authenticate_user(username: str, password: str) -> Optional[Dict[str, str]]:
|
|||||||
timeout=15,
|
timeout=15,
|
||||||
check=False,
|
check=False,
|
||||||
)
|
)
|
||||||
|
if auth_result.returncode != 0:
|
||||||
|
return None
|
||||||
|
if is_upn:
|
||||||
|
# Use the KDC-confirmed account, never assume the UPN prefix is its
|
||||||
|
# sAMAccountName. Keep the ticket private and remove it below.
|
||||||
|
ticket = subprocess.run(["klist", "-c", cache_path], capture_output=True, text=True,
|
||||||
|
env=command_env, timeout=15, check=False)
|
||||||
|
matched = re.search(r"^Default principal:\s*([^\s@\\/]+)@([^\s]+)\s*$", ticket.stdout, re.MULTILINE)
|
||||||
|
if ticket.returncode != 0 or not matched or matched.group(2).casefold() != realm.casefold():
|
||||||
|
return None
|
||||||
|
account = matched.group(1)
|
||||||
|
if is_excluded_user(account):
|
||||||
|
return None
|
||||||
|
canonical_name = f"{workgroup}\\{account}"
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
return None
|
return None
|
||||||
finally:
|
finally:
|
||||||
@@ -375,9 +395,6 @@ def authenticate_user(username: str, password: str) -> Optional[Dict[str, str]]:
|
|||||||
os.remove(cache_path)
|
os.remove(cache_path)
|
||||||
except OSError:
|
except OSError:
|
||||||
pass
|
pass
|
||||||
if auth_result.returncode != 0:
|
|
||||||
return None
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
sid_result = subprocess.run(
|
sid_result = subprocess.run(
|
||||||
["wbinfo", "--name-to-sid", canonical_name],
|
["wbinfo", "--name-to-sid", canonical_name],
|
||||||
|
|||||||
@@ -120,6 +120,23 @@ ensure_user "$AD_WEB_ADMIN_USER" "$AD_WEB_ADMIN_PASSWORD" Preview Administrator
|
|||||||
ensure_user report_svc "$AD_USER_PASSWORD" Report Service
|
ensure_user report_svc "$AD_USER_PASSWORD" Report Service
|
||||||
ensure_user MSOL_sync "$AD_USER_PASSWORD" Directory Sync
|
ensure_user MSOL_sync "$AD_USER_PASSWORD" Directory Sync
|
||||||
|
|
||||||
|
# Windows' post-2000 logon name can differ from the pre-2000 account name,
|
||||||
|
# including a DNS suffix that differs from the domain's Kerberos realm.
|
||||||
|
for user in dave frank; do
|
||||||
|
user_dn=$(samba-tool user show "$user" | sed -n 's/^dn: //p')
|
||||||
|
if [[ $user == dave ]]; then
|
||||||
|
user_upn="david.davis@${AD_DNS_DOMAIN}"
|
||||||
|
else
|
||||||
|
user_upn="frank.foster@people.${AD_DNS_DOMAIN}"
|
||||||
|
fi
|
||||||
|
ldbmodify -H /var/lib/samba/private/sam.ldb <<EOF
|
||||||
|
dn: ${user_dn}
|
||||||
|
changetype: modify
|
||||||
|
replace: userPrincipalName
|
||||||
|
userPrincipalName: ${user_upn}
|
||||||
|
EOF
|
||||||
|
done
|
||||||
|
|
||||||
for group in Finance_Analysts Engineering_Leads FS_Finance FS_Engineering FS_Projects; do
|
for group in Finance_Analysts Engineering_Leads FS_Finance FS_Engineering FS_Projects; do
|
||||||
ensure_group "$group"
|
ensure_group "$group"
|
||||||
done
|
done
|
||||||
|
|||||||
+16
-1
@@ -241,11 +241,20 @@ def main() -> int:
|
|||||||
)
|
)
|
||||||
check(non_admin.status == 200 and non_admin.json().get("role") == "user", "valid non-admin domain user cannot sign in")
|
check(non_admin.status == 200 and non_admin.json().get("role") == "user", "valid non-admin domain user cannot sign in")
|
||||||
user_token = non_admin.json()["token"]
|
user_token = non_admin.json()["token"]
|
||||||
for username in (f"{WORKGROUP}\\alice", f"alice@{DNS_DOMAIN}", f"alice@{WORKGROUP}"):
|
for username in (f"{WORKGROUP}\\alice", f"alice@{DNS_DOMAIN}", f"alice@{DNS_DOMAIN.upper()}"):
|
||||||
formatted = http("/api/login", method="POST", value={"username": username, "password": USER_PASSWORD})
|
formatted = http("/api/login", method="POST", value={"username": username, "password": USER_PASSWORD})
|
||||||
check(formatted.status == 200, f"qualified user login failed for {username}")
|
check(formatted.status == 200, f"qualified user login failed for {username}")
|
||||||
check(formatted.json().get("sid") == non_admin.json()["sid"] and formatted.json().get("role") == "user",
|
check(formatted.json().get("sid") == non_admin.json()["sid"] and formatted.json().get("role") == "user",
|
||||||
"login format changes identity or grants administration")
|
"login format changes identity or grants administration")
|
||||||
|
check(http("/api/login", method="POST", value={"username": f"alice@{WORKGROUP}", "password": USER_PASSWORD}).status == 401,
|
||||||
|
"UPN login accepts a NetBIOS suffix")
|
||||||
|
for account, upn in (("dave", f"david.davis@{DNS_DOMAIN}"), ("frank", f"frank.foster@people.{DNS_DOMAIN}")):
|
||||||
|
legacy = http("/api/login", method="POST", value={"username": f"{WORKGROUP}\\{account}", "password": USER_PASSWORD})
|
||||||
|
modern = http("/api/login", method="POST", value={"username": upn, "password": USER_PASSWORD})
|
||||||
|
check(legacy.status == modern.status == 200, f"UPN alias authentication failed for {upn}")
|
||||||
|
check(modern.json()["user"] == f"{WORKGROUP}\\{account}" and modern.json()["sid"] == legacy.json()["sid"],
|
||||||
|
"UPN prefix is confused with another account")
|
||||||
|
check(modern.json()["role"] == "user", "UPN alias receives unexpected admin access")
|
||||||
for endpoint in ("/api/overview", "/api/access", "/api/storage", "/api/trash", "/api/report", "/api/system"):
|
for endpoint in ("/api/overview", "/api/access", "/api/storage", "/api/trash", "/api/report", "/api/system"):
|
||||||
check(http(endpoint, token=user_token).status == 403, f"non-admin can read {endpoint}")
|
check(http(endpoint, token=user_token).status == 403, f"non-admin can read {endpoint}")
|
||||||
for endpoint in ("/api/access", "/api/actions/backup", "/api/actions/reconciliation", "/api/trash/restore"):
|
for endpoint in ("/api/access", "/api/actions/backup", "/api/actions/reconciliation", "/api/trash/restore"):
|
||||||
@@ -338,6 +347,12 @@ def main() -> int:
|
|||||||
for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'):
|
for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'):
|
||||||
result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json()
|
result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json()
|
||||||
check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text")
|
check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text")
|
||||||
|
for scope in ('name','all','content'):
|
||||||
|
for query in ('Thumbs','Locked','IGNOREDARTIFACT742'):
|
||||||
|
check(http(query_path("/api/documents", {"q":query,"scope":scope}), token=user_token).json()["total"] == 0,
|
||||||
|
"cache or Office lock file appears in document search")
|
||||||
|
catalog_artifacts=engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT count(*) FROM documents WHERE lower(name)='thumbs.db' OR name GLOB '~$*'\").fetchone()[0])").stdout.strip()
|
||||||
|
check(catalog_artifacts == '0', "cache or Office lock file reaches the document queue")
|
||||||
# Guessing an indexed ID is insufficient to get another user's content.
|
# Guessing an indexed ID is insufficient to get another user's content.
|
||||||
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip()
|
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip()
|
||||||
for suffix in ("", "/preview", "/download", "/content"):
|
for suffix in ("", "/preview", "/download", "/content"):
|
||||||
|
|||||||
@@ -50,6 +50,39 @@ font=next(Path('/usr/share/fonts').rglob('NimbusSans-Regular.otf'))
|
|||||||
draw=ImageDraw.Draw(image)
|
draw=ImageDraw.Draw(image)
|
||||||
draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40)
|
draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40)
|
||||||
image.save(root/'scanned-invoice.pdf','PDF',resolution=150)
|
image.save(root/'scanned-invoice.pdf','PDF',resolution=150)
|
||||||
|
|
||||||
|
# Large portrait/landscape images and a small image for responsive preview checks.
|
||||||
|
# Generate them locally so the preview needs no external downloads or image assets.
|
||||||
|
def preview_image(path, width, height):
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
image = Image.new('RGB', (width, height), '#dceaf2')
|
||||||
|
draw = ImageDraw.Draw(image)
|
||||||
|
unit = min(width, height)
|
||||||
|
sun = unit // 9
|
||||||
|
center = (width * 3 // 4, height // 4)
|
||||||
|
draw.ellipse((center[0] - sun, center[1] - sun,
|
||||||
|
center[0] + sun, center[1] + sun), fill='#e7bb57')
|
||||||
|
draw.polygon([(0, height * 3 // 4), (width // 3, height // 3),
|
||||||
|
(width * 3 // 4, height * 4 // 5), (width, height // 2),
|
||||||
|
(width, height), (0, height)], fill='#7e9a82')
|
||||||
|
draw.polygon([(0, height * 9 // 10), (width // 2, height * 3 // 5),
|
||||||
|
(width, height * 4 // 5), (width, height), (0, height)], fill='#59796a')
|
||||||
|
margin = unit // 20
|
||||||
|
label_font = ImageFont.truetype(str(font), max(12, unit // 24))
|
||||||
|
draw.text((margin, margin), f'{width} × {height} px', fill='#304a60', font=label_font)
|
||||||
|
draw.text((margin, height - margin - unit // 20), 'IMAGEPREVIEW742', fill='white', font=label_font)
|
||||||
|
draw.rectangle((0, 0, width - 1, height - 1), outline='#304a60', width=max(2, unit // 100))
|
||||||
|
image.save(path)
|
||||||
|
|
||||||
|
photos = Path('/data/groups/data/Finance/Photos')
|
||||||
|
preview_image(photos/'portrait-4000x6000.jpg', 4000, 6000)
|
||||||
|
preview_image(photos/'landscape-6000x4000.png', 6000, 4000)
|
||||||
|
preview_image(photos/'small-320x240.png', 320, 240)
|
||||||
|
preview_image(Path('/data/private/alice/private-portrait.jpg'), 2400, 3600)
|
||||||
|
|
||||||
|
for folder in (root,Path('/data/private/alice')):
|
||||||
|
for name in ('Thumbs.db','THUMBS.DB','~$Locked.docx','~$Locked.pptx'):
|
||||||
|
(folder/name).write_text('IGNOREDARTIFACT742')
|
||||||
(root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed')
|
(root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed')
|
||||||
for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]:
|
for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]:
|
||||||
with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet:
|
with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet:
|
||||||
|
|||||||
@@ -45,7 +45,34 @@ try{
|
|||||||
await page.setViewportSize({width:390,height:844});
|
await page.setViewportSize({width:390,height:844});
|
||||||
assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844});
|
assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844});
|
||||||
await page.screenshot({path:out+'/02c-live-mobile-pdf-viewer.png',fullPage:false});
|
await page.screenshot({path:out+'/02c-live-mobile-pdf-viewer.png',fullPage:false});
|
||||||
await page.locator('#document-close').click();await page.locator('#reset-filters').click();
|
await page.locator('#document-close').click();
|
||||||
|
// Real seeded JPEG/PNG files exercise the image parser and responsive dialog together.
|
||||||
|
for(const name of ['portrait-4000x6000.jpg','landscape-6000x4000.png','small-320x240.png','private-portrait.jpg']){
|
||||||
|
await page.setViewportSize({width:1500,height:1050});
|
||||||
|
await page.locator('#search-query').fill(name);
|
||||||
|
const card=page.locator('.document-card').filter({hasText:name});
|
||||||
|
await card.waitFor();await card.locator('[data-document]').click();
|
||||||
|
await page.locator('#document-preview img').waitFor();
|
||||||
|
await page.waitForFunction(()=>{const img=document.querySelector('#document-preview img');return img?.complete&&img.naturalWidth>0;});
|
||||||
|
for(const [device,viewport] of [['desktop',{width:1500,height:1050}],['mobile',{width:390,height:844}]]){
|
||||||
|
await page.setViewportSize(viewport);
|
||||||
|
const metrics=await page.evaluate(()=>{
|
||||||
|
const dialog=document.querySelector('#document-dialog'),preview=document.querySelector('#document-preview'),img=preview.querySelector('img');
|
||||||
|
const bounds=img.getBoundingClientRect(),area=preview.getBoundingClientRect();
|
||||||
|
return {ratio:bounds.width/bounds.height,naturalRatio:img.naturalWidth/img.naturalHeight,
|
||||||
|
fits:bounds.width>0&&bounds.height>0&&bounds.left>=area.left&&bounds.top>=area.top&&bounds.right<=area.right+1&&bounds.bottom<=area.bottom+1,
|
||||||
|
scrolls:[dialog,preview].some(element=>element.scrollHeight>element.clientHeight+1||element.scrollWidth>element.clientWidth+1)};
|
||||||
|
});
|
||||||
|
assert.ok(metrics.fits,`${name} must fit the ${device} preview`);
|
||||||
|
assert.ok(Math.abs(metrics.ratio-metrics.naturalRatio)<0.01,'Image must preserve its aspect ratio');
|
||||||
|
assert.equal(metrics.scrolls,false,'Image dialog must not require scrolling');
|
||||||
|
await page.screenshot({path:`${out}/02d-image-${name}-${device}.png`,fullPage:false});
|
||||||
|
}
|
||||||
|
await page.locator('#document-close').click();
|
||||||
|
}
|
||||||
|
const imageText=await page.evaluate(async()=>{const response=await fetch('/api/documents?q=IMAGEPREVIEW742&scope=content');return response.json();});
|
||||||
|
assert.equal(imageText.total,0,'Image fixtures must not be OCRed or content-indexed');
|
||||||
|
await page.locator('#reset-filters').click();
|
||||||
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length>1);
|
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length>1);
|
||||||
assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false);
|
assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false);
|
||||||
await page.screenshot({path:out+'/03-live-mobile-files.png',fullPage:true});
|
await page.screenshot({path:out+'/03-live-mobile-files.png',fullPage:true});
|
||||||
@@ -74,6 +101,6 @@ try{
|
|||||||
await admin.goto(origin+'/');await admin.locator('.document-card').first().waitFor();
|
await admin.goto(origin+'/');await admin.locator('.document-card').first().waitFor();
|
||||||
assert.equal(await admin.locator('#admin-link').isVisible(),true);
|
assert.equal(await admin.locator('#admin-link').isVisible(),true);
|
||||||
assert.deepEqual(errors,[]);
|
assert.deepEqual(errors,[]);
|
||||||
console.log('PASS: live user/admin login, OCR search/preview, original PDF download, mobile layout, /admin boundary, index monitoring and pause/resume.');
|
console.log('PASS: live user/admin login, OCR search/preview, original PDF download, shared/private JPEG/PNG previews, mobile layout, /admin boundary, index monitoring and pause/resume.');
|
||||||
console.log('Screenshots: '+out);
|
console.log('Screenshots: '+out);
|
||||||
}finally{await browser.close();}
|
}finally{await browser.close();}
|
||||||
|
|||||||
@@ -57,7 +57,12 @@ try {
|
|||||||
const parts=url.pathname.split('/'),row=fixtures.find(row=>row.id===parts[3]);
|
const parts=url.pathname.split('/'),row=fixtures.find(row=>row.id===parts[3]);
|
||||||
if(!row||row.id===deniedId){status=404;body={error:'Datei nicht verfügbar'};}
|
if(!row||row.id===deniedId){status=404;body={error:'Datei nicht verfügbar'};}
|
||||||
else if(parts[4]==='preview'){
|
else if(parts[4]==='preview'){
|
||||||
|
if(row.previewWidth){
|
||||||
|
const width=row.previewWidth,height=row.previewHeight;
|
||||||
|
type='image/svg+xml';body=Buffer.from(`<svg xmlns="http://www.w3.org/2000/svg" width="${width}" height="${height}" viewBox="0 0 400 700" preserveAspectRatio="none"><rect width="400" height="700" fill="#dceaf2"/><circle cx="300" cy="120" r="45" fill="#e7bb57"/><path d="M0 450L140 200L310 480L400 370V700H0Z" fill="#7e9a82"/><path d="M0 600L200 400L400 580V700H0Z" fill="#59796a"/><path d="M0 0H400V700H0Z" fill="none" stroke="#304a60" stroke-width="12"/><text x="20" y="40" font-family="Arial" font-size="20">Oben · ${width} × ${height}</text><text x="20" y="675" font-family="Arial" font-size="20" fill="white">Unten · vollständiges Bild</text></svg>`);
|
||||||
|
}else{
|
||||||
type='image/svg+xml';body=Buffer.from(`<svg xmlns="http://www.w3.org/2000/svg" width="700" height="940" viewBox="0 0 700 940"><rect width="700" height="940" fill="white"/><text x="70" y="100" font-family="Arial" font-size="30">RECHNUNG</text><text x="70" y="160" font-family="Arial" font-size="18">Oktober 2026 · FINANZ742</text><path d="M70 200h560M70 500h560" stroke="#aaa"/><text x="70" y="275" font-family="Arial" font-size="20">Büroausstattung</text><text x="70" y="350" font-family="Arial" font-size="18">3 × Bildschirm</text><text x="480" y="350" font-family="Arial" font-size="18">1.500,00 EUR</text><text x="70" y="560" font-family="Arial" font-size="20">Gesamtbetrag: 1.500,00 EUR</text></svg>`);
|
type='image/svg+xml';body=Buffer.from(`<svg xmlns="http://www.w3.org/2000/svg" width="700" height="940" viewBox="0 0 700 940"><rect width="700" height="940" fill="white"/><text x="70" y="100" font-family="Arial" font-size="30">RECHNUNG</text><text x="70" y="160" font-family="Arial" font-size="18">Oktober 2026 · FINANZ742</text><path d="M70 200h560M70 500h560" stroke="#aaa"/><text x="70" y="275" font-family="Arial" font-size="20">Büroausstattung</text><text x="70" y="350" font-family="Arial" font-size="18">3 × Bildschirm</text><text x="480" y="350" font-family="Arial" font-size="18">1.500,00 EUR</text><text x="70" y="560" font-family="Arial" font-size="20">Gesamtbetrag: 1.500,00 EUR</text></svg>`);
|
||||||
|
}
|
||||||
}else if(parts[4]==='content'){type='application/pdf';body=pdf;}else body=row;
|
}else if(parts[4]==='content'){type='application/pdf';body=pdf;}else body=row;
|
||||||
}else if(url.pathname.startsWith('/assets/vendor/pdfjs/6.3.289-app1/')){
|
}else if(url.pathname.startsWith('/assets/vendor/pdfjs/6.3.289-app1/')){
|
||||||
const relative=url.pathname.slice('/assets/vendor/pdfjs/6.3.289-app1/'.length);
|
const relative=url.pathname.slice('/assets/vendor/pdfjs/6.3.289-app1/'.length);
|
||||||
@@ -78,10 +83,13 @@ try {
|
|||||||
assert.equal(await page.locator('nav a[href^="/admin"]').count(),0);
|
assert.equal(await page.locator('nav a[href^="/admin"]').count(),0);
|
||||||
assert.equal(await page.locator('#source-filter').textContent().then(text=>text.includes('Mein Private-Ordner')),true);
|
assert.equal(await page.locator('#source-filter').textContent().then(text=>text.includes('Mein Private-Ordner')),true);
|
||||||
assert.match(await page.locator('#index-progress').innerText(),/1 Datei/);
|
assert.match(await page.locator('#index-progress').innerText(),/1 Datei/);
|
||||||
|
assert.equal(await page.locator('.document-excerpt').count(),0);
|
||||||
|
assert.equal(await page.locator('#documents').innerText().then(text=>text.includes('Referenz FINANZ742')||text.includes('Projektplan mit Schulung')),false);
|
||||||
await shot('01-portal-desktop');
|
await shot('01-portal-desktop');
|
||||||
await page.locator('#search-query').fill('FINANZ742');
|
await page.locator('#search-query').fill('FINANZ742');
|
||||||
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1);
|
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1);
|
||||||
assert.match(await page.locator('.document-card').innerText(),/Rechnung Oktober/);
|
assert.match(await page.locator('.document-card').innerText(),/Rechnung Oktober/);
|
||||||
|
assert.equal(await page.locator('.document-card').innerText().then(text=>text.includes('FINANZ742')),false);
|
||||||
await page.locator('#search-scope').selectOption('content');
|
await page.locator('#search-scope').selectOption('content');
|
||||||
await page.locator('[data-document]').click();
|
await page.locator('[data-document]').click();
|
||||||
await page.locator('#document-dialog').waitFor({state:'visible'});
|
await page.locator('#document-dialog').waitFor({state:'visible'});
|
||||||
@@ -101,11 +109,13 @@ try {
|
|||||||
assert.equal(await page.locator('#document-download').getAttribute('href'),'/api/documents/'+fixtures[0].id+'/download');
|
assert.equal(await page.locator('#document-download').getAttribute('href'),'/api/documents/'+fixtures[0].id+'/download');
|
||||||
await shot('02-portal-document-preview');
|
await shot('02-portal-document-preview');
|
||||||
await page.locator('#document-close').click();
|
await page.locator('#document-close').click();
|
||||||
|
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
await page.locator('#reset-filters').click();await all();
|
await page.locator('#reset-filters').click();await all();
|
||||||
await page.locator(`[data-document="${fixtures[1].id}"]`).click();
|
await page.locator(`[data-document="${fixtures[1].id}"]`).click();
|
||||||
await page.locator('#document-dialog').waitFor({state:'visible'});
|
await page.locator('#document-dialog').waitFor({state:'visible'});
|
||||||
assert.equal(await page.locator('.document-text, #document-content, #document-pdf-viewer').count(),0);
|
assert.equal(await page.locator('.document-text, #document-content, #document-pdf-viewer').count(),0);
|
||||||
assert.equal(await page.locator('#document-dialog').innerText().then(text=>text.includes('Projektplan mit Schulung')),false);
|
assert.equal(await page.locator('#document-dialog').innerText().then(text=>text.includes('Projektplan mit Schulung')),false);
|
||||||
|
assert.equal(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
assert.ok((await page.locator('#document-dialog').boundingBox()).height<400,'Files without a visual preview should have a compact dialog');
|
assert.ok((await page.locator('#document-dialog').boundingBox()).height<400,'Files without a visual preview should have a compact dialog');
|
||||||
await shot('02b-office-without-document-text');
|
await shot('02b-office-without-document-text');
|
||||||
for(const extension of ['md','odt']) {
|
for(const extension of ['md','odt']) {
|
||||||
@@ -117,12 +127,46 @@ try {
|
|||||||
}
|
}
|
||||||
fixtures[1].extension='docx';
|
fixtures[1].extension='docx';
|
||||||
await page.locator('#document-close').click();
|
await page.locator('#document-close').click();
|
||||||
|
// Large images must fit in both dimensions, including after resizing an open dialog.
|
||||||
|
const originalOffice={...fixtures[1]};
|
||||||
|
for(const [orientation,width,height] of [['portrait',4000,7000],['landscape',7000,4000],['small',180,240]]){
|
||||||
|
Object.assign(fixtures[1],{extension:'png',name:'Bildvorschau.png',path:'Dokumente/Ein sehr langer Ordnername für die Prüfung mehrzeiliger Metadaten/Bildvorschau.png',hasPreview:true,previewWidth:width,previewHeight:height,version:width.toString(16).padStart(16,'0')});
|
||||||
|
await page.setViewportSize({width:1500,height:1050});
|
||||||
|
await page.locator(`[data-document="${fixtures[1].id}"]`).click();
|
||||||
|
await page.locator('#document-preview img').waitFor();
|
||||||
|
await page.waitForFunction(()=>{const img=document.querySelector('#document-preview img');return img?.complete&&img.naturalWidth>0;});
|
||||||
|
for(const [device,viewport] of [['desktop',{width:1500,height:1050}],['mobile',{width:390,height:844}],['mobile-landscape',{width:844,height:390}]]){
|
||||||
|
await page.setViewportSize(viewport);
|
||||||
|
const metrics=await page.evaluate(()=>{
|
||||||
|
const dialog=document.querySelector('#document-dialog'),preview=document.querySelector('#document-preview'),img=preview.querySelector('img');
|
||||||
|
const bounds=element=>{const {x,y,width,height}=element.getBoundingClientRect();return {x,y,width,height};};
|
||||||
|
return {dialog:bounds(dialog),preview:bounds(preview),image:bounds(img),naturalWidth:img.naturalWidth,naturalHeight:img.naturalHeight,scrolls:[dialog,preview].map(element=>({x:element.scrollWidth-element.clientWidth,y:element.scrollHeight-element.clientHeight}))};
|
||||||
|
});
|
||||||
|
assert.deepEqual([metrics.naturalWidth,metrics.naturalHeight],[width,height]);
|
||||||
|
for(const scroll of metrics.scrolls){assert.ok(scroll.x<=1,'Image preview must not scroll horizontally');assert.ok(scroll.y<=1,'Image preview must not scroll vertically');}
|
||||||
|
assert.ok(Math.abs(metrics.image.width/metrics.image.height-width/height)<0.01,'Image aspect ratio must be preserved');
|
||||||
|
assert.ok(metrics.image.width<=width&&metrics.image.height<=height,'Small images must not be enlarged');
|
||||||
|
assert.ok(metrics.image.width>0&&metrics.image.height>0,'Image must remain visible');
|
||||||
|
assert.ok(metrics.image.x>=metrics.preview.x&&metrics.image.y>=metrics.preview.y);
|
||||||
|
assert.ok(metrics.image.x+metrics.image.width<=metrics.preview.x+metrics.preview.width+1);
|
||||||
|
assert.ok(metrics.image.y+metrics.image.height<=metrics.preview.y+metrics.preview.height+1);
|
||||||
|
assert.ok(metrics.dialog.y>=0&&metrics.dialog.y+metrics.dialog.height<=viewport.height+1,'Dialog must fit the viewport');
|
||||||
|
assert.equal(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
|
await shot(`06-image-${orientation}-${device}`);
|
||||||
|
}
|
||||||
|
await page.locator('#document-close').click();
|
||||||
|
}
|
||||||
|
delete fixtures[1].previewWidth;delete fixtures[1].previewHeight;
|
||||||
|
Object.assign(fixtures[1],originalOffice);
|
||||||
|
await page.setViewportSize({width:1500,height:1050});
|
||||||
await page.locator('#source-filter').selectOption('private:alice');
|
await page.locator('#source-filter').selectOption('private:alice');
|
||||||
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1);
|
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1);
|
||||||
assert.match(await page.locator('.document-card').innerText(),/Meine Notizen/);
|
assert.match(await page.locator('.document-card').innerText(),/Meine Notizen/);
|
||||||
await page.locator('#reset-filters').click();await all();
|
await page.locator('#reset-filters').click();await all();
|
||||||
await page.locator('#view-list').click();
|
await page.locator('#view-list').click();
|
||||||
assert.equal(await page.locator('#documents').evaluate(element=>element.classList.contains('document-list')),true);
|
assert.equal(await page.locator('#documents').evaluate(element=>element.classList.contains('document-list')),true);
|
||||||
|
assert.equal(await page.locator('.document-excerpt').count(),0);
|
||||||
|
assert.equal(await page.locator('#documents').innerText().then(text=>text.includes('Referenz FINANZ742')||text.includes('Projektplan mit Schulung')),false);
|
||||||
await shot('03-portal-list');
|
await shot('03-portal-list');
|
||||||
// A background OCR completion becomes searchable without reloading.
|
// A background OCR completion becomes searchable without reloading.
|
||||||
fixtures[4].text='Erkanntes Protokoll mit dem Suchwort SCANLIVE742';fixtures[4].snippet=fixtures[4].text;fixtures[4].state='ready';
|
fixtures[4].text='Erkanntes Protokoll mit dem Suchwort SCANLIVE742';fixtures[4].snippet=fixtures[4].text;fixtures[4].state='ready';
|
||||||
@@ -134,6 +178,7 @@ try {
|
|||||||
deniedId=fixtures[4].id;
|
deniedId=fixtures[4].id;
|
||||||
await page.locator('#document-dialog').waitFor({state:'hidden',timeout:6000});
|
await page.locator('#document-dialog').waitFor({state:'hidden',timeout:6000});
|
||||||
assert.match(await page.locator('#toast').innerText(),/nicht verfügbar/);
|
assert.match(await page.locator('#toast').innerText(),/nicht verfügbar/);
|
||||||
|
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
deniedId=''; await page.locator('#reset-filters').click();await all();
|
deniedId=''; await page.locator('#reset-filters').click();await all();
|
||||||
await page.locator('#view-grid').click();
|
await page.locator('#view-grid').click();
|
||||||
for(const width of [900,390]){
|
for(const width of [900,390]){
|
||||||
@@ -141,12 +186,31 @@ try {
|
|||||||
assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false,'Portal overflows at '+width);
|
assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false,'Portal overflows at '+width);
|
||||||
await shot('04-portal-'+width);
|
await shot('04-portal-'+width);
|
||||||
}
|
}
|
||||||
|
await page.evaluate(()=>window.scrollTo(0,300));
|
||||||
await page.locator('[data-document]').first().click();
|
await page.locator('[data-document]').first().click();
|
||||||
await page.locator('#document-dialog').waitFor({state:'visible'});
|
await page.locator('#document-dialog').waitFor({state:'visible'});
|
||||||
assert.equal(await page.evaluate(()=>document.querySelector('#document-dialog').scrollWidth>document.querySelector('#document-dialog').clientWidth),false);
|
assert.equal(await page.evaluate(()=>document.querySelector('#document-dialog').scrollWidth>document.querySelector('#document-dialog').clientWidth),false);
|
||||||
await page.frameLocator('#document-pdf-viewer').locator('.page canvas').first().waitFor();
|
await page.frameLocator('#document-pdf-viewer').locator('.page canvas').first().waitFor();
|
||||||
assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844});
|
assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844});
|
||||||
|
const backgroundScroll=await page.evaluate(()=>scrollY);
|
||||||
|
assert.ok(backgroundScroll>0,'Use a genuinely scrolled background for the modal test');
|
||||||
|
assert.equal(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
|
assert.equal(await page.evaluate(()=>getComputedStyle(document.body).overflowY),'hidden');
|
||||||
|
await page.mouse.move(40,20);await page.mouse.wheel(0,600);await page.waitForTimeout(200);
|
||||||
|
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Header wheel must not scroll the background');
|
||||||
|
const pdfScroll=await page.evaluate(()=>document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer').scrollTop);
|
||||||
|
await page.mouse.move(200,300);await page.mouse.wheel(0,600);
|
||||||
|
await page.waitForFunction(before=>document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer').scrollTop>before,pdfScroll);
|
||||||
|
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'PDF scroll must leave the background stationary');
|
||||||
|
await page.evaluate(()=>{const viewer=document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer');viewer.scrollTop=viewer.scrollHeight;});
|
||||||
|
await page.mouse.wheel(0,600);await page.waitForTimeout(200);
|
||||||
|
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Scrolling past the PDF end must not scroll the background');
|
||||||
await shot('05-portal-mobile-document');await page.locator('#document-close').click();
|
await shot('05-portal-mobile-document');await page.locator('#document-close').click();
|
||||||
|
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
|
||||||
|
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Closing must preserve the original scroll position');
|
||||||
|
await page.mouse.move(200,300);await page.mouse.wheel(0,300);
|
||||||
|
await page.waitForFunction(before=>scrollY>before,backgroundScroll);
|
||||||
|
await page.evaluate(()=>window.scrollTo(0,0));
|
||||||
await page.locator('#logout').click();await page.locator('#login-view').waitFor({state:'visible'});
|
await page.locator('#logout').click();await page.locator('#login-view').waitFor({state:'visible'});
|
||||||
await page.locator('#login-form input[name=username]').fill('alice');
|
await page.locator('#login-form input[name=username]').fill('alice');
|
||||||
await page.locator('#login-form input[name=password]').fill('password');
|
await page.locator('#login-form input[name=password]').fill('password');
|
||||||
@@ -157,6 +221,6 @@ try {
|
|||||||
assert.equal(await page.locator('#admin-link').isVisible(),true);
|
assert.equal(await page.locator('#admin-link').isVisible(),true);
|
||||||
assert.ok(pendingRequests.includes('FINANZ742')&&pendingRequests.includes('SCANLIVE742'));
|
assert.ok(pendingRequests.includes('FINANZ742')&&pendingRequests.includes('SCANLIVE742'));
|
||||||
assert.deepEqual(errors,[]);
|
assert.deepEqual(errors,[]);
|
||||||
console.log('PASS: user login, admin separation, live content search, own-private filter, preview/download links, OCR refresh, revocation, grid/list, desktop/mobile.');
|
console.log('PASS: user login, admin separation, live content search, own-private filter, preview/download links, OCR refresh, revocation, grid/list, desktop/mobile, portrait/landscape image fitting.');
|
||||||
console.log('Screenshots: '+out);
|
console.log('Screenshots: '+out);
|
||||||
} finally {await browser.close();}
|
} finally {await browser.close();}
|
||||||
|
|||||||
@@ -105,6 +105,102 @@ class DocumentTests(DocumentFixture):
|
|||||||
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
|
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
|
||||||
self.find({'q':[query]})
|
self.find({'q':[query]})
|
||||||
|
|
||||||
|
def test_ocr_candidates_match_images_to_their_page_and_ignore_native_text_and_masks(self):
|
||||||
|
body='Short cover\f'+'Native text '*30+'\fTiny footer\fNo text\f'
|
||||||
|
images='page num type width height\n2 0 image 300 300\n3 1 image 500 500\n4 2 smask 500 500\n'
|
||||||
|
self.assertEqual(extract_document.ocr_candidates(body,images,4),[3])
|
||||||
|
self.assertEqual(extract_document.ocr_candidates(body,images,2),[])
|
||||||
|
self.assertEqual(extract_document.ocr_candidates('Header\fNative text '+('word '*30)+'\f','2 0 image 100 100',2),[])
|
||||||
|
|
||||||
|
def test_search_ocr_limits_render_size_and_time_and_does_not_rebuild_the_pdf(self):
|
||||||
|
calls=[]
|
||||||
|
def parser(args,timeout=120):
|
||||||
|
calls.append((args,timeout))
|
||||||
|
return ('Page 3 size: 595 x 842 pts\nPage 7 size: 144 x 288 pts' if args[0]=='pdfinfo' else
|
||||||
|
'Recognized page text' if args[0]=='tesseract' else '')
|
||||||
|
config={'language':'deu+eng','ocr_dpi':300,'ocr_max_dimension':3500,'ocr_page_timeout':30}
|
||||||
|
with mock.patch.object(extract_document,'command',side_effect=parser):
|
||||||
|
body=extract_document.ocr_search_text(Path('input'),[3,7],config)
|
||||||
|
self.assertEqual(body,'Recognized page text\nRecognized page text')
|
||||||
|
self.assertEqual([args[0] for args,_ in calls],['pdfinfo','pdftoppm','tesseract','pdftoppm','tesseract'])
|
||||||
|
for args,timeout in calls[1:]:
|
||||||
|
self.assertEqual(timeout,30)
|
||||||
|
if args[0]=='pdftoppm':
|
||||||
|
self.assertLessEqual(int(args[args.index('-scale-to')+1]),3500)
|
||||||
|
self.assertEqual(args[args.index('-r')+1],'300')
|
||||||
|
self.assertEqual(args[args.index('-f')+1],args[args.index('-l')+1])
|
||||||
|
self.assertEqual(calls[1][0][calls[1][0].index('-f')+1],'3')
|
||||||
|
self.assertEqual(calls[3][0][calls[3][0].index('-f')+1],'7')
|
||||||
|
self.assertEqual(calls[1][0][calls[1][0].index('-scale-to')+1],'3500')
|
||||||
|
self.assertEqual(calls[3][0][calls[3][0].index('-scale-to')+1],'1200')
|
||||||
|
|
||||||
|
def artifact_files(self):
|
||||||
|
files=[]
|
||||||
|
for root,name in ((self.data/'Finance','Thumbs.db'),(self.data/'Finance'/'Child','THUMBS.DB'),
|
||||||
|
(self.data/'Finance','~$report.docx'),(self.data/'Finance'/'Child','~$Slides.PPTX'),
|
||||||
|
(self.private/'alice','thumbs.Db'),(self.private/'alice','~$notes.odt')):
|
||||||
|
path=root/name
|
||||||
|
self.write(path,'IGNOREDARTIFACT742')
|
||||||
|
files.append((path,path.read_bytes(),path.stat().st_ino))
|
||||||
|
return files
|
||||||
|
|
||||||
|
def seed_old_artifact_entries(self):
|
||||||
|
files=self.artifact_files()
|
||||||
|
with mock.patch.object(documents,'is_excluded_filename',return_value=False):
|
||||||
|
self.scan()
|
||||||
|
for path,_,_ in files:
|
||||||
|
self.conn.execute("UPDATE documents SET body='IGNOREDARTIFACT742',state='pending' WHERE name=?",(path.name,))
|
||||||
|
self.conn.commit()
|
||||||
|
return files
|
||||||
|
|
||||||
|
def test_cache_and_lock_files_are_never_cataloged_in_data_or_private(self):
|
||||||
|
files=self.artifact_files()
|
||||||
|
for name in ('Thumbs.db.docx','project-Thumbs.db','~notes.docx'):
|
||||||
|
self.write(self.data/'Finance'/name,'ordinary file')
|
||||||
|
self.scan()
|
||||||
|
for path,original,inode in files:
|
||||||
|
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents WHERE name=?',(path.name,)).fetchone()[0],0)
|
||||||
|
self.assertEqual(path.read_bytes(),original)
|
||||||
|
self.assertEqual(path.stat().st_ino,inode)
|
||||||
|
for name in ('Thumbs.db.docx','project-Thumbs.db','~notes.docx'):
|
||||||
|
self.assertEqual(self.find({'q':[name],'scope':['name']})['total'],1)
|
||||||
|
|
||||||
|
def test_stale_artifacts_never_affect_search_counts_facets_or_direct_access(self):
|
||||||
|
self.seed_old_artifact_entries()
|
||||||
|
result=self.find({'limit':['1']})
|
||||||
|
self.assertEqual((result['total'],len(result['items']),result['hasMore'],result['types'],result['pending']),
|
||||||
|
(2,1,True,['docx','odt'],0))
|
||||||
|
for scope in ('name','all','content'):
|
||||||
|
for query in ('Thumbs','report','IGNOREDARTIFACT742'):
|
||||||
|
self.assertEqual(self.find({'q':[query],'scope':[scope]})['total'],0)
|
||||||
|
for row in self.conn.execute("SELECT id FROM documents WHERE name GLOB '~$*' OR lower(name)='thumbs.db'"):
|
||||||
|
with self.assertRaises(FileNotFoundError):
|
||||||
|
documents.detail(self.conn,IDENTITY,row['id'])
|
||||||
|
for operation in (documents.download,documents.preview):
|
||||||
|
with self.assertRaises(FileNotFoundError), operation(self.conn,IDENTITY,row['id']):
|
||||||
|
pass
|
||||||
|
with mock.patch.object(document_index,'run_job') as job:
|
||||||
|
while document_index.process_next(self.conn):
|
||||||
|
pass
|
||||||
|
job.assert_not_called()
|
||||||
|
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents').fetchone()[0],4)
|
||||||
|
|
||||||
|
def test_upgrade_removes_old_artifacts_and_fts_without_changing_originals_or_pause(self):
|
||||||
|
files=self.seed_old_artifact_entries()
|
||||||
|
documents.control_worker(self.conn,'pause','admin')
|
||||||
|
self.conn.execute('PRAGMA user_version=1')
|
||||||
|
self.conn.commit()
|
||||||
|
documents.ensure_schema(self.conn)
|
||||||
|
self.assertTrue(documents.worker_paused(self.conn))
|
||||||
|
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents').fetchone()[0],4)
|
||||||
|
self.assertEqual(self.conn.execute("SELECT count(*) FROM document_fts WHERE document_fts MATCH 'IGNOREDARTIFACT742'").fetchone()[0],0)
|
||||||
|
self.assertEqual(self.find({'q':['Umsatz'],'scope':['content']})['total'],1)
|
||||||
|
for path,original,inode in files:
|
||||||
|
self.assertEqual(path.read_bytes(),original)
|
||||||
|
self.assertEqual(path.stat().st_ino,inode)
|
||||||
|
documents.ensure_schema(self.conn)
|
||||||
|
self.assertEqual(self.conn.execute('PRAGMA user_version').fetchone()[0],2)
|
||||||
|
|
||||||
def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self):
|
def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self):
|
||||||
for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
|
for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
|
||||||
self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742')
|
self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742')
|
||||||
|
|||||||
+43
-6
@@ -297,7 +297,7 @@ class DomainAuthenticationTests(unittest.TestCase):
|
|||||||
result = web_ui.authenticate_domain_admin("alice", "p@ss word")
|
result = web_ui.authenticate_domain_admin("alice", "p@ss word")
|
||||||
|
|
||||||
self.assertEqual(result, "EXAMPLE\\alice")
|
self.assertEqual(result, "EXAMPLE\\alice")
|
||||||
self.assertEqual(run.call_args_list[0].args[0], ["kinit", "alice@EXAMPLE.COM"])
|
self.assertEqual(run.call_args_list[0].args[0], ["kinit", "--", "alice@EXAMPLE.COM"])
|
||||||
self.assertNotIn("p@ss word", run.call_args_list[0].args[0])
|
self.assertNotIn("p@ss word", run.call_args_list[0].args[0])
|
||||||
|
|
||||||
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
||||||
@@ -314,24 +314,61 @@ class DomainAuthenticationTests(unittest.TestCase):
|
|||||||
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN':'example.com','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN':'example.com','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
||||||
@mock.patch('app.web_ui.subprocess.run')
|
@mock.patch('app.web_ui.subprocess.run')
|
||||||
def test_login_formats_resolve_the_same_identity_and_role(self, run):
|
def test_login_formats_resolve_the_same_identity_and_role(self, run):
|
||||||
for name in ('alice','EXAMPLE\\alice','example\\alice','alice@example.com','alice@EXAMPLE.COM','alice@EXAMPLE'):
|
for name in ('alice','EXAMPLE\\alice','example\\alice','alice@example.com','alice@EXAMPLE.COM'):
|
||||||
with self.subTest(name=name):
|
with self.subTest(name=name):
|
||||||
run.reset_mock()
|
run.reset_mock()
|
||||||
run.side_effect = [mock.Mock(returncode=0,stdout=''),
|
responses = [mock.Mock(returncode=0,stdout=''),
|
||||||
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-1100 SID_USER (1)'),
|
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-1100 SID_USER (1)'),
|
||||||
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-513'),
|
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-513'),
|
||||||
mock.Mock(returncode=0,stdout='EXAMPLE\\alice 1'),
|
mock.Mock(returncode=0,stdout='EXAMPLE\\alice 1'),
|
||||||
mock.Mock(returncode=0,stdout='11100')]
|
mock.Mock(returncode=0,stdout='11100')]
|
||||||
|
if '@' in name:
|
||||||
|
responses.insert(1,mock.Mock(returncode=0,stdout='Default principal: alice@EXAMPLE.COM\n'))
|
||||||
|
run.side_effect = responses
|
||||||
value = web_ui.authenticate_user(name,'password')
|
value = web_ui.authenticate_user(name,'password')
|
||||||
self.assertEqual(value['sub'],'EXAMPLE\\alice')
|
self.assertEqual(value['sub'],'EXAMPLE\\alice')
|
||||||
self.assertEqual(value['role'],'user')
|
self.assertEqual(value['role'],'user')
|
||||||
self.assertEqual(run.call_args_list[0].args[0],['kinit','alice@EXAMPLE.COM'])
|
self.assertEqual(run.call_args_list[0].args[0],['kinit','-C','-E','--',name] if '@' in name else ['kinit','--','alice@EXAMPLE.COM'])
|
||||||
self.assertEqual(run.call_args_list[1].args[0],['wbinfo','--name-to-sid','EXAMPLE\\alice'])
|
self.assertEqual(run.call_args_list[2 if '@' in name else 1].args[0],['wbinfo','--name-to-sid','EXAMPLE\\alice'])
|
||||||
run.reset_mock()
|
run.reset_mock()
|
||||||
for name in ('alice@other.example','OTHER\\alice','MSOL_sync@example.com','krbtgt@example.com'):
|
for name in ('alice@EXAMPLE','OTHER\\alice','MSOL_sync@example.com','krbtgt@example.com','alice@','@example.com','alice@@example.com','EXAMPLE\\alice@example.com'):
|
||||||
self.assertIsNone(web_ui.authenticate_user(name,'password'))
|
self.assertIsNone(web_ui.authenticate_user(name,'password'))
|
||||||
run.assert_not_called()
|
run.assert_not_called()
|
||||||
|
|
||||||
|
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN':'dc.example.com','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
||||||
|
@mock.patch('app.web_ui.subprocess.run')
|
||||||
|
def test_upn_uses_kdc_account_with_different_prefix_and_alternate_dns_suffix(self, run):
|
||||||
|
for upn in ('alice.smith@example.com','alice.smith@people.example.net'):
|
||||||
|
with self.subTest(upn=upn):
|
||||||
|
run.reset_mock()
|
||||||
|
run.side_effect = [mock.Mock(returncode=0,stdout=''),
|
||||||
|
mock.Mock(returncode=0,stdout='Default principal: alice@EXAMPLE.COM\n'),
|
||||||
|
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-1100 SID_USER (1)'),
|
||||||
|
mock.Mock(returncode=0,stdout='S-1-5-21-1-2-3-513'),
|
||||||
|
mock.Mock(returncode=0,stdout='EXAMPLE\\alice 1'),
|
||||||
|
mock.Mock(returncode=0,stdout='11100')]
|
||||||
|
result=web_ui.authenticate_user(upn,'password')
|
||||||
|
self.assertEqual(result['sub'],'EXAMPLE\\alice')
|
||||||
|
self.assertEqual(result['role'],'user')
|
||||||
|
self.assertEqual(run.call_args_list[0].args[0],['kinit','-C','-E','--',upn])
|
||||||
|
self.assertEqual(run.call_args_list[2].args[0],['wbinfo','--name-to-sid','EXAMPLE\\alice'])
|
||||||
|
self.assertEqual(run.call_args_list[1].kwargs['env']['LC_ALL'],'C')
|
||||||
|
self.assertFalse(os.path.exists(run.call_args_list[1].args[0][2]))
|
||||||
|
|
||||||
|
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM'})
|
||||||
|
@mock.patch('app.web_ui.subprocess.run')
|
||||||
|
def test_upn_rejects_failed_password_foreign_ticket_and_system_account_alias(self, run):
|
||||||
|
for code, ticket in ((1,''),(0,'Default principal: alice@OTHER.EXAMPLE\n'),
|
||||||
|
(0,'Default principal: MSOL_sync@EXAMPLE.COM\n'),
|
||||||
|
(0,'Default principal: krbtgt@EXAMPLE.COM\n'),(0,'unreadable ticket')):
|
||||||
|
with self.subTest(code=code,ticket=ticket):
|
||||||
|
run.reset_mock()
|
||||||
|
run.side_effect=[mock.Mock(returncode=code,stdout=''),mock.Mock(returncode=0,stdout=ticket)]
|
||||||
|
self.assertIsNone(web_ui.authenticate_user('alias@example.com','password'))
|
||||||
|
self.assertFalse(any(call.args[0][0]=='wbinfo' for call in run.call_args_list))
|
||||||
|
cache=run.call_args_list[0].kwargs['env']['KRB5CCNAME'].removeprefix('FILE:')
|
||||||
|
self.assertFalse(os.path.exists(cache))
|
||||||
|
|
||||||
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
@mock.patch.dict(os.environ, {'WORKGROUP':'EXAMPLE','REALM':'EXAMPLE.COM','DOMAIN_ADMINS_SID':'S-1-5-21-1-2-3-512'})
|
||||||
@mock.patch('app.web_ui.subprocess.run')
|
@mock.patch('app.web_ui.subprocess.run')
|
||||||
def test_system_accounts_rejected_even_when_resolved_from_an_alias(self, run):
|
def test_system_accounts_rejected_even_when_resolved_from_an_alias(self, run):
|
||||||
|
|||||||
Reference in New Issue
Block a user