Compare commits

..
2 Commits
Author SHA1 Message Date
Ludwig Lehnert fc05509ac7 fts (3) 2026-10-03 17:03:57 +00:00
Ludwig Lehnert 2b23422fa3 fts (2) 2026-10-03 15:43:12 +00:00
11 changed files with 220 additions and 26 deletions
+3
View File
@@ -33,6 +33,9 @@ ACME_HTTP_PORT=80
# DOCUMENT_SCAN_SECONDS=30 # DOCUMENT_SCAN_SECONDS=30
# DOCUMENT_OCR_LANGUAGE=deu+eng # DOCUMENT_OCR_LANGUAGE=deu+eng
# DOCUMENT_OCR_TIMEOUT_SECONDS=600 # DOCUMENT_OCR_TIMEOUT_SECONDS=600
# DOCUMENT_OCR_DPI=300
# DOCUMENT_OCR_MAX_DIMENSION=3500
# DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS=30
# DOCUMENT_MAX_FILE_MB=512 # DOCUMENT_MAX_FILE_MB=512
# DOCUMENT_MAX_PDF_PAGES=500 # DOCUMENT_MAX_PDF_PAGES=500
# TRASH_RETENTION_DAYS=7 # TRASH_RETENTION_DAYS=7
+6 -3
View File
@@ -71,15 +71,15 @@ The Data share uses Samba Windows ACL checks instead of POSIX ACLs. Keep its dat
## File Search and PDF Recognition ## File Search and PDF Recognition
Users sign in at `/` with their existing AD credentials using `username`, `DOMAIN\username`, or `username@domain`. Domain-qualified names accept the configured AD DNS domain/realm; the short NetBIOS domain is also accepted after `@`. The file view provides live search by filename and content, folder and file-type filters, grid/list views, previews, and downloads of the original files. Only Domain Admins see the administration link and can use management pages and APIs under `/admin/...`; old administration URLs redirect there. Administrative APIs are under `/admin/api/...`, while sign-in, session, and document endpoints remain under `/api/...`. This upgrade expires existing web sessions. Users sign in at `/` with their existing AD credentials using `username`, `DOMAIN\username`, or `username@domain`. Domain-qualified names accept the configured AD DNS domain/realm; the short NetBIOS domain is also accepted after `@`. The file view provides live search by filename and content, folder and file-type filters, grid/list views, previews, and downloads of the original files. Grid and list entries show file metadata without extracted content snippets. Only Domain Admins see the administration link and can use management pages and APIs under `/admin/...`; old administration URLs redirect there. Administrative APIs are under `/admin/api/...`, while sign-in, session, and document endpoints remain under `/api/...`. This upgrade expires existing web sessions.
The catalog includes active Data folders and each user's own Private folder. Every search, detail, preview, and download request checks current individual folder permissions. Revoking access or archiving a folder takes effect without waiting for reindexing. Domain Admins can see all active Data folders, but the user file view still shows only their own Private folder. Archived folders, recycle repositories, FSLogix profiles, symlinks, nested mounts, and special files are excluded. Private files additionally require ownership and read/traverse permissions for the signed-in user's mapped Unix identity. The catalog includes active Data folders and each user's own Private folder. Every search, detail, preview, and download request checks current individual folder permissions. Revoking access or archiving a folder takes effect without waiting for reindexing. Domain Admins can see all active Data folders, but the user file view still shows only their own Private folder. Archived folders, recycle repositories, FSLogix profiles, symlinks, nested mounts, and special files are excluded. `Thumbs.db` (case-insensitive) and Office lock files whose names start with `~$` are omitted from results, counts and file-type filters in both Data and Private folders. Upgrades remove their existing catalog entries without deleting original files. Private files additionally require ownership and read/traverse permissions for the signed-in user's mapped Unix identity.
Linux filesystem events update the filename catalog as files change, with periodic scans recovering missed events. Text extraction and OCR run through a persistent background queue; files become searchable by filename before content extraction finishes. Search updates while typing, and open views refresh every two seconds to show completed extraction. Full-text queries match word prefixes and ignore accents; filename-only queries also match substrings. Linux filesystem events update the filename catalog as files change, with periodic scans recovering missed events. Text extraction and OCR run through a persistent background queue; files become searchable by filename before content extraction finishes. Search updates while typing, and open views refresh every two seconds to show completed extraction. Full-text queries match word prefixes and ignore accents; filename-only queries also match substrings.
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content. Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
PDFs containing raster images and pages with little extracted text are queued for local OCR using OCRmyPDF and Tesseract, with German and English enabled by default. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts. Only PDF pages that themselves contain raster images and little extracted text are queued for local OCR using Poppler and Tesseract, with German and English enabled by default. Search OCR renders those pages at 300 DPI, capped at 3500 pixels on the longest side, instead of inheriting high DPI from embedded logos or rebuilding an OCR PDF. Each render and recognition step has a 30-second timeout. Photos can still require a recognition attempt to establish whether they contain text; text-free pages finish without adding search content. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities. Parsers run as an unprivileged service account on temporary copies, with filesystem/network restrictions and resource limits. This requires a Linux kernel with Landlock enabled (Linux 5.13 or newer) on x86-64 or ARM64. If isolation is unavailable, extraction fails closed while filename search remains available; the worker logs the reason. OCR needs no external service or additional container capabilities.
@@ -92,6 +92,9 @@ Domain Admins monitor the worker at `/admin/documents`: current file and phase,
| `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes | | `DOCUMENT_SCAN_SECONDS` | `30` | Recovery scan interval; filesystem events handle intervening changes |
| `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR | | `DOCUMENT_OCR_LANGUAGE` | `deu+eng` | Installed Tesseract languages used for OCR |
| `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job | | `DOCUMENT_OCR_TIMEOUT_SECONDS` | `600` | Maximum runtime for an OCR job |
| `DOCUMENT_OCR_DPI` | `300` | OCR render resolution (150–400 DPI), subject to the pixel cap |
| `DOCUMENT_OCR_MAX_DIMENSION` | `3500` | Maximum OCR image width/height (1500–5000 pixels); originals are unchanged |
| `DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS` | `30` | Maximum runtime per page render or recognition step (5–120 seconds) |
| `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search | | `DOCUMENT_MAX_FILE_MB` | `512` | Largest file copied for extraction; larger files use filename search |
| `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction | | `DOCUMENT_MAX_PDF_PAGES` | `500` | Maximum PDF page count for extraction |
+9 -1
View File
@@ -211,7 +211,10 @@ def run_job(row, source, phase, cancelled=None):
os.chown(workspace, user.pw_uid, user.pw_gid) os.chown(workspace, user.pw_uid, user.pw_gid)
config = {'extension': row['extension'], 'phase': phase, config = {'extension': row['extension'], 'phase': phase,
'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'), 'language': os.getenv('DOCUMENT_OCR_LANGUAGE', 'deu+eng'),
'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout} 'max_pages': setting('DOCUMENT_MAX_PDF_PAGES', 500, 1, 5000), 'timeout': timeout,
'ocr_dpi': setting('DOCUMENT_OCR_DPI', 300, 150, 400),
'ocr_max_dimension': setting('DOCUMENT_OCR_MAX_DIMENSION', 3500, 1500, 5000),
'ocr_page_timeout': setting('DOCUMENT_OCR_PAGE_TIMEOUT_SECONDS', 30, 5, 120)}
for name, content in [('config.json', json.dumps(config))]: for name, content in [('config.json', json.dumps(config))]:
path = os.path.join(workspace, name) path = os.path.join(workspace, name)
with open(path, 'w') as handle: with open(path, 'w') as handle:
@@ -300,6 +303,11 @@ def process_next(conn, prefer_ocr=False, stop=None):
if row is None: if row is None:
worker_state(conn, 'idle') worker_state(conn, 'idle')
return False return False
if documents.is_excluded_filename(row['name']):
conn.execute('DELETE FROM documents WHERE id=?', (row['id'],))
conn.commit()
worker_state(conn, 'idle')
return True
source = conn.execute('SELECT * FROM document_sources WHERE id=?', (row['source_id'],)).fetchone() source = conn.execute('SELECT * FROM document_sources WHERE id=?', (row['source_id'],)).fetchone()
if source is None: if source is None:
return False return False
+17 -6
View File
@@ -25,6 +25,11 @@ except ImportError:
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents')) SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db') SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
MAX_TEXT = 2 * 1024 * 1024 MAX_TEXT = 2 * 1024 * 1024
SEARCHABLE_NAME_SQL = "lower({name}) <> 'thumbs.db' AND {name} NOT GLOB '~$*'"
def is_excluded_filename(name):
return name.casefold() == 'thumbs.db' or name.startswith('~$')
@@ -83,17 +88,20 @@ def ensure_schema(conn):
# Upgrade the derived index only: keep catalog IDs, permissions, previews # Upgrade the derived index only: keep catalog IDs, permissions, previews
# and originals, but remove content and jobs for excluded file types. # and originals, but remove content and jobs for excluded file types.
if conn.execute('PRAGMA user_version').fetchone()[0] < 1: if conn.execute('PRAGMA user_version').fetchone()[0] < 2:
with conn: with conn:
conn.execute('BEGIN IMMEDIATE') conn.execute('BEGIN IMMEDIATE')
if conn.execute('PRAGMA user_version').fetchone()[0] < 1: version = conn.execute('PRAGMA user_version').fetchone()[0]
if version < 1:
content = sorted(CONTENT_SUFFIXES) content = sorted(CONTENT_SUFFIXES)
images = sorted(IMAGE_SUFFIXES) images = sorted(IMAGE_SUFFIXES)
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0, conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview='' state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
THEN 'pending' ELSE 'name-only' END THEN 'pending' ELSE 'name-only' END
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content]) WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
conn.execute('PRAGMA user_version=1') if version < 2:
conn.execute(f"DELETE FROM documents WHERE NOT ({SEARCHABLE_NAME_SQL.format(name='name')})")
conn.execute('PRAGMA user_version=2')
def worker_paused(conn): def worker_paused(conn):
@@ -243,6 +251,10 @@ def allowed_sources(conn, identity):
def discover_file(conn, source, relative, seen=0): def discover_file(conn, source, relative, seen=0):
name = relative.rsplit('/', 1)[-1]
if is_excluded_filename(name):
conn.execute('DELETE FROM documents WHERE source_id=? AND path=?', (source['id'], relative))
return
try: try:
with open_file(source['root'], relative) as (_, info): with open_file(source['root'], relative) as (_, info):
stamp = fingerprint(info) stamp = fingerprint(info)
@@ -254,7 +266,6 @@ def discover_file(conn, source, relative, seen=0):
if seen: if seen:
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative)) conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
return return
name = relative.rsplit('/', 1)[-1]
extension = os.path.splitext(name)[1].lower() extension = os.path.splitext(name)[1].lower()
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only' state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state) conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
@@ -348,7 +359,7 @@ def private_readable(source, relative, uid):
def current_document(conn, document_id, identity): def current_document(conn, document_id, identity):
sources = allowed_sources(conn, identity) sources = allowed_sources(conn, identity)
row = conn.execute('SELECT * FROM documents WHERE id=?', (document_id,)).fetchone() row = conn.execute('SELECT * FROM documents WHERE id=?', (document_id,)).fetchone()
if row is None or row['source_id'] not in sources: if row is None or row['source_id'] not in sources or is_excluded_filename(row['name']):
raise FileNotFoundError('Document not found') raise FileNotFoundError('Document not found')
source = sources[row['source_id']] source = sources[row['source_id']]
if not private_readable(source, row['path'], identity.get('uid')): if not private_readable(source, row['path'], identity.get('uid')):
@@ -397,7 +408,7 @@ def search(conn, identity, params):
for row in conn.execute(f"SELECT id,source_id,path FROM documents WHERE source_id IN ({placeholders}) AND source_id LIKE 'private:%'", ids).fetchall(): for row in conn.execute(f"SELECT id,source_id,path FROM documents WHERE source_id IN ({placeholders}) AND source_id LIKE 'private:%'", ids).fetchall():
if private_readable(selected[row['source_id']], row['path'], identity.get('uid')): if private_readable(selected[row['source_id']], row['path'], identity.get('uid')):
conn.execute('INSERT INTO readable_private VALUES(?)', (row['id'],)) conn.execute('INSERT INTO readable_private VALUES(?)', (row['id'],))
private_condition = "(d.source_id NOT LIKE 'private:%' OR EXISTS (SELECT 1 FROM readable_private r WHERE r.id=d.id))" private_condition = SEARCHABLE_NAME_SQL.format(name='d.name') + " AND (d.source_id NOT LIKE 'private:%' OR EXISTS (SELECT 1 FROM readable_private r WHERE r.id=d.id))"
conditions = [f'd.source_id IN ({placeholders})', private_condition] conditions = [f'd.source_id IN ({placeholders})', private_condition]
values = ids[:] values = ids[:]
from_sql = 'documents d' from_sql = 'documents d'
+50 -10
View File
@@ -4,6 +4,7 @@
import ctypes import ctypes
import errno import errno
import json import json
import math
import os import os
from pathlib import Path from pathlib import Path
import platform import platform
@@ -114,6 +115,47 @@ def office_text(path):
return '\n'.join(text)[:MAX_TEXT] return '\n'.join(text)[:MAX_TEXT]
def ocr_candidates(body, image_list, pages):
"""Only text-poor pages that themselves contain raster images need OCR."""
image_pages = set()
for line in image_list.splitlines():
fields = line.split()
if len(fields) >= 5 and fields[0].isdigit() and fields[2] == 'image':
image_pages.add(int(fields[0]))
return [number for number, text in enumerate(body.split('\f')[:pages], 1)
if number in image_pages and len(re.sub(r'\s+', '', text)) < 80]
def ocr_search_text(source, candidates, config):
# Search needs text only: do not rebuild a PDF or render at embedded-image
# DPI, which can enlarge an entire page because of a small high-DPI logo.
if not candidates:
return ''
sizes = command(['pdfinfo', '-f', '1', '-l', str(max(candidates)), str(source)])
dimensions = {int(number): max(float(width), float(height)) for number,width,height in
re.findall(r'Page\s+(\d+)\s+size:\s+([0-9.]+)\s+x\s+([0-9.]+)\s+pts', sizes)}
text = []
length = 0
dpi = config.get('ocr_dpi', 300)
maximum = config.get('ocr_max_dimension', 3500)
for page in candidates:
# The scale option overrides Poppler's DPI option. Compute the target
# from the physical page size so small pages are not upscaled to the cap.
scale = max(1, min(maximum, math.ceil(dimensions[page] * dpi / 72))) if page in dimensions else maximum
command(['pdftoppm', '-f', str(page), '-l', str(page), '-singlefile',
'-r', str(dpi), '-scale-to', str(scale),
'-gray', '-png', str(source), 'ocr-page'], timeout=config.get('ocr_page_timeout', 30))
fragment = command(['tesseract', 'ocr-page.png', 'stdout', '-l', config['language'],
'--dpi', str(config.get('ocr_dpi', 300))], timeout=config.get('ocr_page_timeout', 30))
fragment = fragment.strip()
if fragment:
text.append(fragment)
length += len(fragment) + 1
if length >= MAX_TEXT:
break
return '\n'.join(text)[:MAX_TEXT]
def extract(config): def extract(config):
suffix = config['extension'] suffix = config['extension']
source = Path('input') source = Path('input')
@@ -124,18 +166,16 @@ def extract(config):
pages = int(found.group(1)) if found else 0 pages = int(found.group(1)) if found else 0
if pages > config['max_pages']: if pages > config['max_pages']:
raise RuntimeError('PDF exceeds page limit') raise RuntimeError('PDF exceeds page limit')
if config['phase'] == 'ocr':
command(['ocrmypdf', '--redo-ocr', '--output-type', 'pdf', '--optimize', '0', '--jobs', '1',
'--language', config['language'], '--tesseract-timeout', '120', '--skip-big', '50',
str(source), 'ocr.pdf'], timeout=config['timeout'] - 15)
source = Path('ocr.pdf')
command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt']) command(['pdftotext', '-enc', 'UTF-8', '-layout', str(source), 'text.txt'])
body = read_text('text.txt') body = read_text('text.txt')
if config['phase'] != 'ocr': image_list = command(['pdfimages', '-list', str(source)])
image_list = command(['pdfimages', '-list', str(source)]) candidates = ocr_candidates(body, image_list, pages)
has_images = bool(re.search(r'^\s*\d+\s+\d+\s+image\s', image_list, re.MULTILINE)) if config['phase'] == 'ocr':
page_text = body.split('\f')[:pages] recognized = ocr_search_text(source, candidates, config)
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text) if recognized:
body = (body.rstrip() + '\n' + recognized)[:MAX_TEXT]
else:
needs_ocr = bool(candidates)
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview']) command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
elif suffix in OFFICE_SUFFIXES: elif suffix in OFFICE_SUFFIXES:
body = office_text(source) body = office_text(source)
+1 -3
View File
@@ -2,7 +2,6 @@
const $ = selector => document.querySelector(selector); const $ = selector => document.querySelector(selector);
const esc = value => String(value ?? "").replace(/[&<>'"]/g, char => ({"&":"&amp;","<":"&lt;",">":"&gt;","'":"&#39;",'"':"&quot;"}[char])); const esc = value => String(value ?? "").replace(/[&<>'"]/g, char => ({"&":"&amp;","<":"&lt;",">":"&gt;","'":"&#39;",'"':"&quot;"}[char]));
const snippet = value => esc(value).replace(/\x01/g, "<mark>").replace(/\x02/g, "</mark>");
const sizeLabel = value => { const sizeLabel = value => {
const units = ["B", "kB", "MB", "GB", "TB"]; const units = ["B", "kB", "MB", "GB", "TB"];
let amount = Number(value), index = 0; let amount = Number(value), index = 0;
@@ -18,7 +17,6 @@ function icon(name) {
}; };
return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[name]}</svg>`; return `<svg class="action-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false">${paths[name]}</svg>`;
} }
const statusLabel = row => ({pending:"Text wird erfasst…",ocr:row.attempts ? "Texterkennung wird erneut versucht…" : "Texterkennung läuft…",failed:"Text konnte noch nicht erfasst werden.","name-only":"Suche nach Dateinamen verfügbar."}[row.state] || "");
let session = null, offset = 0, listMode = false, timer, debounce, request, serial = 0, lastResult = "", activeDocument = null, initialFilters = null; let session = null, offset = 0, listMode = false, timer, debounce, request, serial = 0, lastResult = "", activeDocument = null, initialFilters = null;
const pageSize = 40; const pageSize = 40;
@@ -99,7 +97,7 @@ async function refresh(userAction = true) {
$("#documents").innerHTML = result.items.length ? result.items.map(row => `<article class="document-card"> $("#documents").innerHTML = result.items.length ? result.items.map(row => `<article class="document-card">
<button type="button" class="document-open" data-document="${esc(row.id)}" aria-label="Öffnen: ${esc(row.name)}"> <button type="button" class="document-open" data-document="${esc(row.id)}" aria-label="Öffnen: ${esc(row.name)}">
<div class="document-thumbnail">${row.hasPreview ? `<img src="/api/documents/${row.id}/preview?v=${row.version}" loading="lazy" alt="">` : `<div class="document-placeholder">${icon("file")}<span>${esc((row.extension || "Datei").toUpperCase())}</span></div>`}</div> <div class="document-thumbnail">${row.hasPreview ? `<img src="/api/documents/${row.id}/preview?v=${row.version}" loading="lazy" alt="">` : `<div class="document-placeholder">${icon("file")}<span>${esc((row.extension || "Datei").toUpperCase())}</span></div>`}</div>
<div class="document-description"><h2 title="${esc(row.name)}">${esc(row.name)}</h2><p class="document-path" title="${esc(row.path)}">${esc(row.path)}</p><p class="document-excerpt">${snippet(row.snippet || statusLabel(row))}</p></div> <div class="document-description"><h2 title="${esc(row.name)}">${esc(row.name)}</h2><p class="document-path" title="${esc(row.path)}">${esc(row.path)}</p></div>
</button><footer><span title="${esc(row.source)}">${esc(row.kind === "private" ? "Private" : row.source)} · ${sizeLabel(row.size)}<small>${modifiedLabel(row.modified)} UTC</small></span><a class="button icon-button" href="/api/documents/${row.id}/download" download="${esc(row.name)}" aria-label="Herunterladen: ${esc(row.name)}" title="Herunterladen">${icon("download")}</a></footer> </button><footer><span title="${esc(row.source)}">${esc(row.kind === "private" ? "Private" : row.source)} · ${sizeLabel(row.size)}<small>${modifiedLabel(row.modified)} UTC</small></span><a class="button icon-button" href="/api/documents/${row.id}/download" download="${esc(row.name)}" aria-label="Herunterladen: ${esc(row.name)}" title="Herunterladen">${icon("download")}</a></footer>
</article>`).join("") : '<div class="portal-empty"><h2>Keine Dateien gefunden</h2><p>Suchbegriff oder Filter ändern.</p></div>'; </article>`).join("") : '<div class="portal-empty"><h2>Keine Dateien gefunden</h2><p>Suchbegriff oder Filter ändern.</p></div>';
} }
+2 -3
View File
@@ -6,6 +6,7 @@
* { box-sizing: border-box; } * { box-sizing: border-box; }
[hidden] { display: none !important; } [hidden] { display: none !important; }
body { margin: 0; min-height: 100vh; } body { margin: 0; min-height: 100vh; }
html:has(#document-dialog[open]), body:has(#document-dialog[open]) { overflow: hidden; overscroll-behavior: none; }
a { color: #0645ad; } a { color: #0645ad; }
button, input, select { font: inherit; } button, input, select { font: inherit; }
button, .button { display: inline-block; padding: .4rem .7rem; border: 1px solid #777; color: #111; background: #eee; cursor: pointer; text-decoration: none; white-space: nowrap; } button, .button { display: inline-block; padding: .4rem .7rem; border: 1px solid #777; color: #111; background: #eee; cursor: pointer; text-decoration: none; white-space: nowrap; }
@@ -192,7 +193,6 @@ progress { width: 100%; }
.document-description { padding: .8rem; } .document-description { padding: .8rem; }
.document-description h2 { font-size: 1rem; margin: 0 0 .3rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .document-description h2 { font-size: 1rem; margin: 0 0 .3rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
.document-path { font-size: .8rem; color: #666; margin: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .document-path { font-size: .8rem; color: #666; margin: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
.document-excerpt { height: 3.6em; overflow: hidden; font-size: .85rem; margin: .6rem 0 0; overflow-wrap: anywhere; }
mark { background: #fff0a0; color: #111; } mark { background: #fff0a0; color: #111; }
.document-card footer { display: flex; justify-content: space-between; align-items: center; gap: .5rem; padding: .6rem .8rem; border-top: 1px solid #ccc; } .document-card footer { display: flex; justify-content: space-between; align-items: center; gap: .5rem; padding: .6rem .8rem; border-top: 1px solid #ccc; }
.document-card footer > span { min-width: 0; font-size: .8rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .document-card footer > span { min-width: 0; font-size: .8rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
@@ -203,7 +203,6 @@ mark { background: #fff0a0; color: #111; }
.document-list .document-thumbnail { width: 80px; height: 100px; flex-shrink: 0; border: 0; border-right: 1px solid #ccc; } .document-list .document-thumbnail { width: 80px; height: 100px; flex-shrink: 0; border: 0; border-right: 1px solid #ccc; }
.document-list .document-description { min-width: 0; } .document-list .document-description { min-width: 0; }
.document-list footer { flex-shrink: 0; border: 0; } .document-list footer { flex-shrink: 0; border: 0; }
.document-list .document-excerpt { height: 1.4em; }
.portal-empty { grid-column: 1/-1; padding: 2rem; border: 1px solid #aaa; text-align: center; color: #666; } .portal-empty { grid-column: 1/-1; padding: 2rem; border: 1px solid #aaa; text-align: center; color: #666; }
.portal-empty h2 { color: #111; } .portal-empty h2 { color: #111; }
.portal-pagination { display: flex; align-items: center; justify-content: center; gap: 1rem; margin-top: 1.5rem; } .portal-pagination { display: flex; align-items: center; justify-content: center; gap: 1rem; margin-top: 1.5rem; }
@@ -218,7 +217,7 @@ mark { background: #fff0a0; color: #111; }
.document-detail { height: calc(100% - 95px); min-height: 0; } .document-detail { height: calc(100% - 95px); min-height: 0; }
.document-preview { padding: 1rem; overflow: auto; background: #f5f5f5; } .document-preview { padding: 1rem; overflow: auto; background: #f5f5f5; }
.document-preview img { display: block; width: 100%; height: auto; } .document-preview img { display: block; width: 100%; height: auto; }
.document-dialog.pdf-dialog { width: 100vw; height: 100dvh; max-width: 100vw; max-height: 100dvh; margin: 0; border: 0; } .document-dialog.pdf-dialog { width: 100vw; height: 100dvh; max-width: 100vw; max-height: 100dvh; margin: 0; border: 0; overflow: hidden; }
.document-dialog.pdf-dialog[open] { display: flex; flex-direction: column; } .document-dialog.pdf-dialog[open] { display: flex; flex-direction: column; }
.pdf-dialog .document-dialog-head { flex-shrink: 0; padding: .6rem 1rem; } .pdf-dialog .document-dialog-head { flex-shrink: 0; padding: .6rem 1rem; }
.pdf-dialog .document-detail { flex: 1; height: auto; overflow: hidden; } .pdf-dialog .document-detail { flex: 1; height: auto; overflow: hidden; }
+6
View File
@@ -338,6 +338,12 @@ def main() -> int:
for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'): for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'):
result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json() result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json()
check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text") check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text")
for scope in ('name','all','content'):
for query in ('Thumbs','Locked','IGNOREDARTIFACT742'):
check(http(query_path("/api/documents", {"q":query,"scope":scope}), token=user_token).json()["total"] == 0,
"cache or Office lock file appears in document search")
catalog_artifacts=engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT count(*) FROM documents WHERE lower(name)='thumbs.db' OR name GLOB '~$*'\").fetchone()[0])").stdout.strip()
check(catalog_artifacts == '0', "cache or Office lock file reaches the document queue")
# Guessing an indexed ID is insufficient to get another user's content. # Guessing an indexed ID is insufficient to get another user's content.
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip() hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip()
for suffix in ("", "/preview", "/download", "/content"): for suffix in ("", "/preview", "/download", "/content"):
+3
View File
@@ -50,6 +50,9 @@ font=next(Path('/usr/share/fonts').rglob('NimbusSans-Regular.otf'))
draw=ImageDraw.Draw(image) draw=ImageDraw.Draw(image)
draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40) draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40)
image.save(root/'scanned-invoice.pdf','PDF',resolution=150) image.save(root/'scanned-invoice.pdf','PDF',resolution=150)
for folder in (root,Path('/data/private/alice')):
for name in ('Thumbs.db','THUMBS.DB','~$Locked.docx','~$Locked.pptx'):
(folder/name).write_text('IGNOREDARTIFACT742')
(root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed') (root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed')
for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]: for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]:
with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet: with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet:
+27
View File
@@ -78,10 +78,13 @@ try {
assert.equal(await page.locator('nav a[href^="/admin"]').count(),0); assert.equal(await page.locator('nav a[href^="/admin"]').count(),0);
assert.equal(await page.locator('#source-filter').textContent().then(text=>text.includes('Mein Private-Ordner')),true); assert.equal(await page.locator('#source-filter').textContent().then(text=>text.includes('Mein Private-Ordner')),true);
assert.match(await page.locator('#index-progress').innerText(),/1 Datei/); assert.match(await page.locator('#index-progress').innerText(),/1 Datei/);
assert.equal(await page.locator('.document-excerpt').count(),0);
assert.equal(await page.locator('#documents').innerText().then(text=>text.includes('Referenz FINANZ742')||text.includes('Projektplan mit Schulung')),false);
await shot('01-portal-desktop'); await shot('01-portal-desktop');
await page.locator('#search-query').fill('FINANZ742'); await page.locator('#search-query').fill('FINANZ742');
await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1); await page.waitForFunction(()=>document.querySelectorAll('.document-card').length===1);
assert.match(await page.locator('.document-card').innerText(),/Rechnung Oktober/); assert.match(await page.locator('.document-card').innerText(),/Rechnung Oktober/);
assert.equal(await page.locator('.document-card').innerText().then(text=>text.includes('FINANZ742')),false);
await page.locator('#search-scope').selectOption('content'); await page.locator('#search-scope').selectOption('content');
await page.locator('[data-document]').click(); await page.locator('[data-document]').click();
await page.locator('#document-dialog').waitFor({state:'visible'}); await page.locator('#document-dialog').waitFor({state:'visible'});
@@ -101,11 +104,13 @@ try {
assert.equal(await page.locator('#document-download').getAttribute('href'),'/api/documents/'+fixtures[0].id+'/download'); assert.equal(await page.locator('#document-download').getAttribute('href'),'/api/documents/'+fixtures[0].id+'/download');
await shot('02-portal-document-preview'); await shot('02-portal-document-preview');
await page.locator('#document-close').click(); await page.locator('#document-close').click();
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
await page.locator('#reset-filters').click();await all(); await page.locator('#reset-filters').click();await all();
await page.locator(`[data-document="${fixtures[1].id}"]`).click(); await page.locator(`[data-document="${fixtures[1].id}"]`).click();
await page.locator('#document-dialog').waitFor({state:'visible'}); await page.locator('#document-dialog').waitFor({state:'visible'});
assert.equal(await page.locator('.document-text, #document-content, #document-pdf-viewer').count(),0); assert.equal(await page.locator('.document-text, #document-content, #document-pdf-viewer').count(),0);
assert.equal(await page.locator('#document-dialog').innerText().then(text=>text.includes('Projektplan mit Schulung')),false); assert.equal(await page.locator('#document-dialog').innerText().then(text=>text.includes('Projektplan mit Schulung')),false);
assert.equal(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
assert.ok((await page.locator('#document-dialog').boundingBox()).height<400,'Files without a visual preview should have a compact dialog'); assert.ok((await page.locator('#document-dialog').boundingBox()).height<400,'Files without a visual preview should have a compact dialog');
await shot('02b-office-without-document-text'); await shot('02b-office-without-document-text');
for(const extension of ['md','odt']) { for(const extension of ['md','odt']) {
@@ -123,6 +128,8 @@ try {
await page.locator('#reset-filters').click();await all(); await page.locator('#reset-filters').click();await all();
await page.locator('#view-list').click(); await page.locator('#view-list').click();
assert.equal(await page.locator('#documents').evaluate(element=>element.classList.contains('document-list')),true); assert.equal(await page.locator('#documents').evaluate(element=>element.classList.contains('document-list')),true);
assert.equal(await page.locator('.document-excerpt').count(),0);
assert.equal(await page.locator('#documents').innerText().then(text=>text.includes('Referenz FINANZ742')||text.includes('Projektplan mit Schulung')),false);
await shot('03-portal-list'); await shot('03-portal-list');
// A background OCR completion becomes searchable without reloading. // A background OCR completion becomes searchable without reloading.
fixtures[4].text='Erkanntes Protokoll mit dem Suchwort SCANLIVE742';fixtures[4].snippet=fixtures[4].text;fixtures[4].state='ready'; fixtures[4].text='Erkanntes Protokoll mit dem Suchwort SCANLIVE742';fixtures[4].snippet=fixtures[4].text;fixtures[4].state='ready';
@@ -134,6 +141,7 @@ try {
deniedId=fixtures[4].id; deniedId=fixtures[4].id;
await page.locator('#document-dialog').waitFor({state:'hidden',timeout:6000}); await page.locator('#document-dialog').waitFor({state:'hidden',timeout:6000});
assert.match(await page.locator('#toast').innerText(),/nicht verfügbar/); assert.match(await page.locator('#toast').innerText(),/nicht verfügbar/);
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
deniedId=''; await page.locator('#reset-filters').click();await all(); deniedId=''; await page.locator('#reset-filters').click();await all();
await page.locator('#view-grid').click(); await page.locator('#view-grid').click();
for(const width of [900,390]){ for(const width of [900,390]){
@@ -141,12 +149,31 @@ try {
assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false,'Portal overflows at '+width); assert.equal(await page.evaluate(()=>document.documentElement.scrollWidth>innerWidth),false,'Portal overflows at '+width);
await shot('04-portal-'+width); await shot('04-portal-'+width);
} }
await page.evaluate(()=>window.scrollTo(0,300));
await page.locator('[data-document]').first().click(); await page.locator('[data-document]').first().click();
await page.locator('#document-dialog').waitFor({state:'visible'}); await page.locator('#document-dialog').waitFor({state:'visible'});
assert.equal(await page.evaluate(()=>document.querySelector('#document-dialog').scrollWidth>document.querySelector('#document-dialog').clientWidth),false); assert.equal(await page.evaluate(()=>document.querySelector('#document-dialog').scrollWidth>document.querySelector('#document-dialog').clientWidth),false);
await page.frameLocator('#document-pdf-viewer').locator('.page canvas').first().waitFor(); await page.frameLocator('#document-pdf-viewer').locator('.page canvas').first().waitFor();
assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844}); assert.deepEqual(await page.locator('#document-dialog').boundingBox(),{x:0,y:0,width:390,height:844});
const backgroundScroll=await page.evaluate(()=>scrollY);
assert.ok(backgroundScroll>0,'Use a genuinely scrolled background for the modal test');
assert.equal(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
assert.equal(await page.evaluate(()=>getComputedStyle(document.body).overflowY),'hidden');
await page.mouse.move(40,20);await page.mouse.wheel(0,600);await page.waitForTimeout(200);
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Header wheel must not scroll the background');
const pdfScroll=await page.evaluate(()=>document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer').scrollTop);
await page.mouse.move(200,300);await page.mouse.wheel(0,600);
await page.waitForFunction(before=>document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer').scrollTop>before,pdfScroll);
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'PDF scroll must leave the background stationary');
await page.evaluate(()=>{const viewer=document.querySelector('#document-pdf-viewer').contentDocument.querySelector('#viewerContainer');viewer.scrollTop=viewer.scrollHeight;});
await page.mouse.wheel(0,600);await page.waitForTimeout(200);
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Scrolling past the PDF end must not scroll the background');
await shot('05-portal-mobile-document');await page.locator('#document-close').click(); await shot('05-portal-mobile-document');await page.locator('#document-close').click();
assert.notEqual(await page.evaluate(()=>getComputedStyle(document.documentElement).overflowY),'hidden');
assert.equal(await page.evaluate(()=>scrollY),backgroundScroll,'Closing must preserve the original scroll position');
await page.mouse.move(200,300);await page.mouse.wheel(0,300);
await page.waitForFunction(before=>scrollY>before,backgroundScroll);
await page.evaluate(()=>window.scrollTo(0,0));
await page.locator('#logout').click();await page.locator('#login-view').waitFor({state:'visible'}); await page.locator('#logout').click();await page.locator('#login-view').waitFor({state:'visible'});
await page.locator('#login-form input[name=username]').fill('alice'); await page.locator('#login-form input[name=username]').fill('alice');
await page.locator('#login-form input[name=password]').fill('password'); await page.locator('#login-form input[name=password]').fill('password');
+96
View File
@@ -105,6 +105,102 @@ class DocumentTests(DocumentFixture):
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'): for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
self.find({'q':[query]}) self.find({'q':[query]})
def test_ocr_candidates_match_images_to_their_page_and_ignore_native_text_and_masks(self):
body='Short cover\f'+'Native text '*30+'\fTiny footer\fNo text\f'
images='page num type width height\n2 0 image 300 300\n3 1 image 500 500\n4 2 smask 500 500\n'
self.assertEqual(extract_document.ocr_candidates(body,images,4),[3])
self.assertEqual(extract_document.ocr_candidates(body,images,2),[])
self.assertEqual(extract_document.ocr_candidates('Header\fNative text '+('word '*30)+'\f','2 0 image 100 100',2),[])
def test_search_ocr_limits_render_size_and_time_and_does_not_rebuild_the_pdf(self):
calls=[]
def parser(args,timeout=120):
calls.append((args,timeout))
return ('Page 3 size: 595 x 842 pts\nPage 7 size: 144 x 288 pts' if args[0]=='pdfinfo' else
'Recognized page text' if args[0]=='tesseract' else '')
config={'language':'deu+eng','ocr_dpi':300,'ocr_max_dimension':3500,'ocr_page_timeout':30}
with mock.patch.object(extract_document,'command',side_effect=parser):
body=extract_document.ocr_search_text(Path('input'),[3,7],config)
self.assertEqual(body,'Recognized page text\nRecognized page text')
self.assertEqual([args[0] for args,_ in calls],['pdfinfo','pdftoppm','tesseract','pdftoppm','tesseract'])
for args,timeout in calls[1:]:
self.assertEqual(timeout,30)
if args[0]=='pdftoppm':
self.assertLessEqual(int(args[args.index('-scale-to')+1]),3500)
self.assertEqual(args[args.index('-r')+1],'300')
self.assertEqual(args[args.index('-f')+1],args[args.index('-l')+1])
self.assertEqual(calls[1][0][calls[1][0].index('-f')+1],'3')
self.assertEqual(calls[3][0][calls[3][0].index('-f')+1],'7')
self.assertEqual(calls[1][0][calls[1][0].index('-scale-to')+1],'3500')
self.assertEqual(calls[3][0][calls[3][0].index('-scale-to')+1],'1200')
def artifact_files(self):
files=[]
for root,name in ((self.data/'Finance','Thumbs.db'),(self.data/'Finance'/'Child','THUMBS.DB'),
(self.data/'Finance','~$report.docx'),(self.data/'Finance'/'Child','~$Slides.PPTX'),
(self.private/'alice','thumbs.Db'),(self.private/'alice','~$notes.odt')):
path=root/name
self.write(path,'IGNOREDARTIFACT742')
files.append((path,path.read_bytes(),path.stat().st_ino))
return files
def seed_old_artifact_entries(self):
files=self.artifact_files()
with mock.patch.object(documents,'is_excluded_filename',return_value=False):
self.scan()
for path,_,_ in files:
self.conn.execute("UPDATE documents SET body='IGNOREDARTIFACT742',state='pending' WHERE name=?",(path.name,))
self.conn.commit()
return files
def test_cache_and_lock_files_are_never_cataloged_in_data_or_private(self):
files=self.artifact_files()
for name in ('Thumbs.db.docx','project-Thumbs.db','~notes.docx'):
self.write(self.data/'Finance'/name,'ordinary file')
self.scan()
for path,original,inode in files:
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents WHERE name=?',(path.name,)).fetchone()[0],0)
self.assertEqual(path.read_bytes(),original)
self.assertEqual(path.stat().st_ino,inode)
for name in ('Thumbs.db.docx','project-Thumbs.db','~notes.docx'):
self.assertEqual(self.find({'q':[name],'scope':['name']})['total'],1)
def test_stale_artifacts_never_affect_search_counts_facets_or_direct_access(self):
self.seed_old_artifact_entries()
result=self.find({'limit':['1']})
self.assertEqual((result['total'],len(result['items']),result['hasMore'],result['types'],result['pending']),
(2,1,True,['docx','odt'],0))
for scope in ('name','all','content'):
for query in ('Thumbs','report','IGNOREDARTIFACT742'):
self.assertEqual(self.find({'q':[query],'scope':[scope]})['total'],0)
for row in self.conn.execute("SELECT id FROM documents WHERE name GLOB '~$*' OR lower(name)='thumbs.db'"):
with self.assertRaises(FileNotFoundError):
documents.detail(self.conn,IDENTITY,row['id'])
for operation in (documents.download,documents.preview):
with self.assertRaises(FileNotFoundError), operation(self.conn,IDENTITY,row['id']):
pass
with mock.patch.object(document_index,'run_job') as job:
while document_index.process_next(self.conn):
pass
job.assert_not_called()
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents').fetchone()[0],4)
def test_upgrade_removes_old_artifacts_and_fts_without_changing_originals_or_pause(self):
files=self.seed_old_artifact_entries()
documents.control_worker(self.conn,'pause','admin')
self.conn.execute('PRAGMA user_version=1')
self.conn.commit()
documents.ensure_schema(self.conn)
self.assertTrue(documents.worker_paused(self.conn))
self.assertEqual(self.conn.execute('SELECT count(*) FROM documents').fetchone()[0],4)
self.assertEqual(self.conn.execute("SELECT count(*) FROM document_fts WHERE document_fts MATCH 'IGNOREDARTIFACT742'").fetchone()[0],0)
self.assertEqual(self.find({'q':['Umsatz'],'scope':['content']})['total'],1)
for path,original,inode in files:
self.assertEqual(path.read_bytes(),original)
self.assertEqual(path.stat().st_ino,inode)
documents.ensure_schema(self.conn)
self.assertEqual(self.conn.execute('PRAGMA user_version').fetchone()[0],2)
def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self): def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self):
for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'): for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742') self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742')