fts (1)
This commit is contained in:
@@ -45,6 +45,7 @@ COPY app/audit_store.py /app/audit_store.py
|
||||
COPY app/audit_collector.py /app/audit_collector.py
|
||||
COPY app/trash.py /app/trash.py
|
||||
COPY app/web_ui.py /app/web_ui.py
|
||||
COPY app/document_types.py /app/document_types.py
|
||||
COPY app/documents.py /app/documents.py
|
||||
COPY app/document_index.py /app/document_index.py
|
||||
COPY app/extract_document.py /app/extract_document.py
|
||||
|
||||
@@ -77,7 +77,7 @@ The catalog includes active Data folders and each user's own Private folder. Eve
|
||||
|
||||
Linux filesystem events update the filename catalog as files change, with periodic scans recovering missed events. Text extraction and OCR run through a persistent background queue; files become searchable by filename before content extraction finishes. Search updates while typing, and open views refresh every two seconds to show completed extraction. Full-text queries match word prefixes and ignore accents; filename-only queries also match substrings.
|
||||
|
||||
Content extraction supports PDFs, UTF-8/UTF-16 text files, modern Office formats (`docx`, `xlsx`, `pptx`), and OpenDocument formats (`odt`, `ods`, `odp`). PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Other formats, including legacy binary Office files, remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
||||
Content search indexes only PDFs, Word documents (`docx`, `odt`) and PowerPoint presentations (`pptx`, `odp`). All file types remain searchable by filename, including text, Markdown, spreadsheets and images. Existing extracted content from excluded types is removed from the derived search index on upgrade; original files and filename entries remain intact. PDFs and common images receive a first-page/image thumbnail. PDFs open in a fullscreen dialog with the locally bundled Mozilla PDF.js viewer, including page navigation, zoom, thumbnails and search within native PDF text. PDF loading and byte-range requests check current folder access. Legacy binary Office formats (`doc`, `ppt`) remain searchable by filename. Extracted text is limited to 2 MiB per file; encrypted, damaged, or oversized documents may have no searchable content.
|
||||
|
||||
PDFs containing raster images and pages with little extracted text are queued for local OCR using OCRmyPDF and Tesseract, with German and English enabled by default. Native text remains searchable while OCR is pending. OCR text is used for search. No file type displays a separate extracted document-text panel in its dialog. The PDF viewer displays the original PDF. Downloads always return the original file. No OCR replacement or separate OCR PDF is published. Failed jobs retry with backoff, and pending work survives restarts.
|
||||
|
||||
|
||||
@@ -306,9 +306,9 @@ def process_next(conn, prefer_ocr=False, stop=None):
|
||||
phase = 'ocr' if row['state'] == 'ocr' else 'text'
|
||||
source_label = source['label'] + (' / ' + source['id'][8:] if source['kind'] == 'private' else '')
|
||||
worker_state(conn, phase, row['id'])
|
||||
supported = {'.pdf'} | documents.TEXT_SUFFIXES | documents.OFFICE_SUFFIXES | documents.IMAGE_SUFFIXES
|
||||
supported = documents.CONTENT_SUFFIXES | documents.IMAGE_SUFFIXES
|
||||
if row['extension'] not in supported or row['size'] > setting('DOCUMENT_MAX_FILE_MB', 512, 1, 4096) * 1024 * 1024:
|
||||
conn.execute("UPDATE documents SET state='name-only',indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
|
||||
conn.execute("UPDATE documents SET body='',state='name-only',pages=0,attempts=0,retry_at=0,indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
|
||||
conn.commit()
|
||||
worker_state(conn, 'idle')
|
||||
return True
|
||||
@@ -320,10 +320,11 @@ def process_next(conn, prefer_ocr=False, stop=None):
|
||||
documents.discover_file(conn, source, row['path'])
|
||||
documents.worker_event(conn, 'changed', row['path'], source_label)
|
||||
else:
|
||||
state = 'ocr' if result.get('needsOcr') else 'ready'
|
||||
content = row['extension'] in documents.CONTENT_SUFFIXES
|
||||
state = ('ocr' if row['extension'] == '.pdf' and result.get('needsOcr') else 'ready') if content else 'name-only'
|
||||
conn.execute('''UPDATE documents SET body=?,state=?,pages=?,preview=CASE WHEN ?='' THEN preview ELSE ? END,
|
||||
indexed_at=?,attempts=0,retry_at=0 WHERE id=? AND fingerprint=?''',
|
||||
(str(result.get('body', ''))[:documents.MAX_TEXT], state, result.get('pages', 0),
|
||||
(str(result.get('body', ''))[:documents.MAX_TEXT] if content else '', state, result.get('pages', 0),
|
||||
result.get('preview', ''), result.get('preview', ''), time.time(), row['id'], row['fingerprint']))
|
||||
documents.worker_event(conn, 'queued-ocr' if state == 'ocr' else 'complete', row['path'], source_label)
|
||||
conn.execute('UPDATE document_worker SET processed=processed+1 WHERE id=1')
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Formats eligible for content search, shared by the catalog and parser."""
|
||||
|
||||
OFFICE_SUFFIXES = {'.docx', '.pptx', '.odt', '.odp'}
|
||||
CONTENT_SUFFIXES = {'.pdf'} | OFFICE_SUFFIXES
|
||||
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
|
||||
+35
-11
@@ -14,18 +14,18 @@ try:
|
||||
from . import access_control, reconcile_shares as directory
|
||||
from .account_policy import account_name, is_excluded_user
|
||||
from .state_db import connect_state_db
|
||||
from .document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
except ImportError:
|
||||
import access_control
|
||||
import reconcile_shares as directory
|
||||
from account_policy import account_name, is_excluded_user
|
||||
from state_db import connect_state_db
|
||||
from document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
|
||||
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
|
||||
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
|
||||
MAX_TEXT = 2 * 1024 * 1024
|
||||
TEXT_SUFFIXES = {'.txt', '.md', '.csv', '.tsv', '.json', '.xml', '.html', '.htm', '.log', '.ini', '.yaml', '.yml'}
|
||||
OFFICE_SUFFIXES = {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}
|
||||
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
|
||||
|
||||
|
||||
|
||||
def connect():
|
||||
@@ -81,6 +81,20 @@ def ensure_schema(conn):
|
||||
END;
|
||||
''')
|
||||
|
||||
# Upgrade the derived index only: keep catalog IDs, permissions, previews
|
||||
# and originals, but remove content and jobs for excluded file types.
|
||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
||||
with conn:
|
||||
conn.execute('BEGIN IMMEDIATE')
|
||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
||||
content = sorted(CONTENT_SUFFIXES)
|
||||
images = sorted(IMAGE_SUFFIXES)
|
||||
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
|
||||
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
|
||||
THEN 'pending' ELSE 'name-only' END
|
||||
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
|
||||
conn.execute('PRAGMA user_version=1')
|
||||
|
||||
|
||||
def worker_paused(conn):
|
||||
return bool(conn.execute('SELECT paused FROM document_worker WHERE id=1').fetchone()[0])
|
||||
@@ -241,13 +255,15 @@ def discover_file(conn, source, relative, seen=0):
|
||||
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
|
||||
return
|
||||
name = relative.rsplit('/', 1)[-1]
|
||||
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen)
|
||||
VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
|
||||
extension = os.path.splitext(name)[1].lower()
|
||||
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
|
||||
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
|
||||
VALUES(?,?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
|
||||
name=excluded.name,extension=excluded.extension,size=excluded.size,
|
||||
modified=excluded.modified,fingerprint=excluded.fingerprint,seen=excluded.seen,
|
||||
body='',state='pending',preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
|
||||
(uuid.uuid4().hex, source['id'], relative, name, os.path.splitext(name)[1].lower(),
|
||||
info.st_size, info.st_mtime, stamp, seen))
|
||||
body='',state=excluded.state,preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
|
||||
(uuid.uuid4().hex, source['id'], relative, name, extension,
|
||||
info.st_size, info.st_mtime, stamp, seen, state))
|
||||
|
||||
|
||||
def walk_source(conn, source, watch=None, should_stop=None):
|
||||
@@ -385,7 +401,8 @@ def search(conn, identity, params):
|
||||
conditions = [f'd.source_id IN ({placeholders})', private_condition]
|
||||
values = ids[:]
|
||||
from_sql = 'documents d'
|
||||
snippet_sql = "substr(d.body,1,220)"
|
||||
content_types = ','.join("'" + extension + "'" for extension in sorted(CONTENT_SUFFIXES))
|
||||
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN substr(d.body,1,220) ELSE '' END"
|
||||
if query:
|
||||
if scope == 'name':
|
||||
conditions.append("d.name LIKE ? ESCAPE '\\'")
|
||||
@@ -400,7 +417,14 @@ def search(conn, identity, params):
|
||||
from_sql += ' JOIN document_fts ON document_fts.rowid=d.rowid'
|
||||
conditions.append('document_fts MATCH ?')
|
||||
values.append(match)
|
||||
snippet_sql = "snippet(document_fts,2,char(1),char(2),' … ',32)"
|
||||
if scope == 'content':
|
||||
conditions.append(f'd.extension IN ({content_types})')
|
||||
else:
|
||||
# A stale body from an older worker must never make other file
|
||||
# types appear in content search, even before index cleanup.
|
||||
conditions.append(f"(d.extension IN ({content_types}) OR d.rowid IN (SELECT rowid FROM document_fts WHERE document_fts MATCH ?))")
|
||||
values.append('{name path} : (' + ' AND '.join('"' + word + '"*' for word in words) + ')')
|
||||
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN snippet(document_fts,2,char(1),char(2),' … ',32) ELSE '' END"
|
||||
if extension:
|
||||
conditions.append('d.extension=?')
|
||||
values.append('.' + extension)
|
||||
@@ -429,7 +453,7 @@ def search(conn, identity, params):
|
||||
def detail(conn, identity, document_id):
|
||||
row, source = current_document(conn, document_id, identity)
|
||||
result = public_document(row, source)
|
||||
result['text'] = row['body']
|
||||
result['text'] = row['body'] if row['extension'] in CONTENT_SUFFIXES else ''
|
||||
return result
|
||||
|
||||
|
||||
|
||||
@@ -15,6 +15,11 @@ import warnings
|
||||
import zipfile
|
||||
from xml.etree import ElementTree
|
||||
|
||||
try:
|
||||
from .document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
except ImportError:
|
||||
from document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
|
||||
MAX_TEXT = 2 * 1024 * 1024
|
||||
|
||||
|
||||
@@ -97,7 +102,7 @@ def office_text(path):
|
||||
raise RuntimeError('Office document exceeds extraction limits')
|
||||
for entry in sorted(entries, key=lambda item: item.filename):
|
||||
name = entry.filename
|
||||
wanted = (name.startswith(('word/', 'ppt/slides/', 'xl/worksheets/')) or name in {'xl/sharedStrings.xml', 'content.xml'}) and name.endswith('.xml')
|
||||
wanted = (name.startswith(('word/', 'ppt/slides/')) or name == 'content.xml') and name.endswith('.xml')
|
||||
if not wanted or entry.file_size > 16 * 1024 * 1024:
|
||||
continue
|
||||
node = ElementTree.fromstring(archive.read(entry))
|
||||
@@ -132,9 +137,9 @@ def extract(config):
|
||||
page_text = body.split('\f')[:pages]
|
||||
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text)
|
||||
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
||||
elif suffix in {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}:
|
||||
elif suffix in OFFICE_SUFFIXES:
|
||||
body = office_text(source)
|
||||
elif suffix in {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}:
|
||||
elif suffix in IMAGE_SUFFIXES:
|
||||
from PIL import Image
|
||||
Image.MAX_IMAGE_PIXELS = 25_000_000
|
||||
warnings.simplefilter('error', Image.DecompressionBombWarning)
|
||||
@@ -142,9 +147,7 @@ def extract(config):
|
||||
image.thumbnail((1400,1400))
|
||||
image.convert('RGB').save('preview.jpg', 'JPEG', quality=85)
|
||||
else:
|
||||
body = read_text(source)
|
||||
if suffix in {'.html', '.htm', '.xml'}:
|
||||
body = re.sub(r'<[^>]+>', ' ', body)
|
||||
raise ValueError('File type is not eligible for content extraction')
|
||||
return {'body': body[:MAX_TEXT], 'pages': pages, 'needsOcr': needs_ocr}
|
||||
|
||||
|
||||
|
||||
+13
-4
@@ -329,8 +329,17 @@ def main() -> int:
|
||||
office = eventually("Office document full text", lambda: http(query_path("/api/documents", {"q":"OFFICETEXT742"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=90).json()
|
||||
check(office["items"][0]["extension"] == "docx", "Office full text extraction failed")
|
||||
for marker in ('POWERPOINTONLY742','ODTONLY742','ODPONLY742'):
|
||||
eventually("presentation/document content extraction", lambda: http(query_path("/api/documents", {"q":marker,"scope":"content"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=90)
|
||||
for marker in ('TEXTONLY742','SPREADSHEETONLY742'):
|
||||
for scope in ('all','content'):
|
||||
check(http(query_path("/api/documents", {"q":marker,"scope":scope}), token=user_token).json()["total"] == 0, "excluded file content is searchable")
|
||||
for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'):
|
||||
result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json()
|
||||
check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text")
|
||||
# Guessing an indexed ID is insufficient to get another user's content.
|
||||
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.txt'\").fetchone()[0])").stdout.strip()
|
||||
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip()
|
||||
for suffix in ("", "/preview", "/download", "/content"):
|
||||
check(http("/api/documents/"+hidden_id+suffix, token=user_token).status == 404, "guessed private document ID exposes data")
|
||||
if os.getenv("PLAYWRIGHT_MODULE"):
|
||||
@@ -366,7 +375,7 @@ def main() -> int:
|
||||
http(control_path, method="POST", value={"action":"resume"}, token=token)
|
||||
eventually("paused scan completes after resume", lambda: http("/api/documents/"+queued_scan["id"], token=user_token),
|
||||
lambda r: r.status == 200 and r.json()["state"] == "ready" and "SCANNEDUNIQUE742" in r.json()["text"], timeout=90)
|
||||
eventually("paused filename and content become searchable", lambda: http(query_path("/api/documents", {"q":"PAUSED_QUEUE742"}), token=user_token),
|
||||
eventually("paused filename becomes searchable", lambda: http(query_path("/api/documents", {"q":"paused-note.txt","scope":"name"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json()["total"] == 1, timeout=45)
|
||||
|
||||
announce("legacy GUID-path migration preserves file hashes and inodes")
|
||||
@@ -872,8 +881,8 @@ fi
|
||||
rules(2)
|
||||
allowed("cd Permissions; put /tmp/live-note.txt created.txt")
|
||||
allowed("cd Permissions; put /tmp/live-note.txt existing.txt")
|
||||
engine_run("exec", CLIENT_CONTAINER, "sh", "-c", "printf 'LIVE_CONTENT742 from SMB\n' > /tmp/live-search.txt")
|
||||
allowed("cd Permissions; put /tmp/live-search.txt live-search.txt")
|
||||
engine_run("exec", CLIENT_CONTAINER, "python3", "-c", "import zipfile; z=zipfile.ZipFile('/tmp/live-search.docx','w'); z.writestr('word/document.xml','<document><body><p><t>LIVE_CONTENT742 from SMB</t></p></body></document>'); z.close()")
|
||||
allowed("cd Permissions; put /tmp/live-search.docx live-search.docx")
|
||||
live_row = eventually("SMB write becomes full-text searchable", lambda: http(query_path("/api/documents", {"q":"LIVE_CONTENT742"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=45).json()["items"][0]
|
||||
|
||||
|
||||
@@ -50,6 +50,16 @@ font=next(Path('/usr/share/fonts').rglob('NimbusSans-Regular.otf'))
|
||||
draw=ImageDraw.Draw(image)
|
||||
draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40)
|
||||
image.save(root/'scanned-invoice.pdf','PDF',resolution=150)
|
||||
(root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed')
|
||||
for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]:
|
||||
with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet:
|
||||
spreadsheet.writestr(entry,'<document><t>SPREADSHEETONLY742</t></document>')
|
||||
for name,marker in [('alice','ALICEPRIVATE742'),('bob','BOBPRIVATE742')]:
|
||||
with zipfile.ZipFile(Path('/data/private')/name/'readme.docx','w') as private:
|
||||
private.writestr('word/document.xml','<document><body><p><t>'+marker+'</t></p></body></document>')
|
||||
for suffix,entry,marker in [('pptx','ppt/slides/slide1.xml','POWERPOINTONLY742'),('odt','content.xml','ODTONLY742'),('odp','content.xml','ODPONLY742')]:
|
||||
with zipfile.ZipFile(root/('office-notes.'+suffix),'w') as office:
|
||||
office.writestr(entry,'<document><body><p><t>'+marker+'</t></p></body></document>')
|
||||
with zipfile.ZipFile(root/'office-notes.docx','w') as office:
|
||||
office.writestr('word/document.xml','<document><body><p><t>Office document OFFICETEXT742</t></p></body></document>')
|
||||
PYDOCUMENTS
|
||||
|
||||
@@ -14,8 +14,8 @@ const sources=[{id:'data:finance',label:'Finanzen',kind:'data'},{id:'data:projec
|
||||
const fixtures=[
|
||||
['Rechnung Oktober 2026.pdf','pdf','data:finance','Rechnung für Büroausstattung\nReferenz FINANZ742\nGesamtbetrag 1.500,00 EUR',true],
|
||||
['Projektplan für die Einführung des neuen Dokumentenportals und die Schulung aller Mitarbeiter.docx','docx','data:projects','Projektplan mit Schulung und Zeitplan. Start im Oktober.',false],
|
||||
['Meine Notizen.txt','txt','private:alice','Meine persönlichen Notizen. Termin zur Budgetplanung am Mittwoch.',false],
|
||||
['Umsatzübersicht.csv','csv','data:finance','Quartal,Umsatz\nQ1,120000\nQ2,135000',false],
|
||||
['Meine Notizen.txt','txt','private:alice','',false],
|
||||
['Umsatzübersicht.csv','csv','data:finance','',false],
|
||||
['Gescanntes Protokoll.pdf','pdf','data:projects','',true]
|
||||
].map(([name,extension,sourceId,text,hasPreview],index)=>({id:(index+1).toString(16).padStart(32,'0'),name,extension,sourceId,text,path:'Dokumente/'+name,source:sources.find(source=>source.id===sourceId).label,kind:sources.find(source=>source.id===sourceId).kind,size:143360+index*1024,modified:1791023400-index*86400,state:index===4?'ocr':'ready',pages:extension==='pdf'?3:0,hasPreview,version:'aaaaaaaaaaaaaaaa',snippet:text}));
|
||||
// A small native three-page PDF exercises the real viewer, including its find bar.
|
||||
|
||||
+112
-38
@@ -11,7 +11,7 @@ from types import SimpleNamespace
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
from app import access_control as access, document_index, documents, reconcile_shares as directory, web_ui
|
||||
from app import access_control as access, document_index, documents, extract_document, reconcile_shares as directory, web_ui
|
||||
|
||||
ALICE = 'S-1-5-21-1-2-3-1100'
|
||||
BOB = 'S-1-5-21-1-2-3-1101'
|
||||
@@ -53,10 +53,10 @@ class DocumentFixture(unittest.TestCase):
|
||||
self.policy.commit()
|
||||
for name in ('alice','bob'):
|
||||
(self.private/name).mkdir()
|
||||
self.write(self.data/'Finance'/'forecast.txt','Forecast apple Umsatz München')
|
||||
self.write(self.data/'Finance'/'forecast.docx','Forecast apple Umsatz München')
|
||||
self.write(self.data/'Engineering'/'secret.hidden','Classified secret ROBOT42')
|
||||
self.write(self.private/'alice'/'own.txt','Private personal ALICEONLY')
|
||||
self.write(self.private/'bob'/'other.txt','Private personal BOBONLY')
|
||||
self.write(self.private/'alice'/'own.odt','Private personal ALICEONLY')
|
||||
self.write(self.private/'bob'/'other.odt','Private personal BOBONLY')
|
||||
self.conn = documents.connect()
|
||||
self.addCleanup(self.conn.close)
|
||||
documents.ensure_schema(self.conn)
|
||||
@@ -89,8 +89,8 @@ class DocumentTests(DocumentFixture):
|
||||
def test_search_facets_counts_and_snippets_never_expose_other_users_or_hidden_folders(self):
|
||||
value = self.find()
|
||||
self.assertEqual(value['total'],2)
|
||||
self.assertEqual({row['name'] for row in value['items']},{'forecast.txt','own.txt'})
|
||||
self.assertEqual(value['types'],['txt'])
|
||||
self.assertEqual({row['name'] for row in value['items']},{'forecast.docx','own.odt'})
|
||||
self.assertEqual(value['types'],['docx','odt'])
|
||||
self.assertEqual({source['label'] for source in value['sources']},{'Finance','Private'})
|
||||
self.assertEqual(self.find({'q':['BOBONLY']})['total'],0)
|
||||
self.assertEqual(self.find({'q':['secret']})['total'],0)
|
||||
@@ -105,8 +105,82 @@ class DocumentTests(DocumentFixture):
|
||||
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
|
||||
self.find({'q':[query]})
|
||||
|
||||
def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self):
|
||||
for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
|
||||
self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742')
|
||||
self.scan()
|
||||
for extension in ('.pdf','.docx','.pptx','.odt','.odp'):
|
||||
self.assertEqual(self.document('typed'+extension)['state'],'pending')
|
||||
for extension in ('.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
|
||||
row=self.document('typed'+extension)
|
||||
self.assertEqual(row['state'],'name-only')
|
||||
self.assertEqual(row['body'],'')
|
||||
self.assertEqual(self.find({'q':['typed'+extension],'scope':['name']})['total'],1)
|
||||
self.assertEqual(self.find({'q':['typed'+extension]})['total'],1)
|
||||
self.assertEqual(self.find({'q':['EXCLUDEDBODY742'],'scope':['content']})['total'],0)
|
||||
|
||||
def test_excluded_stale_content_cannot_match_all_or_content_search_or_leak_in_snippets(self):
|
||||
self.write(self.data/'Finance'/'legacy.xlsx','original spreadsheet bytes')
|
||||
self.scan()
|
||||
row=self.document('legacy.xlsx')
|
||||
self.conn.execute("UPDATE documents SET body='STALESPREADSHEET742',state='ready' WHERE id=?",(row['id'],))
|
||||
self.conn.commit()
|
||||
for scope in ('all','content'):
|
||||
self.assertEqual(self.find({'q':['STALESPREADSHEET742'],'scope':[scope]})['total'],0)
|
||||
result=self.find({'q':['legacy'],'scope':['all']})
|
||||
self.assertEqual(result['total'],1)
|
||||
self.assertEqual(result['items'][0]['snippet'],'')
|
||||
self.assertEqual(documents.detail(self.conn,IDENTITY,row['id'])['text'],'')
|
||||
self.assertEqual(next(item for item in self.find()['items'] if item['id']==row['id'])['snippet'],'')
|
||||
|
||||
def test_upgrade_clears_excluded_fts_and_jobs_but_keeps_ids_originals_and_image_previews(self):
|
||||
files=[]
|
||||
for extension in ('.txt','.md','.csv','.xlsx','.ods','.png'):
|
||||
path=self.data/'Finance'/('upgrade'+extension)
|
||||
self.write(path,'original bytes '+extension)
|
||||
files.append((path,path.read_bytes(),path.stat().st_ino))
|
||||
self.scan()
|
||||
ids={self.document(path.name)['id'] for path,_,_ in files}
|
||||
for path,_,_ in files:
|
||||
row=self.document(path.name)
|
||||
self.conn.execute("UPDATE documents SET body='OLDINDEX742',state='failed',attempts=4,retry_at=9999999999,preview=?,pages=7 WHERE id=?", ('retained.jpg' if path.suffix=='.png' else '',row['id']))
|
||||
self.conn.execute('PRAGMA user_version=0')
|
||||
self.conn.commit()
|
||||
paused=documents.control_worker(self.conn,'pause','admin')
|
||||
self.assertTrue(paused['paused'])
|
||||
documents.ensure_schema(self.conn)
|
||||
self.assertTrue(documents.worker_paused(self.conn))
|
||||
self.assertEqual(ids,{self.document(path.name)['id'] for path,_,_ in files})
|
||||
for path,original,inode in files:
|
||||
row=self.document(path.name)
|
||||
self.assertEqual((row['body'],row['state'],row['attempts'],row['retry_at'],row['pages']),('','name-only',0,0,0))
|
||||
self.assertEqual(path.read_bytes(),original)
|
||||
self.assertEqual(path.stat().st_ino,inode)
|
||||
self.assertEqual(self.find({'q':[path.name],'scope':['name']})['total'],1)
|
||||
self.assertEqual(self.document('upgrade.png')['preview'],'retained.jpg')
|
||||
self.assertEqual(self.conn.execute("SELECT count(*) FROM document_fts WHERE document_fts MATCH 'body:OLDINDEX742'").fetchone()[0],0)
|
||||
self.assertEqual(self.find({'q':['Umsatz'],'scope':['content']})['total'],1)
|
||||
self.assertEqual(self.policy.execute('SELECT count(*) FROM folder_permissions').fetchone()[0],2)
|
||||
# Reopening a migrated database leaves completed previews untouched.
|
||||
documents.ensure_schema(self.conn)
|
||||
self.assertEqual(self.document('upgrade.png')['state'],'name-only')
|
||||
|
||||
def test_extractor_rejects_excluded_content_and_preview_jobs_never_index_text(self):
|
||||
for extension in ('.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
|
||||
with self.assertRaises(ValueError):
|
||||
extract_document.extract({'extension':extension})
|
||||
self.write(self.data/'Finance'/'photo.png','image fixture')
|
||||
self.scan()
|
||||
with mock.patch.object(document_index,'run_job',return_value={'body':'UNWANTEDIMAGE742','preview':'photo.jpg','needsOcr':True}) as job:
|
||||
while document_index.process_next(self.conn):
|
||||
pass
|
||||
self.assertEqual(job.call_count,1)
|
||||
row=self.document('photo.png')
|
||||
self.assertEqual((row['state'],row['body'],row['preview']),('name-only','','photo.jpg'))
|
||||
self.assertEqual(self.find({'q':['UNWANTEDIMAGE742']})['total'],0)
|
||||
|
||||
def test_preview_download_and_detail_reject_inaccessible_guessed_ids(self):
|
||||
for name in ('secret.hidden','other.txt'):
|
||||
for name in ('secret.hidden','other.odt'):
|
||||
row = self.document(name)
|
||||
for operation in (documents.detail,):
|
||||
with self.assertRaises(FileNotFoundError):
|
||||
@@ -116,7 +190,7 @@ class DocumentTests(DocumentFixture):
|
||||
pass
|
||||
|
||||
def test_revoke_and_archive_apply_to_all_endpoints_without_reindexing(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
with documents.download(self.conn,IDENTITY,row['id']) as (handle,_):
|
||||
self.assertIn(b'Umsatz',handle.read())
|
||||
self.policy.execute('UPDATE folder_permissions SET level=0 WHERE principalId=?',(ALICE,))
|
||||
@@ -133,7 +207,7 @@ class DocumentTests(DocumentFixture):
|
||||
|
||||
def test_admin_sees_all_data_but_only_own_private_and_excluded_accounts_see_nothing(self):
|
||||
admin = {**IDENTITY,'sub':'EXAMPLE\\admin','sid':ADMIN,'role':'domain-admin'}
|
||||
self.assertEqual({row['name'] for row in self.find(identity=admin)['items']},{'forecast.txt','secret.hidden'})
|
||||
self.assertEqual({row['name'] for row in self.find(identity=admin)['items']},{'forecast.docx','secret.hidden'})
|
||||
for username in ('MSOL_sync','krbtgt'):
|
||||
self.assertEqual(self.find(identity={**IDENTITY,'sub':username})['total'],0)
|
||||
access.cache_users(self.policy,{ALICE:{'sam':'MSOL_sync','name':'Sync'}})
|
||||
@@ -142,11 +216,11 @@ class DocumentTests(DocumentFixture):
|
||||
|
||||
def test_private_owner_must_match_current_unix_identity(self):
|
||||
identity = {**IDENTITY,'uid':os.getuid()+10000}
|
||||
self.assertEqual({row['name'] for row in self.find(identity=identity)['items']},{'forecast.txt'})
|
||||
self.assertEqual({row['name'] for row in self.find(identity=identity)['items']},{'forecast.docx'})
|
||||
|
||||
def test_private_read_permissions_filter_counts_types_and_all_file_endpoints(self):
|
||||
row = self.document('own.txt')
|
||||
(self.private/'alice'/'own.txt').chmod(0)
|
||||
row = self.document('own.odt')
|
||||
(self.private/'alice'/'own.odt').chmod(0)
|
||||
self.assertEqual(self.find()['total'],1)
|
||||
self.assertEqual(self.find({'q':['ALICEONLY']})['total'],0)
|
||||
with self.assertRaises(FileNotFoundError):
|
||||
@@ -156,43 +230,43 @@ class DocumentTests(DocumentFixture):
|
||||
pass
|
||||
|
||||
def test_changed_and_deleted_files_cannot_serve_stale_text_preview_or_download(self):
|
||||
row = self.document('forecast.txt')
|
||||
self.write(self.data/'Finance'/'forecast.txt','Completely new revision')
|
||||
self.assertNotIn('forecast.txt',{item['name'] for item in self.find()['items']})
|
||||
row = self.document('forecast.docx')
|
||||
self.write(self.data/'Finance'/'forecast.docx','Completely new revision')
|
||||
self.assertNotIn('forecast.docx',{item['name'] for item in self.find()['items']})
|
||||
with self.assertRaises(FileNotFoundError):
|
||||
documents.detail(self.conn,IDENTITY,row['id'])
|
||||
self.scan()
|
||||
current = self.document('forecast.txt')
|
||||
current = self.document('forecast.docx')
|
||||
self.assertEqual(current['id'],row['id'])
|
||||
self.assertEqual(current['body'],'')
|
||||
self.assertEqual(current['state'],'pending')
|
||||
self.assertEqual(self.find({'q':['Umsatz']})['total'],0)
|
||||
(self.data/'Finance'/'forecast.txt').unlink()
|
||||
(self.data/'Finance'/'forecast.docx').unlink()
|
||||
self.scan()
|
||||
self.assertEqual(self.find()['total'],1)
|
||||
|
||||
def test_symlinks_trash_fifo_and_path_traversal_are_not_indexed_or_opened(self):
|
||||
folder = self.data/'Finance'
|
||||
os.symlink(self.private/'bob'/'other.txt',folder/'linked.txt')
|
||||
os.symlink(self.private/'bob'/'other.odt',folder/'linked.txt')
|
||||
os.symlink(self.private/'bob',folder/'linked-directory')
|
||||
os.mkfifo(folder/'fifo')
|
||||
self.write(folder/'.trash'/'deleted.txt','deleted contents')
|
||||
self.scan()
|
||||
self.assertEqual(self.find()['total'],2)
|
||||
for path in ('../alice/own.txt','.trash/deleted.txt','linked.txt','linked-directory/other.txt','fifo','/forecast.txt'):
|
||||
for path in ('../alice/own.odt','.trash/deleted.txt','linked.txt','linked-directory/other.odt','fifo','/forecast.docx'):
|
||||
with self.assertRaises((OSError,FileNotFoundError)), documents.open_file(str(folder),path):
|
||||
pass
|
||||
|
||||
def test_queue_phases_retries_and_restart_keep_ocr_work_durable(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
self.conn.execute("UPDATE documents SET extension='.pdf',state='pending' WHERE id=?",(row['id'],))
|
||||
self.conn.commit()
|
||||
with mock.patch.object(document_index,'run_job',return_value={'body':'native footer','pages':1,'needsOcr':True,'preview':''}):
|
||||
self.assertTrue(document_index.process_next(self.conn))
|
||||
self.assertEqual(self.document('forecast.txt')['state'],'ocr')
|
||||
self.assertEqual(self.document('forecast.docx')['state'],'ocr')
|
||||
with mock.patch.object(document_index,'run_job',side_effect=RuntimeError('OCR busy')):
|
||||
self.assertTrue(document_index.process_next(self.conn))
|
||||
queued = self.document('forecast.txt')
|
||||
queued = self.document('forecast.docx')
|
||||
self.assertEqual(queued['state'],'ocr')
|
||||
self.assertEqual(queued['body'],'native footer')
|
||||
self.assertGreater(queued['retry_at'],time.time())
|
||||
@@ -204,22 +278,22 @@ class DocumentTests(DocumentFixture):
|
||||
self.assertEqual(self.find({'q':['OCR742']})['total'],1)
|
||||
|
||||
def test_stale_extraction_result_cannot_overwrite_a_new_version(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
self.conn.execute("UPDATE documents SET state='pending' WHERE id=?",(row['id'],))
|
||||
self.conn.commit()
|
||||
def concurrent_change(*args):
|
||||
self.write(self.data/'Finance'/'forecast.txt','New content')
|
||||
self.write(self.data/'Finance'/'forecast.docx','New content')
|
||||
self.scan()
|
||||
return {'body':'obsolete confidential text','pages':0,'needsOcr':False,'preview':''}
|
||||
with mock.patch.object(document_index,'run_job',side_effect=concurrent_change):
|
||||
document_index.process_next(self.conn)
|
||||
self.assertEqual(self.document('forecast.txt')['body'],'')
|
||||
self.assertEqual(self.document('forecast.txt')['state'],'pending')
|
||||
self.assertEqual(self.document('forecast.docx')['body'],'')
|
||||
self.assertEqual(self.document('forecast.docx')['state'],'pending')
|
||||
|
||||
def test_copy_rejects_growth_before_starting_a_parser(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
source = self.conn.execute('SELECT * FROM document_sources WHERE id=?',(row['source_id'],)).fetchone()
|
||||
original = self.data/'Finance'/'forecast.txt'
|
||||
original = self.data/'Finance'/'forecast.docx'
|
||||
before = original.read_bytes()
|
||||
@contextlib.contextmanager
|
||||
def growing_file(*args):
|
||||
@@ -233,7 +307,7 @@ class DocumentTests(DocumentFixture):
|
||||
self.assertEqual(original.read_bytes(),before)
|
||||
|
||||
def test_pause_survives_restart_and_resume_preserves_queue_and_search(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
self.conn.execute("UPDATE documents SET state='ocr' WHERE id=?",(row['id'],))
|
||||
self.conn.commit()
|
||||
value = documents.control_worker(self.conn,'pause','EXAMPLE\\admin')
|
||||
@@ -243,13 +317,13 @@ class DocumentTests(DocumentFixture):
|
||||
with mock.patch.object(document_index,'run_job') as job:
|
||||
self.assertFalse(document_index.process_next(self.conn))
|
||||
job.assert_not_called()
|
||||
self.assertEqual(self.document('forecast.txt')['state'],'ocr')
|
||||
self.assertEqual(self.document('forecast.docx')['state'],'ocr')
|
||||
self.assertEqual(self.find({'q':['Umsatz']})['total'],1)
|
||||
self.conn.commit()
|
||||
documents.control_worker(self.conn,'resume','EXAMPLE\\admin')
|
||||
with mock.patch.object(document_index,'run_job',return_value={'body':'Resumed OCR742','pages':1,'needsOcr':False,'preview':''}):
|
||||
self.assertTrue(document_index.process_next(self.conn))
|
||||
self.assertEqual(self.document('forecast.txt')['state'],'ready')
|
||||
self.assertEqual(self.document('forecast.docx')['state'],'ready')
|
||||
self.assertEqual(self.find({'q':['OCR742']})['total'],1)
|
||||
status = documents.worker_snapshot(self.conn)
|
||||
self.assertEqual(status['counts']['complete'],4)
|
||||
@@ -259,7 +333,7 @@ class DocumentTests(DocumentFixture):
|
||||
documents.control_worker(self.conn,'delete','admin')
|
||||
|
||||
def test_interrupting_active_job_keeps_ocr_phase_and_attempt_count(self):
|
||||
row = self.document('forecast.txt')
|
||||
row = self.document('forecast.docx')
|
||||
self.conn.execute("UPDATE documents SET state='ocr' WHERE id=?",(row['id'],))
|
||||
self.conn.commit()
|
||||
def pause(*args):
|
||||
@@ -267,15 +341,15 @@ class DocumentTests(DocumentFixture):
|
||||
raise document_index.JobPaused()
|
||||
with mock.patch.object(document_index,'run_job',side_effect=pause):
|
||||
self.assertFalse(document_index.process_next(self.conn))
|
||||
queued = self.document('forecast.txt')
|
||||
queued = self.document('forecast.docx')
|
||||
self.assertEqual((queued['state'],queued['attempts'],queued['retry_at']),('ocr',0,0))
|
||||
self.assertEqual(documents.worker_snapshot(self.conn)['current'],None)
|
||||
|
||||
def test_interrupted_catalog_scan_never_prunes_existing_records(self):
|
||||
source = self.conn.execute('SELECT * FROM document_sources WHERE id=?',('data:'+self.folder_ids['Finance'],)).fetchone()
|
||||
before = self.document('forecast.txt')
|
||||
before = self.document('forecast.docx')
|
||||
self.assertFalse(documents.walk_source(self.conn,source,should_stop=lambda:True))
|
||||
self.assertEqual(self.document('forecast.txt'),before)
|
||||
self.assertEqual(self.document('forecast.docx'),before)
|
||||
|
||||
def test_linux_file_events_detect_atomic_replacement_and_delete(self):
|
||||
watcher = document_index.FileEvents()
|
||||
@@ -358,14 +432,14 @@ class DocumentHttpTests(DocumentFixture):
|
||||
self.assertIn('HttpOnly',headers['Set-Cookie'])
|
||||
|
||||
def test_http_download_original_detail_and_guessed_private_id(self):
|
||||
row=self.document('forecast.txt')
|
||||
row=self.document('forecast.docx')
|
||||
status,headers,body=self.request('/api/documents/'+row['id']+'/download')
|
||||
self.assertEqual(status,200)
|
||||
self.assertIn(b'Umsatz',body)
|
||||
self.assertIn('attachment;',headers['Content-Disposition'])
|
||||
self.assertEqual(headers['Content-Type'],'application/octet-stream')
|
||||
for suffix in ('','/download','/preview','/content'):
|
||||
self.assertEqual(self.request('/api/documents/'+self.document('other.txt')['id']+suffix)[0],404)
|
||||
self.assertEqual(self.request('/api/documents/'+self.document('other.odt')['id']+suffix)[0],404)
|
||||
self.assertEqual(self.request('/api/documents',identity=None)[0],401)
|
||||
|
||||
def test_pdf_content_range_requests_recheck_access_and_keep_original_bytes(self):
|
||||
@@ -388,7 +462,7 @@ class DocumentHttpTests(DocumentFixture):
|
||||
self.assertTrue(headers['Content-Range'].endswith('/'+str(len(data))))
|
||||
for value in ('bytes=999999-','bytes=4-1','bytes=-0','bytes=-','bytes=0-1,3-4','anything'):
|
||||
self.assertEqual(self.request(path,extra_headers={'Range':value})[0],416)
|
||||
self.assertEqual(self.request('/api/documents/'+self.document('forecast.txt')['id']+'/content')[0],404)
|
||||
self.assertEqual(self.request('/api/documents/'+self.document('forecast.docx')['id']+'/content')[0],404)
|
||||
self.policy.execute('UPDATE folder_permissions SET level=0 WHERE principalId=?',(ALICE,))
|
||||
self.policy.commit()
|
||||
self.assertEqual(self.request(path,extra_headers={'Range':'bytes=0-4'})[0],404)
|
||||
|
||||
Reference in New Issue
Block a user