This commit is contained in:
Ludwig Lehnert
2026-10-03 14:53:48 +00:00
parent 98b08f6e57
commit 9944f2be1f
10 changed files with 193 additions and 66 deletions
+5 -4
View File
@@ -306,9 +306,9 @@ def process_next(conn, prefer_ocr=False, stop=None):
phase = 'ocr' if row['state'] == 'ocr' else 'text'
source_label = source['label'] + (' / ' + source['id'][8:] if source['kind'] == 'private' else '')
worker_state(conn, phase, row['id'])
supported = {'.pdf'} | documents.TEXT_SUFFIXES | documents.OFFICE_SUFFIXES | documents.IMAGE_SUFFIXES
supported = documents.CONTENT_SUFFIXES | documents.IMAGE_SUFFIXES
if row['extension'] not in supported or row['size'] > setting('DOCUMENT_MAX_FILE_MB', 512, 1, 4096) * 1024 * 1024:
conn.execute("UPDATE documents SET state='name-only',indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
conn.execute("UPDATE documents SET body='',state='name-only',pages=0,attempts=0,retry_at=0,indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
conn.commit()
worker_state(conn, 'idle')
return True
@@ -320,10 +320,11 @@ def process_next(conn, prefer_ocr=False, stop=None):
documents.discover_file(conn, source, row['path'])
documents.worker_event(conn, 'changed', row['path'], source_label)
else:
state = 'ocr' if result.get('needsOcr') else 'ready'
content = row['extension'] in documents.CONTENT_SUFFIXES
state = ('ocr' if row['extension'] == '.pdf' and result.get('needsOcr') else 'ready') if content else 'name-only'
conn.execute('''UPDATE documents SET body=?,state=?,pages=?,preview=CASE WHEN ?='' THEN preview ELSE ? END,
indexed_at=?,attempts=0,retry_at=0 WHERE id=? AND fingerprint=?''',
(str(result.get('body', ''))[:documents.MAX_TEXT], state, result.get('pages', 0),
(str(result.get('body', ''))[:documents.MAX_TEXT] if content else '', state, result.get('pages', 0),
result.get('preview', ''), result.get('preview', ''), time.time(), row['id'], row['fingerprint']))
documents.worker_event(conn, 'queued-ocr' if state == 'ocr' else 'complete', row['path'], source_label)
conn.execute('UPDATE document_worker SET processed=processed+1 WHERE id=1')
+5
View File
@@ -0,0 +1,5 @@
"""Formats eligible for content search, shared by the catalog and parser."""
OFFICE_SUFFIXES = {'.docx', '.pptx', '.odt', '.odp'}
CONTENT_SUFFIXES = {'.pdf'} | OFFICE_SUFFIXES
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
+35 -11
View File
@@ -14,18 +14,18 @@ try:
from . import access_control, reconcile_shares as directory
from .account_policy import account_name, is_excluded_user
from .state_db import connect_state_db
from .document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
except ImportError:
import access_control
import reconcile_shares as directory
from account_policy import account_name, is_excluded_user
from state_db import connect_state_db
from document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
MAX_TEXT = 2 * 1024 * 1024
TEXT_SUFFIXES = {'.txt', '.md', '.csv', '.tsv', '.json', '.xml', '.html', '.htm', '.log', '.ini', '.yaml', '.yml'}
OFFICE_SUFFIXES = {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
def connect():
@@ -81,6 +81,20 @@ def ensure_schema(conn):
END;
''')
# Upgrade the derived index only: keep catalog IDs, permissions, previews
# and originals, but remove content and jobs for excluded file types.
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
with conn:
conn.execute('BEGIN IMMEDIATE')
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
content = sorted(CONTENT_SUFFIXES)
images = sorted(IMAGE_SUFFIXES)
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
THEN 'pending' ELSE 'name-only' END
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
conn.execute('PRAGMA user_version=1')
def worker_paused(conn):
return bool(conn.execute('SELECT paused FROM document_worker WHERE id=1').fetchone()[0])
@@ -241,13 +255,15 @@ def discover_file(conn, source, relative, seen=0):
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
return
name = relative.rsplit('/', 1)[-1]
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen)
VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
extension = os.path.splitext(name)[1].lower()
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
VALUES(?,?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
name=excluded.name,extension=excluded.extension,size=excluded.size,
modified=excluded.modified,fingerprint=excluded.fingerprint,seen=excluded.seen,
body='',state='pending',preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
(uuid.uuid4().hex, source['id'], relative, name, os.path.splitext(name)[1].lower(),
info.st_size, info.st_mtime, stamp, seen))
body='',state=excluded.state,preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
(uuid.uuid4().hex, source['id'], relative, name, extension,
info.st_size, info.st_mtime, stamp, seen, state))
def walk_source(conn, source, watch=None, should_stop=None):
@@ -385,7 +401,8 @@ def search(conn, identity, params):
conditions = [f'd.source_id IN ({placeholders})', private_condition]
values = ids[:]
from_sql = 'documents d'
snippet_sql = "substr(d.body,1,220)"
content_types = ','.join("'" + extension + "'" for extension in sorted(CONTENT_SUFFIXES))
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN substr(d.body,1,220) ELSE '' END"
if query:
if scope == 'name':
conditions.append("d.name LIKE ? ESCAPE '\\'")
@@ -400,7 +417,14 @@ def search(conn, identity, params):
from_sql += ' JOIN document_fts ON document_fts.rowid=d.rowid'
conditions.append('document_fts MATCH ?')
values.append(match)
snippet_sql = "snippet(document_fts,2,char(1),char(2),' … ',32)"
if scope == 'content':
conditions.append(f'd.extension IN ({content_types})')
else:
# A stale body from an older worker must never make other file
# types appear in content search, even before index cleanup.
conditions.append(f"(d.extension IN ({content_types}) OR d.rowid IN (SELECT rowid FROM document_fts WHERE document_fts MATCH ?))")
values.append('{name path} : (' + ' AND '.join('"' + word + '"*' for word in words) + ')')
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN snippet(document_fts,2,char(1),char(2),' … ',32) ELSE '' END"
if extension:
conditions.append('d.extension=?')
values.append('.' + extension)
@@ -429,7 +453,7 @@ def search(conn, identity, params):
def detail(conn, identity, document_id):
row, source = current_document(conn, document_id, identity)
result = public_document(row, source)
result['text'] = row['body']
result['text'] = row['body'] if row['extension'] in CONTENT_SUFFIXES else ''
return result
+9 -6
View File
@@ -15,6 +15,11 @@ import warnings
import zipfile
from xml.etree import ElementTree
try:
from .document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
except ImportError:
from document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
MAX_TEXT = 2 * 1024 * 1024
@@ -97,7 +102,7 @@ def office_text(path):
raise RuntimeError('Office document exceeds extraction limits')
for entry in sorted(entries, key=lambda item: item.filename):
name = entry.filename
wanted = (name.startswith(('word/', 'ppt/slides/', 'xl/worksheets/')) or name in {'xl/sharedStrings.xml', 'content.xml'}) and name.endswith('.xml')
wanted = (name.startswith(('word/', 'ppt/slides/')) or name == 'content.xml') and name.endswith('.xml')
if not wanted or entry.file_size > 16 * 1024 * 1024:
continue
node = ElementTree.fromstring(archive.read(entry))
@@ -132,9 +137,9 @@ def extract(config):
page_text = body.split('\f')[:pages]
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text)
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
elif suffix in {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}:
elif suffix in OFFICE_SUFFIXES:
body = office_text(source)
elif suffix in {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}:
elif suffix in IMAGE_SUFFIXES:
from PIL import Image
Image.MAX_IMAGE_PIXELS = 25_000_000
warnings.simplefilter('error', Image.DecompressionBombWarning)
@@ -142,9 +147,7 @@ def extract(config):
image.thumbnail((1400,1400))
image.convert('RGB').save('preview.jpg', 'JPEG', quality=85)
else:
body = read_text(source)
if suffix in {'.html', '.htm', '.xml'}:
body = re.sub(r'<[^>]+>', ' ', body)
raise ValueError('File type is not eligible for content extraction')
return {'body': body[:MAX_TEXT], 'pages': pages, 'needsOcr': needs_ocr}