fts (1)
This commit is contained in:
@@ -306,9 +306,9 @@ def process_next(conn, prefer_ocr=False, stop=None):
|
||||
phase = 'ocr' if row['state'] == 'ocr' else 'text'
|
||||
source_label = source['label'] + (' / ' + source['id'][8:] if source['kind'] == 'private' else '')
|
||||
worker_state(conn, phase, row['id'])
|
||||
supported = {'.pdf'} | documents.TEXT_SUFFIXES | documents.OFFICE_SUFFIXES | documents.IMAGE_SUFFIXES
|
||||
supported = documents.CONTENT_SUFFIXES | documents.IMAGE_SUFFIXES
|
||||
if row['extension'] not in supported or row['size'] > setting('DOCUMENT_MAX_FILE_MB', 512, 1, 4096) * 1024 * 1024:
|
||||
conn.execute("UPDATE documents SET state='name-only',indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
|
||||
conn.execute("UPDATE documents SET body='',state='name-only',pages=0,attempts=0,retry_at=0,indexed_at=? WHERE id=? AND fingerprint=?", (time.time(), row['id'], row['fingerprint']))
|
||||
conn.commit()
|
||||
worker_state(conn, 'idle')
|
||||
return True
|
||||
@@ -320,10 +320,11 @@ def process_next(conn, prefer_ocr=False, stop=None):
|
||||
documents.discover_file(conn, source, row['path'])
|
||||
documents.worker_event(conn, 'changed', row['path'], source_label)
|
||||
else:
|
||||
state = 'ocr' if result.get('needsOcr') else 'ready'
|
||||
content = row['extension'] in documents.CONTENT_SUFFIXES
|
||||
state = ('ocr' if row['extension'] == '.pdf' and result.get('needsOcr') else 'ready') if content else 'name-only'
|
||||
conn.execute('''UPDATE documents SET body=?,state=?,pages=?,preview=CASE WHEN ?='' THEN preview ELSE ? END,
|
||||
indexed_at=?,attempts=0,retry_at=0 WHERE id=? AND fingerprint=?''',
|
||||
(str(result.get('body', ''))[:documents.MAX_TEXT], state, result.get('pages', 0),
|
||||
(str(result.get('body', ''))[:documents.MAX_TEXT] if content else '', state, result.get('pages', 0),
|
||||
result.get('preview', ''), result.get('preview', ''), time.time(), row['id'], row['fingerprint']))
|
||||
documents.worker_event(conn, 'queued-ocr' if state == 'ocr' else 'complete', row['path'], source_label)
|
||||
conn.execute('UPDATE document_worker SET processed=processed+1 WHERE id=1')
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Formats eligible for content search, shared by the catalog and parser."""
|
||||
|
||||
OFFICE_SUFFIXES = {'.docx', '.pptx', '.odt', '.odp'}
|
||||
CONTENT_SUFFIXES = {'.pdf'} | OFFICE_SUFFIXES
|
||||
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
|
||||
+35
-11
@@ -14,18 +14,18 @@ try:
|
||||
from . import access_control, reconcile_shares as directory
|
||||
from .account_policy import account_name, is_excluded_user
|
||||
from .state_db import connect_state_db
|
||||
from .document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
except ImportError:
|
||||
import access_control
|
||||
import reconcile_shares as directory
|
||||
from account_policy import account_name, is_excluded_user
|
||||
from state_db import connect_state_db
|
||||
from document_types import CONTENT_SUFFIXES, OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
|
||||
SEARCH_ROOT = os.getenv('DOCUMENT_STATE_ROOT', os.path.join(os.getenv('STATE_ROOT', '/state'), 'documents'))
|
||||
SEARCH_DB = os.path.join(SEARCH_ROOT, 'search.db')
|
||||
MAX_TEXT = 2 * 1024 * 1024
|
||||
TEXT_SUFFIXES = {'.txt', '.md', '.csv', '.tsv', '.json', '.xml', '.html', '.htm', '.log', '.ini', '.yaml', '.yml'}
|
||||
OFFICE_SUFFIXES = {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}
|
||||
IMAGE_SUFFIXES = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}
|
||||
|
||||
|
||||
|
||||
def connect():
|
||||
@@ -81,6 +81,20 @@ def ensure_schema(conn):
|
||||
END;
|
||||
''')
|
||||
|
||||
# Upgrade the derived index only: keep catalog IDs, permissions, previews
|
||||
# and originals, but remove content and jobs for excluded file types.
|
||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
||||
with conn:
|
||||
conn.execute('BEGIN IMMEDIATE')
|
||||
if conn.execute('PRAGMA user_version').fetchone()[0] < 1:
|
||||
content = sorted(CONTENT_SUFFIXES)
|
||||
images = sorted(IMAGE_SUFFIXES)
|
||||
conn.execute(f"""UPDATE documents SET body='',pages=0,attempts=0,retry_at=0,
|
||||
state=CASE WHEN extension IN ({','.join('?' for _ in images)}) AND preview=''
|
||||
THEN 'pending' ELSE 'name-only' END
|
||||
WHERE extension NOT IN ({','.join('?' for _ in content)})""", [*images,*content])
|
||||
conn.execute('PRAGMA user_version=1')
|
||||
|
||||
|
||||
def worker_paused(conn):
|
||||
return bool(conn.execute('SELECT paused FROM document_worker WHERE id=1').fetchone()[0])
|
||||
@@ -241,13 +255,15 @@ def discover_file(conn, source, relative, seen=0):
|
||||
conn.execute('UPDATE documents SET seen=? WHERE source_id=? AND path=?', (seen, source['id'], relative))
|
||||
return
|
||||
name = relative.rsplit('/', 1)[-1]
|
||||
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen)
|
||||
VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
|
||||
extension = os.path.splitext(name)[1].lower()
|
||||
state = 'pending' if extension in CONTENT_SUFFIXES | IMAGE_SUFFIXES else 'name-only'
|
||||
conn.execute('''INSERT INTO documents(id,source_id,path,name,extension,size,modified,fingerprint,seen,state)
|
||||
VALUES(?,?,?,?,?,?,?,?,?,?) ON CONFLICT(source_id,path) DO UPDATE SET
|
||||
name=excluded.name,extension=excluded.extension,size=excluded.size,
|
||||
modified=excluded.modified,fingerprint=excluded.fingerprint,seen=excluded.seen,
|
||||
body='',state='pending',preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
|
||||
(uuid.uuid4().hex, source['id'], relative, name, os.path.splitext(name)[1].lower(),
|
||||
info.st_size, info.st_mtime, stamp, seen))
|
||||
body='',state=excluded.state,preview='',pages=0,attempts=0,retry_at=0,indexed_at=0''',
|
||||
(uuid.uuid4().hex, source['id'], relative, name, extension,
|
||||
info.st_size, info.st_mtime, stamp, seen, state))
|
||||
|
||||
|
||||
def walk_source(conn, source, watch=None, should_stop=None):
|
||||
@@ -385,7 +401,8 @@ def search(conn, identity, params):
|
||||
conditions = [f'd.source_id IN ({placeholders})', private_condition]
|
||||
values = ids[:]
|
||||
from_sql = 'documents d'
|
||||
snippet_sql = "substr(d.body,1,220)"
|
||||
content_types = ','.join("'" + extension + "'" for extension in sorted(CONTENT_SUFFIXES))
|
||||
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN substr(d.body,1,220) ELSE '' END"
|
||||
if query:
|
||||
if scope == 'name':
|
||||
conditions.append("d.name LIKE ? ESCAPE '\\'")
|
||||
@@ -400,7 +417,14 @@ def search(conn, identity, params):
|
||||
from_sql += ' JOIN document_fts ON document_fts.rowid=d.rowid'
|
||||
conditions.append('document_fts MATCH ?')
|
||||
values.append(match)
|
||||
snippet_sql = "snippet(document_fts,2,char(1),char(2),' … ',32)"
|
||||
if scope == 'content':
|
||||
conditions.append(f'd.extension IN ({content_types})')
|
||||
else:
|
||||
# A stale body from an older worker must never make other file
|
||||
# types appear in content search, even before index cleanup.
|
||||
conditions.append(f"(d.extension IN ({content_types}) OR d.rowid IN (SELECT rowid FROM document_fts WHERE document_fts MATCH ?))")
|
||||
values.append('{name path} : (' + ' AND '.join('"' + word + '"*' for word in words) + ')')
|
||||
snippet_sql = f"CASE WHEN d.extension IN ({content_types}) THEN snippet(document_fts,2,char(1),char(2),' … ',32) ELSE '' END"
|
||||
if extension:
|
||||
conditions.append('d.extension=?')
|
||||
values.append('.' + extension)
|
||||
@@ -429,7 +453,7 @@ def search(conn, identity, params):
|
||||
def detail(conn, identity, document_id):
|
||||
row, source = current_document(conn, document_id, identity)
|
||||
result = public_document(row, source)
|
||||
result['text'] = row['body']
|
||||
result['text'] = row['body'] if row['extension'] in CONTENT_SUFFIXES else ''
|
||||
return result
|
||||
|
||||
|
||||
|
||||
@@ -15,6 +15,11 @@ import warnings
|
||||
import zipfile
|
||||
from xml.etree import ElementTree
|
||||
|
||||
try:
|
||||
from .document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
except ImportError:
|
||||
from document_types import OFFICE_SUFFIXES, IMAGE_SUFFIXES
|
||||
|
||||
MAX_TEXT = 2 * 1024 * 1024
|
||||
|
||||
|
||||
@@ -97,7 +102,7 @@ def office_text(path):
|
||||
raise RuntimeError('Office document exceeds extraction limits')
|
||||
for entry in sorted(entries, key=lambda item: item.filename):
|
||||
name = entry.filename
|
||||
wanted = (name.startswith(('word/', 'ppt/slides/', 'xl/worksheets/')) or name in {'xl/sharedStrings.xml', 'content.xml'}) and name.endswith('.xml')
|
||||
wanted = (name.startswith(('word/', 'ppt/slides/')) or name == 'content.xml') and name.endswith('.xml')
|
||||
if not wanted or entry.file_size > 16 * 1024 * 1024:
|
||||
continue
|
||||
node = ElementTree.fromstring(archive.read(entry))
|
||||
@@ -132,9 +137,9 @@ def extract(config):
|
||||
page_text = body.split('\f')[:pages]
|
||||
needs_ocr = has_images and any(len(re.sub(r'\s+', '', page)) < 80 for page in page_text)
|
||||
command(['pdftoppm', '-f', '1', '-l', '1', '-singlefile', '-scale-to', '1400', '-jpeg', str(source), 'preview'])
|
||||
elif suffix in {'.docx', '.xlsx', '.pptx', '.odt', '.ods', '.odp'}:
|
||||
elif suffix in OFFICE_SUFFIXES:
|
||||
body = office_text(source)
|
||||
elif suffix in {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.webp', '.bmp'}:
|
||||
elif suffix in IMAGE_SUFFIXES:
|
||||
from PIL import Image
|
||||
Image.MAX_IMAGE_PIXELS = 25_000_000
|
||||
warnings.simplefilter('error', Image.DecompressionBombWarning)
|
||||
@@ -142,9 +147,7 @@ def extract(config):
|
||||
image.thumbnail((1400,1400))
|
||||
image.convert('RGB').save('preview.jpg', 'JPEG', quality=85)
|
||||
else:
|
||||
body = read_text(source)
|
||||
if suffix in {'.html', '.htm', '.xml'}:
|
||||
body = re.sub(r'<[^>]+>', ' ', body)
|
||||
raise ValueError('File type is not eligible for content extraction')
|
||||
return {'body': body[:MAX_TEXT], 'pages': pages, 'needsOcr': needs_ocr}
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user