This commit is contained in:
Ludwig Lehnert
2026-10-03 14:53:48 +00:00
parent 98b08f6e57
commit 9944f2be1f
10 changed files with 193 additions and 66 deletions
+2 -2
View File
@@ -14,8 +14,8 @@ const sources=[{id:'data:finance',label:'Finanzen',kind:'data'},{id:'data:projec
const fixtures=[
['Rechnung Oktober 2026.pdf','pdf','data:finance','Rechnung für Büroausstattung\nReferenz FINANZ742\nGesamtbetrag 1.500,00 EUR',true],
['Projektplan für die Einführung des neuen Dokumentenportals und die Schulung aller Mitarbeiter.docx','docx','data:projects','Projektplan mit Schulung und Zeitplan. Start im Oktober.',false],
['Meine Notizen.txt','txt','private:alice','Meine persönlichen Notizen. Termin zur Budgetplanung am Mittwoch.',false],
['Umsatzübersicht.csv','csv','data:finance','Quartal,Umsatz\nQ1,120000\nQ2,135000',false],
['Meine Notizen.txt','txt','private:alice','',false],
['Umsatzübersicht.csv','csv','data:finance','',false],
['Gescanntes Protokoll.pdf','pdf','data:projects','',true]
].map(([name,extension,sourceId,text,hasPreview],index)=>({id:(index+1).toString(16).padStart(32,'0'),name,extension,sourceId,text,path:'Dokumente/'+name,source:sources.find(source=>source.id===sourceId).label,kind:sources.find(source=>source.id===sourceId).kind,size:143360+index*1024,modified:1791023400-index*86400,state:index===4?'ocr':'ready',pages:extension==='pdf'?3:0,hasPreview,version:'aaaaaaaaaaaaaaaa',snippet:text}));
// A small native three-page PDF exercises the real viewer, including its find bar.
+112 -38
View File
@@ -11,7 +11,7 @@ from types import SimpleNamespace
import unittest
from unittest import mock
from app import access_control as access, document_index, documents, reconcile_shares as directory, web_ui
from app import access_control as access, document_index, documents, extract_document, reconcile_shares as directory, web_ui
ALICE = 'S-1-5-21-1-2-3-1100'
BOB = 'S-1-5-21-1-2-3-1101'
@@ -53,10 +53,10 @@ class DocumentFixture(unittest.TestCase):
self.policy.commit()
for name in ('alice','bob'):
(self.private/name).mkdir()
self.write(self.data/'Finance'/'forecast.txt','Forecast apple Umsatz München')
self.write(self.data/'Finance'/'forecast.docx','Forecast apple Umsatz München')
self.write(self.data/'Engineering'/'secret.hidden','Classified secret ROBOT42')
self.write(self.private/'alice'/'own.txt','Private personal ALICEONLY')
self.write(self.private/'bob'/'other.txt','Private personal BOBONLY')
self.write(self.private/'alice'/'own.odt','Private personal ALICEONLY')
self.write(self.private/'bob'/'other.odt','Private personal BOBONLY')
self.conn = documents.connect()
self.addCleanup(self.conn.close)
documents.ensure_schema(self.conn)
@@ -89,8 +89,8 @@ class DocumentTests(DocumentFixture):
def test_search_facets_counts_and_snippets_never_expose_other_users_or_hidden_folders(self):
value = self.find()
self.assertEqual(value['total'],2)
self.assertEqual({row['name'] for row in value['items']},{'forecast.txt','own.txt'})
self.assertEqual(value['types'],['txt'])
self.assertEqual({row['name'] for row in value['items']},{'forecast.docx','own.odt'})
self.assertEqual(value['types'],['docx','odt'])
self.assertEqual({source['label'] for source in value['sources']},{'Finance','Private'})
self.assertEqual(self.find({'q':['BOBONLY']})['total'],0)
self.assertEqual(self.find({'q':['secret']})['total'],0)
@@ -105,8 +105,82 @@ class DocumentTests(DocumentFixture):
for query in ('"', '*', ':', '" OR NOT ()', 'x\x00y'):
self.find({'q':[query]})
def test_only_document_types_are_queued_for_content_and_all_types_keep_filename_search(self):
for extension in ('.pdf','.docx','.pptx','.odt','.odp','.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
self.write(self.data/'Finance'/('typed'+extension),'EXCLUDEDBODY742')
self.scan()
for extension in ('.pdf','.docx','.pptx','.odt','.odp'):
self.assertEqual(self.document('typed'+extension)['state'],'pending')
for extension in ('.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
row=self.document('typed'+extension)
self.assertEqual(row['state'],'name-only')
self.assertEqual(row['body'],'')
self.assertEqual(self.find({'q':['typed'+extension],'scope':['name']})['total'],1)
self.assertEqual(self.find({'q':['typed'+extension]})['total'],1)
self.assertEqual(self.find({'q':['EXCLUDEDBODY742'],'scope':['content']})['total'],0)
def test_excluded_stale_content_cannot_match_all_or_content_search_or_leak_in_snippets(self):
self.write(self.data/'Finance'/'legacy.xlsx','original spreadsheet bytes')
self.scan()
row=self.document('legacy.xlsx')
self.conn.execute("UPDATE documents SET body='STALESPREADSHEET742',state='ready' WHERE id=?",(row['id'],))
self.conn.commit()
for scope in ('all','content'):
self.assertEqual(self.find({'q':['STALESPREADSHEET742'],'scope':[scope]})['total'],0)
result=self.find({'q':['legacy'],'scope':['all']})
self.assertEqual(result['total'],1)
self.assertEqual(result['items'][0]['snippet'],'')
self.assertEqual(documents.detail(self.conn,IDENTITY,row['id'])['text'],'')
self.assertEqual(next(item for item in self.find()['items'] if item['id']==row['id'])['snippet'],'')
def test_upgrade_clears_excluded_fts_and_jobs_but_keeps_ids_originals_and_image_previews(self):
files=[]
for extension in ('.txt','.md','.csv','.xlsx','.ods','.png'):
path=self.data/'Finance'/('upgrade'+extension)
self.write(path,'original bytes '+extension)
files.append((path,path.read_bytes(),path.stat().st_ino))
self.scan()
ids={self.document(path.name)['id'] for path,_,_ in files}
for path,_,_ in files:
row=self.document(path.name)
self.conn.execute("UPDATE documents SET body='OLDINDEX742',state='failed',attempts=4,retry_at=9999999999,preview=?,pages=7 WHERE id=?", ('retained.jpg' if path.suffix=='.png' else '',row['id']))
self.conn.execute('PRAGMA user_version=0')
self.conn.commit()
paused=documents.control_worker(self.conn,'pause','admin')
self.assertTrue(paused['paused'])
documents.ensure_schema(self.conn)
self.assertTrue(documents.worker_paused(self.conn))
self.assertEqual(ids,{self.document(path.name)['id'] for path,_,_ in files})
for path,original,inode in files:
row=self.document(path.name)
self.assertEqual((row['body'],row['state'],row['attempts'],row['retry_at'],row['pages']),('','name-only',0,0,0))
self.assertEqual(path.read_bytes(),original)
self.assertEqual(path.stat().st_ino,inode)
self.assertEqual(self.find({'q':[path.name],'scope':['name']})['total'],1)
self.assertEqual(self.document('upgrade.png')['preview'],'retained.jpg')
self.assertEqual(self.conn.execute("SELECT count(*) FROM document_fts WHERE document_fts MATCH 'body:OLDINDEX742'").fetchone()[0],0)
self.assertEqual(self.find({'q':['Umsatz'],'scope':['content']})['total'],1)
self.assertEqual(self.policy.execute('SELECT count(*) FROM folder_permissions').fetchone()[0],2)
# Reopening a migrated database leaves completed previews untouched.
documents.ensure_schema(self.conn)
self.assertEqual(self.document('upgrade.png')['state'],'name-only')
def test_extractor_rejects_excluded_content_and_preview_jobs_never_index_text(self):
for extension in ('.txt','.md','.csv','.xlsx','.ods','.html','.json','.bin'):
with self.assertRaises(ValueError):
extract_document.extract({'extension':extension})
self.write(self.data/'Finance'/'photo.png','image fixture')
self.scan()
with mock.patch.object(document_index,'run_job',return_value={'body':'UNWANTEDIMAGE742','preview':'photo.jpg','needsOcr':True}) as job:
while document_index.process_next(self.conn):
pass
self.assertEqual(job.call_count,1)
row=self.document('photo.png')
self.assertEqual((row['state'],row['body'],row['preview']),('name-only','','photo.jpg'))
self.assertEqual(self.find({'q':['UNWANTEDIMAGE742']})['total'],0)
def test_preview_download_and_detail_reject_inaccessible_guessed_ids(self):
for name in ('secret.hidden','other.txt'):
for name in ('secret.hidden','other.odt'):
row = self.document(name)
for operation in (documents.detail,):
with self.assertRaises(FileNotFoundError):
@@ -116,7 +190,7 @@ class DocumentTests(DocumentFixture):
pass
def test_revoke_and_archive_apply_to_all_endpoints_without_reindexing(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
with documents.download(self.conn,IDENTITY,row['id']) as (handle,_):
self.assertIn(b'Umsatz',handle.read())
self.policy.execute('UPDATE folder_permissions SET level=0 WHERE principalId=?',(ALICE,))
@@ -133,7 +207,7 @@ class DocumentTests(DocumentFixture):
def test_admin_sees_all_data_but_only_own_private_and_excluded_accounts_see_nothing(self):
admin = {**IDENTITY,'sub':'EXAMPLE\\admin','sid':ADMIN,'role':'domain-admin'}
self.assertEqual({row['name'] for row in self.find(identity=admin)['items']},{'forecast.txt','secret.hidden'})
self.assertEqual({row['name'] for row in self.find(identity=admin)['items']},{'forecast.docx','secret.hidden'})
for username in ('MSOL_sync','krbtgt'):
self.assertEqual(self.find(identity={**IDENTITY,'sub':username})['total'],0)
access.cache_users(self.policy,{ALICE:{'sam':'MSOL_sync','name':'Sync'}})
@@ -142,11 +216,11 @@ class DocumentTests(DocumentFixture):
def test_private_owner_must_match_current_unix_identity(self):
identity = {**IDENTITY,'uid':os.getuid()+10000}
self.assertEqual({row['name'] for row in self.find(identity=identity)['items']},{'forecast.txt'})
self.assertEqual({row['name'] for row in self.find(identity=identity)['items']},{'forecast.docx'})
def test_private_read_permissions_filter_counts_types_and_all_file_endpoints(self):
row = self.document('own.txt')
(self.private/'alice'/'own.txt').chmod(0)
row = self.document('own.odt')
(self.private/'alice'/'own.odt').chmod(0)
self.assertEqual(self.find()['total'],1)
self.assertEqual(self.find({'q':['ALICEONLY']})['total'],0)
with self.assertRaises(FileNotFoundError):
@@ -156,43 +230,43 @@ class DocumentTests(DocumentFixture):
pass
def test_changed_and_deleted_files_cannot_serve_stale_text_preview_or_download(self):
row = self.document('forecast.txt')
self.write(self.data/'Finance'/'forecast.txt','Completely new revision')
self.assertNotIn('forecast.txt',{item['name'] for item in self.find()['items']})
row = self.document('forecast.docx')
self.write(self.data/'Finance'/'forecast.docx','Completely new revision')
self.assertNotIn('forecast.docx',{item['name'] for item in self.find()['items']})
with self.assertRaises(FileNotFoundError):
documents.detail(self.conn,IDENTITY,row['id'])
self.scan()
current = self.document('forecast.txt')
current = self.document('forecast.docx')
self.assertEqual(current['id'],row['id'])
self.assertEqual(current['body'],'')
self.assertEqual(current['state'],'pending')
self.assertEqual(self.find({'q':['Umsatz']})['total'],0)
(self.data/'Finance'/'forecast.txt').unlink()
(self.data/'Finance'/'forecast.docx').unlink()
self.scan()
self.assertEqual(self.find()['total'],1)
def test_symlinks_trash_fifo_and_path_traversal_are_not_indexed_or_opened(self):
folder = self.data/'Finance'
os.symlink(self.private/'bob'/'other.txt',folder/'linked.txt')
os.symlink(self.private/'bob'/'other.odt',folder/'linked.txt')
os.symlink(self.private/'bob',folder/'linked-directory')
os.mkfifo(folder/'fifo')
self.write(folder/'.trash'/'deleted.txt','deleted contents')
self.scan()
self.assertEqual(self.find()['total'],2)
for path in ('../alice/own.txt','.trash/deleted.txt','linked.txt','linked-directory/other.txt','fifo','/forecast.txt'):
for path in ('../alice/own.odt','.trash/deleted.txt','linked.txt','linked-directory/other.odt','fifo','/forecast.docx'):
with self.assertRaises((OSError,FileNotFoundError)), documents.open_file(str(folder),path):
pass
def test_queue_phases_retries_and_restart_keep_ocr_work_durable(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
self.conn.execute("UPDATE documents SET extension='.pdf',state='pending' WHERE id=?",(row['id'],))
self.conn.commit()
with mock.patch.object(document_index,'run_job',return_value={'body':'native footer','pages':1,'needsOcr':True,'preview':''}):
self.assertTrue(document_index.process_next(self.conn))
self.assertEqual(self.document('forecast.txt')['state'],'ocr')
self.assertEqual(self.document('forecast.docx')['state'],'ocr')
with mock.patch.object(document_index,'run_job',side_effect=RuntimeError('OCR busy')):
self.assertTrue(document_index.process_next(self.conn))
queued = self.document('forecast.txt')
queued = self.document('forecast.docx')
self.assertEqual(queued['state'],'ocr')
self.assertEqual(queued['body'],'native footer')
self.assertGreater(queued['retry_at'],time.time())
@@ -204,22 +278,22 @@ class DocumentTests(DocumentFixture):
self.assertEqual(self.find({'q':['OCR742']})['total'],1)
def test_stale_extraction_result_cannot_overwrite_a_new_version(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
self.conn.execute("UPDATE documents SET state='pending' WHERE id=?",(row['id'],))
self.conn.commit()
def concurrent_change(*args):
self.write(self.data/'Finance'/'forecast.txt','New content')
self.write(self.data/'Finance'/'forecast.docx','New content')
self.scan()
return {'body':'obsolete confidential text','pages':0,'needsOcr':False,'preview':''}
with mock.patch.object(document_index,'run_job',side_effect=concurrent_change):
document_index.process_next(self.conn)
self.assertEqual(self.document('forecast.txt')['body'],'')
self.assertEqual(self.document('forecast.txt')['state'],'pending')
self.assertEqual(self.document('forecast.docx')['body'],'')
self.assertEqual(self.document('forecast.docx')['state'],'pending')
def test_copy_rejects_growth_before_starting_a_parser(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
source = self.conn.execute('SELECT * FROM document_sources WHERE id=?',(row['source_id'],)).fetchone()
original = self.data/'Finance'/'forecast.txt'
original = self.data/'Finance'/'forecast.docx'
before = original.read_bytes()
@contextlib.contextmanager
def growing_file(*args):
@@ -233,7 +307,7 @@ class DocumentTests(DocumentFixture):
self.assertEqual(original.read_bytes(),before)
def test_pause_survives_restart_and_resume_preserves_queue_and_search(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
self.conn.execute("UPDATE documents SET state='ocr' WHERE id=?",(row['id'],))
self.conn.commit()
value = documents.control_worker(self.conn,'pause','EXAMPLE\\admin')
@@ -243,13 +317,13 @@ class DocumentTests(DocumentFixture):
with mock.patch.object(document_index,'run_job') as job:
self.assertFalse(document_index.process_next(self.conn))
job.assert_not_called()
self.assertEqual(self.document('forecast.txt')['state'],'ocr')
self.assertEqual(self.document('forecast.docx')['state'],'ocr')
self.assertEqual(self.find({'q':['Umsatz']})['total'],1)
self.conn.commit()
documents.control_worker(self.conn,'resume','EXAMPLE\\admin')
with mock.patch.object(document_index,'run_job',return_value={'body':'Resumed OCR742','pages':1,'needsOcr':False,'preview':''}):
self.assertTrue(document_index.process_next(self.conn))
self.assertEqual(self.document('forecast.txt')['state'],'ready')
self.assertEqual(self.document('forecast.docx')['state'],'ready')
self.assertEqual(self.find({'q':['OCR742']})['total'],1)
status = documents.worker_snapshot(self.conn)
self.assertEqual(status['counts']['complete'],4)
@@ -259,7 +333,7 @@ class DocumentTests(DocumentFixture):
documents.control_worker(self.conn,'delete','admin')
def test_interrupting_active_job_keeps_ocr_phase_and_attempt_count(self):
row = self.document('forecast.txt')
row = self.document('forecast.docx')
self.conn.execute("UPDATE documents SET state='ocr' WHERE id=?",(row['id'],))
self.conn.commit()
def pause(*args):
@@ -267,15 +341,15 @@ class DocumentTests(DocumentFixture):
raise document_index.JobPaused()
with mock.patch.object(document_index,'run_job',side_effect=pause):
self.assertFalse(document_index.process_next(self.conn))
queued = self.document('forecast.txt')
queued = self.document('forecast.docx')
self.assertEqual((queued['state'],queued['attempts'],queued['retry_at']),('ocr',0,0))
self.assertEqual(documents.worker_snapshot(self.conn)['current'],None)
def test_interrupted_catalog_scan_never_prunes_existing_records(self):
source = self.conn.execute('SELECT * FROM document_sources WHERE id=?',('data:'+self.folder_ids['Finance'],)).fetchone()
before = self.document('forecast.txt')
before = self.document('forecast.docx')
self.assertFalse(documents.walk_source(self.conn,source,should_stop=lambda:True))
self.assertEqual(self.document('forecast.txt'),before)
self.assertEqual(self.document('forecast.docx'),before)
def test_linux_file_events_detect_atomic_replacement_and_delete(self):
watcher = document_index.FileEvents()
@@ -358,14 +432,14 @@ class DocumentHttpTests(DocumentFixture):
self.assertIn('HttpOnly',headers['Set-Cookie'])
def test_http_download_original_detail_and_guessed_private_id(self):
row=self.document('forecast.txt')
row=self.document('forecast.docx')
status,headers,body=self.request('/api/documents/'+row['id']+'/download')
self.assertEqual(status,200)
self.assertIn(b'Umsatz',body)
self.assertIn('attachment;',headers['Content-Disposition'])
self.assertEqual(headers['Content-Type'],'application/octet-stream')
for suffix in ('','/download','/preview','/content'):
self.assertEqual(self.request('/api/documents/'+self.document('other.txt')['id']+suffix)[0],404)
self.assertEqual(self.request('/api/documents/'+self.document('other.odt')['id']+suffix)[0],404)
self.assertEqual(self.request('/api/documents',identity=None)[0],401)
def test_pdf_content_range_requests_recheck_access_and_keep_original_bytes(self):
@@ -388,7 +462,7 @@ class DocumentHttpTests(DocumentFixture):
self.assertTrue(headers['Content-Range'].endswith('/'+str(len(data))))
for value in ('bytes=999999-','bytes=4-1','bytes=-0','bytes=-','bytes=0-1,3-4','anything'):
self.assertEqual(self.request(path,extra_headers={'Range':value})[0],416)
self.assertEqual(self.request('/api/documents/'+self.document('forecast.txt')['id']+'/content')[0],404)
self.assertEqual(self.request('/api/documents/'+self.document('forecast.docx')['id']+'/content')[0],404)
self.policy.execute('UPDATE folder_permissions SET level=0 WHERE principalId=?',(ALICE,))
self.policy.commit()
self.assertEqual(self.request(path,extra_headers={'Range':'bytes=0-4'})[0],404)