fts (1)
This commit is contained in:
+13
-4
@@ -329,8 +329,17 @@ def main() -> int:
|
||||
office = eventually("Office document full text", lambda: http(query_path("/api/documents", {"q":"OFFICETEXT742"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=90).json()
|
||||
check(office["items"][0]["extension"] == "docx", "Office full text extraction failed")
|
||||
for marker in ('POWERPOINTONLY742','ODTONLY742','ODPONLY742'):
|
||||
eventually("presentation/document content extraction", lambda: http(query_path("/api/documents", {"q":marker,"scope":"content"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=90)
|
||||
for marker in ('TEXTONLY742','SPREADSHEETONLY742'):
|
||||
for scope in ('all','content'):
|
||||
check(http(query_path("/api/documents", {"q":marker,"scope":scope}), token=user_token).json()["total"] == 0, "excluded file content is searchable")
|
||||
for filename in ('index-excluded.txt','index-excluded.xlsx','index-excluded.ods','forecast.csv','readme.txt'):
|
||||
result=http(query_path("/api/documents", {"q":filename,"scope":"name"}), token=user_token).json()
|
||||
check(result["total"] == 1 and result["items"][0]["snippet"] == '', "excluded file cannot be found by name or exposes text")
|
||||
# Guessing an indexed ID is insufficient to get another user's content.
|
||||
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.txt'\").fetchone()[0])").stdout.strip()
|
||||
hidden_id = engine_run("exec", FILES_CONTAINER, "python3", "-c", "import sqlite3; c=sqlite3.connect('/state/documents/search.db'); print(c.execute(\"SELECT id FROM documents WHERE source_id='private:bob' AND name='readme.docx'\").fetchone()[0])").stdout.strip()
|
||||
for suffix in ("", "/preview", "/download", "/content"):
|
||||
check(http("/api/documents/"+hidden_id+suffix, token=user_token).status == 404, "guessed private document ID exposes data")
|
||||
if os.getenv("PLAYWRIGHT_MODULE"):
|
||||
@@ -366,7 +375,7 @@ def main() -> int:
|
||||
http(control_path, method="POST", value={"action":"resume"}, token=token)
|
||||
eventually("paused scan completes after resume", lambda: http("/api/documents/"+queued_scan["id"], token=user_token),
|
||||
lambda r: r.status == 200 and r.json()["state"] == "ready" and "SCANNEDUNIQUE742" in r.json()["text"], timeout=90)
|
||||
eventually("paused filename and content become searchable", lambda: http(query_path("/api/documents", {"q":"PAUSED_QUEUE742"}), token=user_token),
|
||||
eventually("paused filename becomes searchable", lambda: http(query_path("/api/documents", {"q":"paused-note.txt","scope":"name"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json()["total"] == 1, timeout=45)
|
||||
|
||||
announce("legacy GUID-path migration preserves file hashes and inodes")
|
||||
@@ -872,8 +881,8 @@ fi
|
||||
rules(2)
|
||||
allowed("cd Permissions; put /tmp/live-note.txt created.txt")
|
||||
allowed("cd Permissions; put /tmp/live-note.txt existing.txt")
|
||||
engine_run("exec", CLIENT_CONTAINER, "sh", "-c", "printf 'LIVE_CONTENT742 from SMB\n' > /tmp/live-search.txt")
|
||||
allowed("cd Permissions; put /tmp/live-search.txt live-search.txt")
|
||||
engine_run("exec", CLIENT_CONTAINER, "python3", "-c", "import zipfile; z=zipfile.ZipFile('/tmp/live-search.docx','w'); z.writestr('word/document.xml','<document><body><p><t>LIVE_CONTENT742 from SMB</t></p></body></document>'); z.close()")
|
||||
allowed("cd Permissions; put /tmp/live-search.docx live-search.docx")
|
||||
live_row = eventually("SMB write becomes full-text searchable", lambda: http(query_path("/api/documents", {"q":"LIVE_CONTENT742"}), token=user_token),
|
||||
lambda r: r.status == 200 and r.json().get("total") == 1, timeout=45).json()["items"][0]
|
||||
|
||||
|
||||
@@ -50,6 +50,16 @@ font=next(Path('/usr/share/fonts').rglob('NimbusSans-Regular.otf'))
|
||||
draw=ImageDraw.Draw(image)
|
||||
draw.text((120,200),'INVOICE SCAN\nCustomer reference SCANNEDUNIQUE742\nInvoice total 1500 EUR\nPayment due 30 October 2026',fill='black',font=ImageFont.truetype(str(font),44),spacing=40)
|
||||
image.save(root/'scanned-invoice.pdf','PDF',resolution=150)
|
||||
(root/'index-excluded.txt').write_text('TEXTONLY742 should never be content-indexed')
|
||||
for suffix,entry in [('xlsx','xl/sharedStrings.xml'),('ods','content.xml')]:
|
||||
with zipfile.ZipFile(root/('index-excluded.'+suffix),'w') as spreadsheet:
|
||||
spreadsheet.writestr(entry,'<document><t>SPREADSHEETONLY742</t></document>')
|
||||
for name,marker in [('alice','ALICEPRIVATE742'),('bob','BOBPRIVATE742')]:
|
||||
with zipfile.ZipFile(Path('/data/private')/name/'readme.docx','w') as private:
|
||||
private.writestr('word/document.xml','<document><body><p><t>'+marker+'</t></p></body></document>')
|
||||
for suffix,entry,marker in [('pptx','ppt/slides/slide1.xml','POWERPOINTONLY742'),('odt','content.xml','ODTONLY742'),('odp','content.xml','ODPONLY742')]:
|
||||
with zipfile.ZipFile(root/('office-notes.'+suffix),'w') as office:
|
||||
office.writestr(entry,'<document><body><p><t>'+marker+'</t></p></body></document>')
|
||||
with zipfile.ZipFile(root/'office-notes.docx','w') as office:
|
||||
office.writestr('word/document.xml','<document><body><p><t>Office document OFFICETEXT742</t></p></body></document>')
|
||||
PYDOCUMENTS
|
||||
|
||||
Reference in New Issue
Block a user