import base64 import contextlib import functools import hashlib import http.server import io import json import mimetypes import pathlib import re import threading import time import unittest import urllib.parse import wave import zipfile try: from playwright.sync_api import Error as PlaywrightError from playwright.sync_api import sync_playwright except ImportError: PlaywrightError = Exception sync_playwright = None ROOT = pathlib.Path(__file__).resolve().parents[1] def vendor_name(key, pattern): resources = (ROOT / "static/reader-resources.js").read_text(encoding="utf-8") match = re.search(rf"\b{re.escape(key)}:\s*\{{.*?sha256:\s*\"([0-9a-f]+)\"", resources, re.S) if not match: raise RuntimeError(f"missing vendor digest for {key!r}") return pattern.replace("*", match.group(1)[:12], 1) VENDOR_FILES = { "pdf": vendor_name("pdf", "pdf.min.*.mjs"), "pdf_worker": vendor_name("pdfWorker", "pdf.worker.min.*.mjs"), "marked": vendor_name("marked", "marked.min.*.js"), "purify": vendor_name("purify", "purify.min.*.js"), "jszip": vendor_name("jszip", "jszip.min.*.js"), "docx": vendor_name("docx", "docx-preview.min.*.js"), } VENDOR_STEMS = { "pdf": "pdf.min.*.mjs", "pdf_worker": "pdf.worker.min.*.mjs", "marked": "marked.min.*.js", "purify": "purify.min.*.js", "jszip": "jszip.min.*.js", "docx": "docx-preview.min.*.js", } def route_local_vendor_fallback(page): for key, requested in VENDOR_FILES.items(): candidates = sorted((ROOT / "static/vendor").glob(VENDOR_STEMS[key])) if not candidates: continue page.route( f"**/static/vendor/{requested}", lambda route, _request, path=str(candidates[-1]): route.fulfill(path=path), ) SOURCE_URL = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.pdf" PDF_MODULE = """ export const GlobalWorkerOptions = {}; export class PDFWorker { promise = Promise.resolve(); destroy() {} } const wait = () => new Promise((resolve) => setTimeout(resolve, 15)); export function getDocument() { const page = { getViewport({ scale }) { return { width: 600 * scale, height: 800 * scale }; }, getTextContent() { return Promise.resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }); }, render() { window.__pdfActive = (window.__pdfActive || 0) + 1; window.__pdfPeak = Math.max(window.__pdfPeak || 0, window.__pdfActive); return { promise: wait().then(() => { window.__pdfActive -= 1; }) }; }, }; return { promise: Promise.resolve({ numPages: 30, getPage: () => Promise.resolve(page), getOutline: () => Promise.resolve(window.__pdfOutlineEnabled ? [{title: '第一章', dest: [{}], items: []}] : null), getPageIndex: () => Promise.resolve(0) }) }; } """ PDF_TASK_MODULE = """ export const GlobalWorkerOptions = {}; export class PDFWorker { promise = Promise.resolve(); destroy() {} } export function getDocument() { const probe = window.__pdfProbe ||= { loads: 0, destroys: 0, renders: [], cancels: [], releases: {} }; probe.loads++; const pdf = { numPages: 30, getOutline: async () => null, getPage: async (number) => ({ getViewport: ({scale}) => ({width: 600 * scale, height: 800 * scale, convertToViewportPoint: (x, y) => [x, y]}), async getTextContent() { if (probe.holdText) { probe.holdText = false; await new Promise(resolve => { probe.releaseText = resolve; }); } return {items: [{str: number === 1 ? 'obsolete' : number === 2 ? 'current' : 'ordinary', transform: [12, 0, 0, 12, 20, 40], hasEOL: false}]}; }, render() { probe.renders.push(number); return {promise: probe.holdPages?.includes(number) ? new Promise(resolve => { probe.releases[number] = resolve; }) : Promise.resolve(), cancel() { probe.cancels.push(number); }}; }, }), }; return {promise: window.__holdPdfLoad ? new Promise(resolve => { probe.releaseLoad = () => resolve(pdf); }) : Promise.resolve(pdf), destroy() { probe.destroys++; return Promise.resolve(); }}; } """ STORE_SCRIPT = """ window.__storeStartedAt = performance.now(); window.__readerBookmarks = []; window.VoiceOfMLReaderStore = Object.freeze({ get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300)), put: (entry) => { window.__savedReaderProgress = entry; return Promise.resolve(); }, list: () => Promise.resolve([]), remove: () => Promise.resolve(), clearHistory: () => Promise.resolve(), putBookmark: (entry) => { window.__readerBookmarks = window.__readerBookmarks.filter((item) => item.id !== entry.id).concat(entry); return Promise.resolve(); }, listBookmarks: (url) => Promise.resolve(window.__readerBookmarks.filter((item) => item.url === url)), listAllBookmarks: () => Promise.resolve([...window.__readerBookmarks].sort((a, b) => b.createdAt - a.createdAt)), removeBookmark: (id) => { window.__readerBookmarks = window.__readerBookmarks.filter((item) => item.id !== id); return Promise.resolve(); } }); """ MARKED_SCRIPT = "window.marked = { parse: (text) => '

' + text + '

' };" PURIFY_SCRIPT = "window.DOMPurify = { sanitize: (html) => html };" JSZIP_SCRIPT = "window.JSZip = function() {};" EPUB_SCRIPT = """ window.ePub = () => ({ renderTo: (frame) => ({ themes: { register() {}, select() {}, fontSize() {} }, on() {}, prev() {}, next() {}, display: () => new Promise((resolve) => setTimeout(() => { frame.textContent = 'EPUB readable'; resolve(); }, 20)), }) }); """ DOCX_SCRIPT = """ window.docx = { renderAsync: (_bytes, body) => new Promise((resolve) => setTimeout(() => { body.textContent = 'DOCX readable'; resolve(); }, 20)) }; """ PNG_BYTES = bytes.fromhex( "89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c489" "0000000d49444154789c6360f8cfc000000301010018dd8db10000000049454e44ae426082" ) IMAGE_FIXTURES = { "jpg": ("image/jpeg", base64.b64decode("/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAP//////////////////////////////////////////////////////////////////////////////////////2wBDAf//////////////////////////////////////////////////////////////////////////////////////wAARCAABAAEDASIAAhEBAxEB/8QAFQABAQAAAAAAAAAAAAAAAAAAAAf/xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oADAMBAAIQAxAAAAF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABBQJ//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAwEBPwF//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAgEBPwF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQAGPwJ//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABPyF//9oADAMBAAIAAwAAAB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAEDAQE/EB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAECAQE/EB//xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oACAEBAAE/EB//2Q==")), "jpeg": ("image/jpeg", base64.b64decode("/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAP//////////////////////////////////////////////////////////////////////////////////////2wBDAf//////////////////////////////////////////////////////////////////////////////////////wAARCAABAAEDASIAAhEBAxEB/8QAFQABAQAAAAAAAAAAAAAAAAAAAAf/xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oADAMBAAIQAxAAAAF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABBQJ//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAwEBPwF//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAgEBPwF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQAGPwJ//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABPyF//9oADAMBAAIAAwAAAB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAEDAQE/EB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAECAQE/EB//xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oACAEBAAE/EB//2Q==")), "gif": ("image/gif", base64.b64decode("R0lGODlhAQABAIAAAAAAAP///ywAAAAAAQABAAACAUwAOw==")), "bmp": ("image/bmp", bytes.fromhex("424d3a00000000000000360000002800000001000000010000000100180000000000040000000000000000000000000000000000000000000000")), "webp": ("image/webp", base64.b64decode("UklGRiIAAABXRUJQVlA4IBYAAAAwAQCdASoBAAEAAUAmJaQAA3AA/v89WAAAAA==")), } def minimal_pdf(): stream = b"BT /F1 18 Tf 20 100 Td (Reader PDF) Tj ET" objects = [ b"<< /Type /Catalog /Pages 2 0 R >>", b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>", b"<< /Length " + str(len(stream)).encode() + b" >>\nstream\n" + stream + b"\nendstream", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", ] payload = bytearray(b"%PDF-1.4\n%\xe2\xe3\xcf\xd3\n"); offsets = [0] for number, body in enumerate(objects, 1): offsets.append(len(payload)); payload.extend(f"{number} 0 obj\n".encode() + body + b"\nendobj\n") xref = len(payload); payload.extend(f"xref\n0 {len(objects) + 1}\n".encode()); payload.extend(b"0000000000 65535 f \n") for offset in offsets[1:]: payload.extend(f"{offset:010d} 00000 n \n".encode()) payload.extend(f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R >>\nstartxref\n{xref}\n%%EOF\n".encode()) return bytes(payload) def zip_bytes(files, stored_first=None): output = io.BytesIO() with zipfile.ZipFile(output, "w", zipfile.ZIP_DEFLATED) as archive: if stored_first: archive.writestr(stored_first[0], stored_first[1], compress_type=zipfile.ZIP_STORED) for name, body in files.items(): archive.writestr(name, body) return output.getvalue() def zip_bomb_metadata(): # A complete, decompressible archive that exceeds the per-entry ratio limit. return zip_bytes({"bomb.txt": b"x" * (1024 * 1024)}) def minimal_epub(): return zip_bytes({ "META-INF/container.xml": '', "OEBPS/content.opf": 'readerReaderen', "OEBPS/chapter.xhtml": 'Reader

EPUB readable

', }, ("mimetype", "application/epub+zip")) def epub_with_navigation(): return zip_bytes({ "META-INF/container.xml": '', "OEBPS/content.opf": 'reader-e2eReader E2Ezh', "OEBPS/nav.xhtml": '目录', "OEBPS/chapter-1.xhtml": '

第一章

第一章正文

', "OEBPS/chapter-2.xhtml": '

第二章

第二章正文

', }, ("mimetype", "application/epub+zip")) def epub_with_legacy_chm_markup(): return zip_bytes({ "META-INF/container.xml": '', "OEBPS/content.opf": 'legacy-chmLegacy CHMC', "OEBPS/style.css": "p { color: rgb(1, 2, 3); }", "OEBPS/picture.svg": '', "OEBPS/chapter.xhtml": 'Legacy CHMArticle titleLegacy CHM content正文第一段正文第二段图片之后的正文', }, ("mimetype", "application/epub+zip")) def epub_with_many_chapters(count=14): manifest = '' spine = '' links = [] files = {} for index in range(1, count + 1): manifest += f'' spine += f'' links.append(f'
  • 章节 {index}
  • ') files[f"OEBPS/chapter-{index}.xhtml"] = f'

    章节 {index}

    正文 {index}

    ' files.update({ "META-INF/container.xml": '', "OEBPS/content.opf": f'reader-raceReader Racezh{manifest}{spine}', "OEBPS/nav.xhtml": f'', }) return zip_bytes(files, ("mimetype", "application/epub+zip")) def minimal_docx(): return zip_bytes({ "[Content_Types].xml": '', "_rels/.rels": '', "word/document.xml": 'DOCX readable', }) def minimal_wav(): output = io.BytesIO() with wave.open(output, "wb") as audio: audio.setnchannels(1) audio.setsampwidth(2) audio.setframerate(8000) audio.writeframes(b"\0\0" * 800) return output.getvalue() class StaticHandler(http.server.SimpleHTTPRequestHandler): def guess_type(self, path): if path.endswith(".mjs") or path.endswith(".js"): return "text/javascript" return mimetypes.guess_type(path)[0] or "application/octet-stream" def log_message(self, *_args): pass @contextlib.contextmanager def static_server(): handler = functools.partial(StaticHandler, directory=str(ROOT)) server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler) thread = threading.Thread(target=server.serve_forever, daemon=True) thread.start() try: yield f"http://127.0.0.1:{server.server_port}" finally: server.shutdown() server.server_close() thread.join(timeout=5) @unittest.skipIf(sync_playwright is None, "install requirements-test.txt to run Reader performance tests") class ReaderPerformanceTest(unittest.TestCase): @classmethod def setUpClass(cls): cls.server = static_server() cls.origin = cls.server.__enter__() cls.playwright = sync_playwright().start() try: cls.browser = cls.playwright.chromium.launch(headless=True, args=["--no-sandbox"]) except PlaywrightError as error: cls.playwright.stop() cls.server.__exit__(None, None, None) raise unittest.SkipTest(f"Chromium is unavailable: {error}") @classmethod def tearDownClass(cls): cls.browser.close() cls.playwright.stop() cls.server.__exit__(None, None, None) def setUp(self): self.context = self.browser.new_context(viewport={"width": 1440, "height": 900}) self.page = self.context.new_page() self.pdf_requested_at = None self.page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) self.page.route("**/static/vendor/pdf.min.*.mjs", self.route_pdf) self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/pdf", body=b"pdf")) def tearDown(self): self.context.close() def route_pdf(self, route): self.pdf_requested_at = self.page.evaluate("performance.now()") route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE) def open_pdf(self): query = urllib.parse.quote(SOURCE_URL, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded") self.page.locator(".reader-page").nth(29).wait_for(state="attached") def test_reader_starts_with_session_metadata_before_document_load(self): errors = [] self.page.on("pageerror", lambda error: errors.append(str(error))) self.page.add_init_script(""" sessionStorage.setItem('reader-source:metadata-probe', JSON.stringify({ url: 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt', download: 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt', title: 'Metadata title', extension: 'txt', original_extension: 'txt', repo: 'Test', folder: ['Folder'] })); """) self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Metadata readable")) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt", safe="") self.page.goto(f"{self.origin}/static/reader.html?id=metadata-probe&url={source}&ext=txt", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for(state="visible") self.assertEqual(self.page.locator(".reader-text").text_content(), "Metadata readable") self.assertEqual(self.page.locator("#title").text_content(), "Metadata title.txt") self.assertEqual(self.page.locator("#reader-path").text_content(), "Test/Folder") self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "prepare") self.page.locator("html[data-reader-phase='ready']").wait_for(state="attached") self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "ready") self.assertEqual(errors, []) def test_fetch_file_aborts_when_pagehide_disposes_reader(self): self.page.add_init_script(r""" (() => { const nativeFetch = window.fetch.bind(window); const probe = window.__fetchFileProbe = { started: false, aborted: false, result: "" }; window.fetch = (input, init = {}) => { if (!String(input).includes("fetch-file-probe")) return nativeFetch(input, init); probe.started = true; return new Promise((resolve, reject) => { const signal = init && init.signal; const abort = () => { probe.aborted = true; probe.result = "aborted"; reject(new DOMException("Reader disposed", "AbortError")); }; if (signal && signal.aborted) return abort(); if (signal) signal.addEventListener("abort", abort, { once: true }); probe.resolve = () => { probe.result = "fulfilled"; resolve(new Response("probe", { status: 200, headers: { "content-type": "text/plain" } })); }; }); }; })(); """) self.page.unroute("**/api/reader-content**") self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader")) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file.txt", safe="") self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FetchFile", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for(state="visible") self.page.evaluate("""() => { window.__fetchFilePromise = window.fetchFile("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file-probe.txt") .then(() => { window.__fetchFileProbe.result = "fulfilled"; }) .catch((error) => { window.__fetchFileProbe.error = error.name; }); }""") self.page.wait_for_function("() => window.__fetchFileProbe.started === true") self.page.evaluate("""() => { const event = new Event("pagehide"); Object.defineProperty(event, "persisted", { value: true }); window.dispatchEvent(event); }""") self.page.wait_for_timeout(100) self.assertFalse(self.page.evaluate("() => window.__fetchFileProbe.aborted")) self.page.evaluate("window.dispatchEvent(new Event('pagehide'))") self.page.wait_for_function("() => window.__fetchFileProbe.result === 'aborted'", timeout=2000) self.assertEqual(self.page.evaluate("() => window.__fetchFileProbe.error"), "AbortError") self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed") def test_concurrent_fetch_file_callers_receive_complete_bodies(self): self.page.add_init_script(r""" (() => { const nativeFetch = window.fetch.bind(window); window.__concurrentFetchCalls = 0; window.fetch = (input, init = {}) => { if (!String(input).includes("concurrent-fetch-file")) return nativeFetch(input, init); window.__concurrentFetchCalls += 1; return Promise.resolve(new Response("complete shared body", { status: 200, headers: { "content-type": "text/plain" } })); }; })(); """) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file.txt", safe="") self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FetchFile", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for(state="visible") bodies = self.page.evaluate("""async () => { const url = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/concurrent-fetch-file.txt"; const files = await Promise.all([window.fetchFile(url), window.fetchFile(url)]); return Promise.all(files.map((file) => file.text())); }""") self.assertEqual(bodies, ["complete shared body", "complete shared body"]) self.assertEqual(self.page.evaluate("window.__concurrentFetchCalls"), 1) def test_id_only_resolver_is_lifecycle_managed(self): errors = [] self.page.on("pageerror", lambda error: errors.append(str(error))) self.page.add_init_script(r""" (() => { const nativeFetch = window.fetch.bind(window); window.__resolverProbe = { started: false, aborted: false }; window.fetch = (input, init = {}) => { if (!String(input).includes("/api/reader-resolve?id=resolver-abort")) return nativeFetch(input, init); window.__resolverProbe.started = true; return new Promise((resolve, reject) => { const abort = () => { window.__resolverProbe.aborted = true; reject(new DOMException("Reader disposed", "AbortError")); }; if (init.signal?.aborted) return abort(); init.signal?.addEventListener("abort", abort, { once: true }); }); }; })(); """) self.page.goto(f"{self.origin}/static/reader.html?id=resolver-abort", wait_until="domcontentloaded") self.page.wait_for_function("window.__resolverProbe.started") self.page.evaluate("window.dispatchEvent(new Event('pagehide'))") self.page.wait_for_function("window.__resolverProbe.aborted") self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed") self.assertEqual(errors, []) def test_id_only_resolver_failure_uses_reader_error_ui(self): errors = [] self.page.on("pageerror", lambda error: errors.append(str(error))) self.page.route("**/api/reader-resolve?id=resolver-failure", lambda route: route.fulfill(status=503, body="unavailable")) self.page.goto(f"{self.origin}/static/reader.html?id=resolver-failure", wait_until="domcontentloaded") self.page.locator(".reader-error").wait_for(state="visible") self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "failed") self.assertEqual(self.page.locator("#content").get_attribute("data-error-code"), "READER_NETWORK") self.assertEqual(errors, []) def test_id_only_reader_source_uses_authoritative_resolve(self): stored = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/stored.txt" authoritative = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/authoritative.txt" stored_data = { "url": stored, "download": stored, "title": "Stored title", "extension": "txt", "original_extension": "txt", "repo": "Test", "folder": ["Stored"], } self.page.add_init_script( f"sessionStorage.setItem('reader-source:id-only-authority', {json.dumps(json.dumps(stored_data))})" ) self.page.route( "**/api/reader-resolve?id=id-only-authority", lambda route: route.fulfill( status=200, content_type="application/json", body=json.dumps({"url": authoritative, "download": authoritative, "title": "Resolved title", "extension": "txt", "original_extension": "txt", "repo": "Test", "folder": "Authoritative"}), ), ) self.page.unroute("**/api/reader-content**") self.page.route( "**/api/reader-content**", lambda route: route.fulfill( status=200, content_type="text/plain", body=b"Authoritative reader" if "authoritative.txt" in route.request.url else b"Stored reader", ), ) self.page.goto(f"{self.origin}/static/reader.html?id=id-only-authority", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for(state="visible") self.assertEqual(self.page.locator(".reader-text").text_content(), "Authoritative reader") self.assertEqual(self.page.locator("#title").text_content(), "Resolved title.txt") self.assertEqual(self.page.locator("#reader-path").text_content(), "Test/Authoritative") self.assertNotIn("url=", self.page.url) def test_id_only_reader_falls_back_to_session_source_when_resolve_fails(self): stored = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/stored.txt" stored_data = { "url": stored, "download": stored, "title": "Stored title", "extension": "txt", "original_extension": "txt", "repo": "Test", "folder": ["Stored"], } self.page.add_init_script( f"sessionStorage.setItem('reader-source:id-only-fallback', {json.dumps(json.dumps(stored_data))})" ) self.page.route("**/api/reader-resolve?id=id-only-fallback", lambda route: route.fulfill(status=503, body="unavailable")) self.page.unroute("**/api/reader-content**") self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Stored reader")) self.page.goto(f"{self.origin}/static/reader.html?id=id-only-fallback", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for(state="visible") self.assertEqual(self.page.locator(".reader-text").text_content(), "Stored reader") self.assertEqual(self.page.locator("#title").text_content(), "Stored title.txt") def test_document_preparation_overlaps_delayed_history_restore(self): self.open_pdf() store_started = self.page.evaluate("window.__storeStartedAt") self.assertIsNotNone(self.pdf_requested_at) self.assertLess(self.pdf_requested_at - store_started, 250) def test_pdf_rendering_has_bounded_concurrency_and_canvas_memory(self): self.open_pdf() metrics = self.scroll_document() self.assertLessEqual(metrics["peak"], 2) self.assertLessEqual(metrics["rendered"], 11) self.assertGreater(metrics["pixels"], 0) def test_pdf_exposes_lazy_accessible_text(self): self.open_pdf() first_page = self.page.locator(".reader-page").first self.assertEqual(first_page.get_attribute("role"), "region") self.assertEqual(first_page.locator("canvas").get_attribute("aria-hidden"), "true") self.page.wait_for_function("document.querySelector('.reader-page')?.dataset.textReady === '1'") self.assertEqual(first_page.locator(".reader-pdf-text").text_content(), "Accessible PDF text") def test_native_pdf_canvas_does_not_wait_for_text_layer(self): module = PDF_MODULE.replace( "getTextContent() { return Promise.resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }); },", "getTextContent() { return new Promise(resolve => setTimeout(() => resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }), 1000)); },", ) self.page.unroute("**/static/vendor/pdf.min.*.mjs") self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill( content_type="text/javascript", body=module)) self.open_pdf() self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for(timeout=3000) self.assertNotEqual( self.page.locator('.reader-page[data-page="1"]').get_attribute("data-text-ready"), "1" ) self.page.locator('.reader-page[data-page="1"][data-text-ready="1"]').wait_for(timeout=3000) def test_scanned_pdf_bookmark_has_empty_excerpt(self): self.page.route('**/static/vendor/pdf.min.*.mjs', lambda route: route.fulfill( content_type='text/javascript', body=PDF_MODULE.replace("[{ str: 'Accessible PDF text', hasEOL: false }]", '[]'))) self.open_pdf() self.page.wait_for_function("document.querySelector('.reader-page')?.dataset.textReady === '1'") self.assertEqual(self.page.locator('.reader-pdf-text').first.text_content(), '') self.page.locator('#bookmark-ribbon').click() self.assertEqual(self.page.locator('#bookmark-excerpt-input').input_value(), '') self.page.locator('#bookmark-add').click() self.page.wait_for_function('window.__readerBookmarks.length === 1') self.assertEqual(self.page.evaluate('window.__readerBookmarks[0].excerpt'), '') def test_reader_panel_bookmark_search_and_theme(self): self.page.add_init_script("window.__pdfOutlineEnabled = true") self.open_pdf() self.page.locator("#bookmark-ribbon").click() self.assertEqual(self.page.locator("#bookmark-popover").get_attribute("role"), "dialog") self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "true") self.page.locator("#bookmark-add").press("Escape") self.assertTrue(self.page.locator("#bookmark-popover").is_hidden()) self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "false") self.page.locator("#bookmark-ribbon").click() self.assertIn("第 1 / 30 页", self.page.locator("#bookmark-prompt").text_content()) self.page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.page === 1") self.page.locator("#bookmark-add").click() self.page.locator("#history").click() self.assertTrue(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')")) self.assertNotEqual(self.page.locator("#history-panel").evaluate("element => getComputedStyle(element).transitionDuration"), "0s") self.assertTrue(self.page.locator("#toc-tab").is_visible()) self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc") self.assertEqual(self.page.locator(".reader-panel-tabs").get_attribute("role"), "tablist") self.page.locator("#toc-tab").focus() self.page.locator("#toc-tab").press("ArrowRight") self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "bookmarks") self.page.locator("#bookmarks-tab").press("ArrowLeft") self.assertEqual(self.page.locator("#toc-list .panel-item-main").get_attribute("role"), "link") self.assertEqual(self.page.locator("#toc-list .panel-item-main").evaluate("element => getComputedStyle(element).userSelect"), "text") self.assertEqual(self.page.locator("#toc-panel .panel-search-toggle").text_content(), "搜索") self.assertEqual(self.page.locator("#history-panel > footer").count(), 0) self.assertEqual(self.page.locator("#history-panel > header #theme-toggle").count(), 1) self.assertEqual(self.page.locator("#history-panel > header .icon-btn").count(), 0) self.assertEqual(self.page.locator("#history-clear").text_content(), "清空历史") self.assertLess(self.page.locator("#history-clear").evaluate("element => [...element.parentElement.children].indexOf(element)"), self.page.locator("#history-view .panel-search-toggle").evaluate("element => [...element.parentElement.children].indexOf(element)")) self.assertEqual(self.page.locator("#history-clear").evaluate("element => getComputedStyle(element).alignItems"), "center") self.assertEqual(self.page.locator("#history-view .panel-search-toggle").evaluate("element => getComputedStyle(element).transform"), "none") self.page.locator('.reader-panel-tabs button[data-panel="bookmarks"]').click() self.page.locator("#bookmarks-list .panel-item-main").filter(has_text="第 1 / 30 页").wait_for() self.page.locator("#bookmarks-panel .panel-search-toggle").click() self.assertTrue(self.page.locator("#bookmarks-panel .panel-search").evaluate("element => element.classList.contains('is-open')")) self.page.locator("#bookmarks-panel .panel-search").fill("不存在") self.assertTrue(self.page.locator("#bookmarks-list .panel-item").is_hidden()) self.page.locator("#history").click() self.page.locator("#history").click() self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc") self.page.locator('.reader-panel-tabs button[data-panel="bookmarks"]').click() self.page.locator("#history").click() self.page.locator("#history").click() self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc") self.page.locator("#theme-toggle").click() self.assertEqual(self.page.locator("html").get_attribute("data-theme"), "light") self.assertTrue(self.page.locator("html").evaluate("element => element.classList.contains('theme-transition')")) self.page.wait_for_timeout(300) self.assertNotEqual(self.page.locator(".compact-input").first.evaluate("element => getComputedStyle(element).backgroundColor"), "rgb(37, 41, 45)") self.page.locator("#page-prev").hover() self.assertNotEqual(self.page.locator("#page-prev").evaluate("element => getComputedStyle(element).backgroundColor"), "rgb(41, 45, 49)") self.assertEqual(self.page.locator("#zoom").get_attribute("min"), "25") self.assertEqual(self.page.locator("#zoom").get_attribute("max"), "400") def test_reader_controls_fit_viewport_without_overlap_and_work_on_mobile(self): self.open_pdf() def assert_toolbar_layout(): layout = self.page.locator(".reader-toolbar").evaluate("""toolbar => { const view = {width: innerWidth, height: innerHeight}; const selectors = ['#back', '#page-prev', '#page-number', '#page-next', '#zoom-out', '#zoom', '#zoom-in', '#history', '#download']; const rects = selectors.map(selector => { const element = document.querySelector(selector); const rect = element.getBoundingClientRect(); return {selector, left: rect.left, top: rect.top, right: rect.right, bottom: rect.bottom, width: rect.width, height: rect.height, visible: !!(rect.width && rect.height)}; }).filter(item => item.visible); return {toolbar: toolbar.getBoundingClientRect().toJSON(), view, rects}; }""") self.assertGreaterEqual(layout["toolbar"]["height"], 36) for item in layout["rects"]: self.assertGreater(item["width"], 0, item["selector"]) self.assertGreaterEqual(item["left"], 0, item["selector"]) self.assertLessEqual(item["right"], layout["view"]["width"] + 1, item["selector"]) self.assertGreaterEqual(item["top"], 0, item["selector"]) self.assertLessEqual(item["bottom"], layout["toolbar"]["bottom"] + 1, item["selector"]) for index, first in enumerate(layout["rects"]): for second in layout["rects"][index + 1:]: overlap = first["left"] < second["right"] and second["left"] < first["right"] and first["top"] < second["bottom"] and second["top"] < first["bottom"] self.assertFalse(overlap, f'{first["selector"]} overlaps {second["selector"]}') assert_toolbar_layout() self.page.locator("#page-next").click() self.assertEqual(self.page.locator("#page-number").input_value(), "2") self.page.locator("#page-prev").click() self.assertEqual(self.page.locator("#page-number").input_value(), "1") self.page.locator("#zoom-in").click() self.assertEqual(self.page.locator("#zoom").input_value(), "110") self.page.locator("#zoom-out").click() self.assertEqual(self.page.locator("#zoom").input_value(), "100") self.page.locator("#history").click() self.assertTrue(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')")) self.page.locator("#history-close").click() self.assertFalse(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')")) mobile_context = self.browser.new_context(viewport={"width": 390, "height": 844}) mobile_page = mobile_context.new_page() mobile_page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) mobile_page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE)) mobile_page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/pdf", body=b"pdf")) query = urllib.parse.quote(SOURCE_URL, safe="") mobile_page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded") mobile_page.locator(".reader-page").nth(29).wait_for(state="attached") mobile_layout = mobile_page.locator(".reader-toolbar").evaluate("""toolbar => { const view = {width: innerWidth, height: innerHeight}; const rects = [...toolbar.querySelectorAll('button, input, a')].map(element => { const rect = element.getBoundingClientRect(); return {left: rect.left, right: rect.right, top: rect.top, bottom: rect.bottom, width: rect.width, height: rect.height, visible: !!(rect.width && rect.height)}; }).filter(item => item.visible); return {toolbar: toolbar.getBoundingClientRect().toJSON(), view, rects}; }""") self.assertEqual(mobile_layout["view"]["width"], 390) self.assertGreaterEqual(mobile_layout["toolbar"]["height"], 36) for item in mobile_layout["rects"]: self.assertGreaterEqual(item["left"], 0) self.assertLessEqual(item["right"], 390) self.assertLessEqual(item["bottom"], mobile_layout["toolbar"]["bottom"] + 1) mobile_page.locator("#history").click() self.assertTrue(mobile_page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')")) mobile_page.wait_for_timeout(300) panel_box = mobile_page.locator("#history-panel").bounding_box() self.assertIsNotNone(panel_box) self.assertGreaterEqual(panel_box["x"], 0) self.assertLessEqual(panel_box["x"] + panel_box["width"], 390) mobile_context.close() def test_reader_controls_honor_boundaries_and_keyboard_activation(self): self.open_pdf() self.assertEqual(self.page.locator("#page-number").input_value(), "1") self.page.locator("#page-prev").click() self.assertEqual(self.page.locator("#page-number").input_value(), "1") self.page.locator("#page-next").focus() self.page.locator("#page-next").press("Enter") self.assertEqual(self.page.locator("#page-number").input_value(), "2") self.page.locator("#page-number").fill("999") self.page.locator("#page-number").press("Enter") self.page.wait_for_function("() => document.querySelector('#page-number').value === '30'") self.assertEqual(self.page.locator("#page-number").input_value(), "30") self.page.locator("#page-number").fill("0") self.page.locator("#page-number").press("Enter") self.page.wait_for_function("() => document.querySelector('#page-number').value === '1'") self.assertEqual(self.page.locator("#page-number").input_value(), "1") self.page.locator("#zoom").fill("999") self.page.locator("#zoom").press("Enter") self.page.wait_for_function("() => document.querySelector('#zoom').value === '400'") self.assertEqual(self.page.locator("#zoom").input_value(), "400") self.page.locator("#zoom-in").click() self.assertEqual(self.page.locator("#zoom").input_value(), "400") self.page.locator("#zoom").fill("1") self.page.locator("#zoom").press("Enter") self.page.wait_for_function("() => document.querySelector('#zoom').value === '25'") self.assertEqual(self.page.locator("#zoom").input_value(), "25") self.page.locator("#zoom-out").click() self.assertEqual(self.page.locator("#zoom").input_value(), "25") self.page.locator("#history").focus() self.page.locator("#history").press("Enter") self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "true") self.page.locator("#history-close").press("Enter") self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "false") self.assertTrue(self.page.locator("#download").get_attribute("href")) self.assertEqual(self.page.locator("#download").get_attribute("target"), "_blank") self.assertIn("noopener", self.page.locator("#download").get_attribute("rel")) def test_format_modes_expose_matching_controls_and_bookmark_ui(self): cases = [ ("pdf", "pdf", "30 页", ".reader-page", "application/pdf", b"pdf", False, False), ("txt", "text", "已加载", ".reader-text", "text/plain", b"Text readable", True, False), ("md", "markdown", "已加载", ".reader-markdown", "text/markdown", b"# Markdown readable", True, False), ("html", "html", "HTML", "iframe.html-frame", "text/html", b"

    HTML readable

    ", True, False), ("png", "image", "图片", ".reader-image", "image/png", PNG_BYTES, True, False), ("docx", "docx", "DOCX", ".docx-body", "application/vnd.openxmlformats-officedocument.wordprocessingml.document", minimal_docx(), False, False), ("wav", "audio", "音频", ".reader-audio", "audio/wav", minimal_wav(), True, True), ] for extension, mode, status, content_selector, content_type, body, page_hidden, zoom_hidden in cases: with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE)) if extension == "md": page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=MARKED_SCRIPT)) page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT)) if extension == "docx": page.route("**/static/vendor/jszip.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=JSZIP_SCRIPT)) page.route("**/static/vendor/docx-preview.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=DOCX_SCRIPT)) page.route("**/api/reader-content**", lambda route, _request, content_type=content_type, body=body: route.fulfill(status=200, content_type=content_type, body=body)) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/matrix.{extension}" if extension == "docx": source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/docx-native-v1/document.docx" page.route("**/api/reader-resolve?id=capability-matrix", lambda route: route.fulfill( status=200, content_type="application/json", body=json.dumps({"url": source, "extension": extension, "title": "Matrix"}), )) page.goto(f"{self.origin}/static/reader.html?id=capability-matrix", wait_until="domcontentloaded") page.wait_for_function("expected => document.querySelector('#status').textContent === expected", arg=status) self.assertEqual(page.locator(".reader-content").get_attribute("data-mode"), mode) page.locator(content_selector).first.wait_for(state="attached") self.assertTrue(page.locator("#bookmark-ribbon").is_visible()) ribbon = page.locator("#bookmark-ribbon").bounding_box() self.assertIsNotNone(ribbon) self.assertGreaterEqual(ribbon["x"], 0) self.assertLessEqual(ribbon["x"] + ribbon["width"], 390) self.assertEqual(page.locator(".page-controls").is_hidden(), page_hidden) self.assertEqual(page.locator(".zoom-controls").is_hidden(), zoom_hidden) self.assertEqual(page.locator("#full-search-toggle").evaluate("node => node.hidden"), mode in ("image", "audio")) self.assertEqual(page.locator("#media-tab").evaluate("node => node.hidden"), mode != "audio") self.assertFalse(page.locator(".reader-progress-bookmark").evaluate("node => node.hidden")) if not zoom_hidden: page.locator("#zoom-in").click() self.assertEqual(page.locator("#zoom").input_value(), "110") page.locator("#bookmark-ribbon").press("Enter") self.assertTrue(page.locator("#bookmark-popover").is_visible()) page.locator("#bookmark-cancel").press("Escape") self.assertTrue(page.locator("#bookmark-popover").is_hidden()) context.close() def test_video_failure_shows_recoverable_reader_error(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="video/mp4", body=b"invalid video fixture")) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/broken.mp4" page.route(source, lambda route: route.abort()) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=mp4&title=Broken", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertIn("媒体加载失败", page.locator(".reader-error").text_content()) self.assertEqual(page.locator("#status").text_content(), "无法打开") self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_MEDIA") self.assertFalse(page.locator(".reader-loading-indicator").count()) context.close() def test_video_extension_aliases_report_media_errors_consistently(self): for extension in ("mp4", "mov", "video"): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="video/mp4", body=b"invalid video fixture")) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/broken.{extension}" page.route(source, lambda route: route.abort()) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Broken", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertEqual(page.locator("#status").text_content(), "无法打开") context.close() def test_unsupported_format_hides_inapplicable_controls(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/archive.zip" page.route("**/api/reader-resolve?id=unsupported-controls", lambda route: route.fulfill( status=200, content_type="application/json", body=json.dumps({"url": source, "extension": "zip"}), )) page.goto(f"{self.origin}/static/reader.html?id=unsupported-controls", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertIn("此文件暂不支持在线阅读", page.locator(".reader-error").text_content()) self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_UNSUPPORTED") self.assertTrue(page.locator(".page-controls").is_hidden()) self.assertTrue(page.locator(".zoom-controls").is_hidden()) for selector in ("#full-search-toggle", "#media-tab", "#bookmark-ribbon", ".reader-progress-bookmark"): self.assertTrue(page.locator(selector).evaluate("node => node.hidden"), selector) self.assertFalse(page.locator(".reader-loading-indicator").count()) context.close() def test_converted_pdf_pages_reject_invalid_manifest_and_missing_first_page(self): cases = [ ("old-v1", {"version": 1, "kind": "pdf-pages", "pages": [{"page": 1, "path": f"objects/aa/{'a' * 64}/pages/page-000001.webp"}]}), ("bad-manifest", {"version": 2, "kind": "pdf-pages", "page_count": 0}), ("wrong-kind", {"version": 2, "kind": "pdf", "page_count": 1}), ("missing-page-count", {"version": 2, "kind": "pdf-pages"}), ("pages-field", {"version": 2, "kind": "pdf-pages", "page_count": 1, "pages": []}), ] for name, manifest in cases: with self.subTest(case=name): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/page-manifest.json" page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route, _request, manifest=manifest: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))) page.route("https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/**", lambda route: route.fulfill(status=404, body=b"")) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertTrue(page.locator(".reader-error").text_content().strip()) self.assertEqual(page.locator("#status").text_content(), "无法打开") self.assertFalse(page.locator(".reader-loading-indicator").count()) context.close() def test_converted_pdf_pages_report_error_when_later_page_is_missing(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() page.route("https://huggingface.co/**", lambda route: route.abort()) source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef/page-manifest.json" root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" manifest = {"version": 2, "kind": "pdf-pages", "page_count": 4} missing_requests = [] page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))) def serve_page(route, _request): if not route.request.url.endswith("page-000004.webp"): route.fulfill(status=200, content_type="image/webp", body=IMAGE_FIXTURES["webp"][1]) else: missing_requests.append(route.request.url) route.fulfill(status=404, body=b"") page.route(f"{root}/pages/**", serve_page) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded") page.locator(".reader-page img.ready").first.wait_for(timeout=10000) page.locator(".reader-page[data-page='4']").scroll_into_view_if_needed() page.wait_for_function("""() => { const shell = document.querySelector('.reader-page[data-page="4"]'); return shell?.dataset.renderState === 'idle' && Number(shell.dataset.renderRetries) > 0; }""") self.assertTrue(missing_requests) self.assertTrue(page.locator(".reader-page[data-page='1'] img.ready").evaluate( "image => image.complete && image.naturalWidth > 0")) self.assertNotEqual(page.locator(".reader-page[data-page='4']").get_attribute("data-render-state"), "rendered") self.assertEqual(page.locator(".reader-page[data-page='4'] img.ready").count(), 0) context.close() def test_pdf_pages_stalled_early_manifest_uses_proxy(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" source = root + "/page-manifest.json" proxy = [] page.add_init_script(f"""(() => {{ const fetchOriginal = window.fetch.bind(window); window.__earlyManifestAborted = false; window.fetch = (url, options = {{}}) => String(url) === {json.dumps(source)} ? new Promise((resolve, reject) => options.signal.addEventListener('abort', () => {{ window.__earlyManifestAborted = true; reject(new DOMException('cancelled', 'AbortError')); }}, {{once: true}})) : fetchOriginal(url, options); }})()""") page.route("**/api/reader-content**", lambda route: (proxy.append(route.request.url), route.fulfill( content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 2})))) page.route(f"{root}/pages/**", lambda route: route.fulfill( content_type="image/webp", body=IMAGE_FIXTURES["webp"][1])) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded") page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=8000) self.assertTrue(page.evaluate('window.__earlyManifestAborted')) self.assertTrue(proxy) def test_pdf_pages_progressive_prefetch_yields_to_jump(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" source = root + "/page-manifest.json" held, requested = [], [] page.route("**/api/reader-content**", lambda route: route.fulfill( content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 40}))) def serve_page(route): number = int(route.request.url.rsplit("page-", 1)[1].split(".", 1)[0]) requested.append(number) if number == 12: held.append(route) else: route.fulfill(content_type="image/webp", body=IMAGE_FIXTURES["webp"][1]) page.route(f"{root}/pages/**", serve_page) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded") page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=10000) page.wait_for_function("() => document.querySelector('.reader-page[data-page=\"12\"]')") page.wait_for_timeout(2800) self.assertIn(12, requested) self.assertIn(13, requested) self.assertIn(14, requested) with page.expect_response(lambda response: response.url.endswith("page-000035.webp"), timeout=12000): page.locator('.reader-page[data-page="25"]').scroll_into_view_if_needed() page.locator('.reader-page[data-page="25"] img.ready').wait_for(timeout=10000) for route in held: try: route.abort() except PlaywrightError: pass self.assertIn(35, requested) def test_pdf_pages_jump_does_not_wait_for_stalled_background_image(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() page.add_init_script("Object.defineProperty(navigator, 'connection', {value: {saveData: true}, configurable: true})") root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" source = root + "/page-manifest.json" held, requested = [], [] page.route("**/api/reader-content**", lambda route: route.fulfill( content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 40}))) def serve_page(route): number = int(route.request.url.rsplit("page-", 1)[1].split(".", 1)[0]) requested.append(number) if number == 2: held.append(route) else: route.fulfill(content_type="image/webp", body=IMAGE_FIXTURES["webp"][1]) page.route(f"{root}/pages/**", serve_page) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded") page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=10000) page.wait_for_function("""() => document.querySelector('.reader-page[data-page="2"]')?._renderStarted""", timeout=10000) page.locator("#page-number").fill("25") page.locator("#page-number").dispatch_event("change") page.wait_for_function("""() => document.querySelector('#viewport').scrollTop >= document.querySelector('.reader-page[data-page="25"]').offsetTop - 2""", timeout=3000) page.locator('.reader-page[data-page="25"] img.ready').wait_for(timeout=3000) self.assertIn(25, requested) for route in held: try: route.abort() except PlaywrightError: pass def test_converted_pdf_pages_fit_wide_images_without_overlap(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() page.route("https://huggingface.co/**", lambda route: route.abort()) source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef/page-manifest.json" root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" manifest = {"version": 2, "kind": "pdf-pages", "page_count": 3, "toc": [{"title": "第一章", "page": 1, "depth": 0}, {"title": "第二章", "page": 2, "depth": 1}]} wide_page = b'' page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))) page.route(f"{root}/pages/**", lambda route: route.fulfill(status=200, content_type="image/svg+xml", body=wide_page)) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded") page.wait_for_function("""() => { const images = [...document.querySelectorAll('.reader-page img.ready')]; return images.length === 3 && images.every(image => image.complete && image.naturalWidth === 2400 && image.naturalHeight === 3200); }""", timeout=10000) boxes = page.locator(".reader-page").evaluate_all("""pages => pages.slice(0, 3).map(page => { const shell = page.getBoundingClientRect(), image = page.querySelector('img').getBoundingClientRect(); return { shell: { top: shell.top, right: shell.right, bottom: shell.bottom, left: shell.left, width: shell.width }, image: { top: image.top, right: image.right, bottom: image.bottom, left: image.left, width: image.width } }; })""") self.assertEqual(len(boxes), 3) for box in boxes: self.assertLessEqual(box["image"]["width"], box["shell"]["width"] + 1) self.assertGreaterEqual(box["image"]["left"], box["shell"]["left"] - 1) self.assertLessEqual(box["image"]["right"], box["shell"]["right"] + 1) for current, following in zip(boxes, boxes[1:]): self.assertGreaterEqual(following["shell"]["top"], current["image"]["bottom"]) page.locator("#history").click() self.assertEqual(page.locator("#toc-list .toc-item").count(), 2) self.assertIn("第二章 · 第 2 页", page.locator("#toc-list .toc-item").nth(1).text_content()) page.locator("#toc-list .toc-item").nth(1).click() page.wait_for_function("() => document.querySelector('#page-number').value === '2'") context.close() def test_compact_pdf_manifest_virtualizes_and_navigates_to_distant_page(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef" source = root + "/page-manifest.json" manifest = {"version": 2, "kind": "pdf-pages", "source_sha256": "a" * 64, "profile": "test", "page_count": 5000} page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route(source, lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))) page.route(f"{root}/pages/**", lambda route: route.fulfill(status=200, content_type="image/webp", body=IMAGE_FIXTURES["webp"][1])) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Compact", wait_until="domcontentloaded") page.locator(".reader-page img.ready").first.wait_for(timeout=10000) self.assertLessEqual(page.locator(".reader-page img").count(), 25) page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * 0.5") page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 1500") self.assertLessEqual(page.locator(".reader-page").count(), 160) page.locator("#page-number").fill("4000") page.locator("#viewport").evaluate("node => { node.scrollTop += node.clientHeight * 1.2; node.dispatchEvent(new Event('scroll')); }") page.wait_for_timeout(80) self.assertEqual(page.locator("#page-number").input_value(), "4000") page.locator("#page-number").dispatch_event("change") page.locator(".reader-page[data-page='4000'] img.ready").wait_for(timeout=10000) self.assertEqual(page.locator("#page-number").input_value(), "4000") self.assertLessEqual(page.locator(".reader-page").count(), 160) self.assertEqual(page.locator(".reader-page[data-page='1']").count(), 0) page.locator("#page-number").fill("1") page.locator("#page-number").dispatch_event("change") page.locator(".reader-page[data-page='1'] img.ready").wait_for(timeout=10000) self.assertLessEqual(page.locator(".reader-page").count(), 160) self.assertLessEqual(page.locator(".reader-page img").count(), 25) context.close() def test_long_native_pdf_keeps_bounded_page_shells(self): module = PDF_TASK_MODULE.replace("numPages: 30", "numPages: 1200") self.page.unroute("**/static/vendor/pdf.min.*.mjs") self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill( content_type="text/javascript", body=module)) self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(SOURCE_URL, safe='')}&ext=pdf", wait_until="domcontentloaded") self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for() self.page.evaluate("""() => { const run = document.querySelector('.reader-page[data-page="1"] .reader-pdf-text-run'); const range = document.createRange(); range.selectNodeContents(run); getSelection().removeAllRanges(); getSelection().addRange(range); }""") self.page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * .84") self.page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 900") self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 1) self.assertEqual(self.page.evaluate("getSelection().toString()"), self.page.locator('.reader-page[data-page="1"] .reader-pdf-text-run').first.text_content()) self.page.evaluate("() => { getSelection().removeAllRanges(); document.activeElement.blur(); }") self.page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * .92") self.page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 1000") self.page.wait_for_function("() => !document.querySelector('.reader-page[data-page=\"1\"]')") self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 0) self.assertLessEqual(self.page.locator(".reader-page").count(), 160) self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 0) self.page.locator("#page-number").fill("1") self.page.locator("#page-number").dispatch_event("change") self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for() self.assertLessEqual(self.page.locator(".reader-page").count(), 160) def test_reader_toolbar_controls_have_accessible_names_and_state(self): self.open_pdf() controls = self.page.locator(".reader-toolbar button, .reader-toolbar a") names = self.page.locator(".reader-toolbar button:visible, .reader-toolbar a:visible").evaluate_all("""elements => elements.map(element => ({ id: element.id, name: element.getAttribute('aria-label') || element.getAttribute('title') || element.textContent.trim(), type: element.tagName === 'BUTTON' ? element.type : '', target: element.tagName === 'A' ? element.target : '', rel: element.tagName === 'A' ? element.rel : '' }))""") self.assertGreater(controls.count(), 0) for item in names: self.assertTrue(item["name"], item["id"]) if item["type"]: self.assertEqual(item["type"], "button", item["id"]) if item["id"] == "download": self.assertEqual(item["target"], "_blank") self.assertIn("noopener", item["rel"]) self.page.locator("#history").click() self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "true") self.page.locator("#theme-toggle").click() self.assertIn(self.page.locator("#theme-toggle").get_attribute("aria-label"), ("切换到白天模式", "切换到夜间模式")) self.page.locator("#bookmark-ribbon").press("Enter") self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "true") self.page.locator("#bookmark-cancel").press("Escape") self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "false") def test_text_bookmark_uses_progress_excerpt_and_highlights_search(self): self.page.unroute("**/api/reader-content**") text = "\n\n".join(f"第 {index} 段 searchable-{index} 这是用于书签摘要搜索的正文内容。" * 5 for index in range(120)) self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain; charset=utf-8", body=text.encode())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.txt" query = urllib.parse.quote(source, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Bookmark", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for() self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.5; element.dispatchEvent(new Event('scroll')); }") self.page.locator("#bookmark-ribbon").click() self.page.locator("#bookmark-label").fill("我的书签") self.page.locator("#bookmark-excerpt-input").fill("自定义摘要") self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.35; element.dispatchEvent(new Event('scroll')); }") self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0) self.assertIn("阅读进度", self.page.locator("#bookmark-prompt").text_content()) self.assertNotIn("px", self.page.locator("#bookmark-prompt").text_content()) self.assertRegex(self.page.locator("#bookmark-prompt").text_content(), r"阅读进度 \d+\.\d%") self.assertTrue(self.page.locator("#bookmark-excerpt-input").input_value()) prompt_progress = float(self.page.locator("#bookmark-prompt").text_content().split("阅读进度 ", 1)[1].split("%", 1)[0]) self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.8; element.dispatchEvent(new Event('scroll')); }") self.page.locator("#bookmark-add").click() self.page.wait_for_function("() => window.__readerBookmarks.length === 1") bookmark = self.page.evaluate("window.__readerBookmarks[0]") self.assertEqual(bookmark["label"], "我的书签") self.assertEqual(bookmark["excerpt"], "自定义摘要") self.assertGreater(bookmark["progress"], 0) self.assertAlmostEqual(bookmark["progress"], prompt_progress, places=1) self.assertTrue(bookmark["excerpt"]) self.page.evaluate("window.__readerBookmarks.push({id: 'other', url: 'https://example.test/other.txt', title: '另一本书', label: '阅读进度 12.3%', excerpt: '跨书摘要', readerUrl: location.href, createdAt: Date.now() + 1})") self.page.locator("#history").click() self.page.locator("#bookmarks-tab").click() self.page.locator("#bookmarks-list .bookmark-excerpt").wait_for() self.assertEqual(self.page.locator("#bookmarks-list .panel-item").count(), 1) self.page.locator("#bookmarks-all").click() self.page.locator("#bookmarks-list .panel-item").nth(1).wait_for() self.assertEqual(self.page.locator("#bookmarks-all").text_content(), "本书书签") self.assertIn("另一本书", self.page.locator("#bookmarks-list").text_content()) term = bookmark["excerpt"].split()[0][:6] self.page.locator("#bookmarks-panel .panel-search-toggle").click() self.page.locator("#bookmarks-panel .panel-search").fill(term) self.assertGreater(self.page.locator("#bookmarks-list mark.search-match").count(), 0) self.page.locator("#bookmarks-panel .panel-search").fill("") self.page.locator("#bookmarks-list .panel-item-edit").first.click() self.page.locator("#bookmark-label").fill("修改后的标题") self.page.locator("#bookmark-excerpt-input").fill("修改后的摘要") self.page.locator("#bookmark-add").click() self.assertIn("修改后的标题", self.page.locator("#bookmarks-list").text_content()) def test_full_text_search_lists_highlighted_snippets_and_jumps(self): self.page.unroute("**/api/reader-content**") text = ("开头内容。" * 80) + "正文目标词出现在这里,前后都有上下文。" + ("中间内容。" * 120) + "正文目标词再次出现。" self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=text.encode())) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/full-search.txt", safe="") self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FullSearch", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for() self.assertEqual(self.page.locator("#full-search-view").count(), 1) self.assertEqual(self.page.locator(".full-search-toggle").count(), 0) self.page.locator("#history").click() self.page.locator("#full-search-toggle").click() self.assertFalse(self.page.locator("#full-search-view").is_hidden()) self.page.locator("#full-search-toggle").click() self.page.wait_for_timeout(300) self.assertTrue(self.page.locator("#history-panel").is_visible()) self.assertTrue(self.page.locator("#full-search-view").is_hidden()) self.page.locator("#full-search-toggle").click() self.page.locator("#full-search-input").fill("目标词") self.page.locator("#full-search-status").filter(has_text="2 个结果").wait_for() self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 2) self.assertEqual(self.page.locator("#full-search-results mark.search-match").count(), 2) self.assertEqual(self.page.locator(".full-search-highlight").count(), 2) self.assertLessEqual(len(self.page.locator("#full-search-results .full-search-snippet").first.text_content()), 180) self.page.locator("#full-search-input").fill("阅读选项") self.page.locator("#full-search-status").filter(has_text="未找到").wait_for() self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0) self.page.locator("#full-search-input").fill("目标词") self.page.locator("#full-search-status").filter(has_text="2 个结果").wait_for() self.page.locator("#full-search-results .full-search-result").nth(1).click() self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0) def test_pdf_allows_two_bookmarks_on_one_page_and_restores_offsets(self): self.open_pdf() self.page.locator("#bookmark-ribbon").click() self.page.locator("#bookmark-add").click() self.page.locator("#viewport").evaluate("element => { element.scrollTop = 360; element.dispatchEvent(new Event('scroll')); }") self.page.locator("#bookmark-ribbon").click() self.page.locator("#bookmark-add").click() bookmarks = self.page.evaluate("window.__readerBookmarks") self.assertEqual(len(bookmarks), 2) self.assertNotEqual(bookmarks[0]["id"], bookmarks[1]["id"]) self.assertLess(bookmarks[0]["pageOffset"], bookmarks[1]["pageOffset"]) self.page.locator("#history").click() self.page.locator("#bookmarks-tab").click() rows = self.page.locator("#bookmarks-list .panel-item-main") self.assertEqual(rows.count(), 2) self.page.locator("#viewport").evaluate("element => element.scrollTop = 0") rows.nth(1).click() self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 250) rows.nth(0).click() self.assertLess(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 80) def test_bfcache_pageshow_waits_for_restoration_gate(self): delayed_store = STORE_SCRIPT.replace( "get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))", "get: () => new Promise((resolve) => setTimeout(() => resolve({url: location.href, scrollTop: 640, zoom: 100}), 1200))", ) self.page.unroute("**/static/reader-store.js") self.page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=delayed_store)) query = urllib.parse.quote(SOURCE_URL, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded") self.page.evaluate("""() => { for (const type of ["pagehide", "pageshow"]) { const event = new Event(type); Object.defineProperty(event, "persisted", { value: true }); window.dispatchEvent(event); } }""") self.page.wait_for_timeout(650) self.assertNotEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed") self.assertIsNone(self.page.evaluate("window.__savedReaderProgress || null")) self.page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.scrollTop >= 600", timeout=3000) def test_pagehide_before_restoration_does_not_overwrite_progress(self): query = urllib.parse.quote(SOURCE_URL, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded") self.page.evaluate("window.dispatchEvent(new Event('pagehide'))") self.page.wait_for_timeout(150) self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed") self.assertIsNone(self.page.evaluate("window.__savedReaderProgress || null")) def test_blocked_v1_upgrade_does_not_block_document_loading(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) blocker = context.new_page() blocker.goto(f"{self.origin}/static/reader.html", wait_until="domcontentloaded") blocker.evaluate("""async () => { await new Promise((resolve) => { const request = indexedDB.deleteDatabase('voiceofml-reader'); request.onsuccess = request.onerror = request.onblocked = resolve; }); window.__heldDb = await new Promise((resolve, reject) => { const request = indexedDB.open('voiceofml-reader', 1); request.onupgradeneeded = () => { const store = request.result.createObjectStore('entries', {keyPath: 'url'}); store.createIndex('lastReadAt', 'lastReadAt'); }; request.onsuccess = () => resolve(request.result); request.onerror = () => reject(request.error); }); }""") reader = context.new_page() reader.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader")) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/blocked.txt", safe="") reader.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=Blocked", wait_until="domcontentloaded") reader.locator(".reader-text").wait_for(timeout=5000) blocker.evaluate("window.__heldDb.close()") context.close() def test_html_bookmark_at_zero_restores_iframe_top(self): self.page.unroute("**/api/reader-content**") document = b"

    Top bookmark content

    Bottom

    " self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document)) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.html" query = urllib.parse.quote(source, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=html&title=HTML", wait_until="domcontentloaded") self.page.locator(".html-frame").wait_for() self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id: url + '\\0top', url, label: '阅读进度 0.0%', htmlScrollTop: 0, scrollTop: 0, createdAt: 1})", source) self.page.locator("#history").click() self.page.locator("#bookmarks-tab").click() bookmark = self.page.locator("#bookmarks-list .panel-item-main") bookmark.wait_for() self.page.locator(".html-frame").evaluate("frame => frame.contentWindow.scrollTo(0, 900)") bookmark.click() self.assertEqual(self.page.locator(".html-frame").evaluate("frame => frame.contentWindow.scrollY"), 0) def test_stale_bookmark_query_cannot_overwrite_all_bookmarks(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() store = """window.VoiceOfMLReaderStore = Object.freeze({ get: () => Promise.resolve(null), put: () => Promise.resolve(), list: () => Promise.resolve([]), remove: () => Promise.resolve(), clearHistory: () => Promise.resolve(), putBookmark: () => Promise.resolve(), removeBookmark: () => Promise.resolve(), listBookmarks: (url) => new Promise((resolve) => setTimeout(() => resolve([{id:'local', url, title:'Current book', label:'Current mark', createdAt:1}]), 180)), listAllBookmarks: () => new Promise((resolve) => setTimeout(() => resolve([{id:'all', url:'other', readerUrl:location.href, title:'All book', label:'All mark', createdAt:2}]), 10)) });""" page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=store)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader")) source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/race.txt", safe="") page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=Race", wait_until="domcontentloaded") page.locator(".reader-text").wait_for() page.locator("#history").click() page.locator("#bookmarks-tab").click() page.locator("#bookmarks-all").click() page.locator("#bookmarks-list .panel-item-main", has_text="All book").wait_for() page.wait_for_timeout(220) self.assertEqual(page.locator("#bookmarks-list .panel-item").count(), 1) self.assertIn("All book", page.locator("#bookmarks-list .panel-item-main").text_content()) context.close() def test_truncated_epub_reports_source_damage(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=b"PK\x03\x04truncated")) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/damaged.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Damaged", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertTrue(page.locator(".reader-error").text_content().strip()) self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_CORRUPT") context.close() def test_malicious_html_css_and_svg_are_inert_and_keep_safe_text(self): document = b'''

    Safe reader text

    safe link
    ''' self.page.unroute("**/api/reader-content**") self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document)) external = [] self.page.on("request", lambda request: external.append(request.url) if "evil.test" in request.url else None) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/security.html" self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=html&title=Security", wait_until="domcontentloaded") self.page.wait_for_function("() => document.querySelector('#status').textContent === 'HTML'") frame = self.page.locator("iframe.html-frame").content_frame self.assertEqual(frame.locator("#safe").text_content(), "Safe reader text") self.assertEqual(frame.locator("body script,body link,body svg").count(), 0) self.assertEqual(frame.locator("body img[src], body [href^='javascript:']").count(), 0) self.assertEqual(frame.locator("[onclick], [href^='javascript:']").count(), 0) self.assertFalse(self.page.evaluate("window.__unsafe === true")); self.assertEqual(external, []) def test_oversized_chapter_manifest_and_response_use_resource_limit(self): digest = "a" * 64; source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/chapter-manifest.json"; limit = 8 * 1024 * 1024 manifest = {"version": 1, "kind": "epub-chapters", "chapters": [{"index": 1, "path": "chapter.xhtml", "bytes": 10}]} for name, headers, chapter_headers in (("manifest", {"content-length": str(limit + 1)}, None), ("chapter", {}, {"content-length": str(8 * 1024 * 1024 + 1)})): with self.subTest(case=name): context = self.browser.new_context(viewport={"width": 390, "height": 844}); page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) def serve_limited(route, _request, headers=headers, chapter_headers=chapter_headers): if chapter_headers and "chapter.xhtml" in route.request.url: route.fulfill(status=200, content_type="text/html", headers=chapter_headers, body=b"

    chapter

    ") else: route.fulfill(status=200, content_type="application/json", headers=headers, body=json.dumps(manifest)) page.route("**/api/reader-content**", serve_limited) if chapter_headers: page.route("https://huggingface.co/**", lambda route, _request, headers=chapter_headers: route.fulfill(status=200, content_type="text/html", headers=headers, body=b"

    chapter

    ")) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Limits", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000); self.assertIn(page.locator("#content").get_attribute("data-error-code"), ("READER_PARSE", "READER_CORRUPT")); expected_limit = limit if name == "manifest" else 8 * 1024 * 1024; self.assertEqual(page.evaluate("limit => { try { VoiceOfMLReaderSecurity.assertResponseSize({headers:{get:()=>String(limit + 1)}}, limit); return null; } catch (error) { return error.message; } }", expected_limit), "READER_RESOURCE_LIMIT"); context.close() def test_zip_bomb_metadata_is_rejected_before_docx_and_foliate_parsers(self): payload = zip_bomb_metadata() with zipfile.ZipFile(io.BytesIO(payload)) as archive: self.assertIsNone(archive.testzip()) entry = archive.getinfo("bomb.txt") self.assertGreater(entry.file_size / entry.compress_size, 200) for extension in ("epub", "docx"): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/static/vendor/jszip.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.JSZip=function(){window.__archiveParserStarted=true};")) page.route("**/static/vendor/docx-preview.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.docx={renderAsync(){window.__archiveParserStarted=true}};")) page.route("**/static/foliate-reader/view.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="customElements.define('foliate-view', class extends HTMLElement { open() { window.__archiveParserStarted=true; throw new Error('parser must not start'); } });")) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/octet-stream", body=payload)) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bomb.epub" if extension == "epub" else f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{'a' * 64}/docx-native-v1/document.docx" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Bomb", wait_until="domcontentloaded") page.locator(".reader-error").wait_for(timeout=10000) self.assertIn("READER_ARCHIVE_LIMIT", page.locator(".reader-error").text_content()) self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_CORRUPT" if extension == "epub" else "READER_PARSE") self.assertFalse(page.evaluate("window.__archiveParserStarted === true")) context.close() def test_actual_store_broadcasts_progress_and_bookmark_updates_between_readers(self): self.page.unroute("**/static/reader-store.js"); self.page.unroute("**/api/reader-content**") self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Shared reader")) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/shared.txt"; query = urllib.parse.quote(source, safe="") self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Shared", wait_until="domcontentloaded"); self.page.locator(".reader-text").wait_for() other = self.context.new_page(); other.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Shared reader")); other.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Shared", wait_until="domcontentloaded"); other.locator(".reader-text").wait_for(); other.locator("#history").click() self.page.evaluate("url => VoiceOfMLReaderStore.put({url, title:'Shared', extension:'txt', lastReadAt:Date.now(), scrollTop:321})", source) self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id:'shared-bookmark', url, title:'Shared', label:'Shared mark', createdAt:Date.now()})", source) other.wait_for_function("() => [...document.querySelectorAll('#history-list .panel-item-main')].some(item => item.textContent.includes('Shared'))"); other.locator("#bookmarks-tab").click(); other.locator("#bookmarks-list .panel-item-main").filter(has_text="Shared mark").wait_for() def test_html_and_markdown_toc_click_navigation(self): for extension, body in (("html", b"

    HTML one

    space

    HTML two

    "), ("md", b"# Markdown one\n\n## Markdown two")): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}); page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) if extension == "md": page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.marked={parse:()=>'

    Markdown one

    Markdown two

    '};")); page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT)) page.route("**/api/reader-content**", lambda route, _request, body=body: route.fulfill(status=200, content_type="text/html" if extension == "html" else "text/markdown", body=body)) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/toc.{extension}"; page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=TOC", wait_until="domcontentloaded"); page.locator("#history").click(); page.locator("#toc-list .panel-item-main").nth(1).click() page.wait_for_function("() => document.querySelector('iframe') ? document.querySelector('iframe').contentWindow.scrollY > 0 : document.querySelector('#viewport').scrollTop > 0"); context.close() def test_pdf_outline_does_not_delay_ready(self): module = PDF_MODULE.replace("getOutline: () => Promise.resolve(window.__pdfOutlineEnabled ? [{title: '第一章', dest: [{}], items: []}] : null)", "getOutline: () => new Promise(resolve => { window.__releaseOutline = () => resolve([{title: '第一章', dest: [{}], items: []}]); })") self.page.unroute("**/static/vendor/pdf.min.*.mjs") self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(content_type="text/javascript", body=module)) self.open_pdf() self.page.locator("html[data-reader-phase='ready']").wait_for() self.page.locator(".reader-page canvas.ready").first.wait_for() self.assertEqual(self.page.locator("#toc-list .toc-item").count(), 0) self.page.evaluate("window.__releaseOutline()") self.page.locator("#toc-list .toc-item").first.wait_for(state="attached") def test_pdf_outline_click_navigates_to_declared_page(self): self.page.add_init_script("window.__pdfOutlineEnabled = true") module = PDF_MODULE.replace("[{title: '第一章', dest: [{}], items: []}]", "[{title: '第三章', dest: [{}], items: []}]").replace("getPageIndex: () => Promise.resolve(0)", "getPageIndex: () => Promise.resolve(2)"); self.page.unroute("**/static/vendor/pdf.min.*.mjs"); self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module)); self.open_pdf(); self.page.locator("#history").click(); self.page.locator("#toc-list .panel-item-main").click(); self.page.wait_for_function("() => document.querySelector('#page-number').value === '3'") def test_pdf_outline_jump_scrolls_before_target_render_finishes(self): self.page.add_init_script("window.__pdfProbe = { loads: 0, destroys: 0, renders: [], cancels: [], releases: {}, holdPages: [20] }") module = PDF_TASK_MODULE.replace( "getOutline: async () => null,", "getOutline: async () => [{title: '第二十页', dest: [{}], items: []}],\n getPageIndex: async () => 19,", ) self.page.unroute("**/static/vendor/pdf.min.*.mjs") self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module)) self.open_pdf() self.page.locator("#history").click() self.page.locator("#toc-list .panel-item-main").click() self.page.wait_for_function("""() => { const viewport = document.querySelector('#viewport'); return document.querySelector('#page-number').value === '20' && viewport.scrollTop > 1000 && Boolean(window.__pdfProbe.releases[20]); }""") self.assertGreater(self.page.locator("#viewport").evaluate("node => node.scrollTop"), 1000) self.page.evaluate("window.__pdfProbe.releases[20]()") self.page.locator('.reader-page[data-page="20"] canvas.ready').wait_for() def test_pdf_outline_marks_current_entry_when_page_changes(self): self.page.add_init_script("window.__pdfOutlineEnabled = true") module = PDF_MODULE.replace( "[{title: '第一章', dest: [{}], items: []}]", "[{title: '第一章', dest: [{page: 0}], items: []}, {title: '第二章', dest: [{page: 2}], items: []}]", ).replace("getPageIndex: () => Promise.resolve(0)", "getPageIndex: (ref) => Promise.resolve(ref.page || 0)") self.page.unroute("**/static/vendor/pdf.min.*.mjs") self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module)) self.open_pdf() self.page.locator("#history").click() self.page.locator("#toc-list .toc-item").nth(1).wait_for() self.assertTrue(self.page.locator("#toc-list .toc-item").nth(0).evaluate("row => row.classList.contains('is-current')")) self.page.evaluate("""() => { const viewport = document.querySelector('#viewport'); viewport.scrollTop = document.querySelector('.reader-page[data-page="3"]').offsetTop; viewport.dispatchEvent(new Event('scroll')); }""") self.page.wait_for_function("() => document.querySelector('#page-number').value === '3'") self.assertTrue(self.page.locator("#toc-list .toc-item").nth(1).evaluate("row => row.classList.contains('is-current')")) def test_pdf_stale_navigation_and_search_keep_latest_state(self): errors = [] self.page.on("pageerror", lambda error: errors.append(str(error))) self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill( status=200, content_type="text/javascript", body=PDF_TASK_MODULE, )) self.open_pdf() self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'") self.page.evaluate("window.__pdfProbe.holdPages = [20]") self.page.locator("#page-number").fill("20") self.page.locator("#page-number").dispatch_event("change") self.page.wait_for_function("() => Boolean(window.__pdfProbe.releases[20])") self.page.locator("#page-number").fill("3") self.page.locator("#page-number").dispatch_event("change") self.page.wait_for_function("() => Math.abs(document.querySelector('#viewport').scrollTop - document.querySelector('[data-page=\"3\"]').offsetTop) < 2") self.page.evaluate("""async () => { const pending = document.querySelector('[data-page="20"]')._renderPromise; window.__pdfProbe.releases[20](); await pending.catch(() => {}); await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve))); }""") self.assertEqual(self.page.locator("#page-number").input_value(), "3") self.assertAlmostEqual(self.page.locator("#viewport").evaluate("node => node.scrollTop"), self.page.locator('[data-page="3"]').evaluate("node => node.offsetTop"), delta=2) self.page.wait_for_function("() => window.__savedReaderProgress?.page === 3") self.page.locator("#history").click() self.page.locator("#full-search-toggle").click() for action in ("replace", "clear"): with self.subTest(action=action): if action == "clear": # A completed search caches extracted text. Use a fresh # document so cancellation still exercises an in-flight read. self.open_pdf() self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'") self.page.locator("#history").click() self.page.locator("#full-search-toggle").click() self.page.evaluate("window.__pdfProbe.holdText = true; delete window.__pdfProbe.releaseText") self.page.locator("#full-search-input").fill("obsolete") self.page.wait_for_function("() => Boolean(window.__pdfProbe.releaseText)") if action == "replace": self.page.locator("#full-search-input").fill("current") self.page.locator("#full-search-status").filter(has_text="1 个结果").wait_for() else: self.page.locator("#full-search-clear").click() self.page.evaluate("""async () => { window.__pdfProbe.releaseText(); await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve))); }""") self.assertEqual(self.page.locator('[data-page="1"] .full-search-highlight').count(), 0) if action == "replace": self.assertEqual(self.page.locator("#full-search-results .full-search-location").all_text_contents(), ["第 2 页"]) self.assertEqual(self.page.locator("#full-search-results mark").all_text_contents(), ["current"]) self.assertEqual(self.page.locator("#content .full-search-highlight").all_text_contents(), ["current"]) self.assertFalse(self.page.locator("#full-search-next").is_disabled()) else: self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0) self.assertEqual(self.page.locator("#content .full-search-highlight").count(), 0) self.assertEqual(self.page.locator("#full-search-status").text_content(), "输入关键词搜索正文") self.assertTrue(self.page.locator("#full-search-next").is_disabled()) self.assertTrue(self.page.locator("#full-search-prev").is_disabled()) self.assertEqual(errors, []) def test_pdf_disposal_destroys_loading_task_and_cancels_render_queue(self): for phase in ("loading", "rendering"): with self.subTest(phase=phase): context = self.browser.new_context() self.addCleanup(context.close) page = context.new_page() errors = [] page.on("pageerror", lambda error: errors.append(str(error))) page.add_init_script(f"window.__holdPdfLoad = {json.dumps(phase == 'loading')}") page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) requests = [] page.on("request", lambda request: requests.append(request.url) if "reader-content" in request.url or request.url == SOURCE_URL else None) page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_TASK_MODULE)) page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(SOURCE_URL, safe='')}&ext=pdf", wait_until="domcontentloaded") page.wait_for_function("() => Boolean(window.__pdfProbe)") if phase == "rendering": page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'") page.evaluate("""() => { window.__pdfProbe.holdPages = [20, 21, 22]; for (const number of [20, 21, 22]) { document.querySelector(`[data-page="${number}"]`).dispatchEvent(new Event('focus')); } }""") page.wait_for_function("() => Boolean(window.__pdfProbe.releases[20] && window.__pdfProbe.releases[21])") self.assertFalse(page.evaluate("Boolean(window.__pdfProbe.releases[22])")) page.evaluate("window.dispatchEvent(new PageTransitionEvent('pagehide', {persisted: true}))") self.assertEqual(page.evaluate("window.__pdfProbe.destroys"), 0) self.assertEqual(page.evaluate("window.__pdfProbe.cancels"), []) page.evaluate("window.dispatchEvent(new Event('pagehide'))") page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'disposed'") before = page.evaluate("({renders: [...window.__pdfProbe.renders], saved: window.__savedReaderProgress || null})") page.evaluate("""async () => { const pending = [...document.querySelectorAll('.reader-page')].map(node => node._renderPromise).filter(Boolean); window.__pdfProbe.releaseLoad?.(); Object.values(window.__pdfProbe.releases).forEach(resolve => resolve()); await Promise.allSettled(pending); window.dispatchEvent(new Event('pagehide')); }""") page.wait_for_timeout(550) # Cross delayed history restoration and the save debounce. self.assertEqual(page.evaluate("window.__pdfProbe.destroys"), 1) self.assertEqual(page.evaluate("window.__pdfProbe.loads"), 1) self.assertEqual(page.evaluate("window.__pdfProbe.renders"), before["renders"]) self.assertEqual(page.evaluate("window.__savedReaderProgress || null"), before["saved"]) self.assertEqual(page.locator(".reader-error").count(), 0) self.assertEqual(page.locator("html").get_attribute("data-reader-phase"), "disposed") self.assertTrue(page.locator(".reader-page canvas").evaluate_all("nodes => nodes.every(node => node.width === 0 && node.height === 0)")) self.assertEqual(sorted(page.evaluate("window.__pdfProbe.cancels")), [20, 21] if phase == "rendering" else []) if phase == "loading": self.assertEqual(page.locator(".reader-page").count(), 0) self.assertIsNone(before["saved"]) self.assertEqual(requests, []) self.assertEqual(errors, []) context.close() def test_epub_chapters_load_first_lazy_next_and_toc_destination(self): digest = "b" * 64 source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/chapter-manifest.json" manifest = {"version": 1, "kind": "epub-chapters", "chapters": [ {"index": i, "path": f"chapter-{i}.xhtml", "title": f"Chapter {i}", "bytes": 100} for i in range(1, 5) ]} requests, delayed = [], [] self.page.unroute("**/api/reader-content**") def serve_chapter(route, _request): number = next((i for i in range(1, 5) if f"chapter-{i}.xhtml" in route.request.url), None) if number is None: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)) return requests.append(number) route.fulfill(status=200, content_type="text/html", body=f"

    Chapter {number}

    body

    ") self.page.route("**/api/reader-content**", serve_chapter) self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Chapters", wait_until="domcontentloaded") self.page.locator(".reader-epub-chapter[data-chapter='1']").wait_for() self.assertEqual(set(requests), {1, 2, 3, 4}) self.page.locator(".reader-epub-chapter[data-chapter='2']").wait_for() self.page.locator("#history").click() self.page.locator("#toc-list .panel-item-main").nth(2).click() self.page.locator(".reader-epub-chapter[data-chapter='3']").wait_for() self.page.locator(".reader-epub-chapter[data-chapter='4']").wait_for() self.page.evaluate("() => new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve)))") self.assertEqual(self.page.locator(".reader-epub-chapter").evaluate_all("nodes => nodes.map(node => Number(node.dataset.chapter))"), [1, 2, 3, 4]) self.assertEqual(self.page.locator(".reader-chapter-sentinel").count(), 0) self.assertEqual(sorted(requests), [1, 2, 3, 4]) self.assertTrue(self.page.locator("#toc-list .toc-item").nth(2).evaluate("node => node.classList.contains('is-current')")) def test_epub_chapters_serve_nested_bundle_with_resources(self): digest = "c" * 64 base = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/foliate-original-v1/epub-chapters" source = base + "/chapter-manifest.json" chapters = { "chapters/chapter-0001.xhtml": '

    第一章

    one

    '.encode(), "chapters/chapter-0002.xhtml": "

    第二章

    two

    ".encode(), } manifest = {"version": 1, "kind": "epub-chapters", "chapters": [ {"index": i, "title": f"第{i}章", "path": path, "bytes": len(body), "sha256": hashlib.sha256(body).hexdigest()} for i, (path, body) in enumerate(chapters.items(), 1) ]} image_url = base + "/resources/img.png" image_requests, unexpected_requests = [], [] def reject_external(route): unexpected_requests.append(route.request.url) route.abort() def serve_image(route): image_requests.append(route.request.url) route.fulfill(status=200, content_type="image/png", body=PNG_BYTES) def serve_nested(route, _request): url = urllib.parse.parse_qs(urllib.parse.urlsplit(route.request.url).query)["url"][0] if url == source: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest).encode()) elif url.startswith(base + "/") and url[len(base) + 1:] in chapters: route.fulfill(status=200, content_type="text/html", body=chapters[url[len(base) + 1:]]) else: unexpected_requests.append(url) route.fulfill(status=404, body=b"no") self.page.unroute("**/api/reader-content**") self.page.route("**/api/reader-content**", serve_nested) self.page.route("https://huggingface.co/**", reject_external) self.page.route(image_url, serve_image) self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Nested", wait_until="domcontentloaded") first = self.page.locator(".reader-epub-chapter[data-chapter='1']") first.wait_for() self.assertIn("第一章", first.inner_text()) image = first.locator("img") image.wait_for() self.page.wait_for_function("""() => { const image = document.querySelector('.reader-epub-chapter[data-chapter="1"] img'); return image?.complete && image.naturalWidth === 1 && image.naturalHeight === 1; }""") self.assertEqual(image.get_attribute("src"), image_url) self.assertEqual(image_requests, [image_url]) self.page.locator("#history").click() self.page.locator("#toc-list .panel-item-main").nth(1).click() second = self.page.locator(".reader-epub-chapter[data-chapter='2']") second.wait_for() self.assertIn("第二章", second.inner_text()) self.assertEqual(unexpected_requests, []) def test_foliate_normalizes_legacy_chm_markup_and_keeps_resources(self): # This case verifies preservation of source CSS; night colors have a separate contract. self.page.add_init_script("localStorage.setItem('theme', 'light')") with zipfile.ZipFile(io.BytesIO(epub_with_legacy_chm_markup())) as archive: files = {name: archive.read(name) for name in archive.namelist()} chapter_path = "OEBPS/chapters space%20/chapter.xhtml" files["OEBPS/content.opf"] = files["OEBPS/content.opf"].replace( b'href="chapter.xhtml"', f'href="{urllib.parse.quote(chapter_path.removeprefix("OEBPS/"))}"'.encode(), ) resource_files = { f"OEBPS/images %23/{name}": PNG_BYTES for name in ("cover#1.png", "cover space.png", "literal%20.png", "literal%23.png") } resource_files["OEBPS/chapters space%20/local image.png"] = PNG_BYTES images = [] for path in resource_files: relative = (path.removeprefix("OEBPS/chapters space%20/") if path.startswith("OEBPS/chapters space%20/") else "../" + path.removeprefix("OEBPS/")) images.append(f'') chapter = files.pop("OEBPS/chapter.xhtml").decode().replace( 'href="style.css"', 'href="../style.css"', ).replace('src="picture.svg"', 'src="../picture.svg"') files[chapter_path] = chapter.replace( "' + "".join(images) + '查看原图 getComputedStyle(element).color"), "rgb(1, 2, 3)") self.assertTrue((image.get_attribute("src") or "").startswith("blob:")) self.assertGreater(image.bounding_box()["height"], 0) self.page.wait_for_function("""count => performance.getEntriesByType('resource') .filter(entry => entry.name.includes('/api/reader-resource?')).length >= count""", arg=len(resource_files)) self.assertEqual(set(requested_paths), set(resource_files)) resource_paths = self.page.locator(".foliate-continuous article[data-section='0'] svg image").evaluate_all("""images => images.map(image => new URL(image.getAttribute('href'), location.origin).searchParams.get('path'))""") self.assertCountEqual(resource_paths, resource_files) with self.page.expect_popup() as opened: self.page.locator('.foliate-continuous #full-image').click() popup = opened.value popup.wait_for_load_state() self.assertEqual(urllib.parse.parse_qs(urllib.parse.urlsplit(popup.url).query)['path'], ['OEBPS/chapters space%20/local image.png']) popup.wait_for_function('() => document.querySelector("img")?.naturalWidth === 1') popup.close() def test_foliate_navigation_path_dark_links_and_unique_sections(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) self.addCleanup(context.close) page = context.new_page() page.add_init_script("localStorage.setItem('theme', 'dark')") page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_navigation())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/e2e.epub" url = f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=E2E&path=Test%2Fbooks" page.goto(url, wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 2") link = page.locator(".foliate-continuous article[data-section='0'] a").first link.wait_for(state="attached") self.assertIsNotNone(page.locator(".foliate-continuous article[data-section='0']").evaluate("article => article.shadowRoot")) title_display = page.locator("#title").evaluate("element => getComputedStyle(element).display") page.locator(".foliate-continuous article[data-section='0']").evaluate("article => { const style = document.createElement('style'); style.textContent = '#title{display:none!important} a{color:rgb(1,2,3)!important;border-top-style:dotted}'; article.shadowRoot.appendChild(style); }") self.assertEqual(page.locator("#title").evaluate("element => getComputedStyle(element).display"), title_display) self.assertEqual(link.evaluate("element => getComputedStyle(element).borderTopStyle"), "dotted") self.assertEqual(link.evaluate("element => getComputedStyle(element).color"), "rgb(138, 180, 232)") page.locator("#history").click() for theme, color in (("light", "rgb(1, 2, 3)"), ("dark", "rgb(138, 180, 232)")): page.locator("#theme-toggle").click() page.wait_for_function("() => !document.documentElement.classList.contains('theme-transition')") self.assertEqual(page.locator("html").get_attribute("data-theme"), theme) self.assertEqual(link.evaluate("element => getComputedStyle(element).color"), color) page.locator("#history").click() self.assertEqual(page.locator("#reader-path").text_content(), "Test/books") self.assertIsNone(page.locator("#reader-path").get_attribute("hidden")) initial_url = page.url link.click() page.wait_for_function("() => document.querySelector('#viewport').scrollTop > 0") self.assertEqual(page.url, initial_url) self.assertAlmostEqual(page.locator("#one").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2) page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(1).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')") page.wait_for_function("() => document.querySelectorAll('.foliate-continuous article[data-section]').length === 3") sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => item.dataset.section)") self.assertEqual(sections, ["0", "1", "2"]) page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight; element.dispatchEvent(new Event('scroll')); }") page.wait_for_timeout(250) self.assertEqual(page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => item.dataset.section)"), ["0", "1", "2"]) context.close() def test_foliate_rapid_toc_navigation_keeps_latest_destination(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/race.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Race", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.evaluate("""() => { const sections = document.querySelector('foliate-view').book.sections.filter(section => section.linear !== 'no'); for (const [index, delay] of [[6, 350], [10, 20]]) { const original = sections[index].createDocument.bind(sections[index]); sections[index].createDocument = () => new Promise((resolve, reject) => setTimeout(() => original().then(resolve, reject), delay)); } document.querySelectorAll('#toc-list .panel-item-main')[5].click(); document.querySelectorAll('#toc-list .panel-item-main')[9].click(); }""") page.wait_for_timeout(700) self.assertTrue(page.locator("#toc-list .toc-item").nth(9).evaluate("item => item.classList.contains('is-current')")) self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2) sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => Number(item.dataset.section))") self.assertEqual(sections, sorted(set(sections))) context.close() def test_foliate_failed_toc_section_can_retry(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() errors = [] page.on("pageerror", lambda error: errors.append(str(error))) page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/retry.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Retry", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.evaluate("""() => { const sections = document.querySelector('foliate-view').book.sections.filter(item => item.linear !== 'no'); const section = sections[10]; window.__neighborFailures = 0; sections[9].createDocument = () => { window.__neighborFailures++; return Promise.reject(new Error('neighbor unavailable')); }; const original = section.createDocument.bind(section); let attempts = 0; section.createDocument = () => ++attempts === 1 ? Promise.reject(new Error('transient section failure')) : original(); window.__retrySection = () => document.querySelectorAll('#toc-list .panel-item-main')[9].click(); }""") page.evaluate("window.__retrySection()") page.wait_for_timeout(100) page.evaluate("window.__retrySection()") page.locator("#chapter-10").wait_for(timeout=5000) page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2) self.assertGreater(page.evaluate("window.__neighborFailures"), 0) self.assertEqual(page.locator(".reader-error").count(), 0) self.assertEqual(errors, []) context.close() def test_foliate_duplicate_toc_navigation_shares_section_load(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/deduplicate.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Deduplicate", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.evaluate("""() => { const section = document.querySelector('foliate-view').book.sections.filter(item => item.linear !== 'no')[10]; const original = section.createDocument.bind(section); window.__sectionCreates = 0; section.createDocument = () => { window.__sectionCreates += 1; return new Promise((resolve, reject) => setTimeout(() => original().then(resolve, reject), 200)); }; document.querySelectorAll('#toc-list .panel-item-main')[9].click(); document.querySelectorAll('#toc-list .panel-item-main')[9].click(); }""") page.locator("#chapter-10").wait_for(timeout=5000) page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") self.assertEqual(page.evaluate("window.__sectionCreates"), 1) self.assertEqual(page.locator(".foliate-continuous article[data-section='10']").count(), 1) self.assertTrue(page.locator("#toc-list .toc-item").nth(9).evaluate("item => item.classList.contains('is-current')")) context.close() def test_foliate_chapter_buttons_use_continuous_reader_navigation(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/chapter-buttons.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Buttons", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(9).click() page.locator("#chapter-10").wait_for(timeout=5000) page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") page.locator("#history").click() page.locator(".reader-chapter-next").click() page.locator("#chapter-11").wait_for(timeout=5000) page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(11)').classList.contains('is-current')") self.assertAlmostEqual(page.locator("#chapter-11").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2) self.assertTrue(page.locator("#toc-list .toc-item").nth(10).evaluate("item => item.classList.contains('is-current')")) context.close() def test_foliate_toc_retains_groups_and_chapter_buttons_skip_them(self): with zipfile.ZipFile(io.BytesIO(epub_with_navigation())) as archive: files = {name: archive.read(name) for name in archive.namelist()} files['OEBPS/nav.xhtml'] = files['OEBPS/nav.xhtml'].replace( b'
    1. 第一卷
    2. 第二卷
      1. ', b'
    ') self.page.unroute('**/api/reader-content**') self.page.route('**/api/reader-content**', lambda route: route.fulfill( status=200, content_type='application/epub+zip', body=zip_bytes(files))) source = 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/groups.epub' self.page.goto(f'{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe="")}&ext=epub', wait_until='domcontentloaded') self.page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 4") self.page.locator('#history').click() self.assertEqual(self.page.locator('#toc-list [role=heading]').all_text_contents(), ['第一卷', '第二卷']) self.assertEqual(self.page.locator('#toc-list [role=link]').all_text_contents(), ['第一章', '第二章']) self.assertEqual(self.page.locator('#toc-list .toc-group [tabindex]').count(), 0) self.page.locator('#toc-list [role=link]').first.click() self.page.wait_for_function("() => document.querySelector('[data-toc-index=\"1\"]').classList.contains('is-current')") self.page.wait_for_function("() => document.querySelector('#history').getAttribute('aria-expanded') === 'false'") self.page.locator('#history').click() self.page.locator('.reader-chapter-next').click() self.page.wait_for_function("() => document.querySelector('[data-toc-index=\"3\"]').classList.contains('is-current')") self.assertAlmostEqual(self.page.locator('#two').evaluate( "e => e.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2) def test_chm_rebuilt_spine_ignores_previous_asset_positions(self): with zipfile.ZipFile(io.BytesIO(epub_with_navigation())) as archive: files = {name: archive.read(name) for name in archive.namelist()} files['OEBPS/content.opf'] = files['OEBPS/content.opf'].replace( b'', b'') root = 'objects/aa/' + 'a' * 64 + '/' files['META-INF/reader-chm.json'] = json.dumps({'version': 1, 'previous_path': root + 'calibre-chm-epub-v2/document.epub', 'previous_sections': ['OEBPS/nav.xhtml', 'OEBPS/chapter-1.xhtml', 'OEBPS/chapter-2.xhtml']}) self.page.route('**/static/reader-store.js', lambda route: route.fulfill( status=200, content_type='text/javascript', body=STORE_SCRIPT.replace( 'get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))', "get: url => Promise.resolve(url.includes('calibre-chm-epub-v2') ? {foliateSection:2, foliateOffset:100, foliateTocIndex:1} : null)"))) self.page.route('**/api/reader-content**', lambda route: route.fulfill( status=200, content_type='application/epub+zip', body=zip_bytes(files))) source = 'https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/' + root + 'manual-chm-navigation-v2/document.epub' self.page.goto(f'{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe="")}&ext=epub', wait_until='domcontentloaded') self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'") self.page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 0") previous = source.replace('manual-chm-navigation-v2', 'calibre-chm-epub-v2') self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id:'old-chm-mark',url,foliateSection:1,foliateOffset:150,label:'Old chapter one',createdAt:1})", previous) self.page.locator('#history').click() self.page.locator('#bookmarks-tab').click() self.assertEqual(self.page.locator('#bookmarks-list .panel-item-main', has_text='Old chapter one').count(), 0) self.assertIn('manual-chm-navigation-v2', urllib.parse.unquote(self.page.url)) def test_foliate_scroll_updates_toc_on_animation_frame(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll-toc.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=ScrollToc", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(9).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") page.evaluate("""() => { const viewport = document.querySelector('#viewport'), target = document.querySelector('.foliate-continuous article[data-section="11"]').shadowRoot.querySelector('#chapter-11'); viewport.scrollTop += target.getBoundingClientRect().top - viewport.getBoundingClientRect().top + 100; viewport.dispatchEvent(new Event('scroll')); }""") page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(11)').classList.contains('is-current')", timeout=1000) context.close() def test_foliate_resize_keeps_current_text_anchor(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll-anchor.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Anchor", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(9).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") page.evaluate("""() => { const viewport = document.querySelector('#viewport'); viewport.style.overflowAnchor = 'none'; viewport.dispatchEvent(new Event('scroll')); }""") page.wait_for_timeout(50) before = page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top") page.evaluate("""() => { const spacer = document.createElement('div'); spacer.style.height = '600px'; const articles = [...document.querySelectorAll('.foliate-continuous article[data-section]:not(.foliate-section-placeholder)')].filter(article => Number(article.dataset.section) < 10); articles[articles.length - 1].shadowRoot.querySelector('.reader-section-body').appendChild(spacer); }""") page.wait_for_function("top => Math.abs(document.querySelector('.foliate-continuous article[data-section=\"10\"]').shadowRoot.querySelector('#chapter-10').getBoundingClientRect().top - top) < 3", arg=before, timeout=2000) context.close() def test_foliate_virtualizes_distant_sections_and_reloads_them(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/virtual.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Virtual", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") for index in [1, 3, 5, 7, 9, 11, 13]: page.evaluate("i => document.querySelectorAll('#toc-list .panel-item-main')[i].click()", index) page.wait_for_function("i => document.querySelectorAll('#toc-list .toc-item')[i].classList.contains('is-current')", arg=index) page.wait_for_timeout(100) self.assertLessEqual(page.locator(".foliate-continuous article[data-section]:not(.foliate-section-placeholder)").count(), 12) self.assertGreater(page.locator(".foliate-section-placeholder").count(), 0) sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => Number(item.dataset.section))") self.assertEqual(sections, sorted(set(sections))) page.evaluate("() => document.querySelectorAll('#toc-list .panel-item-main')[1].click()") page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')") page.locator(".foliate-continuous article[data-section]:not(.foliate-section-placeholder) #chapter-2").wait_for(timeout=5000) page.wait_for_function("() => document.querySelectorAll('.foliate-continuous article[data-section]:not(.foliate-section-placeholder)').length <= 12") context.close() def test_large_foliate_toc_keeps_a_bounded_dom_window(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters(600))) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/large-toc.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=LargeToc", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#toc-list')?.classList.contains('toc-list-virtualized')") page.locator("#history").click() self.assertLess(page.locator("#toc-list .toc-item").count(), 100) page.evaluate("""() => { const panel = document.querySelector('#toc-panel'); panel.scrollTop = 20000; panel.dispatchEvent(new Event('scroll')); }""") page.wait_for_function("() => [...document.querySelectorAll('#toc-list .toc-item')].some(row => row.textContent.includes('章节 501'))") self.assertLess(page.locator("#toc-list .toc-item").count(), 100) page.locator('#toc-list [data-toc-index="500"] .panel-item-main').click() page.wait_for_function("() => document.querySelector('#toc-list [data-toc-index=\"500\"]')?.classList.contains('is-current')") self.assertEqual(page.locator('#toc-list .is-current').get_attribute('data-toc-index'), '500') context.close() def test_foliate_full_search_uses_continuous_reader_navigation(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/search.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Search", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#full-search-toggle").click() page.locator("#full-search-input").fill("正文 10") page.locator("#full-search-results .full-search-result").first.wait_for(timeout=10000) page.locator("#full-search-results .full-search-result").first.click() page.locator("#chapter-10").wait_for(timeout=5000) page.locator(".foliate-continuous .full-search-highlight").wait_for(timeout=5000) page.wait_for_function("() => !document.querySelector('#history-panel').classList.contains('is-open')") self.assertTrue(page.locator(".foliate-continuous .full-search-highlight").evaluate("element => { const rect = element.getBoundingClientRect(), viewport = document.querySelector('#viewport').getBoundingClientRect(); return rect.bottom > viewport.top && rect.top < viewport.bottom; }")) context.close() def test_foliate_bookmark_restores_continuous_section_position(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Bookmark", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(9).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") page.locator("#bookmark-ribbon").click() page.locator("#bookmark-add").click() page.wait_for_function("() => window.__readerBookmarks.length === 1") bookmark = page.evaluate("window.__readerBookmarks[0]") self.assertEqual(bookmark.get("foliateSection"), 10) page.locator("#viewport").evaluate("element => { element.scrollTop = 0; element.dispatchEvent(new Event('scroll')); }") page.locator("#history").click() page.locator("#bookmarks-tab").click() page.locator("#bookmarks-list .panel-item-main").click() page.wait_for_function("() => !document.querySelector('#history-panel').classList.contains('is-open')") page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3) context.close() def test_foliate_progress_slider_uses_continuous_viewport(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/progress.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Progress", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(3).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(4)').classList.contains('is-current')") page.locator("#history").click() page.locator(".reader-progress-range").dispatch_event("pointerdown") # Seek inside section 12, away from its fractional-pixel top boundary. page.locator(".reader-progress-range").fill("81") page.locator(".reader-progress-range").dispatch_event("pointerup") page.wait_for_function("() => { const article = document.querySelector('.foliate-continuous article[data-section=\"12\"]'), viewport = document.querySelector('#viewport'); if (!article) return false; const marker = viewport.getBoundingClientRect().top + 8, rect = article.getBoundingClientRect(); return rect.top <= marker && rect.bottom > marker; }") self.assertAlmostEqual(float(page.locator(".reader-progress-percent").text_content().rstrip("%")), 81, delta=7) page.locator('.foliate-section-placeholder').first.evaluate("node => node.style.setProperty('--foliate-placeholder-height', `${node.getBoundingClientRect().height + 600}px`)") page.locator(".reader-progress-undo").click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(4)').classList.contains('is-current')") self.assertAlmostEqual(page.locator("#chapter-4").evaluate("node => node.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3) self.assertTrue(page.locator(".reader-progress-undo").is_hidden()) page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 4") context.close() def test_foliate_history_saves_structured_section_position(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/history.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=History", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(9).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')") page.evaluate("window.dispatchEvent(new Event('pagehide'))") page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.foliateSection === 10") self.assertEqual(page.evaluate("window.__savedReaderProgress.foliateTocIndex"), 9) context.close() def test_foliate_history_restores_structured_section_position(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() restored_store = STORE_SCRIPT.replace("get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300)),", "get: () => Promise.resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}),") page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=restored_store)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/restore.epub" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Restore", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)')?.classList.contains('is-current')", timeout=10000) self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3) delayed_store = restored_store.replace( "get: () => Promise.resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}),", "get: () => new Promise(resolve => { window.__releaseHistory = () => resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}); }),", ) page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=delayed_store)) page.reload(wait_until="domcontentloaded") page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14") page.locator("#history").click() page.locator("#toc-list .panel-item-main").nth(1).click() page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')") page.evaluate("window.__releaseHistory()") page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'") page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 2") self.assertAlmostEqual(page.locator("#chapter-2").evaluate("node => node.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3) context.close() def test_reader_store_resets_old_history_and_keeps_current_bookmarks(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.goto(f"{self.origin}/static/reader.html", wait_until="domcontentloaded") page.evaluate("""async () => { await new Promise((resolve) => { const request = indexedDB.deleteDatabase('voiceofml-reader'); request.onsuccess = request.onerror = request.onblocked = resolve; }); await new Promise((resolve, reject) => { const request = indexedDB.open('voiceofml-reader', 1); request.onupgradeneeded = () => { const store = request.result.createObjectStore('entries', {keyPath: 'url'}); store.createIndex('lastReadAt', 'lastReadAt'); }; request.onerror = () => reject(request.error); request.onsuccess = () => { const db = request.result, tx = db.transaction('entries', 'readwrite'); tx.objectStore('entries').put({url: 'legacy', lastReadAt: 1}); tx.oncomplete = () => { db.close(); resolve(); }; }; }); }""") page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader")) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/store.txt" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Store", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'") result = page.evaluate("""async (url) => { const legacy = await VoiceOfMLReaderStore.get('legacy'); await VoiceOfMLReaderStore.putBookmark({id: url + '\\0page:1', url, label: '第 1 页', createdAt: 1}); const historyBeforeClear = await VoiceOfMLReaderStore.list(); const bookmarkEntries = await VoiceOfMLReaderStore.listBookmarks(url); await VoiceOfMLReaderStore.clearHistory(); return {legacy: !!legacy, schema: VoiceOfMLReaderStore.SCHEMA_VERSION, validHistory: historyBeforeClear.every(entry => typeof entry.url === 'string' && entry.schemaVersion === 1), history: (await VoiceOfMLReaderStore.list()).length, bookmarks: (await VoiceOfMLReaderStore.listBookmarks(url)).length, bookmarkSchema: bookmarkEntries[0].schemaVersion}; }""", source) self.assertEqual(result, {"legacy": False, "schema": 1, "validHistory": True, "history": 0, "bookmarks": 1, "bookmarkSchema": 1}) context.close() def test_mobile_pdf_rendering_uses_one_slot_and_seven_canvases(self): self.page.set_viewport_size({"width": 390, "height": 844}) self.open_pdf() metrics = self.scroll_document() self.assertLessEqual(metrics["peak"], 1) self.assertLessEqual(metrics["rendered"], 7) self.assertGreater(metrics["pixels"], 0) def test_high_density_mobile_pdf_uses_backing_scale_without_distortion(self): context = self.browser.new_context( viewport={"width": 390, "height": 844}, device_scale_factor=3 ) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill( status=200, content_type="text/javascript", body=STORE_SCRIPT )) page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill( status=200, content_type="text/javascript", body=PDF_MODULE )) page.route("**/api/reader-content**", lambda route: route.fulfill( status=200, content_type="application/pdf", body=b"pdf" )) query = urllib.parse.quote(SOURCE_URL, safe="") page.goto( f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded", ) page.locator(".reader-page canvas.ready").first.wait_for(timeout=10000) metrics = page.locator(".reader-page").first.evaluate("""shell => { const canvas = shell.querySelector('canvas'); const box = canvas.getBoundingClientRect(); return { dpr: devicePixelRatio, backingWidth: canvas.width, backingHeight: canvas.height, cssWidth: box.width, cssHeight: box.height, shellAspect: shell.getBoundingClientRect().width / shell.getBoundingClientRect().height, cssAspect: box.width / box.height, backingAspect: canvas.width / canvas.height, canvasCssWidth: parseFloat(getComputedStyle(canvas).width), canvasCssHeight: parseFloat(getComputedStyle(canvas).height), }; }""") self.assertEqual(metrics["dpr"], 3) self.assertGreaterEqual(metrics["backingWidth"], metrics["cssWidth"] * 1.9) self.assertGreaterEqual(metrics["backingHeight"], metrics["cssHeight"] * 1.9) self.assertAlmostEqual(metrics["shellAspect"], metrics["cssAspect"], delta=0.01) self.assertAlmostEqual(metrics["cssAspect"], metrics["backingAspect"], delta=0.01) self.assertAlmostEqual(metrics["cssWidth"], metrics["canvasCssWidth"], delta=0.01) self.assertAlmostEqual(metrics["cssHeight"], metrics["canvasCssHeight"], delta=0.01) context.close() def test_mobile_zoom_enlarges_pdf_page_without_resizing_content_shell(self): self.page.set_viewport_size({"width": 390, "height": 844}) self.open_pdf() before = self.page.evaluate("""() => ({ content: document.querySelector('.reader-content').getBoundingClientRect().width, page: document.querySelector('.reader-page').getBoundingClientRect().width, pixels: document.querySelector('.reader-page canvas').width, })""") self.page.locator("#zoom-in").click(click_count=5) self.page.wait_for_function("before => document.querySelector('.reader-page canvas').width > before * 1.45", arg=before["pixels"]) after = self.page.evaluate("""() => ({ content: document.querySelector('.reader-content').getBoundingClientRect().width, page: document.querySelector('.reader-page').getBoundingClientRect().width, pixels: document.querySelector('.reader-page canvas').width, })""") self.assertAlmostEqual(after["content"], before["content"], delta=1) self.assertGreater(after["page"], before["page"] * 1.45) self.assertGreater(after["pixels"], before["pixels"] * 1.45) def test_txt_displays_before_stream_finishes(self): self.page.add_init_script(r""" const nativeFetch = window.fetch.bind(window); window.fetch = (input, init) => { const url = String(input && input.url || input); if (!url.includes('/api/reader-content?url=')) return nativeFetch(input, init); const encode = text => new TextEncoder().encode(text); const bytes = encode('first line\n'); const large = encode('large ASCII chunk\n'.repeat(8192) + 'target\n中文'); return Promise.resolve(new Response(new ReadableStream({ start(controller) { window.__txtStream = controller; controller.enqueue(bytes); window.__txtLargeChunk = () => { controller.enqueue(large.slice(0, -1)); }; window.__txtFinish = () => { controller.enqueue(large.slice(-1)); const ending = encode('\nfinal 中文'); controller.enqueue(ending.slice(0, ending.length - 1)); controller.enqueue(ending.slice(-1)); controller.close(); }; } }), { status: 200, headers: { 'Content-Type': 'text/plain' } })); }; """) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.txt" self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Performance", wait_until="domcontentloaded") self.page.locator(".reader-text").filter(has_text="first line").wait_for(timeout=3000) self.assertNotEqual(self.page.locator("#status").text_content(), "已加载") self.assertEqual(self.page.locator(".reader-text").text_content(), "first line\n") self.page.locator("#history").click() self.page.locator("#full-search-toggle").click() self.page.locator("#full-search-input").fill("first line") self.page.locator(".reader-text mark.full-search-highlight").wait_for() self.assertEqual(self.page.locator(".reader-text mark.full-search-highlight").count(), 1) self.page.evaluate("window.__txtLargeChunk()") expected = "first line\n" + "large ASCII chunk\n" * 8192 + "target\n中" self.page.wait_for_function("() => document.querySelector('.reader-text').textContent.endsWith('target\\n中')") self.assertEqual(self.page.locator(".reader-text").text_content(), expected) self.assertNotEqual(self.page.locator("#status").text_content(), "已加载") self.page.locator("#full-search-clear").click() self.assertEqual(self.page.locator(".reader-text mark").count(), 0) self.assertEqual(self.page.locator(".reader-text").text_content(), expected) self.page.evaluate("window.__txtFinish()") self.page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'") expected += "文\nfinal 中文" self.assertEqual(self.page.locator(".reader-text").text_content(), expected) self.page.locator("#full-search-input").fill("中文") self.page.wait_for_function("() => document.querySelectorAll('#full-search-results .full-search-result').length === 2") self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 2) self.assertEqual(self.page.locator(".reader-text mark.full-search-highlight").all_text_contents(), ["中文", "中文"]) self.page.locator("#full-search-clear").click() self.assertEqual(self.page.locator(".reader-text").text_content(), expected) self.assertEqual(self.page.locator(".full-search-highlight").count(), 0) self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0) def test_text_reader_uses_scroll_mode_without_pagination_controls(self): text = "\n\n".join(f"第 {index} 段内容。" * 120 for index in range(8)) self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=text.encode())) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll.txt" self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Scroll", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for() self.assertEqual(self.page.locator("#reading-mode").count(), 0) self.assertTrue(self.page.locator(".page-controls").is_hidden()) self.assertFalse(self.page.locator(".reader-viewport").evaluate("element => element.classList.contains('is-paginated')")) self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight; element.dispatchEvent(new Event('scroll')); }") self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0) def test_reader_tab_is_hidden_for_document_without_toc(self): self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"No table of contents")) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/no-toc.txt" self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=NoToc", wait_until="domcontentloaded") self.page.locator(".reader-text").wait_for() self.page.locator("#history").click() self.assertTrue(self.page.locator("#toc-tab").is_hidden()) def test_txt_detects_legacy_encodings_and_multibyte_sample_boundary(self): cases = [ (list("中文文本".encode("gb18030")), "Encoding", "中文文本"), (list("中文文本".encode("utf-16")), "Encoding", "中文文本"), (list("AB中文".encode("utf-16le")), "Encoding", "AB中文"), (list("Русский текст".encode("cp1251")), "Русский", "Русский текст"), (list(b"A" * 65535 + "中文".encode("gb18030")), "中文", "A" * 65535 + "中文"), (list("中文".encode("gb18030") + b"\xff" + "文本".encode("gb18030")), "中文", "中文文本"), ] for encoded, title, expected in cases: with self.subTest(title=title, size=len(encoded)): context = self.browser.new_context(viewport={"width": 1440, "height": 900}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.add_init_script(""" window.__txtBytes = new Uint8Array(%s); const nativeFetch = window.fetch.bind(window); window.fetch = (input, init) => { const url = String(input && input.url || input); if (!url.includes('/api/reader-content?url=')) return nativeFetch(input, init); return Promise.resolve(new Response(window.__txtBytes, { status: 200 })); }; """ % json.dumps(encoded)) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/encoding.txt" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title={urllib.parse.quote(title)}", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'") self.assertEqual(page.locator(".reader-text").text_content(), expected) context.close() def test_markdown_extension_aliases_render_content(self): for extension in ("md", "markdown"): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT)) page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=MARKED_SCRIPT)) page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT)) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/markdown", body=b"# Markdown readable")) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/readme.{extension}" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Markdown", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'") self.assertIn("Markdown readable", page.locator(".reader-markdown").inner_text()) context.close() def test_html_aliases_render_safely_with_visible_text(self): document = b'

    HTML readable

    ' for extension in ("html", "htm"): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}, color_scheme="dark") page = context.new_page(); external = [] page.on("request", lambda request: external.append(request.url) if "evil.test" in request.url else None) page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document)) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/page.{extension}" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=HTML", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === 'HTML'") frame = page.locator("iframe.html-frame").content_frame self.assertEqual(frame.locator("#visible").text_content(), "HTML readable") self.assertEqual(frame.locator("script, iframe").count(), 0) self.assertFalse(page.evaluate("window.__unsafe === true")) self.assertEqual(external, []) self.assertNotEqual(frame.locator("#visible").evaluate("e => getComputedStyle(e).color"), "rgb(255, 255, 255)") context.close() def test_image_aliases_decode_real_image_bytes(self): for extension, (content_type, document) in IMAGE_FIXTURES.items(): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() def serve_image(route, _request, mime=content_type, body=document): route.fulfill(status=200, content_type=mime, body=body) page.route("**/api/reader-content**", serve_image) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/image.{extension}" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Image", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === '图片'") self.assertTrue(page.locator(".reader-image").evaluate("image => image.complete && image.naturalWidth === 1 && image.naturalHeight === 1")) context.close() def test_native_media_uses_proxy_controls_and_mobile_layout(self): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() requests = [] page.route("**/api/reader-content**", lambda route: ( requests.append(route.request.url), route.fulfill(status=200, content_type="audio/wav", body=minimal_wav()), )) source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/sound.wav" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=wav&title=Sound", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('audio')?.readyState >= 1") audio = page.locator(".reader-audio") self.assertTrue(audio.evaluate("element => element.controls")) self.assertEqual(audio.get_attribute("preload"), "metadata") self.assertIn("/api/reader-content?url=", audio.get_attribute("src")) self.assertTrue(page.locator(".zoom-controls").is_hidden()) page.locator("#history").click() self.assertTrue(page.locator("#media-tab").is_visible()) self.assertEqual(page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "media") page.locator(".media-panel-bookmark").click() self.assertIn("时间", page.locator("#bookmark-prompt").text_content()) page.locator("#bookmark-add").click() page.locator("#bookmarks-tab").click() page.locator("#bookmarks-list .panel-item-main").wait_for() self.assertIn("时间", page.locator("#bookmarks-list").text_content()) self.assertTrue(requests) context.close() def test_audio_extension_aliases_use_native_reader_controls(self): for extension in ("mp3", "m4a", "flac", "mpga", "audio"): with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 390, "height": 844}) page = context.new_page() page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="audio/wav", body=minimal_wav())) source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/alias.{extension}" page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Audio", wait_until="domcontentloaded") page.wait_for_function("() => document.querySelector('#status').textContent === '音频'") audio = page.locator(".reader-audio") self.assertTrue(audio.evaluate("element => element.controls")) self.assertEqual(page.locator(".reader-content").get_attribute("data-mode"), "audio") self.assertTrue(page.locator(".zoom-controls").is_hidden()) self.assertTrue(page.locator("#bookmark-ribbon").is_visible()) context.close() def test_supported_formats_start_loading_while_history_restores(self): cases = [ ("md", "已加载", ".reader-markdown", ["content", "marked", "purify"]), ("docx", "DOCX", ".docx-body", ["content", "jszip", "docx"]), ("png", "图片", ".reader-image", ["content"]), ] elapsed_by_format = {} for extension, ready_status, selector, expected_requests in cases: with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 1440, "height": 900}) page = context.new_page() requested_at = {} def timed(name, content_type, body): def fulfill(route): requested_at[name] = page.evaluate("performance.now()") page.evaluate("name => (window.__formatRequests ||= []).push(name)", name) route.fulfill(status=200, content_type=content_type, body=body) return fulfill pending_store = STORE_SCRIPT.replace( "get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))", "get: () => new Promise((resolve) => { window.__releaseHistory = () => { window.__historyResolved = true; resolve(null); }; })", ) page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=pending_store)) page.route("**/static/vendor/marked.min.*.js", timed("marked", "text/javascript", MARKED_SCRIPT)) page.route("**/static/vendor/purify.min.*.js", timed("purify", "text/javascript", PURIFY_SCRIPT)) page.route("**/static/vendor/jszip.min.*.js", timed("jszip", "text/javascript", JSZIP_SCRIPT)) page.route("**/static/vendor/epub.min.06eae1574510.js", timed("epub", "text/javascript", EPUB_SCRIPT)) page.route("**/static/vendor/docx-preview.min.*.js", timed("docx", "text/javascript", DOCX_SCRIPT)) content_type = "image/png" if extension == "png" else "application/octet-stream" body = PNG_BYTES if extension == "png" else minimal_docx() if extension == "docx" else minimal_epub() if extension == "epub" else b"Reader benchmark content" page.route("**/api/reader-content**", timed("content", content_type, body)) if extension == "docx": digest = "a" * 64 source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/docx-native-v1/document.docx" else: source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.{extension}" if extension == "png": page.route(source, timed("content", content_type, body)) started = time.perf_counter() page.goto( f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Performance", wait_until="domcontentloaded", ) page.wait_for_function("expected => expected.every(name => window.__formatRequests?.includes(name))", arg=expected_requests) self.assertFalse(page.evaluate("window.__historyResolved === true")) page.evaluate("window.__releaseHistory()") page.wait_for_function( "expected => document.querySelector('#status').textContent === expected", arg=ready_status, ) page.locator(selector).wait_for(state="attached") elapsed_by_format[extension] = (time.perf_counter() - started) * 1000 store_started = page.evaluate("window.__storeStartedAt") self.assertEqual(set(requested_at), set(expected_requests)) self.assertTrue(all(requested_at[name] >= store_started for name in expected_requests)) context.close() print("\n Reader format load times: " + ", ".join( f"{name}={elapsed_by_format[name]:.1f}ms" for name in sorted(elapsed_by_format) )) def test_cold_cache_first_read_with_real_format_engines(self): cases = [ ("txt", b"Cold TXT readable", "text/plain", "已加载", ".reader-text", []), ("md", b"# Cold Markdown", "text/markdown", "已加载", ".reader-markdown", [VENDOR_FILES["marked"], VENDOR_FILES["purify"]]), ("pdf", minimal_pdf(), "application/pdf", "1 页", ".reader-page canvas.ready", [VENDOR_FILES["pdf"], VENDOR_FILES["pdf_worker"]]), ("docx", minimal_docx(), "application/vnd.openxmlformats-officedocument.wordprocessingml.document", "1 页", ".docx-body", [VENDOR_FILES["jszip"], VENDOR_FILES["docx"]]), ("png", PNG_BYTES, "image/png", "图片", ".reader-image", []), ] results = [] for extension, document, content_type, ready_status, selector, engines in cases: with self.subTest(extension=extension): context = self.browser.new_context(viewport={"width": 1440, "height": 900}, service_workers="block") try: page = context.new_page(); session = context.new_cdp_session(page) session.send("Network.enable"); session.send("Network.setCacheDisabled", {"cacheDisabled": True}) responses = []; page.on("response", lambda response: responses.append(response)) route_local_vendor_fallback(page) def serve_document(route): route.fulfill(status=200, content_type=content_type, body=document) page.route("**/api/reader-content**", serve_document) if extension == "docx": digest = "a" * 64 source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/docx-native-v1/document.docx" else: source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/cold.{extension}" if extension == "png": page.route(source, serve_document) started = time.perf_counter() page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Cold", wait_until="domcontentloaded") page.wait_for_function("expected => document.querySelector('#status').textContent === expected", arg=ready_status, timeout=30000) page.locator(selector).wait_for(state="attached", timeout=30000) if extension == "docx": self.assertTrue(page.locator(".page-controls").is_visible()) self.assertEqual(page.locator(".reader-docx-page").count(), 1) elapsed = (time.perf_counter() - started) * 1000 urls = [urllib.parse.urlsplit(response.url).path.rsplit("/", 1)[-1] for response in responses] for engine in engines: self.assertIn(engine, urls) for response in responses: self.assertFalse(response.from_service_worker) entries = page.evaluate("""() => performance.getEntriesByType('resource').map((entry) => ({ name: entry.name, transferSize: entry.transferSize }))""") for engine in engines: if ".worker." in engine: continue entry = next((item for item in entries if item["name"].endswith(engine)), None) self.assertIsNotNone(entry, engine) self.assertGreater(entry["transferSize"], 0, f"{engine} was not transferred on a cold load") byte_count = len(document) + sum( (ROOT / "static/vendor" / engine).stat().st_size if (ROOT / "static/vendor" / engine).exists() else 0 for engine in engines ) results.append((extension, elapsed, byte_count, len(responses))) finally: context.close() print("\n Reader cold-cache first read (real engines):") for extension, elapsed, byte_count, request_count in results: print(f" {extension:<5s} {elapsed:>7.1f}ms {byte_count / 1024:>8.1f}KiB {request_count:>2d} responses") def test_image_proxy_failure_falls_back_to_direct_source(self): source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fallback.png" requests = [] self.page.route(source, lambda route: (requests.append("source"), route.fulfill(status=200, content_type="image/png", body=PNG_BYTES))) self.page.route( "**/api/reader-content**", lambda route: (requests.append("proxy"), route.fulfill(status=404, body=b"")), ) self.page.goto( f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=png&title=Fallback", wait_until="domcontentloaded", ) self.page.wait_for_function("() => document.querySelector('#status').textContent === '图片'") self.assertEqual(requests, ["proxy", "source"]) def scroll_document(self): pages = self.page.locator(".reader-page") for index in range(30): pages.nth(index).scroll_into_view_if_needed() self.page.wait_for_timeout(20) self.page.wait_for_timeout(100) return self.page.locator("body").evaluate("""() => ({ peak: window.__pdfPeak, rendered: document.querySelectorAll('.reader-page[data-render-state="rendered"]').length, pixels: [...document.querySelectorAll('.reader-page canvas')].reduce((sum, canvas) => sum + canvas.width * canvas.height, 0), })""") def load_tests(_loader, _tests, _pattern): return unittest.TestSuite() if __name__ == "__main__": unittest.main()