import base64
import contextlib
import functools
import hashlib
import http.server
import io
import json
import mimetypes
import pathlib
import re
import threading
import time
import unittest
import urllib.parse
import wave
import zipfile
try:
from playwright.sync_api import Error as PlaywrightError
from playwright.sync_api import sync_playwright
except ImportError:
PlaywrightError = Exception
sync_playwright = None
ROOT = pathlib.Path(__file__).resolve().parents[1]
def vendor_name(key, pattern):
resources = (ROOT / "static/reader-resources.js").read_text(encoding="utf-8")
match = re.search(rf"\b{re.escape(key)}:\s*\{{.*?sha256:\s*\"([0-9a-f]+)\"", resources, re.S)
if not match:
raise RuntimeError(f"missing vendor digest for {key!r}")
return pattern.replace("*", match.group(1)[:12], 1)
VENDOR_FILES = {
"pdf": vendor_name("pdf", "pdf.min.*.mjs"),
"pdf_worker": vendor_name("pdfWorker", "pdf.worker.min.*.mjs"),
"marked": vendor_name("marked", "marked.min.*.js"),
"purify": vendor_name("purify", "purify.min.*.js"),
"jszip": vendor_name("jszip", "jszip.min.*.js"),
"docx": vendor_name("docx", "docx-preview.min.*.js"),
}
VENDOR_STEMS = {
"pdf": "pdf.min.*.mjs",
"pdf_worker": "pdf.worker.min.*.mjs",
"marked": "marked.min.*.js",
"purify": "purify.min.*.js",
"jszip": "jszip.min.*.js",
"docx": "docx-preview.min.*.js",
}
def route_local_vendor_fallback(page):
for key, requested in VENDOR_FILES.items():
candidates = sorted((ROOT / "static/vendor").glob(VENDOR_STEMS[key]))
if not candidates:
continue
page.route(
f"**/static/vendor/{requested}",
lambda route, _request, path=str(candidates[-1]): route.fulfill(path=path),
)
SOURCE_URL = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.pdf"
PDF_MODULE = """
export const GlobalWorkerOptions = {};
export class PDFWorker { promise = Promise.resolve(); destroy() {} }
const wait = () => new Promise((resolve) => setTimeout(resolve, 15));
export function getDocument() {
const page = {
getViewport({ scale }) { return { width: 600 * scale, height: 800 * scale }; },
getTextContent() { return Promise.resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }); },
render() {
window.__pdfActive = (window.__pdfActive || 0) + 1;
window.__pdfPeak = Math.max(window.__pdfPeak || 0, window.__pdfActive);
return { promise: wait().then(() => { window.__pdfActive -= 1; }) };
},
};
return { promise: Promise.resolve({ numPages: 30, getPage: () => Promise.resolve(page), getOutline: () => Promise.resolve(window.__pdfOutlineEnabled ? [{title: '第一章', dest: [{}], items: []}] : null), getPageIndex: () => Promise.resolve(0) }) };
}
"""
PDF_TASK_MODULE = """
export const GlobalWorkerOptions = {};
export class PDFWorker { promise = Promise.resolve(); destroy() {} }
export function getDocument() {
const probe = window.__pdfProbe ||= { loads: 0, destroys: 0, renders: [], cancels: [], releases: {} };
probe.loads++;
const pdf = {
numPages: 30,
getOutline: async () => null,
getPage: async (number) => ({
getViewport: ({scale}) => ({width: 600 * scale, height: 800 * scale,
convertToViewportPoint: (x, y) => [x, y]}),
async getTextContent() {
if (probe.holdText) {
probe.holdText = false;
await new Promise(resolve => { probe.releaseText = resolve; });
}
return {items: [{str: number === 1 ? 'obsolete' : number === 2 ? 'current' : 'ordinary',
transform: [12, 0, 0, 12, 20, 40], hasEOL: false}]};
},
render() {
probe.renders.push(number);
return {promise: probe.holdPages?.includes(number)
? new Promise(resolve => { probe.releases[number] = resolve; }) : Promise.resolve(),
cancel() { probe.cancels.push(number); }};
},
}),
};
return {promise: window.__holdPdfLoad
? new Promise(resolve => { probe.releaseLoad = () => resolve(pdf); }) : Promise.resolve(pdf),
destroy() { probe.destroys++; return Promise.resolve(); }};
}
"""
STORE_SCRIPT = """
window.__storeStartedAt = performance.now();
window.__readerBookmarks = [];
window.VoiceOfMLReaderStore = Object.freeze({
get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300)),
put: (entry) => { window.__savedReaderProgress = entry; return Promise.resolve(); }, list: () => Promise.resolve([]), remove: () => Promise.resolve(), clearHistory: () => Promise.resolve(),
putBookmark: (entry) => { window.__readerBookmarks = window.__readerBookmarks.filter((item) => item.id !== entry.id).concat(entry); return Promise.resolve(); },
listBookmarks: (url) => Promise.resolve(window.__readerBookmarks.filter((item) => item.url === url)),
listAllBookmarks: () => Promise.resolve([...window.__readerBookmarks].sort((a, b) => b.createdAt - a.createdAt)),
removeBookmark: (id) => { window.__readerBookmarks = window.__readerBookmarks.filter((item) => item.id !== id); return Promise.resolve(); }
});
"""
MARKED_SCRIPT = "window.marked = { parse: (text) => '
' + text + '
' };"
PURIFY_SCRIPT = "window.DOMPurify = { sanitize: (html) => html };"
JSZIP_SCRIPT = "window.JSZip = function() {};"
EPUB_SCRIPT = """
window.ePub = () => ({ renderTo: (frame) => ({
themes: { register() {}, select() {}, fontSize() {} },
on() {}, prev() {}, next() {},
display: () => new Promise((resolve) => setTimeout(() => {
frame.textContent = 'EPUB readable'; resolve();
}, 20)),
}) });
"""
DOCX_SCRIPT = """
window.docx = { renderAsync: (_bytes, body) => new Promise((resolve) => setTimeout(() => {
body.textContent = 'DOCX readable'; resolve();
}, 20)) };
"""
PNG_BYTES = bytes.fromhex(
"89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c489"
"0000000d49444154789c6360f8cfc000000301010018dd8db10000000049454e44ae426082"
)
IMAGE_FIXTURES = {
"jpg": ("image/jpeg", base64.b64decode("/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAP//////////////////////////////////////////////////////////////////////////////////////2wBDAf//////////////////////////////////////////////////////////////////////////////////////wAARCAABAAEDASIAAhEBAxEB/8QAFQABAQAAAAAAAAAAAAAAAAAAAAf/xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oADAMBAAIQAxAAAAF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABBQJ//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAwEBPwF//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAgEBPwF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQAGPwJ//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABPyF//9oADAMBAAIAAwAAAB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAEDAQE/EB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAECAQE/EB//xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oACAEBAAE/EB//2Q==")),
"jpeg": ("image/jpeg", base64.b64decode("/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAP//////////////////////////////////////////////////////////////////////////////////////2wBDAf//////////////////////////////////////////////////////////////////////////////////////wAARCAABAAEDASIAAhEBAxEB/8QAFQABAQAAAAAAAAAAAAAAAAAAAAf/xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oADAMBAAIQAxAAAAF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABBQJ//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAwEBPwF//8QAFBEBAAAAAAAAAAAAAAAAAAAAAP/aAAgBAgEBPwF//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQAGPwJ//8QAFBABAAAAAAAAAAAAAAAAAAAAAP/aAAgBAQABPyF//9oADAMBAAIAAwAAAB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAEDAQE/EB//xAAUEQEAAAAAAAAAAAAAAAAAAAAA/9oACAECAQE/EB//xAAUEAEAAAAAAAAAAAAAAAAAAAAA/9oACAEBAAE/EB//2Q==")),
"gif": ("image/gif", base64.b64decode("R0lGODlhAQABAIAAAAAAAP///ywAAAAAAQABAAACAUwAOw==")),
"bmp": ("image/bmp", bytes.fromhex("424d3a00000000000000360000002800000001000000010000000100180000000000040000000000000000000000000000000000000000000000")),
"webp": ("image/webp", base64.b64decode("UklGRiIAAABXRUJQVlA4IBYAAAAwAQCdASoBAAEAAUAmJaQAA3AA/v89WAAAAA==")),
}
def minimal_pdf():
stream = b"BT /F1 18 Tf 20 100 Td (Reader PDF) Tj ET"
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
b"<< /Length " + str(len(stream)).encode() + b" >>\nstream\n" + stream + b"\nendstream",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
payload = bytearray(b"%PDF-1.4\n%\xe2\xe3\xcf\xd3\n"); offsets = [0]
for number, body in enumerate(objects, 1):
offsets.append(len(payload)); payload.extend(f"{number} 0 obj\n".encode() + body + b"\nendobj\n")
xref = len(payload); payload.extend(f"xref\n0 {len(objects) + 1}\n".encode()); payload.extend(b"0000000000 65535 f \n")
for offset in offsets[1:]: payload.extend(f"{offset:010d} 00000 n \n".encode())
payload.extend(f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R >>\nstartxref\n{xref}\n%%EOF\n".encode())
return bytes(payload)
def zip_bytes(files, stored_first=None):
output = io.BytesIO()
with zipfile.ZipFile(output, "w", zipfile.ZIP_DEFLATED) as archive:
if stored_first: archive.writestr(stored_first[0], stored_first[1], compress_type=zipfile.ZIP_STORED)
for name, body in files.items(): archive.writestr(name, body)
return output.getvalue()
def zip_bomb_metadata():
# A complete, decompressible archive that exceeds the per-entry ratio limit.
return zip_bytes({"bomb.txt": b"x" * (1024 * 1024)})
def minimal_epub():
return zip_bytes({
"META-INF/container.xml": '',
"OEBPS/content.opf": 'readerReaderen ',
"OEBPS/chapter.xhtml": 'ReaderEPUB readable
',
}, ("mimetype", "application/epub+zip"))
def epub_with_navigation():
return zip_bytes({
"META-INF/container.xml": '',
"OEBPS/content.opf": 'reader-e2eReader E2Ezh ',
"OEBPS/nav.xhtml": '目录',
"OEBPS/chapter-1.xhtml": '第一章
第一章正文
',
"OEBPS/chapter-2.xhtml": '第二章
第二章正文
',
}, ("mimetype", "application/epub+zip"))
def epub_with_legacy_chm_markup():
return zip_bytes({
"META-INF/container.xml": '',
"OEBPS/content.opf": 'legacy-chmLegacy CHMC ',
"OEBPS/style.css": "p { color: rgb(1, 2, 3); }",
"OEBPS/picture.svg": '',
"OEBPS/chapter.xhtml": 'Legacy CHMArticle titleLegacy CHM content正文第一段正文第二段图片之后的正文',
}, ("mimetype", "application/epub+zip"))
def epub_with_many_chapters(count=14):
manifest = ' '
spine = ''
links = []
files = {}
for index in range(1, count + 1):
manifest += f' '
spine += f''
links.append(f'章节 {index}')
files[f"OEBPS/chapter-{index}.xhtml"] = f'章节 {index}
正文 {index}
'
files.update({
"META-INF/container.xml": '',
"OEBPS/content.opf": f'reader-raceReader Racezh{manifest}{spine}',
"OEBPS/nav.xhtml": f'',
})
return zip_bytes(files, ("mimetype", "application/epub+zip"))
def minimal_docx():
return zip_bytes({
"[Content_Types].xml": '',
"_rels/.rels": '',
"word/document.xml": 'DOCX readable',
})
def minimal_wav():
output = io.BytesIO()
with wave.open(output, "wb") as audio:
audio.setnchannels(1)
audio.setsampwidth(2)
audio.setframerate(8000)
audio.writeframes(b"\0\0" * 800)
return output.getvalue()
class StaticHandler(http.server.SimpleHTTPRequestHandler):
def guess_type(self, path):
if path.endswith(".mjs") or path.endswith(".js"):
return "text/javascript"
return mimetypes.guess_type(path)[0] or "application/octet-stream"
def log_message(self, *_args):
pass
@contextlib.contextmanager
def static_server():
handler = functools.partial(StaticHandler, directory=str(ROOT))
server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
yield f"http://127.0.0.1:{server.server_port}"
finally:
server.shutdown()
server.server_close()
thread.join(timeout=5)
@unittest.skipIf(sync_playwright is None, "install requirements-test.txt to run Reader performance tests")
class ReaderPerformanceTest(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.server = static_server()
cls.origin = cls.server.__enter__()
cls.playwright = sync_playwright().start()
try:
cls.browser = cls.playwright.chromium.launch(headless=True, args=["--no-sandbox"])
except PlaywrightError as error:
cls.playwright.stop()
cls.server.__exit__(None, None, None)
raise unittest.SkipTest(f"Chromium is unavailable: {error}")
@classmethod
def tearDownClass(cls):
cls.browser.close()
cls.playwright.stop()
cls.server.__exit__(None, None, None)
def setUp(self):
self.context = self.browser.new_context(viewport={"width": 1440, "height": 900})
self.page = self.context.new_page()
self.pdf_requested_at = None
self.page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
self.page.route("**/static/vendor/pdf.min.*.mjs", self.route_pdf)
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/pdf", body=b"pdf"))
def tearDown(self):
self.context.close()
def route_pdf(self, route):
self.pdf_requested_at = self.page.evaluate("performance.now()")
route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE)
def open_pdf(self):
query = urllib.parse.quote(SOURCE_URL, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded")
self.page.locator(".reader-page").nth(29).wait_for(state="attached")
def test_reader_starts_with_session_metadata_before_document_load(self):
errors = []
self.page.on("pageerror", lambda error: errors.append(str(error)))
self.page.add_init_script("""
sessionStorage.setItem('reader-source:metadata-probe', JSON.stringify({
url: 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt',
download: 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt',
title: 'Metadata title', extension: 'txt', original_extension: 'txt',
repo: 'Test', folder: ['Folder']
}));
""")
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Metadata readable"))
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/metadata.txt", safe="")
self.page.goto(f"{self.origin}/static/reader.html?id=metadata-probe&url={source}&ext=txt", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for(state="visible")
self.assertEqual(self.page.locator(".reader-text").text_content(), "Metadata readable")
self.assertEqual(self.page.locator("#title").text_content(), "Metadata title.txt")
self.assertEqual(self.page.locator("#reader-path").text_content(), "Test/Folder")
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "prepare")
self.page.locator("html[data-reader-phase='ready']").wait_for(state="attached")
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "ready")
self.assertEqual(errors, [])
def test_fetch_file_aborts_when_pagehide_disposes_reader(self):
self.page.add_init_script(r"""
(() => {
const nativeFetch = window.fetch.bind(window);
const probe = window.__fetchFileProbe = { started: false, aborted: false, result: "" };
window.fetch = (input, init = {}) => {
if (!String(input).includes("fetch-file-probe")) return nativeFetch(input, init);
probe.started = true;
return new Promise((resolve, reject) => {
const signal = init && init.signal;
const abort = () => {
probe.aborted = true;
probe.result = "aborted";
reject(new DOMException("Reader disposed", "AbortError"));
};
if (signal && signal.aborted) return abort();
if (signal) signal.addEventListener("abort", abort, { once: true });
probe.resolve = () => {
probe.result = "fulfilled";
resolve(new Response("probe", { status: 200, headers: { "content-type": "text/plain" } }));
};
});
};
})();
""")
self.page.unroute("**/api/reader-content**")
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader"))
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file.txt", safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FetchFile", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for(state="visible")
self.page.evaluate("""() => {
window.__fetchFilePromise = window.fetchFile("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file-probe.txt")
.then(() => { window.__fetchFileProbe.result = "fulfilled"; })
.catch((error) => { window.__fetchFileProbe.error = error.name; });
}""")
self.page.wait_for_function("() => window.__fetchFileProbe.started === true")
self.page.evaluate("""() => {
const event = new Event("pagehide");
Object.defineProperty(event, "persisted", { value: true });
window.dispatchEvent(event);
}""")
self.page.wait_for_timeout(100)
self.assertFalse(self.page.evaluate("() => window.__fetchFileProbe.aborted"))
self.page.evaluate("window.dispatchEvent(new Event('pagehide'))")
self.page.wait_for_function("() => window.__fetchFileProbe.result === 'aborted'", timeout=2000)
self.assertEqual(self.page.evaluate("() => window.__fetchFileProbe.error"), "AbortError")
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed")
def test_concurrent_fetch_file_callers_receive_complete_bodies(self):
self.page.add_init_script(r"""
(() => {
const nativeFetch = window.fetch.bind(window);
window.__concurrentFetchCalls = 0;
window.fetch = (input, init = {}) => {
if (!String(input).includes("concurrent-fetch-file")) return nativeFetch(input, init);
window.__concurrentFetchCalls += 1;
return Promise.resolve(new Response("complete shared body", { status: 200, headers: { "content-type": "text/plain" } }));
};
})();
""")
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fetch-file.txt", safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FetchFile", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for(state="visible")
bodies = self.page.evaluate("""async () => {
const url = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/concurrent-fetch-file.txt";
const files = await Promise.all([window.fetchFile(url), window.fetchFile(url)]);
return Promise.all(files.map((file) => file.text()));
}""")
self.assertEqual(bodies, ["complete shared body", "complete shared body"])
self.assertEqual(self.page.evaluate("window.__concurrentFetchCalls"), 1)
def test_id_only_resolver_is_lifecycle_managed(self):
errors = []
self.page.on("pageerror", lambda error: errors.append(str(error)))
self.page.add_init_script(r"""
(() => {
const nativeFetch = window.fetch.bind(window);
window.__resolverProbe = { started: false, aborted: false };
window.fetch = (input, init = {}) => {
if (!String(input).includes("/api/reader-resolve?id=resolver-abort")) return nativeFetch(input, init);
window.__resolverProbe.started = true;
return new Promise((resolve, reject) => {
const abort = () => { window.__resolverProbe.aborted = true; reject(new DOMException("Reader disposed", "AbortError")); };
if (init.signal?.aborted) return abort();
init.signal?.addEventListener("abort", abort, { once: true });
});
};
})();
""")
self.page.goto(f"{self.origin}/static/reader.html?id=resolver-abort", wait_until="domcontentloaded")
self.page.wait_for_function("window.__resolverProbe.started")
self.page.evaluate("window.dispatchEvent(new Event('pagehide'))")
self.page.wait_for_function("window.__resolverProbe.aborted")
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed")
self.assertEqual(errors, [])
def test_id_only_resolver_failure_uses_reader_error_ui(self):
errors = []
self.page.on("pageerror", lambda error: errors.append(str(error)))
self.page.route("**/api/reader-resolve?id=resolver-failure", lambda route: route.fulfill(status=503, body="unavailable"))
self.page.goto(f"{self.origin}/static/reader.html?id=resolver-failure", wait_until="domcontentloaded")
self.page.locator(".reader-error").wait_for(state="visible")
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "failed")
self.assertEqual(self.page.locator("#content").get_attribute("data-error-code"), "READER_NETWORK")
self.assertEqual(errors, [])
def test_id_only_reader_source_uses_authoritative_resolve(self):
stored = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/stored.txt"
authoritative = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/authoritative.txt"
stored_data = {
"url": stored, "download": stored, "title": "Stored title", "extension": "txt",
"original_extension": "txt", "repo": "Test", "folder": ["Stored"],
}
self.page.add_init_script(
f"sessionStorage.setItem('reader-source:id-only-authority', {json.dumps(json.dumps(stored_data))})"
)
self.page.route(
"**/api/reader-resolve?id=id-only-authority",
lambda route: route.fulfill(
status=200, content_type="application/json",
body=json.dumps({"url": authoritative, "download": authoritative, "title": "Resolved title", "extension": "txt", "original_extension": "txt", "repo": "Test", "folder": "Authoritative"}),
),
)
self.page.unroute("**/api/reader-content**")
self.page.route(
"**/api/reader-content**",
lambda route: route.fulfill(
status=200, content_type="text/plain",
body=b"Authoritative reader" if "authoritative.txt" in route.request.url else b"Stored reader",
),
)
self.page.goto(f"{self.origin}/static/reader.html?id=id-only-authority", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for(state="visible")
self.assertEqual(self.page.locator(".reader-text").text_content(), "Authoritative reader")
self.assertEqual(self.page.locator("#title").text_content(), "Resolved title.txt")
self.assertEqual(self.page.locator("#reader-path").text_content(), "Test/Authoritative")
self.assertNotIn("url=", self.page.url)
def test_id_only_reader_falls_back_to_session_source_when_resolve_fails(self):
stored = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/stored.txt"
stored_data = {
"url": stored, "download": stored, "title": "Stored title", "extension": "txt",
"original_extension": "txt", "repo": "Test", "folder": ["Stored"],
}
self.page.add_init_script(
f"sessionStorage.setItem('reader-source:id-only-fallback', {json.dumps(json.dumps(stored_data))})"
)
self.page.route("**/api/reader-resolve?id=id-only-fallback", lambda route: route.fulfill(status=503, body="unavailable"))
self.page.unroute("**/api/reader-content**")
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Stored reader"))
self.page.goto(f"{self.origin}/static/reader.html?id=id-only-fallback", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for(state="visible")
self.assertEqual(self.page.locator(".reader-text").text_content(), "Stored reader")
self.assertEqual(self.page.locator("#title").text_content(), "Stored title.txt")
def test_document_preparation_overlaps_delayed_history_restore(self):
self.open_pdf()
store_started = self.page.evaluate("window.__storeStartedAt")
self.assertIsNotNone(self.pdf_requested_at)
self.assertLess(self.pdf_requested_at - store_started, 250)
def test_pdf_rendering_has_bounded_concurrency_and_canvas_memory(self):
self.open_pdf()
metrics = self.scroll_document()
self.assertLessEqual(metrics["peak"], 2)
self.assertLessEqual(metrics["rendered"], 11)
self.assertGreater(metrics["pixels"], 0)
def test_pdf_exposes_lazy_accessible_text(self):
self.open_pdf()
first_page = self.page.locator(".reader-page").first
self.assertEqual(first_page.get_attribute("role"), "region")
self.assertEqual(first_page.locator("canvas").get_attribute("aria-hidden"), "true")
self.page.wait_for_function("document.querySelector('.reader-page')?.dataset.textReady === '1'")
self.assertEqual(first_page.locator(".reader-pdf-text").text_content(), "Accessible PDF text")
def test_native_pdf_canvas_does_not_wait_for_text_layer(self):
module = PDF_MODULE.replace(
"getTextContent() { return Promise.resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }); },",
"getTextContent() { return new Promise(resolve => setTimeout(() => resolve({ items: [{ str: 'Accessible PDF text', hasEOL: false }] }), 1000)); },",
)
self.page.unroute("**/static/vendor/pdf.min.*.mjs")
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(
content_type="text/javascript", body=module))
self.open_pdf()
self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for(timeout=3000)
self.assertNotEqual(
self.page.locator('.reader-page[data-page="1"]').get_attribute("data-text-ready"), "1"
)
self.page.locator('.reader-page[data-page="1"][data-text-ready="1"]').wait_for(timeout=3000)
def test_scanned_pdf_bookmark_has_empty_excerpt(self):
self.page.route('**/static/vendor/pdf.min.*.mjs', lambda route: route.fulfill(
content_type='text/javascript', body=PDF_MODULE.replace("[{ str: 'Accessible PDF text', hasEOL: false }]", '[]')))
self.open_pdf()
self.page.wait_for_function("document.querySelector('.reader-page')?.dataset.textReady === '1'")
self.assertEqual(self.page.locator('.reader-pdf-text').first.text_content(), '')
self.page.locator('#bookmark-ribbon').click()
self.assertEqual(self.page.locator('#bookmark-excerpt-input').input_value(), '')
self.page.locator('#bookmark-add').click()
self.page.wait_for_function('window.__readerBookmarks.length === 1')
self.assertEqual(self.page.evaluate('window.__readerBookmarks[0].excerpt'), '')
def test_reader_panel_bookmark_search_and_theme(self):
self.page.add_init_script("window.__pdfOutlineEnabled = true")
self.open_pdf()
self.page.locator("#bookmark-ribbon").click()
self.assertEqual(self.page.locator("#bookmark-popover").get_attribute("role"), "dialog")
self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "true")
self.page.locator("#bookmark-add").press("Escape")
self.assertTrue(self.page.locator("#bookmark-popover").is_hidden())
self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "false")
self.page.locator("#bookmark-ribbon").click()
self.assertIn("第 1 / 30 页", self.page.locator("#bookmark-prompt").text_content())
self.page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.page === 1")
self.page.locator("#bookmark-add").click()
self.page.locator("#history").click()
self.assertTrue(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')"))
self.assertNotEqual(self.page.locator("#history-panel").evaluate("element => getComputedStyle(element).transitionDuration"), "0s")
self.assertTrue(self.page.locator("#toc-tab").is_visible())
self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc")
self.assertEqual(self.page.locator(".reader-panel-tabs").get_attribute("role"), "tablist")
self.page.locator("#toc-tab").focus()
self.page.locator("#toc-tab").press("ArrowRight")
self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "bookmarks")
self.page.locator("#bookmarks-tab").press("ArrowLeft")
self.assertEqual(self.page.locator("#toc-list .panel-item-main").get_attribute("role"), "link")
self.assertEqual(self.page.locator("#toc-list .panel-item-main").evaluate("element => getComputedStyle(element).userSelect"), "text")
self.assertEqual(self.page.locator("#toc-panel .panel-search-toggle").text_content(), "搜索")
self.assertEqual(self.page.locator("#history-panel > footer").count(), 0)
self.assertEqual(self.page.locator("#history-panel > header #theme-toggle").count(), 1)
self.assertEqual(self.page.locator("#history-panel > header .icon-btn").count(), 0)
self.assertEqual(self.page.locator("#history-clear").text_content(), "清空历史")
self.assertLess(self.page.locator("#history-clear").evaluate("element => [...element.parentElement.children].indexOf(element)"), self.page.locator("#history-view .panel-search-toggle").evaluate("element => [...element.parentElement.children].indexOf(element)"))
self.assertEqual(self.page.locator("#history-clear").evaluate("element => getComputedStyle(element).alignItems"), "center")
self.assertEqual(self.page.locator("#history-view .panel-search-toggle").evaluate("element => getComputedStyle(element).transform"), "none")
self.page.locator('.reader-panel-tabs button[data-panel="bookmarks"]').click()
self.page.locator("#bookmarks-list .panel-item-main").filter(has_text="第 1 / 30 页").wait_for()
self.page.locator("#bookmarks-panel .panel-search-toggle").click()
self.assertTrue(self.page.locator("#bookmarks-panel .panel-search").evaluate("element => element.classList.contains('is-open')"))
self.page.locator("#bookmarks-panel .panel-search").fill("不存在")
self.assertTrue(self.page.locator("#bookmarks-list .panel-item").is_hidden())
self.page.locator("#history").click()
self.page.locator("#history").click()
self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc")
self.page.locator('.reader-panel-tabs button[data-panel="bookmarks"]').click()
self.page.locator("#history").click()
self.page.locator("#history").click()
self.assertEqual(self.page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "toc")
self.page.locator("#theme-toggle").click()
self.assertEqual(self.page.locator("html").get_attribute("data-theme"), "light")
self.assertTrue(self.page.locator("html").evaluate("element => element.classList.contains('theme-transition')"))
self.page.wait_for_timeout(300)
self.assertNotEqual(self.page.locator(".compact-input").first.evaluate("element => getComputedStyle(element).backgroundColor"), "rgb(37, 41, 45)")
self.page.locator("#page-prev").hover()
self.assertNotEqual(self.page.locator("#page-prev").evaluate("element => getComputedStyle(element).backgroundColor"), "rgb(41, 45, 49)")
self.assertEqual(self.page.locator("#zoom").get_attribute("min"), "25")
self.assertEqual(self.page.locator("#zoom").get_attribute("max"), "400")
def test_reader_controls_fit_viewport_without_overlap_and_work_on_mobile(self):
self.open_pdf()
def assert_toolbar_layout():
layout = self.page.locator(".reader-toolbar").evaluate("""toolbar => {
const view = {width: innerWidth, height: innerHeight};
const selectors = ['#back', '#page-prev', '#page-number', '#page-next', '#zoom-out', '#zoom', '#zoom-in', '#history', '#download'];
const rects = selectors.map(selector => {
const element = document.querySelector(selector);
const rect = element.getBoundingClientRect();
return {selector, left: rect.left, top: rect.top, right: rect.right, bottom: rect.bottom, width: rect.width, height: rect.height, visible: !!(rect.width && rect.height)};
}).filter(item => item.visible);
return {toolbar: toolbar.getBoundingClientRect().toJSON(), view, rects};
}""")
self.assertGreaterEqual(layout["toolbar"]["height"], 36)
for item in layout["rects"]:
self.assertGreater(item["width"], 0, item["selector"])
self.assertGreaterEqual(item["left"], 0, item["selector"])
self.assertLessEqual(item["right"], layout["view"]["width"] + 1, item["selector"])
self.assertGreaterEqual(item["top"], 0, item["selector"])
self.assertLessEqual(item["bottom"], layout["toolbar"]["bottom"] + 1, item["selector"])
for index, first in enumerate(layout["rects"]):
for second in layout["rects"][index + 1:]:
overlap = first["left"] < second["right"] and second["left"] < first["right"] and first["top"] < second["bottom"] and second["top"] < first["bottom"]
self.assertFalse(overlap, f'{first["selector"]} overlaps {second["selector"]}')
assert_toolbar_layout()
self.page.locator("#page-next").click()
self.assertEqual(self.page.locator("#page-number").input_value(), "2")
self.page.locator("#page-prev").click()
self.assertEqual(self.page.locator("#page-number").input_value(), "1")
self.page.locator("#zoom-in").click()
self.assertEqual(self.page.locator("#zoom").input_value(), "110")
self.page.locator("#zoom-out").click()
self.assertEqual(self.page.locator("#zoom").input_value(), "100")
self.page.locator("#history").click()
self.assertTrue(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')"))
self.page.locator("#history-close").click()
self.assertFalse(self.page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')"))
mobile_context = self.browser.new_context(viewport={"width": 390, "height": 844})
mobile_page = mobile_context.new_page()
mobile_page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
mobile_page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE))
mobile_page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/pdf", body=b"pdf"))
query = urllib.parse.quote(SOURCE_URL, safe="")
mobile_page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded")
mobile_page.locator(".reader-page").nth(29).wait_for(state="attached")
mobile_layout = mobile_page.locator(".reader-toolbar").evaluate("""toolbar => {
const view = {width: innerWidth, height: innerHeight};
const rects = [...toolbar.querySelectorAll('button, input, a')].map(element => {
const rect = element.getBoundingClientRect();
return {left: rect.left, right: rect.right, top: rect.top, bottom: rect.bottom, width: rect.width, height: rect.height, visible: !!(rect.width && rect.height)};
}).filter(item => item.visible);
return {toolbar: toolbar.getBoundingClientRect().toJSON(), view, rects};
}""")
self.assertEqual(mobile_layout["view"]["width"], 390)
self.assertGreaterEqual(mobile_layout["toolbar"]["height"], 36)
for item in mobile_layout["rects"]:
self.assertGreaterEqual(item["left"], 0)
self.assertLessEqual(item["right"], 390)
self.assertLessEqual(item["bottom"], mobile_layout["toolbar"]["bottom"] + 1)
mobile_page.locator("#history").click()
self.assertTrue(mobile_page.locator("#history-panel").evaluate("element => element.classList.contains('is-open')"))
mobile_page.wait_for_timeout(300)
panel_box = mobile_page.locator("#history-panel").bounding_box()
self.assertIsNotNone(panel_box)
self.assertGreaterEqual(panel_box["x"], 0)
self.assertLessEqual(panel_box["x"] + panel_box["width"], 390)
mobile_context.close()
def test_reader_controls_honor_boundaries_and_keyboard_activation(self):
self.open_pdf()
self.assertEqual(self.page.locator("#page-number").input_value(), "1")
self.page.locator("#page-prev").click()
self.assertEqual(self.page.locator("#page-number").input_value(), "1")
self.page.locator("#page-next").focus()
self.page.locator("#page-next").press("Enter")
self.assertEqual(self.page.locator("#page-number").input_value(), "2")
self.page.locator("#page-number").fill("999")
self.page.locator("#page-number").press("Enter")
self.page.wait_for_function("() => document.querySelector('#page-number').value === '30'")
self.assertEqual(self.page.locator("#page-number").input_value(), "30")
self.page.locator("#page-number").fill("0")
self.page.locator("#page-number").press("Enter")
self.page.wait_for_function("() => document.querySelector('#page-number').value === '1'")
self.assertEqual(self.page.locator("#page-number").input_value(), "1")
self.page.locator("#zoom").fill("999")
self.page.locator("#zoom").press("Enter")
self.page.wait_for_function("() => document.querySelector('#zoom').value === '400'")
self.assertEqual(self.page.locator("#zoom").input_value(), "400")
self.page.locator("#zoom-in").click()
self.assertEqual(self.page.locator("#zoom").input_value(), "400")
self.page.locator("#zoom").fill("1")
self.page.locator("#zoom").press("Enter")
self.page.wait_for_function("() => document.querySelector('#zoom').value === '25'")
self.assertEqual(self.page.locator("#zoom").input_value(), "25")
self.page.locator("#zoom-out").click()
self.assertEqual(self.page.locator("#zoom").input_value(), "25")
self.page.locator("#history").focus()
self.page.locator("#history").press("Enter")
self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "true")
self.page.locator("#history-close").press("Enter")
self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "false")
self.assertTrue(self.page.locator("#download").get_attribute("href"))
self.assertEqual(self.page.locator("#download").get_attribute("target"), "_blank")
self.assertIn("noopener", self.page.locator("#download").get_attribute("rel"))
def test_format_modes_expose_matching_controls_and_bookmark_ui(self):
cases = [
("pdf", "pdf", "30 页", ".reader-page", "application/pdf", b"pdf", False, False),
("txt", "text", "已加载", ".reader-text", "text/plain", b"Text readable", True, False),
("md", "markdown", "已加载", ".reader-markdown", "text/markdown", b"# Markdown readable", True, False),
("html", "html", "HTML", "iframe.html-frame", "text/html", b"HTML readable
", True, False),
("png", "image", "图片", ".reader-image", "image/png", PNG_BYTES, True, False),
("docx", "docx", "DOCX", ".docx-body", "application/vnd.openxmlformats-officedocument.wordprocessingml.document", minimal_docx(), False, False),
("wav", "audio", "音频", ".reader-audio", "audio/wav", minimal_wav(), True, True),
]
for extension, mode, status, content_selector, content_type, body, page_hidden, zoom_hidden in cases:
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_MODULE))
if extension == "md":
page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=MARKED_SCRIPT))
page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT))
if extension == "docx":
page.route("**/static/vendor/jszip.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=JSZIP_SCRIPT))
page.route("**/static/vendor/docx-preview.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=DOCX_SCRIPT))
page.route("**/api/reader-content**", lambda route, _request, content_type=content_type, body=body: route.fulfill(status=200, content_type=content_type, body=body))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/matrix.{extension}"
if extension == "docx":
source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/docx-native-v1/document.docx"
page.route("**/api/reader-resolve?id=capability-matrix", lambda route: route.fulfill(
status=200, content_type="application/json",
body=json.dumps({"url": source, "extension": extension, "title": "Matrix"}),
))
page.goto(f"{self.origin}/static/reader.html?id=capability-matrix", wait_until="domcontentloaded")
page.wait_for_function("expected => document.querySelector('#status').textContent === expected", arg=status)
self.assertEqual(page.locator(".reader-content").get_attribute("data-mode"), mode)
page.locator(content_selector).first.wait_for(state="attached")
self.assertTrue(page.locator("#bookmark-ribbon").is_visible())
ribbon = page.locator("#bookmark-ribbon").bounding_box()
self.assertIsNotNone(ribbon)
self.assertGreaterEqual(ribbon["x"], 0)
self.assertLessEqual(ribbon["x"] + ribbon["width"], 390)
self.assertEqual(page.locator(".page-controls").is_hidden(), page_hidden)
self.assertEqual(page.locator(".zoom-controls").is_hidden(), zoom_hidden)
self.assertEqual(page.locator("#full-search-toggle").evaluate("node => node.hidden"), mode in ("image", "audio"))
self.assertEqual(page.locator("#media-tab").evaluate("node => node.hidden"), mode != "audio")
self.assertFalse(page.locator(".reader-progress-bookmark").evaluate("node => node.hidden"))
if not zoom_hidden:
page.locator("#zoom-in").click()
self.assertEqual(page.locator("#zoom").input_value(), "110")
page.locator("#bookmark-ribbon").press("Enter")
self.assertTrue(page.locator("#bookmark-popover").is_visible())
page.locator("#bookmark-cancel").press("Escape")
self.assertTrue(page.locator("#bookmark-popover").is_hidden())
context.close()
def test_video_failure_shows_recoverable_reader_error(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="video/mp4", body=b"invalid video fixture"))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/broken.mp4"
page.route(source, lambda route: route.abort())
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=mp4&title=Broken", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertIn("媒体加载失败", page.locator(".reader-error").text_content())
self.assertEqual(page.locator("#status").text_content(), "无法打开")
self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_MEDIA")
self.assertFalse(page.locator(".reader-loading-indicator").count())
context.close()
def test_video_extension_aliases_report_media_errors_consistently(self):
for extension in ("mp4", "mov", "video"):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="video/mp4", body=b"invalid video fixture"))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/broken.{extension}"
page.route(source, lambda route: route.abort())
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Broken", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertEqual(page.locator("#status").text_content(), "无法打开")
context.close()
def test_unsupported_format_hides_inapplicable_controls(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/archive.zip"
page.route("**/api/reader-resolve?id=unsupported-controls", lambda route: route.fulfill(
status=200, content_type="application/json", body=json.dumps({"url": source, "extension": "zip"}),
))
page.goto(f"{self.origin}/static/reader.html?id=unsupported-controls", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertIn("此文件暂不支持在线阅读", page.locator(".reader-error").text_content())
self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_UNSUPPORTED")
self.assertTrue(page.locator(".page-controls").is_hidden())
self.assertTrue(page.locator(".zoom-controls").is_hidden())
for selector in ("#full-search-toggle", "#media-tab", "#bookmark-ribbon", ".reader-progress-bookmark"):
self.assertTrue(page.locator(selector).evaluate("node => node.hidden"), selector)
self.assertFalse(page.locator(".reader-loading-indicator").count())
context.close()
def test_converted_pdf_pages_reject_invalid_manifest_and_missing_first_page(self):
cases = [
("old-v1", {"version": 1, "kind": "pdf-pages", "pages": [{"page": 1, "path": f"objects/aa/{'a' * 64}/pages/page-000001.webp"}]}),
("bad-manifest", {"version": 2, "kind": "pdf-pages", "page_count": 0}),
("wrong-kind", {"version": 2, "kind": "pdf", "page_count": 1}),
("missing-page-count", {"version": 2, "kind": "pdf-pages"}),
("pages-field", {"version": 2, "kind": "pdf-pages", "page_count": 1, "pages": []}),
]
for name, manifest in cases:
with self.subTest(case=name):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/page-manifest.json"
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route, _request, manifest=manifest: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)))
page.route("https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/**", lambda route: route.fulfill(status=404, body=b""))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertTrue(page.locator(".reader-error").text_content().strip())
self.assertEqual(page.locator("#status").text_content(), "无法打开")
self.assertFalse(page.locator(".reader-loading-indicator").count())
context.close()
def test_converted_pdf_pages_report_error_when_later_page_is_missing(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
page.route("https://huggingface.co/**", lambda route: route.abort())
source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef/page-manifest.json"
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
manifest = {"version": 2, "kind": "pdf-pages", "page_count": 4}
missing_requests = []
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)))
def serve_page(route, _request):
if not route.request.url.endswith("page-000004.webp"):
route.fulfill(status=200, content_type="image/webp", body=IMAGE_FIXTURES["webp"][1])
else:
missing_requests.append(route.request.url)
route.fulfill(status=404, body=b"")
page.route(f"{root}/pages/**", serve_page)
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded")
page.locator(".reader-page img.ready").first.wait_for(timeout=10000)
page.locator(".reader-page[data-page='4']").scroll_into_view_if_needed()
page.wait_for_function("""() => {
const shell = document.querySelector('.reader-page[data-page="4"]');
return shell?.dataset.renderState === 'idle' && Number(shell.dataset.renderRetries) > 0;
}""")
self.assertTrue(missing_requests)
self.assertTrue(page.locator(".reader-page[data-page='1'] img.ready").evaluate(
"image => image.complete && image.naturalWidth > 0"))
self.assertNotEqual(page.locator(".reader-page[data-page='4']").get_attribute("data-render-state"), "rendered")
self.assertEqual(page.locator(".reader-page[data-page='4'] img.ready").count(), 0)
context.close()
def test_pdf_pages_stalled_early_manifest_uses_proxy(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
source = root + "/page-manifest.json"
proxy = []
page.add_init_script(f"""(() => {{
const fetchOriginal = window.fetch.bind(window);
window.__earlyManifestAborted = false;
window.fetch = (url, options = {{}}) => String(url) === {json.dumps(source)}
? new Promise((resolve, reject) => options.signal.addEventListener('abort', () => {{
window.__earlyManifestAborted = true;
reject(new DOMException('cancelled', 'AbortError'));
}}, {{once: true}}))
: fetchOriginal(url, options);
}})()""")
page.route("**/api/reader-content**", lambda route: (proxy.append(route.request.url), route.fulfill(
content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 2}))))
page.route(f"{root}/pages/**", lambda route: route.fulfill(
content_type="image/webp", body=IMAGE_FIXTURES["webp"][1]))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded")
page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=8000)
self.assertTrue(page.evaluate('window.__earlyManifestAborted'))
self.assertTrue(proxy)
def test_pdf_pages_progressive_prefetch_yields_to_jump(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
source = root + "/page-manifest.json"
held, requested = [], []
page.route("**/api/reader-content**", lambda route: route.fulfill(
content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 40})))
def serve_page(route):
number = int(route.request.url.rsplit("page-", 1)[1].split(".", 1)[0])
requested.append(number)
if number == 12:
held.append(route)
else:
route.fulfill(content_type="image/webp", body=IMAGE_FIXTURES["webp"][1])
page.route(f"{root}/pages/**", serve_page)
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded")
page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=10000)
page.wait_for_function("() => document.querySelector('.reader-page[data-page=\"12\"]')")
page.wait_for_timeout(2800)
self.assertIn(12, requested)
self.assertIn(13, requested)
self.assertIn(14, requested)
with page.expect_response(lambda response: response.url.endswith("page-000035.webp"), timeout=12000):
page.locator('.reader-page[data-page="25"]').scroll_into_view_if_needed()
page.locator('.reader-page[data-page="25"] img.ready').wait_for(timeout=10000)
for route in held:
try: route.abort()
except PlaywrightError: pass
self.assertIn(35, requested)
def test_pdf_pages_jump_does_not_wait_for_stalled_background_image(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
page.add_init_script("Object.defineProperty(navigator, 'connection', {value: {saveData: true}, configurable: true})")
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
source = root + "/page-manifest.json"
held, requested = [], []
page.route("**/api/reader-content**", lambda route: route.fulfill(
content_type="application/json", body=json.dumps({"version": 2, "kind": "pdf-pages", "page_count": 40})))
def serve_page(route):
number = int(route.request.url.rsplit("page-", 1)[1].split(".", 1)[0])
requested.append(number)
if number == 2:
held.append(route)
else:
route.fulfill(content_type="image/webp", body=IMAGE_FIXTURES["webp"][1])
page.route(f"{root}/pages/**", serve_page)
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages", wait_until="domcontentloaded")
page.locator('.reader-page[data-page="1"] img.ready').wait_for(timeout=10000)
page.wait_for_function("""() => document.querySelector('.reader-page[data-page="2"]')?._renderStarted""", timeout=10000)
page.locator("#page-number").fill("25")
page.locator("#page-number").dispatch_event("change")
page.wait_for_function("""() => document.querySelector('#viewport').scrollTop >= document.querySelector('.reader-page[data-page="25"]').offsetTop - 2""", timeout=3000)
page.locator('.reader-page[data-page="25"] img.ready').wait_for(timeout=3000)
self.assertIn(25, requested)
for route in held:
try: route.abort()
except PlaywrightError: pass
def test_converted_pdf_pages_fit_wide_images_without_overlap(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
page.route("https://huggingface.co/**", lambda route: route.abort())
source = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef/page-manifest.json"
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
manifest = {"version": 2, "kind": "pdf-pages", "page_count": 3, "toc": [{"title": "第一章", "page": 1, "depth": 0}, {"title": "第二章", "page": 2, "depth": 1}]}
wide_page = b''
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)))
page.route(f"{root}/pages/**", lambda route: route.fulfill(status=200, content_type="image/svg+xml", body=wide_page))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Converted", wait_until="domcontentloaded")
page.wait_for_function("""() => {
const images = [...document.querySelectorAll('.reader-page img.ready')];
return images.length === 3 && images.every(image =>
image.complete && image.naturalWidth === 2400 && image.naturalHeight === 3200);
}""", timeout=10000)
boxes = page.locator(".reader-page").evaluate_all("""pages => pages.slice(0, 3).map(page => {
const shell = page.getBoundingClientRect(), image = page.querySelector('img').getBoundingClientRect();
return { shell: { top: shell.top, right: shell.right, bottom: shell.bottom, left: shell.left, width: shell.width }, image: { top: image.top, right: image.right, bottom: image.bottom, left: image.left, width: image.width } };
})""")
self.assertEqual(len(boxes), 3)
for box in boxes:
self.assertLessEqual(box["image"]["width"], box["shell"]["width"] + 1)
self.assertGreaterEqual(box["image"]["left"], box["shell"]["left"] - 1)
self.assertLessEqual(box["image"]["right"], box["shell"]["right"] + 1)
for current, following in zip(boxes, boxes[1:]):
self.assertGreaterEqual(following["shell"]["top"], current["image"]["bottom"])
page.locator("#history").click()
self.assertEqual(page.locator("#toc-list .toc-item").count(), 2)
self.assertIn("第二章 · 第 2 页", page.locator("#toc-list .toc-item").nth(1).text_content())
page.locator("#toc-list .toc-item").nth(1).click()
page.wait_for_function("() => document.querySelector('#page-number').value === '2'")
context.close()
def test_compact_pdf_manifest_virtualizes_and_navigates_to_distant_page(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
root = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/" + "a" * 64 + "/1234567890abcdef"
source = root + "/page-manifest.json"
manifest = {"version": 2, "kind": "pdf-pages", "source_sha256": "a" * 64,
"profile": "test", "page_count": 5000}
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route(source, lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest)))
page.route(f"{root}/pages/**", lambda route: route.fulfill(status=200, content_type="image/webp", body=IMAGE_FIXTURES["webp"][1]))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=pdf-pages&title=Compact", wait_until="domcontentloaded")
page.locator(".reader-page img.ready").first.wait_for(timeout=10000)
self.assertLessEqual(page.locator(".reader-page img").count(), 25)
page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * 0.5")
page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 1500")
self.assertLessEqual(page.locator(".reader-page").count(), 160)
page.locator("#page-number").fill("4000")
page.locator("#viewport").evaluate("node => { node.scrollTop += node.clientHeight * 1.2; node.dispatchEvent(new Event('scroll')); }")
page.wait_for_timeout(80)
self.assertEqual(page.locator("#page-number").input_value(), "4000")
page.locator("#page-number").dispatch_event("change")
page.locator(".reader-page[data-page='4000'] img.ready").wait_for(timeout=10000)
self.assertEqual(page.locator("#page-number").input_value(), "4000")
self.assertLessEqual(page.locator(".reader-page").count(), 160)
self.assertEqual(page.locator(".reader-page[data-page='1']").count(), 0)
page.locator("#page-number").fill("1")
page.locator("#page-number").dispatch_event("change")
page.locator(".reader-page[data-page='1'] img.ready").wait_for(timeout=10000)
self.assertLessEqual(page.locator(".reader-page").count(), 160)
self.assertLessEqual(page.locator(".reader-page img").count(), 25)
context.close()
def test_long_native_pdf_keeps_bounded_page_shells(self):
module = PDF_TASK_MODULE.replace("numPages: 30", "numPages: 1200")
self.page.unroute("**/static/vendor/pdf.min.*.mjs")
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(
content_type="text/javascript", body=module))
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(SOURCE_URL, safe='')}&ext=pdf", wait_until="domcontentloaded")
self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for()
self.page.evaluate("""() => {
const run = document.querySelector('.reader-page[data-page="1"] .reader-pdf-text-run');
const range = document.createRange();
range.selectNodeContents(run);
getSelection().removeAllRanges();
getSelection().addRange(range);
}""")
self.page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * .84")
self.page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 900")
self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 1)
self.assertEqual(self.page.evaluate("getSelection().toString()"),
self.page.locator('.reader-page[data-page="1"] .reader-pdf-text-run').first.text_content())
self.page.evaluate("() => { getSelection().removeAllRanges(); document.activeElement.blur(); }")
self.page.locator("#viewport").evaluate("node => node.scrollTop = node.scrollHeight * .92")
self.page.wait_for_function("() => Number(document.querySelector('#page-number').value) > 1000")
self.page.wait_for_function("() => !document.querySelector('.reader-page[data-page=\"1\"]')")
self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 0)
self.assertLessEqual(self.page.locator(".reader-page").count(), 160)
self.assertEqual(self.page.locator('.reader-page[data-page="1"]').count(), 0)
self.page.locator("#page-number").fill("1")
self.page.locator("#page-number").dispatch_event("change")
self.page.locator('.reader-page[data-page="1"] canvas.ready').wait_for()
self.assertLessEqual(self.page.locator(".reader-page").count(), 160)
def test_reader_toolbar_controls_have_accessible_names_and_state(self):
self.open_pdf()
controls = self.page.locator(".reader-toolbar button, .reader-toolbar a")
names = self.page.locator(".reader-toolbar button:visible, .reader-toolbar a:visible").evaluate_all("""elements => elements.map(element => ({
id: element.id,
name: element.getAttribute('aria-label') || element.getAttribute('title') || element.textContent.trim(),
type: element.tagName === 'BUTTON' ? element.type : '',
target: element.tagName === 'A' ? element.target : '',
rel: element.tagName === 'A' ? element.rel : ''
}))""")
self.assertGreater(controls.count(), 0)
for item in names:
self.assertTrue(item["name"], item["id"])
if item["type"]:
self.assertEqual(item["type"], "button", item["id"])
if item["id"] == "download":
self.assertEqual(item["target"], "_blank")
self.assertIn("noopener", item["rel"])
self.page.locator("#history").click()
self.assertEqual(self.page.locator("#history").get_attribute("aria-expanded"), "true")
self.page.locator("#theme-toggle").click()
self.assertIn(self.page.locator("#theme-toggle").get_attribute("aria-label"), ("切换到白天模式", "切换到夜间模式"))
self.page.locator("#bookmark-ribbon").press("Enter")
self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "true")
self.page.locator("#bookmark-cancel").press("Escape")
self.assertEqual(self.page.locator("#bookmark-ribbon").get_attribute("aria-expanded"), "false")
def test_text_bookmark_uses_progress_excerpt_and_highlights_search(self):
self.page.unroute("**/api/reader-content**")
text = "\n\n".join(f"第 {index} 段 searchable-{index} 这是用于书签摘要搜索的正文内容。" * 5 for index in range(120))
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain; charset=utf-8", body=text.encode()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.txt"
query = urllib.parse.quote(source, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Bookmark", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for()
self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.5; element.dispatchEvent(new Event('scroll')); }")
self.page.locator("#bookmark-ribbon").click()
self.page.locator("#bookmark-label").fill("我的书签")
self.page.locator("#bookmark-excerpt-input").fill("自定义摘要")
self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.35; element.dispatchEvent(new Event('scroll')); }")
self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0)
self.assertIn("阅读进度", self.page.locator("#bookmark-prompt").text_content())
self.assertNotIn("px", self.page.locator("#bookmark-prompt").text_content())
self.assertRegex(self.page.locator("#bookmark-prompt").text_content(), r"阅读进度 \d+\.\d%")
self.assertTrue(self.page.locator("#bookmark-excerpt-input").input_value())
prompt_progress = float(self.page.locator("#bookmark-prompt").text_content().split("阅读进度 ", 1)[1].split("%", 1)[0])
self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight * 0.8; element.dispatchEvent(new Event('scroll')); }")
self.page.locator("#bookmark-add").click()
self.page.wait_for_function("() => window.__readerBookmarks.length === 1")
bookmark = self.page.evaluate("window.__readerBookmarks[0]")
self.assertEqual(bookmark["label"], "我的书签")
self.assertEqual(bookmark["excerpt"], "自定义摘要")
self.assertGreater(bookmark["progress"], 0)
self.assertAlmostEqual(bookmark["progress"], prompt_progress, places=1)
self.assertTrue(bookmark["excerpt"])
self.page.evaluate("window.__readerBookmarks.push({id: 'other', url: 'https://example.test/other.txt', title: '另一本书', label: '阅读进度 12.3%', excerpt: '跨书摘要', readerUrl: location.href, createdAt: Date.now() + 1})")
self.page.locator("#history").click()
self.page.locator("#bookmarks-tab").click()
self.page.locator("#bookmarks-list .bookmark-excerpt").wait_for()
self.assertEqual(self.page.locator("#bookmarks-list .panel-item").count(), 1)
self.page.locator("#bookmarks-all").click()
self.page.locator("#bookmarks-list .panel-item").nth(1).wait_for()
self.assertEqual(self.page.locator("#bookmarks-all").text_content(), "本书书签")
self.assertIn("另一本书", self.page.locator("#bookmarks-list").text_content())
term = bookmark["excerpt"].split()[0][:6]
self.page.locator("#bookmarks-panel .panel-search-toggle").click()
self.page.locator("#bookmarks-panel .panel-search").fill(term)
self.assertGreater(self.page.locator("#bookmarks-list mark.search-match").count(), 0)
self.page.locator("#bookmarks-panel .panel-search").fill("")
self.page.locator("#bookmarks-list .panel-item-edit").first.click()
self.page.locator("#bookmark-label").fill("修改后的标题")
self.page.locator("#bookmark-excerpt-input").fill("修改后的摘要")
self.page.locator("#bookmark-add").click()
self.assertIn("修改后的标题", self.page.locator("#bookmarks-list").text_content())
def test_full_text_search_lists_highlighted_snippets_and_jumps(self):
self.page.unroute("**/api/reader-content**")
text = ("开头内容。" * 80) + "正文目标词出现在这里,前后都有上下文。" + ("中间内容。" * 120) + "正文目标词再次出现。"
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=text.encode()))
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/full-search.txt", safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=FullSearch", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for()
self.assertEqual(self.page.locator("#full-search-view").count(), 1)
self.assertEqual(self.page.locator(".full-search-toggle").count(), 0)
self.page.locator("#history").click()
self.page.locator("#full-search-toggle").click()
self.assertFalse(self.page.locator("#full-search-view").is_hidden())
self.page.locator("#full-search-toggle").click()
self.page.wait_for_timeout(300)
self.assertTrue(self.page.locator("#history-panel").is_visible())
self.assertTrue(self.page.locator("#full-search-view").is_hidden())
self.page.locator("#full-search-toggle").click()
self.page.locator("#full-search-input").fill("目标词")
self.page.locator("#full-search-status").filter(has_text="2 个结果").wait_for()
self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 2)
self.assertEqual(self.page.locator("#full-search-results mark.search-match").count(), 2)
self.assertEqual(self.page.locator(".full-search-highlight").count(), 2)
self.assertLessEqual(len(self.page.locator("#full-search-results .full-search-snippet").first.text_content()), 180)
self.page.locator("#full-search-input").fill("阅读选项")
self.page.locator("#full-search-status").filter(has_text="未找到").wait_for()
self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0)
self.page.locator("#full-search-input").fill("目标词")
self.page.locator("#full-search-status").filter(has_text="2 个结果").wait_for()
self.page.locator("#full-search-results .full-search-result").nth(1).click()
self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0)
def test_pdf_allows_two_bookmarks_on_one_page_and_restores_offsets(self):
self.open_pdf()
self.page.locator("#bookmark-ribbon").click()
self.page.locator("#bookmark-add").click()
self.page.locator("#viewport").evaluate("element => { element.scrollTop = 360; element.dispatchEvent(new Event('scroll')); }")
self.page.locator("#bookmark-ribbon").click()
self.page.locator("#bookmark-add").click()
bookmarks = self.page.evaluate("window.__readerBookmarks")
self.assertEqual(len(bookmarks), 2)
self.assertNotEqual(bookmarks[0]["id"], bookmarks[1]["id"])
self.assertLess(bookmarks[0]["pageOffset"], bookmarks[1]["pageOffset"])
self.page.locator("#history").click()
self.page.locator("#bookmarks-tab").click()
rows = self.page.locator("#bookmarks-list .panel-item-main")
self.assertEqual(rows.count(), 2)
self.page.locator("#viewport").evaluate("element => element.scrollTop = 0")
rows.nth(1).click()
self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 250)
rows.nth(0).click()
self.assertLess(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 80)
def test_bfcache_pageshow_waits_for_restoration_gate(self):
delayed_store = STORE_SCRIPT.replace(
"get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))",
"get: () => new Promise((resolve) => setTimeout(() => resolve({url: location.href, scrollTop: 640, zoom: 100}), 1200))",
)
self.page.unroute("**/static/reader-store.js")
self.page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=delayed_store))
query = urllib.parse.quote(SOURCE_URL, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded")
self.page.evaluate("""() => {
for (const type of ["pagehide", "pageshow"]) {
const event = new Event(type);
Object.defineProperty(event, "persisted", { value: true });
window.dispatchEvent(event);
}
}""")
self.page.wait_for_timeout(650)
self.assertNotEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed")
self.assertIsNone(self.page.evaluate("window.__savedReaderProgress || null"))
self.page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.scrollTop >= 600", timeout=3000)
def test_pagehide_before_restoration_does_not_overwrite_progress(self):
query = urllib.parse.quote(SOURCE_URL, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance", wait_until="domcontentloaded")
self.page.evaluate("window.dispatchEvent(new Event('pagehide'))")
self.page.wait_for_timeout(150)
self.assertEqual(self.page.locator("html").get_attribute("data-reader-phase"), "disposed")
self.assertIsNone(self.page.evaluate("window.__savedReaderProgress || null"))
def test_blocked_v1_upgrade_does_not_block_document_loading(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
blocker = context.new_page()
blocker.goto(f"{self.origin}/static/reader.html", wait_until="domcontentloaded")
blocker.evaluate("""async () => {
await new Promise((resolve) => { const request = indexedDB.deleteDatabase('voiceofml-reader'); request.onsuccess = request.onerror = request.onblocked = resolve; });
window.__heldDb = await new Promise((resolve, reject) => { const request = indexedDB.open('voiceofml-reader', 1); request.onupgradeneeded = () => { const store = request.result.createObjectStore('entries', {keyPath: 'url'}); store.createIndex('lastReadAt', 'lastReadAt'); }; request.onsuccess = () => resolve(request.result); request.onerror = () => reject(request.error); });
}""")
reader = context.new_page()
reader.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader"))
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/blocked.txt", safe="")
reader.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=Blocked", wait_until="domcontentloaded")
reader.locator(".reader-text").wait_for(timeout=5000)
blocker.evaluate("window.__heldDb.close()")
context.close()
def test_html_bookmark_at_zero_restores_iframe_top(self):
self.page.unroute("**/api/reader-content**")
document = b"Top bookmark content
Bottom
"
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.html"
query = urllib.parse.quote(source, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=html&title=HTML", wait_until="domcontentloaded")
self.page.locator(".html-frame").wait_for()
self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id: url + '\\0top', url, label: '阅读进度 0.0%', htmlScrollTop: 0, scrollTop: 0, createdAt: 1})", source)
self.page.locator("#history").click()
self.page.locator("#bookmarks-tab").click()
bookmark = self.page.locator("#bookmarks-list .panel-item-main")
bookmark.wait_for()
self.page.locator(".html-frame").evaluate("frame => frame.contentWindow.scrollTo(0, 900)")
bookmark.click()
self.assertEqual(self.page.locator(".html-frame").evaluate("frame => frame.contentWindow.scrollY"), 0)
def test_stale_bookmark_query_cannot_overwrite_all_bookmarks(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
store = """window.VoiceOfMLReaderStore = Object.freeze({
get: () => Promise.resolve(null), put: () => Promise.resolve(), list: () => Promise.resolve([]), remove: () => Promise.resolve(), clearHistory: () => Promise.resolve(),
putBookmark: () => Promise.resolve(), removeBookmark: () => Promise.resolve(),
listBookmarks: (url) => new Promise((resolve) => setTimeout(() => resolve([{id:'local', url, title:'Current book', label:'Current mark', createdAt:1}]), 180)),
listAllBookmarks: () => new Promise((resolve) => setTimeout(() => resolve([{id:'all', url:'other', readerUrl:location.href, title:'All book', label:'All mark', createdAt:2}]), 10))
});"""
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=store))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader"))
source = urllib.parse.quote("https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/race.txt", safe="")
page.goto(f"{self.origin}/static/reader.html?url={source}&ext=txt&title=Race", wait_until="domcontentloaded")
page.locator(".reader-text").wait_for()
page.locator("#history").click()
page.locator("#bookmarks-tab").click()
page.locator("#bookmarks-all").click()
page.locator("#bookmarks-list .panel-item-main", has_text="All book").wait_for()
page.wait_for_timeout(220)
self.assertEqual(page.locator("#bookmarks-list .panel-item").count(), 1)
self.assertIn("All book", page.locator("#bookmarks-list .panel-item-main").text_content())
context.close()
def test_truncated_epub_reports_source_damage(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=b"PK\x03\x04truncated"))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/damaged.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Damaged", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertTrue(page.locator(".reader-error").text_content().strip())
self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_CORRUPT")
context.close()
def test_malicious_html_css_and_svg_are_inert_and_keep_safe_text(self):
document = b'''Safe reader text
'''
self.page.unroute("**/api/reader-content**")
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document))
external = []
self.page.on("request", lambda request: external.append(request.url) if "evil.test" in request.url else None)
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/security.html"
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=html&title=Security", wait_until="domcontentloaded")
self.page.wait_for_function("() => document.querySelector('#status').textContent === 'HTML'")
frame = self.page.locator("iframe.html-frame").content_frame
self.assertEqual(frame.locator("#safe").text_content(), "Safe reader text")
self.assertEqual(frame.locator("body script,body link,body svg").count(), 0)
self.assertEqual(frame.locator("body img[src], body [href^='javascript:']").count(), 0)
self.assertEqual(frame.locator("[onclick], [href^='javascript:']").count(), 0)
self.assertFalse(self.page.evaluate("window.__unsafe === true")); self.assertEqual(external, [])
def test_oversized_chapter_manifest_and_response_use_resource_limit(self):
digest = "a" * 64; source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/chapter-manifest.json"; limit = 8 * 1024 * 1024
manifest = {"version": 1, "kind": "epub-chapters", "chapters": [{"index": 1, "path": "chapter.xhtml", "bytes": 10}]}
for name, headers, chapter_headers in (("manifest", {"content-length": str(limit + 1)}, None), ("chapter", {}, {"content-length": str(8 * 1024 * 1024 + 1)})):
with self.subTest(case=name):
context = self.browser.new_context(viewport={"width": 390, "height": 844}); page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
def serve_limited(route, _request, headers=headers, chapter_headers=chapter_headers):
if chapter_headers and "chapter.xhtml" in route.request.url:
route.fulfill(status=200, content_type="text/html", headers=chapter_headers, body=b"chapter
")
else:
route.fulfill(status=200, content_type="application/json", headers=headers, body=json.dumps(manifest))
page.route("**/api/reader-content**", serve_limited)
if chapter_headers: page.route("https://huggingface.co/**", lambda route, _request, headers=chapter_headers: route.fulfill(status=200, content_type="text/html", headers=headers, body=b"chapter
"))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Limits", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000); self.assertIn(page.locator("#content").get_attribute("data-error-code"), ("READER_PARSE", "READER_CORRUPT")); expected_limit = limit if name == "manifest" else 8 * 1024 * 1024; self.assertEqual(page.evaluate("limit => { try { VoiceOfMLReaderSecurity.assertResponseSize({headers:{get:()=>String(limit + 1)}}, limit); return null; } catch (error) { return error.message; } }", expected_limit), "READER_RESOURCE_LIMIT"); context.close()
def test_zip_bomb_metadata_is_rejected_before_docx_and_foliate_parsers(self):
payload = zip_bomb_metadata()
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
self.assertIsNone(archive.testzip())
entry = archive.getinfo("bomb.txt")
self.assertGreater(entry.file_size / entry.compress_size, 200)
for extension in ("epub", "docx"):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/static/vendor/jszip.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.JSZip=function(){window.__archiveParserStarted=true};"))
page.route("**/static/vendor/docx-preview.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.docx={renderAsync(){window.__archiveParserStarted=true}};"))
page.route("**/static/foliate-reader/view.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="customElements.define('foliate-view', class extends HTMLElement { open() { window.__archiveParserStarted=true; throw new Error('parser must not start'); } });"))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/octet-stream", body=payload))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bomb.epub" if extension == "epub" else f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{'a' * 64}/docx-native-v1/document.docx"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Bomb", wait_until="domcontentloaded")
page.locator(".reader-error").wait_for(timeout=10000)
self.assertIn("READER_ARCHIVE_LIMIT", page.locator(".reader-error").text_content())
self.assertEqual(page.locator("#content").get_attribute("data-error-code"), "READER_CORRUPT" if extension == "epub" else "READER_PARSE")
self.assertFalse(page.evaluate("window.__archiveParserStarted === true"))
context.close()
def test_actual_store_broadcasts_progress_and_bookmark_updates_between_readers(self):
self.page.unroute("**/static/reader-store.js"); self.page.unroute("**/api/reader-content**")
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Shared reader"))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/shared.txt"; query = urllib.parse.quote(source, safe="")
self.page.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Shared", wait_until="domcontentloaded"); self.page.locator(".reader-text").wait_for()
other = self.context.new_page(); other.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Shared reader")); other.goto(f"{self.origin}/static/reader.html?url={query}&ext=txt&title=Shared", wait_until="domcontentloaded"); other.locator(".reader-text").wait_for(); other.locator("#history").click()
self.page.evaluate("url => VoiceOfMLReaderStore.put({url, title:'Shared', extension:'txt', lastReadAt:Date.now(), scrollTop:321})", source)
self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id:'shared-bookmark', url, title:'Shared', label:'Shared mark', createdAt:Date.now()})", source)
other.wait_for_function("() => [...document.querySelectorAll('#history-list .panel-item-main')].some(item => item.textContent.includes('Shared'))"); other.locator("#bookmarks-tab").click(); other.locator("#bookmarks-list .panel-item-main").filter(has_text="Shared mark").wait_for()
def test_html_and_markdown_toc_click_navigation(self):
for extension, body in (("html", b"HTML one
space
HTML two
"), ("md", b"# Markdown one\n\n## Markdown two")):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844}); page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
if extension == "md":
page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body="window.marked={parse:()=>'Markdown one
Markdown two
'};")); page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT))
page.route("**/api/reader-content**", lambda route, _request, body=body: route.fulfill(status=200, content_type="text/html" if extension == "html" else "text/markdown", body=body))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/toc.{extension}"; page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=TOC", wait_until="domcontentloaded"); page.locator("#history").click(); page.locator("#toc-list .panel-item-main").nth(1).click()
page.wait_for_function("() => document.querySelector('iframe') ? document.querySelector('iframe').contentWindow.scrollY > 0 : document.querySelector('#viewport').scrollTop > 0"); context.close()
def test_pdf_outline_does_not_delay_ready(self):
module = PDF_MODULE.replace("getOutline: () => Promise.resolve(window.__pdfOutlineEnabled ? [{title: '第一章', dest: [{}], items: []}] : null)",
"getOutline: () => new Promise(resolve => { window.__releaseOutline = () => resolve([{title: '第一章', dest: [{}], items: []}]); })")
self.page.unroute("**/static/vendor/pdf.min.*.mjs")
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(content_type="text/javascript", body=module))
self.open_pdf()
self.page.locator("html[data-reader-phase='ready']").wait_for()
self.page.locator(".reader-page canvas.ready").first.wait_for()
self.assertEqual(self.page.locator("#toc-list .toc-item").count(), 0)
self.page.evaluate("window.__releaseOutline()")
self.page.locator("#toc-list .toc-item").first.wait_for(state="attached")
def test_pdf_outline_click_navigates_to_declared_page(self):
self.page.add_init_script("window.__pdfOutlineEnabled = true")
module = PDF_MODULE.replace("[{title: '第一章', dest: [{}], items: []}]", "[{title: '第三章', dest: [{}], items: []}]").replace("getPageIndex: () => Promise.resolve(0)", "getPageIndex: () => Promise.resolve(2)"); self.page.unroute("**/static/vendor/pdf.min.*.mjs"); self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module)); self.open_pdf(); self.page.locator("#history").click(); self.page.locator("#toc-list .panel-item-main").click(); self.page.wait_for_function("() => document.querySelector('#page-number').value === '3'")
def test_pdf_outline_jump_scrolls_before_target_render_finishes(self):
self.page.add_init_script("window.__pdfProbe = { loads: 0, destroys: 0, renders: [], cancels: [], releases: {}, holdPages: [20] }")
module = PDF_TASK_MODULE.replace(
"getOutline: async () => null,",
"getOutline: async () => [{title: '第二十页', dest: [{}], items: []}],\n getPageIndex: async () => 19,",
)
self.page.unroute("**/static/vendor/pdf.min.*.mjs")
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module))
self.open_pdf()
self.page.locator("#history").click()
self.page.locator("#toc-list .panel-item-main").click()
self.page.wait_for_function("""() => {
const viewport = document.querySelector('#viewport');
return document.querySelector('#page-number').value === '20' &&
viewport.scrollTop > 1000 && Boolean(window.__pdfProbe.releases[20]);
}""")
self.assertGreater(self.page.locator("#viewport").evaluate("node => node.scrollTop"), 1000)
self.page.evaluate("window.__pdfProbe.releases[20]()")
self.page.locator('.reader-page[data-page="20"] canvas.ready').wait_for()
def test_pdf_outline_marks_current_entry_when_page_changes(self):
self.page.add_init_script("window.__pdfOutlineEnabled = true")
module = PDF_MODULE.replace(
"[{title: '第一章', dest: [{}], items: []}]",
"[{title: '第一章', dest: [{page: 0}], items: []}, {title: '第二章', dest: [{page: 2}], items: []}]",
).replace("getPageIndex: () => Promise.resolve(0)", "getPageIndex: (ref) => Promise.resolve(ref.page || 0)")
self.page.unroute("**/static/vendor/pdf.min.*.mjs")
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route, _request, module=module: route.fulfill(status=200, content_type="text/javascript", body=module))
self.open_pdf()
self.page.locator("#history").click()
self.page.locator("#toc-list .toc-item").nth(1).wait_for()
self.assertTrue(self.page.locator("#toc-list .toc-item").nth(0).evaluate("row => row.classList.contains('is-current')"))
self.page.evaluate("""() => {
const viewport = document.querySelector('#viewport');
viewport.scrollTop = document.querySelector('.reader-page[data-page="3"]').offsetTop;
viewport.dispatchEvent(new Event('scroll'));
}""")
self.page.wait_for_function("() => document.querySelector('#page-number').value === '3'")
self.assertTrue(self.page.locator("#toc-list .toc-item").nth(1).evaluate("row => row.classList.contains('is-current')"))
def test_pdf_stale_navigation_and_search_keep_latest_state(self):
errors = []
self.page.on("pageerror", lambda error: errors.append(str(error)))
self.page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(
status=200, content_type="text/javascript", body=PDF_TASK_MODULE,
))
self.open_pdf()
self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'")
self.page.evaluate("window.__pdfProbe.holdPages = [20]")
self.page.locator("#page-number").fill("20")
self.page.locator("#page-number").dispatch_event("change")
self.page.wait_for_function("() => Boolean(window.__pdfProbe.releases[20])")
self.page.locator("#page-number").fill("3")
self.page.locator("#page-number").dispatch_event("change")
self.page.wait_for_function("() => Math.abs(document.querySelector('#viewport').scrollTop - document.querySelector('[data-page=\"3\"]').offsetTop) < 2")
self.page.evaluate("""async () => {
const pending = document.querySelector('[data-page="20"]')._renderPromise;
window.__pdfProbe.releases[20](); await pending.catch(() => {});
await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve)));
}""")
self.assertEqual(self.page.locator("#page-number").input_value(), "3")
self.assertAlmostEqual(self.page.locator("#viewport").evaluate("node => node.scrollTop"),
self.page.locator('[data-page="3"]').evaluate("node => node.offsetTop"), delta=2)
self.page.wait_for_function("() => window.__savedReaderProgress?.page === 3")
self.page.locator("#history").click()
self.page.locator("#full-search-toggle").click()
for action in ("replace", "clear"):
with self.subTest(action=action):
if action == "clear":
# A completed search caches extracted text. Use a fresh
# document so cancellation still exercises an in-flight read.
self.open_pdf()
self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'")
self.page.locator("#history").click()
self.page.locator("#full-search-toggle").click()
self.page.evaluate("window.__pdfProbe.holdText = true; delete window.__pdfProbe.releaseText")
self.page.locator("#full-search-input").fill("obsolete")
self.page.wait_for_function("() => Boolean(window.__pdfProbe.releaseText)")
if action == "replace":
self.page.locator("#full-search-input").fill("current")
self.page.locator("#full-search-status").filter(has_text="1 个结果").wait_for()
else:
self.page.locator("#full-search-clear").click()
self.page.evaluate("""async () => {
window.__pdfProbe.releaseText();
await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve)));
}""")
self.assertEqual(self.page.locator('[data-page="1"] .full-search-highlight').count(), 0)
if action == "replace":
self.assertEqual(self.page.locator("#full-search-results .full-search-location").all_text_contents(), ["第 2 页"])
self.assertEqual(self.page.locator("#full-search-results mark").all_text_contents(), ["current"])
self.assertEqual(self.page.locator("#content .full-search-highlight").all_text_contents(), ["current"])
self.assertFalse(self.page.locator("#full-search-next").is_disabled())
else:
self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0)
self.assertEqual(self.page.locator("#content .full-search-highlight").count(), 0)
self.assertEqual(self.page.locator("#full-search-status").text_content(), "输入关键词搜索正文")
self.assertTrue(self.page.locator("#full-search-next").is_disabled())
self.assertTrue(self.page.locator("#full-search-prev").is_disabled())
self.assertEqual(errors, [])
def test_pdf_disposal_destroys_loading_task_and_cancels_render_queue(self):
for phase in ("loading", "rendering"):
with self.subTest(phase=phase):
context = self.browser.new_context()
self.addCleanup(context.close)
page = context.new_page()
errors = []
page.on("pageerror", lambda error: errors.append(str(error)))
page.add_init_script(f"window.__holdPdfLoad = {json.dumps(phase == 'loading')}")
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
requests = []
page.on("request", lambda request: requests.append(request.url) if "reader-content" in request.url or request.url == SOURCE_URL else None)
page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PDF_TASK_MODULE))
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(SOURCE_URL, safe='')}&ext=pdf", wait_until="domcontentloaded")
page.wait_for_function("() => Boolean(window.__pdfProbe)")
if phase == "rendering":
page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'")
page.evaluate("""() => {
window.__pdfProbe.holdPages = [20, 21, 22];
for (const number of [20, 21, 22]) {
document.querySelector(`[data-page="${number}"]`).dispatchEvent(new Event('focus'));
}
}""")
page.wait_for_function("() => Boolean(window.__pdfProbe.releases[20] && window.__pdfProbe.releases[21])")
self.assertFalse(page.evaluate("Boolean(window.__pdfProbe.releases[22])"))
page.evaluate("window.dispatchEvent(new PageTransitionEvent('pagehide', {persisted: true}))")
self.assertEqual(page.evaluate("window.__pdfProbe.destroys"), 0)
self.assertEqual(page.evaluate("window.__pdfProbe.cancels"), [])
page.evaluate("window.dispatchEvent(new Event('pagehide'))")
page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'disposed'")
before = page.evaluate("({renders: [...window.__pdfProbe.renders], saved: window.__savedReaderProgress || null})")
page.evaluate("""async () => {
const pending = [...document.querySelectorAll('.reader-page')].map(node => node._renderPromise).filter(Boolean);
window.__pdfProbe.releaseLoad?.();
Object.values(window.__pdfProbe.releases).forEach(resolve => resolve());
await Promise.allSettled(pending);
window.dispatchEvent(new Event('pagehide'));
}""")
page.wait_for_timeout(550) # Cross delayed history restoration and the save debounce.
self.assertEqual(page.evaluate("window.__pdfProbe.destroys"), 1)
self.assertEqual(page.evaluate("window.__pdfProbe.loads"), 1)
self.assertEqual(page.evaluate("window.__pdfProbe.renders"), before["renders"])
self.assertEqual(page.evaluate("window.__savedReaderProgress || null"), before["saved"])
self.assertEqual(page.locator(".reader-error").count(), 0)
self.assertEqual(page.locator("html").get_attribute("data-reader-phase"), "disposed")
self.assertTrue(page.locator(".reader-page canvas").evaluate_all("nodes => nodes.every(node => node.width === 0 && node.height === 0)"))
self.assertEqual(sorted(page.evaluate("window.__pdfProbe.cancels")), [20, 21] if phase == "rendering" else [])
if phase == "loading":
self.assertEqual(page.locator(".reader-page").count(), 0)
self.assertIsNone(before["saved"])
self.assertEqual(requests, [])
self.assertEqual(errors, [])
context.close()
def test_epub_chapters_load_first_lazy_next_and_toc_destination(self):
digest = "b" * 64
source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/chapter-manifest.json"
manifest = {"version": 1, "kind": "epub-chapters", "chapters": [
{"index": i, "path": f"chapter-{i}.xhtml", "title": f"Chapter {i}", "bytes": 100}
for i in range(1, 5)
]}
requests, delayed = [], []
self.page.unroute("**/api/reader-content**")
def serve_chapter(route, _request):
number = next((i for i in range(1, 5) if f"chapter-{i}.xhtml" in route.request.url), None)
if number is None:
route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest))
return
requests.append(number)
route.fulfill(status=200, content_type="text/html", body=f"Chapter {number}
body
")
self.page.route("**/api/reader-content**", serve_chapter)
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Chapters", wait_until="domcontentloaded")
self.page.locator(".reader-epub-chapter[data-chapter='1']").wait_for()
self.assertEqual(set(requests), {1, 2, 3, 4})
self.page.locator(".reader-epub-chapter[data-chapter='2']").wait_for()
self.page.locator("#history").click()
self.page.locator("#toc-list .panel-item-main").nth(2).click()
self.page.locator(".reader-epub-chapter[data-chapter='3']").wait_for()
self.page.locator(".reader-epub-chapter[data-chapter='4']").wait_for()
self.page.evaluate("() => new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve)))")
self.assertEqual(self.page.locator(".reader-epub-chapter").evaluate_all("nodes => nodes.map(node => Number(node.dataset.chapter))"), [1, 2, 3, 4])
self.assertEqual(self.page.locator(".reader-chapter-sentinel").count(), 0)
self.assertEqual(sorted(requests), [1, 2, 3, 4])
self.assertTrue(self.page.locator("#toc-list .toc-item").nth(2).evaluate("node => node.classList.contains('is-current')"))
def test_epub_chapters_serve_nested_bundle_with_resources(self):
digest = "c" * 64
base = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/foliate-original-v1/epub-chapters"
source = base + "/chapter-manifest.json"
chapters = {
"chapters/chapter-0001.xhtml": '第一章

one
'.encode(),
"chapters/chapter-0002.xhtml": "第二章
two
".encode(),
}
manifest = {"version": 1, "kind": "epub-chapters", "chapters": [
{"index": i, "title": f"第{i}章", "path": path, "bytes": len(body),
"sha256": hashlib.sha256(body).hexdigest()}
for i, (path, body) in enumerate(chapters.items(), 1)
]}
image_url = base + "/resources/img.png"
image_requests, unexpected_requests = [], []
def reject_external(route):
unexpected_requests.append(route.request.url)
route.abort()
def serve_image(route):
image_requests.append(route.request.url)
route.fulfill(status=200, content_type="image/png", body=PNG_BYTES)
def serve_nested(route, _request):
url = urllib.parse.parse_qs(urllib.parse.urlsplit(route.request.url).query)["url"][0]
if url == source:
route.fulfill(status=200, content_type="application/json", body=json.dumps(manifest).encode())
elif url.startswith(base + "/") and url[len(base) + 1:] in chapters:
route.fulfill(status=200, content_type="text/html", body=chapters[url[len(base) + 1:]])
else:
unexpected_requests.append(url)
route.fulfill(status=404, body=b"no")
self.page.unroute("**/api/reader-content**")
self.page.route("**/api/reader-content**", serve_nested)
self.page.route("https://huggingface.co/**", reject_external)
self.page.route(image_url, serve_image)
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub-chapters&title=Nested", wait_until="domcontentloaded")
first = self.page.locator(".reader-epub-chapter[data-chapter='1']")
first.wait_for()
self.assertIn("第一章", first.inner_text())
image = first.locator("img")
image.wait_for()
self.page.wait_for_function("""() => {
const image = document.querySelector('.reader-epub-chapter[data-chapter="1"] img');
return image?.complete && image.naturalWidth === 1 && image.naturalHeight === 1;
}""")
self.assertEqual(image.get_attribute("src"), image_url)
self.assertEqual(image_requests, [image_url])
self.page.locator("#history").click()
self.page.locator("#toc-list .panel-item-main").nth(1).click()
second = self.page.locator(".reader-epub-chapter[data-chapter='2']")
second.wait_for()
self.assertIn("第二章", second.inner_text())
self.assertEqual(unexpected_requests, [])
def test_foliate_normalizes_legacy_chm_markup_and_keeps_resources(self):
# This case verifies preservation of source CSS; night colors have a separate contract.
self.page.add_init_script("localStorage.setItem('theme', 'light')")
with zipfile.ZipFile(io.BytesIO(epub_with_legacy_chm_markup())) as archive:
files = {name: archive.read(name) for name in archive.namelist()}
chapter_path = "OEBPS/chapters space%20/chapter.xhtml"
files["OEBPS/content.opf"] = files["OEBPS/content.opf"].replace(
b'href="chapter.xhtml"',
f'href="{urllib.parse.quote(chapter_path.removeprefix("OEBPS/"))}"'.encode(),
)
resource_files = {
f"OEBPS/images %23/{name}": PNG_BYTES
for name in ("cover#1.png", "cover space.png", "literal%20.png", "literal%23.png")
}
resource_files["OEBPS/chapters space%20/local image.png"] = PNG_BYTES
images = []
for path in resource_files:
relative = (path.removeprefix("OEBPS/chapters space%20/")
if path.startswith("OEBPS/chapters space%20/")
else "../" + path.removeprefix("OEBPS/"))
images.append(f'')
chapter = files.pop("OEBPS/chapter.xhtml").decode().replace(
'href="style.css"', 'href="../style.css"',
).replace('src="picture.svg"', 'src="../picture.svg"')
files[chapter_path] = chapter.replace(
"'
+ "".join(images) + '查看原图 getComputedStyle(element).color"), "rgb(1, 2, 3)")
self.assertTrue((image.get_attribute("src") or "").startswith("blob:"))
self.assertGreater(image.bounding_box()["height"], 0)
self.page.wait_for_function("""count => performance.getEntriesByType('resource')
.filter(entry => entry.name.includes('/api/reader-resource?')).length >= count""",
arg=len(resource_files))
self.assertEqual(set(requested_paths), set(resource_files))
resource_paths = self.page.locator(".foliate-continuous article[data-section='0'] svg image").evaluate_all("""images =>
images.map(image => new URL(image.getAttribute('href'), location.origin).searchParams.get('path'))""")
self.assertCountEqual(resource_paths, resource_files)
with self.page.expect_popup() as opened:
self.page.locator('.foliate-continuous #full-image').click()
popup = opened.value
popup.wait_for_load_state()
self.assertEqual(urllib.parse.parse_qs(urllib.parse.urlsplit(popup.url).query)['path'],
['OEBPS/chapters space%20/local image.png'])
popup.wait_for_function('() => document.querySelector("img")?.naturalWidth === 1')
popup.close()
def test_foliate_navigation_path_dark_links_and_unique_sections(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
self.addCleanup(context.close)
page = context.new_page()
page.add_init_script("localStorage.setItem('theme', 'dark')")
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_navigation()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/e2e.epub"
url = f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=E2E&path=Test%2Fbooks"
page.goto(url, wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 2")
link = page.locator(".foliate-continuous article[data-section='0'] a").first
link.wait_for(state="attached")
self.assertIsNotNone(page.locator(".foliate-continuous article[data-section='0']").evaluate("article => article.shadowRoot"))
title_display = page.locator("#title").evaluate("element => getComputedStyle(element).display")
page.locator(".foliate-continuous article[data-section='0']").evaluate("article => { const style = document.createElement('style'); style.textContent = '#title{display:none!important} a{color:rgb(1,2,3)!important;border-top-style:dotted}'; article.shadowRoot.appendChild(style); }")
self.assertEqual(page.locator("#title").evaluate("element => getComputedStyle(element).display"), title_display)
self.assertEqual(link.evaluate("element => getComputedStyle(element).borderTopStyle"), "dotted")
self.assertEqual(link.evaluate("element => getComputedStyle(element).color"), "rgb(138, 180, 232)")
page.locator("#history").click()
for theme, color in (("light", "rgb(1, 2, 3)"), ("dark", "rgb(138, 180, 232)")):
page.locator("#theme-toggle").click()
page.wait_for_function("() => !document.documentElement.classList.contains('theme-transition')")
self.assertEqual(page.locator("html").get_attribute("data-theme"), theme)
self.assertEqual(link.evaluate("element => getComputedStyle(element).color"), color)
page.locator("#history").click()
self.assertEqual(page.locator("#reader-path").text_content(), "Test/books")
self.assertIsNone(page.locator("#reader-path").get_attribute("hidden"))
initial_url = page.url
link.click()
page.wait_for_function("() => document.querySelector('#viewport').scrollTop > 0")
self.assertEqual(page.url, initial_url)
self.assertAlmostEqual(page.locator("#one").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2)
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(1).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')")
page.wait_for_function("() => document.querySelectorAll('.foliate-continuous article[data-section]').length === 3")
sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => item.dataset.section)")
self.assertEqual(sections, ["0", "1", "2"])
page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight; element.dispatchEvent(new Event('scroll')); }")
page.wait_for_timeout(250)
self.assertEqual(page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => item.dataset.section)"), ["0", "1", "2"])
context.close()
def test_foliate_rapid_toc_navigation_keeps_latest_destination(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/race.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Race", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.evaluate("""() => {
const sections = document.querySelector('foliate-view').book.sections.filter(section => section.linear !== 'no');
for (const [index, delay] of [[6, 350], [10, 20]]) {
const original = sections[index].createDocument.bind(sections[index]);
sections[index].createDocument = () => new Promise((resolve, reject) => setTimeout(() => original().then(resolve, reject), delay));
}
document.querySelectorAll('#toc-list .panel-item-main')[5].click();
document.querySelectorAll('#toc-list .panel-item-main')[9].click();
}""")
page.wait_for_timeout(700)
self.assertTrue(page.locator("#toc-list .toc-item").nth(9).evaluate("item => item.classList.contains('is-current')"))
self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2)
sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => Number(item.dataset.section))")
self.assertEqual(sections, sorted(set(sections)))
context.close()
def test_foliate_failed_toc_section_can_retry(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
errors = []
page.on("pageerror", lambda error: errors.append(str(error)))
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/retry.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Retry", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.evaluate("""() => {
const sections = document.querySelector('foliate-view').book.sections.filter(item => item.linear !== 'no');
const section = sections[10];
window.__neighborFailures = 0;
sections[9].createDocument = () => { window.__neighborFailures++; return Promise.reject(new Error('neighbor unavailable')); };
const original = section.createDocument.bind(section); let attempts = 0;
section.createDocument = () => ++attempts === 1 ? Promise.reject(new Error('transient section failure')) : original();
window.__retrySection = () => document.querySelectorAll('#toc-list .panel-item-main')[9].click();
}""")
page.evaluate("window.__retrySection()")
page.wait_for_timeout(100)
page.evaluate("window.__retrySection()")
page.locator("#chapter-10").wait_for(timeout=5000)
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2)
self.assertGreater(page.evaluate("window.__neighborFailures"), 0)
self.assertEqual(page.locator(".reader-error").count(), 0)
self.assertEqual(errors, [])
context.close()
def test_foliate_duplicate_toc_navigation_shares_section_load(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/deduplicate.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Deduplicate", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.evaluate("""() => {
const section = document.querySelector('foliate-view').book.sections.filter(item => item.linear !== 'no')[10];
const original = section.createDocument.bind(section); window.__sectionCreates = 0;
section.createDocument = () => { window.__sectionCreates += 1; return new Promise((resolve, reject) => setTimeout(() => original().then(resolve, reject), 200)); };
document.querySelectorAll('#toc-list .panel-item-main')[9].click();
document.querySelectorAll('#toc-list .panel-item-main')[9].click();
}""")
page.locator("#chapter-10").wait_for(timeout=5000)
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
self.assertEqual(page.evaluate("window.__sectionCreates"), 1)
self.assertEqual(page.locator(".foliate-continuous article[data-section='10']").count(), 1)
self.assertTrue(page.locator("#toc-list .toc-item").nth(9).evaluate("item => item.classList.contains('is-current')"))
context.close()
def test_foliate_chapter_buttons_use_continuous_reader_navigation(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/chapter-buttons.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Buttons", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(9).click()
page.locator("#chapter-10").wait_for(timeout=5000)
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
page.locator("#history").click()
page.locator(".reader-chapter-next").click()
page.locator("#chapter-11").wait_for(timeout=5000)
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(11)').classList.contains('is-current')")
self.assertAlmostEqual(page.locator("#chapter-11").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2)
self.assertTrue(page.locator("#toc-list .toc-item").nth(10).evaluate("item => item.classList.contains('is-current')"))
context.close()
def test_foliate_toc_retains_groups_and_chapter_buttons_skip_them(self):
with zipfile.ZipFile(io.BytesIO(epub_with_navigation())) as archive:
files = {name: archive.read(name) for name in archive.namelist()}
files['OEBPS/nav.xhtml'] = files['OEBPS/nav.xhtml'].replace(
b'- 第一卷
- 第二卷
- ', b'
')
self.page.unroute('**/api/reader-content**')
self.page.route('**/api/reader-content**', lambda route: route.fulfill(
status=200, content_type='application/epub+zip', body=zip_bytes(files)))
source = 'https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/groups.epub'
self.page.goto(f'{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe="")}&ext=epub', wait_until='domcontentloaded')
self.page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 4")
self.page.locator('#history').click()
self.assertEqual(self.page.locator('#toc-list [role=heading]').all_text_contents(), ['第一卷', '第二卷'])
self.assertEqual(self.page.locator('#toc-list [role=link]').all_text_contents(), ['第一章', '第二章'])
self.assertEqual(self.page.locator('#toc-list .toc-group [tabindex]').count(), 0)
self.page.locator('#toc-list [role=link]').first.click()
self.page.wait_for_function("() => document.querySelector('[data-toc-index=\"1\"]').classList.contains('is-current')")
self.page.wait_for_function("() => document.querySelector('#history').getAttribute('aria-expanded') === 'false'")
self.page.locator('#history').click()
self.page.locator('.reader-chapter-next').click()
self.page.wait_for_function("() => document.querySelector('[data-toc-index=\"3\"]').classList.contains('is-current')")
self.assertAlmostEqual(self.page.locator('#two').evaluate(
"e => e.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=2)
def test_chm_rebuilt_spine_ignores_previous_asset_positions(self):
with zipfile.ZipFile(io.BytesIO(epub_with_navigation())) as archive:
files = {name: archive.read(name) for name in archive.namelist()}
files['OEBPS/content.opf'] = files['OEBPS/content.opf'].replace(
b'', b'')
root = 'objects/aa/' + 'a' * 64 + '/'
files['META-INF/reader-chm.json'] = json.dumps({'version': 1,
'previous_path': root + 'calibre-chm-epub-v2/document.epub',
'previous_sections': ['OEBPS/nav.xhtml', 'OEBPS/chapter-1.xhtml', 'OEBPS/chapter-2.xhtml']})
self.page.route('**/static/reader-store.js', lambda route: route.fulfill(
status=200, content_type='text/javascript', body=STORE_SCRIPT.replace(
'get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))',
"get: url => Promise.resolve(url.includes('calibre-chm-epub-v2') ? {foliateSection:2, foliateOffset:100, foliateTocIndex:1} : null)")))
self.page.route('**/api/reader-content**', lambda route: route.fulfill(
status=200, content_type='application/epub+zip', body=zip_bytes(files)))
source = 'https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/' + root + 'manual-chm-navigation-v2/document.epub'
self.page.goto(f'{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe="")}&ext=epub', wait_until='domcontentloaded')
self.page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'")
self.page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 0")
previous = source.replace('manual-chm-navigation-v2', 'calibre-chm-epub-v2')
self.page.evaluate("url => VoiceOfMLReaderStore.putBookmark({id:'old-chm-mark',url,foliateSection:1,foliateOffset:150,label:'Old chapter one',createdAt:1})", previous)
self.page.locator('#history').click()
self.page.locator('#bookmarks-tab').click()
self.assertEqual(self.page.locator('#bookmarks-list .panel-item-main', has_text='Old chapter one').count(), 0)
self.assertIn('manual-chm-navigation-v2', urllib.parse.unquote(self.page.url))
def test_foliate_scroll_updates_toc_on_animation_frame(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll-toc.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=ScrollToc", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(9).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
page.evaluate("""() => { const viewport = document.querySelector('#viewport'), target = document.querySelector('.foliate-continuous article[data-section="11"]').shadowRoot.querySelector('#chapter-11'); viewport.scrollTop += target.getBoundingClientRect().top - viewport.getBoundingClientRect().top + 100; viewport.dispatchEvent(new Event('scroll')); }""")
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(11)').classList.contains('is-current')", timeout=1000)
context.close()
def test_foliate_resize_keeps_current_text_anchor(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll-anchor.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Anchor", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(9).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
page.evaluate("""() => { const viewport = document.querySelector('#viewport'); viewport.style.overflowAnchor = 'none'; viewport.dispatchEvent(new Event('scroll')); }""")
page.wait_for_timeout(50)
before = page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top")
page.evaluate("""() => { const spacer = document.createElement('div'); spacer.style.height = '600px'; const articles = [...document.querySelectorAll('.foliate-continuous article[data-section]:not(.foliate-section-placeholder)')].filter(article => Number(article.dataset.section) < 10); articles[articles.length - 1].shadowRoot.querySelector('.reader-section-body').appendChild(spacer); }""")
page.wait_for_function("top => Math.abs(document.querySelector('.foliate-continuous article[data-section=\"10\"]').shadowRoot.querySelector('#chapter-10').getBoundingClientRect().top - top) < 3", arg=before, timeout=2000)
context.close()
def test_foliate_virtualizes_distant_sections_and_reloads_them(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/virtual.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Virtual", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
for index in [1, 3, 5, 7, 9, 11, 13]:
page.evaluate("i => document.querySelectorAll('#toc-list .panel-item-main')[i].click()", index)
page.wait_for_function("i => document.querySelectorAll('#toc-list .toc-item')[i].classList.contains('is-current')", arg=index)
page.wait_for_timeout(100)
self.assertLessEqual(page.locator(".foliate-continuous article[data-section]:not(.foliate-section-placeholder)").count(), 12)
self.assertGreater(page.locator(".foliate-section-placeholder").count(), 0)
sections = page.locator(".foliate-continuous article[data-section]").evaluate_all("items => items.map(item => Number(item.dataset.section))")
self.assertEqual(sections, sorted(set(sections)))
page.evaluate("() => document.querySelectorAll('#toc-list .panel-item-main')[1].click()")
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')")
page.locator(".foliate-continuous article[data-section]:not(.foliate-section-placeholder) #chapter-2").wait_for(timeout=5000)
page.wait_for_function("() => document.querySelectorAll('.foliate-continuous article[data-section]:not(.foliate-section-placeholder)').length <= 12")
context.close()
def test_large_foliate_toc_keeps_a_bounded_dom_window(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters(600)))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/large-toc.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=LargeToc", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#toc-list')?.classList.contains('toc-list-virtualized')")
page.locator("#history").click()
self.assertLess(page.locator("#toc-list .toc-item").count(), 100)
page.evaluate("""() => { const panel = document.querySelector('#toc-panel'); panel.scrollTop = 20000; panel.dispatchEvent(new Event('scroll')); }""")
page.wait_for_function("() => [...document.querySelectorAll('#toc-list .toc-item')].some(row => row.textContent.includes('章节 501'))")
self.assertLess(page.locator("#toc-list .toc-item").count(), 100)
page.locator('#toc-list [data-toc-index="500"] .panel-item-main').click()
page.wait_for_function("() => document.querySelector('#toc-list [data-toc-index=\"500\"]')?.classList.contains('is-current')")
self.assertEqual(page.locator('#toc-list .is-current').get_attribute('data-toc-index'), '500')
context.close()
def test_foliate_full_search_uses_continuous_reader_navigation(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/search.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Search", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#full-search-toggle").click()
page.locator("#full-search-input").fill("正文 10")
page.locator("#full-search-results .full-search-result").first.wait_for(timeout=10000)
page.locator("#full-search-results .full-search-result").first.click()
page.locator("#chapter-10").wait_for(timeout=5000)
page.locator(".foliate-continuous .full-search-highlight").wait_for(timeout=5000)
page.wait_for_function("() => !document.querySelector('#history-panel').classList.contains('is-open')")
self.assertTrue(page.locator(".foliate-continuous .full-search-highlight").evaluate("element => { const rect = element.getBoundingClientRect(), viewport = document.querySelector('#viewport').getBoundingClientRect(); return rect.bottom > viewport.top && rect.top < viewport.bottom; }"))
context.close()
def test_foliate_bookmark_restores_continuous_section_position(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/bookmark.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Bookmark", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(9).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
page.locator("#bookmark-ribbon").click()
page.locator("#bookmark-add").click()
page.wait_for_function("() => window.__readerBookmarks.length === 1")
bookmark = page.evaluate("window.__readerBookmarks[0]")
self.assertEqual(bookmark.get("foliateSection"), 10)
page.locator("#viewport").evaluate("element => { element.scrollTop = 0; element.dispatchEvent(new Event('scroll')); }")
page.locator("#history").click()
page.locator("#bookmarks-tab").click()
page.locator("#bookmarks-list .panel-item-main").click()
page.wait_for_function("() => !document.querySelector('#history-panel').classList.contains('is-open')")
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3)
context.close()
def test_foliate_progress_slider_uses_continuous_viewport(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/progress.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Progress", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(3).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(4)').classList.contains('is-current')")
page.locator("#history").click()
page.locator(".reader-progress-range").dispatch_event("pointerdown")
# Seek inside section 12, away from its fractional-pixel top boundary.
page.locator(".reader-progress-range").fill("81")
page.locator(".reader-progress-range").dispatch_event("pointerup")
page.wait_for_function("() => { const article = document.querySelector('.foliate-continuous article[data-section=\"12\"]'), viewport = document.querySelector('#viewport'); if (!article) return false; const marker = viewport.getBoundingClientRect().top + 8, rect = article.getBoundingClientRect(); return rect.top <= marker && rect.bottom > marker; }")
self.assertAlmostEqual(float(page.locator(".reader-progress-percent").text_content().rstrip("%")), 81, delta=7)
page.locator('.foliate-section-placeholder').first.evaluate("node => node.style.setProperty('--foliate-placeholder-height', `${node.getBoundingClientRect().height + 600}px`)")
page.locator(".reader-progress-undo").click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(4)').classList.contains('is-current')")
self.assertAlmostEqual(page.locator("#chapter-4").evaluate("node => node.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3)
self.assertTrue(page.locator(".reader-progress-undo").is_hidden())
page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 4")
context.close()
def test_foliate_history_saves_structured_section_position(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/history.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=History", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(9).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)').classList.contains('is-current')")
page.evaluate("window.dispatchEvent(new Event('pagehide'))")
page.wait_for_function("() => window.__savedReaderProgress && window.__savedReaderProgress.foliateSection === 10")
self.assertEqual(page.evaluate("window.__savedReaderProgress.foliateTocIndex"), 9)
context.close()
def test_foliate_history_restores_structured_section_position(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
restored_store = STORE_SCRIPT.replace("get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300)),", "get: () => Promise.resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}),")
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=restored_store))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="application/epub+zip", body=epub_with_many_chapters()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/restore.epub"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=epub&title=Restore", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(10)')?.classList.contains('is-current')", timeout=10000)
self.assertAlmostEqual(page.locator("#chapter-10").evaluate("element => element.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3)
delayed_store = restored_store.replace(
"get: () => Promise.resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}),",
"get: () => new Promise(resolve => { window.__releaseHistory = () => resolve({foliateSection: 10, foliateOffset: 0, foliateTocIndex: 9, zoom: 1}); }),",
)
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=delayed_store))
page.reload(wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelectorAll('#toc-list .toc-item').length === 14")
page.locator("#history").click()
page.locator("#toc-list .panel-item-main").nth(1).click()
page.wait_for_function("() => document.querySelector('#toc-list .toc-item:nth-child(2)').classList.contains('is-current')")
page.evaluate("window.__releaseHistory()")
page.wait_for_function("() => document.documentElement.dataset.readerPhase === 'ready'")
page.wait_for_function("() => window.__savedReaderProgress?.foliateSection === 2")
self.assertAlmostEqual(page.locator("#chapter-2").evaluate("node => node.getBoundingClientRect().top - document.querySelector('#viewport').getBoundingClientRect().top"), 8, delta=3)
context.close()
def test_reader_store_resets_old_history_and_keeps_current_bookmarks(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.goto(f"{self.origin}/static/reader.html", wait_until="domcontentloaded")
page.evaluate("""async () => {
await new Promise((resolve) => { const request = indexedDB.deleteDatabase('voiceofml-reader'); request.onsuccess = request.onerror = request.onblocked = resolve; });
await new Promise((resolve, reject) => {
const request = indexedDB.open('voiceofml-reader', 1);
request.onupgradeneeded = () => { const store = request.result.createObjectStore('entries', {keyPath: 'url'}); store.createIndex('lastReadAt', 'lastReadAt'); };
request.onerror = () => reject(request.error);
request.onsuccess = () => { const db = request.result, tx = db.transaction('entries', 'readwrite'); tx.objectStore('entries').put({url: 'legacy', lastReadAt: 1}); tx.oncomplete = () => { db.close(); resolve(); }; };
});
}""")
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"Reader"))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/store.txt"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Store", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'")
result = page.evaluate("""async (url) => {
const legacy = await VoiceOfMLReaderStore.get('legacy');
await VoiceOfMLReaderStore.putBookmark({id: url + '\\0page:1', url, label: '第 1 页', createdAt: 1});
const historyBeforeClear = await VoiceOfMLReaderStore.list();
const bookmarkEntries = await VoiceOfMLReaderStore.listBookmarks(url);
await VoiceOfMLReaderStore.clearHistory();
return {legacy: !!legacy, schema: VoiceOfMLReaderStore.SCHEMA_VERSION,
validHistory: historyBeforeClear.every(entry => typeof entry.url === 'string' && entry.schemaVersion === 1),
history: (await VoiceOfMLReaderStore.list()).length, bookmarks: (await VoiceOfMLReaderStore.listBookmarks(url)).length,
bookmarkSchema: bookmarkEntries[0].schemaVersion};
}""", source)
self.assertEqual(result, {"legacy": False, "schema": 1, "validHistory": True, "history": 0, "bookmarks": 1, "bookmarkSchema": 1})
context.close()
def test_mobile_pdf_rendering_uses_one_slot_and_seven_canvases(self):
self.page.set_viewport_size({"width": 390, "height": 844})
self.open_pdf()
metrics = self.scroll_document()
self.assertLessEqual(metrics["peak"], 1)
self.assertLessEqual(metrics["rendered"], 7)
self.assertGreater(metrics["pixels"], 0)
def test_high_density_mobile_pdf_uses_backing_scale_without_distortion(self):
context = self.browser.new_context(
viewport={"width": 390, "height": 844}, device_scale_factor=3
)
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(
status=200, content_type="text/javascript", body=STORE_SCRIPT
))
page.route("**/static/vendor/pdf.min.*.mjs", lambda route: route.fulfill(
status=200, content_type="text/javascript", body=PDF_MODULE
))
page.route("**/api/reader-content**", lambda route: route.fulfill(
status=200, content_type="application/pdf", body=b"pdf"
))
query = urllib.parse.quote(SOURCE_URL, safe="")
page.goto(
f"{self.origin}/static/reader.html?url={query}&ext=pdf&title=Performance",
wait_until="domcontentloaded",
)
page.locator(".reader-page canvas.ready").first.wait_for(timeout=10000)
metrics = page.locator(".reader-page").first.evaluate("""shell => {
const canvas = shell.querySelector('canvas');
const box = canvas.getBoundingClientRect();
return {
dpr: devicePixelRatio,
backingWidth: canvas.width,
backingHeight: canvas.height,
cssWidth: box.width,
cssHeight: box.height,
shellAspect: shell.getBoundingClientRect().width / shell.getBoundingClientRect().height,
cssAspect: box.width / box.height,
backingAspect: canvas.width / canvas.height,
canvasCssWidth: parseFloat(getComputedStyle(canvas).width),
canvasCssHeight: parseFloat(getComputedStyle(canvas).height),
};
}""")
self.assertEqual(metrics["dpr"], 3)
self.assertGreaterEqual(metrics["backingWidth"], metrics["cssWidth"] * 1.9)
self.assertGreaterEqual(metrics["backingHeight"], metrics["cssHeight"] * 1.9)
self.assertAlmostEqual(metrics["shellAspect"], metrics["cssAspect"], delta=0.01)
self.assertAlmostEqual(metrics["cssAspect"], metrics["backingAspect"], delta=0.01)
self.assertAlmostEqual(metrics["cssWidth"], metrics["canvasCssWidth"], delta=0.01)
self.assertAlmostEqual(metrics["cssHeight"], metrics["canvasCssHeight"], delta=0.01)
context.close()
def test_mobile_zoom_enlarges_pdf_page_without_resizing_content_shell(self):
self.page.set_viewport_size({"width": 390, "height": 844})
self.open_pdf()
before = self.page.evaluate("""() => ({
content: document.querySelector('.reader-content').getBoundingClientRect().width,
page: document.querySelector('.reader-page').getBoundingClientRect().width,
pixels: document.querySelector('.reader-page canvas').width,
})""")
self.page.locator("#zoom-in").click(click_count=5)
self.page.wait_for_function("before => document.querySelector('.reader-page canvas').width > before * 1.45", arg=before["pixels"])
after = self.page.evaluate("""() => ({
content: document.querySelector('.reader-content').getBoundingClientRect().width,
page: document.querySelector('.reader-page').getBoundingClientRect().width,
pixels: document.querySelector('.reader-page canvas').width,
})""")
self.assertAlmostEqual(after["content"], before["content"], delta=1)
self.assertGreater(after["page"], before["page"] * 1.45)
self.assertGreater(after["pixels"], before["pixels"] * 1.45)
def test_txt_displays_before_stream_finishes(self):
self.page.add_init_script(r"""
const nativeFetch = window.fetch.bind(window);
window.fetch = (input, init) => {
const url = String(input && input.url || input);
if (!url.includes('/api/reader-content?url=')) return nativeFetch(input, init);
const encode = text => new TextEncoder().encode(text);
const bytes = encode('first line\n');
const large = encode('large ASCII chunk\n'.repeat(8192) + 'target\n中文');
return Promise.resolve(new Response(new ReadableStream({
start(controller) {
window.__txtStream = controller;
controller.enqueue(bytes);
window.__txtLargeChunk = () => {
controller.enqueue(large.slice(0, -1));
};
window.__txtFinish = () => {
controller.enqueue(large.slice(-1));
const ending = encode('\nfinal 中文');
controller.enqueue(ending.slice(0, ending.length - 1));
controller.enqueue(ending.slice(-1));
controller.close();
};
}
}), { status: 200, headers: { 'Content-Type': 'text/plain' } }));
};
""")
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.txt"
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Performance", wait_until="domcontentloaded")
self.page.locator(".reader-text").filter(has_text="first line").wait_for(timeout=3000)
self.assertNotEqual(self.page.locator("#status").text_content(), "已加载")
self.assertEqual(self.page.locator(".reader-text").text_content(), "first line\n")
self.page.locator("#history").click()
self.page.locator("#full-search-toggle").click()
self.page.locator("#full-search-input").fill("first line")
self.page.locator(".reader-text mark.full-search-highlight").wait_for()
self.assertEqual(self.page.locator(".reader-text mark.full-search-highlight").count(), 1)
self.page.evaluate("window.__txtLargeChunk()")
expected = "first line\n" + "large ASCII chunk\n" * 8192 + "target\n中"
self.page.wait_for_function("() => document.querySelector('.reader-text').textContent.endsWith('target\\n中')")
self.assertEqual(self.page.locator(".reader-text").text_content(), expected)
self.assertNotEqual(self.page.locator("#status").text_content(), "已加载")
self.page.locator("#full-search-clear").click()
self.assertEqual(self.page.locator(".reader-text mark").count(), 0)
self.assertEqual(self.page.locator(".reader-text").text_content(), expected)
self.page.evaluate("window.__txtFinish()")
self.page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'")
expected += "文\nfinal 中文"
self.assertEqual(self.page.locator(".reader-text").text_content(), expected)
self.page.locator("#full-search-input").fill("中文")
self.page.wait_for_function("() => document.querySelectorAll('#full-search-results .full-search-result').length === 2")
self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 2)
self.assertEqual(self.page.locator(".reader-text mark.full-search-highlight").all_text_contents(), ["中文", "中文"])
self.page.locator("#full-search-clear").click()
self.assertEqual(self.page.locator(".reader-text").text_content(), expected)
self.assertEqual(self.page.locator(".full-search-highlight").count(), 0)
self.assertEqual(self.page.locator("#full-search-results .full-search-result").count(), 0)
def test_text_reader_uses_scroll_mode_without_pagination_controls(self):
text = "\n\n".join(f"第 {index} 段内容。" * 120 for index in range(8))
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=text.encode()))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/scroll.txt"
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=Scroll", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for()
self.assertEqual(self.page.locator("#reading-mode").count(), 0)
self.assertTrue(self.page.locator(".page-controls").is_hidden())
self.assertFalse(self.page.locator(".reader-viewport").evaluate("element => element.classList.contains('is-paginated')"))
self.page.locator("#viewport").evaluate("element => { element.scrollTop = element.scrollHeight; element.dispatchEvent(new Event('scroll')); }")
self.assertGreater(self.page.locator("#viewport").evaluate("element => element.scrollTop"), 0)
def test_reader_tab_is_hidden_for_document_without_toc(self):
self.page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/plain", body=b"No table of contents"))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/no-toc.txt"
self.page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title=NoToc", wait_until="domcontentloaded")
self.page.locator(".reader-text").wait_for()
self.page.locator("#history").click()
self.assertTrue(self.page.locator("#toc-tab").is_hidden())
def test_txt_detects_legacy_encodings_and_multibyte_sample_boundary(self):
cases = [
(list("中文文本".encode("gb18030")), "Encoding", "中文文本"),
(list("中文文本".encode("utf-16")), "Encoding", "中文文本"),
(list("AB中文".encode("utf-16le")), "Encoding", "AB中文"),
(list("Русский текст".encode("cp1251")), "Русский", "Русский текст"),
(list(b"A" * 65535 + "中文".encode("gb18030")), "中文", "A" * 65535 + "中文"),
(list("中文".encode("gb18030") + b"\xff" + "文本".encode("gb18030")), "中文", "中文文本"),
]
for encoded, title, expected in cases:
with self.subTest(title=title, size=len(encoded)):
context = self.browser.new_context(viewport={"width": 1440, "height": 900})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.add_init_script("""
window.__txtBytes = new Uint8Array(%s);
const nativeFetch = window.fetch.bind(window);
window.fetch = (input, init) => {
const url = String(input && input.url || input);
if (!url.includes('/api/reader-content?url=')) return nativeFetch(input, init);
return Promise.resolve(new Response(window.__txtBytes, { status: 200 }));
};
""" % json.dumps(encoded))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/encoding.txt"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=txt&title={urllib.parse.quote(title)}", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'")
self.assertEqual(page.locator(".reader-text").text_content(), expected)
context.close()
def test_markdown_extension_aliases_render_content(self):
for extension in ("md", "markdown"):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=STORE_SCRIPT))
page.route("**/static/vendor/marked.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=MARKED_SCRIPT))
page.route("**/static/vendor/purify.min.*.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=PURIFY_SCRIPT))
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/markdown", body=b"# Markdown readable"))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/readme.{extension}"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Markdown", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === '已加载'")
self.assertIn("Markdown readable", page.locator(".reader-markdown").inner_text())
context.close()
def test_html_aliases_render_safely_with_visible_text(self):
document = b'HTML readable
'
for extension in ("html", "htm"):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844}, color_scheme="dark")
page = context.new_page(); external = []
page.on("request", lambda request: external.append(request.url) if "evil.test" in request.url else None)
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="text/html", body=document))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/page.{extension}"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=HTML", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === 'HTML'")
frame = page.locator("iframe.html-frame").content_frame
self.assertEqual(frame.locator("#visible").text_content(), "HTML readable")
self.assertEqual(frame.locator("script, iframe").count(), 0)
self.assertFalse(page.evaluate("window.__unsafe === true"))
self.assertEqual(external, [])
self.assertNotEqual(frame.locator("#visible").evaluate("e => getComputedStyle(e).color"), "rgb(255, 255, 255)")
context.close()
def test_image_aliases_decode_real_image_bytes(self):
for extension, (content_type, document) in IMAGE_FIXTURES.items():
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
def serve_image(route, _request, mime=content_type, body=document):
route.fulfill(status=200, content_type=mime, body=body)
page.route("**/api/reader-content**", serve_image)
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/image.{extension}"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Image", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === '图片'")
self.assertTrue(page.locator(".reader-image").evaluate("image => image.complete && image.naturalWidth === 1 && image.naturalHeight === 1"))
context.close()
def test_native_media_uses_proxy_controls_and_mobile_layout(self):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
requests = []
page.route("**/api/reader-content**", lambda route: (
requests.append(route.request.url),
route.fulfill(status=200, content_type="audio/wav", body=minimal_wav()),
))
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/sound.wav"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=wav&title=Sound", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('audio')?.readyState >= 1")
audio = page.locator(".reader-audio")
self.assertTrue(audio.evaluate("element => element.controls"))
self.assertEqual(audio.get_attribute("preload"), "metadata")
self.assertIn("/api/reader-content?url=", audio.get_attribute("src"))
self.assertTrue(page.locator(".zoom-controls").is_hidden())
page.locator("#history").click()
self.assertTrue(page.locator("#media-tab").is_visible())
self.assertEqual(page.locator('.reader-panel-tabs button[aria-selected="true"]').get_attribute("data-panel"), "media")
page.locator(".media-panel-bookmark").click()
self.assertIn("时间", page.locator("#bookmark-prompt").text_content())
page.locator("#bookmark-add").click()
page.locator("#bookmarks-tab").click()
page.locator("#bookmarks-list .panel-item-main").wait_for()
self.assertIn("时间", page.locator("#bookmarks-list").text_content())
self.assertTrue(requests)
context.close()
def test_audio_extension_aliases_use_native_reader_controls(self):
for extension in ("mp3", "m4a", "flac", "mpga", "audio"):
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 390, "height": 844})
page = context.new_page()
page.route("**/api/reader-content**", lambda route: route.fulfill(status=200, content_type="audio/wav", body=minimal_wav()))
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/alias.{extension}"
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Audio", wait_until="domcontentloaded")
page.wait_for_function("() => document.querySelector('#status').textContent === '音频'")
audio = page.locator(".reader-audio")
self.assertTrue(audio.evaluate("element => element.controls"))
self.assertEqual(page.locator(".reader-content").get_attribute("data-mode"), "audio")
self.assertTrue(page.locator(".zoom-controls").is_hidden())
self.assertTrue(page.locator("#bookmark-ribbon").is_visible())
context.close()
def test_supported_formats_start_loading_while_history_restores(self):
cases = [
("md", "已加载", ".reader-markdown", ["content", "marked", "purify"]),
("docx", "DOCX", ".docx-body", ["content", "jszip", "docx"]),
("png", "图片", ".reader-image", ["content"]),
]
elapsed_by_format = {}
for extension, ready_status, selector, expected_requests in cases:
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 1440, "height": 900})
page = context.new_page()
requested_at = {}
def timed(name, content_type, body):
def fulfill(route):
requested_at[name] = page.evaluate("performance.now()")
page.evaluate("name => (window.__formatRequests ||= []).push(name)", name)
route.fulfill(status=200, content_type=content_type, body=body)
return fulfill
pending_store = STORE_SCRIPT.replace(
"get: () => new Promise((resolve) => setTimeout(() => resolve(null), 300))",
"get: () => new Promise((resolve) => { window.__releaseHistory = () => { window.__historyResolved = true; resolve(null); }; })",
)
page.route("**/static/reader-store.js", lambda route: route.fulfill(status=200, content_type="text/javascript", body=pending_store))
page.route("**/static/vendor/marked.min.*.js", timed("marked", "text/javascript", MARKED_SCRIPT))
page.route("**/static/vendor/purify.min.*.js", timed("purify", "text/javascript", PURIFY_SCRIPT))
page.route("**/static/vendor/jszip.min.*.js", timed("jszip", "text/javascript", JSZIP_SCRIPT))
page.route("**/static/vendor/epub.min.06eae1574510.js", timed("epub", "text/javascript", EPUB_SCRIPT))
page.route("**/static/vendor/docx-preview.min.*.js", timed("docx", "text/javascript", DOCX_SCRIPT))
content_type = "image/png" if extension == "png" else "application/octet-stream"
body = PNG_BYTES if extension == "png" else minimal_docx() if extension == "docx" else minimal_epub() if extension == "epub" else b"Reader benchmark content"
page.route("**/api/reader-content**", timed("content", content_type, body))
if extension == "docx":
digest = "a" * 64
source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/docx-native-v1/document.docx"
else:
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/performance.{extension}"
if extension == "png":
page.route(source, timed("content", content_type, body))
started = time.perf_counter()
page.goto(
f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Performance",
wait_until="domcontentloaded",
)
page.wait_for_function("expected => expected.every(name => window.__formatRequests?.includes(name))", arg=expected_requests)
self.assertFalse(page.evaluate("window.__historyResolved === true"))
page.evaluate("window.__releaseHistory()")
page.wait_for_function(
"expected => document.querySelector('#status').textContent === expected",
arg=ready_status,
)
page.locator(selector).wait_for(state="attached")
elapsed_by_format[extension] = (time.perf_counter() - started) * 1000
store_started = page.evaluate("window.__storeStartedAt")
self.assertEqual(set(requested_at), set(expected_requests))
self.assertTrue(all(requested_at[name] >= store_started for name in expected_requests))
context.close()
print("\n Reader format load times: " + ", ".join(
f"{name}={elapsed_by_format[name]:.1f}ms" for name in sorted(elapsed_by_format)
))
def test_cold_cache_first_read_with_real_format_engines(self):
cases = [
("txt", b"Cold TXT readable", "text/plain", "已加载", ".reader-text", []),
("md", b"# Cold Markdown", "text/markdown", "已加载", ".reader-markdown", [VENDOR_FILES["marked"], VENDOR_FILES["purify"]]),
("pdf", minimal_pdf(), "application/pdf", "1 页", ".reader-page canvas.ready", [VENDOR_FILES["pdf"], VENDOR_FILES["pdf_worker"]]),
("docx", minimal_docx(), "application/vnd.openxmlformats-officedocument.wordprocessingml.document", "1 页", ".docx-body", [VENDOR_FILES["jszip"], VENDOR_FILES["docx"]]),
("png", PNG_BYTES, "image/png", "图片", ".reader-image", []),
]
results = []
for extension, document, content_type, ready_status, selector, engines in cases:
with self.subTest(extension=extension):
context = self.browser.new_context(viewport={"width": 1440, "height": 900}, service_workers="block")
try:
page = context.new_page(); session = context.new_cdp_session(page)
session.send("Network.enable"); session.send("Network.setCacheDisabled", {"cacheDisabled": True})
responses = []; page.on("response", lambda response: responses.append(response))
route_local_vendor_fallback(page)
def serve_document(route):
route.fulfill(status=200, content_type=content_type, body=document)
page.route("**/api/reader-content**", serve_document)
if extension == "docx":
digest = "a" * 64
source = f"https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/objects/aa/{digest}/docx-native-v1/document.docx"
else:
source = f"https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/cold.{extension}"
if extension == "png": page.route(source, serve_document)
started = time.perf_counter()
page.goto(f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext={extension}&title=Cold", wait_until="domcontentloaded")
page.wait_for_function("expected => document.querySelector('#status').textContent === expected", arg=ready_status, timeout=30000)
page.locator(selector).wait_for(state="attached", timeout=30000)
if extension == "docx":
self.assertTrue(page.locator(".page-controls").is_visible())
self.assertEqual(page.locator(".reader-docx-page").count(), 1)
elapsed = (time.perf_counter() - started) * 1000
urls = [urllib.parse.urlsplit(response.url).path.rsplit("/", 1)[-1] for response in responses]
for engine in engines: self.assertIn(engine, urls)
for response in responses: self.assertFalse(response.from_service_worker)
entries = page.evaluate("""() => performance.getEntriesByType('resource').map((entry) => ({ name: entry.name, transferSize: entry.transferSize }))""")
for engine in engines:
if ".worker." in engine: continue
entry = next((item for item in entries if item["name"].endswith(engine)), None)
self.assertIsNotNone(entry, engine)
self.assertGreater(entry["transferSize"], 0, f"{engine} was not transferred on a cold load")
byte_count = len(document) + sum(
(ROOT / "static/vendor" / engine).stat().st_size
if (ROOT / "static/vendor" / engine).exists() else 0
for engine in engines
)
results.append((extension, elapsed, byte_count, len(responses)))
finally:
context.close()
print("\n Reader cold-cache first read (real engines):")
for extension, elapsed, byte_count, request_count in results:
print(f" {extension:<5s} {elapsed:>7.1f}ms {byte_count / 1024:>8.1f}KiB {request_count:>2d} responses")
def test_image_proxy_failure_falls_back_to_direct_source(self):
source = "https://huggingface.co/datasets/VoiceOfML/Test/resolve/main/fallback.png"
requests = []
self.page.route(source, lambda route: (requests.append("source"), route.fulfill(status=200, content_type="image/png", body=PNG_BYTES)))
self.page.route(
"**/api/reader-content**",
lambda route: (requests.append("proxy"), route.fulfill(status=404, body=b"")),
)
self.page.goto(
f"{self.origin}/static/reader.html?url={urllib.parse.quote(source, safe='')}&ext=png&title=Fallback",
wait_until="domcontentloaded",
)
self.page.wait_for_function("() => document.querySelector('#status').textContent === '图片'")
self.assertEqual(requests, ["proxy", "source"])
def scroll_document(self):
pages = self.page.locator(".reader-page")
for index in range(30):
pages.nth(index).scroll_into_view_if_needed()
self.page.wait_for_timeout(20)
self.page.wait_for_timeout(100)
return self.page.locator("body").evaluate("""() => ({
peak: window.__pdfPeak,
rendered: document.querySelectorAll('.reader-page[data-render-state="rendered"]').length,
pixels: [...document.querySelectorAll('.reader-page canvas')].reduce((sum, canvas) => sum + canvas.width * canvas.height, 0),
})""")
def load_tests(_loader, _tests, _pattern):
return unittest.TestSuite()
if __name__ == "__main__":
unittest.main()