Search / static /reader-contract.js
vomebook's picture
feat: consume versioned v3 PDF preview and text layers
d471240 verified
Raw History Blame Contribute Delete
20.1 kB
(function (root) {
"use strict";
function preloadPdfOpeningResources() {
// Classic-script and module consumers share the same in-flight preload.
if (root.__VOICE_PDF_PRELOAD__) return;
try {
const params = new URLSearchParams(location.search);
const id = params.get("id") || "";
let source = params.get("url") || "";
if (!source && id) {
for (const key of [`reader-source:${id}`, `reader-resolve:${id}`]) {
const value = JSON.parse(sessionStorage.getItem(key) || "null");
if (value?.url) {
source = value.url;
break;
}
}
}
if (!source) return;
const manifestUrl = new URL(source, location.origin);
const sourceInfo = pdfPageSource(manifestUrl.href, location.origin, location.origin);
if (!sourceInfo) return;
const security = root.VoiceOfMLReaderSecurity;
if (!security) return;
const pageUrl = new URL(sourceInfo.pageUrl(1));
const controller = new AbortController(),
directSource = ["huggingface.co", "hf-mirror.com"].includes(manifestUrl.hostname),
timer = setTimeout(() => controller.abort(), directSource ? 1500 : 120000);
let disposed = false;
let preload;
const dispose = () => {
if (disposed) return;
disposed = true;
clearTimeout(timer);
controller.abort();
if (preload?.image) {
preload.image.src = "";
preload.image = null;
}
root.removeEventListener?.("pagehide", onPageHide);
};
const onPageHide = (event) => {
if (!event.persisted) dispose();
};
const manifestRequest = fetch(manifestUrl.href, {
priority: "low",
signal: controller.signal
})
.then(async (response) => {
const bytes = await security.readBytes(response, security.LIMITS.manifestBytes);
return new Response(bytes, {
status: response.status,
statusText: response.statusText,
headers: response.headers
});
})
.finally(() => clearTimeout(timer));
preload = {
manifestUrl: manifestUrl.href,
manifest: manifestRequest,
pageUrl: pageUrl.href,
image: null,
dispose
};
preload.manifest.catch(() => {});
root.addEventListener?.("pagehide", onPageHide);
root.__VOICE_PDF_PRELOAD__ = preload;
} catch (_) {}
}
const ReaderMode = Object.freeze({
UNSUPPORTED: 0,
ORIGINAL: 1,
CONVERTED: 2,
PENDING: 3,
FAILED: 4
});
const modes = Object.freeze({
pdf: "pdf",
"pdf-pages": "pdf-pages",
"image-pages": "image-pages",
epub: "foliate",
mobi: "foliate",
azw: "foliate",
azw3: "foliate",
fb2: "foliate",
fbz: "foliate",
"epub-chapters": "epub-chapters",
docx: "docx",
html: "html",
htm: "html",
txt: "text",
md: "markdown",
markdown: "markdown",
vcf: "text",
ini: "text",
jpg: "image",
jpeg: "image",
png: "image",
gif: "image",
bmp: "image",
tif: "image",
tiff: "image",
webp: "image",
mp3: "audio",
wav: "audio",
m4a: "audio",
flac: "audio",
mpga: "audio",
wma: "audio",
ape: "audio",
amr: "audio",
audio: "audio",
mp4: "video",
mov: "video",
video: "video",
swf: "swf"
});
const articleExtensions = Object.freeze(Object.keys(modes));
const features = Object.freeze({
pdf: { toc: true, search: true, zoom: true, bookmarks: true, pagination: true, media: false },
"pdf-pages": {
toc: true,
search: true,
zoom: true,
bookmarks: true,
pagination: true,
media: false
},
"image-pages": {
toc: false,
search: false,
zoom: true,
bookmarks: true,
pagination: true,
media: false
},
foliate: {
toc: true,
search: true,
zoom: true,
bookmarks: true,
pagination: false,
media: false
},
"epub-chapters": {
toc: true,
search: true,
zoom: true,
bookmarks: true,
pagination: false,
media: false
},
docx: { toc: true, search: true, zoom: true, bookmarks: true, pagination: true, media: false },
html: { toc: true, search: true, zoom: true, bookmarks: true, pagination: false, media: false },
text: {
toc: false,
search: true,
zoom: true,
bookmarks: true,
pagination: false,
media: false
},
markdown: {
toc: true,
search: true,
zoom: true,
bookmarks: true,
pagination: false,
media: false
},
image: {
toc: false,
search: false,
zoom: true,
bookmarks: true,
pagination: false,
media: false
},
audio: {
toc: false,
search: false,
zoom: false,
bookmarks: true,
pagination: false,
media: true
},
unsupported: {
toc: false,
search: false,
zoom: false,
bookmarks: false,
pagination: false,
media: false
},
video: {
toc: false,
search: false,
zoom: false,
bookmarks: true,
pagination: false,
media: true
},
swf: {
toc: false,
search: false,
zoom: false,
bookmarks: true,
pagination: false,
media: true
}
});
function capability(extension) {
const normalized = String(extension || "").toLowerCase();
const mode = modes[normalized];
return Object.freeze({
extension: normalized,
mode: mode || null,
readerMode: mode ? ReaderMode.ORIGINAL : ReaderMode.UNSUPPORTED,
article: !!mode,
features: Object.freeze({
...(features[mode] || features.unsupported)
})
});
}
function clampNumber(value, minimum, maximum, fallback) {
const numeric = Math.round(Number(value));
return Number.isFinite(numeric) ? Math.min(maximum, Math.max(minimum, numeric)) : fallback;
}
const assetRoot = String.raw`objects/[0-9a-f]{2}/[0-9a-f]{64}/`;
const v3BucketPathPattern = new RegExp(`^${assetRoot}[0-9a-f]{16}/(?:reading-manifest\\.json|page-map\\.json\\.gz|text-layer-manifest\\.json|text-review-manifest\\.json|text/(?:page-[0-9]{6}\\.json\\.gz|book-text\\.json\\.gz|partition-[0-9]{6}-[0-9]{6}-manifest\\.json)|preview/(?:page-[0-9]{6}\\.(?:png|webp|jpeg)|partition-[0-9]{6}-[0-9]{6}-manifest\\.json))$`);
const assetVersion = String.raw`(?:[0-9a-f]{16}/)?`;
const assetPages = String.raw`(?:page-manifest\.json|pages/page-[0-9]{6}\.(?:webp|jxl))`;
const assetOcr = String.raw`(?:ocr-manifest\.json|ocr/(?:page-[0-9]{6}\.json\.gz|book-text\.json\.gz))`;
const assetDocument = String.raw`(?:linearized\.pdf|document\.(?:pdf|epub|mobi|azw|azw3|fb2|docx|html|txt|md|webp|swf)|book\.epub|audio\.(?:mp3|wav|m4a|flac|mpga)|video\.(?:mp4|mov))`;
const assetChapters = String.raw`(?:chapter-manifest\.json|epub-chapters/(?:chapter-manifest\.json|chapters/chapter-[0-9]{4}\.xhtml|resources/[A-Za-z0-9._~%+\-/]+|epub-search-index\.json\.gz))`;
const v2ChapterPathPattern = new RegExp(`^chapters/ebook/(?:epub|mobi|azw3|fb2|chm)/[0-9a-f]{64}/[0-9a-f]{16}/(?:chapter-manifest\\.json|epub-search-index\\.json\\.gz|chapters/chapter-[0-9]{4}\\.xhtml|resources/(?!\\.{1,2}(?:/|$))[^/]+(?:/(?!\\.{1,2}(?:/|$))[^/]+)*)$`);
const ebookBucketPathPattern = new RegExp(`^ebook-chapters/${assetRoot}[a-z0-9-]+/[a-z0-9-]+-epub-chapters-v[0-9]+-bucket/epub-chapters/(?:chapter-manifest\\.json|chapters/chapter-[0-9]{4}\\.xhtml|resources/(?!\\.{1,2}(?:/|$))[^/]+(?:/(?!\\.{1,2}(?:/|$))[^/]+)*|epub-search-index\\.json\\.gz)$`);
const assetPrimaryPattern = new RegExp(
`^${assetRoot}${assetVersion}(?:${assetPages}|(?:[a-z0-9-]+/)?${assetDocument})$`
);
const assetSourcePattern = new RegExp(
`^(?:pdf_manifest\\.json|${assetRoot}${assetVersion}(?:${assetPages}|${assetOcr}|(?:[a-z0-9-]+/)?(?:${assetDocument}|${assetChapters})))$`
);
const bucketPathPattern = new RegExp(`^${assetRoot}${assetVersion}(?:${assetPages}|${assetOcr})$`);
const v2ObjectPathPattern = new RegExp(`^${assetRoot}${assetVersion}(?:${assetPages}|${assetOcr})$`);
const staticBucketPathPattern = new RegExp(`^(?:${assetRoot}(?:[0-9a-f]{16}/)?(?:[a-z0-9-]+/)?(?:document\\.(?:docx|html|txt|md|webp|jpg|jpeg|png|gif|bmp|swf)|audio\\.(?:mp3|wav|m4a|flac|mpga)|video\\.(?:mp4|mov))|documents/(?:text|web|spreadsheet|office|pdf)/[a-z0-9_-]+/[0-9a-f]{64}/document\\.(?:txt|md|html|docx|pdf|vcf|ini)|media/(?:audio|video|swf)/[a-z0-9_-]+/[0-9a-f]{64}/(?:audio\\.[a-z0-9]+|video\\.[a-z0-9]+|document\\.swf)|pages/image/(?:jpg|jpeg|png|bmp|tif|tiff|webp)/[0-9a-f]{64}/(?:page-manifest\\.json|pages/page-[0-9]{6}\\.webp)|derived/[A-Za-z0-9._-]+/[a-z0-9]{32}/document\\.pdf|chapters/ebook/(?:epub|mobi|azw3|fb2|chm)/[0-9a-f]{64}/[0-9a-f]{16}/(?:chapter-manifest\\.json|epub-search-index\\.json\\.gz))$`);
const versionedBucketPathPattern = new RegExp(`^${assetRoot}[0-9a-f]{16}/(?:${assetPages}|${assetOcr})$`);
const assetBase = "https://huggingface.co/datasets/vomebook/Reader-Assets/resolve/main/";
const assetModes = Object.freeze({
p: "pdf",
e: "epub",
d: "docx",
h: "html",
t: "txt",
k: "md",
i: "webp",
a: "audio",
v: "video",
f: "swf"
});
function isBucketPath(path, versioned = false, bucket = "vomebook/pdf-pages-v2") {
if (bucket === "vomebook/reader-assets-v2") return staticBucketPathPattern.test(path) || v2ChapterPathPattern.test(path) || v2ObjectPathPattern.test(path);
if (bucket !== "vomebook/pdf-pages-v2") return false;
return /^objects\/[0-9a-f]{2}\/[0-9a-f]{64}\/[0-9a-f]{16}\/document\.pdf$/.test(path) ||
(versioned ? versionedBucketPathPattern : bucketPathPattern).test(path) ||
v3BucketPathPattern.test(path) ||
staticBucketPathPattern.test(path) ||
false;
}
function isAssetSourcePath(path) {
const prefix = /^\/datasets\/vomebook\/Reader-Assets\/resolve\/[^/]+\//.exec(path);
return !!prefix && assetSourcePattern.test(path.slice(prefix[0].length));
}
// One parser for opening preload, manifest validation and page navigation.
// Dataset assets may be unversioned; the Bucket API requires a version.
function pdfPageSource(raw, base, bucketOrigin, filename = "page-manifest.json", bucketName = "vomebook/pdf-pages-v2") {
try {
const url = new URL(raw, base);
if (url.username || url.password || url.hash) return null;
const bucket = [bucketOrigin, "https://voiceofml-search.hf.space"].includes(url.origin) &&
url.pathname === "/api/reader-bucket-resource";
const bucketPrefix = `/buckets/${bucketName}/resolve/`;
const directBucket = url.origin === "https://huggingface.co" &&
url.pathname.startsWith(bucketPrefix) && !url.search;
let path;
if (bucket) {
if (url.searchParams.getAll("path").length !== 1 ||
[...url.searchParams.keys()].some((key) => key !== "path")) return null;
path = url.searchParams.get("path") || "";
if (!isBucketPath(path, true, bucketName)) return null;
} else if (directBucket) {
path = decodeURIComponent(url.pathname.slice(bucketPrefix.length));
if (!isBucketPath(path, true, bucketName)) return null;
} else {
if (
url.protocol !== "https:" ||
!["huggingface.co", "hf-mirror.com"].includes(url.hostname) ||
url.search
)
return null;
const prefix = /^\/datasets\/vomebook\/Reader-Assets\/resolve\/[^/]+\//.exec(url.pathname);
if (!prefix) return null;
path = decodeURIComponent(url.pathname.slice(prefix[0].length));
if (!isBucketPath(path, false, bucketName)) return null;
}
if (!["page-manifest.json", "ocr-manifest.json", "reading-manifest.json"].includes(filename) ||
!path.endsWith(`/${filename}`)) return null;
const rootPath = path.slice(0, -filename.length - 1);
const pathPrefix = bucket || directBucket ? "" : url.pathname.slice(0, -path.length);
return {
root: rootPath,
assetUrl(relativePath) {
if (!isBucketPath(relativePath, bucket || directBucket, bucketName) ||
!relativePath.startsWith(rootPath.split("/").slice(0, 3).join("/") + "/"))
throw new Error("PDF_ASSET_INVALID");
const target = new URL(url.href);
if (bucket) target.searchParams.set("path", relativePath);
else if (directBucket) target.pathname = `${bucketPrefix}${relativePath}`;
else target.pathname = `${pathPrefix}${relativePath}`;
return target.href;
},
pageUrl(page) {
if (!Number.isInteger(page) || page < 1 || page > 999999)
throw new Error("PDF_PAGE_INVALID");
const filename = `pages/page-${String(page).padStart(6, "0")}.webp`;
return this.assetUrl(`${rootPath}/${filename}`);
}
};
} catch (_) {
return null;
}
}
function imagePageSource(raw, base, bucketOrigin) {
return pdfPageSource(raw, base, bucketOrigin, "page-manifest.json", "vomebook/reader-assets-v2");
}
function txtRelativePath(path) {
const value = String(path || "");
const dot = value.lastIndexOf(".");
return (dot > value.lastIndexOf("/") ? value.slice(0, dot) : value) + ".txt";
}
function assetFields(asset, bucketBase) {
const path = String(asset?.p || "");
const ocrPath = String(asset?.o || "");
const ocrBucket = asset?.b === "vomebook/pdf-pages-v2" || asset?.ob === "vomebook/pdf-pages-v2";
const ocrUrl = ocrPath.endsWith("/ocr-manifest.json") && ocrBucket && isBucketPath(ocrPath, true)
? `${bucketBase}?path=${encodeURIComponent(ocrPath)}`
: "";
if (asset?.s === 3 && ocrUrl)
return { ReaderOcrManifest: ocrUrl, ReaderOcrMode: String(asset.om || "") };
const bucketName = String(asset?.b || "vomebook/reader-assets-v2");
const bucket = isBucketPath(path, true, bucketName);
const chapterBucket = asset?.cb === "vomebook/reader-assets-v2" &&
(ebookBucketPathPattern.test(String(asset?.c || "")) || v2ChapterPathPattern.test(String(asset?.c || "")));
if (
!asset ||
asset.s !== 2 ||
!Object.prototype.hasOwnProperty.call(assetModes, asset.m) ||
!(asset.b ? bucket : bucket || assetPrimaryPattern.test(path)) ||
(asset?.b && bucketName !== "vomebook/pdf-pages-v2" && bucketName !== "vomebook/reader-assets-v2") ||
(asset?.cb && !chapterBucket)
)
return null;
const nativeExtension =
/(?:^|\/)document\.(epub|mobi|azw|azw3|fb2)$/i.exec(path)?.[1]?.toLowerCase() || "epub";
const extension =
asset.m === "e"
? nativeExtension
: asset.m === "p" && /\/(?:page|reading)-manifest\.json$/.test(path)
? "pdf-pages"
: asset.m === "i" && path.endsWith("page-manifest.json")
? "image-pages"
: assetModes[asset.m];
return {
ReaderLink: chapterBucket
? `${bucketBase}?path=${encodeURIComponent(asset.c)}`
: bucket
? `${bucketBase}?path=${encodeURIComponent(path)}`
: assetBase + path,
ReaderExtension: asset.c ? "epub-chapters" : extension,
ReaderChapterManifest: asset.c
? chapterBucket
? `${bucketBase}?path=${encodeURIComponent(asset.c)}`
: assetBase + asset.c
: "",
ReaderFallback: asset.f ? assetBase + asset.f : "",
ReaderOcrManifest: ocrUrl || (ocrPath.endsWith("/ocr-manifest.json") ? assetBase + ocrPath : ""),
ReaderOcrMode: String(asset.om || ""),
ReaderPdfDocument: asset.pd && ["vomebook/reader-assets-v2", "vomebook/pdf-pages-v2"].includes(asset.pdb) &&
isBucketPath(asset.pd, true, asset.pdb) && asset.pd.endsWith("/document.pdf")
? `${bucketBase}?path=${encodeURIComponent(asset.pd)}` : ""
};
}
function readerUrl(record, basePath) {
const source = record && (record.ReaderLink || record.readerLink || record.Link || record.link);
const readerExtension =
record &&
(record.ReaderExtension || record.readerExtension || record.Extension || record.extension);
if (!source || capability(readerExtension).readerMode === ReaderMode.UNSUPPORTED) return "";
const originalSource =
record.DownloadLink || record.downloadLink || record.Link || record.link || "";
const shortId = /^https:\/\/huggingface\.co\/datasets\/VoiceOfML\//.test(originalSource)
? shortSourceId(originalSource)
: "";
const title =
(record.File || record.name || "") +
(record.Extension || record.extension ? "." + (record.Extension || record.extension) : "");
const params = new URLSearchParams(
shortId
? { id: shortId, title, ext: readerExtension || "" }
: { url: source, title, ext: readerExtension || "" }
);
if (!shortId && (record.DownloadLink || record.downloadLink))
params.set("download", record.DownloadLink || record.downloadLink);
if (!shortId && (record.OcrUrl || record.ocrUrl))
params.set("ocr", record.OcrUrl || record.ocrUrl);
if (!shortId && (record.ReaderOcrManifest || record.readerOcrManifest))
params.set("ocr_manifest", record.ReaderOcrManifest || record.readerOcrManifest);
if (!shortId && (record.ReaderFallback || record.readerFallback))
params.set("fallback", record.ReaderFallback || record.readerFallback);
if (!shortId && record.ReaderPdfDocument) params.set("pdf_document", record.ReaderPdfDocument);
if (
!shortId &&
(record.ReaderChapterManifest ||
record.readerChapterManifest ||
record.ChapterManifest ||
record.chapterManifest)
)
params.set(
"chapter_manifest",
record.ReaderChapterManifest ||
record.readerChapterManifest ||
record.ChapterManifest ||
record.chapterManifest
);
if (!shortId && (record.ReturnUrl || record.returnUrl))
params.set("return", record.ReturnUrl || record.returnUrl);
const repo = String(record.Repo || record.repo || "")
.split("/")
.pop();
const folder = Array.isArray(record.Folder || record.folder)
? (record.Folder || record.folder).join("/")
: "";
if (!shortId && repo) params.set("path", repo + (folder ? "/" + folder : ""));
if (!shortId && (record.FolderUrl || record.folderUrl))
params.set("folder_url", record.FolderUrl || record.folderUrl);
return (basePath || "/static/reader.html") + "?" + params.toString();
}
function canonicalSourceUrl(value) {
// Normalize path bytes, not decoded URLs: %2F must remain distinct from /,
// and %2528 must never become %28. Query/fragment delimiters stay intact.
return value.replace(/^(https:\/\/huggingface\.co\/datasets\/VoiceOfML\/)([^?#]*)/, (_, prefix, path) =>
prefix + path.replace(/%[0-9a-fA-F]{2}|[^A-Za-z0-9._~/-]/gu, (token) => {
if (/^%[0-9a-f]{2}$/i.test(token)) {
const char = String.fromCharCode(parseInt(token.slice(1), 16));
return /^[A-Za-z0-9._~-]$/.test(char) ? char : token.toUpperCase();
}
return encodeURIComponent(token).replace(/[!'()*]/g, (char) =>
"%" + char.charCodeAt(0).toString(16).toUpperCase());
})
);
}
function shortSourceId(value) {
let hash = 1469598103934665603n;
for (const byte of new TextEncoder().encode(canonicalSourceUrl(value))) {
hash ^= BigInt(byte);
hash = BigInt.asUintN(64, hash * 1099511628211n);
}
return hash.toString(36).padStart(13, "0");
}
root.VoiceOfMLReader = Object.freeze({
ReaderMode,
articleExtensions,
capability,
features,
clampNumber,
assetFields,
isAssetSourcePath,
isBucketPath,
pdfPageSource,
imagePageSource,
txtRelativePath,
canonicalSourceUrl,
shortSourceId,
readerUrl
});
preloadPdfOpeningResources();
})(typeof self !== "undefined" ? self : window);