Spaces:
Running
Running
Download src/lib/reconstructor.js from sizzlebop/pulpie-webgpu: direct link, hf CLI and curl.
- Browser
- Download file 5.12 kB
-
https://huggingface.co/spaces/sizzlebop/pulpie-webgpu/resolve/main/src/lib/reconstructor.js
- Command line
-
hf download hf://spaces/sizzlebop/pulpie-webgpu/src/lib/reconstructor.js
-
curl -L -o reconstructor.js https://huggingface.co/spaces/sizzlebop/pulpie-webgpu/resolve/main/src/lib/reconstructor.js
5.12 kB
| /** | |
| * Pulpie WebUI - Main Content Reconstructor | |
| * | |
| * Reconstructs main-content HTML from classifier predictions and converts | |
| * to clean Markdown using Turndown with image link retention and tracking-pixel removal. | |
| */ | |
| import TurndownService from 'turndown'; | |
| import { ITEM_ID_ATTR } from './html-simplifier.js'; | |
| // Matches tracking pixels, 1x1 gifs, and empty spacer images | |
| const SPACER_IMG_REGEX = | |
| /<img\b[^>]*?(?:(?:width|height)\s*=\s*["']?\s*1\b|src\s*=\s*["'][^"']*(?:trans(?:parent)?|spacer|blank|pixel|1x1|clear)[^"']*["']|src\s*=\s*["']data:image[^"']*["'])[^>]*>/gi; | |
| /** | |
| * Initialize and configure TurndownService for clean GitHub-style Markdown. | |
| * @returns {TurndownService} | |
| */ | |
| function createTurndownService() { | |
| const turndown = new TurndownService({ | |
| headingStyle: 'atx', | |
| hr: '---', | |
| bulletListMarker: '-', | |
| codeBlockStyle: 'fenced', | |
| emDelimiter: '*', | |
| }); | |
| // Keep pre/code formatting | |
| turndown.addRule('fencedCodeBlock', { | |
| filter: ['pre'], | |
| replacement: function (content, node) { | |
| const code = node.querySelector('code'); | |
| const language = code?.className?.match(/(?:lang|language)-(\S+)/)?.[1] || ''; | |
| const text = (code || node).textContent.replace(/\n+$/, ''); | |
| return `\n\n\`\`\`${language}\n${text}\n\`\`\`\n\n`; | |
| }, | |
| }); | |
| // Preserve tables | |
| turndown.addRule('tables', { | |
| filter: ['table'], | |
| replacement: function (content, node) { | |
| // Basic markdown table reconstruction | |
| const rows = Array.from(node.querySelectorAll('tr')); | |
| if (rows.length === 0) return ''; | |
| const tableLines = []; | |
| let isHeader = true; | |
| for (const row of rows) { | |
| const cells = Array.from(row.querySelectorAll('th, td')).map((c) => | |
| c.textContent.trim().replace(/\|/g, '\\|') | |
| ); | |
| if (cells.length === 0) continue; | |
| tableLines.push(`| ${cells.join(' | ')} |`); | |
| if (isHeader) { | |
| tableLines.push(`| ${cells.map(() => '---').join(' | ')} |`); | |
| isHeader = false; | |
| } | |
| } | |
| return `\n\n${tableLines.join('\n')}\n\n`; | |
| }, | |
| }); | |
| // Clean images | |
| turndown.addRule('imagesWithAlts', { | |
| filter: 'img', | |
| replacement: function (content, node) { | |
| const alt = node.getAttribute('alt') || ''; | |
| const src = node.getAttribute('src') || ''; | |
| if (!src) return ''; | |
| return ``; | |
| }, | |
| }); | |
| return turndown; | |
| } | |
| const _turndownInstance = createTurndownService(); | |
| /** | |
| * Strip invisible tracking pixels and spacer images. | |
| * @param {string} html | |
| * @returns {string} | |
| */ | |
| export function stripSpacerImages(html) { | |
| return html.replace(SPACER_IMG_REGEX, ''); | |
| } | |
| /** | |
| * Reconstruct main HTML and convert to Markdown based on model predictions. | |
| * | |
| * @param {Document} doc - The DOM containing _item_id markers | |
| * @param {Record<string, 'main' | 'other'>} labels - Map of item_id -> label | |
| * @returns {{ html: string, markdown: string }} | |
| */ | |
| export function reconstructMainContent(doc, labels) { | |
| // Clone the document body to prevent altering the original state | |
| const clonedBody = doc.body.cloneNode(true); | |
| // Map each _item_id to its element | |
| const idToElement = new Map(); | |
| const allTagged = clonedBody.querySelectorAll(`[${ITEM_ID_ATTR}]`); | |
| for (const el of allTagged) { | |
| const id = el.getAttribute(ITEM_ID_ATTR); | |
| if (id && !idToElement.has(id)) { | |
| idToElement.set(id, el); | |
| } | |
| } | |
| // Set of all elements to preserve (main content + ancestors + descendants) | |
| const elementsToRemain = new Set(); | |
| for (const [itemId, label] of Object.entries(labels)) { | |
| if (label !== 'main') continue; | |
| const el = idToElement.get(itemId); | |
| if (!el) continue; | |
| // Add element and all its descendants | |
| elementsToRemain.add(el); | |
| const descendants = el.querySelectorAll('*'); | |
| for (const d of descendants) { | |
| elementsToRemain.add(d); | |
| } | |
| // Add ancestors up to body | |
| let ancestor = el.parentElement; | |
| while (ancestor && ancestor !== clonedBody.parentElement) { | |
| elementsToRemain.add(ancestor); | |
| ancestor = ancestor.parentElement; | |
| } | |
| } | |
| // If no elements were kept (e.g. all classified as boilerplate), retain fallback | |
| if (elementsToRemain.size === 0) { | |
| return { | |
| html: '<p><em>No main content identified.</em></p>', | |
| markdown: '*No main content identified.*', | |
| }; | |
| } | |
| // Prune nodes that are not in elementsToRemain | |
| function pruneNode(node) { | |
| const children = Array.from(node.children); | |
| for (const child of children) { | |
| if (!elementsToRemain.has(child)) { | |
| child.remove(); | |
| } else { | |
| pruneNode(child); | |
| } | |
| } | |
| } | |
| pruneNode(clonedBody); | |
| // Clean _item_id attributes from final output | |
| const remainingTagged = clonedBody.querySelectorAll(`[${ITEM_ID_ATTR}]`); | |
| for (const el of remainingTagged) { | |
| el.removeAttribute(ITEM_ID_ATTR); | |
| } | |
| const rawExtractedHtml = clonedBody.innerHTML.trim(); | |
| const cleanedHtml = stripSpacerImages(rawExtractedHtml); | |
| const markdown = _turndownInstance.turndown(cleanedHtml).trim(); | |
| return { | |
| html: cleanedHtml, | |
| markdown, | |
| }; | |
| } | |