book-ocr-studio / scripts /render_public_docs.py
moebiusT7's picture
Source beta 2026-09-23, revision 5
0209b2f verified
Raw History Blame Contribute Delete
2.79 kB
"""Render public Markdown originals to static HTML; --check detects stale pages."""
import argparse
import html
import re
from pathlib import Path
import markdown2
ROOT = Path(__file__).resolve().parents[1]
PAGES = {'INSTALL.md': 'install.html', 'RELEASE_NOTES.md': 'release-notes.html',
'THIRD_PARTY_NOTICES.md': 'licenses.html', 'LICENSE': 'license.html',
'BENCHMARKS.md': 'benchmarks.html', 'CONNECTORS.md': 'connectors.html', 'docs/legal/PUBLIC_SCOPE.md': 'legal-scope.html'}
def render(name):
source = (ROOT/name).read_text(encoding='utf-8')
# Strip fence language hints so optional Pygments cannot change generated bytes.
source = re.sub(r'^```[^\n]+$', '```', source, flags=re.MULTILINE)
title = source.splitlines()[0].lstrip('# ').strip() if name != 'LICENSE' else 'Application license — AGPL-3.0-only'
body = '<pre>'+html.escape(source)+'</pre>' if name == 'LICENSE' else markdown2.markdown(source, extras=['fenced-code-blocks', 'tables'], safe_mode='escape')
for old, new in PAGES.items():
body = body.replace('href="'+old+'"', 'href="'+new+'"')
return ('<!doctype html>\n<html lang="en"><head><meta charset="utf-8">'
'<meta name="viewport" content="width=device-width,initial-scale=1"><title>'+html.escape(title)+'</title>'
'<style>body{margin:0;background:#f5f4ef;color:#18342f;font:17px/1.65 system-ui,sans-serif}'
'main{max-width:900px;margin:auto;padding:35px 24px}a{color:#215947}h1{line-height:1.2}'
'pre{background:#e8ebe3;padding:18px;overflow-x:auto;white-space:pre-wrap;overflow-wrap:anywhere}'
'code{font-size:.88em}table{border-collapse:collapse;display:block;overflow:auto}'
'td,th{padding:9px;border:1px solid #becbbf;text-align:left}nav{margin-bottom:32px}'
'</style></head><body><main><nav><a href="index.html">Book OCR Studio</a> · '
'<a href="install.html">Install</a> · <a href="benchmarks.html">Comparison</a> · '
'<a href="release-notes.html">Release notes</a></nav>'+body+
'<hr><p><a href="'+html.escape(name)+'" download>Download original text</a></p></main></body></html>\n')
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--check', action='store_true')
args = parser.parse_args()
for source, target in PAGES.items():
output = render(source)
path = ROOT/target
if args.check:
if not path.exists() or path.read_text(encoding='utf-8') != output:
raise SystemExit('Stale public HTML: '+target)
else:
path.write_text(output, encoding='utf-8')
print('Public HTML pages match Markdown originals:', len(PAGES))
if __name__ == '__main__':
main()