File size: 3,992 Bytes
b68816f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 | """Run the parsing pipeline without setting six environment variables by hand.
py scripts/parse.py # Windows
python3 scripts/parse.py # Linux
py scripts/parse.py data/other # a different folder
py scripts/parse.py --backend vlm # force a backend
Why Python rather than shell. This wrapper was written twice first — `parse.sh`
for bash and `parse.ps1` for PowerShell — and inside twenty minutes hit two traps
that are peculiar to shells:
- CRLF. `core.autocrlf=true` rewrites `.sh` on Windows checkout, and bash then
dies on line one with `set: pipefail`, a message that names nothing.
- A trailing backslash. A path containing spaces gets quoted by PowerShell, the
final backslash escapes the closing quote, and the NEXT argument is swallowed
— `--backend pipeline` ended up inside the value of `--input`.
Neither is possible here: `subprocess` takes an argument list and never goes
through a shell. One file, the same on every platform, and readable by anyone on
the team without knowing both shells.
This is a development aid. Production calls the module directly
(`python -m src.knowledge_parsing.run`) — parsing is an offline, admin-triggered
batch job, not something a person runs from a terminal.
"""
from __future__ import annotations
import os
import shutil
import subprocess
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
# Model weights and temp files are kept OFF C: — it has under 2 GB free.
MODELS = Path("D:/mineru-models")
ENVIRONMENT = {
"HF_HOME": str(MODELS / "hf"),
"MODELSCOPE_CACHE": str(MODELS / "modelscope"),
"TMP": str(MODELS / "tmp"),
"TEMP": str(MODELS / "tmp"),
"PYTHONIOENCODING": "utf-8",
"UV_CACHE_DIR": "D:/uv-cache",
}
# `--no-sync` is MANDATORY, not a preference. Plain `uv run` syncs the venv to
# uv.lock first, and `mineru` is not locked there — so it would UNINSTALL MinerU
# before running anything. This is also the repo's standard form; see CLAUDE.md.
UV = ["uv", "run", "--no-sync", "python"]
def _environment() -> dict[str, str]:
"""Variables the caller already set are NOT overridden."""
env = os.environ.copy()
for k, v in ENVIRONMENT.items():
env.setdefault(k, v)
return env
def main() -> int:
if shutil.which("uv") is None:
print("uv is not on PATH.", file=sys.stderr)
return 1
env = _environment()
if subprocess.run([*UV, "-c", "import mineru"], cwd=ROOT, env=env,
capture_output=True).returncode != 0:
print("MinerU is not installed in this repo's venv.", file=sys.stderr)
print(" UV_CACHE_DIR=D:/uv-cache uv pip install --python .venv "
"mineru==3.4.4", file=sys.stderr)
return 1
args = list(sys.argv[1:])
# A first argument that is not a flag is taken as the input folder.
if args and not args[0].startswith("-"):
input_dir = args.pop(0)
else:
input_dir = str(ROOT / "data" / "knowledge_docs")
# The backend is chosen from whether a GPU is present, and ALWAYS printed —
# no artifact should ever carry a backend nobody noticed.
if "--backend" not in args:
# flush is required: without it the parent's message appears AFTER the
# child's output, and the reader assumes the backend was chosen late.
if shutil.which("nvidia-smi"):
print(">> GPU detected - using the default hybrid/high", flush=True)
else:
print(">> NO GPU detected - falling back to pipeline")
print(" (no image descriptions; multiplication copied as the "
"letter 'x')", flush=True)
args += ["--backend", "pipeline"]
command = [*UV, "-m", "src.knowledge_parsing.run",
"--input", input_dir, *args]
return subprocess.run(command, cwd=ROOT, env=env).returncode
if __name__ == "__main__":
raise SystemExit(main())
|