czty's picture
Add files using upload-large-folder tool
4140be3 verified
Raw
History Blame Contribute Delete
4.45 kB
#!/usr/bin/env python3
"""
Annotate runtime environment for Bioconda tool JSON files.
Default targets:
- output/bioconda_source_tool/bioconda_t0_core_tools.json
- output/bioconda_source_tool/bioconda_t1_domain_tools.json
- output/bioconda_source_tool/bioconda_t2_on_demand_tools.json
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from typing import Dict, List, Tuple
def _lower_list(values: List[str]) -> List[str]:
return [str(v).strip().lower() for v in values if str(v).strip()]
def infer_runtime(package_name: str, dependencies: List[str], summary: str, description: str) -> Tuple[str, str]:
pkg = package_name.lower()
deps = _lower_list(dependencies)
text = f"{summary} {description}".lower()
# Rule 1: R ecosystem
if pkg.startswith("r-") or pkg.startswith("bioconductor-"):
return "R", "package_name prefix indicates R/Bioconductor ecosystem"
if any(d.startswith("r-") or "r-base" in d for d in deps):
return "R", "dependencies include r-base or r-* packages"
# Rule 2: Perl ecosystem
if pkg.startswith("perl-") or any(d.startswith("perl-") or d == "perl" for d in deps):
return "Perl", "package/dependencies indicate Perl runtime"
# Rule 3: Java ecosystem
if any("openjdk" in d or "default-jre" in d or d == "java" for d in deps):
return "Java", "dependencies include Java runtime (openjdk/jre)"
# Rule 4: Python ecosystem
if any("python" in d for d in deps):
return "Python", "dependencies include python"
py_keywords = ("python", "pypi", "scanpy", "scvelo", "scvi", "anndata")
if any(k in text for k in py_keywords):
return "Python", "summary/description contains Python ecosystem keywords"
# Rule 5: Mostly compiled binary style
compiled_markers = ("libgcc", "libstdcxx", "htslib", "zlib", "gcc", "gxx")
if any(any(m in d for m in compiled_markers) for d in deps):
return "Compiled", "dependencies suggest compiled/native executable"
return "Other", "no strong signal for Python/R/Perl/Java; marked as Other"
def annotate_file(json_path: Path, inplace: bool = True) -> Dict[str, int]:
if not json_path.exists():
return {"total": 0, "updated": 0}
with json_path.open("r", encoding="utf-8") as f:
rows = json.load(f)
if not isinstance(rows, list):
raise ValueError(f"JSON root must be list: {json_path}")
updated = 0
for row in rows:
if not isinstance(row, dict):
continue
runtime, reason = infer_runtime(
package_name=str(row.get("package_name", "")).strip(),
dependencies=list(row.get("dependencies", []) or []),
summary=str(row.get("summary", "")).strip(),
description=str(row.get("description", "")).strip(),
)
row["execution_environment"] = runtime
row["execution_environment_reason"] = reason
updated += 1
if inplace:
with json_path.open("w", encoding="utf-8") as f:
json.dump(rows, f, ensure_ascii=False, indent=2)
return {"total": len(rows), "updated": updated}
def main() -> None:
parser = argparse.ArgumentParser(description="Add execution environment annotations to tool JSON files.")
parser.add_argument(
"--files",
nargs="*",
default=[
"/225040511/project/BioScientist/agent_system/toolbase/output/bioconda_source_tool/bioconda_t0_core_tools.json",
"/225040511/project/BioScientist/agent_system/toolbase/output/bioconda_source_tool/bioconda_t1_domain_tools.json",
"/225040511/project/BioScientist/agent_system/toolbase/output/bioconda_source_tool/bioconda_t2_on_demand_tools.json",
],
help="Target JSON files.",
)
args = parser.parse_args()
results = {}
for p in args.files:
path = Path(p)
stats = annotate_file(path, inplace=True)
results[str(path)] = stats
print(f"{path}: updated {stats['updated']}/{stats['total']}")
report_path = Path("/225040511/project/BioScientist/agent_system/toolbase/output/bioconda_source_tool/runtime_annotation_report.json")
report_path.parent.mkdir(parents=True, exist_ok=True)
with report_path.open("w", encoding="utf-8") as f:
json.dump(results, f, ensure_ascii=False, indent=2)
print(f"Saved report: {report_path}")
if __name__ == "__main__":
main()