Sepideh2027's picture
Update app.py
ba601a9 verified
Raw History Blame Contribute Delete
5.24 kB
# app.py
import gradio as gr
import torch
import numpy as np
from transformers import AutoTokenizer, AutoModelForSequenceClassification
import requests
import json
# Ψ§Ϊ―Ψ± Ω…Ψ―Ω„ ChemBERTa Ψ―Ψ± Spaces Ω…Ψ΄Ϊ©Ω„ Ψ―Ψ§Ψ΄Ψͺ، Ψ§Ψ² یک Ω…Ψ―Ω„ Ψ³Ψ§Ψ―Ω‡β€ŒΨͺΨ± Ψ§Ψ³Ψͺفاده Ω…ΫŒβ€ŒΪ©Ω†ΫŒΩ…
MODEL_NAME = "nlpaueb/bert-base-uncased-chemical" # Ω…Ψ―Ω„ Ψ΄ΫŒΩ…ΫŒΨ§ΫŒΫŒ Ψ³Ψ¨Ϊ©β€ŒΨͺΨ±
try:
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForSequenceClassification.from_pretrained(MODEL_NAME, num_labels=1)
except:
# Fallback Ω…Ψ―Ω„
tokenizer = AutoTokenizer.from_pretrained("bert-base-uncased")
model = None
def predict_properties(smiles):
"""
Predict molecular properties using a chemical language model
"""
# Basic validation
if not smiles or len(smiles) < 2:
return "Please enter a valid SMILES string"
# Try to use PubChem API for real data
pubchem_data = {}
try:
pubchem_url = f"https://pubchem.ncbi.nlm.nih.gov/rest/pug/compound/smiles/{smiles}/property/MolecularWeight,LogP,XLogP/JSON"
response = requests.get(pubchem_url, timeout=5)
if response.status_code == 200:
pubchem_data = response.json()
except:
pass
# Simulate ML prediction if model is available
if model:
inputs = tokenizer(smiles, return_tensors="pt", truncation=True, padding=True)
with torch.no_grad():
outputs = model(**inputs)
prediction = outputs.logits.item()
else:
# Fallback: simple heuristic based on SMILES length
prediction = len(smiles) * 0.1
# Generate report
report = f"""
# πŸ§ͺ Molecular Property Prediction
## πŸ“‹ Input
**SMILES:** `{smiles}`
## πŸ“Š Predicted Properties
### 1. Molecular Weight
"""
if pubchem_data and 'PropertyTable' in pubchem_data:
props = pubchem_data['PropertyTable']['Properties'][0]
report += f"- **Experimental:** {props.get('MolecularWeight', 'N/A')} g/mol\n"
else:
# Estimate from SMILES
weight_est = len(smiles.replace('=', '').replace('#', '')) * 13.5
report += f"- **Estimated:** {weight_est:.2f} g/mol (from SMILES pattern)\n"
report += f"- **ML Prediction Score:** {prediction:.4f}\n"
report += """
### 2. Lipophilicity (LogP)
"""
if pubchem_data and 'PropertyTable' in pubchem_data:
props = pubchem_data['PropertyTable']['Properties'][0]
logp = props.get('XLogP', props.get('LogP', 'N/A'))
report += f"- **Experimental LogP:** {logp}\n"
else:
# Simple heuristic
logp_est = smiles.count('C') * 0.5 + smiles.count('O') * (-0.3) + smiles.count('N') * (-0.4)
report += f"- **Estimated LogP:** {logp_est:.2f}\n"
report += f"""
### 3. Drug-likeness Assessment
#### Lipinski's Rule of 5 (Heuristic):
- **Molecular weight:** {'βœ“ < 500' if weight_est < 500 else 'βœ— > 500'}
- **LogP:** {'βœ“ < 5' if abs(logp_est) < 5 else 'βœ— > 5'}
- **H-bond donors:** Estimated {'< 5' if smiles.count('O') + smiles.count('N') < 5 else '> 5'}
- **H-bond acceptors:** Estimated {'< 10' if smiles.count('O') + smiles.count('N') < 10 else '> 10'}
### 4. Structural Features
- **Ring structures:** {'Present' if '(' in smiles or '1' in smiles else 'Not detected'}
- **Double bonds:** {smiles.count('=')} detected
- **Triple bonds:** {smiles.count('#')} detected
- **Charged groups:** {'Possible' if '.' in smiles or '+' in smiles or '-' in smiles else 'Not detected'}
## πŸ”¬ Scientific Context
This tool demonstrates:
1. **SMILES string** interpretation
2. **Molecular property** prediction using ML
3. **Drug-likeness** assessment
4. Integration with **PubChem API** for experimental data
## 🎯 Research Applications
- **Virtual screening** of compound libraries
- **Lead optimization** in drug discovery
- **QSAR model** development
- **Chemical space** exploration
---
*Note: This is a demonstration tool. For actual drug discovery, use established platforms like RDKit, Schrodinger, or MOE.*
"""
return report
# Create Gradio interface
demo = gr.Interface(
fn=predict_properties,
inputs=gr.Textbox(
label="Enter SMILES String",
placeholder="Example: CCO (ethanol), CC(=O)OC1=CC=CC=C1C(=O)O (aspirin)",
lines=2
),
outputs=gr.Markdown(label="Analysis Report"),
title="🧬 Molecular Property Predictor",
description="""A machine learning tool for predicting molecular properties from SMILES strings.
Demonstrates computational chemistry methods for drug discovery research.""",
examples=[
["CCO"], # Ethanol
["CC(=O)OC1=CC=CC=C1C(=O)O"], # Aspirin
["CN1C=NC2=C1C(=O)N(C(=O)N2C)C"], # Caffeine
["CC(C)CC1=CC=C(C=C1)C(C)C(=O)O"], # Ibuprofen
["C1=CC=C(C=C1)C=O"], # Benzaldehyde
],
theme="soft"
)
# Add academic credentials
demo.footer = """
**Academic Context:**
This tool demonstrates computational chemistry methods relevant to Prof. Natasha Bazelj's research in:
- Machine learning for drug discovery
- Molecular property prediction
- Virtual screening pipelines
- QSAR model development
"""
if __name__ == "__main__":
demo.launch(debug=True)