Spaces:
Running on Zero
Running on Zero
Sequence is the only input, fixed at 66 aa; sampling settings pinned
Browse files- README.md +12 -6
- app.py +16 -27
- pipeline.py +19 -2
README.md
CHANGED
|
@@ -25,6 +25,11 @@ sections below document the ones that exist.
|
|
| 25 |
Predict how an Intrinsically Disordered Peptide (IDP) behaves as its
|
| 26 |
concentration rises, directly from sequence.
|
| 27 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
Enter a sequence and this section runs the three stages of the CELL-FM condensate
|
| 29 |
pipeline end to end:
|
| 30 |
|
|
@@ -57,9 +62,10 @@ probability 0.8, and the raw per-image calls as a downloadable CSV.
|
|
| 57 |
|
| 58 |
### Presets
|
| 59 |
|
| 60 |
-
The dropdown is loaded with NUP98
|
| 61 |
-
mutation panel (F→S, F→Y, charge and spacer variants).
|
| 62 |
-
against WT is the quickest way to see the model respond
|
|
|
|
| 63 |
|
| 64 |
---
|
| 65 |
|
|
@@ -71,10 +77,10 @@ fallback (verified bit-identical to the Triton path: same AUC, same AAC, same
|
|
| 71 |
256/256 per-image calls) — but a CPU run of the defaults would take hours and
|
| 72 |
time out. Use ZeroGPU or a GPU hardware tier.
|
| 73 |
|
| 74 |
-
The
|
| 75 |
(measured: 11.5 s per batch of 64, plus ~9 s of one-off model loading). Runtime
|
| 76 |
-
scales linearly in both
|
| 77 |
-
|
| 78 |
|
| 79 |
On ZeroGPU the per-call budget is set by `ZEROGPU_DURATION` (default 300 s), which
|
| 80 |
covers the defaults with room to spare.
|
|
|
|
| 25 |
Predict how an Intrinsically Disordered Peptide (IDP) behaves as its
|
| 26 |
concentration rises, directly from sequence.
|
| 27 |
|
| 28 |
+
The only input is the sequence, and it must be **exactly 66 amino acids** — the
|
| 29 |
+
entire CondenSeq library the model was trained on is 66-mers (all 14,578 of them),
|
| 30 |
+
so other lengths are off-distribution. Sampling is fixed at 512 concentration
|
| 31 |
+
levels and 100 ODE steps, seed 6, matching the offline runs.
|
| 32 |
+
|
| 33 |
Enter a sequence and this section runs the three stages of the CELL-FM condensate
|
| 34 |
pipeline end to end:
|
| 35 |
|
|
|
|
| 62 |
|
| 63 |
### Presets
|
| 64 |
|
| 65 |
+
The dropdown is loaded with the NUP98 IDP — wild type (the default) plus its full
|
| 66 |
+
mutation panel, 92 sequences of 66 aa (F→S, F→Y, charge and spacer variants).
|
| 67 |
+
Comparing a mutant's AUC against WT is the quickest way to see the model respond
|
| 68 |
+
to sequence grammar.
|
| 69 |
|
| 70 |
---
|
| 71 |
|
|
|
|
| 77 |
256/256 per-image calls) — but a CPU run of the defaults would take hours and
|
| 78 |
time out. Use ZeroGPU or a GPU hardware tier.
|
| 79 |
|
| 80 |
+
The fixed settings — 512 images at 100 ODE steps — take **~85 s on an A40**
|
| 81 |
(measured: 11.5 s per batch of 64, plus ~9 s of one-off model loading). Runtime
|
| 82 |
+
scales linearly in both, so a section that needs to be faster should lower them in
|
| 83 |
+
`pipeline.py` (`NUM_IMAGES`, `NUM_STEPS`).
|
| 84 |
|
| 85 |
On ZeroGPU the per-call budget is set by `ZEROGPU_DURATION` (default 300 s), which
|
| 86 |
covers the defaults with room to spare.
|
app.py
CHANGED
|
@@ -58,8 +58,6 @@ INK = "#0b0b0b"
|
|
| 58 |
MUTED = "#52514e"
|
| 59 |
SURFACE = "#fcfcfb"
|
| 60 |
|
| 61 |
-
MAX_SEQUENCE_LEN = 2048
|
| 62 |
-
|
| 63 |
|
| 64 |
def load_presets():
|
| 65 |
"""NUP98 WT plus its mutation panel, as demo starting points."""
|
|
@@ -166,19 +164,13 @@ def results_table(result: pipeline.Result) -> pd.DataFrame:
|
|
| 166 |
|
| 167 |
|
| 168 |
@GPU_DECORATOR
|
| 169 |
-
def analyse(sequence,
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
|
| 175 |
-
|
| 176 |
-
num_images=int(num_images),
|
| 177 |
-
num_steps=int(num_steps),
|
| 178 |
-
batch_size=int(batch_size),
|
| 179 |
-
seed=int(seed),
|
| 180 |
-
progress=progress,
|
| 181 |
-
)
|
| 182 |
|
| 183 |
table = results_table(result)
|
| 184 |
csv_path = os.path.join("/tmp", "pred_vs_intensity.csv")
|
|
@@ -216,20 +208,17 @@ def build_condensate_titration_section():
|
|
| 216 |
with gr.Row():
|
| 217 |
with gr.Column(scale=2):
|
| 218 |
preset = gr.Dropdown(
|
| 219 |
-
choices=list(PRESETS), value="NUP98 WT", label="
|
| 220 |
-
info="NUP98
|
| 221 |
)
|
| 222 |
sequence = gr.Textbox(
|
| 223 |
-
value=DEFAULT_SEQUENCE, lines=4,
|
| 224 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 225 |
)
|
| 226 |
-
with gr.Accordion("Sampling settings", open=False):
|
| 227 |
-
num_images = gr.Slider(64, 1024, value=512, step=64, label="Images (concentration levels)",
|
| 228 |
-
info="More levels give a smoother curve and cost proportionally more time.")
|
| 229 |
-
num_steps = gr.Slider(20, 200, value=100, step=10, label="ODE steps per image")
|
| 230 |
-
batch_size = gr.Slider(16, 256, value=64, step=16, label="Batch size")
|
| 231 |
-
seed = gr.Number(value=pipeline.DEFAULT_SEED, precision=0, label="Random seed",
|
| 232 |
-
info="Generation starts from a random latent; fix the seed for a repeatable curve.")
|
| 233 |
run_button = gr.Button("Run analysis", variant="primary")
|
| 234 |
|
| 235 |
with gr.Column(scale=3):
|
|
@@ -255,7 +244,7 @@ def build_condensate_titration_section():
|
|
| 255 |
preset.change(lambda name: PRESETS[name], inputs=preset, outputs=sequence)
|
| 256 |
run_button.click(
|
| 257 |
analyse,
|
| 258 |
-
inputs=
|
| 259 |
outputs=[curve_plot, sample_plot, summary, table, download],
|
| 260 |
)
|
| 261 |
|
|
|
|
| 58 |
MUTED = "#52514e"
|
| 59 |
SURFACE = "#fcfcfb"
|
| 60 |
|
|
|
|
|
|
|
| 61 |
|
| 62 |
def load_presets():
|
| 63 |
"""NUP98 WT plus its mutation panel, as demo starting points."""
|
|
|
|
| 164 |
|
| 165 |
|
| 166 |
@GPU_DECORATOR
|
| 167 |
+
def analyse(sequence, progress=gr.Progress()):
|
| 168 |
+
try:
|
| 169 |
+
sequence = pipeline.clean_sequence(sequence)
|
| 170 |
+
except ValueError as exc:
|
| 171 |
+
raise gr.Error(str(exc)) from exc
|
| 172 |
+
|
| 173 |
+
result = pipeline.run(sequence, progress=progress)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 174 |
|
| 175 |
table = results_table(result)
|
| 176 |
csv_path = os.path.join("/tmp", "pred_vs_intensity.csv")
|
|
|
|
| 208 |
with gr.Row():
|
| 209 |
with gr.Column(scale=2):
|
| 210 |
preset = gr.Dropdown(
|
| 211 |
+
choices=list(PRESETS), value="NUP98 WT", label="Load a sequence",
|
| 212 |
+
info="NUP98 IDP and its mutation panel; fills the box below, which you can edit.",
|
| 213 |
)
|
| 214 |
sequence = gr.Textbox(
|
| 215 |
+
value=DEFAULT_SEQUENCE, lines=4,
|
| 216 |
+
label=f"Protein sequence ({pipeline.SEQUENCE_LENGTH} aa)",
|
| 217 |
+
info=(
|
| 218 |
+
f"Exactly {pipeline.SEQUENCE_LENGTH} single-letter amino acids — the whole "
|
| 219 |
+
"CondenSeq library is 66-mers. FASTA headers are stripped."
|
| 220 |
+
),
|
| 221 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 222 |
run_button = gr.Button("Run analysis", variant="primary")
|
| 223 |
|
| 224 |
with gr.Column(scale=3):
|
|
|
|
| 244 |
preset.change(lambda name: PRESETS[name], inputs=preset, outputs=sequence)
|
| 245 |
run_button.click(
|
| 246 |
analyse,
|
| 247 |
+
inputs=sequence,
|
| 248 |
outputs=[curve_plot, sample_plot, summary, table, download],
|
| 249 |
)
|
| 250 |
|
pipeline.py
CHANGED
|
@@ -48,11 +48,22 @@ WINDOW_DIVISOR = 8 # window = num_images // 8, matching ana_all_mutation_log_sc
|
|
| 48 |
|
| 49 |
VALID_AA = set("ACDEFGHIKLMNPQRSTVWY")
|
| 50 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
# Generation starts from a random latent, so the curve moves a little between runs.
|
| 52 |
# The offline scripts fixed this at 6; keeping the same default makes the demo's
|
| 53 |
# numbers repeatable and comparable to them.
|
| 54 |
DEFAULT_SEED = 6
|
| 55 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
|
| 57 |
def resolve_weights(filename: str) -> str:
|
| 58 |
"""Local directory if CELLFM_LOCAL_WEIGHTS is set, else pull from the model repo."""
|
|
@@ -191,6 +202,12 @@ def clean_sequence(sequence: str) -> str:
|
|
| 191 |
f"Sequence contains non-standard residues: {', '.join(bad)}. "
|
| 192 |
"Use the 20 standard amino acids."
|
| 193 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 194 |
return seq
|
| 195 |
|
| 196 |
|
|
@@ -298,8 +315,8 @@ def sample_panel(images, intensities, k=6):
|
|
| 298 |
return images[idx, 1].astype(np.float32) / 65535.0, intensities[idx]
|
| 299 |
|
| 300 |
|
| 301 |
-
def run(sequence, num_images=
|
| 302 |
-
progress=None) -> Result:
|
| 303 |
"""The whole pipeline: sequence in, titration curve and its areas out."""
|
| 304 |
sequence = clean_sequence(sequence)
|
| 305 |
|
|
|
|
| 48 |
|
| 49 |
VALID_AA = set("ACDEFGHIKLMNPQRSTVWY")
|
| 50 |
|
| 51 |
+
# CondenSeq is a fixed-length library: all 14,578 training sequences are exactly
|
| 52 |
+
# 66 residues. Anything else is off-distribution, so the demo requires this length
|
| 53 |
+
# rather than silently padding or cropping.
|
| 54 |
+
SEQUENCE_LENGTH = 66
|
| 55 |
+
|
| 56 |
# Generation starts from a random latent, so the curve moves a little between runs.
|
| 57 |
# The offline scripts fixed this at 6; keeping the same default makes the demo's
|
| 58 |
# numbers repeatable and comparable to them.
|
| 59 |
DEFAULT_SEED = 6
|
| 60 |
|
| 61 |
+
# Fixed sampling settings. The concentration ladder length and ODE step count match
|
| 62 |
+
# the offline runs, so demo numbers stay comparable to the published ones.
|
| 63 |
+
NUM_IMAGES = 512
|
| 64 |
+
NUM_STEPS = 100
|
| 65 |
+
BATCH_SIZE = 64
|
| 66 |
+
|
| 67 |
|
| 68 |
def resolve_weights(filename: str) -> str:
|
| 69 |
"""Local directory if CELLFM_LOCAL_WEIGHTS is set, else pull from the model repo."""
|
|
|
|
| 202 |
f"Sequence contains non-standard residues: {', '.join(bad)}. "
|
| 203 |
"Use the 20 standard amino acids."
|
| 204 |
)
|
| 205 |
+
if len(seq) != SEQUENCE_LENGTH:
|
| 206 |
+
raise ValueError(
|
| 207 |
+
f"Sequence is {len(seq)} residues; CELL-FM CondenSeq takes exactly "
|
| 208 |
+
f"{SEQUENCE_LENGTH}. The whole training library is {SEQUENCE_LENGTH}-mers, "
|
| 209 |
+
"so other lengths are outside what the model has seen."
|
| 210 |
+
)
|
| 211 |
return seq
|
| 212 |
|
| 213 |
|
|
|
|
| 315 |
return images[idx, 1].astype(np.float32) / 65535.0, intensities[idx]
|
| 316 |
|
| 317 |
|
| 318 |
+
def run(sequence, num_images=NUM_IMAGES, num_steps=NUM_STEPS, batch_size=BATCH_SIZE,
|
| 319 |
+
seed=DEFAULT_SEED, progress=None) -> Result:
|
| 320 |
"""The whole pipeline: sequence in, titration curve and its areas out."""
|
| 321 |
sequence = clean_sequence(sequence)
|
| 322 |
|