BoHuangLab commited on
Commit
09d5e2c
·
verified ·
1 Parent(s): d6e8796

Sequence is the only input, fixed at 66 aa; sampling settings pinned

Browse files
Files changed (3) hide show
  1. README.md +12 -6
  2. app.py +16 -27
  3. pipeline.py +19 -2
README.md CHANGED
@@ -25,6 +25,11 @@ sections below document the ones that exist.
25
  Predict how an Intrinsically Disordered Peptide (IDP) behaves as its
26
  concentration rises, directly from sequence.
27
 
 
 
 
 
 
28
  Enter a sequence and this section runs the three stages of the CELL-FM condensate
29
  pipeline end to end:
30
 
@@ -57,9 +62,10 @@ probability 0.8, and the raw per-image calls as a downloadable CSV.
57
 
58
  ### Presets
59
 
60
- The dropdown is loaded with NUP98's FG-repeat region — wild type plus its full
61
- mutation panel (F→S, F→Y, charge and spacer variants). Comparing a mutant's AUC
62
- against WT is the quickest way to see the model respond to sequence grammar.
 
63
 
64
  ---
65
 
@@ -71,10 +77,10 @@ fallback (verified bit-identical to the Triton path: same AUC, same AAC, same
71
  256/256 per-image calls) — but a CPU run of the defaults would take hours and
72
  time out. Use ZeroGPU or a GPU hardware tier.
73
 
74
- The default settings — 512 images at 100 ODE steps — take **~85 s on an A40**
75
  (measured: 11.5 s per batch of 64, plus ~9 s of one-off model loading). Runtime
76
- scales linearly in both settings; lower either in **Sampling settings** for a
77
- faster, noisier curve.
78
 
79
  On ZeroGPU the per-call budget is set by `ZEROGPU_DURATION` (default 300 s), which
80
  covers the defaults with room to spare.
 
25
  Predict how an Intrinsically Disordered Peptide (IDP) behaves as its
26
  concentration rises, directly from sequence.
27
 
28
+ The only input is the sequence, and it must be **exactly 66 amino acids** — the
29
+ entire CondenSeq library the model was trained on is 66-mers (all 14,578 of them),
30
+ so other lengths are off-distribution. Sampling is fixed at 512 concentration
31
+ levels and 100 ODE steps, seed 6, matching the offline runs.
32
+
33
  Enter a sequence and this section runs the three stages of the CELL-FM condensate
34
  pipeline end to end:
35
 
 
62
 
63
  ### Presets
64
 
65
+ The dropdown is loaded with the NUP98 IDP — wild type (the default) plus its full
66
+ mutation panel, 92 sequences of 66 aa (F→S, F→Y, charge and spacer variants).
67
+ Comparing a mutant's AUC against WT is the quickest way to see the model respond
68
+ to sequence grammar.
69
 
70
  ---
71
 
 
77
  256/256 per-image calls) — but a CPU run of the defaults would take hours and
78
  time out. Use ZeroGPU or a GPU hardware tier.
79
 
80
+ The fixed settings — 512 images at 100 ODE steps — take **~85 s on an A40**
81
  (measured: 11.5 s per batch of 64, plus ~9 s of one-off model loading). Runtime
82
+ scales linearly in both, so a section that needs to be faster should lower them in
83
+ `pipeline.py` (`NUM_IMAGES`, `NUM_STEPS`).
84
 
85
  On ZeroGPU the per-call budget is set by `ZEROGPU_DURATION` (default 300 s), which
86
  covers the defaults with room to spare.
app.py CHANGED
@@ -58,8 +58,6 @@ INK = "#0b0b0b"
58
  MUTED = "#52514e"
59
  SURFACE = "#fcfcfb"
60
 
61
- MAX_SEQUENCE_LEN = 2048
62
-
63
 
64
  def load_presets():
65
  """NUP98 WT plus its mutation panel, as demo starting points."""
@@ -166,19 +164,13 @@ def results_table(result: pipeline.Result) -> pd.DataFrame:
166
 
167
 
168
  @GPU_DECORATOR
169
- def analyse(sequence, num_images, num_steps, batch_size, seed, progress=gr.Progress()):
170
- sequence = pipeline.clean_sequence(sequence)
171
- if len(sequence) > MAX_SEQUENCE_LEN:
172
- raise gr.Error(f"Sequence is {len(sequence)} aa; the model takes at most {MAX_SEQUENCE_LEN}.")
173
-
174
- result = pipeline.run(
175
- sequence,
176
- num_images=int(num_images),
177
- num_steps=int(num_steps),
178
- batch_size=int(batch_size),
179
- seed=int(seed),
180
- progress=progress,
181
- )
182
 
183
  table = results_table(result)
184
  csv_path = os.path.join("/tmp", "pred_vs_intensity.csv")
@@ -216,20 +208,17 @@ def build_condensate_titration_section():
216
  with gr.Row():
217
  with gr.Column(scale=2):
218
  preset = gr.Dropdown(
219
- choices=list(PRESETS), value="NUP98 WT", label="Preset",
220
- info="NUP98 FG-repeat WT and its mutation panel. Editing the box below overrides this.",
221
  )
222
  sequence = gr.Textbox(
223
- value=DEFAULT_SEQUENCE, lines=4, label="Protein sequence",
224
- info="Single-letter amino acids; FASTA headers are stripped.",
 
 
 
 
225
  )
226
- with gr.Accordion("Sampling settings", open=False):
227
- num_images = gr.Slider(64, 1024, value=512, step=64, label="Images (concentration levels)",
228
- info="More levels give a smoother curve and cost proportionally more time.")
229
- num_steps = gr.Slider(20, 200, value=100, step=10, label="ODE steps per image")
230
- batch_size = gr.Slider(16, 256, value=64, step=16, label="Batch size")
231
- seed = gr.Number(value=pipeline.DEFAULT_SEED, precision=0, label="Random seed",
232
- info="Generation starts from a random latent; fix the seed for a repeatable curve.")
233
  run_button = gr.Button("Run analysis", variant="primary")
234
 
235
  with gr.Column(scale=3):
@@ -255,7 +244,7 @@ def build_condensate_titration_section():
255
  preset.change(lambda name: PRESETS[name], inputs=preset, outputs=sequence)
256
  run_button.click(
257
  analyse,
258
- inputs=[sequence, num_images, num_steps, batch_size, seed],
259
  outputs=[curve_plot, sample_plot, summary, table, download],
260
  )
261
 
 
58
  MUTED = "#52514e"
59
  SURFACE = "#fcfcfb"
60
 
 
 
61
 
62
  def load_presets():
63
  """NUP98 WT plus its mutation panel, as demo starting points."""
 
164
 
165
 
166
  @GPU_DECORATOR
167
+ def analyse(sequence, progress=gr.Progress()):
168
+ try:
169
+ sequence = pipeline.clean_sequence(sequence)
170
+ except ValueError as exc:
171
+ raise gr.Error(str(exc)) from exc
172
+
173
+ result = pipeline.run(sequence, progress=progress)
 
 
 
 
 
 
174
 
175
  table = results_table(result)
176
  csv_path = os.path.join("/tmp", "pred_vs_intensity.csv")
 
208
  with gr.Row():
209
  with gr.Column(scale=2):
210
  preset = gr.Dropdown(
211
+ choices=list(PRESETS), value="NUP98 WT", label="Load a sequence",
212
+ info="NUP98 IDP and its mutation panel; fills the box below, which you can edit.",
213
  )
214
  sequence = gr.Textbox(
215
+ value=DEFAULT_SEQUENCE, lines=4,
216
+ label=f"Protein sequence ({pipeline.SEQUENCE_LENGTH} aa)",
217
+ info=(
218
+ f"Exactly {pipeline.SEQUENCE_LENGTH} single-letter amino acids — the whole "
219
+ "CondenSeq library is 66-mers. FASTA headers are stripped."
220
+ ),
221
  )
 
 
 
 
 
 
 
222
  run_button = gr.Button("Run analysis", variant="primary")
223
 
224
  with gr.Column(scale=3):
 
244
  preset.change(lambda name: PRESETS[name], inputs=preset, outputs=sequence)
245
  run_button.click(
246
  analyse,
247
+ inputs=sequence,
248
  outputs=[curve_plot, sample_plot, summary, table, download],
249
  )
250
 
pipeline.py CHANGED
@@ -48,11 +48,22 @@ WINDOW_DIVISOR = 8 # window = num_images // 8, matching ana_all_mutation_log_sc
48
 
49
  VALID_AA = set("ACDEFGHIKLMNPQRSTVWY")
50
 
 
 
 
 
 
51
  # Generation starts from a random latent, so the curve moves a little between runs.
52
  # The offline scripts fixed this at 6; keeping the same default makes the demo's
53
  # numbers repeatable and comparable to them.
54
  DEFAULT_SEED = 6
55
 
 
 
 
 
 
 
56
 
57
  def resolve_weights(filename: str) -> str:
58
  """Local directory if CELLFM_LOCAL_WEIGHTS is set, else pull from the model repo."""
@@ -191,6 +202,12 @@ def clean_sequence(sequence: str) -> str:
191
  f"Sequence contains non-standard residues: {', '.join(bad)}. "
192
  "Use the 20 standard amino acids."
193
  )
 
 
 
 
 
 
194
  return seq
195
 
196
 
@@ -298,8 +315,8 @@ def sample_panel(images, intensities, k=6):
298
  return images[idx, 1].astype(np.float32) / 65535.0, intensities[idx]
299
 
300
 
301
- def run(sequence, num_images=512, num_steps=100, batch_size=64, seed=DEFAULT_SEED,
302
- progress=None) -> Result:
303
  """The whole pipeline: sequence in, titration curve and its areas out."""
304
  sequence = clean_sequence(sequence)
305
 
 
48
 
49
  VALID_AA = set("ACDEFGHIKLMNPQRSTVWY")
50
 
51
+ # CondenSeq is a fixed-length library: all 14,578 training sequences are exactly
52
+ # 66 residues. Anything else is off-distribution, so the demo requires this length
53
+ # rather than silently padding or cropping.
54
+ SEQUENCE_LENGTH = 66
55
+
56
  # Generation starts from a random latent, so the curve moves a little between runs.
57
  # The offline scripts fixed this at 6; keeping the same default makes the demo's
58
  # numbers repeatable and comparable to them.
59
  DEFAULT_SEED = 6
60
 
61
+ # Fixed sampling settings. The concentration ladder length and ODE step count match
62
+ # the offline runs, so demo numbers stay comparable to the published ones.
63
+ NUM_IMAGES = 512
64
+ NUM_STEPS = 100
65
+ BATCH_SIZE = 64
66
+
67
 
68
  def resolve_weights(filename: str) -> str:
69
  """Local directory if CELLFM_LOCAL_WEIGHTS is set, else pull from the model repo."""
 
202
  f"Sequence contains non-standard residues: {', '.join(bad)}. "
203
  "Use the 20 standard amino acids."
204
  )
205
+ if len(seq) != SEQUENCE_LENGTH:
206
+ raise ValueError(
207
+ f"Sequence is {len(seq)} residues; CELL-FM CondenSeq takes exactly "
208
+ f"{SEQUENCE_LENGTH}. The whole training library is {SEQUENCE_LENGTH}-mers, "
209
+ "so other lengths are outside what the model has seen."
210
+ )
211
  return seq
212
 
213
 
 
315
  return images[idx, 1].astype(np.float32) / 65535.0, intensities[idx]
316
 
317
 
318
+ def run(sequence, num_images=NUM_IMAGES, num_steps=NUM_STEPS, batch_size=BATCH_SIZE,
319
+ seed=DEFAULT_SEED, progress=None) -> Result:
320
  """The whole pipeline: sequence in, titration curve and its areas out."""
321
  sequence = clean_sequence(sequence)
322