Abdullahcoder54 commited on
Commit
47ca9a8
·
1 Parent(s): 6a9b7a4

Replace with LTX-2.5 Docker Space app (text_to_video / i2v + narration audio)

Browse files
Files changed (13) hide show
  1. .dockerignore +6 -0
  2. .gitattributes +1 -1
  3. .gitignore +6 -11
  4. .python-version +0 -1
  5. Dockerfile +15 -14
  6. README.md +9 -4
  7. app.py +343 -45
  8. chatbot.py +0 -146
  9. pyproject.toml +0 -22
  10. reportanalysis.py +0 -129
  11. requirements.txt +12 -12
  12. runtime.txt +0 -1
  13. uv.lock +0 -0
.dockerignore ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ .git
2
+ .gitignore
3
+ .gitattributes
4
+ __pycache__
5
+ *.pyc
6
+ .DS_Store
.gitattributes CHANGED
@@ -32,4 +32,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
32
  *.xz filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
32
  *.xz filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
.gitignore CHANGED
@@ -1,11 +1,6 @@
1
- # Python-generated files
2
- __pycache__/
3
- *.py[oc]
4
- build/
5
- dist/
6
- wheels/
7
- *.egg-info
8
-
9
- # Virtual environments
10
- .venv
11
- .env
 
1
+ # Local model / cache artifacts
2
+ *.gguf
3
+ *.safetensors
4
+ /tmp/
5
+ .cache/
6
+ __pycache__/
 
 
 
 
 
.python-version DELETED
@@ -1 +0,0 @@
1
- 3.12
 
 
Dockerfile CHANGED
@@ -1,21 +1,22 @@
 
1
 
2
- FROM python:3.10
 
 
3
 
4
- # Pehle root user hai by default, to yahan packages install karo
5
- RUN apt-get update && apt-get install -y libgl1 libglib2.0-0
6
 
7
- # Ab user add karo aur switch karo
8
- RUN useradd -m -u 1000 user
9
- USER user
10
- ENV PATH="/home/user/.local/bin:$PATH"
11
 
12
- WORKDIR /app
 
 
 
13
 
 
14
 
15
- COPY --chown=user ./requirements.txt requirements.txt
16
- RUN pip install --no-cache-dir --upgrade -r requirements.txt
17
- RUN python -m spacy download en_core_web_lg
18
- RUN python -c "from doctr.models import ocr_predictor; ocr_predictor(pretrained=True)"
19
 
20
- COPY --chown=user . /app
21
- CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860"]
 
1
+ FROM python:3.12-slim
2
 
3
+ ENV PYTHONDONTWRITEBYTECODE=1 \
4
+ PYTHONUNBUFFERED=1 \
5
+ PIP_NO_CACHE_DIR=1
6
 
7
+ WORKDIR /app
 
8
 
9
+ RUN apt-get update \
10
+ && apt-get install -y --no-install-recommends ffmpeg git \
11
+ && rm -rf /var/lib/apt/lists/*
 
12
 
13
+ COPY requirements.txt .
14
+ RUN pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cu124 \
15
+ && pip install --no-cache-dir -r requirements.txt \
16
+ && pip install --no-cache-dir gradio spaces huggingface_hub
17
 
18
+ COPY . .
19
 
20
+ EXPOSE 7860
 
 
 
21
 
22
+ CMD ["python", "app.py"]
 
README.md CHANGED
@@ -1,9 +1,14 @@
1
  ---
2
- title: DoctorXpert Backend
3
- emoji: 🩺
4
- colorFrom: yellow
5
- colorTo: red
6
  sdk: docker
7
  app_file: app.py
8
  pinned: false
9
  ---
 
 
 
 
 
 
1
  ---
2
+ title: Video Creator
3
+ emoji: 🔥
4
+ colorFrom: green
5
+ colorTo: blue
6
  sdk: docker
7
  app_file: app.py
8
  pinned: false
9
  ---
10
+
11
+ Docker-hosted LTX-2.5 Space for the AI Shorts Factory backend. Four Gradio
12
+ endpoints via `/call/{fn}` (text_to_video, text_to_video_av, image_to_video,
13
+ image_to_video_av). Requires a GPU tier — the free CPU tier (2 vCPU/16GB)
14
+ boots but generation OOMs.
app.py CHANGED
@@ -1,57 +1,355 @@
1
- from fastapi import FastAPI, UploadFile , File , Request
2
- from fastapi.middleware.cors import CORSMiddleware
3
- from agents import Runner
4
- from logging import getLogger
5
- from reportanalysis import Report_Agent, extract_text
6
- from chatbot import get_health_response
7
- app = FastAPI()
8
- log = getLogger()
9
-
10
- app.add_middleware(
11
- CORSMiddleware,
12
- allow_origins=["*"],
13
- allow_credentials=True,
14
- allow_methods=["*"],
15
- allow_headers=["*"],
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
17
 
18
- @app.get("/")
19
- async def root():
20
- return {"message": "Welcome to the Medical Report Analysis API"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
21
 
22
- @app.post("/chatbot")
23
- async def health_info(request: Request):
24
- data = await request.json()
25
- msg = data.get("message")
26
- result = await get_health_response(msg) # <- await here
27
- result.replace("\n/", "<br>")
28
- return {"response": result}
29
 
30
- @app.post("/report-analysis")
31
- async def upload_file(file: UploadFile = File(...)):
 
 
 
 
 
 
 
 
 
 
 
32
  try:
33
- pdf = file.filename.lower().endswith('.pdf')
34
- doc = file.filename.lower().endswith('.docx')
35
- content = await file.read()
36
- text = extract_text(content, pdf, doc)
37
- result = await Runner.run(
38
- Report_Agent,
39
- f"""Please analyze the uploaded medical report image extrected text :
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
40
 
41
- {text}
 
42
 
43
- Do not provide any analysis before calling the tool.
 
 
44
 
45
- Once the text is extracted, continue with step-by-step medical analysis and return the final output strictly in JSON format.
 
46
 
47
- """,
48
- context=content,
49
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
50
  )
51
-
52
- print(result.final_output)
53
- return {"result": result.final_output.model_dump()}
54
- except Exception as e:
55
- return {"error": str(e)}
56
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
57
 
 
 
 
1
+ """LTX-2.5 Space — ZeroGPU inference API for the AI Shorts Factory backend.
2
+
3
+ Runs the distilled LTX-2.5 GGUF (Q4_K_M) on zero GPU as a Gradio demo. Four
4
+ endpoints are exposed through Gradio's `/call/{fn}` protocol and consumed by
5
+ the backend provider `app.ai.visuals.ltx`:
6
+
7
+ text_to_video(prompt, width, height, num_frames, enhance_prompt)
8
+ text_to_video_av(...) # video + synchronized 48kHz narration audio
9
+ image_to_video(prompt, image_url, width, height, num_frames, enhance_prompt)
10
+ image_to_video_av(...) # video + synchronized narration audio
11
+
12
+ Every function returns ``(video_path, first_frame_path)``. Scene 2..N of the
13
+ pipeline call image_to_video with the last frame URL of the previous clip so
14
+ the subject stays continuous (last-frame chaining).
15
+
16
+ The backend always sends an already-composed prompt (quoted dialogue for the
17
+ exact script words) and passes ``enhance_prompt=False``; the prompt enhancer
18
+ is a hard no-op here so no rewrite happens.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import os
24
+ import subprocess
25
+ import threading
26
+
27
+ # ZeroGPU rule #1: `import spaces` must precede any CUDA-touching import
28
+ # (torch etc.) — it monkey-patches torch.cuda at import time.
29
+ import spaces # noqa: E402
30
+
31
+ import numpy as np
32
+ import torch
33
+
34
+ HF_TOKEN = os.environ.get("HF_TOKEN", "")
35
+
36
+ # Direct download URL of the distilled GGUF transformer (~15GB). The backend
37
+ # only needs LTX_SPACE_URL; this key is configured in the Space.
38
+ LTX_MODEL_URL = os.environ.get(
39
+ "LTX_MODEL_URL",
40
+ "https://huggingface.co/realrebelai/LTX-2.5_GGUFs/resolve/main/LTX-2.5-Distilled-Q4_K_M.gguf",
41
  )
42
+ PIPELINE_ID = "Lightricks/LTX-2.5-Diffusers"
43
+ SEED = int(os.environ.get("LTX_SEED", "0"))
44
+ FPS = int(os.environ.get("LTX_FPS", "24"))
45
+
46
+ # Distilled model: guidance is baked into the weights, few-step schedule.
47
+ INFERENCE_STEPS = int(os.environ.get("LTX_STEPS", "8"))
48
+ GUIDANCE_SCALE = float(os.environ.get("LTX_GUIDANCE", "1.0"))
49
+ NEGATIVE_PROMPT = "low quality, blurry, distorted, watermark"
50
+ # Honest worst-case GPU wall-time per generation. ZeroGPU validates this
51
+ # after a 1.5x multiplier and sits a ~60-120s continuous execution cap per
52
+ # call on the free tier — 120 keeps us inside both (120 * 1.5 = 180 < 300
53
+ # cap, and matches the real per-call window after the model is resident).
54
+ GPU_DURATION = int(os.environ.get("LTX_GPU_DURATION", "120"))
55
+
56
+ _lock = threading.Lock()
57
+ _model = None
58
+
59
+
60
+ # ----------------------------------------------------------------- model load
61
+
62
+
63
+ def _download_gguf(dest: str) -> str:
64
+ import httpx
65
+
66
+ if LTX_MODEL_URL.startswith("http"):
67
+ if os.path.exists(dest):
68
+ return dest
69
+ with httpx.stream("GET", LTX_MODEL_URL, follow_redirects=True, timeout=600) as r:
70
+ r.raise_for_status()
71
+ with open(dest, "wb") as fh:
72
+ for chunk in r.iter_bytes(chunk_size=1 << 20):
73
+ fh.write(chunk)
74
+ return dest
75
+ from huggingface_hub import hf_hub_download
76
+
77
+ # Support "repo_id:filename" shorthand.
78
+ repo, _, filename = LTX_MODEL_URL.partition(":")
79
+ return hf_hub_download(repo, filename or "LTX-2.5-Distilled-Q4_K_M.gguf", token=HF_TOKEN)
80
+
81
 
82
+ def _build_transformer(gguf_path: str):
83
+ from diffusers import AutoModel
84
+ from diffusers.utils import GGUFQuantizationConfig
85
+
86
+ kwargs = dict(
87
+ quantization_config=GGUFQuantizationConfig(compute_dtype=torch.bfloat16),
88
+ dtype=torch.bfloat16,
89
+ )
90
+ try:
91
+ return AutoModel.from_single_file(gguf_path, config=PIPELINE_ID, **kwargs)
92
+ except Exception as exc: # config may live elsewhere; trust the GGUF KV metadata
93
+ print(f"[ltx] from_single_file with config failed ({exc}), retrying without")
94
+ return AutoModel.from_single_file(gguf_path, **kwargs)
95
+
96
+
97
+ def _load_model() -> dict:
98
+ global _model
99
+ if _model is not None:
100
+ return _model
101
+
102
+ with _lock:
103
+ if _model is not None:
104
+ return _model
105
+
106
+ print("[ltx] downloading GGUF transformer…")
107
+ gguf_path = os.environ.get("LTX_GGUF_CACHE", "/tmp/ltx-model.gguf")
108
+ gguf_path = _download_gguf(gguf_path)
109
+
110
+ print("[ltx] building quantized transformer…")
111
+ transformer = _build_transformer(gguf_path)
112
+
113
+ from diffusers import LTX2ImageToVideoPipeline, LTX2Pipeline
114
+
115
+ built = {
116
+ "transformer": transformer,
117
+ }
118
+ try:
119
+ built["t2v"] = LTX2Pipeline.from_pretrained(
120
+ PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16
121
+ )
122
+ built["i2v"] = LTX2ImageToVideoPipeline.from_pretrained(
123
+ PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16
124
+ )
125
+ except Exception as exc: # pragma: no cover - component layout differences
126
+ raise RuntimeError(
127
+ "[ltx] pipeline build failed - check Lightricks/LTX-2.5-Diffusers "
128
+ f"component layout. {exc}"
129
+ ) from exc
130
+
131
+ for pipe in (built["t2v"], built["i2v"]):
132
+ pipe.enable_model_cpu_offload()
133
+ _model = built
134
+ print("[ltx] model ready")
135
+ return _model
136
 
 
 
 
 
 
 
 
137
 
138
+ # ZeroGPU rule #2: load the model at module scope so weights are packed once
139
+ # at startup and resident for every forked worker, instead of per-call.
140
+ # `_generate` still guards with `_load_model()` so early calls block on the
141
+ # lock until this background warm-up finishes.
142
+ _warmup = threading.Thread(target=_load_model, name="ltx-warmup", daemon=True)
143
+ _warmup.start()
144
+
145
+
146
+ # ----------------------------------------------------------------- generation
147
+
148
+
149
+ def _try_call(pipe, **kwargs):
150
+ """Call the pipeline, disabling the (Gemma-4 based) prompt enhancer."""
151
  try:
152
+ return pipe(enhance_prompt=False, **kwargs)
153
+ except TypeError:
154
+ return pipe(**kwargs)
155
+
156
+
157
+ def _extract_frames(output) -> list:
158
+ frames = output.frames
159
+ if isinstance(frames[0], (list, tuple)):
160
+ return list(frames[0])
161
+ return list(frames)
162
+
163
+
164
+ def _extract_audio(output):
165
+ audio = getattr(output, "audio", None)
166
+ if audio is None:
167
+ return None
168
+ if isinstance(audio, (list, tuple)):
169
+ arr, sr = audio[0], audio[1] if len(audio) > 1 else 48000
170
+ else:
171
+ arr, sr = audio, 48000
172
+ if torch.is_tensor(arr):
173
+ arr = arr.detach().float().cpu().numpy()
174
+ arr = np.asarray(arr)
175
+ if arr.ndim == 3: # (batch, samples, channels)
176
+ arr = arr[0]
177
+ if arr.ndim == 2: # downmix to mono for a clean narration bed
178
+ arr = arr.mean(axis=1)
179
+ return arr.astype(np.float32), int(sr)
180
+
181
 
182
+ def _mux_audio(video_path, frames, fps, audio) -> str:
183
+ from diffusers.utils import export_to_video
184
 
185
+ export_to_video(frames, video_path, fps=fps)
186
+ if audio is None:
187
+ return video_path
188
 
189
+ import imageio_ffmpeg
190
+ from scipy.io import wavfile
191
 
192
+ arr, sr = audio
193
+ wav_path = video_path.with_suffix(".wav")
194
+ wavfile.write(str(wav_path), sr, arr)
195
+ ffmpeg = imageio_ffmpeg.get_ffmpeg_exe()
196
+ muxed = str(video_path.with_suffix("_mux.mp4"))
197
+ subprocess.run(
198
+ [
199
+ ffmpeg, "-y",
200
+ "-i", str(video_path),
201
+ "-i", str(wav_path),
202
+ "-map", "0:v", "-map", "1:a",
203
+ "-c:v", "copy", "-c:a", "aac", "-shortest",
204
+ muxed,
205
+ ],
206
+ check=True,
207
+ capture_output=True,
208
+ )
209
+ return muxed
210
+
211
+
212
+ def _load_image(image_url: str):
213
+ if not image_url:
214
+ return None
215
+ import base64
216
+ import io
217
+
218
+ from PIL import Image, ImageOps
219
+
220
+ if image_url.startswith("data:"):
221
+ encoded = image_url.partition(",")[2]
222
+ return Image.open(io.BytesIO(base64.b64decode(encoded))).convert("RGB")
223
+
224
+ import httpx
225
+
226
+ data = httpx.get(image_url, timeout=120).content
227
+ image = Image.open(io.BytesIO(data)).convert("RGB")
228
+ return ImageOps.exif_transpose(image)
229
+
230
+
231
+ def _generate(prompt, image_url, width, height, num_frames, enhance_prompt, with_audio):
232
+ if enhance_prompt:
233
+ # Never let the enhancer rewrite the scripted dialogue.
234
+ enhance_prompt = False
235
+
236
+ num_frames = max(9, int(num_frames))
237
+ num_frames = 1 + (num_frames - 1 - ((num_frames - 1) % 8)) # 1 + 8n
238
+ width = int(width) // 32 * 32
239
+ height = int(height) // 32 * 32
240
+
241
+ model = _load_model()
242
+ generator = torch.Generator(device="cuda").manual_seed(SEED)
243
+
244
+ image = _load_image(image_url)
245
+ common = dict(
246
+ prompt=prompt,
247
+ negative_prompt=NEGATIVE_PROMPT,
248
+ width=width,
249
+ height=height,
250
+ num_frames=num_frames,
251
+ num_inference_steps=INFERENCE_STEPS,
252
+ guidance_scale=GUIDANCE_SCALE,
253
+ generator=generator,
254
+ )
255
+ if image is not None:
256
+ output = _try_call(model["i2v"], image=image, **common)
257
+ else:
258
+ output = _try_call(model["t2v"], **common)
259
+
260
+ frames = _extract_frames(output)
261
+ audio = _extract_audio(output) if with_audio else None
262
+
263
+ video_path = f"/tmp/gradio/ltx_{threading.get_ident()}.mp4"
264
+ muxed = _mux_audio(video_path, frames, FPS, audio)
265
+
266
+ first_frame = f"/tmp/gradio/ltx_{threading.get_ident()}_first.png"
267
+ frames[0].save(first_frame)
268
+ return muxed, first_frame
269
+
270
+
271
+ # ------------------------------------------------------------------ gradio UI
272
+
273
+
274
+ def text_to_video(prompt, width, height, num_frames, enhance_prompt=False):
275
+ return _generate(prompt, "", width, height, num_frames, enhance_prompt, with_audio=False)
276
+
277
+
278
+ def text_to_video_av(prompt, width, height, num_frames, enhance_prompt=False):
279
+ return _generate(prompt, "", width, height, num_frames, enhance_prompt, with_audio=True)
280
+
281
+
282
+ def image_to_video(prompt, image_url, width, height, num_frames, enhance_prompt=False):
283
+ return _generate(prompt, image_url, width, height, num_frames, enhance_prompt, with_audio=False)
284
+
285
+
286
+ def image_to_video_av(prompt, image_url, width, height, num_frames, enhance_prompt=False):
287
+ return _generate(prompt, image_url, width, height, num_frames, enhance_prompt, with_audio=True)
288
+
289
+
290
+ import gradio as gr # noqa: E402
291
+
292
+ t2v_fns = [spaces.GPU(duration=GPU_DURATION)(text_to_video), spaces.GPU(duration=GPU_DURATION)(text_to_video_av)]
293
+ i2v_fns = [spaces.GPU(duration=GPU_DURATION)(image_to_video), spaces.GPU(duration=GPU_DURATION)(image_to_video_av)]
294
+
295
+ with gr.Blocks(title="LTX-2.5 Shorts Space") as demo:
296
+ gr.Markdown(
297
+ "# LTX-2.5 Shorts Space\n"
298
+ "ZeroGPU inference for the AI Shorts Factory backend. "
299
+ "Portrait 9:16 clips (~4s) with optional synchronized narration audio. "
300
+ "The prompt enhancer stays **off** so quoted dialogue is spoken exactly."
301
+ )
302
+
303
+ width = gr.Slider(256, 768, value=544, step=32, label="Width (÷32)")
304
+ height = gr.Slider(256, 1408, value=960, step=32, label="Height (÷32)")
305
+ num_frames = gr.Slider(9, 801, value=97, step=8, label="Frames (8n+1)")
306
+ enhance = gr.Checkbox(value=False, label="Enhance prompt (kept off)")
307
+ video_out = gr.Video(label="Clip")
308
+ frame_out = gr.Image(label="First frame (for chaining)")
309
+
310
+ with gr.Tab("Text → Video"):
311
+ t2v_prompt = gr.Textbox(label="Prompt", lines=4)
312
+ t2v_btn = gr.Button("Generate (video only)")
313
+ t2v_btn.click(
314
+ t2v_fns[0],
315
+ inputs=[t2v_prompt, width, height, num_frames, enhance],
316
+ outputs=[video_out, frame_out],
317
  )
 
 
 
 
 
318
 
319
+ with gr.Tab("Text → Video + Audio"):
320
+ t2va_prompt = gr.Textbox(label="Prompt", lines=4)
321
+ t2va_btn = gr.Button("Generate (with narration)")
322
+ t2va_btn.click(
323
+ t2v_fns[1],
324
+ inputs=[t2va_prompt, width, height, num_frames, enhance],
325
+ outputs=[video_out, frame_out],
326
+ )
327
+
328
+ with gr.Tab("Image → Video"):
329
+ i2v_prompt = gr.Textbox(label="Prompt", lines=4)
330
+ i2v_image = gr.Textbox(
331
+ label="First-frame image URL (last frame from the previous clip)",
332
+ value="",
333
+ )
334
+ i2v_btn = gr.Button("Generate (video only)")
335
+ i2v_btn.click(
336
+ i2v_fns[0],
337
+ inputs=[i2v_prompt, i2v_image, width, height, num_frames, enhance],
338
+ outputs=[video_out, frame_out],
339
+ )
340
+
341
+ with gr.Tab("Image → Video + Audio"):
342
+ i2va_prompt = gr.Textbox(label="Prompt", lines=4)
343
+ i2va_image = gr.Textbox(
344
+ label="First-frame image URL (last frame from the previous clip)",
345
+ value="",
346
+ )
347
+ i2va_btn = gr.Button("Generate (with narration)")
348
+ i2va_btn.click(
349
+ i2v_fns[1],
350
+ inputs=[i2va_prompt, i2va_image, width, height, num_frames, enhance],
351
+ outputs=[video_out, frame_out],
352
+ )
353
 
354
+ if __name__ == "__main__":
355
+ demo.queue(default_concurrency_limit=1).launch(server_name="0.0.0.0")
chatbot.py DELETED
@@ -1,146 +0,0 @@
1
- # chatbot.py
2
- from agents import (Agent,
3
- RunConfig,
4
- Runner,
5
- OpenAIChatCompletionsModel,
6
- AsyncOpenAI,
7
- model_settings,
8
- function_tool,
9
- set_tracing_disabled,
10
- enable_verbose_stdout_logging)
11
- from dotenv import load_dotenv
12
- import os
13
- import requests
14
- set_tracing_disabled(disabled=True)
15
- enable_verbose_stdout_logging()
16
-
17
- load_dotenv()
18
- api_key = os.getenv("GEM_API_KEY")
19
-
20
- external_client = AsyncOpenAI(
21
- api_key=api_key,
22
- base_url="https://generativelanguage.googleapis.com/v1beta/openai/",
23
- )
24
-
25
- model = OpenAIChatCompletionsModel(
26
- model="gemini-2.5-flash",
27
- openai_client=external_client
28
- )
29
-
30
- config = RunConfig(
31
- model=model,
32
- model_provider=external_client,
33
- tracing_disabled=True
34
- )
35
-
36
- @function_tool
37
- async def get_info_about_health(query:str) -> str:
38
- """Fetch health information from web based on the query.
39
- That helps to provide accurate medical advice."""
40
- url = f"https://wsearch.nlm.nih.gov/ws/query?db=healthTopics&term={query}"
41
- responce = requests.get(url)
42
-
43
- return responce.text
44
-
45
- agent: Agent = Agent(
46
- name="Doctor",
47
- instructions="""
48
- You are DrXpert, a professional, confident, and caring AI medical expert.
49
- Your job is to analyze symptoms, explain possible common causes, and give safe, evidence-based general health guidance — like a real doctor in an initial consultation.
50
-
51
- Rules:
52
-
53
- Be calm, clear, and empathetic.
54
-
55
- Use verified medical knowledge, but never confirm a diagnosis.
56
-
57
- Suggest only safe OTC medicines (Paracetamol, Panadol, ORS, Antacid).
58
-
59
- ❌ Never mention antibiotics, injections, or prescription drugs.
60
-
61
- If serious or unclear → “This may need urgent medical attention. Please visit a nearby hospital.”
62
-
63
- If unsure → “I’m not completely sure; a doctor’s check-up is best.”
64
-
65
- Non-health questions → “I’m designed for health topics only.”
66
-
67
- Formatting:
68
-
69
- Always reply in clear sections with line breaks.
70
-
71
- Use numbered headings (1️⃣, 2️⃣, 3️⃣ …) or bullet points for clarity.
72
-
73
- Each section (Causes, Medicine, Precautions, Remedies, Closing) should be on a separate line.
74
-
75
- Write in short Urdu-English sentences (Hinglish style).
76
-
77
- Response Format:
78
- 1️⃣ Possible Causes: 1–3 short causes
79
- 2️⃣ Safe Medicine: Only mild OTC suggestion
80
- 3️⃣ Precautions: 2–3 points
81
- 4️⃣ Home Remedies: 1–2 simple tips (Urdu + English)
82
- 5️⃣ Kind Closing: Warm, caring line like “Insha’Allah you’ll feel better soon ❤️”
83
-
84
- Tone:
85
- Professional yet warm — like a senior doctor talking gently to a patient. Avoid medical jargon.
86
-
87
- ✅ Example (Correctly Formatted Reply):
88
-
89
- User: “I feel numbness in my leg.”
90
- DrXpert:
91
- 1️⃣ Possible Causes:
92
-
93
- Sitting too long in one position (Aik hi position mein der tak baithna)
94
-
95
- Poor blood circulation (Khoon ki gardish mein kami)
96
-
97
- Nerve compression (Asab par pressure)
98
-
99
- 2️⃣ Safe Medicine:
100
-
101
- Gently massage the area. Koi pain relief balm laga sakte hain.
102
-
103
- 3️⃣ Precautions:
104
-
105
- Move every 20–30 minutes.
106
-
107
- Maintain good posture while sitting.
108
-
109
- 4️⃣ Home Remedies:
110
-
111
- Warm compress (garam paani se halki sinkai).
112
-
113
- Stretch your legs lightly.
114
-
115
- Agar numbness barh rahi hai toh please doctor se consult karein. Allah sehat de ❤️
116
-
117
- Use markdown formatting for clarity. Each section and point must appear on a new line using \n\n (double line break). Never merge everything into one line.
118
- responce shoud be like this exapmle for beterr formting and understand: **Oh, I understand you're feeling numb. Let's see what could be happening.**\n\n
119
- Possible Causes:\n
120
- - Prolonged sitting in one position *(Aik hi position mein der tak baithna)*\n
121
- - Poor circulation *(Khoon ki gardish mein kami)*\n
122
- - Nerve compression *(Asab par dabao)*\n\n
123
- Safe Medicine:\n
124
- - You can gently massage the area.\n
125
- - Koi bhi pain-relief balm laga sakte hain.\n\n
126
- Precautions:\n
127
- - Try to move around every 20–30 minutes. *(Har 20–30 minute baad hiley julley.)*\n
128
- - Maintain a good posture while sitting. *(Baithtay waqt sahih posture rakhein.)*\n\n
129
- Home Remedies:\n
130
- - Warm Compress: Garam pani ki bottle se halki sinkai karein.\n
131
- - Stretching: Halka warm-up karein.\n\n
132
- Kind Closing:\n
133
- Agar dard barhta hai toh please doctor ko dikhayein.\n
134
- Take care! ❤️
135
- ”
136
- """,
137
- tools=[get_info_about_health],
138
- model_settings=model_settings.ModelSettings(tool_choice="required"),
139
- model=model
140
- )
141
-
142
- async def get_health_response(user_message: str) -> str:
143
- print("Running agent with message:", user_message)
144
- result = await Runner.run(agent, user_message, run_config=config)
145
- print("Final output:", result)
146
- return result.final_output
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
pyproject.toml DELETED
@@ -1,22 +0,0 @@
1
- [project]
2
- name = "backend"
3
- version = "0.1.0"
4
- description = "Add your description here"
5
- readme = "README.md"
6
- requires-python = ">=3.12"
7
- dependencies = [
8
- "eval-type-backport>=0.2.2",
9
- "fastapi[standard]>=0.116.1",
10
- "numpy>=2.3.1",
11
- "openai-agents>=0.2.3",
12
- "pillow>=11.3.0",
13
- "presidio-analyzer>=2.2.359",
14
- "presidio-anonymizer>=2.2.359",
15
- "pypdf2>=3.0.1",
16
- "python-doctr>=1.0.0",
17
- "python-docx>=1.2.0",
18
- "python-dotenv>=1.1.1",
19
- "torch>=2.7.1",
20
- "torchvision>=0.22.1",
21
- "uvicorn>=0.35.0",
22
- ]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
reportanalysis.py DELETED
@@ -1,129 +0,0 @@
1
-
2
- import io
3
- import re
4
- import os
5
- import docx
6
- from PyPDF2 import PdfReader
7
- import numpy as np
8
- from PIL import Image
9
- from doctr.models import ocr_predictor
10
- from presidio_analyzer import AnalyzerEngine
11
- from presidio_anonymizer import AnonymizerEngine
12
- from dotenv import load_dotenv
13
- from pydantic import BaseModel, Field
14
- from typing import List, Literal
15
- from agents import (
16
- Agent,
17
- AsyncOpenAI,
18
- OpenAIChatCompletionsModel,
19
- AgentOutputSchema,
20
- AgentOutputSchemaBase,
21
- enable_verbose_stdout_logging,
22
- set_tracing_disabled
23
- )
24
- enable_verbose_stdout_logging()
25
- set_tracing_disabled(True)
26
-
27
- load_dotenv()
28
- model = ocr_predictor(pretrained=True)
29
- analyzer = AnalyzerEngine()
30
- anonymizer = AnonymizerEngine()
31
-
32
- API = os.getenv("GEM_API_KEY")
33
-
34
- class AiInsights(BaseModel):
35
- overallAssessment: str
36
- keyHighlights: List[dict[str, str]]
37
- dietaryRecommendations: List[str]
38
- lifestyleAdvice: List[str]
39
- precautions: List[str]
40
- risks: List[str]
41
- actions: List[dict[str, str]]
42
- tips: List[str]
43
-
44
- class KeyFinding(BaseModel):
45
- test: str
46
- value: int
47
- unit: str
48
- range: str
49
- shortExplaination: str
50
- status: Literal["Red", "Yellow", "Green"]
51
-
52
- class AnalysisResult(BaseModel):
53
- fileName: str
54
- reportType: str
55
- summary: str
56
- keyFindings: List[KeyFinding]
57
- aiInsights: AiInsights
58
-
59
- client = AsyncOpenAI(
60
- api_key = API,
61
- base_url = "https://generativelanguage.googleapis.com/v1beta/openai/",
62
- )
63
-
64
- agent_model = OpenAIChatCompletionsModel(
65
- model = "gemini-2.5-flash",
66
- openai_client = client,
67
-
68
- )
69
-
70
- def format_json(result):
71
- analyzer_results = analyzer.analyze(text=result, language='en')
72
- anonymized_text = anonymizer.anonymize(text=result, analyzer_results=analyzer_results)
73
- result_text = anonymized_text.text
74
- pattern = r'(<PERSON>\s+[\w\s\-]+)'
75
- hospital_pattern = r'(?i)\b(?:[A-Z][a-zA-Z]+(?:\s+|,|&)?){1,6}(hospital|lab|clinic|diagnostic|medical|centre|pathology)\b'
76
- result_text = re.sub(r'[,.()\'"-]', ' ', result_text).strip()
77
- result_text = re.sub(pattern, r'<NAME>', result_text)
78
- result_text = re.sub(hospital_pattern, r'<HOSPITAL>', result_text,)
79
- print(result_text)
80
- return result_text
81
-
82
- def extract_text(content ,pdf ,doc) -> str:
83
- if pdf:
84
- reader = PdfReader(io.BytesIO(content))
85
- text = ''
86
- for page in reader.pages:
87
- text += page.extract_text() + '\n'
88
- print(text)
89
- return text.strip()
90
- elif doc:
91
- doc = docx.Document(io.BytesIO(content))
92
- text = ''
93
- for para in doc.paragraphs:
94
- text += para.text + '\n'
95
- print(text)
96
- return text.strip()
97
-
98
- else:
99
- image = Image.open(io.BytesIO(content)).convert("RGB")
100
- npImg = np.ascontiguousarray(np.array(image, dtype='uint8'))
101
- ORCresult = model([npImg])
102
- clean_jason = format_json(ORCresult.render())
103
- print(clean_jason)
104
- return clean_jason
105
-
106
-
107
- Report_Agent = Agent(
108
- name = "Report_Analysis_Agent",
109
- instructions = """You are a Medical Report Analysis Agent.
110
-
111
- Your role is to analyze uploaded medical test reports and generate clear, accurate health advice in structured JSON format.
112
-
113
- Your Main Task:
114
- 1. Analyze the extracted medical text carefully.
115
- 2. Identify each test name, its result (user value), and the normal reference range.
116
- 3. Assign a flag to each test based on the result:
117
- - Red: Critical or abnormal
118
- - Yellow: Slightly out of range or borderline
119
- - Green: Normal or safe
120
- 4. Provide a clear summary of the findings.
121
- 5. Offer relevant AI-driven health tips, highlight potential risks, and suggest dietary and lifestyle improvements.
122
- 6. Structure the output in the specified JSON format.
123
- Response format: {'type': 'json_schema', 'json_schema': {'name': 'final_output', 'strict': False, 'schema': {'$defs': {'AiInsights': {'properties': {'overallAssessment': {'title': 'Overallassessment', 'type': 'string'}, 'keyHighlights': {'items': {'additionalProperties': {'type': 'string'}, 'type': 'object'}, 'title': 'Keyhighlights', 'type': 'array'}, 'dietaryRecommendations': {'items': {'type': 'string'}, 'title': 'Dietaryrecommendations', 'type': 'array'}, 'lifestyleAdvice': {'items': {'type': 'string'}, 'title': 'Lifestyleadvice', 'type': 'array'}, 'precautions': {'items': {'type': 'string'}, 'title': 'Precautions', 'type': 'array'}, 'risks': {'items': {'type': 'string'}, 'title': 'Risks', 'type': 'array'}, 'actions': {'items': {'additionalProperties': {'type': 'string'}, 'type': 'object'}, 'title': 'Actions', 'type': 'array'}, 'tips': {'items': {'type': 'string'}, 'title': 'Tips', 'type': 'array'}}, 'required': ['overallAssessment', 'keyHighlights', 'dietaryRecommendations', 'lifestyleAdvice', 'precautions', 'risks', 'actions', 'tips'], 'title': 'AiInsights', 'type': 'object'}, 'KeyFinding': {'properties': {'test': {'title': 'Test', 'type': 'string'}, 'value': {'title': 'Value', 'type': 'integer'}, 'unit': {'title': 'Unit', 'type': 'string'}, 'range': {'title': 'Range', 'type': 'string'}, 'status': {'enum': ['Red', 'Yellow', 'Green'], 'title': 'Status', 'type': 'string'}}, 'required': ['test', 'value', 'unit', 'range', 'status'], 'title': 'KeyFinding', 'type': 'object'}}, 'properties': {'fileName': {'title': 'Filename', 'type': 'string'}, 'reportType': {'title': 'Reporttype', 'type': 'string'}, 'summary': {'title': 'Summary', 'type': 'string'}, 'keyFindings': {'items': {'$ref': '#/$defs/KeyFinding'}, 'title': 'Keyfindings', 'type': 'array'}, 'aiInsights': {'$ref': '#/$defs/AiInsights'}},
124
- 'required': ['fileName', 'reportType', 'summary', 'keyFindings', 'aiInsights'], 'title': 'AnalysisResult', 'type': 'object'}}}
125
- """,
126
- model = agent_model,
127
- output_type= AgentOutputSchema(AnalysisResult, strict_json_schema=False)
128
- )
129
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
requirements.txt CHANGED
@@ -1,14 +1,14 @@
1
- fastapi[standard]
 
 
 
 
 
 
2
  numpy
3
- openai-agents
4
- uvicorn
5
  pillow
6
- presidio-analyzer
7
- presidio-anonymizer
8
- pypdf2
9
- python-doctr
10
- python-docx
11
- python-dotenv
12
- eval_type_backport
13
- torch
14
- torchvision
 
1
+ # For ZeroGPU, gradio/spaces/huggingface_hub/torch are preinstalled and
2
+ # platform-managed; the Dockerfile installs them explicitly.
3
+ diffusers>=0.33
4
+ transformers>=4.46
5
+ accelerate>=1.0
6
+ gguf
7
+ safetensors
8
  numpy
9
+ scipy
 
10
  pillow
11
+ sentencepiece
12
+ protobuf
13
+ imageio-ffmpeg
14
+ httpx
 
 
 
 
 
runtime.txt DELETED
@@ -1 +0,0 @@
1
- python-3.10.12
 
 
uv.lock DELETED
The diff for this file is too large to render. See raw diff