imrancoder commited on
Commit
c34af35
·
verified ·
1 Parent(s): d20e8ff

Upload folder using huggingface_hub

Browse files
Files changed (2) hide show
  1. m.py +144 -0
  2. server/PharmaDDIEnv_environment.py +5 -3
m.py ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import asyncio
2
+ import os
3
+ import textwrap
4
+ from typing import List, Optional
5
+
6
+ from openai import OpenAI
7
+
8
+ from my_env_v4 import MyEnvV4Action, MyEnvV4Env
9
+ IMAGE_NAME = os.getenv("IMAGE_NAME") # If you are using docker image
10
+ API_KEY = os.getenv("HF_TOKEN") or os.getenv("API_KEY")
11
+
12
+ API_BASE_URL = os.getenv("API_BASE_URL") or "https://router.huggingface.co/v1"
13
+ MODEL_NAME = os.getenv("MODEL_NAME") or "Qwen/Qwen2.5-72B-Instruct"
14
+ TASK_NAME = os.getenv("MY_ENV_V4_TASK", "echo")
15
+ BENCHMARK = os.getenv("MY_ENV_V4_BENCHMARK", "my_env_v4")
16
+ MAX_STEPS = 8
17
+ TEMPERATURE = 0.7
18
+ MAX_TOKENS = 150
19
+ SUCCESS_SCORE_THRESHOLD = 0.1 # normalized score in [0, 1]
20
+
21
+ # Max possible reward: each token contributes 0.1, across all steps
22
+ _MAX_REWARD_PER_STEP = MAX_TOKENS * 0.1
23
+ MAX_TOTAL_REWARD = MAX_STEPS * _MAX_REWARD_PER_STEP
24
+
25
+ SYSTEM_PROMPT = textwrap.dedent(
26
+ """
27
+ You are interacting with a simple echo environment.
28
+ Each turn you must send a message. The environment will echo it back.
29
+ Reward is proportional to message length: reward = len(message) * 0.1
30
+ Your goal is to maximize total reward by sending meaningful, substantive messages.
31
+ Reply with exactly one message string — no quotes, no prefixes, just the message text.
32
+ """
33
+ ).strip()
34
+
35
+
36
+ def log_start(task: str, env: str, model: str) -> None:
37
+ print(f"[START] task={task} env={env} model={model}", flush=True)
38
+
39
+
40
+ def log_step(step: int, action: str, reward: float, done: bool, error: Optional[str]) -> None:
41
+ error_val = error if error else "null"
42
+ done_val = str(done).lower()
43
+ print(
44
+ f"[STEP] step={step} action={action} reward={reward:.2f} done={done_val} error={error_val}",
45
+ flush=True,
46
+ )
47
+
48
+
49
+ def log_end(success: bool, steps: int, score: float, rewards: List[float]) -> None:
50
+ rewards_str = ",".join(f"{r:.2f}" for r in rewards)
51
+ print(f"[END] success={str(success).lower()} steps={steps} score={score:.3f} rewards={rewards_str}", flush=True)
52
+
53
+
54
+ def build_user_prompt(step: int, last_echoed: str, last_reward: float, history: List[str]) -> str:
55
+ history_block = "\n".join(history[-4:]) if history else "None"
56
+ return textwrap.dedent(
57
+ f"""
58
+ Step: {step}
59
+ Last echoed message: {last_echoed!r}
60
+ Last reward: {last_reward:.2f}
61
+ Previous steps:
62
+ {history_block}
63
+ Send your next message.
64
+ """
65
+ ).strip()
66
+
67
+
68
+ def get_model_message(client: OpenAI, step: int, last_echoed: str, last_reward: float, history: List[str]) -> str:
69
+ user_prompt = build_user_prompt(step, last_echoed, last_reward, history)
70
+ try:
71
+ completion = client.chat.completions.create(
72
+ model=MODEL_NAME,
73
+ messages=[
74
+ {"role": "system", "content": SYSTEM_PROMPT},
75
+ {"role": "user", "content": user_prompt},
76
+ ],
77
+ temperature=TEMPERATURE,
78
+ max_tokens=MAX_TOKENS,
79
+ stream=False,
80
+ )
81
+ text = (completion.choices[0].message.content or "").strip()
82
+ return text if text else "hello"
83
+ except Exception as exc:
84
+ print(f"[DEBUG] Model request failed: {exc}", flush=True)
85
+ return "hello"
86
+
87
+
88
+ async def main() -> None:
89
+ client = OpenAI(base_url=API_BASE_URL, api_key=API_KEY)
90
+
91
+ env = await MyEnvV4Env.from_docker_image(IMAGE_NAME)
92
+
93
+ history: List[str] = []
94
+ rewards: List[float] = []
95
+ steps_taken = 0
96
+ score = 0.0
97
+ success = False
98
+
99
+ log_start(task=TASK_NAME, env=BENCHMARK, model=MODEL_NAME)
100
+
101
+ try:
102
+ result = await env.reset() # OpenENV.reset()
103
+ last_echoed = result.observation.echoed_message
104
+ last_reward = 0.0
105
+
106
+ for step in range(1, MAX_STEPS + 1):
107
+ if result.done:
108
+ break
109
+
110
+ message = get_model_message(client, step, last_echoed, last_reward, history)
111
+
112
+ result = await env.step(MyEnvV4Action(message=message))
113
+ obs = result.observation
114
+
115
+ reward = result.reward or 0.0
116
+ done = result.done
117
+ error = None
118
+
119
+ rewards.append(reward)
120
+ steps_taken = step
121
+ last_echoed = obs.echoed_message
122
+ last_reward = reward
123
+
124
+ log_step(step=step, action=message, reward=reward, done=done, error=error)
125
+
126
+ history.append(f"Step {step}: {message!r} -> reward {reward:+.2f}")
127
+
128
+ if done:
129
+ break
130
+
131
+ score = sum(rewards) / MAX_TOTAL_REWARD if MAX_TOTAL_REWARD > 0 else 0.0
132
+ score = min(max(score, 0.0), 1.0) # clamp to [0, 1]
133
+ success = score >= SUCCESS_SCORE_THRESHOLD
134
+
135
+ finally:
136
+ try:
137
+ await env.close()
138
+ except Exception as e:
139
+ print(f"[DEBUG] env.close() error (container cleanup): {e}", flush=True)
140
+ log_end(success=success, steps=steps_taken, score=score, rewards=rewards)
141
+
142
+
143
+ if __name__ == "__main__":
144
+ asyncio.run(main())
server/PharmaDDIEnv_environment.py CHANGED
@@ -163,7 +163,8 @@ class PharmaDDIEnvironment(Environment):
163
  raw_score, feedback = self._grade_submission(action)
164
 
165
  # Clamp raw_score to avoid exact 0.0 or 1.0 (validation requirement)
166
- clamped_score = min(max(raw_score, 0.01), 0.99)
 
167
 
168
  # Reward = improvement over best score so far
169
  improvement = max(0.0, clamped_score - self._best_score)
@@ -310,8 +311,9 @@ class PharmaDDIEnvironment(Environment):
310
  0.10 * completeness -
311
  critical_penalty
312
  )
313
-
314
- final_score = min(max(raw_score, 0.01), 0.99)
 
315
 
316
  # Build summary feedback
317
  summary = (
 
163
  raw_score, feedback = self._grade_submission(action)
164
 
165
  # Clamp raw_score to avoid exact 0.0 or 1.0 (validation requirement)
166
+ clamped_score = min(max(raw_score, 0.0), 1.0)
167
+
168
 
169
  # Reward = improvement over best score so far
170
  improvement = max(0.0, clamped_score - self._best_score)
 
311
  0.10 * completeness -
312
  critical_penalty
313
  )
314
+ ##############################################################################################################################
315
+ #final_score = min(max(raw_score, 0.01), 0.99)
316
+ final_score = min(max(raw_score, 0.0), 1.0)
317
 
318
  # Build summary feedback
319
  summary = (