Eval v6: restore _fetch_first_rows
Browse files- eval_securecoder.py +17 -0
eval_securecoder.py
CHANGED
|
@@ -178,6 +178,23 @@ def _messages_from_any(row, kind):
|
|
| 178 |
messages.append({"role": "user", "content": str(content)})
|
| 179 |
break
|
| 180 |
return messages, tools
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 181 |
def _build_tool_prompts(rows: list[dict]) -> list[dict]:
|
| 182 |
"""Build (prompt, tools) tuples via the shared _messages_from_any helper.
|
| 183 |
Hermes rows use ``conversations`` (from/value); OpenAI-style use ``messages``
|
|
|
|
| 178 |
messages.append({"role": "user", "content": str(content)})
|
| 179 |
break
|
| 180 |
return messages, tools
|
| 181 |
+
def _fetch_first_rows(repo: str, config: str | None, split: str, n: int) -> list[dict]:
|
| 182 |
+
"""Stream up to ``n`` rows from a Hub dataset. ``config`` is the dataset config
|
| 183 |
+
name (e.g. ``func_calling`` for Hermes FC); pass ``None`` for datasets that
|
| 184 |
+
have only the default config."""
|
| 185 |
+
from datasets import load_dataset
|
| 186 |
+
kwargs: dict[str, Any] = {"split": split, "streaming": True}
|
| 187 |
+
if config:
|
| 188 |
+
kwargs["name"] = config
|
| 189 |
+
ds = load_dataset(repo, token=os.environ.get("HF_TOKEN"), **kwargs)
|
| 190 |
+
out = []
|
| 191 |
+
for row in ds:
|
| 192 |
+
out.append(dict(row))
|
| 193 |
+
if len(out) >= n:
|
| 194 |
+
break
|
| 195 |
+
return out
|
| 196 |
+
|
| 197 |
+
|
| 198 |
def _build_tool_prompts(rows: list[dict]) -> list[dict]:
|
| 199 |
"""Build (prompt, tools) tuples via the shared _messages_from_any helper.
|
| 200 |
Hermes rows use ``conversations`` (from/value); OpenAI-style use ``messages``
|