Spaces:
Running
Running
Download tests/test_latency.py from ArchitSharma/InferScale-Sim: direct link, hf CLI and curl.
- Browser
- Download file 589 Bytes
-
https://huggingface.co/spaces/ArchitSharma/InferScale-Sim/resolve/main/tests/test_latency.py
- Command line
-
hf download hf://spaces/ArchitSharma/InferScale-Sim/tests/test_latency.py
-
curl -L -o test_latency.py https://huggingface.co/spaces/ArchitSharma/InferScale-Sim/resolve/main/tests/test_latency.py
589 Bytes
| from inferscale.latency import AnalyticalLatencyModel | |
| from inferscale.profiles import get_accelerator, get_model | |
| def test_prefill_grows_with_tokens(): | |
| lm = AnalyticalLatencyModel(get_model("Llama-3.1-8B"), get_accelerator("L4")) | |
| assert lm.prefill_seconds([1024]) > lm.prefill_seconds([128]) | |
| def test_quantization_reduces_weight_memory(): | |
| fp16 = AnalyticalLatencyModel(get_model("Llama-3.1-8B"), get_accelerator("L4"), "fp16") | |
| int4 = AnalyticalLatencyModel(get_model("Llama-3.1-8B"), get_accelerator("L4"), "int4") | |
| assert int4.model_weight_gb < fp16.model_weight_gb | |