Instructions to use decula/sd with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use decula/sd with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-9B") model = PeftModel.from_pretrained(base_model, "decula/sd") - Transformers
How to use decula/sd with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="decula/sd")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("decula/sd", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use decula/sd with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "decula/sd" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "decula/sd", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/decula/sd
- SGLang
How to use decula/sd with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "decula/sd" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "decula/sd", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "decula/sd" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "decula/sd", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use decula/sd with Docker Model Runner:
docker model run hf.co/decula/sd
decula commited on
Commit ·
43a4e1b
1
Parent(s): 70c5d10
modify 3b
Browse files
3b.py
CHANGED
|
@@ -9,8 +9,8 @@ HAS_GPU = False
|
|
| 9 |
|
| 10 |
# Model title and context size limit
|
| 11 |
ctx_limit = 2000
|
| 12 |
-
title = "RWKV-5-World-
|
| 13 |
-
model_file = "rwkv-5-h-world-
|
| 14 |
|
| 15 |
# Get the GPU count
|
| 16 |
try:
|
|
@@ -130,26 +130,33 @@ examples = [
|
|
| 130 |
User: Hello Edward. What have you been up to recently?
|
| 131 |
|
| 132 |
Edward:''', 333, 1, 0.3, 0, 1],
|
| 133 |
-
[generate_prompt("
|
| 134 |
-
['''
|
| 135 |
-
但愿大宇宙能够忽略这个误差。
|
| 136 |
-
程心和关一帆进入了飞船,智子最后也进来了。她早就不再穿那身华丽的和服了,她现在身着迷彩服,再次成为一名轻捷精悍的战士,她的身上佩带着许多武器和生存装备,最引人注目的是那把插在背后的武士刀。
|
| 137 |
-
“放心,我在,你们就在!”智子对两位人类朋友说。
|
| 138 |
-
聚变发动机启动了,推进器发出幽幽的蓝光,飞船缓缓地穿过了宇宙之门。
|
| 139 |
-
小宇宙中只剩下漂流瓶和生态球。漂流瓶隐没于黑暗里,在一千米见方的宇宙中,只有生态球里的小太阳发出一点光芒。在这个小小的生命世界中,几只清澈的水球在零重力环境中静静地飘浮着,有一条小鱼从一只水球中蹦出,跃入另一只水球,轻盈地穿游于绿藻之间。在一小块陆地上的草丛中,有一滴露珠从一片草叶上脱离,旋转着飘起,向太空中折射出一缕晶莹的阳光。''', 333, 1, 0.3, 0, 1],
|
| 140 |
]
|
| 141 |
|
| 142 |
##########################################################################
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 143 |
|
| 144 |
# Gradio blocks
|
| 145 |
with gr.Blocks(title=title) as demo:
|
| 146 |
gr.HTML(f"<div style=\"text-align: center;\">\n<h1>RWKV-5 World v2 - {title}</h1>\n</div>")
|
| 147 |
with gr.Tab("Raw Generation"):
|
| 148 |
-
gr.Markdown(f"This is RWKV-5 World v2 with
|
| 149 |
with gr.Row():
|
| 150 |
with gr.Column():
|
| 151 |
prompt = gr.Textbox(lines=2, label="Prompt", value="")
|
| 152 |
-
token_count = gr.Slider(10,
|
| 153 |
temperature = gr.Slider(0.2, 2.0, label="Temperature", step=0.1, value=1.0)
|
| 154 |
top_p = gr.Slider(0.0, 1.0, label="Top P", step=0.05, value=0.3)
|
| 155 |
presence_penalty = gr.Slider(0.0, 1.0, label="Presence Penalty", step=0.1, value=1)
|
|
@@ -165,4 +172,4 @@ with gr.Blocks(title=title) as demo:
|
|
| 165 |
data.click(lambda x: x, [data], [prompt, token_count, temperature, top_p, presence_penalty, count_penalty])
|
| 166 |
|
| 167 |
# Gradio launch
|
| 168 |
-
demo.launch(share=
|
|
|
|
| 9 |
|
| 10 |
# Model title and context size limit
|
| 11 |
ctx_limit = 2000
|
| 12 |
+
title = "RWKV-5-World-3B-v2-20231025-ctx4096"
|
| 13 |
+
model_file = "rwkv-5-h-world-3B"
|
| 14 |
|
| 15 |
# Get the GPU count
|
| 16 |
try:
|
|
|
|
| 130 |
User: Hello Edward. What have you been up to recently?
|
| 131 |
|
| 132 |
Edward:''', 333, 1, 0.3, 0, 1],
|
| 133 |
+
[generate_prompt(""), 333, 1, 0.3, 0, 1],
|
| 134 |
+
['''''', 333, 1, 0.3, 0, 1],
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 135 |
]
|
| 136 |
|
| 137 |
##########################################################################
|
| 138 |
+
port=7860
|
| 139 |
+
use_frpc=True
|
| 140 |
+
frpconfigfile="7680.ini"
|
| 141 |
+
import subprocess
|
| 142 |
+
|
| 143 |
+
def install_Frpc(port, frpconfigfile, use_frpc):
|
| 144 |
+
if use_frpc:
|
| 145 |
+
subprocess.run(['chmod', '+x', './frpc'], check=True)
|
| 146 |
+
print(f'正在启动frp ,端口{port}')
|
| 147 |
+
subprocess.Popen(['./frpc', '-c', frpconfigfile])
|
| 148 |
+
|
| 149 |
+
install_Frpc('7860',frpconfigfile,use_frpc)
|
| 150 |
|
| 151 |
# Gradio blocks
|
| 152 |
with gr.Blocks(title=title) as demo:
|
| 153 |
gr.HTML(f"<div style=\"text-align: center;\">\n<h1>RWKV-5 World v2 - {title}</h1>\n</div>")
|
| 154 |
with gr.Tab("Raw Generation"):
|
| 155 |
+
gr.Markdown(f"This is RWKV-5 World v2 with 3B params - a 100% attention-free RNN [RWKV-LM](https://github.com/BlinkDL/RWKV-LM). Supports all 100+ world languages and code. And we have [200+ Github RWKV projects](https://github.com/search?o=desc&p=1&q=rwkv&s=updated&type=Repositories). *** Please try examples first (bottom of page) *** (edit them to use your question). Demo limited to ctxlen {ctx_limit}.")
|
| 156 |
with gr.Row():
|
| 157 |
with gr.Column():
|
| 158 |
prompt = gr.Textbox(lines=2, label="Prompt", value="")
|
| 159 |
+
token_count = gr.Slider(10, 10000, label="Max Tokens", step=100, value=10000)
|
| 160 |
temperature = gr.Slider(0.2, 2.0, label="Temperature", step=0.1, value=1.0)
|
| 161 |
top_p = gr.Slider(0.0, 1.0, label="Top P", step=0.05, value=0.3)
|
| 162 |
presence_penalty = gr.Slider(0.0, 1.0, label="Presence Penalty", step=0.1, value=1)
|
|
|
|
| 172 |
data.click(lambda x: x, [data], [prompt, token_count, temperature, top_p, presence_penalty, count_penalty])
|
| 173 |
|
| 174 |
# Gradio launch
|
| 175 |
+
demo.launch(share=False)
|