ProCreations's picture
Accelerate full 40-step FP8 generation with native precision, measured quality and real-time demo
1081be0 verified
Raw History Blame Contribute Delete
3.91 kB
"""Capture a real wall-clock30s image-only stream, keeping all generation waits."""
import json,time,threading,subprocess
from pathlib import Path
import numpy as np
import torch
from PIL import Image
import imageio_ffmpeg
import sys
sys.path.insert(0,str(Path(__file__).resolve().parents[1]))
from fp8_runtime import load_pipeline
from acceleration import accelerate_pipeline
from prompts import VIDEO
import sys
ROOT=Path(__file__).resolve().parents[1];sys.path.insert(0,str(ROOT))
OUT=Path(__file__).parent/'demo';OUT.mkdir(parents=True,exist_ok=True)
@torch.inference_mode()
def main():
pipe=accelerate_pipeline(load_pipeline('/home/user/models/qwen-image-2.1-b3179ad',ROOT/'release'/'transformer'))
warm=pipe(prompt=VIDEO[0],width=1024,height=1024,num_inference_steps=40,generator=torch.Generator('cuda').manual_seed(50000)).images[0].convert('RGB')
warm.save(OUT/'initial.png')
ffmpeg=imageio_ffmpeg.get_ffmpeg_exe()
cmd=[ffmpeg,'-y','-loglevel','error','-f','rawvideo','-pix_fmt','rgb24','-s','1024x1024','-r','30','-i','-',
'-an','-c:v','libx264','-preset','veryfast','-crf','17','-pix_fmt','yuv420p','-movflags','+faststart',str(OUT/'realtime-30s.mp4')]
encoder=subprocess.Popen(cmd,stdin=subprocess.PIPE)
shared={'frame':np.asarray(warm).tobytes(),'id':'initial'};lock=threading.Lock();ready=threading.Event()
frames=[];events=[];completed_images=[]
start_holder={}
def record():
start=time.perf_counter();start_holder['start']=start;ready.set()
for n in range(900):
deadline=start+n/30
remaining=deadline-time.perf_counter()
if remaining>0:time.sleep(remaining)
with lock:frame=shared['frame'];fid=shared['id']
sampled=time.perf_counter()-start
encoder.stdin.write(frame)
frames.append({'frame':n,'nominal_seconds':n/30,'sampled_seconds':sampled,'image':fid})
encoder.stdin.close()
thread=threading.Thread(target=record);thread.start();ready.wait()
start=start_holder['start']
for i in range(1,10):
if time.perf_counter()-start>=30:break
prompt=VIDEO[i%len(VIDEO)]
before=time.perf_counter()-start
im=pipe(prompt=prompt,width=1024,height=1024,num_inference_steps=40,generator=torch.Generator('cuda').manual_seed(50000+i)).images[0].convert('RGB')
torch.cuda.synchronize();complete=time.perf_counter()-start
payload=np.asarray(im).tobytes()
with lock:shared.update(frame=payload,id=f'image-{i}')
published=time.perf_counter()-start
completed_images.append((f'image-{i}',im))
events.append({'id':f'image-{i}','prompt':prompt,'seed':50000+i,'request_start_seconds':before,
'completed_seconds':complete,'display_update_seconds':published,'seconds':complete-before,'appears_within_video':published<30})
print(json.dumps(events[-1]),flush=True)
thread.join();rc=encoder.wait();assert rc==0
# Save diagnostic PNGs after capture so file compression does not delay requests.
for fid,im in completed_images:im.save(OUT/f'{fid}.png')
lags=[x['sampled_seconds']-x['nominal_seconds'] for x in frames]
(OUT/'capture_receipt.json').write_text(json.dumps({'duration':30,'fps':30,'frames':900,'width':1024,'height':1024,'steps':40,
'diagnostic_png_writes':'Deferred until capture ends; video display updates occur immediately on generation completion.',
'initial_image':'One completed warmup image visible at time0; all subsequent image changes were captured live after actual generation completion. No time compression, overlays, text, transitions or audio.',
'events':events,'max_capture_lag_seconds':max(lags),'frames_log':frames},indent=2))
assert max(lags)<.25,('Capture lag exceeded250ms',max(lags))
print('VIDEO_COMPLETE',max(lags),flush=True)
if __name__=='__main__':main()