Promote latest kernel artifacts to main
Browse files- README.md +0 -9
- build/{torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so} +2 -2
- build/torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so +0 -0
- build/torch211-cxx11-cu128-x86_64-linux/_ops.py +3 -3
- build/torch211-cxx11-cu128-x86_64-linux/metadata.json +2 -2
- build/torch211-cxx11-cu130-aarch64-linux/__init__.py +67 -0
- build/{torch212-cxx11-cu132-x86_64-linux → torch211-cxx11-cu130-aarch64-linux}/_flashrt_nvfp4_cuda_c4d802d.abi3.so +2 -2
- build/torch211-cxx11-cu130-aarch64-linux/_ops.py +6 -0
- build/torch211-cxx11-cu130-aarch64-linux/flashrt_nvfp4/__init__.py +14 -0
- build/torch211-cxx11-cu130-aarch64-linux/metadata.json +32 -0
- build/{torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so} +2 -2
- build/torch211-cxx11-cu130-x86_64-linux/_ops.py +3 -3
- build/torch211-cxx11-cu130-x86_64-linux/metadata.json +3 -2
- build/torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so +3 -0
- build/torch212-cxx11-cu130-x86_64-linux/_ops.py +3 -3
- build/torch212-cxx11-cu130-x86_64-linux/metadata.json +3 -2
- build/torch212-cxx11-cu132-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so +3 -0
- build/torch212-cxx11-cu132-x86_64-linux/_ops.py +3 -3
- build/torch212-cxx11-cu132-x86_64-linux/metadata.json +3 -2
README.md
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
# flashrt/flashrt-nvfp4
|
| 2 |
-
|
| 3 |
-
This repository is a compatibility mirror for older `kernels` clients
|
| 4 |
-
that resolve repositories through the default Hugging Face model repo API.
|
| 5 |
-
|
| 6 |
-
Canonical Kernel Hub repo: https://huggingface.co/kernels/flashrt/flashrt-nvfp4
|
| 7 |
-
|
| 8 |
-
Do not edit this mirror by hand. It is generated from the Kernel Hub
|
| 9 |
-
`vN` branches and contains the same `build/**` artifacts.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
build/{torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so}
RENAMED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1c2e72487799c2955fffa6ae64245e6a690211a52c1abde76f5f7e189fca764
|
| 3 |
+
size 95168
|
build/torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so
DELETED
|
Binary file (95.2 kB)
|
|
|
build/torch211-cxx11-cu128-x86_64-linux/_ops.py
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
import torch
|
| 2 |
-
from . import
|
| 3 |
-
ops = torch.ops.
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
-
return f"
|
|
|
|
| 1 |
import torch
|
| 2 |
+
from . import _flashrt_nvfp4_cuda_9e33734
|
| 3 |
+
ops = torch.ops._flashrt_nvfp4_cuda_9e33734
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
+
return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
|
build/torch211-cxx11-cu128-x86_64-linux/metadata.json
CHANGED
|
@@ -1,13 +1,13 @@
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
-
"id": "
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
-
"12.
|
| 11 |
]
|
| 12 |
}
|
| 13 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
+
"id": "_flashrt_nvfp4_cuda_9e33734",
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
+
"12.0a"
|
| 11 |
]
|
| 12 |
}
|
| 13 |
}
|
build/torch211-cxx11-cu130-aarch64-linux/__init__.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""FlashRT NVFP4 layout kernels."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
from typing import Optional
|
| 6 |
+
|
| 7 |
+
import torch
|
| 8 |
+
|
| 9 |
+
from ._ops import add_op_namespace_prefix, ops
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
@torch.library.register_fake(add_op_namespace_prefix("nvfp4_sf_linear_to_swizzled"))
|
| 13 |
+
def _nvfp4_sf_linear_to_swizzled_fake(
|
| 14 |
+
scales: torch.Tensor,
|
| 15 |
+
out: torch.Tensor,
|
| 16 |
+
D: int,
|
| 17 |
+
is_sfb: bool = False,
|
| 18 |
+
) -> None:
|
| 19 |
+
if scales.dim() != 2:
|
| 20 |
+
raise RuntimeError("scales must have shape (rows, D / 16)")
|
| 21 |
+
return None
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def nvfp4_sf_swizzled_bytes(rows: int, D: int) -> int:
|
| 25 |
+
"""Return byte count for a CUTLASS Sm1xx NVFP4 swizzled SF buffer."""
|
| 26 |
+
|
| 27 |
+
if rows <= 0:
|
| 28 |
+
raise ValueError("rows must be positive")
|
| 29 |
+
if D <= 0 or D % 16 != 0:
|
| 30 |
+
raise ValueError("D must be positive and divisible by 16")
|
| 31 |
+
n_blocks = D // 16
|
| 32 |
+
n_row_super = (rows + 127) // 128
|
| 33 |
+
n_col_super = (n_blocks + 3) // 4
|
| 34 |
+
return n_row_super * n_col_super * 512
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def nvfp4_sf_linear_to_swizzled(
|
| 38 |
+
scales: torch.Tensor,
|
| 39 |
+
*,
|
| 40 |
+
out: Optional[torch.Tensor] = None,
|
| 41 |
+
is_sfb: bool = False,
|
| 42 |
+
) -> torch.Tensor:
|
| 43 |
+
"""Convert linear NVFP4 scale bytes to CUTLASS Sm1xx swizzled layout.
|
| 44 |
+
|
| 45 |
+
``scales`` must be contiguous CUDA ``torch.uint8`` with shape
|
| 46 |
+
``(rows, D / 16)``. If ``out`` is omitted, a flat ``torch.uint8`` output
|
| 47 |
+
tensor with ``nvfp4_sf_swizzled_bytes(rows, D)`` bytes is allocated.
|
| 48 |
+
"""
|
| 49 |
+
|
| 50 |
+
if scales.dim() != 2:
|
| 51 |
+
raise ValueError("scales must have shape (rows, D / 16)")
|
| 52 |
+
rows = scales.shape[0]
|
| 53 |
+
D = scales.shape[1] * 16
|
| 54 |
+
if out is None:
|
| 55 |
+
out = torch.zeros(
|
| 56 |
+
(nvfp4_sf_swizzled_bytes(rows, D),),
|
| 57 |
+
device=scales.device,
|
| 58 |
+
dtype=torch.uint8,
|
| 59 |
+
)
|
| 60 |
+
ops.nvfp4_sf_linear_to_swizzled(scales, out, D, is_sfb)
|
| 61 |
+
return out
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
__all__ = [
|
| 65 |
+
"nvfp4_sf_linear_to_swizzled",
|
| 66 |
+
"nvfp4_sf_swizzled_bytes",
|
| 67 |
+
]
|
build/{torch212-cxx11-cu132-x86_64-linux → torch211-cxx11-cu130-aarch64-linux}/_flashrt_nvfp4_cuda_c4d802d.abi3.so
RENAMED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1c206040d7bd2911cca52e0c788a9c981e8161991db90f2870c906d210fdc628
|
| 3 |
+
size 167824
|
build/torch211-cxx11-cu130-aarch64-linux/_ops.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from . import _flashrt_nvfp4_cuda_c4d802d
|
| 3 |
+
ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
|
| 4 |
+
|
| 5 |
+
def add_op_namespace_prefix(op_name: str):
|
| 6 |
+
return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
|
build/torch211-cxx11-cu130-aarch64-linux/flashrt_nvfp4/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import ctypes
|
| 2 |
+
import importlib.util
|
| 3 |
+
import sys
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
def _import_from_path(file_path: Path):
|
| 7 |
+
path_hash = '{:x}'.format(ctypes.c_size_t(hash(file_path.absolute())).value)
|
| 8 |
+
spec = importlib.util.spec_from_file_location(path_hash, file_path)
|
| 9 |
+
module = importlib.util.module_from_spec(spec)
|
| 10 |
+
sys.modules[path_hash] = module
|
| 11 |
+
spec.loader.exec_module(module)
|
| 12 |
+
return module
|
| 13 |
+
|
| 14 |
+
globals().update(vars(_import_from_path(Path(__file__).parent.parent / '__init__.py')))
|
build/torch211-cxx11-cu130-aarch64-linux/metadata.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "flashrt-nvfp4",
|
| 3 |
+
"id": "_flashrt_nvfp4_cuda_c4d802d",
|
| 4 |
+
"version": 1,
|
| 5 |
+
"license": "Apache-2.0",
|
| 6 |
+
"python-depends": [],
|
| 7 |
+
"backend": {
|
| 8 |
+
"type": "cuda",
|
| 9 |
+
"archs": [
|
| 10 |
+
"11.0a"
|
| 11 |
+
]
|
| 12 |
+
},
|
| 13 |
+
"digest": {
|
| 14 |
+
"algorithm": "sha256",
|
| 15 |
+
"files": {
|
| 16 |
+
"__init__.py": "aB242czTWLzSGSTrkJY75ln0cEILoGwF8bHScJPgvMc=",
|
| 17 |
+
"_flashrt_nvfp4_cuda_c4d802d.abi3.so": "HCBgQNe9KRHMpS4MeIqcmB6BYZkduQ8ocMkG0hD9xig=",
|
| 18 |
+
"_ops.py": "9FbAFa8lhnvONtloChdas+yY4tDVqcicGsvmPNAKMr4=",
|
| 19 |
+
"flashrt_nvfp4/__init__.py": "v6p5XMfQzddhi1fLSAw4HX9CyS0rQsidvu9VsT01xi4="
|
| 20 |
+
}
|
| 21 |
+
},
|
| 22 |
+
"provenance": {
|
| 23 |
+
"kernel": {
|
| 24 |
+
"sha": "9e33734",
|
| 25 |
+
"dirty": false
|
| 26 |
+
},
|
| 27 |
+
"validation": {
|
| 28 |
+
"torch": "2.11.0+cu130",
|
| 29 |
+
"cuda": "13.0"
|
| 30 |
+
}
|
| 31 |
+
}
|
| 32 |
+
}
|
build/{torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so}
RENAMED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:17c97cca9d88bd01b62199b514c722bb955c23a6290f40c875aa6a2eb31d9aa2
|
| 3 |
+
size 104424
|
build/torch211-cxx11-cu130-x86_64-linux/_ops.py
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
import torch
|
| 2 |
-
from . import
|
| 3 |
-
ops = torch.ops.
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
-
return f"
|
|
|
|
| 1 |
import torch
|
| 2 |
+
from . import _flashrt_nvfp4_cuda_9e33734
|
| 3 |
+
ops = torch.ops._flashrt_nvfp4_cuda_9e33734
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
+
return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
|
build/torch211-cxx11-cu130-x86_64-linux/metadata.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
-
"id": "
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
-
"
|
|
|
|
| 11 |
]
|
| 12 |
}
|
| 13 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
+
"id": "_flashrt_nvfp4_cuda_9e33734",
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
+
"11.0a",
|
| 11 |
+
"12.0a"
|
| 12 |
]
|
| 13 |
}
|
| 14 |
}
|
build/torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3bc3ea1097842b95184b411c5ecad33a36e1d289a1ef65c67a60c48a71229220
|
| 3 |
+
size 119280
|
build/torch212-cxx11-cu130-x86_64-linux/_ops.py
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
import torch
|
| 2 |
-
from . import
|
| 3 |
-
ops = torch.ops.
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
-
return f"
|
|
|
|
| 1 |
import torch
|
| 2 |
+
from . import _flashrt_nvfp4_cuda_9e33734
|
| 3 |
+
ops = torch.ops._flashrt_nvfp4_cuda_9e33734
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
+
return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
|
build/torch212-cxx11-cu130-x86_64-linux/metadata.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
-
"id": "
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
-
"
|
|
|
|
| 11 |
]
|
| 12 |
}
|
| 13 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
+
"id": "_flashrt_nvfp4_cuda_9e33734",
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
+
"11.0a",
|
| 11 |
+
"12.0a"
|
| 12 |
]
|
| 13 |
}
|
| 14 |
}
|
build/torch212-cxx11-cu132-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bf1db69d0f2ea2d97282d2999ec69cb339559e855eae29d2009463a50237fe20
|
| 3 |
+
size 119280
|
build/torch212-cxx11-cu132-x86_64-linux/_ops.py
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
import torch
|
| 2 |
-
from . import
|
| 3 |
-
ops = torch.ops.
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
-
return f"
|
|
|
|
| 1 |
import torch
|
| 2 |
+
from . import _flashrt_nvfp4_cuda_9e33734
|
| 3 |
+
ops = torch.ops._flashrt_nvfp4_cuda_9e33734
|
| 4 |
|
| 5 |
def add_op_namespace_prefix(op_name: str):
|
| 6 |
"""
|
| 7 |
Prefix op by namespace.
|
| 8 |
"""
|
| 9 |
+
return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
|
build/torch212-cxx11-cu132-x86_64-linux/metadata.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
-
"id": "
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
-
"
|
|
|
|
| 11 |
]
|
| 12 |
}
|
| 13 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"name": "flashrt-nvfp4",
|
| 3 |
+
"id": "_flashrt_nvfp4_cuda_9e33734",
|
| 4 |
"version": 1,
|
| 5 |
"license": "Apache-2.0",
|
| 6 |
"python-depends": [],
|
| 7 |
"backend": {
|
| 8 |
"type": "cuda",
|
| 9 |
"archs": [
|
| 10 |
+
"11.0a",
|
| 11 |
+
"12.0a"
|
| 12 |
]
|
| 13 |
}
|
| 14 |
}
|