liangsu9988 commited on
Commit
e018baf
·
verified ·
1 Parent(s): 17bf6d5

Promote latest kernel artifacts to main

Browse files
README.md DELETED
@@ -1,9 +0,0 @@
1
- # flashrt/flashrt-nvfp4
2
-
3
- This repository is a compatibility mirror for older `kernels` clients
4
- that resolve repositories through the default Hugging Face model repo API.
5
-
6
- Canonical Kernel Hub repo: https://huggingface.co/kernels/flashrt/flashrt-nvfp4
7
-
8
- Do not edit this mirror by hand. It is generated from the Kernel Hub
9
- `vN` branches and contains the same `build/**` artifacts.
 
 
 
 
 
 
 
 
 
 
build/{torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2d8e6c0b823b0bebe75d49f7a9bd4806378107c013e043df41ca3d587c55e913
3
- size 100344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1c2e72487799c2955fffa6ae64245e6a690211a52c1abde76f5f7e189fca764
3
+ size 95168
build/torch211-cxx11-cu128-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so DELETED
Binary file (95.2 kB)
 
build/torch211-cxx11-cu128-x86_64-linux/_ops.py CHANGED
@@ -1,9 +1,9 @@
1
  import torch
2
- from . import _flashrt_nvfp4_cuda_c4d802d
3
- ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
- return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
 
1
  import torch
2
+ from . import _flashrt_nvfp4_cuda_9e33734
3
+ ops = torch.ops._flashrt_nvfp4_cuda_9e33734
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
+ return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
build/torch211-cxx11-cu128-x86_64-linux/metadata.json CHANGED
@@ -1,13 +1,13 @@
1
  {
2
  "name": "flashrt-nvfp4",
3
- "id": "_flashrt_nvfp4_cuda_c4d802d",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
- "12.0"
11
  ]
12
  }
13
  }
 
1
  {
2
  "name": "flashrt-nvfp4",
3
+ "id": "_flashrt_nvfp4_cuda_9e33734",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
+ "12.0a"
11
  ]
12
  }
13
  }
build/torch211-cxx11-cu130-aarch64-linux/__init__.py ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """FlashRT NVFP4 layout kernels."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional
6
+
7
+ import torch
8
+
9
+ from ._ops import add_op_namespace_prefix, ops
10
+
11
+
12
+ @torch.library.register_fake(add_op_namespace_prefix("nvfp4_sf_linear_to_swizzled"))
13
+ def _nvfp4_sf_linear_to_swizzled_fake(
14
+ scales: torch.Tensor,
15
+ out: torch.Tensor,
16
+ D: int,
17
+ is_sfb: bool = False,
18
+ ) -> None:
19
+ if scales.dim() != 2:
20
+ raise RuntimeError("scales must have shape (rows, D / 16)")
21
+ return None
22
+
23
+
24
+ def nvfp4_sf_swizzled_bytes(rows: int, D: int) -> int:
25
+ """Return byte count for a CUTLASS Sm1xx NVFP4 swizzled SF buffer."""
26
+
27
+ if rows <= 0:
28
+ raise ValueError("rows must be positive")
29
+ if D <= 0 or D % 16 != 0:
30
+ raise ValueError("D must be positive and divisible by 16")
31
+ n_blocks = D // 16
32
+ n_row_super = (rows + 127) // 128
33
+ n_col_super = (n_blocks + 3) // 4
34
+ return n_row_super * n_col_super * 512
35
+
36
+
37
+ def nvfp4_sf_linear_to_swizzled(
38
+ scales: torch.Tensor,
39
+ *,
40
+ out: Optional[torch.Tensor] = None,
41
+ is_sfb: bool = False,
42
+ ) -> torch.Tensor:
43
+ """Convert linear NVFP4 scale bytes to CUTLASS Sm1xx swizzled layout.
44
+
45
+ ``scales`` must be contiguous CUDA ``torch.uint8`` with shape
46
+ ``(rows, D / 16)``. If ``out`` is omitted, a flat ``torch.uint8`` output
47
+ tensor with ``nvfp4_sf_swizzled_bytes(rows, D)`` bytes is allocated.
48
+ """
49
+
50
+ if scales.dim() != 2:
51
+ raise ValueError("scales must have shape (rows, D / 16)")
52
+ rows = scales.shape[0]
53
+ D = scales.shape[1] * 16
54
+ if out is None:
55
+ out = torch.zeros(
56
+ (nvfp4_sf_swizzled_bytes(rows, D),),
57
+ device=scales.device,
58
+ dtype=torch.uint8,
59
+ )
60
+ ops.nvfp4_sf_linear_to_swizzled(scales, out, D, is_sfb)
61
+ return out
62
+
63
+
64
+ __all__ = [
65
+ "nvfp4_sf_linear_to_swizzled",
66
+ "nvfp4_sf_swizzled_bytes",
67
+ ]
build/{torch212-cxx11-cu132-x86_64-linux → torch211-cxx11-cu130-aarch64-linux}/_flashrt_nvfp4_cuda_c4d802d.abi3.so RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:91f6297b7871d47faf300ef1308b9a340666ab3afaa6b46b9a2d4d314b77255f
3
- size 111104
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1c206040d7bd2911cca52e0c788a9c981e8161991db90f2870c906d210fdc628
3
+ size 167824
build/torch211-cxx11-cu130-aarch64-linux/_ops.py ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ import torch
2
+ from . import _flashrt_nvfp4_cuda_c4d802d
3
+ ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
4
+
5
+ def add_op_namespace_prefix(op_name: str):
6
+ return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
build/torch211-cxx11-cu130-aarch64-linux/flashrt_nvfp4/__init__.py ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import ctypes
2
+ import importlib.util
3
+ import sys
4
+ from pathlib import Path
5
+
6
+ def _import_from_path(file_path: Path):
7
+ path_hash = '{:x}'.format(ctypes.c_size_t(hash(file_path.absolute())).value)
8
+ spec = importlib.util.spec_from_file_location(path_hash, file_path)
9
+ module = importlib.util.module_from_spec(spec)
10
+ sys.modules[path_hash] = module
11
+ spec.loader.exec_module(module)
12
+ return module
13
+
14
+ globals().update(vars(_import_from_path(Path(__file__).parent.parent / '__init__.py')))
build/torch211-cxx11-cu130-aarch64-linux/metadata.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "flashrt-nvfp4",
3
+ "id": "_flashrt_nvfp4_cuda_c4d802d",
4
+ "version": 1,
5
+ "license": "Apache-2.0",
6
+ "python-depends": [],
7
+ "backend": {
8
+ "type": "cuda",
9
+ "archs": [
10
+ "11.0a"
11
+ ]
12
+ },
13
+ "digest": {
14
+ "algorithm": "sha256",
15
+ "files": {
16
+ "__init__.py": "aB242czTWLzSGSTrkJY75ln0cEILoGwF8bHScJPgvMc=",
17
+ "_flashrt_nvfp4_cuda_c4d802d.abi3.so": "HCBgQNe9KRHMpS4MeIqcmB6BYZkduQ8ocMkG0hD9xig=",
18
+ "_ops.py": "9FbAFa8lhnvONtloChdas+yY4tDVqcicGsvmPNAKMr4=",
19
+ "flashrt_nvfp4/__init__.py": "v6p5XMfQzddhi1fLSAw4HX9CyS0rQsidvu9VsT01xi4="
20
+ }
21
+ },
22
+ "provenance": {
23
+ "kernel": {
24
+ "sha": "9e33734",
25
+ "dirty": false
26
+ },
27
+ "validation": {
28
+ "torch": "2.11.0+cu130",
29
+ "cuda": "13.0"
30
+ }
31
+ }
32
+ }
build/{torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_c4d802d.abi3.so → torch211-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8b0a24f8bbe4406e273fb631b33cabc777ef02ee23b23f1be618b3441bdc2ec9
3
- size 111104
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:17c97cca9d88bd01b62199b514c722bb955c23a6290f40c875aa6a2eb31d9aa2
3
+ size 104424
build/torch211-cxx11-cu130-x86_64-linux/_ops.py CHANGED
@@ -1,9 +1,9 @@
1
  import torch
2
- from . import _flashrt_nvfp4_cuda_c4d802d
3
- ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
- return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
 
1
  import torch
2
+ from . import _flashrt_nvfp4_cuda_9e33734
3
+ ops = torch.ops._flashrt_nvfp4_cuda_9e33734
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
+ return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
build/torch211-cxx11-cu130-x86_64-linux/metadata.json CHANGED
@@ -1,13 +1,14 @@
1
  {
2
  "name": "flashrt-nvfp4",
3
- "id": "_flashrt_nvfp4_cuda_c4d802d",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
- "12.0"
 
11
  ]
12
  }
13
  }
 
1
  {
2
  "name": "flashrt-nvfp4",
3
+ "id": "_flashrt_nvfp4_cuda_9e33734",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
+ "11.0a",
11
+ "12.0a"
12
  ]
13
  }
14
  }
build/torch212-cxx11-cu130-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3bc3ea1097842b95184b411c5ecad33a36e1d289a1ef65c67a60c48a71229220
3
+ size 119280
build/torch212-cxx11-cu130-x86_64-linux/_ops.py CHANGED
@@ -1,9 +1,9 @@
1
  import torch
2
- from . import _flashrt_nvfp4_cuda_c4d802d
3
- ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
- return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
 
1
  import torch
2
+ from . import _flashrt_nvfp4_cuda_9e33734
3
+ ops = torch.ops._flashrt_nvfp4_cuda_9e33734
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
+ return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
build/torch212-cxx11-cu130-x86_64-linux/metadata.json CHANGED
@@ -1,13 +1,14 @@
1
  {
2
  "name": "flashrt-nvfp4",
3
- "id": "_flashrt_nvfp4_cuda_c4d802d",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
- "12.0"
 
11
  ]
12
  }
13
  }
 
1
  {
2
  "name": "flashrt-nvfp4",
3
+ "id": "_flashrt_nvfp4_cuda_9e33734",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
+ "11.0a",
11
+ "12.0a"
12
  ]
13
  }
14
  }
build/torch212-cxx11-cu132-x86_64-linux/_flashrt_nvfp4_cuda_9e33734.abi3.so ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bf1db69d0f2ea2d97282d2999ec69cb339559e855eae29d2009463a50237fe20
3
+ size 119280
build/torch212-cxx11-cu132-x86_64-linux/_ops.py CHANGED
@@ -1,9 +1,9 @@
1
  import torch
2
- from . import _flashrt_nvfp4_cuda_c4d802d
3
- ops = torch.ops._flashrt_nvfp4_cuda_c4d802d
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
- return f"_flashrt_nvfp4_cuda_c4d802d::{op_name}"
 
1
  import torch
2
+ from . import _flashrt_nvfp4_cuda_9e33734
3
+ ops = torch.ops._flashrt_nvfp4_cuda_9e33734
4
 
5
  def add_op_namespace_prefix(op_name: str):
6
  """
7
  Prefix op by namespace.
8
  """
9
+ return f"_flashrt_nvfp4_cuda_9e33734::{op_name}"
build/torch212-cxx11-cu132-x86_64-linux/metadata.json CHANGED
@@ -1,13 +1,14 @@
1
  {
2
  "name": "flashrt-nvfp4",
3
- "id": "_flashrt_nvfp4_cuda_c4d802d",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
- "12.0"
 
11
  ]
12
  }
13
  }
 
1
  {
2
  "name": "flashrt-nvfp4",
3
+ "id": "_flashrt_nvfp4_cuda_9e33734",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "python-depends": [],
7
  "backend": {
8
  "type": "cuda",
9
  "archs": [
10
+ "11.0a",
11
+ "12.0a"
12
  ]
13
  }
14
  }