Squash of 54 work commits (Sep 1–12):6251f7cai-hub: body domain — live pose packets ride the realtime sessionea50c77chat_ui: the feed's session gets its profile brief backf51b5f3ai-body: the crate for the native SAM 3D Body port, with its weights reader8211ae6ai-body: the MHR rig and the pose head's parameter decoding, oracle-exact9e343a8ai-body: the DINOv3 ViT-H+/16 backbone, crop and ray conditioning; Metal gains rope-half and affine layer norm69d842cai-body: the promptable pose decoder and its refinement loop, oracle-matched on Metal66e5e2fai-hub: SAM 3D Body runs natively — `sam3dbody` on the body domain, oracle-matched end to enda634198ai-hub: the body-native commit carried a peer's in-flight hub hunks; put them back where they were9ff44e8ai-hub: the body-native wiring, this time only the lane's hunks6a1c16bai-body: third-party notices — what the port is implemented after, and what it is notd78411aai-body: the per-step work moves to the GPUb22259bai-body: the context stays on the GPU; only the pose token leaves the loop346f31fai-body: flash attention for the head-dim-64 blocks45b5b98ai-body: the crop size is a runtime knob, and the loop reports where its time goes4be6d19ai-body: the test modules import the grid constants they still use7598346ai-body: tensor-core GEMMs for the backbone, and the rig's correctives only where they counta9ce596ai-body: the crop warp runs across cores8964ba6ai-body: an FP8 backbone mode, off by default, measured against the oraclea2aaa8fai-body: the FP8 bias rides a column-broadcast add on the deviced53c77dmetal: a device-resident ViT stack, and the body backbone rides itd006d0ametal: resident f32 linears keep their weight on the device525ba1cmetal: a device-resident two-way decoder layer, and the body decoder rides itc9e6d88ai-body: the hands pass — hand crops, the hand decoder, the hand-mode rig and the wrist fusion62dff26ai-body: the mask prompt — a person's segmentation mask conditions the body passa648cf8ai-hub: body session options — hands, detect, persons=N8c568dfai-hub: drop the SAM 3D Body reference worker backend7ff875aai-hub: keep a peer's in-flight beats/notes/local work out of the body commits31e5faaai-hub: local model runner, licence acknowledgements, a shared install panel; Beat This!, Basic Pitch and the Salamander drum-kit entriesb94bc58ai-services: the wire, the app port and the panel state — one conversation, many apps2acb798ai-services: wire v2 — endpoints, receiver-side caps, result disposition8ae0ffbai-services: the engine core — registry, router and conversation, tested against a scripted model2308736ai-services: the real models behind the engine feature — local through the hub, Claude, and nonec3f631dlivepipe: one reusable pipe from a camera to a fleet node and backff62db3ai libs: the runtime env-var cleanup — precision is a per-caller policy, not an environment side channel04a94efrealtime: one service-log line when a live session opens and one when it closes0ecb81cai models: the model-crates env-var cleanup — 172 research knobs gone, the unset default is the code4ca36c1ai hub + services: the assistant's model comes from wherever it is resident — the fleet chat box, with tools, then the local weights432121eaichat engine + wm: launch, then use — the assistant continues in the same turn once the app it started is on the bus7a5bf69ai-hub registry: the Salamander drumkit samples come from the makepad.nl mirror — the GitHub repo only carries the .sfz files102ffc5ai-services: messages on the bus — a manifest declares topics, the engine subscribes on a tool's behalf or by ToolResult.subscribe, a service publishes Message frames, an idle conversation wakes on a message as an event turn under rate laws; the WM bus forwards the new frames; every app that matches the wire gets its arma837792hub + flow: a whitespace-only chat completion is retried once and then fails instead of passing as an answer; a flow's model is a fleet model id unless it names a weight file on disk; chat models show under the text domain in /v1/modelsbc6c620hub + flow: what the chat review found — the in-process route retries an empty completion too, a node says whether its prefill opened thinking so a brief-mode answer is never discarded, a preferred model falls back to normal election when no node has it, discovery keeps looking for the preferred model until patience runs out75c3441hub: the PRO 6000 serves image as well as chat and textad5e98bhub registry: flux2-dev's VRAM estimate is its measured peak, 30 GBc7241e0hub: a node that evicted every resident releases its cached allocator pool before refusing a load or publishing usable VRAM30575f0flow: route generation by request workload1be1e21ai-hub: gate downloads by disk capacity and recover fleet admission df6b394 filesystem_watcher, bounded_http, ai services: live and tool prerequisites 79ebdb9 ai-hub: add a native Pixal3D image-to-3D backend 0ba0d74 ai-hub: propagate typed refusals under reject queue policy cc6c872 Speed up H3 conditioning and video decoding e512059 Fix Qwen vision residency and generated material colors 2864f68 ai-hub http client: bound every plain TCP connect to 3 s per address 3d93229 ai: CUDA is a Linux/Windows-only dependency; the hub library defaults to llm + stt Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
86 lines
4 KiB
Python
86 lines
4 KiB
Python
"""Validate the native CUDA NAF sampler against dense PyTorch operations.
|
|
|
|
Build kernels/pixal.cu with nvcc -shared -O3, then pass the resulting library.
|
|
This is an offline test oracle; production inference never imports Python.
|
|
"""
|
|
import argparse
|
|
import ctypes
|
|
import json
|
|
import time
|
|
|
|
import torch
|
|
import torch.nn.functional as F
|
|
|
|
|
|
def dense_reference(q, k, v, width, low_width, heads, kernel):
|
|
qc, vc = q.shape[-1], v.shape[-1]
|
|
qd, vd = qc // heads, vc // heads
|
|
dilation = width // low_width
|
|
q = q.reshape(width, width, heads, qd).permute(2, 0, 1, 3).reshape(heads, width*width, qd)
|
|
def unfold(x, channels):
|
|
x = x.reshape(low_width, low_width, heads, channels).permute(2, 3, 0, 1)
|
|
x = F.interpolate(x, size=(width, width), mode="nearest-exact")
|
|
x = F.unfold(x, kernel, dilation=dilation, padding=(kernel//2)*dilation)
|
|
return x.reshape(heads, channels, kernel*kernel, width*width).permute(0, 3, 2, 1)
|
|
keys = unfold(k, qd)
|
|
scores = (q.unsqueeze(2)*keys).sum(-1) * qd**-0.5
|
|
attn = scores.softmax(-1)
|
|
del keys, scores
|
|
chunks = []
|
|
# Bound the oracle's unfold workspace without changing the math.
|
|
values = v.reshape(low_width*low_width, heads, vd)
|
|
for start in range(0, vd, 16):
|
|
part = values[:, :, start:start+16].reshape(low_width*low_width, -1)
|
|
val = unfold(part, part.shape[-1]//heads)
|
|
chunks.append((attn.unsqueeze(-1)*val).sum(-2))
|
|
out = torch.cat(chunks, dim=-1).permute(1,0,2).reshape(width,width,vc)
|
|
return out.permute(2,0,1).unsqueeze(0)
|
|
|
|
|
|
def main():
|
|
parser=argparse.ArgumentParser()
|
|
parser.add_argument("library")
|
|
parser.add_argument("--benchmark", action="store_true")
|
|
args=parser.parse_args()
|
|
lib=ctypes.CDLL(args.library)
|
|
op=lib.makepad_cuda_pixal_naf_sample_f32
|
|
op.argtypes=[ctypes.c_void_p]*5+[ctypes.c_uint32]*9+[ctypes.c_void_p]
|
|
op.restype=ctypes.c_int
|
|
torch.manual_seed(19)
|
|
torch.backends.cuda.matmul.allow_tf32=False
|
|
def native(q,k,v,uv,width,low,heads,kernel):
|
|
out=torch.empty((len(uv),v.shape[-1]),device="cuda",dtype=torch.float32)
|
|
status=op(q.data_ptr(),k.data_ptr(),v.data_ptr(),uv.data_ptr(),out.data_ptr(),
|
|
len(uv),width,width,low,low,heads,q.shape[-1],v.shape[-1],kernel,
|
|
torch.cuda.current_stream().cuda_stream)
|
|
if status: raise RuntimeError(f"CUDA launch status {status}")
|
|
return out
|
|
for width,low,qc,vc,heads,kernel in [(8,2,8,16,2,3),(32,4,256,1024,4,9),(48,12,32,64,4,5)]:
|
|
q=torch.randn(width*width,qc,device="cuda")
|
|
k=torch.randn(low*low,qc,device="cuda")
|
|
v=torch.randn(low*low,vc,device="cuda")
|
|
uv=torch.rand(73,2,device="cuda")*1.4-0.2
|
|
uv[:4]=torch.tensor([[0,0],[1,1],[0.5,0.5],[-1,2]],device="cuda")
|
|
dense=dense_reference(q,k,v,width,low,heads,kernel)
|
|
expected=F.grid_sample(dense,(uv*2-1).view(1,-1,1,2),padding_mode="border",align_corners=False)
|
|
expected=expected[0,:,:,0].T.contiguous()
|
|
actual=native(q,k,v,uv,width,low,heads,kernel)
|
|
torch.testing.assert_close(actual,expected,rtol=3e-5,atol=3e-6)
|
|
print(json.dumps({"width":width,"max_error":float((actual-expected).abs().max())}),flush=True)
|
|
if args.benchmark:
|
|
for width,low,count in [(512,32,12000),(512,64,30000),(1024,64,30000)]:
|
|
q=torch.randn(width*width,256,device="cuda")
|
|
k=torch.randn(low*low,256,device="cuda")
|
|
v=torch.randn(low*low,1024,device="cuda")
|
|
uv=torch.rand(count,2,device="cuda")
|
|
for _ in range(2): actual=native(q,k,v,uv,width,low,4,9)
|
|
torch.cuda.synchronize()
|
|
start=time.perf_counter()
|
|
for _ in range(5): actual=native(q,k,v,uv,width,low,4,9)
|
|
torch.cuda.synchronize()
|
|
print(json.dumps({"width":width,"low":low,"voxels":count,
|
|
"sample_ms":(time.perf_counter()-start)*200,
|
|
"output_bytes":actual.numel()*4,"dense_output_bytes":width*width*1024*4}),flush=True)
|
|
|
|
|
|
if __name__=="__main__": main()
|