diff --git a/sam3/perflib/fused.py b/sam3/perflib/fused.py index 6800cca64..4f7455657 100644 --- a/sam3/perflib/fused.py +++ b/sam3/perflib/fused.py @@ -9,7 +9,14 @@ def addmm_act(activation, linear, mat1): if torch.is_grad_enabled(): - raise ValueError("Expected grad to be disabled.") + # Training path: the fused kernel below is inference-only (detached + # weights, bf16 cast), so fall back to standard autograd-friendly ops. + y = linear(mat1) + if activation in [torch.nn.functional.relu, torch.nn.ReLU]: + return torch.nn.functional.relu(y) + if activation in [torch.nn.functional.gelu, torch.nn.GELU]: + return torch.nn.functional.gelu(y) + raise ValueError(f"Unexpected activation {activation}") self = linear.bias.detach() mat2 = linear.weight.detach() self = self.to(torch.bfloat16)