pytorch-image-models/timm/models/layers/split_attn.py

""" Split Attention Conv2d (for ResNeSt Models)

Paper: `ResNeSt: Split-Attention Networks` - /https://arxiv.org/abs/2004.08955

Adapted from original PyTorch impl at https://github.com/zhanghang1989/ResNeSt

Modified for torchscript compat, performance, and consistency with timm by Ross Wightman
"""
import torch
import torch.nn.functional as F
from torch import nn

from .helpers import make_divisible


class RadixSoftmax(nn.Module):
    def __init__(self, radix, cardinality):
        super(RadixSoftmax, self).__init__()
        self.radix = radix
        self.cardinality = cardinality

    def forward(self, x):
        batch = x.size(0)
        if self.radix > 1:
            x = x.view(batch, self.cardinality, self.radix, -1).transpose(1, 2)
            x = F.softmax(x, dim=1)
            x = x.reshape(batch, -1)
        else:
            x = torch.sigmoid(x)
        return x


class SplitAttn(nn.Module):
    """Split-Attention (aka Splat)
    """
    def __init__(self, in_channels, out_channels=None, kernel_size=3, stride=1, padding=None,
                 dilation=1, groups=1, bias=False, radix=2, rd_ratio=0.25, rd_channels=None, rd_divisor=8,
                 act_layer=nn.ReLU, norm_layer=None, drop_block=None, **kwargs):
        super(SplitAttn, self).__init__()
        out_channels = out_channels or in_channels
        self.radix = radix
        self.drop_block = drop_block
        mid_chs = out_channels * radix
        if rd_channels is None:
            attn_chs = make_divisible(in_channels * radix * rd_ratio, min_value=32, divisor=rd_divisor)
        else:
            attn_chs = rd_channels * radix

        padding = kernel_size // 2 if padding is None else padding
        self.conv = nn.Conv2d(
            in_channels, mid_chs, kernel_size, stride, padding, dilation,
            groups=groups * radix, bias=bias, **kwargs)
        self.bn0 = norm_layer(mid_chs) if norm_layer else nn.Identity()
        self.act0 = act_layer(inplace=True)
        self.fc1 = nn.Conv2d(out_channels, attn_chs, 1, groups=groups)
        self.bn1 = norm_layer(attn_chs) if norm_layer else nn.Identity()
        self.act1 = act_layer(inplace=True)
        self.fc2 = nn.Conv2d(attn_chs, mid_chs, 1, groups=groups)
        self.rsoftmax = RadixSoftmax(radix, groups)

    def forward(self, x):
        x = self.conv(x)
        x = self.bn0(x)
        if self.drop_block is not None:
            x = self.drop_block(x)
        x = self.act0(x)

        B, RC, H, W = x.shape
        if self.radix > 1:
            x = x.reshape((B, self.radix, RC // self.radix, H, W))
            x_gap = x.sum(dim=1)
        else:
            x_gap = x
        x_gap = x_gap.mean((2, 3), keepdim=True)
        x_gap = self.fc1(x_gap)
        x_gap = self.bn1(x_gap)
        x_gap = self.act1(x_gap)
        x_attn = self.fc2(x_gap)

        x_attn = self.rsoftmax(x_attn).view(B, -1, 1, 1)
        if self.radix > 1:
            out = (x * x_attn.reshape((B, self.radix, RC // self.radix, 1, 1))).sum(dim=1)
        else:
            out = x * x_attn
        return out.contiguous()
Add ResNeSt models 5 years ago			`""" Split Attention Conv2d (for ResNeSt Models)`

			Paper: `ResNeSt: Split-Attention Networks` - /https://arxiv.org/abs/2004.08955

			`Adapted from original PyTorch impl at https://github.com/zhanghang1989/ResNeSt`

			`Modified for torchscript compat, performance, and consistency with timm by Ross Wightman`
			`"""`
			`import torch`
			`import torch.nn.functional as F`
			`from torch import nn`

Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`from .helpers import make_divisible`

Add ResNeSt models 5 years ago
			`class RadixSoftmax(nn.Module):`
			`def __init__(self, radix, cardinality):`
			`super(RadixSoftmax, self).__init__()`
			`self.radix = radix`
			`self.cardinality = cardinality`

			`def forward(self, x):`
			`batch = x.size(0)`
			`if self.radix > 1:`
			`x = x.view(batch, self.cardinality, self.radix, -1).transpose(1, 2)`
			`x = F.softmax(x, dim=1)`
			`x = x.reshape(batch, -1)`
			`else:`
			`x = torch.sigmoid(x)`
			`return x`


Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`class SplitAttn(nn.Module):`
			`"""Split-Attention (aka Splat)`
Add ResNeSt models 5 years ago			`"""`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`def __init__(self, in_channels, out_channels=None, kernel_size=3, stride=1, padding=None,`
			`dilation=1, groups=1, bias=False, radix=2, rd_ratio=0.25, rd_channels=None, rd_divisor=8,`
Add ResNeSt models 5 years ago			`act_layer=nn.ReLU, norm_layer=None, drop_block=None, **kwargs):`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`super(SplitAttn, self).__init__()`
			`out_channels = out_channels or in_channels`
Add ResNeSt models 5 years ago			`self.radix = radix`
Improve dropblock impl, add fast variant, and better AMP speed, inplace, batchwise... few ResNeSt cleanups 5 years ago			`self.drop_block = drop_block`
			`mid_chs = out_channels * radix`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`if rd_channels is None:`
			`attn_chs = make_divisible(in_channels * radix * rd_ratio, min_value=32, divisor=rd_divisor)`
			`else:`
			`attn_chs = rd_channels * radix`
Improve dropblock impl, add fast variant, and better AMP speed, inplace, batchwise... few ResNeSt cleanups 5 years ago
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`padding = kernel_size // 2 if padding is None else padding`
Add ResNeSt models 5 years ago			`self.conv = nn.Conv2d(`
			`in_channels, mid_chs, kernel_size, stride, padding, dilation,`
			`groups=groups * radix, bias=bias, **kwargs)`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`self.bn0 = norm_layer(mid_chs) if norm_layer else nn.Identity()`
Add ResNeSt models 5 years ago			`self.act0 = act_layer(inplace=True)`
Improve dropblock impl, add fast variant, and better AMP speed, inplace, batchwise... few ResNeSt cleanups 5 years ago			`self.fc1 = nn.Conv2d(out_channels, attn_chs, 1, groups=groups)`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`self.bn1 = norm_layer(attn_chs) if norm_layer else nn.Identity()`
Add ResNeSt models 5 years ago			`self.act1 = act_layer(inplace=True)`
Improve dropblock impl, add fast variant, and better AMP speed, inplace, batchwise... few ResNeSt cleanups 5 years ago			`self.fc2 = nn.Conv2d(attn_chs, mid_chs, 1, groups=groups)`
Add ResNeSt models 5 years ago			`self.rsoftmax = RadixSoftmax(radix, groups)`

			`def forward(self, x):`
			`x = self.conv(x)`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`x = self.bn0(x)`
Add ResNeSt models 5 years ago			`if self.drop_block is not None:`
			`x = self.drop_block(x)`
			`x = self.act0(x)`

			`B, RC, H, W = x.shape`
			`if self.radix > 1:`
			`x = x.reshape((B, self.radix, RC // self.radix, H, W))`
Improve dropblock impl, add fast variant, and better AMP speed, inplace, batchwise... few ResNeSt cleanups 5 years ago			`x_gap = x.sum(dim=1)`
Add ResNeSt models 5 years ago			`else:`
			`x_gap = x`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`x_gap = x_gap.mean((2, 3), keepdim=True)`
Add ResNeSt models 5 years ago			`x_gap = self.fc1(x_gap)`
Add non-local and BAT attention. Merge attn and self-attn factories into one. Add attention references to README. Add mlp 'mode' to ECA. 3 years ago			`x_gap = self.bn1(x_gap)`
Add ResNeSt models 5 years ago			`x_gap = self.act1(x_gap)`
			`x_attn = self.fc2(x_gap)`

Missed one of the abalation model entrypoints, update README 5 years ago			`x_attn = self.rsoftmax(x_attn).view(B, -1, 1, 1)`
Add ResNeSt models 5 years ago			`if self.radix > 1:`
			`out = (x * x_attn.reshape((B, self.radix, RC // self.radix, 1, 1))).sum(dim=1)`
			`else:`
			`out = x * x_attn`
			`return out.contiguous()`