smart_augmentation/higher/model.py

import math
import torch
import torch.nn as nn
import torch.nn.functional as F

## Basic CNN ##
class LeNet_F(nn.Module):
    def __init__(self, num_inp, num_out):
        super(LeNet_F, self).__init__()
        self._params = nn.ParameterDict({
            'w1': nn.Parameter(torch.zeros(20, num_inp, 5, 5)),
            'b1': nn.Parameter(torch.zeros(20)),
            'w2': nn.Parameter(torch.zeros(50, 20, 5, 5)),
            'b2': nn.Parameter(torch.zeros(50)),
            #'w3': nn.Parameter(torch.zeros(500,4*4*50)), #num_imp=1
            'w3': nn.Parameter(torch.zeros(500,5*5*50)), #num_imp=3
            'b3': nn.Parameter(torch.zeros(500)),
            'w4': nn.Parameter(torch.zeros(num_out, 500)),
            'b4': nn.Parameter(torch.zeros(num_out))
        })
        self.initialize()


    def initialize(self):
        nn.init.kaiming_uniform_(self._params["w1"], a=math.sqrt(5))
        nn.init.kaiming_uniform_(self._params["w2"], a=math.sqrt(5))
        nn.init.kaiming_uniform_(self._params["w3"], a=math.sqrt(5))
        nn.init.kaiming_uniform_(self._params["w4"], a=math.sqrt(5))

    def forward(self, x):
        #print("Start Shape ", x.shape)
        out = F.relu(F.conv2d(input=x, weight=self._params["w1"], bias=self._params["b1"]))
        #print("Shape ", out.shape)
        out = F.max_pool2d(out, 2)
        #print("Shape ", out.shape)
        out = F.relu(F.conv2d(input=out, weight=self._params["w2"], bias=self._params["b2"]))
        #print("Shape ", out.shape)
        out = F.max_pool2d(out, 2)
        #print("Shape ", out.shape)
        out = out.view(out.size(0), -1)
        #print("Shape ", out.shape)
        out = F.relu(F.linear(out, self._params["w3"], self._params["b3"]))
        #print("Shape ", out.shape)
        out = F.linear(out, self._params["w4"], self._params["b4"])
        #print("Shape ", out.shape)
        #return F.log_softmax(out, dim=1)
        return out

    def __getitem__(self, key):
        return self._params[key]

    def __str__(self):
        return "LeNet"

class LeNet(nn.Module):
    def __init__(self, num_inp, num_out):
        super(LeNet, self).__init__()
        self.conv1 = nn.Conv2d(num_inp, 20, 5)
        self.pool = nn.MaxPool2d(2, 2)
        self.conv2 = nn.Conv2d(20, 50, 5)
        self.pool2 = nn.MaxPool2d(2, 2)
        self.fc1 = nn.Linear(5*5*50, 500)
        self.fc2 = nn.Linear(500, num_out)

    def forward(self, x):
        x = self.pool(F.relu(self.conv1(x)))
        x = self.pool2(F.relu(self.conv2(x)))
        x = x.view(x.size(0), -1)
        x = F.relu(self.fc1(x))
        x = self.fc2(x)
        return x

    def __str__(self):
        return "LeNet"

## MobileNetv2 ##

def _make_divisible(v, divisor, min_value=None):
    """
    This function is taken from the original tf repo.
    It ensures that all layers have a channel number that is divisible by 8
    It can be seen here:
    https://github.com/tensorflow/models/blob/master/research/slim/nets/mobilenet/mobilenet.py
    :param v:
    :param divisor:
    :param min_value:
    :return:
    """
    if min_value is None:
        min_value = divisor
    new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
    # Make sure that round down does not go down by more than 10%.
    if new_v < 0.9 * v:
        new_v += divisor
    return new_v


class ConvBNReLU(nn.Sequential):
    def __init__(self, in_planes, out_planes, kernel_size=3, stride=1, groups=1):
        padding = (kernel_size - 1) // 2
        super(ConvBNReLU, self).__init__(
            nn.Conv2d(in_planes, out_planes, kernel_size, stride, padding, groups=groups, bias=False),
            nn.BatchNorm2d(out_planes),
            nn.ReLU6(inplace=True)
        )


class InvertedResidual(nn.Module):
    def __init__(self, inp, oup, stride, expand_ratio):
        super(InvertedResidual, self).__init__()
        self.stride = stride
        assert stride in [1, 2]

        hidden_dim = int(round(inp * expand_ratio))
        self.use_res_connect = self.stride == 1 and inp == oup

        layers = []
        if expand_ratio != 1:
            # pw
            layers.append(ConvBNReLU(inp, hidden_dim, kernel_size=1))
        layers.extend([
            # dw
            ConvBNReLU(hidden_dim, hidden_dim, stride=stride, groups=hidden_dim),
            # pw-linear
            nn.Conv2d(hidden_dim, oup, 1, 1, 0, bias=False),
            nn.BatchNorm2d(oup),
        ])
        self.conv = nn.Sequential(*layers)

    def forward(self, x):
        if self.use_res_connect:
            return x + self.conv(x)
        else:
            return self.conv(x)


class MobileNetV2(nn.Module):
    def __init__(self,
                 num_classes=1000,
                 width_mult=1.0,
                 inverted_residual_setting=None,
                 round_nearest=8,
                 block=None):
        """
        MobileNet V2 main class
        Args:
            num_classes (int): Number of classes
            width_mult (float): Width multiplier - adjusts number of channels in each layer by this amount
            inverted_residual_setting: Network structure
            round_nearest (int): Round the number of channels in each layer to be a multiple of this number
            Set to 1 to turn off rounding
            block: Module specifying inverted residual building block for mobilenet
        """
        super(MobileNetV2, self).__init__()

        if block is None:
            block = InvertedResidual
        input_channel = 32
        last_channel = 1280

        if inverted_residual_setting is None:
            inverted_residual_setting = [
                # t, c, n, s
                [1, 16, 1, 1],
                [6, 24, 2, 2],
                [6, 32, 3, 2],
                [6, 64, 4, 2],
                [6, 96, 3, 1],
                [6, 160, 3, 2],
                [6, 320, 1, 1],
            ]

        # only check the first element, assuming user knows t,c,n,s are required
        if len(inverted_residual_setting) == 0 or len(inverted_residual_setting[0]) != 4:
            raise ValueError("inverted_residual_setting should be non-empty "
                             "or a 4-element list, got {}".format(inverted_residual_setting))

        # building first layer
        input_channel = _make_divisible(input_channel * width_mult, round_nearest)
        self.last_channel = _make_divisible(last_channel * max(1.0, width_mult), round_nearest)
        features = [ConvBNReLU(3, input_channel, stride=2)]
        # building inverted residual blocks
        for t, c, n, s in inverted_residual_setting:
            output_channel = _make_divisible(c * width_mult, round_nearest)
            for i in range(n):
                stride = s if i == 0 else 1
                features.append(block(input_channel, output_channel, stride, expand_ratio=t))
                input_channel = output_channel
        # building last several layers
        features.append(ConvBNReLU(input_channel, self.last_channel, kernel_size=1))
        # make it nn.Sequential
        self.features = nn.Sequential(*features)

        # building classifier
        self.classifier = nn.Sequential(
            nn.Dropout(0.2),
            nn.Linear(self.last_channel, num_classes),
        )

        # weight initialization
        for m in self.modules():
            if isinstance(m, nn.Conv2d):
                nn.init.kaiming_normal_(m.weight, mode='fan_out')
                if m.bias is not None:
                    nn.init.zeros_(m.bias)
            elif isinstance(m, nn.BatchNorm2d):
                nn.init.ones_(m.weight)
                nn.init.zeros_(m.bias)
            elif isinstance(m, nn.Linear):
                nn.init.normal_(m.weight, 0, 0.01)
                nn.init.zeros_(m.bias)

    def _forward_impl(self, x):
        # This exists since TorchScript doesn't support inheritance, so the superclass method
        # (this one) needs to have a name other than `forward` that can be accessed in a subclass
        x = self.features(x)
        x = x.mean([2, 3])
        x = self.classifier(x)
        return x

    def forward(self, x):
        return self._forward_impl(x)

    def __str__(self):
        return "MobileNetV2"

## ResNet ##
def conv3x3(in_planes, out_planes, stride=1, groups=1, dilation=1):
    """3x3 convolution with padding"""
    return nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,
                     padding=dilation, groups=groups, bias=False, dilation=dilation)


def conv1x1(in_planes, out_planes, stride=1):
    """1x1 convolution"""
    return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)


class BasicBlock(nn.Module):
    expansion = 1
    __constants__ = ['downsample']

    def __init__(self, inplanes, planes, stride=1, downsample=None, groups=1,
                 base_width=64, dilation=1, norm_layer=None):
        super(BasicBlock, self).__init__()
        if norm_layer is None:
            norm_layer = nn.BatchNorm2d
        if groups != 1 or base_width != 64:
            raise ValueError('BasicBlock only supports groups=1 and base_width=64')
        if dilation > 1:
            raise NotImplementedError("Dilation > 1 not supported in BasicBlock")
        # Both self.conv1 and self.downsample layers downsample the input when stride != 1
        self.conv1 = conv3x3(inplanes, planes, stride)
        self.bn1 = norm_layer(planes)
        self.relu = nn.ReLU(inplace=True)
        self.conv2 = conv3x3(planes, planes)
        self.bn2 = norm_layer(planes)
        self.downsample = downsample
        self.stride = stride

    def forward(self, x):
        identity = x

        out = self.conv1(x)
        out = self.bn1(out)
        out = self.relu(out)

        out = self.conv2(out)
        out = self.bn2(out)

        if self.downsample is not None:
            identity = self.downsample(x)

        out += identity
        out = self.relu(out)

        return out


class Bottleneck(nn.Module):
    expansion = 4
    __constants__ = ['downsample']

    def __init__(self, inplanes, planes, stride=1, downsample=None, groups=1,
                 base_width=64, dilation=1, norm_layer=None):
        super(Bottleneck, self).__init__()
        if norm_layer is None:
            norm_layer = nn.BatchNorm2d
        width = int(planes * (base_width / 64.)) * groups
        # Both self.conv2 and self.downsample layers downsample the input when stride != 1
        self.conv1 = conv1x1(inplanes, width)
        self.bn1 = norm_layer(width)
        self.conv2 = conv3x3(width, width, stride, groups, dilation)
        self.bn2 = norm_layer(width)
        self.conv3 = conv1x1(width, planes * self.expansion)
        self.bn3 = norm_layer(planes * self.expansion)
        self.relu = nn.ReLU(inplace=True)
        self.downsample = downsample
        self.stride = stride

    def forward(self, x):
        identity = x

        out = self.conv1(x)
        out = self.bn1(out)
        out = self.relu(out)

        out = self.conv2(out)
        out = self.bn2(out)
        out = self.relu(out)

        out = self.conv3(out)
        out = self.bn3(out)

        if self.downsample is not None:
            identity = self.downsample(x)

        out += identity
        out = self.relu(out)

        return out

#ResNet18 : block=BasicBlock, layers=[2, 2, 2, 2]
class ResNet(nn.Module):

    def __init__(self, block, layers, num_classes=1000, zero_init_residual=False,
                 groups=1, width_per_group=64, replace_stride_with_dilation=None,
                 norm_layer=None):
        super(ResNet, self).__init__()
        if norm_layer is None:
            norm_layer = nn.BatchNorm2d
        self._norm_layer = norm_layer

        self.inplanes = 64
        self.dilation = 1
        if replace_stride_with_dilation is None:
            # each element in the tuple indicates if we should replace
            # the 2x2 stride with a dilated convolution instead
            replace_stride_with_dilation = [False, False, False]
        if len(replace_stride_with_dilation) != 3:
            raise ValueError("replace_stride_with_dilation should be None "
                             "or a 3-element tuple, got {}".format(replace_stride_with_dilation))
        self.groups = groups
        self.base_width = width_per_group
        self.conv1 = nn.Conv2d(3, self.inplanes, kernel_size=7, stride=2, padding=3,
                               bias=False)
        self.bn1 = norm_layer(self.inplanes)
        self.relu = nn.ReLU(inplace=True)
        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
        self.layer1 = self._make_layer(block, 64, layers[0])
        self.layer2 = self._make_layer(block, 128, layers[1], stride=2,
                                       dilate=replace_stride_with_dilation[0])
        self.layer3 = self._make_layer(block, 256, layers[2], stride=2,
                                       dilate=replace_stride_with_dilation[1])
        self.layer4 = self._make_layer(block, 512, layers[3], stride=2,
                                       dilate=replace_stride_with_dilation[2])
        self.avgpool = nn.AdaptiveAvgPool2d((1, 1))
        self.fc = nn.Linear(512 * block.expansion, num_classes)

        for m in self.modules():
            if isinstance(m, nn.Conv2d):
                nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')
            elif isinstance(m, (nn.BatchNorm2d, nn.GroupNorm)):
                nn.init.constant_(m.weight, 1)
                nn.init.constant_(m.bias, 0)

        # Zero-initialize the last BN in each residual branch,
        # so that the residual branch starts with zeros, and each residual block behaves like an identity.
        # This improves the model by 0.2~0.3% according to https://arxiv.org/abs/1706.02677
        if zero_init_residual:
            for m in self.modules():
                if isinstance(m, Bottleneck):
                    nn.init.constant_(m.bn3.weight, 0)
                elif isinstance(m, BasicBlock):
                    nn.init.constant_(m.bn2.weight, 0)

    def _make_layer(self, block, planes, blocks, stride=1, dilate=False):
        norm_layer = self._norm_layer
        downsample = None
        previous_dilation = self.dilation
        if dilate:
            self.dilation *= stride
            stride = 1
        if stride != 1 or self.inplanes != planes * block.expansion:
            downsample = nn.Sequential(
                conv1x1(self.inplanes, planes * block.expansion, stride),
                norm_layer(planes * block.expansion),
            )

        layers = []
        layers.append(block(self.inplanes, planes, stride, downsample, self.groups,
                            self.base_width, previous_dilation, norm_layer))
        self.inplanes = planes * block.expansion
        for _ in range(1, blocks):
            layers.append(block(self.inplanes, planes, groups=self.groups,
                                base_width=self.base_width, dilation=self.dilation,
                                norm_layer=norm_layer))

        return nn.Sequential(*layers)

    def _forward_impl(self, x):
        # See note [TorchScript super()]
        x = self.conv1(x)
        x = self.bn1(x)
        x = self.relu(x)
        x = self.maxpool(x)

        x = self.layer1(x)
        x = self.layer2(x)
        x = self.layer3(x)
        x = self.layer4(x)

        x = self.avgpool(x)
        x = torch.flatten(x, 1)
        x = self.fc(x)

        return x

    def forward(self, x):
        return self._forward_impl(x)

## Wide ResNet ##
#https://github.com/xternalz/WideResNet-pytorch/blob/master/wideresnet.py
#https://github.com/arcelien/pba/blob/master/pba/wrn.py
#https://github.com/szagoruyko/wide-residual-networks/blob/master/pytorch/resnet.py

class BasicBlock(nn.Module):
    def __init__(self, in_planes, out_planes, stride, dropRate=0.0):
        super(BasicBlock, self).__init__()
        self.bn1 = nn.BatchNorm2d(in_planes)
        self.relu1 = nn.ReLU(inplace=True)
        self.conv1 = nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,
                               padding=1, bias=False)
        self.bn2 = nn.BatchNorm2d(out_planes)
        self.relu2 = nn.ReLU(inplace=True)
        self.conv2 = nn.Conv2d(out_planes, out_planes, kernel_size=3, stride=1,
                               padding=1, bias=False)
        self.droprate = dropRate
        self.equalInOut = (in_planes == out_planes)
        self.convShortcut = (not self.equalInOut) and nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride,
                               padding=0, bias=False) or None
    def forward(self, x):
        if not self.equalInOut:
            x = self.relu1(self.bn1(x))
        else:
            out = self.relu1(self.bn1(x))
        out = self.relu2(self.bn2(self.conv1(out if self.equalInOut else x)))
        if self.droprate > 0:
            out = F.dropout(out, p=self.droprate, training=self.training)
        out = self.conv2(out)
        return torch.add(x if self.equalInOut else self.convShortcut(x), out)

class NetworkBlock(nn.Module):
    def __init__(self, nb_layers, in_planes, out_planes, block, stride, dropRate=0.0):
        super(NetworkBlock, self).__init__()
        self.layer = self._make_layer(block, in_planes, out_planes, nb_layers, stride, dropRate)
    def _make_layer(self, block, in_planes, out_planes, nb_layers, stride, dropRate):
        layers = []
        for i in range(int(nb_layers)):
            layers.append(block(i == 0 and in_planes or out_planes, out_planes, i == 0 and stride or 1, dropRate))
        return nn.Sequential(*layers)
    def forward(self, x):
        return self.layer(x)

#wrn_size: 32 = WRN-28-2 ? 160 = WRN-28-10
class WideResNet(nn.Module):
    #def __init__(self, depth, num_classes, widen_factor=1, dropRate=0.0):
    def __init__(self, num_classes, wrn_size, depth=28, dropRate=0.0):
        super(WideResNet, self).__init__()

        self.kernel_size = wrn_size
        self.depth=depth
        filter_size = 3
        nChannels = [min(self.kernel_size, 16), self.kernel_size, self.kernel_size * 2, self.kernel_size * 4]
        strides = [1, 2, 2]  # stride for each resblock

        #nChannels = [16, 16*widen_factor, 32*widen_factor, 64*widen_factor]
        assert((depth - 4) % 6 == 0)
        n = (depth - 4) / 6
        block = BasicBlock
        # 1st conv before any network block
        self.conv1 = nn.Conv2d(filter_size, nChannels[0], kernel_size=3, stride=1,
                               padding=1, bias=False)
        # 1st block
        self.block1 = NetworkBlock(n, nChannels[0], nChannels[1], block, strides[0], dropRate)
        # 2nd block
        self.block2 = NetworkBlock(n, nChannels[1], nChannels[2], block, strides[1], dropRate)
        # 3rd block
        self.block3 = NetworkBlock(n, nChannels[2], nChannels[3], block, strides[2], dropRate)
        # global average pooling and classifier
        self.bn1 = nn.BatchNorm2d(nChannels[3])
        self.relu = nn.ReLU(inplace=True)
        self.fc = nn.Linear(nChannels[3], num_classes)
        self.nChannels = nChannels[3]

        for m in self.modules():
            if isinstance(m, nn.Conv2d):
                nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')
            elif isinstance(m, nn.BatchNorm2d):
                m.weight.data.fill_(1)
                m.bias.data.zero_()
            elif isinstance(m, nn.Linear):
                m.bias.data.zero_()
    def forward(self, x):
        out = self.conv1(x)
        out = self.block1(out)
        out = self.block2(out)
        out = self.block3(out)
        out = self.relu(self.bn1(out))
        out = F.avg_pool2d(out, 8)
        out = out.view(-1, self.nChannels)
        return self.fc(out)

    def architecture(self):
        return super(WideResNet, self).__str__()

    def __str__(self):
        return "WideResNet(s{}-d{})".format(self.kernel_size, self.depth)
Initial Commit 2019-11-08 11:28:06 -05:00			`import math`
			`import torch`
			`import torch.nn as nn`
			`import torch.nn.functional as F`

Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`## Basic CNN ##`
Tests consomation memoire/temps + methode KL divergence (UDA) 2019-12-06 14:13:28 -05:00			`class LeNet_F(nn.Module):`
Initial Commit 2019-11-08 11:28:06 -05:00			`def __init__(self, num_inp, num_out):`
Tests consomation memoire/temps + methode KL divergence (UDA) 2019-12-06 14:13:28 -05:00			`super(LeNet_F, self).__init__()`
Initial Commit 2019-11-08 11:28:06 -05:00			`self._params = nn.ParameterDict({`
			`'w1': nn.Parameter(torch.zeros(20, num_inp, 5, 5)),`
			`'b1': nn.Parameter(torch.zeros(20)),`
			`'w2': nn.Parameter(torch.zeros(50, 20, 5, 5)),`
			`'b2': nn.Parameter(torch.zeros(50)),`
			`#'w3': nn.Parameter(torch.zeros(500,4450)), #num_imp=1`
			`'w3': nn.Parameter(torch.zeros(500,5550)), #num_imp=3`
			`'b3': nn.Parameter(torch.zeros(500)),`
			`'w4': nn.Parameter(torch.zeros(num_out, 500)),`
			`'b4': nn.Parameter(torch.zeros(num_out))`
			`})`
			`self.initialize()`


			`def initialize(self):`
			`nn.init.kaiming_uniform_(self._params["w1"], a=math.sqrt(5))`
			`nn.init.kaiming_uniform_(self._params["w2"], a=math.sqrt(5))`
			`nn.init.kaiming_uniform_(self._params["w3"], a=math.sqrt(5))`
			`nn.init.kaiming_uniform_(self._params["w4"], a=math.sqrt(5))`

			`def forward(self, x):`
			`#print("Start Shape ", x.shape)`
			`out = F.relu(F.conv2d(input=x, weight=self._params["w1"], bias=self._params["b1"]))`
			`#print("Shape ", out.shape)`
			`out = F.max_pool2d(out, 2)`
			`#print("Shape ", out.shape)`
			`out = F.relu(F.conv2d(input=out, weight=self._params["w2"], bias=self._params["b2"]))`
			`#print("Shape ", out.shape)`
			`out = F.max_pool2d(out, 2)`
			`#print("Shape ", out.shape)`
			`out = out.view(out.size(0), -1)`
			`#print("Shape ", out.shape)`
			`out = F.relu(F.linear(out, self._params["w3"], self._params["b3"]))`
			`#print("Shape ", out.shape)`
			`out = F.linear(out, self._params["w4"], self._params["b4"])`
			`#print("Shape ", out.shape)`
Test KL divergence from UDA 2019-12-06 10:44:18 -05:00			`#return F.log_softmax(out, dim=1)`
			`return out`
Initial Commit 2019-11-08 11:28:06 -05:00
			`def __getitem__(self, key):`
			`return self._params[key]`

			`def __str__(self):`
Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`return "LeNet"`

Tests consomation memoire/temps + methode KL divergence (UDA) 2019-12-06 14:13:28 -05:00			`class LeNet(nn.Module):`
			`def __init__(self, num_inp, num_out):`
			`super(LeNet, self).__init__()`
			`self.conv1 = nn.Conv2d(num_inp, 20, 5)`
			`self.pool = nn.MaxPool2d(2, 2)`
			`self.conv2 = nn.Conv2d(20, 50, 5)`
			`self.pool2 = nn.MaxPool2d(2, 2)`
			`self.fc1 = nn.Linear(5550, 500)`
			`self.fc2 = nn.Linear(500, num_out)`

			`def forward(self, x):`
			`x = self.pool(F.relu(self.conv1(x)))`
			`x = self.pool2(F.relu(self.conv2(x)))`
			`x = x.view(x.size(0), -1)`
			`x = F.relu(self.fc1(x))`
			`x = self.fc2(x)`
			`return x`

			`def __str__(self):`
			`return "LeNet"`

			`## MobileNetv2 ##`

			`def _make_divisible(v, divisor, min_value=None):`
			`"""`
			`This function is taken from the original tf repo.`
			`It ensures that all layers have a channel number that is divisible by 8`
			`It can be seen here:`
			`https://github.com/tensorflow/models/blob/master/research/slim/nets/mobilenet/mobilenet.py`
			`:param v:`
			`:param divisor:`
			`:param min_value:`
			`:return:`
			`"""`
			`if min_value is None:`
			`min_value = divisor`
			`new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)`
			`# Make sure that round down does not go down by more than 10%.`
			`if new_v < 0.9 * v:`
			`new_v += divisor`
			`return new_v`


			`class ConvBNReLU(nn.Sequential):`
			`def __init__(self, in_planes, out_planes, kernel_size=3, stride=1, groups=1):`
			`padding = (kernel_size - 1) // 2`
			`super(ConvBNReLU, self).__init__(`
			`nn.Conv2d(in_planes, out_planes, kernel_size, stride, padding, groups=groups, bias=False),`
			`nn.BatchNorm2d(out_planes),`
			`nn.ReLU6(inplace=True)`
			`)`


			`class InvertedResidual(nn.Module):`
			`def __init__(self, inp, oup, stride, expand_ratio):`
			`super(InvertedResidual, self).__init__()`
			`self.stride = stride`
			`assert stride in [1, 2]`

			`hidden_dim = int(round(inp * expand_ratio))`
			`self.use_res_connect = self.stride == 1 and inp == oup`

			`layers = []`
			`if expand_ratio != 1:`
			`# pw`
			`layers.append(ConvBNReLU(inp, hidden_dim, kernel_size=1))`
			`layers.extend([`
			`# dw`
			`ConvBNReLU(hidden_dim, hidden_dim, stride=stride, groups=hidden_dim),`
			`# pw-linear`
			`nn.Conv2d(hidden_dim, oup, 1, 1, 0, bias=False),`
			`nn.BatchNorm2d(oup),`
			`])`
			`self.conv = nn.Sequential(*layers)`

			`def forward(self, x):`
			`if self.use_res_connect:`
			`return x + self.conv(x)`
			`else:`
			`return self.conv(x)`


			`class MobileNetV2(nn.Module):`
			`def __init__(self,`
			`num_classes=1000,`
			`width_mult=1.0,`
			`inverted_residual_setting=None,`
			`round_nearest=8,`
			`block=None):`
			`"""`
			`MobileNet V2 main class`
			`Args:`
			`num_classes (int): Number of classes`
			`width_mult (float): Width multiplier - adjusts number of channels in each layer by this amount`
			`inverted_residual_setting: Network structure`
			`round_nearest (int): Round the number of channels in each layer to be a multiple of this number`
			`Set to 1 to turn off rounding`
			`block: Module specifying inverted residual building block for mobilenet`
			`"""`
			`super(MobileNetV2, self).__init__()`

			`if block is None:`
			`block = InvertedResidual`
			`input_channel = 32`
			`last_channel = 1280`

			`if inverted_residual_setting is None:`
			`inverted_residual_setting = [`
			`# t, c, n, s`
			`[1, 16, 1, 1],`
			`[6, 24, 2, 2],`
			`[6, 32, 3, 2],`
			`[6, 64, 4, 2],`
			`[6, 96, 3, 1],`
			`[6, 160, 3, 2],`
			`[6, 320, 1, 1],`
			`]`

			`# only check the first element, assuming user knows t,c,n,s are required`
			`if len(inverted_residual_setting) == 0 or len(inverted_residual_setting[0]) != 4:`
			`raise ValueError("inverted_residual_setting should be non-empty "`
			`"or a 4-element list, got {}".format(inverted_residual_setting))`

			`# building first layer`
			`input_channel = _make_divisible(input_channel * width_mult, round_nearest)`
			`self.last_channel = _make_divisible(last_channel * max(1.0, width_mult), round_nearest)`
			`features = [ConvBNReLU(3, input_channel, stride=2)]`
			`# building inverted residual blocks`
			`for t, c, n, s in inverted_residual_setting:`
			`output_channel = _make_divisible(c * width_mult, round_nearest)`
			`for i in range(n):`
			`stride = s if i == 0 else 1`
			`features.append(block(input_channel, output_channel, stride, expand_ratio=t))`
			`input_channel = output_channel`
			`# building last several layers`
			`features.append(ConvBNReLU(input_channel, self.last_channel, kernel_size=1))`
			`# make it nn.Sequential`
			`self.features = nn.Sequential(*features)`

			`# building classifier`
			`self.classifier = nn.Sequential(`
			`nn.Dropout(0.2),`
			`nn.Linear(self.last_channel, num_classes),`
			`)`

			`# weight initialization`
			`for m in self.modules():`
			`if isinstance(m, nn.Conv2d):`
			`nn.init.kaiming_normal_(m.weight, mode='fan_out')`
			`if m.bias is not None:`
			`nn.init.zeros_(m.bias)`
			`elif isinstance(m, nn.BatchNorm2d):`
			`nn.init.ones_(m.weight)`
			`nn.init.zeros_(m.bias)`
			`elif isinstance(m, nn.Linear):`
			`nn.init.normal_(m.weight, 0, 0.01)`
			`nn.init.zeros_(m.bias)`

			`def _forward_impl(self, x):`
			`# This exists since TorchScript doesn't support inheritance, so the superclass method`
			# (this one) needs to have a name other than `forward` that can be accessed in a subclass
			`x = self.features(x)`
			`x = x.mean([2, 3])`
			`x = self.classifier(x)`
			`return x`

			`def forward(self, x):`
			`return self._forward_impl(x)`

			`def __str__(self):`
			`return "MobileNetV2"`

Ajout ResNet18 2019-12-06 16:54:08 -05:00			`## ResNet ##`
			`def conv3x3(in_planes, out_planes, stride=1, groups=1, dilation=1):`
			`"""3x3 convolution with padding"""`
			`return nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,`
			`padding=dilation, groups=groups, bias=False, dilation=dilation)`


			`def conv1x1(in_planes, out_planes, stride=1):`
			`"""1x1 convolution"""`
			`return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)`


			`class BasicBlock(nn.Module):`
			`expansion = 1`
			`__constants__ = ['downsample']`

			`def __init__(self, inplanes, planes, stride=1, downsample=None, groups=1,`
			`base_width=64, dilation=1, norm_layer=None):`
			`super(BasicBlock, self).__init__()`
			`if norm_layer is None:`
			`norm_layer = nn.BatchNorm2d`
			`if groups != 1 or base_width != 64:`
			`raise ValueError('BasicBlock only supports groups=1 and base_width=64')`
			`if dilation > 1:`
			`raise NotImplementedError("Dilation > 1 not supported in BasicBlock")`
			`# Both self.conv1 and self.downsample layers downsample the input when stride != 1`
			`self.conv1 = conv3x3(inplanes, planes, stride)`
			`self.bn1 = norm_layer(planes)`
			`self.relu = nn.ReLU(inplace=True)`
			`self.conv2 = conv3x3(planes, planes)`
			`self.bn2 = norm_layer(planes)`
			`self.downsample = downsample`
			`self.stride = stride`

			`def forward(self, x):`
			`identity = x`

			`out = self.conv1(x)`
			`out = self.bn1(out)`
			`out = self.relu(out)`

			`out = self.conv2(out)`
			`out = self.bn2(out)`

			`if self.downsample is not None:`
			`identity = self.downsample(x)`

			`out += identity`
			`out = self.relu(out)`

			`return out`


			`class Bottleneck(nn.Module):`
			`expansion = 4`
			`__constants__ = ['downsample']`

			`def __init__(self, inplanes, planes, stride=1, downsample=None, groups=1,`
			`base_width=64, dilation=1, norm_layer=None):`
			`super(Bottleneck, self).__init__()`
			`if norm_layer is None:`
			`norm_layer = nn.BatchNorm2d`
			`width = int(planes * (base_width / 64.)) * groups`
			`# Both self.conv2 and self.downsample layers downsample the input when stride != 1`
			`self.conv1 = conv1x1(inplanes, width)`
			`self.bn1 = norm_layer(width)`
			`self.conv2 = conv3x3(width, width, stride, groups, dilation)`
			`self.bn2 = norm_layer(width)`
			`self.conv3 = conv1x1(width, planes * self.expansion)`
			`self.bn3 = norm_layer(planes * self.expansion)`
			`self.relu = nn.ReLU(inplace=True)`
			`self.downsample = downsample`
			`self.stride = stride`

			`def forward(self, x):`
			`identity = x`

			`out = self.conv1(x)`
			`out = self.bn1(out)`
			`out = self.relu(out)`

			`out = self.conv2(out)`
			`out = self.bn2(out)`
			`out = self.relu(out)`

			`out = self.conv3(out)`
			`out = self.bn3(out)`

			`if self.downsample is not None:`
			`identity = self.downsample(x)`

			`out += identity`
			`out = self.relu(out)`

			`return out`

			`#ResNet18 : block=BasicBlock, layers=[2, 2, 2, 2]`
			`class ResNet(nn.Module):`

			`def __init__(self, block, layers, num_classes=1000, zero_init_residual=False,`
			`groups=1, width_per_group=64, replace_stride_with_dilation=None,`
			`norm_layer=None):`
			`super(ResNet, self).__init__()`
			`if norm_layer is None:`
			`norm_layer = nn.BatchNorm2d`
			`self._norm_layer = norm_layer`

			`self.inplanes = 64`
			`self.dilation = 1`
			`if replace_stride_with_dilation is None:`
			`# each element in the tuple indicates if we should replace`
			`# the 2x2 stride with a dilated convolution instead`
			`replace_stride_with_dilation = [False, False, False]`
			`if len(replace_stride_with_dilation) != 3:`
			`raise ValueError("replace_stride_with_dilation should be None "`
			`"or a 3-element tuple, got {}".format(replace_stride_with_dilation))`
			`self.groups = groups`
			`self.base_width = width_per_group`
			`self.conv1 = nn.Conv2d(3, self.inplanes, kernel_size=7, stride=2, padding=3,`
			`bias=False)`
			`self.bn1 = norm_layer(self.inplanes)`
			`self.relu = nn.ReLU(inplace=True)`
			`self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)`
			`self.layer1 = self._make_layer(block, 64, layers[0])`
			`self.layer2 = self._make_layer(block, 128, layers[1], stride=2,`
			`dilate=replace_stride_with_dilation[0])`
			`self.layer3 = self._make_layer(block, 256, layers[2], stride=2,`
			`dilate=replace_stride_with_dilation[1])`
			`self.layer4 = self._make_layer(block, 512, layers[3], stride=2,`
			`dilate=replace_stride_with_dilation[2])`
			`self.avgpool = nn.AdaptiveAvgPool2d((1, 1))`
			`self.fc = nn.Linear(512 * block.expansion, num_classes)`

			`for m in self.modules():`
			`if isinstance(m, nn.Conv2d):`
			`nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')`
			`elif isinstance(m, (nn.BatchNorm2d, nn.GroupNorm)):`
			`nn.init.constant_(m.weight, 1)`
			`nn.init.constant_(m.bias, 0)`

			`# Zero-initialize the last BN in each residual branch,`
			`# so that the residual branch starts with zeros, and each residual block behaves like an identity.`
			`# This improves the model by 0.2~0.3% according to https://arxiv.org/abs/1706.02677`
			`if zero_init_residual:`
			`for m in self.modules():`
			`if isinstance(m, Bottleneck):`
			`nn.init.constant_(m.bn3.weight, 0)`
			`elif isinstance(m, BasicBlock):`
			`nn.init.constant_(m.bn2.weight, 0)`

			`def _make_layer(self, block, planes, blocks, stride=1, dilate=False):`
			`norm_layer = self._norm_layer`
			`downsample = None`
			`previous_dilation = self.dilation`
			`if dilate:`
			`self.dilation *= stride`
			`stride = 1`
			`if stride != 1 or self.inplanes != planes * block.expansion:`
			`downsample = nn.Sequential(`
			`conv1x1(self.inplanes, planes * block.expansion, stride),`
			`norm_layer(planes * block.expansion),`
			`)`

			`layers = []`
			`layers.append(block(self.inplanes, planes, stride, downsample, self.groups,`
			`self.base_width, previous_dilation, norm_layer))`
			`self.inplanes = planes * block.expansion`
			`for _ in range(1, blocks):`
			`layers.append(block(self.inplanes, planes, groups=self.groups,`
			`base_width=self.base_width, dilation=self.dilation,`
			`norm_layer=norm_layer))`

			`return nn.Sequential(*layers)`

			`def _forward_impl(self, x):`
			`# See note [TorchScript super()]`
			`x = self.conv1(x)`
			`x = self.bn1(x)`
			`x = self.relu(x)`
			`x = self.maxpool(x)`

			`x = self.layer1(x)`
			`x = self.layer2(x)`
			`x = self.layer3(x)`
			`x = self.layer4(x)`

			`x = self.avgpool(x)`
			`x = torch.flatten(x, 1)`
			`x = self.fc(x)`

			`return x`

			`def forward(self, x):`
			`return self._forward_impl(x)`

Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`## Wide ResNet ##`
			`#https://github.com/xternalz/WideResNet-pytorch/blob/master/wideresnet.py`
			`#https://github.com/arcelien/pba/blob/master/pba/wrn.py`
Amelioration visualisation des proba 2019-11-13 16:18:53 -05:00			`#https://github.com/szagoruyko/wide-residual-networks/blob/master/pytorch/resnet.py`
Resultats avec early stop sur data test 2019-11-14 13:13:03 -05:00
Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`class BasicBlock(nn.Module):`
			`def __init__(self, in_planes, out_planes, stride, dropRate=0.0):`
			`super(BasicBlock, self).__init__()`
			`self.bn1 = nn.BatchNorm2d(in_planes)`
			`self.relu1 = nn.ReLU(inplace=True)`
			`self.conv1 = nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,`
			`padding=1, bias=False)`
			`self.bn2 = nn.BatchNorm2d(out_planes)`
			`self.relu2 = nn.ReLU(inplace=True)`
			`self.conv2 = nn.Conv2d(out_planes, out_planes, kernel_size=3, stride=1,`
			`padding=1, bias=False)`
			`self.droprate = dropRate`
			`self.equalInOut = (in_planes == out_planes)`
			`self.convShortcut = (not self.equalInOut) and nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride,`
			`padding=0, bias=False) or None`
			`def forward(self, x):`
			`if not self.equalInOut:`
			`x = self.relu1(self.bn1(x))`
			`else:`
			`out = self.relu1(self.bn1(x))`
			`out = self.relu2(self.bn2(self.conv1(out if self.equalInOut else x)))`
			`if self.droprate > 0:`
			`out = F.dropout(out, p=self.droprate, training=self.training)`
			`out = self.conv2(out)`
			`return torch.add(x if self.equalInOut else self.convShortcut(x), out)`

			`class NetworkBlock(nn.Module):`
			`def __init__(self, nb_layers, in_planes, out_planes, block, stride, dropRate=0.0):`
			`super(NetworkBlock, self).__init__()`
			`self.layer = self._make_layer(block, in_planes, out_planes, nb_layers, stride, dropRate)`
			`def _make_layer(self, block, in_planes, out_planes, nb_layers, stride, dropRate):`
			`layers = []`
			`for i in range(int(nb_layers)):`
			`layers.append(block(i == 0 and in_planes or out_planes, out_planes, i == 0 and stride or 1, dropRate))`
			`return nn.Sequential(*layers)`
			`def forward(self, x):`
			`return self.layer(x)`

Test WRN Brutus 2019-12-04 16:34:02 -05:00			`#wrn_size: 32 = WRN-28-2 ? 160 = WRN-28-10`
Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`class WideResNet(nn.Module):`
			`#def __init__(self, depth, num_classes, widen_factor=1, dropRate=0.0):`
			`def __init__(self, num_classes, wrn_size, depth=28, dropRate=0.0):`
			`super(WideResNet, self).__init__()`

Amelioration visualisation des proba 2019-11-13 16:18:53 -05:00			`self.kernel_size = wrn_size`
			`self.depth=depth`
Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`filter_size = 3`
Amelioration visualisation des proba 2019-11-13 16:18:53 -05:00			`nChannels = [min(self.kernel_size, 16), self.kernel_size, self.kernel_size * 2, self.kernel_size * 4]`
Ajout WideResNet (A tester) pour comparaison a PBA 2019-11-13 11:06:44 -05:00			`strides = [1, 2, 2] # stride for each resblock`

			`#nChannels = [16, 16widen_factor, 32widen_factor, 64*widen_factor]`
			`assert((depth - 4) % 6 == 0)`
			`n = (depth - 4) / 6`
			`block = BasicBlock`
			`# 1st conv before any network block`
			`self.conv1 = nn.Conv2d(filter_size, nChannels[0], kernel_size=3, stride=1,`
			`padding=1, bias=False)`
			`# 1st block`
			`self.block1 = NetworkBlock(n, nChannels[0], nChannels[1], block, strides[0], dropRate)`
			`# 2nd block`
			`self.block2 = NetworkBlock(n, nChannels[1], nChannels[2], block, strides[1], dropRate)`
			`# 3rd block`
			`self.block3 = NetworkBlock(n, nChannels[2], nChannels[3], block, strides[2], dropRate)`
			`# global average pooling and classifier`
			`self.bn1 = nn.BatchNorm2d(nChannels[3])`
			`self.relu = nn.ReLU(inplace=True)`
			`self.fc = nn.Linear(nChannels[3], num_classes)`
			`self.nChannels = nChannels[3]`

			`for m in self.modules():`
			`if isinstance(m, nn.Conv2d):`
			`nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')`
			`elif isinstance(m, nn.BatchNorm2d):`
			`m.weight.data.fill_(1)`
			`m.bias.data.zero_()`
			`elif isinstance(m, nn.Linear):`
			`m.bias.data.zero_()`
			`def forward(self, x):`
			`out = self.conv1(x)`
			`out = self.block1(out)`
			`out = self.block2(out)`
			`out = self.block3(out)`
			`out = self.relu(self.bn1(out))`
			`out = F.avg_pool2d(out, 8)`
			`out = out.view(-1, self.nChannels)`
Amelioration visualisation des proba 2019-11-13 16:18:53 -05:00			`return self.fc(out)`

			`def architecture(self):`
			`return super(WideResNet, self).__str__()`

			`def __str__(self):`
Resultats avec early stop sur data test 2019-11-14 13:13:03 -05:00			`return "WideResNet(s{}-d{})".format(self.kernel_size, self.depth)`