bitsandbytes-rocm/bitsandbytes/nn/modules.py

# Copyright (c) Facebook, Inc. and its affiliates.
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
from typing import Optional, TypeVar, Union, overload

import torch
import torch.nn.functional as F
from torch import Tensor, device, dtype, nn

import bitsandbytes as bnb
from bitsandbytes.optim import GlobalOptimManager

T = TypeVar("T", bound="torch.nn.Module")


class StableEmbedding(torch.nn.Embedding):
    def __init__(
        self,
        num_embeddings: int,
        embedding_dim: int,
        padding_idx: Optional[int] = None,
        max_norm: Optional[float] = None,
        norm_type: float = 2.0,
        scale_grad_by_freq: bool = False,
        sparse: bool = False,
        _weight: Optional[Tensor] = None,
        device=None,
        dtype=None,
    ) -> None:
        super().__init__(
            num_embeddings,
            embedding_dim,
            padding_idx,
            max_norm,
            norm_type,
            scale_grad_by_freq,
            sparse,
            _weight,
            device,
            dtype,
        )
        self.norm = torch.nn.LayerNorm(embedding_dim, device=device)
        GlobalOptimManager.get_instance().register_module_override(
            self, "weight", {"optim_bits": 32}
        )

    def reset_parameters(self) -> None:
        torch.nn.init.xavier_uniform_(self.weight)
        self._fill_padding_idx_with_zero()

    """ !!! This is a redefinition of _fill_padding_idx_with_zero in torch.nn.Embedding
        to make the Layer compatible with Pytorch < 1.9.
        This means that if this changes in future PyTorch releases this need to change too
        which is cumbersome. However, with this we can ensure compatibility with previous
        PyTorch releases.
    """

    def _fill_padding_idx_with_zero(self) -> None:
        if self.padding_idx is not None:
            with torch.no_grad():
                self.weight[self.padding_idx].fill_(0)

    def forward(self, input: Tensor) -> Tensor:
        emb = F.embedding(
            input,
            self.weight,
            self.padding_idx,
            self.max_norm,
            self.norm_type,
            self.scale_grad_by_freq,
            self.sparse,
        )

        # always apply layer norm in full precision
        emb = emb.to(torch.get_default_dtype())

        return self.norm(emb).to(self.weight.dtype)


class Embedding(torch.nn.Embedding):
    def __init__(
        self,
        num_embeddings: int,
        embedding_dim: int,
        padding_idx: Optional[int] = None,
        max_norm: Optional[float] = None,
        norm_type: float = 2.0,
        scale_grad_by_freq: bool = False,
        sparse: bool = False,
        _weight: Optional[Tensor] = None,
    ) -> None:
        super().__init__(
            num_embeddings,
            embedding_dim,
            padding_idx,
            max_norm,
            norm_type,
            scale_grad_by_freq,
            sparse,
            _weight,
        )
        GlobalOptimManager.get_instance().register_module_override(
            self, "weight", {"optim_bits": 32}
        )

    def reset_parameters(self) -> None:
        torch.nn.init.xavier_uniform_(self.weight)
        self._fill_padding_idx_with_zero()

    """ !!! This is a redefinition of _fill_padding_idx_with_zero in torch.nn.Embedding
        to make the Layer compatible with Pytorch < 1.9.
        This means that if this changes in future PyTorch releases this need to change too
        which is cumbersome. However, with this we can ensure compatibility with previous
        PyTorch releases.
    """

    def _fill_padding_idx_with_zero(self) -> None:
        if self.padding_idx is not None:
            with torch.no_grad():
                self.weight[self.padding_idx].fill_(0)

    def forward(self, input: Tensor) -> Tensor:
        emb = F.embedding(
            input,
            self.weight,
            self.padding_idx,
            self.max_norm,
            self.norm_type,
            self.scale_grad_by_freq,
            self.sparse,
        )

        return emb


class Int8Params(torch.nn.Parameter):
    def __new__(
        cls,
        data=None,
        requires_grad=True,
        has_fp16_weights=False,
        CB=None,
        SCB=None,
    ):
        cls.has_fp16_weights = has_fp16_weights
        cls.CB = None
        cls.SCB = None
        if data is None:
            data = torch.empty(0)
        return torch.Tensor._make_subclass(cls, data, requires_grad)

    def cuda(self, device):
        if self.has_fp16_weights:
            return super().cuda(device)
        else:
            # we store the 8-bit rows-major weight
            # we convert this weight to the turning/ampere weight during the first inference pass
            B = self.data.contiguous().half().cuda(device)
            CB, CBt, SCB, SCBt, coo_tensorB = bnb.functional.double_quant(B)
            del CBt
            del SCBt
            self.data = CB
            setattr(self, "CB", CB)
            setattr(self, "SCB", SCB)

        return self

    @overload
    def to(
        self: T,
        device: Optional[Union[int, device]] = ...,
        dtype: Optional[Union[dtype, str]] = ...,
        non_blocking: bool = ...,
    ) -> T:
        ...

    @overload
    def to(self: T, dtype: Union[dtype, str], non_blocking: bool = ...) -> T:
        ...

    @overload
    def to(self: T, tensor: Tensor, non_blocking: bool = ...) -> T:
        ...

    def to(self, *args, **kwargs):
        device, dtype, non_blocking, convert_to_format = torch._C._nn._parse_to(
            *args, **kwargs
        )

        if (
            device is not None
            and device.type == "cuda"
            and self.data.device.type == "cpu"
        ):
            return self.cuda(device)
        else:
            new_param = Int8Params(
                super().to(
                    device=device, dtype=dtype, non_blocking=non_blocking
                ),
                requires_grad=self.requires_grad,
                has_fp16_weights=self.has_fp16_weights,
            )
            new_param.CB = self.CB
            new_param.SCB = self.SCB

            return new_param


class Linear8bitLt(nn.Linear):
    def __init__(self, input_features, output_features, bias=True, has_fp16_weights=True,
                       memory_efficient_backward=False, threshold=0.0, index=None):
        super().__init__(input_features, output_features, bias)
        assert not memory_efficient_backward, "memory_efficient_backward is no longer required and the argument is deprecated in 0.37.0 and will be removed in 0.39.0"
        self.state = bnb.MatmulLtState()
        self.index = index

        self.state.threshold = threshold
        self.state.has_fp16_weights = has_fp16_weights
        self.state.memory_efficient_backward = memory_efficient_backward
        if threshold > 0.0 and not has_fp16_weights:
            self.state.use_pool = True

        self.weight = Int8Params(self.weight.data, has_fp16_weights=has_fp16_weights, requires_grad=has_fp16_weights)

    def init_8bit_state(self):
        self.state.CB = self.weight.CB
        self.state.SCB = self.weight.SCB
        self.weight.CB = None
        self.weight.SCB = None

    def forward(self, x: torch.Tensor):
        self.state.is_training = self.training
        if self.weight.CB is not None:
            self.init_8bit_state()

        # weights are cast automatically as Int8Params, but the bias has to be cast manually
        if self.bias is not None and self.bias.dtype != x.dtype:
            self.bias.data = self.bias.data.to(x.dtype)

        out = bnb.matmul(x, self.weight, bias=self.bias, state=self.state)
        if not self.state.has_fp16_weights:
            if self.state.CB is not None and self.state.CxB is not None:
                # we converted 8-bit row major to turing/ampere format in the first inference pass
                # we no longer need the row-major weight
                del self.state.CB
                self.weight.data = self.state.CxB
        return out
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`# Copyright (c) Facebook, Inc. and its affiliates.`
			`#`
			`# This source code is licensed under the MIT license found in the`
Initial commit 2021-10-06 02:16:20 +00:00			`# LICENSE file in the root directory of this source tree.`
Remove unused imports 2022-10-27 11:32:01 +00:00			`from typing import Optional, TypeVar, Union, overload`
Initial commit 2021-10-06 02:16:20 +00:00
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`import torch`
Initial commit 2021-10-06 02:16:20 +00:00			`import torch.nn.functional as F`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`from torch import Tensor, device, dtype, nn`
Initial commit 2021-10-06 02:16:20 +00:00
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`import bitsandbytes as bnb`
Initial commit 2021-10-06 02:16:20 +00:00			`from bitsandbytes.optim import GlobalOptimManager`

ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`T = TypeVar("T", bound="torch.nn.Module")`

Most tests passing. 2022-07-22 21:41:05 +00:00
Initial commit 2021-10-06 02:16:20 +00:00			`class StableEmbedding(torch.nn.Embedding):`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`def __init__(`
			`self,`
			`num_embeddings: int,`
			`embedding_dim: int,`
			`padding_idx: Optional[int] = None,`
			`max_norm: Optional[float] = None,`
			`norm_type: float = 2.0,`
			`scale_grad_by_freq: bool = False,`
			`sparse: bool = False,`
			`_weight: Optional[Tensor] = None,`
add device and dtype parameters to StableEmbedding 2022-11-04 21:05:30 +00:00			`device=None,`
			`dtype=None,`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`) -> None:`
Simplify statements into equivalent, modern variants via pyupgrade --py37-plus. The changes e.g. are subclassing from object, calling super() with super(ThisClass, self), or old-style syntax formatting. 2022-10-27 11:14:13 +00:00			`super().__init__(`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`num_embeddings,`
			`embedding_dim,`
			`padding_idx,`
			`max_norm,`
			`norm_type,`
			`scale_grad_by_freq,`
			`sparse,`
			`_weight,`
add device and dtype parameters to StableEmbedding 2022-11-04 21:05:30 +00:00			`device,`
			`dtype,`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`)`
add device and dtype parameters to StableEmbedding 2022-11-04 21:05:30 +00:00			`self.norm = torch.nn.LayerNorm(embedding_dim, device=device)`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`GlobalOptimManager.get_instance().register_module_override(`
			`self, "weight", {"optim_bits": 32}`
			`)`
Initial commit 2021-10-06 02:16:20 +00:00
			`def reset_parameters(self) -> None:`
			`torch.nn.init.xavier_uniform_(self.weight)`
			`self._fill_padding_idx_with_zero()`

ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`""" !!! This is a redefinition of _fill_padding_idx_with_zero in torch.nn.Embedding`
Initial commit 2021-10-06 02:16:20 +00:00			`to make the Layer compatible with Pytorch < 1.9.`
			`This means that if this changes in future PyTorch releases this need to change too`
			`which is cumbersome. However, with this we can ensure compatibility with previous`
			`PyTorch releases.`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`"""`

Initial commit 2021-10-06 02:16:20 +00:00			`def _fill_padding_idx_with_zero(self) -> None:`
			`if self.padding_idx is not None:`
			`with torch.no_grad():`
			`self.weight[self.padding_idx].fill_(0)`

			`def forward(self, input: Tensor) -> Tensor:`
			`emb = F.embedding(`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`input,`
			`self.weight,`
			`self.padding_idx,`
			`self.max_norm,`
			`self.norm_type,`
			`self.scale_grad_by_freq,`
			`self.sparse,`
			`)`
Initial commit 2021-10-06 02:16:20 +00:00
add device and dtype parameters to StableEmbedding 2022-11-04 21:05:30 +00:00			`# always apply layer norm in full precision`
			`emb = emb.to(torch.get_default_dtype())`

			`return self.norm(emb).to(self.weight.dtype)`
Added module override, bnb.nn.Embedding #13 #15 #19 2021-11-29 17:32:13 +00:00

			`class Embedding(torch.nn.Embedding):`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`def __init__(`
			`self,`
			`num_embeddings: int,`
			`embedding_dim: int,`
			`padding_idx: Optional[int] = None,`
			`max_norm: Optional[float] = None,`
			`norm_type: float = 2.0,`
			`scale_grad_by_freq: bool = False,`
			`sparse: bool = False,`
			`_weight: Optional[Tensor] = None,`
			`) -> None:`
Simplify statements into equivalent, modern variants via pyupgrade --py37-plus. The changes e.g. are subclassing from object, calling super() with super(ThisClass, self), or old-style syntax formatting. 2022-10-27 11:14:13 +00:00			`super().__init__(`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`num_embeddings,`
			`embedding_dim,`
			`padding_idx,`
			`max_norm,`
			`norm_type,`
			`scale_grad_by_freq,`
			`sparse,`
			`_weight,`
			`)`
			`GlobalOptimManager.get_instance().register_module_override(`
			`self, "weight", {"optim_bits": 32}`
			`)`
Added module override, bnb.nn.Embedding #13 #15 #19 2021-11-29 17:32:13 +00:00
			`def reset_parameters(self) -> None:`
			`torch.nn.init.xavier_uniform_(self.weight)`
			`self._fill_padding_idx_with_zero()`

ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`""" !!! This is a redefinition of _fill_padding_idx_with_zero in torch.nn.Embedding`
Added module override, bnb.nn.Embedding #13 #15 #19 2021-11-29 17:32:13 +00:00			`to make the Layer compatible with Pytorch < 1.9.`
			`This means that if this changes in future PyTorch releases this need to change too`
			`which is cumbersome. However, with this we can ensure compatibility with previous`
			`PyTorch releases.`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`"""`

Added module override, bnb.nn.Embedding #13 #15 #19 2021-11-29 17:32:13 +00:00			`def _fill_padding_idx_with_zero(self) -> None:`
			`if self.padding_idx is not None:`
			`with torch.no_grad():`
			`self.weight[self.padding_idx].fill_(0)`

			`def forward(self, input: Tensor) -> Tensor:`
			`emb = F.embedding(`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`input,`
			`self.weight,`
			`self.padding_idx,`
			`self.max_norm,`
			`self.norm_type,`
			`self.scale_grad_by_freq,`
			`self.sparse,`
			`)`
Added module override, bnb.nn.Embedding #13 #15 #19 2021-11-29 17:32:13 +00:00
			`return emb`
Most tests passing. 2022-07-22 21:41:05 +00:00
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00
Most tests passing. 2022-07-22 21:41:05 +00:00			`class Int8Params(torch.nn.Parameter):`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`def __new__(`
reran black with linelength 80 for greater readability 2022-08-01 16:32:47 +00:00			`cls,`
			`data=None,`
			`requires_grad=True,`
			`has_fp16_weights=False,`
			`CB=None,`
			`SCB=None,`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`):`
Most tests passing. 2022-07-22 21:41:05 +00:00			`cls.has_fp16_weights = has_fp16_weights`
			`cls.CB = None`
			`cls.SCB = None`
			`if data is None:`
			`data = torch.empty(0)`
			`return torch.Tensor._make_subclass(cls, data, requires_grad)`

			`def cuda(self, device):`
			`if self.has_fp16_weights:`
			`return super().cuda(device)`
			`else:`
			`# we store the 8-bit rows-major weight`
			`# we convert this weight to the turning/ampere weight during the first inference pass`
			`B = self.data.contiguous().half().cuda(device)`
			`CB, CBt, SCB, SCBt, coo_tensorB = bnb.functional.double_quant(B)`
			`del CBt`
memory efficient fp16 backward 2022-08-25 16:09:23 +00:00			`del SCBt`
Most tests passing. 2022-07-22 21:41:05 +00:00			`self.data = CB`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`setattr(self, "CB", CB)`
			`setattr(self, "SCB", SCB)`
Most tests passing. 2022-07-22 21:41:05 +00:00
			`return self`

			`@overload`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`def to(`
			`self: T,`
			`device: Optional[Union[int, device]] = ...,`
			`dtype: Optional[Union[dtype, str]] = ...,`
			`non_blocking: bool = ...,`
			`) -> T:`
Most tests passing. 2022-07-22 21:41:05 +00:00			`...`

			`@overload`
			`def to(self: T, dtype: Union[dtype, str], non_blocking: bool = ...) -> T:`
			`...`

			`@overload`
			`def to(self: T, tensor: Tensor, non_blocking: bool = ...) -> T:`
			`...`

			`def to(self, args, *kwargs):`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`device, dtype, non_blocking, convert_to_format = torch._C._nn._parse_to(`
			`args, *kwargs`
			`)`

			`if (`
			`device is not None`
			`and device.type == "cuda"`
			`and self.data.device.type == "cpu"`
			`):`
			`return self.cuda(device)`
Most tests passing. 2022-07-22 21:41:05 +00:00			`else:`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`new_param = Int8Params(`
reran black with linelength 80 for greater readability 2022-08-01 16:32:47 +00:00			`super().to(`
			`device=device, dtype=dtype, non_blocking=non_blocking`
			`),`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`requires_grad=self.requires_grad,`
			`has_fp16_weights=self.has_fp16_weights,`
			`)`
Most tests passing. 2022-07-22 21:41:05 +00:00			`new_param.CB = self.CB`
			`new_param.SCB = self.SCB`

			`return new_param`


			`class Linear8bitLt(nn.Linear):`
Added Int8 matmul support for all GPUs. Full backward support. 2023-02-02 04:09:31 +00:00			`def __init__(self, input_features, output_features, bias=True, has_fp16_weights=True,`
			`memory_efficient_backward=False, threshold=0.0, index=None):`
			`super().__init__(input_features, output_features, bias)`
			`assert not memory_efficient_backward, "memory_efficient_backward is no longer required and the argument is deprecated in 0.37.0 and will be removed in 0.39.0"`
Most tests passing. 2022-07-22 21:41:05 +00:00			`self.state = bnb.MatmulLtState()`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`self.index = index`
Most tests passing. 2022-07-22 21:41:05 +00:00
			`self.state.threshold = threshold`
			`self.state.has_fp16_weights = has_fp16_weights`
add memory effcient backward option 2022-09-11 02:51:29 +00:00			`self.state.memory_efficient_backward = memory_efficient_backward`
Most tests passing. 2022-07-22 21:41:05 +00:00			`if threshold > 0.0 and not has_fp16_weights:`
			`self.state.use_pool = True`

Added Int8 matmul support for all GPUs. Full backward support. 2023-02-02 04:09:31 +00:00			`self.weight = Int8Params(self.weight.data, has_fp16_weights=has_fp16_weights, requires_grad=has_fp16_weights)`
Most tests passing. 2022-07-22 21:41:05 +00:00
			`def init_8bit_state(self):`
			`self.state.CB = self.weight.CB`
			`self.state.SCB = self.weight.SCB`
			`self.weight.CB = None`
			`self.weight.SCB = None`

Added Int8 matmul support for all GPUs. Full backward support. 2023-02-02 04:09:31 +00:00			`def forward(self, x: torch.Tensor):`
Most tests passing. 2022-07-22 21:41:05 +00:00			`self.state.is_training = self.training`
ran black and isort for coherent code formatting 2022-08-01 10:31:48 +00:00			`if self.weight.CB is not None:`
			`self.init_8bit_state()`
Fixed bug in Linear8bitLt, when the bias is None. 2022-08-17 10:45:57 +00:00
			`# weights are cast automatically as Int8Params, but the bias has to be cast manually`
Added Int8 matmul support for all GPUs. Full backward support. 2023-02-02 04:09:31 +00:00			`if self.bias is not None and self.bias.dtype != x.dtype:`
			`self.bias.data = self.bias.data.to(x.dtype)`
Most tests passing. 2022-07-22 21:41:05 +00:00
Added fused bias to matmullt. 2022-08-16 19:00:54 +00:00			`out = bnb.matmul(x, self.weight, bias=self.bias, state=self.state)`
add memory effcient backward option 2022-09-11 02:51:29 +00:00			`if not self.state.has_fp16_weights:`
Added Int8 matmul support for all GPUs. Full backward support. 2023-02-02 04:09:31 +00:00			`if self.state.CB is not None and self.state.CxB is not None:`
add memory effcient backward option 2022-09-11 02:51:29 +00:00			`# we converted 8-bit row major to turing/ampere format in the first inference pass`
			`# we no longer need the row-major weight`
			`del self.state.CB`
			`self.weight.data = self.state.CxB`
Most tests passing. 2022-07-22 21:41:05 +00:00			`return out`