{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":35332,"databundleVersionId":3723648},{"sourceType":"competition","sourceId":15696,"databundleVersionId":907058}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import random\n\nimport os\n\nfrom tqdm.notebook import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nfrom torch import optim\nfrom torch.nn import functional as F\n\nfrom transformers import (\n    AutoTokenizer, \n    AutoConfig, \n    AutoModel, \n    get_linear_schedule_with_warmup,\n    get_cosine_schedule_with_warmup\n)\n\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n\nimport seaborn as sns\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:29:34.623142Z","iopub.execute_input":"2026-02-18T10:29:34.623489Z","iopub.status.idle":"2026-02-18T10:29:41.465262Z","shell.execute_reply.started":"2026-02-18T10:29:34.623456Z","shell.execute_reply":"2026-02-18T10:29:41.463205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torchview\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:34:48.783173Z","iopub.execute_input":"2026-02-18T10:34:48.783542Z","iopub.status.idle":"2026-02-18T10:35:00.257917Z","shell.execute_reply.started":"2026-02-18T10:34:48.783493Z","shell.execute_reply":"2026-02-18T10:35:00.256612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchview import draw_graph\nimport graphviz\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:00.260755Z","iopub.execute_input":"2026-02-18T10:35:00.261234Z","iopub.status.idle":"2026-02-18T10:35:00.315478Z","shell.execute_reply.started":"2026-02-18T10:35:00.261183Z","shell.execute_reply":"2026-02-18T10:35:00.314331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained('bert-base-uncased')\n\ntokenizer\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:00.334092Z","iopub.execute_input":"2026-02-18T10:35:00.334847Z","iopub.status.idle":"2026-02-18T10:35:03.295496Z","shell.execute_reply.started":"2026-02-18T10:35:00.334816Z","shell.execute_reply":"2026-02-18T10:35:03.294318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i,k in enumerate(tokenizer.vocab):\n    print (k, tokenizer.vocab[k])\n    if i == 16:\n        break\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:03.296958Z","iopub.execute_input":"2026-02-18T10:35:03.297404Z","iopub.status.idle":"2026-02-18T10:35:03.528215Z","shell.execute_reply.started":"2026-02-18T10:35:03.297357Z","shell.execute_reply":"2026-02-18T10:35:03.526992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corpus = [\n    'Привет. Добро пожаловать на тренировку по машинному обучению', \n    'Hello. Welcome to machine learning training', \n    'Hello. Welcome to machine [MASK] [MASK]', \n    'aaneo', \n    'bbneo'\n]\nbatch = tokenizer.batch_encode_plus(corpus, \n                                    max_length=128, \n                                    padding=True, \n                                    pad_to_max_length=True)\n\nfor k in batch:\n    print ('\\n')\n    print (k)\n    for i in range(len(corpus)):\n        print (batch[k][i])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:03.530402Z","iopub.execute_input":"2026-02-18T10:35:03.531339Z","iopub.status.idle":"2026-02-18T10:35:03.542755Z","shell.execute_reply.started":"2026-02-18T10:35:03.531270Z","shell.execute_reply":"2026-02-18T10:35:03.541595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bert = AutoModel.from_pretrained('bert-base-uncased')\n\nbert\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:03.545018Z","iopub.execute_input":"2026-02-18T10:35:03.545367Z","iopub.status.idle":"2026-02-18T10:35:06.719289Z","shell.execute_reply.started":"2026-02-18T10:35:03.545336Z","shell.execute_reply":"2026-02-18T10:35:06.718126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_ids = torch.tensor(batch['input_ids'])\nattention_mask = torch.tensor(batch['attention_mask'])\n\ngraph = draw_graph(bert, \n                   input_data=(input_ids, \n                               attention_mask), \n                   expand_nested=True)\ngraph.visual_graph\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:06.721030Z","iopub.execute_input":"2026-02-18T10:35:06.721469Z","iopub.status.idle":"2026-02-18T10:35:07.705982Z","shell.execute_reply.started":"2026-02-18T10:35:06.721439Z","shell.execute_reply":"2026-02-18T10:35:07.704773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bert.embeddings\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:13.169710Z","iopub.execute_input":"2026-02-18T10:35:13.170731Z","iopub.status.idle":"2026-02-18T10:35:13.177229Z","shell.execute_reply.started":"2026-02-18T10:35:13.170691Z","shell.execute_reply":"2026-02-18T10:35:13.175925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ws = bert.embeddings.position_embeddings.weight\nws = ws.detach().numpy()\n\nws\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:35:17.627348Z","iopub.execute_input":"2026-02-18T10:35:17.628053Z","iopub.status.idle":"2026-02-18T10:35:17.636156Z","shell.execute_reply.started":"2026-02-18T10:35:17.628017Z","shell.execute_reply":"2026-02-18T10:35:17.634986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in [10, 100, 200, 700]:\n    plt.figure(figsize=(20, 5))\n    plt.plot(ws[:, i])\n    plt.grid()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:04.256472Z","iopub.execute_input":"2026-02-13T09:23:04.258708Z","iopub.status.idle":"2026-02-13T09:23:05.238384Z","shell.execute_reply.started":"2026-02-13T09:23:04.258660Z","shell.execute_reply":"2026-02-13T09:23:05.236965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def weights_ini_radians():\n    \n    p = np.arange(512)[:, np.newaxis]\n    i = np.arange(768)[np.newaxis, :]\n    \n    ws = 1 / np.power(10000, (2 * (i // 2)) / np.float32(768))\n    ws = p * ws\n\n    return ws\n\nws = weights_ini_radians()\n\nfor i in [10, 100, 200, 700]:\n    plt.figure(figsize=(20, 5))\n    plt.plot(ws[:, i])\n    plt.grid()\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:05.239926Z","iopub.execute_input":"2026-02-13T09:23:05.240269Z","iopub.status.idle":"2026-02-13T09:23:06.111942Z","shell.execute_reply.started":"2026-02-13T09:23:05.240237Z","shell.execute_reply":"2026-02-13T09:23:06.110859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def weights_ini_sin_and_cos(ws):\n    \n    ws[:, 0::2] = np.sin(ws[:, 0::2])\n    ws[:, 1::2] = np.cos(ws[:, 1::2])\n\n    return ws\n\nws = weights_ini_sin_and_cos(ws)\n\nfor i in [10, 100, 200, 700]:\n    plt.figure(figsize=(20, 5))\n    plt.plot(ws[:, i])\n    plt.grid()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:06.117865Z","iopub.execute_input":"2026-02-13T09:23:06.118323Z","iopub.status.idle":"2026-02-13T09:23:07.123681Z","shell.execute_reply.started":"2026-02-13T09:23:06.118289Z","shell.execute_reply":"2026-02-13T09:23:07.122613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evs = bert.embeddings(input_ids, \n                      attention_mask)\n\nevs.shape, evs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.125306Z","iopub.execute_input":"2026-02-13T09:23:07.126410Z","iopub.status.idle":"2026-02-13T09:23:07.152736Z","shell.execute_reply.started":"2026-02-13T09:23:07.126353Z","shell.execute_reply":"2026-02-13T09:23:07.151730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class OneHeadSelfAttention(nn.Module):\n    \n    def __init__(self, \n                 hidden_size, \n                 dropout_probability):\n        super().__init__()\n\n        self.scale = hidden_size ** 0.5\n        self.q = nn.Linear(hidden_size, hidden_size)\n        self.k = nn.Linear(hidden_size, hidden_size)\n        self.v = nn.Linear(hidden_size, hidden_size)\n        self.dropout = nn.Dropout(dropout_probability)\n\n    def forward(self, x, attention_mask):\n        \n        q = self.q(x)\n        k = self.k(x)\n        v = self.v(x)\n        \n        attention_scores = torch.matmul(q, k.transpose(-2, -1)) / self.scale\n\n        if attention_mask.dim() == 2:\n            attention_mask = attention_mask[:, None, :]\n        attention_mask = (1.0 - attention_mask) * -10000.0\n        attention_scores = attention_scores + attention_mask\n        \n        attention_ps = F.softmax(attention_scores, dim=-1)\n        attention_ps = self.dropout(attention_ps)\n        context = torch.matmul(attention_ps, v)\n        \n        return context\n\nclass MultiHeadSelfAttention(nn.Module):\n    \n    def __init__(self, \n                 hidden_size, \n                 dropout_probability, \n                 num_attention_heads=12):\n        super().__init__()\n        \n        self.num_attention_heads = num_attention_heads\n        self.head_dim = hidden_size // num_attention_heads\n        \n        self.scale = self.head_dim ** 0.5\n        self.q = nn.Linear(hidden_size, hidden_size)\n        self.k = nn.Linear(hidden_size, hidden_size)\n        self.v = nn.Linear(hidden_size, hidden_size)\n        self.dropout = nn.Dropout(dropout_probability)\n\n    def transpose(self, x):\n        \n        new_shape = x.size()[:2] + (self.num_attention_heads, self.head_dim)\n        x = x.view(*new_shape)\n        \n        x = x.permute(0, 2, 1, 3)\n        \n        return x\n\n    def forward(self, x, attention_mask):\n\n        q = self.q(x) ## (N, L, 768)\n        k = self.k(x)\n        v = self.v(x)\n        \n        q = self.transpose(q) ## (N, 12, L, 768/12)\n        k = self.transpose(k)\n        v = self.transpose(v)\n\n        attention_scores = torch.matmul(q, k.transpose(-2, -1)) / self.scale\n\n        if attention_mask.dim() == 2:\n            attention_mask = attention_mask[:, None, None, :]\n        attention_mask = (1.0 - attention_mask) * -10000.0\n        attention_scores = attention_scores + attention_mask\n\n        attention_ps = F.softmax(attention_scores, dim=-1)\n        attention_ps = self.dropout(attention_ps)\n        context = torch.matmul(attention_ps, v)\n\n        context = context.permute(0, 2, 1, 3).contiguous()\n        context = context.view(\n            x.size()[0], x.size()[1], \n            self.num_attention_heads * self.head_dim\n        )\n        \n        return context\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.154027Z","iopub.execute_input":"2026-02-13T09:23:07.154347Z","iopub.status.idle":"2026-02-13T09:23:07.171147Z","shell.execute_reply.started":"2026-02-13T09:23:07.154317Z","shell.execute_reply":"2026-02-13T09:23:07.169898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BertSelfOutput(nn.Module):\n    \n    def __init__(self, hidden_size, dropout_probability):\n        super().__init__()\n        \n        self.dense = nn.Linear(hidden_size, hidden_size)\n        self.norm = nn.LayerNorm(hidden_size, eps=1e-12)\n        self.dropout = nn.Dropout(dropout_probability)\n\n    def forward(self, hidden_states, input_tensor):\n        \n        hidden_states = self.dense(hidden_states)\n        hidden_states = self.dropout(hidden_states)\n        hidden_states = self.norm(hidden_states + input_tensor)\n        \n        return hidden_states\n\nclass BertAttention(nn.Module):\n    \n    def __init__(self, hidden_size, dropout_probability):\n        super().__init__()\n        \n        self.self = MultiHeadSelfAttention(hidden_size, dropout_probability)\n        self.output = BertSelfOutput(hidden_size, dropout_probability)\n\n    def forward(self, x, attention_mask):\n        \n        self_output = self.self(x, attention_mask)\n        attention_output = self.output(self_output, x)\n        \n        return attention_output\n\nclass BertIntermediate(nn.Module):\n    \n    def __init__(self, hidden_size, intermediate_size):\n        super().__init__()\n        \n        self.dense = nn.Linear(hidden_size, intermediate_size)\n        self.intermediate_act_fn = nn.GELU()\n\n    def forward(self, hidden_states):\n        \n        hidden_states = self.dense(hidden_states)\n        hidden_states = self.intermediate_act_fn(hidden_states)\n        \n        return hidden_states\n\nclass BertOutput(nn.Module):\n    \n    def __init__(self, intermediate_size, hidden_size, dropout_probability):\n        super().__init__()\n        \n        self.dense = nn.Linear(intermediate_size, hidden_size)\n        self.LayerNorm = nn.LayerNorm(hidden_size, eps=1e-12)\n        self.dropout = nn.Dropout(dropout_probability)\n\n    def forward(self, hidden_states, input_tensor):\n        \n        hidden_states = self.dense(hidden_states)\n        hidden_states = self.dropout(hidden_states)\n        hidden_states = self.LayerNorm(hidden_states + input_tensor)\n        \n        return hidden_states\n\nclass BertEncoderLayer(nn.Module):\n    \n    def __init__(self, hidden_size=768, intermediate_size=3072, dropout_probability=.1):\n        super().__init__()\n        \n        self.attention = BertAttention(hidden_size, dropout_probability)\n        self.intermediate = BertIntermediate(hidden_size, intermediate_size)\n        self.output = BertOutput(intermediate_size, hidden_size, dropout_probability)\n\n    def forward(self, x, attention_mask):\n\n        attention_output = self.attention(x, attention_mask) ## (N, L, 768) \n        intermediate_output = self.intermediate(attention_output) ## (N, L, 768*4) \n        output = self.output(intermediate_output, attention_output) ## (N, L, 768) \n        \n        return output\n\nencoder = BertEncoderLayer()\nout = encoder(evs, \n              attention_mask)\n\nout.shape, out\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.172883Z","iopub.execute_input":"2026-02-13T09:23:07.173626Z","iopub.status.idle":"2026-02-13T09:23:07.325559Z","shell.execute_reply.started":"2026-02-13T09:23:07.173575Z","shell.execute_reply":"2026-02-13T09:23:07.324322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"out = bert(input_ids, \n           attention_mask, \n           return_dict=False)\n\nout[0].shape, out[0]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.326850Z","iopub.execute_input":"2026-02-13T09:23:07.327218Z","iopub.status.idle":"2026-02-13T09:23:07.872876Z","shell.execute_reply.started":"2026-02-13T09:23:07.327173Z","shell.execute_reply":"2026-02-13T09:23:07.871642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"out[1].shape, out[1]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.874309Z","iopub.execute_input":"2026-02-13T09:23:07.874812Z","iopub.status.idle":"2026-02-13T09:23:07.883568Z","shell.execute_reply.started":"2026-02-13T09:23:07.874763Z","shell.execute_reply":"2026-02-13T09:23:07.882510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xs = torch.rand(4, 5)\n\nml = nn.Linear(5, 1)\nout = ml(xs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.884808Z","iopub.execute_input":"2026-02-13T09:23:07.885214Z","iopub.status.idle":"2026-02-13T09:23:07.893721Z","shell.execute_reply.started":"2026-02-13T09:23:07.885182Z","shell.execute_reply":"2026-02-13T09:23:07.892841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ys = torch.randint(0, 2, (4,), dtype=torch.float32)\n\nloss_fn = nn.BCEWithLogitsLoss()\nloss = loss_fn(out.flatten(), \n               ys)\nloss.backward()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.894965Z","iopub.execute_input":"2026-02-13T09:23:07.895934Z","iopub.status.idle":"2026-02-13T09:23:07.912399Z","shell.execute_reply.started":"2026-02-13T09:23:07.895895Z","shell.execute_reply":"2026-02-13T09:23:07.911386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights = ml.weight.data.clone()\n\nweights\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.913776Z","iopub.execute_input":"2026-02-13T09:23:07.914213Z","iopub.status.idle":"2026-02-13T09:23:07.922628Z","shell.execute_reply.started":"2026-02-13T09:23:07.914157Z","shell.execute_reply":"2026-02-13T09:23:07.921500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grads = ml.weight.grad.data.clone()\n\ngrads\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.923902Z","iopub.execute_input":"2026-02-13T09:23:07.924204Z","iopub.status.idle":"2026-02-13T09:23:07.934683Z","shell.execute_reply.started":"2026-02-13T09:23:07.924174Z","shell.execute_reply":"2026-02-13T09:23:07.933531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = optim.SGD(ml.parameters(), lr=.01, momentum=.9, dampening=0)\noptimizer.step()\n\nml.weight\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:07.936296Z","iopub.execute_input":"2026-02-13T09:23:07.936893Z","iopub.status.idle":"2026-02-13T09:23:08.431256Z","shell.execute_reply.started":"2026-02-13T09:23:07.936843Z","shell.execute_reply":"2026-02-13T09:23:08.430211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"v = torch.zeros_like(weights)\nv = .9 * v + (1 - 0) * grads\n\nprint (weights - .01 * v)\nv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.432737Z","iopub.execute_input":"2026-02-13T09:23:08.433343Z","iopub.status.idle":"2026-02-13T09:23:08.444127Z","shell.execute_reply.started":"2026-02-13T09:23:08.433309Z","shell.execute_reply":"2026-02-13T09:23:08.443003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer.param_groups\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.445611Z","iopub.execute_input":"2026-02-13T09:23:08.445933Z","iopub.status.idle":"2026-02-13T09:23:08.454460Z","shell.execute_reply.started":"2026-02-13T09:23:08.445903Z","shell.execute_reply":"2026-02-13T09:23:08.453385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer.state\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.455970Z","iopub.execute_input":"2026-02-13T09:23:08.456760Z","iopub.status.idle":"2026-02-13T09:23:08.467086Z","shell.execute_reply.started":"2026-02-13T09:23:08.456707Z","shell.execute_reply":"2026-02-13T09:23:08.465926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer.zero_grad()\n\nml.weight.grad\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.468591Z","iopub.execute_input":"2026-02-13T09:23:08.469054Z","iopub.status.idle":"2026-02-13T09:23:08.474845Z","shell.execute_reply.started":"2026-02-13T09:23:08.469006Z","shell.execute_reply":"2026-02-13T09:23:08.473809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"out = ml(xs)\n\nloss = loss_fn(out.flatten(), \n               ys)\nloss = loss\n\nloss.backward()\n\nml.weight.grad\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.476222Z","iopub.execute_input":"2026-02-13T09:23:08.477026Z","iopub.status.idle":"2026-02-13T09:23:08.488917Z","shell.execute_reply.started":"2026-02-13T09:23:08.476975Z","shell.execute_reply":"2026-02-13T09:23:08.487734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer.zero_grad()\n\nfor i in range(10):\n    \n    out = ml(xs)\n    \n    loss = loss_fn(out.flatten(), \n                   ys)\n    loss = loss / 10\n    loss.backward()\n    \n    print (ml.weight.grad)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.490088Z","iopub.execute_input":"2026-02-13T09:23:08.490880Z","iopub.status.idle":"2026-02-13T09:23:08.507479Z","shell.execute_reply.started":"2026-02-13T09:23:08.490833Z","shell.execute_reply":"2026-02-13T09:23:08.506473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.tensor(1e-8, dtype=torch.float16), torch.tensor(1e-8, dtype=torch.float32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.508806Z","iopub.execute_input":"2026-02-13T09:23:08.509145Z","iopub.status.idle":"2026-02-13T09:23:08.519024Z","shell.execute_reply.started":"2026-02-13T09:23:08.509115Z","shell.execute_reply":"2026-02-13T09:23:08.517861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = torch.cpu.amp.GradScaler()\n\nloss = torch.tensor(1e-8)  \nscaler.scale(loss)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.520539Z","iopub.execute_input":"2026-02-13T09:23:08.521367Z","iopub.status.idle":"2026-02-13T09:23:08.529015Z","shell.execute_reply.started":"2026-02-13T09:23:08.521328Z","shell.execute_reply":"2026-02-13T09:23:08.527951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scheduler = get_linear_schedule_with_warmup(\n    optimizer, \n    num_training_steps=2000, \n    num_warmup_steps=400\n)\n\nl = []\nfor step in range(2000):\n    scheduler.step()\n    l += [scheduler.get_lr()[0]]\n\nplt.figure(figsize=(20, 5))\nplt.plot(l)\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.530328Z","iopub.execute_input":"2026-02-13T09:23:08.530710Z","iopub.status.idle":"2026-02-13T09:23:08.769473Z","shell.execute_reply.started":"2026-02-13T09:23:08.530670Z","shell.execute_reply":"2026-02-13T09:23:08.768456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scheduler = get_cosine_schedule_with_warmup(\n    optimizer, \n    num_training_steps=2000, \n    num_warmup_steps=400, \n    num_cycles=8\n)\n\nl = []\nfor step in range(2000):\n    scheduler.step()\n    l += [scheduler.get_lr()[0]]\n\nplt.figure(figsize=(20, 5))\nplt.plot(l)\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:08.770898Z","iopub.execute_input":"2026-02-13T09:23:08.771331Z","iopub.status.idle":"2026-02-13T09:23:09.010448Z","shell.execute_reply.started":"2026-02-13T09:23:08.771281Z","shell.execute_reply":"2026-02-13T09:23:09.009299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class NeuralNet(nn.Module):\n    \n    def __init__(self):\n        super(NeuralNet, self).__init__()\n        \n        self.linear = nn.Linear(5, 5)\n        self.out = nn.Linear(5, 1)\n    \n    def forward(self, \n                x):\n        \n        linear = F.relu(self.linear(x))\n        out = self.out(linear)\n        \n        return out\n\nml = NeuralNet()\n\noptimizer = optim.SGD([\n    {'params': ml.linear.parameters(), 'lr': 0.01}, \n    {'params': ml.out.parameters(), 'lr': 0.001} \n], momentum=0.9)\n\noptimizer.param_groups\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:09.011814Z","iopub.execute_input":"2026-02-13T09:23:09.012134Z","iopub.status.idle":"2026-02-13T09:23:09.029702Z","shell.execute_reply.started":"2026-02-13T09:23:09.012103Z","shell.execute_reply":"2026-02-13T09:23:09.028267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\n    \"../input/competitions/amex-default-prediction/train_data.csv\", \n    nrows=13*16\n)\n\ntrain.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:29:57.466464Z","iopub.execute_input":"2026-02-18T10:29:57.467583Z","iopub.status.idle":"2026-02-18T10:29:57.531360Z","shell.execute_reply.started":"2026-02-18T10:29:57.467489Z","shell.execute_reply":"2026-02-18T10:29:57.530168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cats = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndate = ['S_2']\n\nnums = [c for c in train.columns if c not in cats+date+['customer_ID']]\ncr = \"0000099d6bd597052cdcda90ffabf56573fe9d7c79be5fbac11a8ed792feb62a\"\nxs = train[train[\"customer_ID\"] == cr][nums].values\n\nprint (xs.shape)\nxs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:32:27.963791Z","iopub.execute_input":"2026-02-18T10:32:27.964138Z","iopub.status.idle":"2026-02-18T10:32:27.978014Z","shell.execute_reply.started":"2026-02-18T10:32:27.964108Z","shell.execute_reply":"2026-02-18T10:32:27.976960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.linalg.norm(xs[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:33:27.687578Z","iopub.execute_input":"2026-02-18T10:33:27.687931Z","iopub.status.idle":"2026-02-18T10:33:27.695709Z","shell.execute_reply.started":"2026-02-18T10:33:27.687903Z","shell.execute_reply":"2026-02-18T10:33:27.694428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xs = np.nan_to_num(xs)\n\nnp.linalg.norm(xs[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-18T10:34:22.674867Z","iopub.execute_input":"2026-02-18T10:34:22.675226Z","iopub.status.idle":"2026-02-18T10:34:22.683731Z","shell.execute_reply.started":"2026-02-18T10:34:22.675195Z","shell.execute_reply":"2026-02-18T10:34:22.682494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\n    \"../input/competitions/nfl-big-data-bowl-2020/train.csv\", \n    usecols = ['GameId', 'PlayId', 'Team', \n               'X', 'Y', 'S', 'A'], \n    nrows=22*16\n)\n\ntrain.head(22)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:09.031108Z","iopub.execute_input":"2026-02-13T09:23:09.032143Z","iopub.status.idle":"2026-02-13T09:23:09.076997Z","shell.execute_reply.started":"2026-02-13T09:23:09.032087Z","shell.execute_reply":"2026-02-13T09:23:09.075920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## https://www.kaggle.com/code/robikscube/nfl-big-data-bowl-plotting-player-position  \n\ndef create_football_field(linenumbers=True,\n                          endzones=True,\n                          highlight_line=False,\n                          highlight_line_number=50,\n                          highlighted_name='Line of Scrimmage',\n                          fifty_is_los=False,\n                          figsize=(12*2, 6.33*2)):\n    \n    \"\"\"\n    Function that plots the football field for viewing plays.\n    Allows for showing or hiding endzones.\n    \"\"\"\n    \n    rect = patches.Rectangle((0, 0), 120, 53.3, linewidth=0.1,\n                             edgecolor='r', facecolor='darkgreen', zorder=0,  alpha=0.5)\n\n    fig, ax = plt.subplots(1, figsize=figsize)\n    ax.add_patch(rect)\n\n    plt.plot([10, 10, 10, 20, 20, 30, 30, 40, 40, 50, 50, 60, 60, 70, 70, 80,\n              80, 90, 90, 100, 100, 110, 110, 120, 0, 0, 120, 120],\n             [0, 0, 53.3, 53.3, 0, 0, 53.3, 53.3, 0, 0, 53.3, 53.3, 0, 0, 53.3,\n              53.3, 0, 0, 53.3, 53.3, 0, 0, 53.3, 53.3, 53.3, 0, 0, 53.3],\n             color='white')\n    \n    if fifty_is_los:\n        plt.plot([60, 60], [0, 53.3], color='gold')\n        plt.text(62, 50, '<- Player Yardline at Snap', color='gold')\n    if endzones:\n        ez1 = patches.Rectangle((0, 0), 10, 53.3,\n                                linewidth=0.1,\n                                edgecolor='r',\n                                facecolor='blue',\n                                alpha=0.2,\n                                zorder=0)\n        ez2 = patches.Rectangle((110, 0), 120, 53.3,\n                                linewidth=0.1,\n                                edgecolor='r',\n                                facecolor='blue',\n                                alpha=0.2,\n                                zorder=0)\n        ax.add_patch(ez1)\n        ax.add_patch(ez2)\n    \n    plt.xlim(0, 120)\n    plt.ylim(-5, 58.3)\n    plt.axis('off')\n    \n    if linenumbers:\n        for x in range(20, 110, 10):\n            numb = x\n            if x > 50:\n                numb = 120 - x\n            plt.text(x, 5, str(numb - 10),\n                     horizontalalignment='center',\n                     fontsize=20, \n                     color='white')\n            plt.text(x - 0.95, 53.3 - 5, str(numb - 10),\n                     horizontalalignment='center',\n                     fontsize=20, \n                     color='white', \n                     rotation=180)\n    if endzones:\n        hash_range = range(11, 110)\n    else:\n        hash_range = range(1, 120)\n\n    for x in hash_range:\n        ax.plot([x, x], [0.4, 0.7], color='white')\n        ax.plot([x, x], [53.0, 52.5], color='white')\n        ax.plot([x, x], [22.91, 23.57], color='white')\n        ax.plot([x, x], [29.73, 30.39], color='white')\n\n    if highlight_line:\n        hl = highlight_line_number + 10\n        plt.plot([hl, hl], [0, 53.3], color='yellow')\n        plt.text(hl + 2, 50, '<- {}'.format(highlighted_name),\n                 color='yellow')\n    \n    return fig, ax\n\ncreate_football_field()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:09.078266Z","iopub.execute_input":"2026-02-13T09:23:09.078972Z","iopub.status.idle":"2026-02-13T09:23:09.761968Z","shell.execute_reply.started":"2026-02-13T09:23:09.078938Z","shell.execute_reply":"2026-02-13T09:23:09.760797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## https://www.kaggle.com/code/robikscube/nfl-big-data-bowl-plotting-player-position  \n\nfig, ax = create_football_field()\n\ntrain.query(\"PlayId == 20170907000118 and Team == 'away'\") \\\n    .plot(x='X', y='Y', kind='scatter', ax=ax, color='orange', s=30, legend='Away')\n\ntrain.query(\"PlayId == 20170907000118 and Team == 'home'\") \\\n    .plot(x='X', y='Y', kind='scatter', ax=ax, color='blue', s=30, legend='Home')\n\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:09.763566Z","iopub.execute_input":"2026-02-13T09:23:09.764410Z","iopub.status.idle":"2026-02-13T09:23:10.486206Z","shell.execute_reply.started":"2026-02-13T09:23:09.764359Z","shell.execute_reply":"2026-02-13T09:23:10.484902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mx = train[train['PlayId'] == 20170907000118][['X', 'Y', 'S', 'A']].values\n\nprint (mx.shape)\nmx\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:10.487665Z","iopub.execute_input":"2026-02-13T09:23:10.488042Z","iopub.status.idle":"2026-02-13T09:23:10.500267Z","shell.execute_reply.started":"2026-02-13T09:23:10.488009Z","shell.execute_reply":"2026-02-13T09:23:10.499013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def distance_matrix(X, Y):\n    \n    mx = np.zeros((22, 22))\n    \n    for i in range(22):\n        for j in range(i+1, 22):\n            d = np.sqrt((X[i] - X[j])**2 + (Y[i] - Y[j])**2)\n            \n            mx[i, j] = d\n            mx[j, i] = d\n    \n    return mx\n\nx = train[train['PlayId'] == 20170907000118][['X']].values\ny = train[train['PlayId'] == 20170907000118][['Y']].values\n\nmx = distance_matrix(x, y)\n\nprint (mx.shape)\nmx\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T09:23:10.501340Z","iopub.execute_input":"2026-02-13T09:23:10.501754Z","iopub.status.idle":"2026-02-13T09:23:10.525969Z","shell.execute_reply.started":"2026-02-13T09:23:10.501707Z","shell.execute_reply":"2026-02-13T09:23:10.524745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}