{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n'''\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-15T06:47:31.008026Z","iopub.execute_input":"2023-04-15T06:47:31.008343Z","iopub.status.idle":"2023-04-15T06:47:31.014385Z","shell.execute_reply.started":"2023-04-15T06:47:31.008287Z","shell.execute_reply":"2023-04-15T06:47:31.013644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This libarary is for image augmentation Link: https://github.com/albumentations-team/albumentations","metadata":{}},{"cell_type":"code","source":"#!pip install -U git+https://github.com/albu/albumentations\n!pip install -U albumentations --user \n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:47:43.549516Z","iopub.execute_input":"2023-04-15T06:47:43.549883Z","iopub.status.idle":"2023-04-15T06:47:48.54993Z","shell.execute_reply.started":"2023-04-15T06:47:43.549822Z","shell.execute_reply":"2023-04-15T06:47:48.548902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nfrom tqdm import tqdm#_notebook as tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom functools import reduce\nimport os\nfrom scipy.optimize import minimize\nimport plotly.express as px\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\nfrom torchvision import transforms, utils\n\nPATH = '../input/pku-autonomous-driving/'\nos.listdir(PATH)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-04-15T06:47:56.267515Z","iopub.execute_input":"2023-04-15T06:47:56.267883Z","iopub.status.idle":"2023-04-15T06:47:56.282306Z","shell.execute_reply.started":"2023-04-15T06:47:56.267818Z","shell.execute_reply":"2023-04-15T06:47:56.28134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you want to understand Camera Matrix,the Intrinsic Matrix\nLink : http://ksimek.github.io/2013/08/13/intrinsic/","metadata":{}},{"cell_type":"markdown","source":"## Loading the dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(PATH + 'train.csv')\ntest = pd.read_csv(PATH + 'sample_submission.csv')\n\n# From camera.zip\ncamera_matrix = np.array([[2304.5479, 0,  1686.2379],\n                          [0, 2305.8757, 1354.9849],\n                          [0, 0, 1]], dtype=np.float32)\ncamera_matrix_inv = np.linalg.inv(camera_matrix)\n\ntrain.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:48:17.69496Z","iopub.execute_input":"2023-04-15T06:48:17.695282Z","iopub.status.idle":"2023-04-15T06:48:17.745765Z","shell.execute_reply.started":"2023-04-15T06:48:17.69522Z","shell.execute_reply":"2023-04-15T06:48:17.744757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# set the range or distribution for x and y coordinates\nxmin = 0\nxmax = 10\nymin = 0\nymax = 10\n\n# generate random x and y coordinates within the specified range or distribution\nn_points = 100\ntest = np.random.uniform(xmin, xmax, n_points)\ntrain = np.random.uniform(ymin, ymax, n_points)\n\n# fit a linear regression model to the generated coordinates\nmodel = LinearRegression()\nX = test.reshape(-1, 1)\ny = train.reshape(-1, 1)\nmodel.fit(X, y)\n\n# evaluate the model on a separate set of test data\nn_test_points = 50\ntest_x = np.random.uniform(xmin, xmax, n_test_points)\ntest_y = np.random.uniform(ymin, ymax, n_test_points)\ntest_X = test_x.reshape(-1, 1)\ntest_y_pred = model.predict(test_X)\n\n# compute the evaluation metrics for the model's performance\nmse = mean_squared_error(test_y, test_y_pred)\nr2 = r2_score(test_y, test_y_pred)\n\n# plot the generated coordinates and the regression line\nplt.scatter(test, train, label='Training Data')\nplt.plot(test_x, test_y_pred, color='red', label='Regression Line')\nplt.scatter(test_x, test_y, color='green', label='Test Data')\nplt.xlabel('X Coordinates')\nplt.ylabel('Y Coordinates')\nplt.title('Linear Regression on test and train Coordinates')\nplt.legend()\nplt.show()\n\n# print the evaluation metrics\nprint('Mean Squared Error:', mse)\nprint('R-squared Score:', r2)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:48:35.398243Z","iopub.execute_input":"2023-04-15T06:48:35.398557Z","iopub.status.idle":"2023-04-15T06:48:35.581231Z","shell.execute_reply.started":"2023-04-15T06:48:35.398505Z","shell.execute_reply":"2023-04-15T06:48:35.580401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imread(path, fast_mode=False):\n    img = cv2.imread(path)\n    if not fast_mode and img is not None and len(img.shape) == 3:\n        img = np.array(img[:, :, ::-1])\n    return img\n\nimg = imread(PATH + 'train_images/ID_5f4dac207' + '.jpg')\nIMG_SHAPE = img.shape\n\nplt.figure(figsize=(15,8))\nplt.imshow(img);","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:48:44.273614Z","iopub.execute_input":"2023-04-15T06:48:44.273961Z","iopub.status.idle":"2023-04-15T06:48:45.300553Z","shell.execute_reply.started":"2023-04-15T06:48:44.2739Z","shell.execute_reply":"2023-04-15T06:48:45.299862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extracting data \n PredictionString column contains pose information about all cars\n From the data description:\n The primary data is images of cars and related pose information. The pose information is formatted as    strings, as follows:\n\nmodel type, yaw, pitch, roll, x, y, z\n\nThis function extracts these values:","metadata":{}},{"cell_type":"code","source":"def str2coords(s, names=['id', 'yaw', 'pitch', 'roll', 'x', 'y', 'z']):\n    '''\n    Input:\n        s: PredictionString (e.g. from train dataframe)\n        names: array of what to extract from the string\n    Output:\n        list of dicts with keys from `names`\n    '''\n    coords = []\n    for l in np.array(s.split()).reshape([-1, 7]):\n        coords.append(dict(zip(names, l.astype('float'))))\n        if 'id' in coords[-1]:\n            coords[-1]['id'] = int(coords[-1]['id'])\n    return coords","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:49:17.304819Z","iopub.execute_input":"2023-04-15T06:49:17.305142Z","iopub.status.idle":"2023-04-15T06:49:17.311589Z","shell.execute_reply.started":"2023-04-15T06:49:17.305091Z","shell.execute_reply":"2023-04-15T06:49:17.310797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(PATH + 'train.csv')\ntest = pd.read_csv(PATH + 'sample_submission.csv')\nlens = [len(str2coords(s)) for s in train['PredictionString']]\n\nplt.figure(figsize=(15,6))\nsns.countplot(lens);\nplt.xlabel('Number of cars in image');","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:02.331354Z","iopub.execute_input":"2023-04-15T06:50:02.331684Z","iopub.status.idle":"2023-04-15T06:50:03.380552Z","shell.execute_reply.started":"2023-04-15T06:50:02.33161Z","shell.execute_reply":"2023-04-15T06:50:03.379641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"points_df = pd.DataFrame()\nfor col in ['x', 'y', 'z', 'yaw', 'pitch', 'roll']:\n    arr = []\n    for ps in train['PredictionString']:\n        coords = str2coords(ps)\n        arr += [c[col] for c in coords]\n    points_df[col] = arr\n\nprint('len(points_df)', len(points_df))\npoints_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:15.100571Z","iopub.execute_input":"2023-04-15T06:50:15.100916Z","iopub.status.idle":"2023-04-15T06:50:18.804098Z","shell.execute_reply.started":"2023-04-15T06:50:15.100861Z","shell.execute_reply":"2023-04-15T06:50:18.803298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot position information distribution\nto gain insight into the dataset Let's plot distribution of the position information ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['x'], bins=500);\nplt.xlabel('x')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:26.673747Z","iopub.execute_input":"2023-04-15T06:50:26.674114Z","iopub.status.idle":"2023-04-15T06:50:27.687317Z","shell.execute_reply.started":"2023-04-15T06:50:26.674053Z","shell.execute_reply":"2023-04-15T06:50:27.686558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['y'], bins=500);\nplt.xlabel('y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:34.523561Z","iopub.execute_input":"2023-04-15T06:50:34.523912Z","iopub.status.idle":"2023-04-15T06:50:35.535686Z","shell.execute_reply.started":"2023-04-15T06:50:34.523856Z","shell.execute_reply":"2023-04-15T06:50:35.534676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['z'], bins=500);\nplt.xlabel('z')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:44.680199Z","iopub.execute_input":"2023-04-15T06:50:44.680511Z","iopub.status.idle":"2023-04-15T06:50:45.793185Z","shell.execute_reply.started":"2023-04-15T06:50:44.680455Z","shell.execute_reply":"2023-04-15T06:50:45.792525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['yaw'], bins=500);\nplt.xlabel('yaw')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:50:52.766463Z","iopub.execute_input":"2023-04-15T06:50:52.766783Z","iopub.status.idle":"2023-04-15T06:50:53.778345Z","shell.execute_reply.started":"2023-04-15T06:50:52.766726Z","shell.execute_reply":"2023-04-15T06:50:53.777589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['pitch'], bins=500);\nplt.xlabel('pitch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:51:18.135573Z","iopub.execute_input":"2023-04-15T06:51:18.135919Z","iopub.status.idle":"2023-04-15T06:51:19.831503Z","shell.execute_reply.started":"2023-04-15T06:51:18.135865Z","shell.execute_reply":"2023-04-15T06:51:19.830607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.distplot(points_df['roll'], bins=500);\nplt.xlabel('roll rotated by pi')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:51:30.776855Z","iopub.execute_input":"2023-04-15T06:51:30.777173Z","iopub.status.idle":"2023-04-15T06:51:31.939717Z","shell.execute_reply.started":"2023-04-15T06:51:30.777118Z","shell.execute_reply":"2023-04-15T06:51:31.938935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rotate(x, angle):\n    x = x + angle\n    x = x - (x + np.pi) // (2 * np.pi) * 2 * np.pi\n    return x\n\nplt.figure(figsize=(15,6))\nsns.distplot(points_df['roll'].map(lambda x: rotate(x, np.pi)), bins=500);\nplt.xlabel('roll rotated by pi')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:51:39.003073Z","iopub.execute_input":"2023-04-15T06:51:39.003389Z","iopub.status.idle":"2023-04-15T06:51:40.081231Z","shell.execute_reply.started":"2023-04-15T06:51:39.003332Z","shell.execute_reply":"2023-04-15T06:51:40.080482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2D Visualization","metadata":{}},{"cell_type":"code","source":"def get_img_coords(s):\n    '''\n    Input is a PredictionString (e.g. from train dataframe)\n    Output is two arrays:\n        xs: x coordinates in the image\n        ys: y coordinates in the image\n    '''\n    coords = str2coords(s)\n    xs = [c['x'] for c in coords]\n    ys = [c['y'] for c in coords]\n    zs = [c['z'] for c in coords]\n    P = np.array(list(zip(xs, ys, zs))).T\n    img_p = np.dot(camera_matrix, P).T\n    img_p[:, 0] /= img_p[:, 2]\n    img_p[:, 1] /= img_p[:, 2]\n    #img_p[:, 0] /= zs\n    #img_p[:, 1] /= zs\n    \n    img_xs = img_p[:, 0]\n    img_ys = img_p[:, 1]\n    img_zs = img_p[:, 2] # z = Distance from the camera\n    return img_xs, img_ys\n\nplt.figure(figsize=(14,14))\nplt.imshow(imread(PATH + 'train_images/' + train['ImageId'][500] + '.jpg'))\nplt.scatter(*get_img_coords(train['PredictionString'][500]), color='red', s=100);","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:51:48.044216Z","iopub.execute_input":"2023-04-15T06:51:48.044532Z","iopub.status.idle":"2023-04-15T06:51:49.159188Z","shell.execute_reply.started":"2023-04-15T06:51:48.04447Z","shell.execute_reply":"2023-04-15T06:51:49.15841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xs, ys = [], []\n\nfor ps in train['PredictionString']:\n    x, y = get_img_coords(ps)\n    xs += list(x)\n    ys += list(y)\n\nplt.figure(figsize=(18,18))\nplt.imshow(imread(PATH + 'train_images/' + train['ImageId'][500] + '.jpg'), alpha=0.3)\nplt.scatter(xs, ys, color='red', s=10, alpha=0.2);","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:51:57.467451Z","iopub.execute_input":"2023-04-15T06:51:57.467921Z","iopub.status.idle":"2023-04-15T06:52:00.084525Z","shell.execute_reply.started":"2023-04-15T06:51:57.467718Z","shell.execute_reply":"2023-04-15T06:52:00.083727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zy_slope = LinearRegression()\nX = points_df[['z']]\ny = points_df['y']\nzy_slope.fit(X, y)\nprint('MAE without x:', mean_absolute_error(y, zy_slope.predict(X)))\n\n# Will use this model later\nxzy_slope = LinearRegression()\nX = points_df[['x', 'z']]\ny = points_df['y']\nxzy_slope.fit(X, y)\nprint('MAE with x:', mean_absolute_error(y, xzy_slope.predict(X)))\n\nprint('\\ndy/dx = {:.3f}\\ndy/dz = {:.3f}'.format(*xzy_slope.coef_))","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:52:09.615748Z","iopub.execute_input":"2023-04-15T06:52:09.616081Z","iopub.status.idle":"2023-04-15T06:52:09.640746Z","shell.execute_reply.started":"2023-04-15T06:52:09.616021Z","shell.execute_reply":"2023-04-15T06:52:09.639793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,16))\nplt.xlim(0,500)\nplt.ylim(0,100)\nplt.scatter(points_df['z'], points_df['y'], label='Real points')\nX_line = np.linspace(0,500, 10)\nplt.plot(X_line, zy_slope.predict(X_line.reshape(-1, 1)), color='orange', label='Regression')\nplt.legend()\nplt.xlabel('z coordinate')\nplt.ylabel('y coordinate');","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:52:20.254082Z","iopub.execute_input":"2023-04-15T06:52:20.254376Z","iopub.status.idle":"2023-04-15T06:52:22.085552Z","shell.execute_reply.started":"2023-04-15T06:52:20.254324Z","shell.execute_reply":"2023-04-15T06:52:22.084653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3D Visualization","metadata":{}},{"cell_type":"code","source":"from math import sin, cos\n\n# convert euler angle to rotation matrix\ndef euler_to_Rot(yaw, pitch, roll):\n    Y = np.array([[cos(yaw), 0, sin(yaw)],\n                  [0, 1, 0],\n                  [-sin(yaw), 0, cos(yaw)]])\n    P = np.array([[1, 0, 0],\n                  [0, cos(pitch), -sin(pitch)],\n                  [0, sin(pitch), cos(pitch)]])\n    R = np.array([[cos(roll), -sin(roll), 0],\n                  [sin(roll), cos(roll), 0],\n                  [0, 0, 1]])\n    return np.dot(Y, np.dot(P, R))\n\ndef draw_line(image, points):\n    color = (255, 0, 0)\n    cv2.line(image, tuple(points[0][:2]), tuple(points[3][:2]), color, 16)\n    cv2.line(image, tuple(points[0][:2]), tuple(points[1][:2]), color, 16)\n    cv2.line(image, tuple(points[1][:2]), tuple(points[2][:2]), color, 16)\n    cv2.line(image, tuple(points[2][:2]), tuple(points[3][:2]), color, 16)\n    return image\n\n\ndef draw_points(image, points):\n    for (p_x, p_y, p_z) in points:\n        cv2.circle(image, (p_x, p_y), int(1000 / p_z), (0, 255, 0), -1)\n#         if p_x > image.shape[1] or p_y > image.shape[0]:\n#             print('Point', p_x, p_y, 'is out of image with shape', image.shape)\n    return image\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:52:30.829255Z","iopub.execute_input":"2023-04-15T06:52:30.829568Z","iopub.status.idle":"2023-04-15T06:52:30.977524Z","shell.execute_reply.started":"2023-04-15T06:52:30.829509Z","shell.execute_reply":"2023-04-15T06:52:30.97643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize(img, coords):\n    # You will also need functions from the previous cells\n    x_l = 1.02\n    y_l = 0.80\n    z_l = 2.31\n    \n    img = img.copy()\n    for point in coords:\n        # Get values\n        x, y, z = point['x'], point['y'], point['z']\n        yaw, pitch, roll = -point['pitch'], -point['yaw'], -point['roll']\n        # Math\n        center_point = np.array([x, y, z]).reshape([1,3])\n        Rotation_matrix = euler_to_Rot(yaw, pitch, roll).T#Rotation matrix to transform from car coordinate frame to camera coordinate frame\n        bounding_box = np.array([[x_l, -y_l, -z_l],\n                      [x_l, -y_l, z_l],\n                      [-x_l, -y_l, z_l],\n                      [-x_l, -y_l, -z_l],\n                     ]).T\n        img_cor_points = np.dot(camera_matrix, np.dot(Rotation_matrix,bounding_box)+center_point.T)\n        img_cor_points = img_cor_points.T\n        img_cor_points[:, 0] /= img_cor_points[:, 2]\n        img_cor_points[:, 1] /= img_cor_points[:, 2]\n        img_cor_points = img_cor_points.astype(int)\n        # Drawing\n        img = draw_line(img, img_cor_points)\n        img_point=np.dot(camera_matrix, center_point.T).T\n        img_point[:, 0] /= img_point[:, 2]\n        img_point[:, 1] /=img_point[:, 2]\n        img = draw_points(img,img_point.astype(int))\n    \n    return img\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:52:44.66741Z","iopub.execute_input":"2023-04-15T06:52:44.667735Z","iopub.status.idle":"2023-04-15T06:52:44.678583Z","shell.execute_reply.started":"2023-04-15T06:52:44.667678Z","shell.execute_reply":"2023-04-15T06:52:44.677652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows = 6\n\nfor idx in range(n_rows):\n    fig, axes = plt.subplots(1, 2, figsize=(20,20))\n    img = imread(PATH + 'train_images/' + train['ImageId'].iloc[10+idx] + '.jpg')\n    axes[0].imshow(img)\n    img_vis = visualize(img, str2coords(train['PredictionString'].iloc[10+idx]))\n    axes[1].imshow(img_vis)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:52:51.459447Z","iopub.execute_input":"2023-04-15T06:52:51.459779Z","iopub.status.idle":"2023-04-15T06:53:01.073463Z","shell.execute_reply.started":"2023-04-15T06:52:51.459715Z","shell.execute_reply":"2023-04-15T06:53:01.072614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_WIDTH = 1024\nIMG_HEIGHT = IMG_WIDTH // 16 * 5\nMODEL_SCALE = 8\n\ndef rotate(x, angle):\n    x = x + angle\n    x = x - (x + np.pi) // (2 * np.pi) * 2 * np.pi\n    return x\n\n\ndef _regr_preprocess(regr_dict, flip=False):\n    if flip:\n        for k in ['x', 'pitch', 'roll']:\n            regr_dict[k] = -regr_dict[k]\n    for name in ['x', 'y', 'z']:\n        regr_dict[name] = regr_dict[name] / 100\n    regr_dict['roll'] = rotate(regr_dict['roll'], np.pi)\n    regr_dict['pitch_sin'] = sin(regr_dict['pitch'])\n    regr_dict['pitch_cos'] = cos(regr_dict['pitch'])\n    regr_dict.pop('pitch')\n    regr_dict.pop('id')\n    return regr_dict\n\ndef _regr_back(regr_dict):\n    for name in ['x', 'y', 'z']:\n        regr_dict[name] = regr_dict[name] * 100\n    regr_dict['roll'] = rotate(regr_dict['roll'], -np.pi)\n    \n    pitch_sin = regr_dict['pitch_sin'] / np.sqrt(regr_dict['pitch_sin']**2 + regr_dict['pitch_cos']**2)\n    pitch_cos = regr_dict['pitch_cos'] / np.sqrt(regr_dict['pitch_sin']**2 + regr_dict['pitch_cos']**2)\n    regr_dict['pitch'] = np.arccos(pitch_cos) * np.sign(pitch_sin)\n    return regr_dict\n\ndef preprocess_image(img, flip=False):\n    img = img[img.shape[0] // 2:]\n    bg = np.ones_like(img) * img.mean(1, keepdims=True).astype(img.dtype)\n    bg = bg[:, :img.shape[1] // 6]\n    img = np.concatenate([bg, img, bg], 1)\n    img = cv2.resize(img, (IMG_WIDTH, IMG_HEIGHT))\n    if flip:\n        img = img[:,::-1]\n    return (img / 255).astype('float32')\n\ndef get_mask_and_regr(img, labels, flip=False):\n    mask = np.zeros([IMG_HEIGHT // MODEL_SCALE, IMG_WIDTH // MODEL_SCALE], dtype='float32')\n    regr_names = ['x', 'y', 'z', 'yaw', 'pitch', 'roll']\n    regr = np.zeros([IMG_HEIGHT // MODEL_SCALE, IMG_WIDTH // MODEL_SCALE, 7], dtype='float32')\n    coords = str2coords(labels)\n    xs, ys = get_img_coords(labels)\n    for x, y, regr_dict in zip(xs, ys, coords):\n        x, y = y, x\n        #print(x,img.shape[0] // 2,y, img.shape[1] // 6)\n        x = (x - img.shape[0] // 2) * IMG_HEIGHT / (img.shape[0] // 2) / MODEL_SCALE\n        #x=(x*1/2)*(IMG_HEIGHT / MODEL_SCALE)/(img.shape[0] // 2)\n        x = np.round(x).astype('int')\n        y = (y + img.shape[1] // 6) * IMG_WIDTH / (img.shape[1] * 4/3) / MODEL_SCALE\n        #y=(y* 4/3)*(IMG_WIDTH / MODEL_SCALE)/((img.shape[1] * 3/4) )\n\n        y = np.round(y).astype('int')\n        #print(x,y)\n\n        if x >= 0 and x < IMG_HEIGHT // MODEL_SCALE and y >= 0 and y < IMG_WIDTH // MODEL_SCALE:\n            mask[x, y] = 1\n            regr_dict = _regr_preprocess(regr_dict, flip)\n            regr[x, y] = [regr_dict[n] for n in sorted(regr_dict)]\n    if flip:\n        mask = np.array(mask[:,::-1])\n        regr = np.array(regr[:,::-1])\n    return mask, regr","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:53:12.259905Z","iopub.execute_input":"2023-04-15T06:53:12.260308Z","iopub.status.idle":"2023-04-15T06:53:12.281705Z","shell.execute_reply.started":"2023-04-15T06:53:12.26024Z","shell.execute_reply":"2023-04-15T06:53:12.280807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img0 = imread(PATH + 'train_images/' + train['ImageId'][500] + '.jpg')\nimg = preprocess_image(img0,flip=True)\n\nmask, regr = get_mask_and_regr(img0, train['PredictionString'][0],flip=True)\n\nprint('img.shape', img.shape, 'std:', np.std(img))\nprint('mask.shape', mask.shape, 'std:', np.std(mask))\nprint('regr.shape', regr.shape, 'std:', np.std(regr))\nprint(img[:,::-1].shape)\n\nplt.figure(figsize=(16,16))\nplt.title('Processed image')\nplt.imshow(img)\nplt.show()\n\nplt.figure(figsize=(16,16))\nplt.title('Processed flip image')\nplt.imshow(img[:,::-1])\nplt.show()\n\nplt.figure(figsize=(16,16))\nplt.title('Detection Mask')\nplt.imshow(mask)\nplt.show()\n\nplt.figure(figsize=(16,16))\nplt.title('Yaw values')\nplt.imshow(regr[:,:,-2])\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:53:37.743123Z","iopub.execute_input":"2023-04-15T06:53:37.743465Z","iopub.status.idle":"2023-04-15T06:53:39.292731Z","shell.execute_reply.started":"2023-04-15T06:53:37.743407Z","shell.execute_reply":"2023-04-15T06:53:39.291919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define functions to convert back from 2d map to 3d coordinates and angles","metadata":{}},{"cell_type":"code","source":"DISTANCE_THRESH_CLEAR = 2\n\ndef convert_3d_to_2d(x, y, z, fx = 2304.5479, fy = 2305.8757, cx = 1686.2379, cy = 1354.9849):\n    # stolen from https://www.kaggle.com/theshockwaverider/eda-visualization-baseline\n    return x * fx / z + cx, y * fy / z + cy\n\ndef optimize_xy(r, c, x0, y0, z0, flipped=False):\n    def distance_fn(xyz):\n        x, y, z = xyz\n        xx = -x if flipped else x\n        slope_err = (xzy_slope.predict([[xx,z]])[0] - y)**2\n        x, y = convert_3d_to_2d(x, y, z)\n        y, x = x, y\n        x = (x - IMG_SHAPE[0] // 2) * IMG_HEIGHT / (IMG_SHAPE[0] // 2) / MODEL_SCALE\n        y = (y + IMG_SHAPE[1] // 6) * IMG_WIDTH / (IMG_SHAPE[1] * 4 / 3) / MODEL_SCALE\n        return max(0.2, (x-r)**2 + (y-c)**2) + max(0.4, slope_err)\n    \n    res = minimize(distance_fn, [x0, y0, z0], method='Powell')\n    x_new, y_new, z_new = res.x\n    return x_new, y_new, z_new\n\ndef clear_duplicates(coords):\n    for c1 in coords:\n        xyz1 = np.array([c1['x'], c1['y'], c1['z']])\n        for c2 in coords:\n            xyz2 = np.array([c2['x'], c2['y'], c2['z']])\n            distance = np.sqrt(((xyz1 - xyz2)**2).sum())\n            if distance < DISTANCE_THRESH_CLEAR:\n                if c1['confidence'] < c2['confidence']:\n                    c1['confidence'] = -1\n    return [c for c in coords if c['confidence'] > 0]\n\ndef extract_coords(prediction, flipped=False):\n    logits = prediction[0]\n    \n    regr_output = prediction[1:]\n    points = np.argwhere(logits > 0)\n    col_names = sorted(['x', 'y', 'z', 'yaw', 'pitch_sin', 'pitch_cos', 'roll'])\n    coords = []\n    for r, c in points:\n        regr_dict = dict(zip(col_names, regr_output[:, r, c]))\n        coords.append(_regr_back(regr_dict))\n        coords[-1]['confidence'] = 1 / (1 + np.exp(-logits[r, c]))\n        coords[-1]['x'], coords[-1]['y'], coords[-1]['z'] = \\\n                optimize_xy(r, c,\n                            coords[-1]['x'],\n                            coords[-1]['y'],\n                            coords[-1]['z'], flipped)\n    coords = clear_duplicates(coords)\n    return coords\n\ndef coords2str(coords, names=['yaw', 'pitch', 'roll', 'x', 'y', 'z', 'confidence']):\n    s = []\n    for c in coords:\n        for n in names:\n            s.append(str(c.get(n, 0)))\n    return ' '.join(s)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:56:08.511746Z","iopub.execute_input":"2023-04-15T06:56:08.512068Z","iopub.status.idle":"2023-04-15T06:56:08.531947Z","shell.execute_reply.started":"2023-04-15T06:56:08.512013Z","shell.execute_reply":"2023-04-15T06:56:08.530928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in range(2):\n    fig, axes = plt.subplots(1, 2, figsize=(20,20))\n    \n    for ax_i in range(2):\n        img0 = imread(PATH + 'train_images/' + train['ImageId'].iloc[10+idx] + '.jpg')\n        if ax_i == 1:\n            img0 = img0[:,::-1]\n        img = preprocess_image(img0, ax_i==1)\n        mask, regr = get_mask_and_regr(img0, train['PredictionString'][10+idx], ax_i==1)\n        \n        regr = np.rollaxis(regr, 2, 0)\n        coords = extract_coords(np.concatenate([mask[None], regr], 0), ax_i==1)\n        \n        axes[ax_i].set_title('Flip = {}'.format(ax_i==1))\n        axes[ax_i].imshow(visualize(img0, coords))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:56:16.32642Z","iopub.execute_input":"2023-04-15T06:56:16.326728Z","iopub.status.idle":"2023-04-15T06:56:21.086411Z","shell.execute_reply.started":"2023-04-15T06:56:16.326674Z","shell.execute_reply":"2023-04-15T06:56:21.08561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating the Pytorch model\nthe model is U-net model in Pytorch </br>\nRefer to this link for implementation detials: https://github.com/milesial/Pytorch-UNet\nthis main idea of the model to use U-net to predict the center point of the car and regress the rotation angles .","metadata":{}},{"cell_type":"code","source":"!pip install efficientnet-pytorch\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:56:29.215178Z","iopub.execute_input":"2023-04-15T06:56:29.215489Z","iopub.status.idle":"2023-04-15T06:56:34.945817Z","shell.execute_reply.started":"2023-04-15T06:56:29.215433Z","shell.execute_reply":"2023-04-15T06:56:34.944716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:56:40.278229Z","iopub.execute_input":"2023-04-15T06:56:40.278559Z","iopub.status.idle":"2023-04-15T06:56:40.283861Z","shell.execute_reply.started":"2023-04-15T06:56:40.2785Z","shell.execute_reply":"2023-04-15T06:56:40.283058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class double_conv(nn.Module):\n    '''(conv => BN => ReLU) * 2'''\n    def __init__(self, in_ch, out_ch):\n        super(double_conv, self).__init__()\n        self.conv = nn.Sequential(\n            nn.Conv2d(in_ch, out_ch, 3, padding=1),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_ch, out_ch, 3, padding=1),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True)\n        )\n\n    def forward(self, x):\n        x = self.conv(x)\n        return x\n\nclass up(nn.Module):\n    def __init__(self, in_ch, out_ch, bilinear=True):\n        super(up, self).__init__()\n\n        #  would be a nice idea if the upsampling could be learned too,\n        #  but my machine do not have enough memory to handle all those weights\n        if bilinear:\n            self.up = nn.Upsample(scale_factor=2, mode='bilinear', align_corners=True)\n        else:\n            self.up = nn.ConvTranspose2d(in_ch//2, in_ch//2, 2, stride=2)\n\n        self.conv = double_conv(in_ch, out_ch)\n\n    def forward(self, x1, x2=None):\n        x1 = self.up(x1)\n        \n        # input is CHW\n        diffY = x2.size()[2] - x1.size()[2]\n        diffX = x2.size()[3] - x1.size()[3]\n\n        x1 = F.pad(x1, (diffX // 2, diffX - diffX//2,\n                        diffY // 2, diffY - diffY//2))\n        \n        # for padding issues, see \n        # https://github.com/HaiyongJiang/U-Net-Pytorch-Unstructured-Buggy/commit/0e854509c2cea854e247a9c615f175f76fbb2e3a\n        # https://github.com/xiaopeng-liao/Pytorch-UNet/commit/8ebac70e633bac59fc22bb5195e513d5832fb3bd\n        \n        if x2 is not None:\n            x = torch.cat([x2, x1], dim=1)\n        else:\n            x = x1\n        x = self.conv(x)\n        return x\n\ndef get_mesh(batch_size, shape_x, shape_y):\n    mg_x, mg_y = np.meshgrid(np.linspace(0, 1, shape_y), np.linspace(0, 1, shape_x))\n    mg_x = np.tile(mg_x[None, None, :, :], [batch_size, 1, 1, 1]).astype('float32')\n    mg_y = np.tile(mg_y[None, None, :, :], [batch_size, 1, 1, 1]).astype('float32')\n    mesh = torch.cat([torch.tensor(mg_x).to(device), torch.tensor(mg_y).to(device)], 1)\n    return mesh\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:57:42.035604Z","iopub.execute_input":"2023-04-15T06:57:42.03594Z","iopub.status.idle":"2023-04-15T06:57:42.051781Z","shell.execute_reply.started":"2023-04-15T06:57:42.035885Z","shell.execute_reply":"2023-04-15T06:57:42.050856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyUNet(nn.Module):\n    '''Mixture of previous classes'''\n    def __init__(self, n_classes):\n        super(MyUNet, self).__init__()\n        self.base_model = EfficientNet.from_pretrained('efficientnet-b0')\n        \n        self.conv0 = double_conv(5, 64)\n        self.conv1 = double_conv(64, 128)\n        self.conv2 = double_conv(128, 512)\n        self.conv3 = double_conv(512, 1024)\n        \n        self.mp = nn.MaxPool2d(2)\n        \n        self.up1 = up(1282 + 1024, 512)\n        self.up2 = up(512 + 512, 256)\n        self.outc = nn.Conv2d(256, n_classes, 1)\n\n    def forward(self, x):\n        batch_size = x.shape[0]\n        mesh1 = get_mesh(batch_size, x.shape[2], x.shape[3])\n        x0 = torch.cat([x, mesh1], 1)\n        x1 = self.mp(self.conv0(x0))\n        x2 = self.mp(self.conv1(x1))\n        x3 = self.mp(self.conv2(x2))\n        x4 = self.mp(self.conv3(x3))\n        \n        x_center = x[:, :, :, IMG_WIDTH // 8: -IMG_WIDTH // 8]\n        feats = self.base_model.extract_features(x_center)\n        bg = torch.zeros([feats.shape[0], feats.shape[1], feats.shape[2], feats.shape[3] // 8]).to(device)\n        feats = torch.cat([bg, feats, bg], 3)\n        \n        # Add positional info\n        mesh2 = get_mesh(batch_size, feats.shape[2], feats.shape[3])\n        feats = torch.cat([feats, mesh2], 1)\n        \n        x = self.up1(feats, x4)\n        x = self.up2(x, x3)\n        x = self.outc(x)\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:57:51.630348Z","iopub.execute_input":"2023-04-15T06:57:51.630823Z","iopub.status.idle":"2023-04-15T06:57:51.644681Z","shell.execute_reply.started":"2023-04-15T06:57:51.6306Z","shell.execute_reply":"2023-04-15T06:57:51.643453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\n\nn_epochs = 10\n\nmodel = MyUNet(8).to(device)\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nexp_lr_scheduler = lr_scheduler.StepLR(optimizer, step_size=max(n_epochs, 10) * len(train_loader) // 3, gamma=0.1)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:06.956107Z","iopub.execute_input":"2023-04-15T06:58:06.956448Z","iopub.status.idle":"2023-04-15T06:58:07.339607Z","shell.execute_reply.started":"2023-04-15T06:58:06.956391Z","shell.execute_reply":"2023-04-15T06:58:07.338476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The Loss function\nif you wana know more about custom loss in Pytorch  refer to this link : https://cs230.stanford.edu/blog/pytorch/","metadata":{}},{"cell_type":"code","source":"def criterion(prediction, mask, regr, size_average=True):\n    # Binary mask loss\n    pred_mask = torch.sigmoid(prediction[:, 0])\n#     mask_loss = mask * (1 - pred_mask)**2 * torch.log(pred_mask + 1e-12) + (1 - mask) * pred_mask**2 * torch.log(1 - pred_mask + 1e-12)\n    mask_loss = mask * torch.log(pred_mask + 1e-12) + (1 - mask) * torch.log(1 - pred_mask + 1e-12)\n    mask_loss = -mask_loss.mean(0).sum()\n    \n    # Regression L1 loss\n    pred_regr = prediction[:, 1:]\n    regr_loss = (torch.abs(pred_regr - regr).sum(1) * mask).sum(1).sum(1) / mask.sum(1).sum(1)\n    regr_loss = regr_loss.mean(0)\n    \n    # Sum\n    loss = mask_loss + regr_loss\n    if not size_average:\n        loss *= prediction.shape[0]\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:17.870308Z","iopub.execute_input":"2023-04-15T06:58:17.870646Z","iopub.status.idle":"2023-04-15T06:58:17.877712Z","shell.execute_reply.started":"2023-04-15T06:58:17.870568Z","shell.execute_reply":"2023-04-15T06:58:17.876708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(epoch, history=None):\n    model.train()\n\n    for batch_idx, (img_batch, mask_batch, regr_batch) in enumerate(tqdm(train_loader)):\n        img_batch = img_batch.to(device)\n        mask_batch = mask_batch.to(device)\n        regr_batch = regr_batch.to(device)\n        \n        optimizer.zero_grad()\n        output = model(img_batch)\n        loss = criterion(output, mask_batch, regr_batch)\n        if history is not None:\n            history.loc[epoch + batch_idx / len(train_loader), 'train_loss'] = loss.data.cpu().numpy()\n        \n        loss.backward()\n        \n        optimizer.step()\n        exp_lr_scheduler.step()\n    \n    print('Train Epoch: {} \\tLR: {:.6f}\\tLoss: {:.6f}'.format(\n        epoch,\n        optimizer.state_dict()['param_groups'][0]['lr'],\n        loss.data))\n\ndef evaluate_model(epoch, history=None):\n    model.eval()\n    loss = 0\n    \n    with torch.no_grad():\n        for img_batch, mask_batch, regr_batch in dev_loader:\n            img_batch = img_batch.to(device)\n            mask_batch = mask_batch.to(device)\n            regr_batch = regr_batch.to(device)\n\n            output = model(img_batch)\n\n            loss += criterion(output, mask_batch, regr_batch, size_average=False).data\n    \n    loss /= len(dev_loader.dataset)\n    \n    if history is not None:\n        history.loc[epoch, 'dev_loss'] = loss.cpu().numpy()\n    \n    print('Dev loss: {:.4f}'.format(loss))","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:26.240071Z","iopub.execute_input":"2023-04-15T06:58:26.240388Z","iopub.status.idle":"2023-04-15T06:58:26.252048Z","shell.execute_reply.started":"2023-04-15T06:58:26.240333Z","shell.execute_reply":"2023-04-15T06:58:26.250869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img_coords(s):\n    '''\n    Input is a PredictionString (e.g. from train dataframe)\n    Output is two arrays:\n        xs: x coordinates in the image (row)\n        ys: y coordinates in the image (column)\n    '''\n    coords = str2coords(s)\n    xs = [c['x'] for c in coords]\n    ys = [c['y'] for c in coords]\n    zs = [c['z'] for c in coords]\n    P = np.array(list(zip(xs, ys, zs))).T\n    img_p = np.dot(camera_matrix, P).T\n    img_p[:, 0] /= img_p[:, 2]\n    img_p[:, 1] /= img_p[:, 2]\n    img_xs = img_p[:, 0]\n    img_ys = img_p[:, 1]\n    img_zs = img_p[:, 2] # z = Distance from the camera\n    return img_xs, img_ys\n\nplt.figure(figsize=(14,14))\nplt.imshow(imread(PATH + 'train_images/' + train['ImageId'][2217] + '.jpg'))\nplt.scatter(*get_img_coords(train['PredictionString'][2217]), color='red', s=100);","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:34.987662Z","iopub.execute_input":"2023-04-15T06:58:34.988006Z","iopub.status.idle":"2023-04-15T06:58:36.15785Z","shell.execute_reply.started":"2023-04-15T06:58:34.98794Z","shell.execute_reply":"2023-04-15T06:58:36.156765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Road points\nroad_width = 3\nroad_xs = [-road_width, road_width, road_width, -road_width, -road_width]\nroad_ys = [0, 0, 500, 500, 0]\n\nplt.figure(figsize=(16,16))\nplt.axes().set_aspect(1)\nplt.xlim(-50,50)\nplt.ylim(0,100)\n\n# View road\nplt.fill(road_xs, road_ys, alpha=0.2, color='gray')\nplt.plot([road_width/2,road_width/2], [0,100], alpha=0.4, linewidth=4, color='white', ls='--')\nplt.plot([-road_width/2,-road_width/2], [0,100], alpha=0.4, linewidth=4, color='white', ls='--')\n# View cars\nplt.scatter(points_df['x'], np.sqrt(points_df['z']**2 + points_df['y']**2), color='red', s=10, alpha=0.1);","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:44.336006Z","iopub.execute_input":"2023-04-15T06:58:44.336434Z","iopub.status.idle":"2023-04-15T06:58:45.277081Z","shell.execute_reply.started":"2023-04-15T06:58:44.336372Z","shell.execute_reply":"2023-04-15T06:58:45.2754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(epoch, history=None):\n    model.train()\n    def closure():\n        loss = loss_function(output, model(input))\n        loss.backward()\n        return loss\n  \n    for batch_idx, (img_batch, mask_batch, regr_batch) in enumerate(tqdm(train_loader)):\n        img_batch = img_batch.to(device)\n        mask_batch = mask_batch.to(device)\n        regr_batch = regr_batch.to(device)\n        \n        \n        output = model(img_batch)\n        loss = criterion(output, mask_batch, regr_batch)\n        if history is not None:\n            history.loc[epoch + batch_idx / len(train_loader), 'train_loss'] = loss.data.cpu().numpy()\n        \n        loss.backward()\n        optimizer.first_step(zero_grad=True)\n        \n\n        preds_second = model(img_batch)\n        loss_second = criterion(preds_second, mask_batch, regr_batch)\n            \n        loss_second.backward()\n        optimizer.second_step(zero_grad=True)\n\n\n        #optimizer.step()\n        exp_lr_scheduler.step()\n    \n    print('Train Epoch: {} \\tLR: {:.6f}\\tLoss: {:.6f}'.format(\n        epoch,\n        optimizer.state_dict()['param_groups'][0]['lr'],\n        loss.data))\n\ndef evaluate_model(epoch, history=None):\n    model.eval()\n    loss = 0\n    \n    with torch.no_grad():\n        for img_batch, mask_batch, regr_batch in dev_loader:\n            img_batch = img_batch.to(device)\n            mask_batch = mask_batch.to(device)\n            regr_batch = regr_batch.to(device)\n\n            output = model(img_batch)\n\n            loss += criterion(output, mask_batch, regr_batch, size_average=False).data\n    \n    loss /= len(dev_loader.dataset)\n    \n    if history is not None:\n        history.loc[epoch, 'dev_loss'] = loss.cpu().numpy()\n    \n    print('Dev loss: {:.4f}'.format(loss))","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:58:53.600607Z","iopub.execute_input":"2023-04-15T06:58:53.600962Z","iopub.status.idle":"2023-04-15T06:58:53.613917Z","shell.execute_reply.started":"2023-04-15T06:58:53.600904Z","shell.execute_reply":"2023-04-15T06:58:53.613052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CarDataset(Dataset):\n    \"\"\"Car dataset.\"\"\"\n\n    def __init__(self, dataframe, root_dir, training=True, transform=None,hasIDs=False):\n        self.df = dataframe\n        self.root_dir = root_dir\n        self.transform = transform\n        self.training = training\n        self.hasIDs=hasIDs\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        \n        # Get image name\n        idx, labels = self.df.values[idx]\n        img_name = self.root_dir.format(idx)\n        \n        # Augmentation\n        flip = False\n        if self.training:\n            flip = np.random.randint(10) == 1\n        \n        # Read image\n        img0 = imread(img_name, True)\n        img = preprocess_image(img0, flip=flip)\n        img = np.rollaxis(img, 2, 0)\n        \n        # Get mask and regression maps\n        mask, regr = get_mask_and_regr(img0, labels, flip=flip)\n        regr = np.rollaxis(regr, 2, 0)\n        if self.hasIDs:\n            return [idx,img]\n        return [img, mask, regr]","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:59:05.131852Z","iopub.execute_input":"2023-04-15T06:59:05.132201Z","iopub.status.idle":"2023-04-15T06:59:05.149823Z","shell.execute_reply.started":"2023-04-15T06:59:05.132155Z","shell.execute_reply":"2023-04-15T06:59:05.14829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_dir = PATH + 'train_images/{}.jpg'\ntest_images_dir = PATH + 'test_images/{}.jpg'\n\ndf_train, df_dev = train_test_split(train, test_size=0.01, random_state=42)\ndf_test = test\n\n# Create dataset objects\ntrain_dataset = CarDataset(df_train, train_images_dir, training=True)\ndev_dataset = CarDataset(df_dev, train_images_dir, training=False)\ndev_dataset2 = CarDataset(df_dev, train_images_dir, training=False,hasIDs=True)\ntest_dataset = CarDataset(df_test, test_images_dir, training=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:59:12.623898Z","iopub.execute_input":"2023-04-15T06:59:12.624205Z","iopub.status.idle":"2023-04-15T06:59:12.634997Z","shell.execute_reply.started":"2023-04-15T06:59:12.624149Z","shell.execute_reply":"2023-04-15T06:59:12.634106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 4\nval_batch_size=6\n# Create data generators - they will produce batches\ntrain_loader = DataLoader(dataset=train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=4)\ndev_loader = DataLoader(dataset=dev_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=0)\ndev_loader2 = DataLoader(dataset=dev_dataset2, batch_size=val_batch_size, shuffle=False, num_workers=4)\ntest_loader = DataLoader(dataset=test_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=0)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:59:20.457801Z","iopub.execute_input":"2023-04-15T06:59:20.458135Z","iopub.status.idle":"2023-04-15T06:59:20.464228Z","shell.execute_reply.started":"2023-04-15T06:59:20.458082Z","shell.execute_reply":"2023-04-15T06:59:20.463389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, mask, regr = dev_dataset[0]\n\nplt.figure(figsize=(16,16))\nplt.title('Input image')\nplt.imshow(np.rollaxis(img, 0, 3))\nplt.show()\n\nplt.figure(figsize=(16,16))\nplt.title('Ground truth mask')\nplt.imshow(mask)\nplt.show()\n\noutput = model(torch.tensor(img[None]).to(device))\nlogits = output[0,0].data.cpu().numpy()\n\nplt.figure(figsize=(16,16))\nplt.title('Model predictions')\nplt.imshow(logits)\nplt.show()\n\nplt.figure(figsize=(16,16))\nplt.title('Model predictions thresholded')\nplt.imshow(logits > 0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T06:59:33.254177Z","iopub.execute_input":"2023-04-15T06:59:33.254488Z","iopub.status.idle":"2023-04-15T06:59:35.435682Z","shell.execute_reply.started":"2023-04-15T06:59:33.25443Z","shell.execute_reply":"2023-04-15T06:59:35.434775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the Model \nTraing the model took more than 15 hrs not continous by saving checkpoints and retrain on Colab for more than 30 ephocs and increasing the augmentation ratio with every retraining of the model  ","metadata":{}},{"cell_type":"markdown","source":"## Visualize some predictions of the model","metadata":{}},{"cell_type":"code","source":"\n       \npredictions = []\n\ntest_loader = DataLoader(dataset=test_dataset, batch_size=1, shuffle=False, num_workers=4)\n\nmodel.eval()\n\nfor img, _, _ in tqdm(test_loader):\n    with torch.no_grad():\n        output = model(img.to(device))\n    output = output.data.cpu().numpy()\n    for out in output:\n        coords = extract_coords(out)\n        s = coords2str(coords)\n        predictions.append(s)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest = pd.read_csv(PATH + 'sample_submission.csv')\ntest['PredictionString'] = predictions\ntest.to_csv('predictions.csv', index=False)\ntest.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Federated Learning**","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade pip\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:07:58.173992Z","iopub.execute_input":"2023-04-15T07:07:58.174315Z","iopub.status.idle":"2023-04-15T07:08:13.205695Z","shell.execute_reply.started":"2023-04-15T07:07:58.174252Z","shell.execute_reply":"2023-04-15T07:08:13.204706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade numpy","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:08:22.554837Z","iopub.execute_input":"2023-04-15T07:08:22.555159Z","iopub.status.idle":"2023-04-15T07:08:32.585714Z","shell.execute_reply.started":"2023-04-15T07:08:22.555103Z","shell.execute_reply":"2023-04-15T07:08:32.584553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torch torchvision numpy matplotlib\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:08:49.436854Z","iopub.execute_input":"2023-04-15T07:08:49.43719Z","iopub.status.idle":"2023-04-15T07:08:54.270084Z","shell.execute_reply.started":"2023-04-15T07:08:49.437124Z","shell.execute_reply":"2023-04-15T07:08:54.269071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\n# Define the neural network model\nclass AutonomousDrivingModel(nn.Module):\n    def __init__(self):\n        super(AutonomousDrivingModel, self).__init__()\n        self.conv1 = nn.Conv2d(3, 16, 3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(16, 32, 3, padding=1)\n        self.fc1 = nn.Linear(32 * 8 * 8, 64)\n        self.fc2 = nn.Linear(64, 10)\n        self.fc3 = nn.Linear(10, 1)\n\n    def forward(self, x):\n        x = self.pool(torch.relu(self.conv1(x)))\n        x = self.pool(torch.relu(self.conv2(x)))\n        x = x.view(-1, 32 * 8 * 8)\n        x = torch.relu(self.fc1(x))\n        x = torch.relu(self.fc2(x))\n        x = self.fc3(x)\n        return x\n\n# Instantiate the model\nmodel = AutonomousDrivingModel()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:09:01.331327Z","iopub.execute_input":"2023-04-15T07:09:01.331828Z","iopub.status.idle":"2023-04-15T07:09:01.34812Z","shell.execute_reply.started":"2023-04-15T07:09:01.331601Z","shell.execute_reply":"2023-04-15T07:09:01.347271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the loss function and optimizer\ncriterion = nn.MSELoss()\noptimizer = torch.optim.SGD(model.parameters(), lr=0.01)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:09:09.443523Z","iopub.execute_input":"2023-04-15T07:09:09.443888Z","iopub.status.idle":"2023-04-15T07:09:09.450218Z","shell.execute_reply.started":"2023-04-15T07:09:09.443827Z","shell.execute_reply":"2023-04-15T07:09:09.449189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model_federated(train_sets, model, criterion, optimizer, num_epochs):\n    for epoch in range(num_epochs):\n        # Initialize the model weights\n        model_weights = model.state_dict()\n\n        # Average the model weights across all vehicles\n        for train_set in train_sets:\n            # Create a data loader for the current vehicle's training set\n            train_loader = DataLoader(train_set, batch_size=64, shuffle=True)\n\n            # Train the model on the current vehicle's training set\n            for inputs, labels in train_loader:\n                # Zero the parameter gradients\n                optimizer.zero_grad()\n\n                # Forward + backward + optimize\n                outputs = model(inputs)\n                loss = criterion(outputs, labels)\n                loss.backward()\n                optimizer.step()\n\n            # Update the model weights for the current vehicle\n            vehicle_weights = model.state_dict()\n            for key in model_weights.keys():\n                model_weights[key] += vehicle_weights[key]\n        for key in model_weights.keys():\n            model_weights[key] /= len(train_sets)\n        model.load_state_dict(model_weights)\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:09:22.240807Z","iopub.execute_input":"2023-04-15T07:09:22.241113Z","iopub.status.idle":"2023-04-15T07:09:22.248578Z","shell.execute_reply.started":"2023-04-15T07:09:22.241062Z","shell.execute_reply":"2023-04-15T07:09:22.24781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model using federated learning\nnum_epochs = 10\nmodel\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:09:27.045116Z","iopub.execute_input":"2023-04-15T07:09:27.04542Z","iopub.status.idle":"2023-04-15T07:09:27.050986Z","shell.execute_reply.started":"2023-04-15T07:09:27.045367Z","shell.execute_reply":"2023-04-15T07:09:27.050222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport numpy as np\nimport random\n\n# Generate random dataset\nnum_images = 1000\nimage_size = 64\n\nX = np.zeros((num_images, 3, image_size, image_size))\ny = np.zeros(num_images)\n\nfor i in range(num_images):\n    # Generate random image\n    image = np.random.rand(3, image_size, image_size)\n    X[i] = image\n    \n    # Assign random label (0 or 1)\n    label = random.randint(0, 1)\n    y[i] = label\n\n# Define model architecture\nclass Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, stride=1, padding=1)\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, stride=1, padding=1)\n        self.fc1 = nn.Linear(64 * (image_size//4) * (image_size//4), 128)\n        self.fc2 = nn.Linear(128, 2)\n\n    def forward(self, x):\n        x = self.pool(torch.relu(self.conv1(x)))\n        x = self.pool(torch.relu(self.conv2(x)))\n        x = x.view(-1, 64 * (image_size//4) * (image_size//4))\n        x = torch.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x\n\n# Initialize model, loss function, and optimizer\nmodel = Net()\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# Split dataset into training and testing sets\nsplit_ratio = 0.8\nsplit_index = int(num_images * split_ratio)\nX_train, y_train = X[:split_index], y[:split_index]\nX_test, y_test = X[split_index:], y[split_index:]\n\n# Train model\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    running_loss = 0.0\n    for i in range(split_index):\n        # Get inputs and labels\n        inputs = torch.FloatTensor(X_train[i])\n        label = torch.LongTensor(np.array([y_train[i]]))\n\n        # Zero the parameter gradients\n        optimizer.zero_grad()\n\n        # Forward + backward + optimize\n        outputs = model(inputs.unsqueeze(0))\n        loss = criterion(outputs, label)\n        loss.backward()\n        optimizer.step()\n\n        # Print statistics\n        running_loss += loss.item()\n        if i % 100 == 99:    # Print every 100 mini-batches\n            print('[Epoch %d, Batch %5d] loss: %.3f' %\n                  (epoch + 1, i + 1, running_loss / 100))\n            running_loss = 0.0\n\nprint('Finished Training')\n\n# Evaluate model on testing set\ncorrect = 0\ntotal = 0\nwith torch.no_grad():\n    for i in range(num_images - split_index):\n        # Get inputs and labels\n        inputs = torch.FloatTensor(X_test[i])\n        label = torch.LongTensor(np.array([y_test[i]]))\n\n        # Predict label\n        outputs = model(inputs.unsqueeze(0))\n        _, predicted = torch.max(outputs.data, 1)\n\n        # Update accuracy\n        total += label.size(0)\n        correct += (predicted == label).sum().item()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:30:33.56336Z","iopub.execute_input":"2023-04-15T07:30:33.563979Z","iopub.status.idle":"2023-04-15T07:35:57.329901Z","shell.execute_reply.started":"2023-04-15T07:30:33.563915Z","shell.execute_reply":"2023-04-15T07:35:57.329059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Define the number of epochs, number of clients, and number of rounds\nnum_epochs = 10\nnum_clients = 5\nnum_rounds = 10\n\n# Generate some random data for training\ntrain_data = np.random.rand(100, 2)\ntrain_labels = np.random.randint(0, 2, size=100)\n\n# Initialize the model weights randomly\nweights = np.random.rand(2)\n\n# Define the learning rate and the fraction of clients to be selected per round\nlr = 0.1\nfrac_clients = 0.5\n\n# Define a function for training the model on a single client's data\ndef train_on_client(client_data, client_labels, weights):\n    for epoch in range(num_epochs):\n        for i in range(len(client_data)):\n            prediction = np.dot(client_data[i], weights)\n            error = client_labels[i] - prediction\n            weights += lr * error * client_data[i]\n    return weights\n\n# Define a function for selecting a fraction of clients randomly for each round\ndef select_clients(num_clients, frac_clients):\n    num_selected = int(num_clients * frac_clients)\n    selected_clients = np.random.choice(num_clients, size=num_selected, replace=False)\n    return selected_clients\n\n# Initialize the list to store the accuracies for each round\naccuracies = []\n\n# Run the Federated Learning process for the specified number of rounds\nfor round in range(num_rounds):\n    # Select a random fraction of clients for this round\n    selected_clients = select_clients(num_clients, frac_clients)\n    \n    # Train the model on the selected clients' data\n    for client in selected_clients:\n        weights = train_on_client(train_data[client::num_clients], train_labels[client::num_clients], weights)\n    \n    # Evaluate the model on the test data\n    test_data = np.random.rand(100, 2)\n    test_labels = np.random.randint(0, 2, size=100)\n    predictions = np.dot(test_data, weights)\n    predicted_labels = np.round(predictions)\n    accuracy = np.sum(predicted_labels == test_labels) / len(test_labels)\n    accuracies.append(accuracy)\n    \n# Plot the accuracies as a line graph\nplt.plot(range(num_rounds), accuracies)\nplt.xlabel('Round')\nplt.ylabel('Accuracy')\nplt.title('Federated Learning Accuracy')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:39:00.790723Z","iopub.execute_input":"2023-04-15T07:39:00.79106Z","iopub.status.idle":"2023-04-15T07:39:00.967026Z","shell.execute_reply.started":"2023-04-15T07:39:00.791007Z","shell.execute_reply":"2023-04-15T07:39:00.966256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Initialize the lists to store the accuracies for each model\nlr_accuracies = []\nfl_accuracies = []\n\n# Train the Linear Regression model on the training data and evaluate on the test data\nlr_weights = train_linear_regression(train_data, train_labels, lr_weights)\npredictions = np.dot(test_data, lr_weights)\npredicted_labels = np.round(predictions)\naccuracy = np.sum(predicted_labels == test_labels) / len(test_labels)\nlr_accuracies.append((0, accuracy))\n\n# Run the Federated Learning process for the specified number of rounds and evaluate on the test data\nfor round in range(num_rounds):\n    # Select a random fraction of vehicles for this round\n    selected_vehicles = select_vehicles(num_vehicles, frac_vehicles)\n    \n    # Train the model on the selected vehicles' data\n    for vehicle in selected_vehicles:\n        fl_weights = train_on_vehicle(train_data[vehicle::num_vehicles], train_labels[vehicle::num_vehicles], fl_weights)\n    \n    # Evaluate the model on the test data\n    predictions = np.dot(test_data, fl_weights)\n    predicted_labels = np.round(predictions)\n    accuracy = np.sum(predicted_labels == test_labels) / len(test_labels)\n    fl_accuracies.append((round + 1, accuracy))\n    \n# Plot the accuracies as separate graphs for comparison\n\nlr_x, lr_y = zip(*lr_accuracies)\nfl_x, fl_y = zip(*fl_accuracies)\n\n# Plot the accuracies for each model\nlr_x, lr_y = zip(*lr_accuracies)\nfl_x, fl_y = zip(*fl_accuracies)\n\nplt.plot([1, 4], [0, 2])\nplt.plot(lr_x, lr_y, 'ro-', label='')\nplt.legend(loc='lower right')\nplt.xlabel('Rounds')\nplt.ylabel('Accuracy')\nplt.title('Linear Regression Model Accuracy')\nplt.show()\n# Plot the accuracies for each model\nlr_x, lr_y = zip(*lr_accuracies)\nfl_x, fl_y = zip(*fl_accuracies)\n\nplt.plot([0, 8], [0, 4], 'b-', label='Federated Learning')\n\nplt.legend(loc='lower right')\nplt.xlabel('Rounds')\nplt.ylabel('Accuracy')\nplt.title('Federated Learning Model Accuracy')\nplt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T08:12:20.170931Z","iopub.execute_input":"2023-04-15T08:12:20.171227Z","iopub.status.idle":"2023-04-15T08:12:20.667366Z","shell.execute_reply.started":"2023-04-15T08:12:20.171173Z","shell.execute_reply":"2023-04-15T08:12:20.666665Z"},"trusted":true},"execution_count":null,"outputs":[]}]}