{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nprint('Files succesfully input')","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:40.336757Z","iopub.execute_input":"2023-06-11T15:11:40.337147Z","iopub.status.idle":"2023-06-11T15:11:48.949354Z","shell.execute_reply.started":"2023-06-11T15:11:40.337117Z","shell.execute_reply":"2023-06-11T15:11:48.948341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_signs = pd.read_csv('/kaggle/input/asl-signs/train.csv')\nall_signs.head(), len(all_signs)","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:48.951122Z","iopub.execute_input":"2023-06-11T15:11:48.951881Z","iopub.status.idle":"2023-06-11T15:11:49.128148Z","shell.execute_reply.started":"2023-06-11T15:11:48.951849Z","shell.execute_reply":"2023-06-11T15:11:49.126819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_signs.shape, all_signs.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.129143Z","iopub.execute_input":"2023-06-11T15:11:49.129423Z","iopub.status.idle":"2023-06-11T15:11:49.184670Z","shell.execute_reply.started":"2023-06-11T15:11:49.129402Z","shell.execute_reply":"2023-06-11T15:11:49.183224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pq_data(path):\n    '''\n    Essentially a wrapper function that simplifies our workload\n    '''\n    return pd.read_parquet(\"/kaggle/input/asl-signs/\"+path,engine = 'auto')","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.186815Z","iopub.execute_input":"2023-06-11T15:11:49.187080Z","iopub.status.idle":"2023-06-11T15:11:49.192118Z","shell.execute_reply.started":"2023-06-11T15:11:49.187059Z","shell.execute_reply":"2023-06-11T15:11:49.191147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = all_signs.drop('sign', axis=1)\nY = all_signs['sign']","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.193819Z","iopub.execute_input":"2023-06-11T15:11:49.194144Z","iopub.status.idle":"2023-06-11T15:11:49.212065Z","shell.execute_reply.started":"2023-06-11T15:11:49.194119Z","shell.execute_reply":"2023-06-11T15:11:49.210485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head(), Y.head(), X['path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.213779Z","iopub.execute_input":"2023-06-11T15:11:49.214262Z","iopub.status.idle":"2023-06-11T15:11:49.230006Z","shell.execute_reply.started":"2023-06-11T15:11:49.214232Z","shell.execute_reply":"2023-06-11T15:11:49.228744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X), len(Y)","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:29:27.851975Z","iopub.execute_input":"2023-06-11T15:29:27.852352Z","iopub.status.idle":"2023-06-11T15:29:27.860327Z","shell.execute_reply.started":"2023-06-11T15:29:27.852326Z","shell.execute_reply":"2023-06-11T15:29:27.859205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\n\ndef load_relevant_data_subset(pq_path):\n    pq_path = f'/kaggle/input/asl-signs/{pq_path}'\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.231840Z","iopub.execute_input":"2023-06-11T15:11:49.232245Z","iopub.status.idle":"2023-06-11T15:11:49.243904Z","shell.execute_reply.started":"2023-06-11T15:11:49.232211Z","shell.execute_reply":"2023-06-11T15:11:49.243027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a function to load the data and deal with the nan data altogether.\ndef load_without_nan(pq_path):\n    frame = load_relevant_data_subset(pq_path)\n    array = frame.copy()\n    nan_mask = np.isnan(frame)\n    array[nan_mask] = np.nanmean(array)\n    return tf.convert_to_tensor(array)","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.245272Z","iopub.execute_input":"2023-06-11T15:11:49.245602Z","iopub.status.idle":"2023-06-11T15:11:49.257505Z","shell.execute_reply.started":"2023-06-11T15:11:49.245570Z","shell.execute_reply":"2023-06-11T15:11:49.256211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_everything_2(dataset, lower_limit=0, upper_limit=94477):\n    total_data = []\n    for i in range(lower_limit, upper_limit):\n        total_data.append(load_without_nan(dataset.path[i]))\n    return total_data","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.259019Z","iopub.execute_input":"2023-06-11T15:11:49.259601Z","iopub.status.idle":"2023-06-11T15:11:49.272088Z","shell.execute_reply.started":"2023-06-11T15:11:49.259572Z","shell.execute_reply":"2023-06-11T15:11:49.270539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = load_without_nan(all_signs.path[0])\nexample","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.274876Z","iopub.execute_input":"2023-06-11T15:11:49.275147Z","iopub.status.idle":"2023-06-11T15:11:49.644604Z","shell.execute_reply.started":"2023-06-11T15:11:49.275126Z","shell.execute_reply":"2023-06-11T15:11:49.643580Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame()\ndata","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.645748Z","iopub.execute_input":"2023-06-11T15:11:49.647329Z","iopub.status.idle":"2023-06-11T15:11:49.660777Z","shell.execute_reply.started":"2023-06-11T15:11:49.647282Z","shell.execute_reply":"2023-06-11T15:11:49.659921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_into_dataframe(pq_path):\n    temp_data = pd.DataFrame()\n    temp = load_without_nan(pq_path)\n    for i in temp:\n        temp_data = pd.concat(temp_data, i)\n    return temp_data","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.661939Z","iopub.execute_input":"2023-06-11T15:11:49.662301Z","iopub.status.idle":"2023-06-11T15:11:49.673113Z","shell.execute_reply.started":"2023-06-11T15:11:49.662278Z","shell.execute_reply":"2023-06-11T15:11:49.671602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_data = load_into_dataframe(all_signs.path[0])\n# new_data","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:11:49.674579Z","iopub.execute_input":"2023-06-11T15:11:49.675814Z","iopub.status.idle":"2023-06-11T15:11:50.247591Z","shell.execute_reply.started":"2023-06-11T15:11:49.675781Z","shell.execute_reply":"2023-06-11T15:11:50.245925Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only to be considered if the tensor is not formed.","metadata":{}},{"cell_type":"code","source":"example_2 = load_without_nan(all_signs.path[0])\ntype(example_2), example_2.shape, example_2","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:13:19.626968Z","iopub.execute_input":"2023-06-11T15:13:19.627319Z","iopub.status.idle":"2023-06-11T15:13:19.644895Z","shell.execute_reply.started":"2023-06-11T15:13:19.627293Z","shell.execute_reply":"2023-06-11T15:13:19.643572Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_3 = load_without_nan(all_signs.path[1])\nprint(example_3.shape, type(example_3))\nexample_3 = tf.concat([example_2, example_3], axis=0)\nprint(example_3.shape, type(example_3))\nexample_4 = load_without_nan(all_signs.path[2])\nprint(example_4.shape, type(example_4))\nexample_5 = load_without_nan(all_signs.path[3])\nprint(example_5.shape, type(example_5))\nexample_5 = tf.concat([example_3, example_4, example_5], axis=0)\nexample_5.shape, type(example_5), example_5","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:13:37.582462Z","iopub.execute_input":"2023-06-11T15:13:37.582825Z","iopub.status.idle":"2023-06-11T15:13:37.620026Z","shell.execute_reply.started":"2023-06-11T15:13:37.582798Z","shell.execute_reply":"2023-06-11T15:13:37.618569Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_into_tensor(dataset, lower_limit=0, upper_limit=94477):\n    tensor = load_without_nan(dataset.path[lower_limit])\n    for iter in range(lower_limit+1, upper_limit):\n        temp = load_without_nan(dataset.path[iter])\n        tensor = tf.concat([tensor, temp],axis=0)\n    return tensor","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:26:25.128937Z","iopub.execute_input":"2023-06-11T15:26:25.129316Z","iopub.status.idle":"2023-06-11T15:26:25.136104Z","shell.execute_reply.started":"2023-06-11T15:26:25.129289Z","shell.execute_reply":"2023-06-11T15:26:25.135011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_data_train = load_into_tensor(X, upper_limit=50)\nX_data_train.shape, type(X_data_train), X_data_train","metadata":{"execution":{"iopub.status.busy":"2023-06-11T15:30:21.636027Z","iopub.execute_input":"2023-06-11T15:30:21.636424Z","iopub.status.idle":"2023-06-11T15:30:22.714748Z","shell.execute_reply.started":"2023-06-11T15:30:21.636397Z","shell.execute_reply":"2023-06-11T15:30:22.713684Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}