{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nLABEL_FILE='/kaggle/input/asl-signs/train.csv'\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-08T04:02:11.150161Z","iopub.execute_input":"2023-03-08T04:02:11.151146Z","iopub.status.idle":"2023-03-08T04:02:11.190048Z","shell.execute_reply.started":"2023-03-08T04:02:11.151080Z","shell.execute_reply":"2023-03-08T04:02:11.188804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Objective\n**Purpose of this notebook to analyze distributin of label values for asl-signs competition.**","metadata":{}},{"cell_type":"code","source":"label_df = pd.read_csv(LABEL_FILE)\nlabel_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:02:20.398046Z","iopub.execute_input":"2023-03-08T04:02:20.398879Z","iopub.status.idle":"2023-03-08T04:02:20.662152Z","shell.execute_reply.started":"2023-03-08T04:02:20.398838Z","shell.execute_reply":"2023-03-08T04:02:20.660557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.array(label_df['sign'])","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:03:20.469791Z","iopub.execute_input":"2023-03-08T04:03:20.470194Z","iopub.status.idle":"2023-03-08T04:03:20.477591Z","shell.execute_reply.started":"2023-03-08T04:03:20.470159Z","shell.execute_reply":"2023-03-08T04:03:20.476180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels, counts = np.unique(labels, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:05:55.668266Z","iopub.execute_input":"2023-03-08T04:05:55.668701Z","iopub.status.idle":"2023-03-08T04:05:55.735102Z","shell.execute_reply.started":"2023-03-08T04:05:55.668653Z","shell.execute_reply":"2023-03-08T04:05:55.733000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_counts = np.array(counts)\nprint(f'Sample size for {len(label_counts)} labels range between {np.min(label_counts)} and {np.max(label_counts)}')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:11:08.367619Z","iopub.execute_input":"2023-03-08T04:11:08.368074Z","iopub.status.idle":"2023-03-08T04:11:08.374579Z","shell.execute_reply.started":"2023-03-08T04:11:08.368026Z","shell.execute_reply":"2023-03-08T04:11:08.373381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:06:24.693215Z","iopub.execute_input":"2023-03-08T04:06:24.693828Z","iopub.status.idle":"2023-03-08T04:06:24.699751Z","shell.execute_reply.started":"2023-03-08T04:06:24.693777Z","shell.execute_reply":"2023-03-08T04:06:24.698190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(unique_labels, counts)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:07:53.963580Z","iopub.execute_input":"2023-03-08T04:07:53.964758Z","iopub.status.idle":"2023-03-08T04:07:56.578012Z","shell.execute_reply.started":"2023-03-08T04:07:53.964677Z","shell.execute_reply":"2023-03-08T04:07:56.576732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion\n**Total sample size is 94477**\n\n**There are 250 labels and sample size for each label varies between 299 and 415. Sample size is relatively small to train a classification model effectively**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = np.random.rand(94477,2)\nX_train, X_test, y_train, y_test = train_test_split(X, labels, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:39:21.759176Z","iopub.execute_input":"2023-03-08T04:39:21.759591Z","iopub.status.idle":"2023-03-08T04:39:21.780421Z","shell.execute_reply.started":"2023-03-08T04:39:21.759553Z","shell.execute_reply":"2023-03-08T04:39:21.779207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train labels \nunique_train_labels, train_label_counts = np.unique(y_train, return_counts=True)\nlabel_counts = np.array(train_label_counts)\nprint(f'Sample size for {len(label_counts)} labels range between {np.min(label_counts)} and {np.max(label_counts)}')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:39:23.611378Z","iopub.execute_input":"2023-03-08T04:39:23.611820Z","iopub.status.idle":"2023-03-08T04:39:23.662356Z","shell.execute_reply.started":"2023-03-08T04:39:23.611781Z","shell.execute_reply":"2023-03-08T04:39:23.661120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train labels \nunique_test_labels, test_label_counts = np.unique(y_test, return_counts=True)\nlabel_counts = np.array(test_label_counts)\nprint(f'Sample size for {len(label_counts)} labels range between {np.min(label_counts)} and {np.max(label_counts)}')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:39:25.653180Z","iopub.execute_input":"2023-03-08T04:39:25.653622Z","iopub.status.idle":"2023-03-08T04:39:25.675108Z","shell.execute_reply.started":"2023-03-08T04:39:25.653583Z","shell.execute_reply":"2023-03-08T04:39:25.673738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"training size :\", len(y_train))\nprint(\"test size :\", len(y_test))","metadata":{"execution":{"iopub.status.busy":"2023-03-08T04:39:41.536301Z","iopub.execute_input":"2023-03-08T04:39:41.536755Z","iopub.status.idle":"2023-03-08T04:39:41.543523Z","shell.execute_reply.started":"2023-03-08T04:39:41.536713Z","shell.execute_reply":"2023-03-08T04:39:41.541916Z"},"trusted":true},"execution_count":null,"outputs":[]}]}