{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames: \n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-20T13:48:02.038254Z","iopub.execute_input":"2023-03-20T13:48:02.038603Z","iopub.status.idle":"2023-03-20T13:48:02.061340Z","shell.execute_reply.started":"2023-03-20T13:48:02.038572Z","shell.execute_reply":"2023-03-20T13:48:02.060441Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reducing the Size of Datatset\n- This notebook works on reducing the size of a face point dataset by selecting the most important points.\n- Reducing dataset size while maintaining accuracy is crucial in deep learning.\n- Using Euclidean distance as a measure to identify and extract the most important data points.\n\nIn short, creating a smaller, more efficient dataset without sacrificing accuracy.\n\n## Overview\n","metadata":{}},{"cell_type":"markdown","source":"## Load the Dataset","metadata":{}},{"cell_type":"code","source":"dfTrain = pd.read_csv('/kaggle/input/asl-signs/train.csv')\ndfTrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:02.064879Z","iopub.execute_input":"2023-03-20T13:48:02.065140Z","iopub.status.idle":"2023-03-20T13:48:02.274367Z","shell.execute_reply.started":"2023-03-20T13:48:02.065114Z","shell.execute_reply":"2023-03-20T13:48:02.273273Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:02.275713Z","iopub.execute_input":"2023-03-20T13:48:02.276544Z","iopub.status.idle":"2023-03-20T13:48:02.283923Z","shell.execute_reply.started":"2023-03-20T13:48:02.276505Z","shell.execute_reply":"2023-03-20T13:48:02.282629Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the data from Parquet Files","metadata":{}},{"cell_type":"code","source":"def video_to_dataframe(dfTrain, index):\n    df = dfTrain.iloc[index,:]\n    label = dfTrain.iloc[index,-1]\n    path = '/kaggle/input/asl-signs/' + df.path\n    videoData = pd.read_parquet(path)\n    \n    videoData = videoData[videoData.type == 'face']\n    frames = videoData.frame.unique()[0]\n    videoData = videoData[videoData.frame == frames]\n    return videoData, label\n\nblowdata, _ = video_to_dataframe(dfTrain, index = 0)\nwaitdata, _ = video_to_dataframe(dfTrain, index = 1)\nblowdata.tail()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:02.285482Z","iopub.execute_input":"2023-03-20T13:48:02.286123Z","iopub.status.idle":"2023-03-20T13:48:02.405663Z","shell.execute_reply.started":"2023-03-20T13:48:02.286063Z","shell.execute_reply":"2023-03-20T13:48:02.404320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plotFace(data, title):\n    plt.scatter(data.x,-data.y,c=data.z)\n    plt.title(title)\n    plt.show()\n\nfor i, index in enumerate(dfTrain.index):\n    if i > 5: break\n    data, label = video_to_dataframe(dfTrain, index = index)\n    plotFace(data, f'{label}')","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:02.409565Z","iopub.execute_input":"2023-03-20T13:48:02.409898Z","iopub.status.idle":"2023-03-20T13:48:04.397539Z","shell.execute_reply.started":"2023-03-20T13:48:02.409867Z","shell.execute_reply":"2023-03-20T13:48:04.396314Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating a Point Specific Dataset","metadata":{}},{"cell_type":"code","source":"'''\nfor i, index in tqdm(enumerate(dfTrain.index)):\n    if i < 12196: continue\n\n    data, label = video_to_dataframe(dfTrain, index = index)\n    data['label'] = label\n    data['points'] = data.x.apply(lambda x: f'({x},')\n    data.points = data.points + data.y.apply(lambda y:f'{y})')\n    \n    if pointWiseData is None:\n        pointWiseData = pd.DataFrame(data.points.copy()).T\n        pointWiseData.index = [i]\n        display(pointWiseData)\n    else:\n        tmpdf = pd.DataFrame(data.points).T\n        tmpdf.index = [i]\n        pointWiseData = pointWiseData.append(tmpdf,ignore_index=True)\n\n\npointWiseData.head()\n'''\n\n# This files has the first 10k records\npointWiseData = pd.read_csv('/kaggle/input/pointwisedata/pointwisedata.csv',index_col=[0])\npointWiseData.dropna(inplace=True)\npointWiseData.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:04.398974Z","iopub.execute_input":"2023-03-20T13:48:04.399573Z","iopub.status.idle":"2023-03-20T13:48:14.226663Z","shell.execute_reply.started":"2023-03-20T13:48:04.399531Z","shell.execute_reply":"2023-03-20T13:48:14.225527Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pointWiseData.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:14.228485Z","iopub.execute_input":"2023-03-20T13:48:14.228867Z","iopub.status.idle":"2023-03-20T13:48:14.235951Z","shell.execute_reply.started":"2023-03-20T13:48:14.228830Z","shell.execute_reply":"2023-03-20T13:48:14.234746Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing nan values\npointWiseData = pointWiseData[pointWiseData['0'] != '(nan,nan)']","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:14.237931Z","iopub.execute_input":"2023-03-20T13:48:14.238363Z","iopub.status.idle":"2023-03-20T13:48:14.344269Z","shell.execute_reply.started":"2023-03-20T13:48:14.238327Z","shell.execute_reply":"2023-03-20T13:48:14.343188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Understanding the Algorithm","metadata":{}},{"cell_type":"code","source":"# How the dataset looks\nprint(f\"\\033[1mOutlook:\")\ndf = pd.DataFrame(\n    [\n        ['(0.1,0.2)','(0.2,0.3)'],\n        ['(0.5,0.4)','(0.6,0.7)']\n    ]\n)\nprint(df)\n# '(x,y)'\n\n# Examples\n\n# Euclidian distance\na1 = np.array([0.39812421,0.79790165,0.94341754])\nb1 = np.array([0.54758138,0.93444764,0.26444269])\nc1 = np.array([[0.25163294,0.90127458,0.31619953],[0.44758138,0.03444764,0.86444269],[0.39812421,0.79790165,0.94341754]])\nprint()\nprint(\"\\033[1mEuclidian Distance:\")\n\nprint(f\"a1: {a1}\")\n\nprint(f\"b1: {b1}\")\n\nprint(f\"\\033[1mEuclidian Distance between a1 and b1 is {np.linalg.norm(a1 - b1)}\\n\")\n\ndf = np.random.rand(5, 3)\ndisplay(df)\nprint(df.shape)\nprint()\n\n# Max along axis\nprint(\"\\033[1mMaximum along the axis:\")\nprint()\nprint(f\"Maximum along the 0 axis is {np.max(df, axis = 0)}\\n\")\n\n# BroadCasting\nprint(\"\\033[1mBroadcasting:\")\n\nprint(f\"c1: \")\nprint(c1)\n\nprint(f\"b1: \")\nprint(b1)\n\nbroadcasted = np.broadcast_arrays(c1, b1)\nprint()\nprint(f\"Broadcast b1 wrt c1 is \\n {broadcasted[1]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:14.345995Z","iopub.execute_input":"2023-03-20T13:48:14.346382Z","iopub.status.idle":"2023-03-20T13:48:14.365714Z","shell.execute_reply.started":"2023-03-20T13:48:14.346345Z","shell.execute_reply":"2023-03-20T13:48:14.364687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Algorithm","metadata":{}},{"cell_type":"code","source":"import re\nimport tqdm\nimport time\nimport concurrent.futures\n\n# convert the object to float type\ndef intify(data):\n    result = data.apply(\n        lambda y: np.array(\n            [float(r) for r in re.findall(r'\\d+\\.\\d+', y)]\n        )\n    )\n    return pd.DataFrame(\n        np.array(\n            result.tolist()\n        ).reshape(-1,2)\n    )\n\n# Main algo - Euclidian dist\ndef euclidianDist(data, row):\n    return np.linalg.norm(data - row)\n\n# Get the max\ndef minmaxPoints(data):\n    return np.max(data, axis = 0), np.argmax(data, axis = 0)\n\n# iterate along rows\ndef excecute(col_data):\n    col_data = intify(col_data)\n\n    tmpdata = []\n    for row in col_data.index:\n        tmpdata.append(\n            euclidianDist(\n                col_data, \n                col_data.loc[row]\n        )\n    ) \n    return tmpdata\n    \ndef getResults(res):\n    maxInd = np.argpartition(res, -n_largest)[-n_largest:]\n    # minInd = np.argpartition(-res, -n_largest)[-n_largest:]\n    return maxInd\n    \n# Get the important points\ndef impPoints(data, n_largest):\n    iterCols = data.columns\n    # multiprocessing\n    with concurrent.futures.ProcessPoolExecutor() as exe:\n        result = list(tqdm.tqdm(exe.map(excecute, list(map(lambda ic: data[ic], iterCols))), total = len(iterCols)))  \n        \n    result = np.array(result).T\n    print(result.shape)\n        \n    with concurrent.futures.ProcessPoolExecutor() as exe2:\n        maxIndices = []\n        outputs = [exe2.submit(getResults, res).result() for res in result]\n        \n        for max_ in outputs:\n            maxIndices.append(max_)\n            \n    return np.array(maxIndices).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:14.367067Z","iopub.execute_input":"2023-03-20T13:48:14.367737Z","iopub.status.idle":"2023-03-20T13:48:14.382068Z","shell.execute_reply.started":"2023-03-20T13:48:14.367697Z","shell.execute_reply":"2023-03-20T13:48:14.380806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_largest = 10\nmaxResult = impPoints(pointWiseData.loc[:5000],n_largest)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:48:19.932552Z","iopub.execute_input":"2023-03-20T13:48:19.933016Z","iopub.status.idle":"2023-03-20T14:02:16.384163Z","shell.execute_reply.started":"2023-03-20T13:48:19.932964Z","shell.execute_reply":"2023-03-20T14:02:16.382855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\n\ndef mostReq(result):\n    vals = Counter(np.array(result).flatten())\n    points = vals.most_common(150)\n    points = [item[0] for item in points]\n    points.sort()\n    return points\n\npoints = mostReq(maxResult)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:02:16.386830Z","iopub.execute_input":"2023-03-20T14:02:16.387146Z","iopub.status.idle":"2023-03-20T14:02:16.412819Z","shell.execute_reply.started":"2023-03-20T14:02:16.387114Z","shell.execute_reply":"2023-03-20T14:02:16.411900Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Results\n- We have performed for 5000 records out of the 94000 records. The results are noted below.","metadata":{}},{"cell_type":"code","source":"np.array(points)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:02:16.414193Z","iopub.execute_input":"2023-03-20T14:02:16.414761Z","iopub.status.idle":"2023-03-20T14:02:16.422405Z","shell.execute_reply.started":"2023-03-20T14:02:16.414723Z","shell.execute_reply":"2023-03-20T14:02:16.421195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meshpoints = {\n    'silhouette': [10, 338, 297, 332, 284,251, 389, 356, 454, 323, 361,288,397,365,379,378,400,\n                   377,152,148,176,149,150,136,172,58,132,93,234,127,162,21,54,103,67,109],\n     'lipsUpperOuter': [61, 40, 39, 37, 0, 267, 269, 409],\n     'lipsLowerOuter': [91, 181, 84, 17, 314, 405, 375],\n     'lipsUpperInner': [78, 80, 82, 13, 311, 415],\n     'lipsLowerInner': [78, 88, 87, 14, 317, 318, 308],\n     'rightEyeUpper0': [161, 159, 157],\n     'rightEyeLower0': [7, 144, 153, 155],\n     'rightEyeUpper1': [30, 27, 56],\n     'rightEyeLower1': [25, 24, 22, 112],\n     'rightEyeUpper2': [225, 223, 221],\n     'rightEyeLower2': [31, 229, 231, 233],\n     'rightEyeLower3': [143, 117, 119, 121, 245],\n     'rightEyebrowUpper': [156, 70, 63, 105, 66, 107, 193],\n     'rightEyebrowLower': [124, 46, 53, 52, 65],\n     'rightEyeIris': [474, 476],\n     'leftEyeUpper0': [388, 386, 384],\n     'leftEyeLower0': [249, 373, 380, 382],\n     'leftEyeUpper1': [260, 257, 286],\n     'leftEyeLower1': [255, 254, 252, 341],\n     'leftEyeUpper2': [445, 444, 442, 413],\n     'leftEyeLower2': [261, 449, 451, 453],\n     'leftEyeLower3': [340, 347, 349, 357],\n     'leftEyebrowUpper': [383, 300, 293, 334, 296, 336, 417],\n     'leftEyebrowLower': [353, 276, 283, 282, 295],\n     'leftEyeIris': [469, 471],\n     'midwayBetweenEyes': [],\n     'noseTip': [1],\n     'noseBottom': [],\n     'noseRightCorner': [],\n     'noseLeftCorner': [],\n     'rightCheek': [],\n     'leftCheek': [],\n}","metadata":{"execution":{"iopub.status.busy":"2023-03-20T19:14:30.514537Z","iopub.execute_input":"2023-03-20T19:14:30.515124Z","iopub.status.idle":"2023-03-20T19:14:30.526085Z","shell.execute_reply.started":"2023-03-20T19:14:30.515087Z","shell.execute_reply":"2023-03-20T19:14:30.524771Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nData = pd.Series(data=[len(v) for v in meshpoints.values()],index=meshpoints.keys())\nnData.sort_values(inplace = True, ascending = False)\n\nnData = nData[nData.apply(lambda x : x > 0)]\nplt.figure(figsize=(10,5))\nplt.title('Number of points for each Category')\nplt.barh(nData.index, nData.values);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-20T19:33:31.255401Z","iopub.execute_input":"2023-03-20T19:33:31.255840Z","iopub.status.idle":"2023-03-20T19:33:31.630192Z","shell.execute_reply.started":"2023-03-20T19:33:31.255799Z","shell.execute_reply":"2023-03-20T19:33:31.629237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plotting the Results","metadata":{}},{"cell_type":"code","source":"for i, index in enumerate(dfTrain.index):\n    if i > 5: break\n    data, label = video_to_dataframe(dfTrain, index = index)\n    data = data.loc[points]\n    plotFace(data, f'{label}')","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:02:16.425232Z","iopub.execute_input":"2023-03-20T14:02:16.425534Z","iopub.status.idle":"2023-03-20T14:02:17.846113Z","shell.execute_reply.started":"2023-03-20T14:02:16.425507Z","shell.execute_reply":"2023-03-20T14:02:17.845173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\n- We are noting the points that showed the highest displacement among the initial 5000 records.\n\n- Although the primary facial characteristics were recorded, it is noteworthy that the eyes were not captured, implying that there was little to no eye movement.","metadata":{}},{"cell_type":"markdown","source":"## Next Steps\n\n- [] Apply a better algorithm.\n- [] Compare the model results after reducing the size of the face points.","metadata":{}}]}