{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:19.851191Z","iopub.execute_input":"2023-06-09T19:47:19.852335Z","iopub.status.idle":"2023-06-09T19:47:19.862233Z","shell.execute_reply.started":"2023-06-09T19:47:19.852261Z","shell.execute_reply":"2023-06-09T19:47:19.859298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy import signal \nimport os,gc\nimport pyarrow.parquet as pq\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:19.866600Z","iopub.execute_input":"2023-06-09T19:47:19.867274Z","iopub.status.idle":"2023-06-09T19:47:19.892286Z","shell.execute_reply.started":"2023-06-09T19:47:19.867222Z","shell.execute_reply":"2023-06-09T19:47:19.890305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-09T19:47:19.895330Z","iopub.execute_input":"2023-06-09T19:47:19.896342Z","iopub.status.idle":"2023-06-09T19:47:19.910623Z","shell.execute_reply.started":"2023-06-09T19:47:19.896288Z","shell.execute_reply":"2023-06-09T19:47:19.908572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'\ntrain_path = 'train'\ntest_path = 'test'\ndefog_path = 'defog'\ntdcsfog_path = 'tdcsfog'\n# sample = pd.read_csv(os.path.join(path,'sample_submission.csv'))\n# sample.head()\n# sample.info()\nunlabeled_path = 'unlabeled'\n\ndefog_m = pd.read_csv(path+'/defog_metadata.csv')\n#defog_m.head()\ndaily_m = pd.read_csv(path+'/daily_metadata.csv')\n#daily_m.head()\n\ngait_daily_subj = set(defog_m['Subject']).intersection(daily_m['Subject'])\nnormal_daily_subj = set(daily_m['Subject']) - set(defog_m['Subject'])\nnormal_daily_ids = daily_m.query('Subject.isin(@normal_daily_subj) ')['Id']\nnormal_daily_files = [f+'.parquet' for f in normal_daily_ids]\ndefog_m = None\ndaily_m = None\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:19.915019Z","iopub.execute_input":"2023-06-09T19:47:19.915810Z","iopub.status.idle":"2023-06-09T19:47:21.104604Z","shell.execute_reply.started":"2023-06-09T19:47:19.915752Z","shell.execute_reply":"2023-06-09T19:47:21.102281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Normal people Daily Data","metadata":{}},{"cell_type":"code","source":"daily_path = os.path.join(path,unlabeled_path)\ndaily_normal_df = None\n\n\nfor dirname, _, filenames in os.walk(daily_path):\n    for i,filename in enumerate(filenames):\n        if filename in normal_daily_files:\n            df = pq.read_table(os.path.join(dirname, filename)).to_pandas()\n            df.dropna(inplace=True)\n            df['Id']=filename[0:filename.index('.')]\n            df = df.astype({'AccV':'float32','AccML':'float32',\n                            'AccAP':'float32','Time':'int32'})\n            if daily_normal_df is None:\n                daily_normal_df = df.copy()\n            else:\n                daily_normal_df=pd.concat([daily_normal_df,df],ignore_index=True)\n        if i==10: break\n\ndaily_normal_df.head()         \n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:21.107115Z","iopub.execute_input":"2023-06-09T19:47:21.107682Z","iopub.status.idle":"2023-06-09T19:47:36.069725Z","shell.execute_reply.started":"2023-06-09T19:47:21.107647Z","shell.execute_reply":"2023-06-09T19:47:36.068703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects = pd.read_csv(os.path.join(path,'subjects.csv'))\n# subjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:36.071016Z","iopub.execute_input":"2023-06-09T19:47:36.071408Z","iopub.status.idle":"2023-06-09T19:47:36.075947Z","shell.execute_reply.started":"2023-06-09T19:47:36.071382Z","shell.execute_reply":"2023-06-09T19:47:36.074947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects.Visit.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:36.077487Z","iopub.execute_input":"2023-06-09T19:47:36.078086Z","iopub.status.idle":"2023-06-09T19:47:36.090711Z","shell.execute_reply.started":"2023-06-09T19:47:36.078057Z","shell.execute_reply":"2023-06-09T19:47:36.089516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks = pd.read_csv(os.path.join(path,'tasks.csv'))\n# tasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:36.092098Z","iopub.execute_input":"2023-06-09T19:47:36.092407Z","iopub.status.idle":"2023-06-09T19:47:36.104241Z","shell.execute_reply.started":"2023-06-09T19:47:36.092381Z","shell.execute_reply":"2023-06-09T19:47:36.102841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks = pd.read_csv(os.path.join(path,'events.csv'))\n# tasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:36.108573Z","iopub.execute_input":"2023-06-09T19:47:36.108952Z","iopub.status.idle":"2023-06-09T19:47:36.117687Z","shell.execute_reply.started":"2023-06-09T19:47:36.108922Z","shell.execute_reply":"2023-06-09T19:47:36.116732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_cs_path = os.path.join(path,train_path,tdcsfog_path)\ntdcsfog_df = [] \n\nfor dirname, _, filenames in os.walk(sample_cs_path):\n    for filename in filenames:\n        df = pd.read_csv(os.path.join(dirname, filename))\n        df['Id']=filename[0:filename.index('.')]\n        tdcsfog_df.append(df)\n\ntdcsfog_df = pd.concat(tdcsfog_df,ignore_index=True)\ntdcsfog_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:36.119720Z","iopub.execute_input":"2023-06-09T19:47:36.120054Z","iopub.status.idle":"2023-06-09T19:47:48.385453Z","shell.execute_reply.started":"2023-06-09T19:47:36.120025Z","shell.execute_reply":"2023-06-09T19:47:48.384319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df.StartHesitation.unique()\ntdcsfog_df.Turn.unique()\ntdcsfog_df.Walking.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:48.386949Z","iopub.execute_input":"2023-06-09T19:47:48.387383Z","iopub.status.idle":"2023-06-09T19:47:48.518448Z","shell.execute_reply.started":"2023-06-09T19:47:48.387342Z","shell.execute_reply":"2023-06-09T19:47:48.516876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_cs_path = os.path.join(path,train_path,defog_path)\ndefog_df = [] \n\nfor dirname, _, filenames in os.walk(sample_cs_path):\n    for filename in filenames:\n        df = pd.read_csv(os.path.join(dirname, filename))\n        df['Id']=filename[0:filename.index('.')]\n        defog_df.append(df)\n\ndefog_df = pd.concat(defog_df,ignore_index=True)\ndefog_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:47:48.520175Z","iopub.execute_input":"2023-06-09T19:47:48.521022Z","iopub.status.idle":"2023-06-09T19:48:08.002874Z","shell.execute_reply.started":"2023-06-09T19:47:48.520985Z","shell.execute_reply":"2023-06-09T19:48:08.001568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['Time','AccV','AccML','AccAP']\ntargets = ['StartHesitation','Turn','Walking']\naccmeasurs = ['AccV','AccML','AccAP']\ndefog_df.Valid.unique()\ndefog_df.Task.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:08.004931Z","iopub.execute_input":"2023-06-09T19:48:08.005387Z","iopub.status.idle":"2023-06-09T19:48:08.167249Z","shell.execute_reply.started":"2023-06-09T19:48:08.005346Z","shell.execute_reply":"2023-06-09T19:48:08.165565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df = defog_df.query('Valid==True and Task==True')\ndefog_df = defog_df.drop(['Valid','Task'],axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:08.168551Z","iopub.execute_input":"2023-06-09T19:48:08.168882Z","iopub.status.idle":"2023-06-09T19:48:09.045895Z","shell.execute_reply.started":"2023-06-09T19:48:08.168852Z","shell.execute_reply":"2023-06-09T19:48:09.044635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = pd.concat([tdcsfog_df,defog_df])\nall_train_data = all_train_data.astype({'Time':'int32','Turn':'int8','Walking':'int8',\\\n                                        'StartHesitation':'int8','AccV':'float32',\\\n                                        'AccML':'float32','AccAP':'float32'})\ntdcsfog_df = None\ndefog_df = None\nall_train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:09.047398Z","iopub.execute_input":"2023-06-09T19:48:09.047751Z","iopub.status.idle":"2023-06-09T19:48:09.887931Z","shell.execute_reply.started":"2023-06-09T19:48:09.047722Z","shell.execute_reply":"2023-06-09T19:48:09.886974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nscaler = MinMaxScaler(feature_range=(-1,1))\nscaler.fit(all_train_data[accmeasurs])\nall_train_data[accmeasurs] = scaler.transform(all_train_data[accmeasurs]).astype('float32')","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:09.889621Z","iopub.execute_input":"2023-06-09T19:48:09.889984Z","iopub.status.idle":"2023-06-09T19:48:10.434486Z","shell.execute_reply.started":"2023-06-09T19:48:09.889954Z","shell.execute_reply":"2023-06-09T19:48:10.433383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.describe()\n\n# standardize the data\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:10.435704Z","iopub.execute_input":"2023-06-09T19:48:10.436014Z","iopub.status.idle":"2023-06-09T19:48:12.328288Z","shell.execute_reply.started":"2023-06-09T19:48:10.435987Z","shell.execute_reply":"2023-06-09T19:48:12.327076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for id,group in all_train_data.groupby('Id')[features+targets]:\n#     ax1 = sns.lineplot(x= 'Time',y='AccV',hue='StartHesitation',data=group.iloc[0:4000])\n#     peaks,_ = signal.find_peaks(group['AccV'],prominence=(None,None))\n#     sns.scatterplot(x='Time',y='AccV',data=group.iloc[peaks],ax=ax1)\n#    plt.show()\n    ax2= sns.lineplot(x= 'Time',y='AccV',hue='Turn',data=group.iloc[0:4000])\n    peaks,_ = signal.find_peaks(group.iloc[0:4000,1],distance=30)\n    sns.scatterplot(x='Time',y='AccV',data=group.iloc[0:4000].iloc[peaks],ax=ax2)\n    plt.show()\n#     sns.lineplot(x= 'Time',y='AccV',hue='Walking',data=group.iloc[0:4000])\n#     plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:12.329897Z","iopub.execute_input":"2023-06-09T19:48:12.330334Z","iopub.status.idle":"2023-06-09T19:48:14.759360Z","shell.execute_reply.started":"2023-06-09T19:48:12.330290Z","shell.execute_reply":"2023-06-09T19:48:14.757960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"from pandas.api.indexers import BaseIndexer\nclass PeakIndexer(BaseIndexer):\n    def __init__(self,peaks,index):\n        super().__init__(index_array=index,step=1)\n        self.peaks = peaks\n        self.current = 0\n    def get_window_bounds(self,\\\n    num_values: int = 0,\\\n    min_periods: int = None,\\\n    center: bool = None,\\\n    closed: str = None,\\\n    step: int = 1\n) -> tuple[np.ndarray, np.ndarray]:\n        if len(self.peaks) == 0:\n            return np.copy(self.index_array),np.copy(self.index_array)\n        start = np.zeros_like(self.index_array//self.step)\n        end = np.zeros_like(self.index_array//self.step)\n        for i in range(len(start)):\n            if i <= self.peaks[-1]:\n                index = np.where(self.peaks>=i)[0][0]\n                if index == 0:\n                    start[i]= 0 \n                    end[i]=self.peaks[index]\n                elif index == len(self.peaks)-1:\n                    start[i] = self.peaks[-2]\n                    end[i] = self.peaks[-1]\n                else:\n                    start[i]= self.peaks[index-1]\n                    end[i] = self.peaks[index]\n            else:\n                start[i] = self.peaks[-1]\n                end[i] = self.index_array[-1]\n        return start,end\n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:14.760923Z","iopub.execute_input":"2023-06-09T19:48:14.761366Z","iopub.status.idle":"2023-06-09T19:48:14.774529Z","shell.execute_reply.started":"2023-06-09T19:48:14.761333Z","shell.execute_reply":"2023-06-09T19:48:14.773548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef findSlice(df,dfn,thresh):\n    size = len(df)\n    sizen = len(dfn)\n    x = np.arange(0,sizen,size,dtype=np.int32)\n    blocks = [(x[i],x[i+1]) for i in range(len(x)-1)]\n    optimum_block=blocks[0]\n    start,end = blocks[0]\n    min_cross = abs(np.cross(df[accmeasurs].values, dfn[accmeasurs].iloc[start:end].values).sum())\n    for start,end in blocks[1:]:\n        dot_product = np.cross(df[accmeasurs].values, dfn[accmeasurs].iloc[start:end].values).sum()\n        if abs(dot_product)<min_cross:\n            min_cross = abs(dot_product)\n            optimum_block = start,end\n#            print(f'resetting current min {min_cross},calculated cross{dot_product}')        \n        if abs(min_cross) < thresh: \n            print(f'found block at {start},{end}')\n            return start,end\n    print(f'scanned all data and returning closest match {start},{end}')\n    return optimum_block","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:14.775832Z","iopub.execute_input":"2023-06-09T19:48:14.776857Z","iopub.status.idle":"2023-06-09T19:48:14.794893Z","shell.execute_reply.started":"2023-06-09T19:48:14.776814Z","shell.execute_reply":"2023-06-09T19:48:14.793805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_normal_df.set_index('Time',inplace=True)\ndef crossCorrelateWithNormal(df,thresh=None):\n    size = len(df)\n    if thresh is not None:\n        start,end = findSlice(df,daily_normal_df,thresh)\n    else:\n        start = 1264140  \n        end = start + size\n    ndf = daily_normal_df.iloc[start:end]\n    c = signal.correlate(df[accmeasurs],ndf[accmeasurs],method='fft',mode='same')\n    df[['ccorr_Accv','ccorr_AccML','ccorr_AccAP']]=c\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:14.796236Z","iopub.execute_input":"2023-06-09T19:48:14.796717Z","iopub.status.idle":"2023-06-09T19:48:15.181862Z","shell.execute_reply.started":"2023-06-09T19:48:14.796665Z","shell.execute_reply":"2023-06-09T19:48:15.179783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef addRollingColumns(df,name):\n    name_avgH = name + '_windowAvg'\n    name_minH = name + '_windowMin'\n    name_maxH = name + '_windowMax'\n    name_stdH = name + '_windowStd'\n    peaks,_ = signal.find_peaks(df[name],distance=30)\n    indexer = PeakIndexer(peaks,df.index)\n    #window = round(np.diff(peaks).max())\n\n    df[name_avgH] = df[name].rolling(indexer,step=1).mean()\n    #df[name_avgH] = df[name_avgH].fillna(df.iloc[peaks[0],list(df.columns).index(name_avgH) ])\n    df[name_avgH] = df[name_avgH].astype('float32')\n    df[name_maxH] = df[name].rolling(indexer,step=1).max()\n    #df[name_maxH] = df[name_maxH].fillna(df.iloc[peaks[0],list(df.columns).index(name_maxH) ])\n    df[name_maxH] = df[name_maxH].astype('float32')\n    df[name_minH] = df[name].rolling(indexer).min()\n    #df[name_minH] = df[name_minH].fillna(df.iloc[peaks[0],list(df.columns).index(name_minH) ])\n    df[name_minH] = df[name_minH].astype('float32')\n    df[name_stdH] = df[name].rolling(indexer).std()\n    #df[name_stdH] = df[name_stdH].fillna(df.iloc[peaks[0],list(df.columns).index(name_stdH) ])\n    df[name_stdH] = df[name_stdH].astype('float32')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:15.183877Z","iopub.execute_input":"2023-06-09T19:48:15.184292Z","iopub.status.idle":"2023-06-09T19:48:15.198015Z","shell.execute_reply.started":"2023-06-09T19:48:15.184254Z","shell.execute_reply":"2023-06-09T19:48:15.196687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numpy.linalg import norm\ndef addSimilarities(df,name1,name2):\n    col = name1+\"_sim_\"+name2\n    df.iloc[df[name1]==0,df.columns.get_loc(name1)]=.000001\n    df.iloc[df[name2]==0,df.columns.get_loc(name2)]=.000001\n    value = df[name1]*df[name2]/(abs(df[name1])*abs(df[name2]))\n    df[col] = value\n    df[col] = df[col].astype('float32')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:15.199541Z","iopub.execute_input":"2023-06-09T19:48:15.199954Z","iopub.status.idle":"2023-06-09T19:48:15.215663Z","shell.execute_reply.started":"2023-06-09T19:48:15.199926Z","shell.execute_reply":"2023-06-09T19:48:15.214456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def addNorm(df,n1,n2,n3):\n    col = 'norm_'+n1+'_'+n2+'_'+n3\n    val = df[n1]**2 + df[n2]**2 + df[n3]**2\n    df[col] = val**0.5\n    df[col] = df[col].astype('float32')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:15.224743Z","iopub.execute_input":"2023-06-09T19:48:15.225131Z","iopub.status.idle":"2023-06-09T19:48:15.233224Z","shell.execute_reply.started":"2023-06-09T19:48:15.225101Z","shell.execute_reply":"2023-06-09T19:48:15.231950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def featureEngineer(data,scanMatchesThresh = None):\n    \n    first = None\n    engineered_dt = []\n    for cid,group in data:\n        df = group.set_index('Time')\n        df = crossCorrelateWithNormal(df,scanMatchesThresh)\n        df = addNorm(df,'AccV','AccML','AccAP')\n        df = addRollingColumns(df,'AccV')\n        df = addRollingColumns(df,'AccML')\n        df = addRollingColumns(df,'AccAP')\n        df = addSimilarities(df,'AccV','AccML')\n        df = addSimilarities(df,'AccV','AccAP')\n        df = addSimilarities(df,'AccAP','AccML')\n        df = addSimilarities(df,'AccV_sim_AccML','AccAP_sim_AccML')\n        df = addSimilarities(df,'AccV_sim_AccAP','AccAP_sim_AccML')\n        \n        df['Id']=cid\n        df = df.reset_index(names='Time')\n        engineered_dt.append(df.copy())\n    engineered_dt = pd.concat(engineered_dt,ignore_index=True)\n    \n    return engineered_dt\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:15.234663Z","iopub.execute_input":"2023-06-09T19:48:15.235135Z","iopub.status.idle":"2023-06-09T19:48:15.249635Z","shell.execute_reply.started":"2023-06-09T19:48:15.235083Z","shell.execute_reply":"2023-06-09T19:48:15.248727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = featureEngineer(all_train_data.groupby('Id'))\nall_features = [feature for feature in all_train_data.columns if feature !='Id' and feature not in targets and feature !='Time']\nall_features","metadata":{"execution":{"iopub.status.busy":"2023-06-09T19:48:15.250671Z","iopub.execute_input":"2023-06-09T19:48:15.250987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sh = all_train_data.corr(numeric_only=True)[['StartHesitation']].apply(abs).sort_values(by='StartHesitation')\ntu = all_train_data.corr(numeric_only=True)[['Turn']].apply(abs).sort_values(by='Turn')\nwk = all_train_data.corr(numeric_only=True)[['Walking']].apply(abs).sort_values(by='Walking')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sh\ntu\nwk","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_thresh = 0.1\nselected_features = [f[1].name for f in sh.iterrows() if f[1].iloc[0]>=corr_thresh]+\\\n                    [f[1].name for f in tu.iterrows() if f[1].iloc[0]>=corr_thresh]+\\\n                    [f[1].name for f in wk.iterrows() if f[1].iloc[0]>=corr_thresh]\nselected_features = list(set(selected_features)-set(targets))\nselected_features.remove('Time')\nselected_features","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.isinf(all_train_data[selected_features]).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_metadata = pd.read_csv(os.path.join(path,'tdcsfog_metadata.csv'))\n# tdcsfog_metadata.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas.api.types import is_numeric_dtype\ndef imputeInf(data):\n    for col in data.columns:\n        if is_numeric_dtype(data[col]):\n            idx_inf = np.isinf(data[col])\n            idx_neginf = np.isneginf(data[col])\n            if idx_inf.sum() >0 :\n                maximum = data.loc[data[col]!=np.inf,col].max()\n                data.loc[idx_inf,col] = maximum\n            elif idx_neginf.sum() >0:\n                minimum = data.loc[data[col]!=-np.inf,col].min()\n                data.loc[idx_neginf,col] = minimum\n            \n    return data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = imputeInf(all_train_data)\nall_train_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_Id_Time_cols(data):\n    data['Id'] = data['Id']+'_'+data['Time'].apply(str)\n    data.drop('Time',inplace=True,axis=1)\n    for col in data.columns:\n        if is_numeric_dtype(data[col]):\n            data[col]= data[col].astype('float32')\n    return data\n\nall_train_data = merge_Id_Time_cols(all_train_data)\ngc.collect()\nall_train_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.fillna(0,inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\nnum_clust = [1,2,3,4,5,6,7]\nmodels = []\ninertias = []\nlabels = []\nfor num in num_clust:\n    model = KMeans(n_clusters=num,random_state=0,n_init=\"auto\").fit(all_train_data[selected_features])\n    inertias.append(model.inertia_)\n    labels.append(model.labels_)\n    models.append(model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alpha = 0\ny = [inertias[i]/inertias[0] + alpha*num_clust[i] for i in range(len(num_clust))]\nax= sns.lineplot(x=num_clust,y=y,label='scaled inertia')\nsns.lineplot(x=num_clust,y=inertias,ax=ax,label='inertia')\nplt.yscale('log')\nplt.xlabel('number of clusters')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m_idx = 3\nkmean_model = models[m_idx]\nall_train_data['cluster'] = labels[m_idx].copy()\nmodels=None\nlabels = None\nall_train_data['cluster'] = all_train_data['cluster'].astype('int8')\nall_features.append('cluster')\nselected_features.append('cluster')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = all_train_data.groupby(['StartHesitation','Turn','Walking'])['Id'].agg('count')\nsampleCount = count[[(1,0,0),(0,1,0),(0,0,1)]].min()\n# nameRef = {(1,0,0):'StartHesitation',(0,1,0):'Turn',(0,0,1):'Walking'}\n# nameOfmax = nameRef[count[count==sampleCount].index[0]]\n# nameOfmax","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"noGait = all_train_data.query('StartHesitation==0 and Walking==0 and Turn==0').sample(sampleCount)\nhesit = all_train_data.query('StartHesitation == 1').sample(sampleCount)\nturn = all_train_data.query('Turn == 1').sample(sampleCount)\nwalk = all_train_data.query('Walking == 1').sample(sampleCount)\n\n# for name in nameRef.values():\n#     if name == nameOfmax:\n#         patch = all_train_data.query(f'{name}==1')\n#     else:\n#         patch = all_train_data.query(f'{name}==1').sample()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = pd.concat([noGait,hesit,turn,walk]).reset_index(drop=True)\nnoGait = None\nhesit = None\nturn = None\nwalk = None\n\n# all_train_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## create multi class column","metadata":{}},{"cell_type":"code","source":"model_classes = 'classes'\nclass_toindex = {1:0,2:1,3:2}\n\nall_train_data[model_classes]=0\nall_train_data.loc[all_train_data.StartHesitation==1,model_classes]=1\nall_train_data.loc[all_train_data.Turn==1,model_classes]=2\nall_train_data.loc[all_train_data.Walking==1,model_classes]=3\nall_train_data[model_classes]= all_train_data[model_classes].astype('int8')\nall_train_data.drop(columns=targets,index=1,inplace=True)\nall_train_data.head()\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import psutil\n# import os\n\n# my_pid = os.getpid()\n\n# for proc in psutil.process_iter():\n#     try:\n#         # Get process name & pid from process object.\n#         processName = proc.name()\n#         processID = proc.pid\n\n#         if proc.pid == my_pid:\n#             print(\"I am not suicidal\")\n#             continue\n\n#         if processName.startswith(\"python3.10[175]\"): # adapt this line to your needs\n#             print(f\"I will kill {processName}[{processID}] : {''.join(proc.cmdline())})\")\n#             #proc.kill()\n#     except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess) as e:\n#         print(e)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import backend as K\ndef precision_xgb(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision_keras = true_positives / (predicted_positives + K.epsilon())\n    return precision_keras","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_io\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.model_selection import train_test_split\n#from sklearn.model_selection import GridSearchCV\nfrom sklearn.experimental import enable_halving_search_cv  # noqa\nfrom sklearn.model_selection import HalvingGridSearchCV\nfrom sklearn import metrics\n#from sklearn.metrics import make_scorer,cohen_kappa_score\nX_tr,X_tst, Y_tr,Y_tst = train_test_split(all_train_data[selected_features],all_train_data[model_classes])\nmodel = XGBClassifier(objective= 'multi:softmax',min_child_weight=1.0,reg_alpha=0.01,\n                      learning_rate=0.1,\n                      verbosity=1,random_state=69,n_jobs=16)\n\nparams = {'max_depth':[8],'n_estimators':[200],'min_child_weight':[5],\n          'gamma':[0.25],'alpha' : [0.001],'subsample':[.8],\n          'colsample_bytree':[0.8]}\n\nxgb = HalvingGridSearchCV(model,params,refit=True,verbose=1,\n                    error_score='raise',aggressive_elimination=True,\n                          scoring='precision_macro',random_state=69,\n                    cv=5,return_train_score=True)\nxgb.fit(X_tr.values,Y_tr.values)\nprint('final result')\nprint(xgb.best_estimator_)\nprint(xgb.best_params_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost, GridSearchCV estimator","metadata":{}},{"cell_type":"code","source":"#Evaluation \nout_eval = xgb.predict(X_tst.values)\neval_precision = metrics.precision_score(Y_tst,out_eval,average='weighted')\neval_accuracy = metrics.accuracy_score(Y_tst,out_eval)\neval_confmat = metrics.confusion_matrix(Y_tst,out_eval)\nprint(f'the evaluation precision score is: {eval_precision}')\nprint(f'the evaluation accuracy score is: {eval_accuracy}')\nprint(f'the evaluation confusion matrix is : {eval_confmat}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing the model and generating the output","metadata":{}},{"cell_type":"code","source":"defog_test_dir = os.path.join(path,test_path,defog_path)\ntdcsfog_test_dir = os.path.join(path,test_path,tdcsfog_path)\ndefog_tst_df = []\ntdcsfog_tst_df = []\nfor dirname, _, filenames in os.walk(defog_test_dir):\n    for filename in filenames:\n        df= pd.read_csv(os.path.join(dirname,filename))\n        df['Id'] = filename[0:filename.index('.')]\n        defog_tst_df.append(df.copy())\nfor dirname, _, filenames in os.walk(tdcsfog_test_dir):\n    for filename in filenames:\n        df= pd.read_csv(os.path.join(dirname,filename))\n        df['Id'] = filename[0:filename.index('.')]\n        tdcsfog_tst_df.append(df.copy())\n\ndefog_tst_df = pd.concat(defog_tst_df)\ntdcsfog_tst_df = pd.concat(tdcsfog_tst_df)\ntst_df = pd.concat([tdcsfog_tst_df,defog_tst_df],ignore_index=True)\ntst_df[accmeasurs] = scaler.transform(tst_df[accmeasurs])\ntst_df = featureEngineer(tst_df.groupby('Id'))\ntst_df = imputeInf(tst_df)\ntst_df = tst_df.fillna(0)\ntst_df = merge_Id_Time_cols(tst_df)\nselected_features.remove('cluster')\ntst_df['cluster'] = kmean_model.predict(tst_df[selected_features])\nselected_features.append('cluster')\ntdcsfog_tst_df = None\ndefog_tst_df = None\ntst_df.head()\noutput_tst = xgb.predict(tst_df[selected_features].values)\nsubmission_array = np.zeros((len(output_tst),3))\nsubmission_array[np.argwhere(output_tst==1),class_toindex[1]] =1\nsubmission_array[np.argwhere(output_tst==2),class_toindex[2]] =1\nsubmission_array[np.argwhere(output_tst==3),class_toindex[3]] =1\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_array.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(data = submission_array,columns=targets)\ndf['Id']=tst_df['Id']\ndf = df[['Id','StartHesitation','Turn','Walking']]\ndf.to_csv('submission.csv',index=False)\ndf.to_csv('/kaggle/working/XGBsubmission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}