{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport scipy.sparse as sp\n\nimport pickle\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom scipy.stats import shapiro, skew, kurtosis","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-08T19:12:35.578434Z","iopub.execute_input":"2022-09-08T19:12:35.578936Z","iopub.status.idle":"2022-09-08T19:12:36.322708Z","shell.execute_reply.started":"2022-09-08T19:12:35.578840Z","shell.execute_reply":"2022-09-08T19:12:36.321494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_obj(path):\n    with open(path, 'rb') as file:\n        obj = pickle.load(file)\n    return obj\n\ncite_feature_idx2name = load_obj(\"../input/msci-sparse-datasettraintest/cite_feature_idx2name.pkl\")\nlen(cite_feature_idx2name)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:12:37.098767Z","iopub.execute_input":"2022-09-08T19:12:37.100023Z","iopub.status.idle":"2022-09-08T19:12:37.129210Z","shell.execute_reply.started":"2022-09-08T19:12:37.099962Z","shell.execute_reply":"2022-09-08T19:12:37.127825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input = sp.load_npz(\"../input/msci-sparse-datasettraintest/train_cite_input_xsparse.npz\")\n(nsamples, nfeatures) = train_cite_input.get_shape()\nprint(\"Number of Samples:\", nsamples)\nprint(\"Number of features:\", nfeatures)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:12:39.101729Z","iopub.execute_input":"2022-09-08T19:12:39.102172Z","iopub.status.idle":"2022-09-08T19:12:53.818866Z","shell.execute_reply.started":"2022-09-08T19:12:39.102135Z","shell.execute_reply":"2022-09-08T19:12:53.817624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input = sp.csc_matrix(train_cite_input)\n(nsamples, nfeatures) = train_cite_input.get_shape()\nprint(\"Number of Samples:\", nsamples)\nprint(\"Number of features:\", nfeatures)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:12:53.821442Z","iopub.execute_input":"2022-09-08T19:12:53.822637Z","iopub.status.idle":"2022-09-08T19:13:00.497182Z","shell.execute_reply.started":"2022-09-08T19:12:53.822590Z","shell.execute_reply":"2022-09-08T19:13:00.495633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_feature_stats(sparse_data, nfeatures):\n    feat_stats=[]\n    for i in range(nfeatures):\n        if i%10000 ==0:\n            print(i)\n\n        data = sparse_data[:, i].data\n        nnz = len(data)\n        mean_value = 0\n        min_value = 0\n        max_value = 0\n        std_value = 0\n        sk=0\n        kurt = 0\n        \n        if nnz > 0:\n            max_value = np.max(data)\n            min_value = np.min(data)\n            mean_value = np.mean(data)\n            std_value = np.std(data)\n            sk = skew(data)\n            kurt = kurtosis(data)\n\n        feat_stats.append({\n            'feat': i,\n            'nnz': nnz,\n            'mean': mean_value,\n            'std': std_value,\n            'min_value': min_value,\n            'max_value': max_value,\n            'skew': sk,\n            'kurtosis': kurt\n        })\n    feat_stats = pd.DataFrame.from_dict(feat_stats)\n    return feat_stats","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:21:13.752054Z","iopub.execute_input":"2022-09-08T19:21:13.752469Z","iopub.status.idle":"2022-09-08T19:21:13.762843Z","shell.execute_reply.started":"2022-09-08T19:21:13.752436Z","shell.execute_reply":"2022-09-08T19:21:13.761493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_stats = get_feature_stats(train_cite_input, nfeatures)\nfeat_stats['featname'] = feat_stats.feat.apply(lambda k: cite_feature_idx2name[k])\nfeat_stats['coverage'] = 100*feat_stats['nnz']/nsamples\n\nfeat_stats.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:13:00.511580Z","iopub.execute_input":"2022-09-08T19:13:00.512021Z","iopub.status.idle":"2022-09-08T19:13:18.500684Z","shell.execute_reply.started":"2022-09-08T19:13:00.511908Z","shell.execute_reply":"2022-09-08T19:13:18.499094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_stats.describe()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:13:18.502386Z","iopub.execute_input":"2022-09-08T19:13:18.502796Z","iopub.status.idle":"2022-09-08T19:13:18.555118Z","shell.execute_reply.started":"2022-09-08T19:13:18.502761Z","shell.execute_reply":"2022-09-08T19:13:18.553887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of features with no records:\", len(feat_stats[feat_stats.nnz == 0]))\nprint()\nfeat_stats[feat_stats.nnz == 0].head()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:13:18.557074Z","iopub.execute_input":"2022-09-08T19:13:18.557889Z","iopub.status.idle":"2022-09-08T19:13:18.581723Z","shell.execute_reply.started":"2022-09-08T19:13:18.557846Z","shell.execute_reply":"2022-09-08T19:13:18.580839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(1, 5 , figsize=(17, 5))\n\nax[0].set_title(\"count distribution \\nof each feature\")\nax[1].set_title(\"mean distribution \\nof each feature\")\nax[2].set_title(\"std distribution \\nof each feature\")\nax[3].set_title(\"min-value distribution \\nof each feature\")\nax[4].set_title(\"max-value distribution \\nof each feature\")\n\nax[0].hist(feat_stats[feat_stats.coverage>0.01]['coverage'], 100)\nax[1].hist(feat_stats[feat_stats['mean']>0]['mean'], 100)\nax[2].hist(feat_stats[feat_stats['std']>0]['std'], 100)\nax[3].hist(feat_stats[feat_stats['min_value']>0]['min_value'], 100)\nax[4].hist(feat_stats[feat_stats['max_value']>0]['max_value'], 100)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:13:18.582952Z","iopub.execute_input":"2022-09-08T19:13:18.586328Z","iopub.status.idle":"2022-09-08T19:13:20.010585Z","shell.execute_reply.started":"2022-09-08T19:13:18.586289Z","shell.execute_reply":"2022-09-08T19:13:20.009298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(feat_stats[feat_stats.coverage>0.9])/nsamples","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:13:20.011980Z","iopub.execute_input":"2022-09-08T19:13:20.012360Z","iopub.status.idle":"2022-09-08T19:13:20.023693Z","shell.execute_reply.started":"2022-09-08T19:13:20.012326Z","shell.execute_reply":"2022-09-08T19:13:20.022266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. 75% of the features have coverage of < 37% of train dataset\n2. Many features have <1 variance\n3. 22% of the features have coverage of >=90%\n\n--> 22% of the features are more frequent\n\nHigh available features will be seen more during the training process and the gradient comare to the rare features.","metadata":{}},{"cell_type":"markdown","source":"# lets check distributions for the frequent features","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots(1, 5 , figsize=(17, 5))\n\nax[0].set_title(\"count distribution \\nof each feature\")\nax[1].set_title(\"mean distribution \\nof each feature\")\nax[2].set_title(\"std distribution \\nof each feature\")\nax[3].set_title(\"min-value distribution \\nof each feature\")\nax[4].set_title(\"max-value distribution \\nof each feature\")\n\nax[0].hist(feat_stats[feat_stats.coverage>90]['coverage'], 100)\nax[1].hist(feat_stats[feat_stats.coverage>90]['mean'], 100)\nax[2].hist(feat_stats[feat_stats.coverage>90]['std'], 100)\nax[3].hist(feat_stats[feat_stats.coverage>90]['min_value'], 100)\nax[4].hist(feat_stats[feat_stats.coverage>90]['max_value'], 100)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:14:08.286499Z","iopub.execute_input":"2022-09-08T19:14:08.286994Z","iopub.status.idle":"2022-09-08T19:14:09.843363Z","shell.execute_reply.started":"2022-09-08T19:14:08.286955Z","shell.execute_reply":"2022-09-08T19:14:09.841985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(1, 5 , figsize=(17, 5))\n\nax[0].set_title(\"count distribution \\nof each feature\")\nax[1].set_title(\"mean distribution \\nof each feature\")\nax[2].set_title(\"std distribution \\nof each feature\")\nax[3].set_title(\"min-value distribution \\nof each feature\")\nax[4].set_title(\"max-value distribution \\nof each feature\")\n\nax[0].hist(feat_stats[feat_stats.coverage<10]['coverage'], 100)\nax[1].hist(feat_stats[feat_stats.coverage<10]['mean'], 100)\nax[2].hist(feat_stats[feat_stats.coverage<10]['std'], 100)\nax[3].hist(feat_stats[feat_stats.coverage<10]['min_value'], 100)\nax[4].hist(feat_stats[feat_stats.coverage<10]['max_value'], 100)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:14:10.074414Z","iopub.execute_input":"2022-09-08T19:14:10.076464Z","iopub.status.idle":"2022-09-08T19:14:11.575727Z","shell.execute_reply.started":"2022-09-08T19:14:10.076402Z","shell.execute_reply":"2022-09-08T19:14:11.574801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"More frequent features have high mean and deviation compared to the low frequent features","metadata":{}},{"cell_type":"markdown","source":"# lets analyse target values","metadata":{}},{"cell_type":"code","source":"train_cite_target = sp.load_npz(\"../input/msci-sparse-datasettraintest/train_cite_target_xsparse.npz\")\ntrain_cite_target = sp.csc_matrix(train_cite_target)\ntrain_cite_target","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:14:12.968043Z","iopub.execute_input":"2022-09-08T19:14:12.968449Z","iopub.status.idle":"2022-09-08T19:14:13.540711Z","shell.execute_reply.started":"2022-09-08T19:14:12.968417Z","shell.execute_reply":"2022-09-08T19:14:13.539525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(nsamples, nfeatures) = train_cite_target.get_shape()\nprint(\"Number of Samples:\", nsamples)\nprint(\"Number of features:\", nfeatures)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:17:47.659444Z","iopub.execute_input":"2022-09-08T19:17:47.660384Z","iopub.status.idle":"2022-09-08T19:17:47.666789Z","shell.execute_reply.started":"2022-09-08T19:17:47.660331Z","shell.execute_reply":"2022-09-08T19:17:47.665344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_stats = get_feature_stats(train_cite_target, nfeatures)\nfeat_stats['coverage'] = 100*feat_stats['nnz']/nsamples\nfeat_stats.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:21:27.707265Z","iopub.execute_input":"2022-09-08T19:21:27.707711Z","iopub.status.idle":"2022-09-08T19:21:27.970405Z","shell.execute_reply.started":"2022-09-08T19:21:27.707675Z","shell.execute_reply":"2022-09-08T19:21:27.969111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_stats.describe()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:21:30.162345Z","iopub.execute_input":"2022-09-08T19:21:30.162786Z","iopub.status.idle":"2022-09-08T19:21:30.205957Z","shell.execute_reply.started":"2022-09-08T19:21:30.162747Z","shell.execute_reply":"2022-09-08T19:21:30.204533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(1, 4, figsize=(17, 5))\n\nax[0].set_title(\"target mean distribution\")\nax[1].set_title(\"target std distribution\")\nax[2].set_title(\"target min-value distribution\")\nax[3].set_title(\"target max-value distribution\")\n\nax[0].hist(feat_stats['mean'], 100)\nax[1].hist(feat_stats['std'], 100)\nax[2].hist(feat_stats['min_value'], 100)\nax[3].hist(feat_stats['max_value'], 100)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T19:21:32.528388Z","iopub.execute_input":"2022-09-08T19:21:32.529535Z","iopub.status.idle":"2022-09-08T19:21:33.864464Z","shell.execute_reply.started":"2022-09-08T19:21:32.529484Z","shell.execute_reply":"2022-09-08T19:21:33.863096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. target distribution is far away from normal.\n2. Outliers in the target distribution\n3. Training process may be effected due the outliers.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}