{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train=pd.read_csv('../input/train.csv')#origin by chen du\ntest=pd.read_csv('../input/test.csv')\nsub=pd.read_csv('../input/sample_submission.csv')\ntarget=train['Target']\ntrain.columns\n#target.value_counts()如果我们是写成target=train['Target']那么target是Series类型就可以使用value_counts()方法\n#target.values\n#data=pd.concat([train,test],axis=0)\n#data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0616bd05f1c992e557699aa2b03053087f68447"},"cell_type":"code","source":"!pip list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc93f5dde03a8664debcabe0f38d1da489b33b2e"},"cell_type":"code","source":"#train.shape\n\n#len(train_feat)\n#len(test_feat)\nnew_train_feats,new_test_feats=[],[]\nfor i in train.columns:\n    if train[i].isnull().sum()/train.shape[0]>0.6:\n        continue\n    elif i=='Target':\n        continue\n    else:\n        new_train_feats.append(i)\n\nfor i in test.columns:\n    if test[i].isnull().sum()/test.shape[0]>0.6:\n        continue\n    else:\n        new_test_feats.append(i)\n\n#len(new_train_feats)\ntrain=train[new_train_feats]\ntest=test[new_test_feats]\n#train.isnull().sum().sort_values()\n#test.isnull().sum().sort_values()\n#train[['meaneduc','SQBmeaned']].describe()\n#train['meaneduc'].dtypes\n#train['SQBmeaned'].dtypes\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n#sns.kdeplot(train['meaneduc'],shade=True)#从频率分布图可以看到，数据分布有很强的集中性，主要集中在6-9之间，我们以7.5作为填充值填充空值\ntrain['meaneduc'].fillna(7.5,inplace=True)\n#train['meaneduc'].isnull().sum()\n#sns.kdeplot(train['SQBmeaned'],shade=True)#以50为填充值\ntrain['SQBmeaned'].fillna(50,inplace=True)\n#train['SQBmeaned'].isnull().sum()\n#sns.kdeplot(test['meaneduc'],shade=True)\ntest['meaneduc'].fillna(8,inplace=True)\n#test['meaneduc'].isnull().sum()\n#sns.kdeplot(train['SQBmeaned'],shade=True)#以40填充\ntest['SQBmeaned'].fillna(40,inplace=True)\n'''\n缺失值处理完毕\n'''\n'''train.isnull().any().any()\ntest.isnull().any().any()'''#检验,如果处理正确这两行输出应该是False\n#plt.plot(train['meaneduc'].values)\n'''\n-----------------处理训练集不同类别样本失衡问题\n'''\nfrom imblearn.over_sampling import RandomOverSampler\nros=RandomOverSampler(random_state=0)\ntrain_cols=train.columns\ntest_cols=test.columns\n#train_feat.shape\n#pd.Series(train_label).value_counts()#现在四个类都是5996个样本\n#train_df.head()\n#train_df.shape\n\n#data.shape\n\ndata_category_cols=['idhogar','dependency','edjefe','edjefa']\ntrain_id=train['Id']\ntest_id=test['Id']\n#train_id\n#test_id\ntrain_pure_feat=train[[i for i in train_cols if i!='Id']]\ntest_pure_feat=test[[i for i in test_cols if i!='Id']]\n#train_pure_feat.shape\n#test_pure_feat.shape\n\n#data=pd.get_dummies(data)\n#data.shape\n#data.iloc[0,:]\n#data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0f9c509d4b44fd2dcb2c68886e1eaf86b7d4619"},"cell_type":"code","source":"data=pd.concat([train_pure_feat,test_pure_feat],axis=0)\n#data.shape\n#data.info()\ndata=pd.get_dummies(data)\ndata.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b1da27dfdc9460f3c08903af428faf1e1e9f4c2"},"cell_type":"code","source":"train=data.iloc[0:9557,:]\n#train.shape\ntest=data.iloc[9557:,:]\n#test.head()\ntotal_train=pd.concat([train,target],axis=1)\ntotal_train.head()\n\ntrain_x,train_y=ros.fit_sample(train,target)\n#train_x.shape\n#train_y.shape\nfrom sklearn.model_selection import train_test_split\nxtrain,xvalid,ytrain,yvalid=train_test_split(train_x,train_y,test_size=0.2,random_state=0)\n#xtrain.shape\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom sklearn.ensemble import RandomForestClassifier\nrdc=RandomForestClassifier(n_estimators=20,n_jobs=-1,random_state=0).fit(xtrain,ytrain)\npred1=rdc.predict(xvalid)\npred1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aea5f1ae5b4a4cad9980775a14ef542cc32c7fe8"},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error,classification_report,precision_score,recall_score,f1_score,\nerror=mean_squared_error(pred1,yvalid)\n#error\nreport=classification_report(yvalid,pred1)\n#report\n\nprint(recall_score(yvalid,pred1,average='category'))\nprint(f1_score(yvalid,pred1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7dacb679a3a35ac144350d198ffd5838a2820fb"},"cell_type":"code","source":"pred2=rdc.predict(test)\nlen(pred2)==test_id.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1360f105acaaf2df6652e3f707b12af2cc59f64f"},"cell_type":"code","source":"result=pd.concat([test_id,pd.Series(pred2,name='Target')],axis=1)\nresult\nsub","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72b2d48a6bcb61dccbeffada4d6aa968aef4fd35"},"cell_type":"code","source":"result.to_csv('result.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eddb9736c206b142c1763bcf5da14b899a1afe76"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}