{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np # linear algebra\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.ensemble import RandomForestClassifier\n%matplotlib inline\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T12:03:54.715395Z","iopub.execute_input":"2022-07-24T12:03:54.716157Z","iopub.status.idle":"2022-07-24T12:03:55.080486Z","shell.execute_reply.started":"2022-07-24T12:03:54.716118Z","shell.execute_reply":"2022-07-24T12:03:55.079372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4장","metadata":{}},{"cell_type":"markdown","source":"titanic_setting","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nsubmission = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')\nsubmission\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:55.082316Z","iopub.execute_input":"2022-07-24T12:03:55.082644Z","iopub.status.idle":"2022-07-24T12:03:55.094821Z","shell.execute_reply.started":"2022-07-24T12:03:55.082613Z","shell.execute_reply":"2022-07-24T12:03:55.093986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"titanic_head","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\ntitanic.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:55.095949Z","iopub.execute_input":"2022-07-24T12:03:55.096574Z","iopub.status.idle":"2022-07-24T12:03:55.125637Z","shell.execute_reply.started":"2022-07-24T12:03:55.096540Z","shell.execute_reply":"2022-07-24T12:03:55.124846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"histogram, bin","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.histplot(data=titanic, x='age', bins = 20, binrange = (0, 80));","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:55.127319Z","iopub.execute_input":"2022-07-24T12:03:55.127996Z","iopub.status.idle":"2022-07-24T12:03:55.393953Z","shell.execute_reply.started":"2022-07-24T12:03:55.127966Z","shell.execute_reply":"2022-07-24T12:03:55.392611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"histogram, hue","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.histplot(data=titanic, x='age', hue='alive', multiple='stack');","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:55.396128Z","iopub.execute_input":"2022-07-24T12:03:55.396594Z","iopub.status.idle":"2022-07-24T12:03:55.708031Z","shell.execute_reply.started":"2022-07-24T12:03:55.396552Z","shell.execute_reply":"2022-07-24T12:03:55.706856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"displot: 수치형 데이터 하나의 분포를 나타내는 그래프\n\n파라미터 조정으로 histplot(히스토그램), kdeplot(커널밀도함수)를 모두 나타낼 수 있음","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.displot(data=titanic, x='age')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:55.709410Z","iopub.execute_input":"2022-07-24T12:03:55.709749Z","iopub.status.idle":"2022-07-24T12:03:56.005450Z","shell.execute_reply.started":"2022-07-24T12:03:55.709716Z","shell.execute_reply":"2022-07-24T12:03:56.004458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.displot(data=titanic, x='age', kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:56.007126Z","iopub.execute_input":"2022-07-24T12:03:56.007488Z","iopub.status.idle":"2022-07-24T12:03:56.268866Z","shell.execute_reply.started":"2022-07-24T12:03:56.007454Z","shell.execute_reply":"2022-07-24T12:03:56.267791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.displot(data=titanic, x='age', kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:56.270218Z","iopub.execute_input":"2022-07-24T12:03:56.270581Z","iopub.status.idle":"2022-07-24T12:03:56.571129Z","shell.execute_reply.started":"2022-07-24T12:03:56.270553Z","shell.execute_reply":"2022-07-24T12:03:56.570349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rugplot: 커널밀도 함수와 함께 샘플의 위치를 나타낼 때 사용\n\nhttps://blog.daum.net/tlos6733/136","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.kdeplot(data=titanic, x='age')\nsns.rugplot(data=titanic, x='age')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:56.574222Z","iopub.execute_input":"2022-07-24T12:03:56.574759Z","iopub.status.idle":"2022-07-24T12:03:56.795032Z","shell.execute_reply.started":"2022-07-24T12:03:56.574725Z","shell.execute_reply":"2022-07-24T12:03:56.793984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"barplot: 부트스트랩 추정값까지 나타냄, 둘 이상의 파라미터의 관계성 확인 가능","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='class', y='fare', data=titanic); #평균값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:56.798336Z","iopub.execute_input":"2022-07-24T12:03:56.798678Z","iopub.status.idle":"2022-07-24T12:03:57.020003Z","shell.execute_reply.started":"2022-07-24T12:03:56.798650Z","shell.execute_reply":"2022-07-24T12:03:57.018600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"hue를 적절히 이용하여 3개까지 가능","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='embarked', y='fare', hue = 'class', data=titanic); #평균값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:57.022204Z","iopub.execute_input":"2022-07-24T12:03:57.023228Z","iopub.status.idle":"2022-07-24T12:03:57.422130Z","shell.execute_reply.started":"2022-07-24T12:03:57.023175Z","shell.execute_reply":"2022-07-24T12:03:57.421085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='sibsp', y='survived', data=titanic); #평균값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:57.423527Z","iopub.execute_input":"2022-07-24T12:03:57.424095Z","iopub.status.idle":"2022-07-24T12:03:57.782759Z","shell.execute_reply.started":"2022-07-24T12:03:57.424063Z","shell.execute_reply":"2022-07-24T12:03:57.781602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"parch를 \"0 vs 1이상\"으로 나누어 살펴보면 더 좋을 것 같은데, 방법이 없을 지","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\ntitanic = sns.load_dataset('titanic')\n\ntitanic['parch10'] = titanic['parch'].map({0: '0',\n                                           1: '1', \n                                           2: '1', \n                                           3: '1', \n                                           4: '1', \n                                           5: '1', \n                                           6: '1'})\n\ntitanic['sibsp10'] = titanic['sibsp'].map({0: '0',\n                                           1: '1', \n                                           2: '1', \n                                           3: '1', \n                                           4: '1', \n                                           5: '1', \n                                           6: '1'})\n\n\n\n#--------------------------------------------------\n#titanic['parch1'] = titanic['parch'].apply(lambda x: x == 0)\n#titanic['parch0'] = titanic['parch'].apply(lambda x: x >= 1)\n\n#titanic.loc[titanic[\"parch\"] == 0]\n#titanic[titanic[\"parch\"] == 0 and titanic[\"survived\"]==1]\n\n\n#parch0 = 0\n#parch1 = 0\n\n#while titanic.parch == 0;\n #   parch0 = parch0 + 1\n    \n#while titanic.parch >= 1;\n #   parch1 = parch1 + 1\n\n#titanic_pivot = titanic.pivot(index = 'sibsp10', \n #                             columns = 'parch10', \n  #                            values = 'survived')\n#titanic_pivot","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:17:48.013144Z","iopub.execute_input":"2022-07-24T12:17:48.013884Z","iopub.status.idle":"2022-07-24T12:17:48.032462Z","shell.execute_reply.started":"2022-07-24T12:17:48.013842Z","shell.execute_reply":"2022-07-24T12:17:48.031005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='sibsp10', y='survived', data=titanic)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:17:52.334917Z","iopub.execute_input":"2022-07-24T12:17:52.335364Z","iopub.status.idle":"2022-07-24T12:17:52.540922Z","shell.execute_reply.started":"2022-07-24T12:17:52.335297Z","shell.execute_reply":"2022-07-24T12:17:52.539400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='parch10', y='survived', data=titanic)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:17:55.268861Z","iopub.execute_input":"2022-07-24T12:17:55.269278Z","iopub.status.idle":"2022-07-24T12:17:55.476631Z","shell.execute_reply.started":"2022-07-24T12:17:55.269240Z","shell.execute_reply":"2022-07-24T12:17:55.475397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='alone', y='survived', data=titanic)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:18:56.735045Z","iopub.execute_input":"2022-07-24T12:18:56.735449Z","iopub.status.idle":"2022-07-24T12:18:57.095936Z","shell.execute_reply.started":"2022-07-24T12:18:56.735415Z","shell.execute_reply":"2022-07-24T12:18:57.094722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://ncdg.tistory.com/1","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='parch', y='survived', data=titanic); #평균값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:57.833451Z","iopub.execute_input":"2022-07-24T12:03:57.834412Z","iopub.status.idle":"2022-07-24T12:03:58.153942Z","shell.execute_reply.started":"2022-07-24T12:03:57.834371Z","shell.execute_reply":"2022-07-24T12:03:58.152851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='class', y='fare', data=titanic, estimator=np.median); #중앙값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:58.155477Z","iopub.execute_input":"2022-07-24T12:03:58.155801Z","iopub.status.idle":"2022-07-24T12:03:58.486561Z","shell.execute_reply.started":"2022-07-24T12:03:58.155771Z","shell.execute_reply":"2022-07-24T12:03:58.485810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='class', y='fare', data=titanic, estimator=np.max); #최댓값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:58.487512Z","iopub.execute_input":"2022-07-24T12:03:58.488290Z","iopub.status.idle":"2022-07-24T12:03:58.836595Z","shell.execute_reply.started":"2022-07-24T12:03:58.488253Z","shell.execute_reply":"2022-07-24T12:03:58.835747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.barplot(x='class', y='fare', data=titanic, estimator=np.min); #최솟값","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:58.837584Z","iopub.execute_input":"2022-07-24T12:03:58.838652Z","iopub.status.idle":"2022-07-24T12:03:59.047628Z","shell.execute_reply.started":"2022-07-24T12:03:58.838616Z","shell.execute_reply":"2022-07-24T12:03:59.046280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"여러 그래프를 한 화면에 그려서 비교하는 경우\n-> 아래쪽에 그린 포인트플롯이 막대그래프에 비해 유리","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.boxplot(x='class', y='age', data=titanic)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:59.048878Z","iopub.execute_input":"2022-07-24T12:03:59.049221Z","iopub.status.idle":"2022-07-24T12:03:59.243649Z","shell.execute_reply.started":"2022-07-24T12:03:59.049175Z","shell.execute_reply":"2022-07-24T12:03:59.242805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.violinplot(x='class', y='age', data=titanic);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:59.244786Z","iopub.execute_input":"2022-07-24T12:03:59.245239Z","iopub.status.idle":"2022-07-24T12:03:59.448762Z","shell.execute_reply.started":"2022-07-24T12:03:59.245210Z","shell.execute_reply":"2022-07-24T12:03:59.447962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.violinplot(x='class', y='age', hue='sex', data=titanic, split=True);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:59.449902Z","iopub.execute_input":"2022-07-24T12:03:59.450356Z","iopub.status.idle":"2022-07-24T12:03:59.688624Z","shell.execute_reply.started":"2022-07-24T12:03:59.450328Z","shell.execute_reply":"2022-07-24T12:03:59.687856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.countplot(x='class', data=titanic);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:59.689791Z","iopub.execute_input":"2022-07-24T12:03:59.690249Z","iopub.status.idle":"2022-07-24T12:03:59.844249Z","shell.execute_reply.started":"2022-07-24T12:03:59.690221Z","shell.execute_reply":"2022-07-24T12:03:59.843116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.countplot(y='class', data=titanic);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:03:59.845467Z","iopub.execute_input":"2022-07-24T12:03:59.846100Z","iopub.status.idle":"2022-07-24T12:04:00.000934Z","shell.execute_reply.started":"2022-07-24T12:03:59.846061Z","shell.execute_reply":"2022-07-24T12:04:00.000072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ntitanic = sns.load_dataset('titanic')\nsns.countplot(data=titanic, x='pclass', hue='alive');","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:00.002096Z","iopub.execute_input":"2022-07-24T12:04:00.002562Z","iopub.status.idle":"2022-07-24T12:04:00.193057Z","shell.execute_reply.started":"2022-07-24T12:04:00.002532Z","shell.execute_reply":"2022-07-24T12:04:00.191859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"카운트플롯은 기본적으로 하나의 파라미터에 대한 csv의 기본값(모든 파라밑에 대해 인원수를 나타내는 csv라면 인원수)을 나타내줌.\n하지만 바플롯은 두 개의 파라미터간의 관계를 보여주기 때문에 파라미터 두 개 입력","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ntitanic = sns.load_dataset('titanic')\nx = [10, 60, 30]\nlabels = ['A', 'B', 'C']\n          \nplt.pie(x=x, labels=labels, autopct='%.1f%%');\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:00.194388Z","iopub.execute_input":"2022-07-24T12:04:00.194724Z","iopub.status.idle":"2022-07-24T12:04:00.292439Z","shell.execute_reply.started":"2022-07-24T12:04:00.194693Z","shell.execute_reply":"2022-07-24T12:04:00.290588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nflights = sns.load_dataset('flights')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:00.294844Z","iopub.execute_input":"2022-07-24T12:04:00.295426Z","iopub.status.idle":"2022-07-24T12:04:01.574947Z","shell.execute_reply.started":"2022-07-24T12:04:00.295377Z","shell.execute_reply":"2022-07-24T12:04:01.573393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flights.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:01.576467Z","iopub.execute_input":"2022-07-24T12:04:01.576902Z","iopub.status.idle":"2022-07-24T12:04:01.588011Z","shell.execute_reply.started":"2022-07-24T12:04:01.576861Z","shell.execute_reply":"2022-07-24T12:04:01.586909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"연도의 월별 승객 수를 알기 위해서 달을 행으로, 연을 열로, 합산할 데이터를 승객 수로 지정하기","metadata":{}},{"cell_type":"code","source":"flights_pivot = flights.pivot(index='month', \n                            columns = 'year', \n                            values='passengers')\nflights_pivot","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:01.589697Z","iopub.execute_input":"2022-07-24T12:04:01.590745Z","iopub.status.idle":"2022-07-24T12:04:01.614132Z","shell.execute_reply.started":"2022-07-24T12:04:01.590699Z","shell.execute_reply":"2022-07-24T12:04:01.613362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"추이를 더 쉽게 파악하기 위해서는 히트맵으로 표현","metadata":{}},{"cell_type":"code","source":"sns.heatmap(data=flights_pivot)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:01.618770Z","iopub.execute_input":"2022-07-24T12:04:01.619399Z","iopub.status.idle":"2022-07-24T12:04:01.917716Z","shell.execute_reply.started":"2022-07-24T12:04:01.619364Z","shell.execute_reply":"2022-07-24T12:04:01.916505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"두 데이터 간의 관계는 라인플롯으로","metadata":{}},{"cell_type":"code","source":"sns.lineplot(x='year', y='passengers', data=flights);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:01.919396Z","iopub.execute_input":"2022-07-24T12:04:01.920403Z","iopub.status.idle":"2022-07-24T12:04:02.342291Z","shell.execute_reply.started":"2022-07-24T12:04:01.920359Z","shell.execute_reply":"2022-07-24T12:04:02.341381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tips = sns.load_dataset('tips')\ntips.head","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:02.343366Z","iopub.execute_input":"2022-07-24T12:04:02.344527Z","iopub.status.idle":"2022-07-24T12:04:03.378171Z","shell.execute_reply.started":"2022-07-24T12:04:02.344486Z","shell.execute_reply":"2022-07-24T12:04:03.377098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='total_bill', y='tip', data=tips);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:03.379712Z","iopub.execute_input":"2022-07-24T12:04:03.380420Z","iopub.status.idle":"2022-07-24T12:04:03.559660Z","shell.execute_reply.started":"2022-07-24T12:04:03.380373Z","shell.execute_reply":"2022-07-24T12:04:03.558522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='total_bill', y='tip', hue='time', data=tips)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:03.561096Z","iopub.execute_input":"2022-07-24T12:04:03.562143Z","iopub.status.idle":"2022-07-24T12:04:03.791010Z","shell.execute_reply.started":"2022-07-24T12:04:03.562104Z","shell.execute_reply":"2022-07-24T12:04:03.789943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.regplot(x='total_bill', y='tip', data=tips);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:03.792255Z","iopub.execute_input":"2022-07-24T12:04:03.792689Z","iopub.status.idle":"2022-07-24T12:04:04.068467Z","shell.execute_reply.started":"2022-07-24T12:04:03.792655Z","shell.execute_reply":"2022-07-24T12:04:04.067580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.regplot(x='total_bill', y='tip',ci=99, data=tips);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.069778Z","iopub.execute_input":"2022-07-24T12:04:04.070107Z","iopub.status.idle":"2022-07-24T12:04:04.340091Z","shell.execute_reply.started":"2022-07-24T12:04:04.070077Z","shell.execute_reply":"2022-07-24T12:04:04.339106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5장","metadata":{}},{"cell_type":"markdown","source":"회기 평가지표 구하는 코드","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, mean_squared_log_error, r2_score\n\ntrue = np.array([1, 2, 3, 2, 3, 5, 4, 6, 5, 6, 7, 8, 8])\npreds = np.array([1, 1, 2, 2, 3, 4, 4, 5, 5, 7, 7, 6, 8])\n\nMAE = mean_absolute_error(true, preds)\nMSE = mean_squared_error(true, preds)\nRMSE = np.sqrt(MSE)\nMSLE = mean_squared_log_error(true, preds)\nRMSLE = np.sqrt(MSLE)\nR2 = r2_score(true, preds)\n\nprint(f'MAE:\\t {MAE: .4f}') #여기서 .4의 역할이 무엇일까....\nprint(f'MSE:\\t {MSE: .4f}')\nprint(f'RMSE:\\t {RMSE: .4f}')\nprint(f'MSLE:\\t {MSLE: .4f}')\nprint(f'RMSLE:\\t {RMSLE: .4f}')\nprint(f'R2:\\t {R2: .4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.341617Z","iopub.execute_input":"2022-07-24T12:04:04.342289Z","iopub.status.idle":"2022-07-24T12:04:04.357027Z","shell.execute_reply.started":"2022-07-24T12:04:04.342244Z","shell.execute_reply":"2022-07-24T12:04:04.355875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"판다스 원본 데이터 만들기","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nheight_weight_dict = {'키': [1.7, 1.5, 1.8], '몸무게': [75, 55, 60]}\ndf = pd.DataFrame(height_weight_dict, index=['광일', '혜성', '덕수'])\n\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.358457Z","iopub.execute_input":"2022-07-24T12:04:04.359816Z","iopub.status.idle":"2022-07-24T12:04:04.370349Z","shell.execute_reply.started":"2022-07-24T12:04:04.359770Z","shell.execute_reply":"2022-07-24T12:04:04.369226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"K폴드 교차검증과 층화K폴드 교차검증","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold\n\ny = np.array(['스팸']*5 + ['일반']*45)\n\nfolds = KFold(n_splits = 5, shuffle = True)\n\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(y)):\n    print(f'Fold {idx+1} 훈련 데이터 타깃값:')\n    print(y[train_idx])\n    print(f'Fold {idx+1} 검증 데이터 타깃값:')\n    print(y[valid_idx], '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.371677Z","iopub.execute_input":"2022-07-24T12:04:04.371981Z","iopub.status.idle":"2022-07-24T12:04:04.386699Z","shell.execute_reply.started":"2022-07-24T12:04:04.371956Z","shell.execute_reply":"2022-07-24T12:04:04.385586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import StratifiedKFold\n\nx = np.array(range(50))\ny = np.array(['스팸']*5 + ['일반']*45)\n\nfolds = StratifiedKFold(n_splits = 5, shuffle = True)\n\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(x, y)):\n    print(f'Fold {idx+1} 훈련 데이터 타깃값:')\n    print(y[train_idx])\n    print(f'Fold {idx+1} 검증 데이터 타깃값:')\n    print(y[valid_idx], '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.387982Z","iopub.execute_input":"2022-07-24T12:04:04.388660Z","iopub.status.idle":"2022-07-24T12:04:04.399683Z","shell.execute_reply.started":"2022-07-24T12:04:04.388621Z","shell.execute_reply":"2022-07-24T12:04:04.398798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"선형회귀모델_데이터생성","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nnp.random.seed(0)\n\nw0 = 5\nw1 = 2\nnoise = np.random.randn(100, 1)\n\nx = 4*np.random.rand(100, 1)\ny = w1*x +w0 +noise\n\nplt.scatter(x,y);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.400871Z","iopub.execute_input":"2022-07-24T12:04:04.401667Z","iopub.status.idle":"2022-07-24T12:04:04.586167Z","shell.execute_reply.started":"2022-07-24T12:04:04.401637Z","shell.execute_reply":"2022-07-24T12:04:04.585333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"선형회귀모델_모델훈련","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nlinear_reg_model = LinearRegression()\nlinear_reg_model.fit(x,y)\n\nprint('y절편(w0):', linear_reg_model.intercept_)\nprint('회기계수(w1):', linear_reg_model.coef_)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.587813Z","iopub.execute_input":"2022-07-24T12:04:04.588214Z","iopub.status.idle":"2022-07-24T12:04:04.605977Z","shell.execute_reply.started":"2022-07-24T12:04:04.588172Z","shell.execute_reply":"2022-07-24T12:04:04.604701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"선형회귀모델_회귀선확인\n\npredict()함수 이용하여 타깃값 예측","metadata":{}},{"cell_type":"code","source":"y_pred = linear_reg_model.predict(x)\n\nplt.scatter(x, y)\nplt.plot(x, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.607627Z","iopub.execute_input":"2022-07-24T12:04:04.608574Z","iopub.status.idle":"2022-07-24T12:04:04.804748Z","shell.execute_reply.started":"2022-07-24T12:04:04.608527Z","shell.execute_reply":"2022-07-24T12:04:04.803507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"로지스틱 회귀모델 (선형회귀 방식을 응용하여 분류 문제에 회기 사용)\n\n시그모이드 함수 그래프 이용 (x값이 작을수록 함숫값이 0에 가까움, x값이 0이면 함숫값 0.5, x값 클수록 함숫값 1에 가까움)\n\n-> 시그모이드 함수를 이용하여 구한 타깃값일 확률이 0.5 미만이면 음성으로 분류, 초과면 양성으로 분류","metadata":{}},{"cell_type":"markdown","source":"결정 트리 (분류와 회귀 모두에 사용 가능)\n\n여러 개의 노드(질문)로 결정\n\n분할 조건에 따라 만들어지는 트리 모양(=동작 효율) 달라짐\n\n분할을 통해 범주 내의 불순도(엔트로피, 지니 불순도를 최소로 만듦\n\n결정트리 정확도 산출 방법 -> 결정트리들로 결정하는 코드는 어떻게 쓰는 것일까","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_breast_cancer\n\ncancer_data = load_breast_cancer()\nX_train, X_test, y_train, y_test = train_test_split(cancer_data['data'], \n                                                    cancer_data['target'],\n                                                    stratify = cancer_data['target'], \n                                                    test_size = 0.4, \n                                                    random_state = 42)\n\ndecisiontree = DecisionTreeClassifier(random_state = 42)\ndecisiontree.fit(X_train, y_train)\n\naccuracy = decisiontree.score(X_test, y_test)\n\nprint(f'결정트리정확도: {accuracy: .3f}')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.806084Z","iopub.execute_input":"2022-07-24T12:04:04.806425Z","iopub.status.idle":"2022-07-24T12:04:04.905745Z","shell.execute_reply.started":"2022-07-24T12:04:04.806394Z","shell.execute_reply":"2022-07-24T12:04:04.904675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_breast_cancer\n\ncancer_data = load_breast_cancer()\nX_train, X_test, y_train, y_test = train_test_split(cancer_data['data'], \n                                                    cancer_data['target'],\n                                                    stratify = cancer_data['target'], \n                                                    test_size = 0.4, \n                                                    random_state = 42)\n\ndecisiontree = DecisionTreeClassifier(random_state = 42)\ndecisiontree.fit(X_train, y_train)\n\n\nprint(decisiontree.fit)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:04:04.907185Z","iopub.execute_input":"2022-07-24T12:04:04.907745Z","iopub.status.idle":"2022-07-24T12:04:04.930230Z","shell.execute_reply.started":"2022-07-24T12:04:04.907711Z","shell.execute_reply":"2022-07-24T12:04:04.929174Z"},"trusted":true},"execution_count":null,"outputs":[]}]}