{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T13:06:09.632872Z","iopub.execute_input":"2022-07-25T13:06:09.633437Z","iopub.status.idle":"2022-07-25T13:06:09.676798Z","shell.execute_reply.started":"2022-07-25T13:06:09.633329Z","shell.execute_reply":"2022-07-25T13:06:09.675193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 왜 log(y)를 사용하면 왼쪽으로 쏠린 분포의 데이터가 정규 분포에 가까워지는가? 라는 질문에 대한 실험을 조금 해보았습니다!","metadata":{}},{"cell_type":"markdown","source":"먼저 데이터를 불러옵니다.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:09.679497Z","iopub.execute_input":"2022-07-25T13:06:09.679853Z","iopub.status.idle":"2022-07-25T13:06:09.763260Z","shell.execute_reply.started":"2022-07-25T13:06:09.679819Z","shell.execute_reply":"2022-07-25T13:06:09.762065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"그 다음으로 관련 라이브러리를 불러옵니다.","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:09.764809Z","iopub.execute_input":"2022-07-25T13:06:09.765161Z","iopub.status.idle":"2022-07-25T13:06:10.438272Z","shell.execute_reply.started":"2022-07-25T13:06:09.765129Z","shell.execute_reply":"2022-07-25T13:06:10.436996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"트레인 데이터를 확인해보면 카운트 속성의 평균과 중앙값이 나타납니다. 이것을 뒤에서 참고할 예정입니다.","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:10.440644Z","iopub.execute_input":"2022-07-25T13:06:10.441190Z","iopub.status.idle":"2022-07-25T13:06:10.504340Z","shell.execute_reply.started":"2022-07-25T13:06:10.441148Z","shell.execute_reply":"2022-07-25T13:06:10.503001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"전체 구간에 대해서 히스토그램을 그려보려고 하였습니다.(흐릿하게 160이 넘어가는 선이 하나 있음을 볼 수 있습니다)","metadata":{}},{"cell_type":"code","source":"plt.rc('font', size = 15)\nsns.histplot(train['count'], bins = 976)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:10.507326Z","iopub.execute_input":"2022-07-25T13:06:10.507650Z","iopub.status.idle":"2022-07-25T13:06:12.186486Z","shell.execute_reply.started":"2022-07-25T13:06:10.507620Z","shell.execute_reply":"2022-07-25T13:06:12.184740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"대여 횟수 중 가장 높은 빈도의 대여 횟수를 정리해보았습니다. 결론적으로 1번 대여한 경우가 105번, 2번이 132, 5번 대여한 경우가 가장 많았습니다.(이것이 위에서 흐릿하게 나타나는 부분입니다)","metadata":{}},{"cell_type":"code","source":"count_values = train['count'].value_counts().sort_index()\ntop50 = count_values[:50]\ntop50","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:12.188236Z","iopub.execute_input":"2022-07-25T13:06:12.188730Z","iopub.status.idle":"2022-07-25T13:06:12.205723Z","shell.execute_reply.started":"2022-07-25T13:06:12.188687Z","shell.execute_reply":"2022-07-25T13:06:12.203849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1번 대여한 경우가 105번,이를 로그로 변환하면 4.65...가 됩니다.","metadata":{}},{"cell_type":"code","source":"np.log(105) # 1번 등장한 데이터의 갯수는 105개 -> 로그 변환하면 4.65","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:12.207482Z","iopub.execute_input":"2022-07-25T13:06:12.209718Z","iopub.status.idle":"2022-07-25T13:06:12.219002Z","shell.execute_reply.started":"2022-07-25T13:06:12.209613Z","shell.execute_reply":"2022-07-25T13:06:12.217756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"145를 로그로 변환하면 4.97... 실제 카운트 횟수는 40의 차이가 나지만 로그로 변환하면 0.32 정도의 차이로 그 차이가 줄어듭니다.","metadata":{}},{"cell_type":"code","source":"np.log(145) # 카운트의 중앙값은 145 -> 로그 변환하면 4.9767.. \n# 105에서 145까지 0.32 밖에 차이가 나지 않도록 촘촘하게 분포가 됨\n# 따라서 중앙값을 기준으로 정규 분포에 가까운 형태로 변환됨\n# 참고 : 1 -> 로그 변환하면 -> 0, 2 -> 0.693147, 3 -> 1.098612으로 변환되는데 낮은 숫자들은 로그 변환하면 넓게 퍼짐","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:12.220476Z","iopub.execute_input":"2022-07-25T13:06:12.220759Z","iopub.status.idle":"2022-07-25T13:06:12.234065Z","shell.execute_reply.started":"2022-07-25T13:06:12.220733Z","shell.execute_reply":"2022-07-25T13:06:12.232761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이어서 대여 횟수 1회부터 50회까지를 로그로 취한 결과를 확인해보면 확실히 편차가 적다는 것을 알 수 있습니다.","metadata":{}},{"cell_type":"code","source":"np.log(top50)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:12.235397Z","iopub.execute_input":"2022-07-25T13:06:12.235889Z","iopub.status.idle":"2022-07-25T13:06:12.249732Z","shell.execute_reply.started":"2022-07-25T13:06:12.235843Z","shell.execute_reply":"2022-07-25T13:06:12.248699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"반대로 827회 대여한 날은 하루, 830번도 하루, 이런 식으로 높은 대여량을 보인 날들은 며칠 없습니다.","metadata":{}},{"cell_type":"code","source":"tail = count_values[-50:]\ntail","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:12.250941Z","iopub.execute_input":"2022-07-25T13:06:12.251278Z","iopub.status.idle":"2022-07-25T13:06:12.262776Z","shell.execute_reply.started":"2022-07-25T13:06:12.251244Z","shell.execute_reply":"2022-07-25T13:06:12.260768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"여기까지 내용을 참고해서 카운트값을 로그를 취한 결과가 몇 번씩 등장했나 확인을 해도, 결국 위에서 확인한 것과 동일하게 나오긴 합니다. 즉, 1.609438은 5번 대여한 날이겠죠?","metadata":{}},{"cell_type":"code","source":"np.log(train['count']).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:22.078289Z","iopub.execute_input":"2022-07-25T13:06:22.078740Z","iopub.status.idle":"2022-07-25T13:06:22.090885Z","shell.execute_reply.started":"2022-07-25T13:06:22.078706Z","shell.execute_reply":"2022-07-25T13:06:22.090142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이를 조금 더 시각적으로 표현하면 아래와 같습니다. 그리고 그것을 다시 로그로 취하면 이렇게 나타납니다. 높이의 순서는 동일하지만, 높이의 편차는 다르다는 것을 확인할 수 있습니다.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,2))\ntop50.plot.bar()\nplt.figure(figsize = (20,2))\nnp.log(top50).plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:25.120465Z","iopub.execute_input":"2022-07-25T13:06:25.120894Z","iopub.status.idle":"2022-07-25T13:06:26.280274Z","shell.execute_reply.started":"2022-07-25T13:06:25.120860Z","shell.execute_reply":"2022-07-25T13:06:26.279547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"위의 것은 처음 50개의 데이터만 시각화하였고, 전체적으로 시각화하면 아래와 같습니다.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,2))\ncount_values.plot.bar()\nplt.figure(figsize = (20,2))\nnp.log(count_values).plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:28.680227Z","iopub.execute_input":"2022-07-25T13:06:28.681742Z","iopub.status.idle":"2022-07-25T13:06:45.447313Z","shell.execute_reply.started":"2022-07-25T13:06:28.681683Z","shell.execute_reply":"2022-07-25T13:06:45.445749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이제 뒷 부분에 있는 대여량이 많은 날들(800번 이상 대여한 날들)을 살펴보면 1번, 2번, 3번 정도 밖에 나타나지 않습니다. 그리고 이를 로그로 변환하면 0 ~ 1.xx로 변환되는데, 위에서 살펴보았을 때 40의 차이가 로그로 변환한 후 0.32밖에 되지 않았던 것과 큰 차이가 납니다. 결국 작은 값은 넓게 벌려주고, 큰 값은 촘촘하게 만들어주는 모습을 볼 수 있습니다.  \n여러분은 이해가 되셨는지 모르겠지만, 저는 실험적으로 '아..그렇네?'라고 생각하게 되었답니다ㅎㅎ  \n읽어주셔서 감사합니다!! :)","metadata":{}},{"cell_type":"code","source":"# 카운트 값이 800번 이상인 데이터는 1 ~ 3 건 정도임\n# 이 값들을 로그 변환하면 1번 -> 0, 2번 -> 0.693147, 3번 -> 1.098612 로 변환됨 따라서 히스토그램으로 분포를 살펴보면 앞 부분에 띄엄띄엄한 형태로 표현됨\nplt.figure(figsize = (20,2))\ntail.plot.bar()\nplt.figure(figsize = (20,2))\nnp.log(tail).plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:06:45.449504Z","iopub.execute_input":"2022-07-25T13:06:45.449880Z","iopub.status.idle":"2022-07-25T13:06:46.634766Z","shell.execute_reply.started":"2022-07-25T13:06:45.449853Z","shell.execute_reply":"2022-07-25T13:06:46.633424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(np.log(train['count']))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:07:00.411340Z","iopub.execute_input":"2022-07-25T13:07:00.411759Z","iopub.status.idle":"2022-07-25T13:07:00.728531Z","shell.execute_reply.started":"2022-07-25T13:07:00.411725Z","shell.execute_reply":"2022-07-25T13:07:00.727274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tail50 = count_values[-50:]\ntail50","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:07:00.730740Z","iopub.execute_input":"2022-07-25T13:07:00.732578Z","iopub.status.idle":"2022-07-25T13:07:00.744201Z","shell.execute_reply.started":"2022-07-25T13:07:00.732526Z","shell.execute_reply":"2022-07-25T13:07:00.742259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.log(tail50)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:07:00.855153Z","iopub.execute_input":"2022-07-25T13:07:00.855789Z","iopub.status.idle":"2022-07-25T13:07:00.865182Z","shell.execute_reply.started":"2022-07-25T13:07:00.855756Z","shell.execute_reply":"2022-07-25T13:07:00.864253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}