{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About this Project \n\n- Lecture name : 2022 K-Digital Training- Big Data analysis using AI Data Platforms\n- 교과목명 : 빅데이터 분석 및 시각화, AI개발 기초, 인공지능 프로그래밍\n- Project name : Kaggle Forecasting competition using Bike Sharing Demand Data\n- Poject Eng Date : July 19th 2022\n- Name : Hyung Ju Cha\n","metadata":{"_uuid":"341cc156b678b0918a022729669f6afa0305e468"}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:33:31.700836Z","iopub.execute_input":"2022-07-14T06:33:31.701580Z","iopub.status.idle":"2022-07-14T06:33:31.712487Z","shell.execute_reply.started":"2022-07-14T06:33:31.701505Z","shell.execute_reply":"2022-07-14T06:33:31.711039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 01. Loading necessary libraries","metadata":{}},{"cell_type":"code","source":"# Ignore  the warnings\nimport warnings\nwarnings.filterwarnings('always')\nwarnings.filterwarnings('ignore')\n\n# data visualisation and manipulation\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import style\nimport seaborn as sns\nimport missingno as msno\n#configure\n# sets matplotlib to inline and displays graphs below the corressponding cell.\n% matplotlib inline  \nstyle.use('fivethirtyeight')\nsns.set(style='whitegrid',color_codes=True)\n\n#import the necessary modelling algos.\n\n#classifiaction.\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC,SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier,GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.naive_bayes import GaussianNB\n\n#regression\nfrom sklearn.linear_model import LinearRegression,Ridge,Lasso,RidgeCV\nfrom sklearn.ensemble import RandomForestRegressor,BaggingRegressor,GradientBoostingRegressor,AdaBoostRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor\nimport xgboost as xgb\nimport lightgbm as lgb\n\n#model selection\nfrom sklearn.model_selection import train_test_split,cross_validate, cross_val_predict\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn.model_selection import GridSearchCV\n\n#evaluation metrics\nfrom sklearn.metrics import mean_squared_log_error,mean_squared_error, r2_score,mean_absolute_error # for regression\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score  # for classification\n\nprint(\"pandas version :\", pd.__version__)\nprint(\"numpy version :\", np.__version__)\nprint(\"seaborn version :\", sns.__version__)\nprint(\"xgboost version :\", xgb.__version__)\nprint(\"lightgbm version :\", lgb.__version__)\n ","metadata":{"_uuid":"3714683c93de6fe9db52ff4973510692a1d10b33","execution":{"iopub.status.busy":"2022-07-14T06:33:31.797534Z","iopub.execute_input":"2022-07-14T06:33:31.798125Z","iopub.status.idle":"2022-07-14T06:33:33.449617Z","shell.execute_reply.started":"2022-07-14T06:33:31.798074Z","shell.execute_reply":"2022-07-14T06:33:33.448407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 02. Loading Data","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv(r'../input/train.csv')\ntest=pd.read_csv(r'../input/test.csv')\ndf=train.copy()\ntest_df=test.copy()\ndf.head()","metadata":{"_uuid":"29ee14f4c117e6ac694e37305a46a8d56a118822","execution":{"iopub.status.busy":"2022-07-14T06:33:33.451143Z","iopub.execute_input":"2022-07-14T06:33:33.451511Z","iopub.status.idle":"2022-07-14T06:33:33.580744Z","shell.execute_reply.started":"2022-07-14T06:33:33.451441Z","shell.execute_reply":"2022-07-14T06:33:33.579805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 03. checking data","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:33:33.582209Z","iopub.execute_input":"2022-07-14T06:33:33.582580Z","iopub.status.idle":"2022-07-14T06:33:33.599467Z","shell.execute_reply.started":"2022-07-14T06:33:33.582508Z","shell.execute_reply":"2022-07-14T06:33:33.598723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:33:33.600377Z","iopub.execute_input":"2022-07-14T06:33:33.600595Z","iopub.status.idle":"2022-07-14T06:33:33.617106Z","shell.execute_reply.started":"2022-07-14T06:33:33.600560Z","shell.execute_reply":"2022-07-14T06:33:33.615358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- At this moment it seems there are no null values","metadata":{}},{"cell_type":"markdown","source":"## step 04. Exploratory data analysis\n- visualization\n- date based\n- Changing train data directly might cause confusion \n- Make a copy. (Exploratory data analysis)\n- Very small dataset\n    + We can use entire dataset\n    + sample a part of entire data","metadata":{}},{"cell_type":"code","source":"df.columns.unique()","metadata":{"_uuid":"4e374b33b0947c5e71b524344ca98972028b4e80","execution":{"iopub.status.busy":"2022-07-14T06:33:33.618653Z","iopub.execute_input":"2022-07-14T06:33:33.619000Z","iopub.status.idle":"2022-07-14T06:33:33.627840Z","shell.execute_reply.started":"2022-07-14T06:33:33.618941Z","shell.execute_reply":"2022-07-14T06:33:33.626646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### A SHORT DESCRIPTION OF THE FEATURES.\n\ndatetime - hourly date + timestamp  \n\nseason -  1 = spring, 2 = summer, 3 = fall, 4 = winter \n\nholiday - whether the day is considered a holiday\n\nworkingday - whether the day is neither a weekend nor holiday\n\nweather -\n\n1: Clear, Few clouds, Partly cloudy, Partly cloudy \n\n2: Mist + Cloudy, Mist + Broken clouds, Mist + Few clouds, Mist \n\n3: Light Snow, Light Rain + Thunderstorm + Scattered clouds, Light Rain + Scattered clouds \n\n4: Heavy Rain + Ice Pallets + Thunderstorm + Mist, Snow + Fog \n\ntemp - temperature in Celsius\n\natemp - \"feels like\" temperature in Celsius\n\nhumidity - relative humidity\n\nwindspeed - wind speed\n\ncasual - number of non-registered user rentals initiated\n\nregistered - number of registered user rentals initiated\n\ncount - number of total rentals","metadata":{"_uuid":"3209eb63e926c5a9a172ac856e8cbe234be5bc08"}},{"cell_type":"code","source":"df.info()","metadata":{"_uuid":"aae2ea236fd1d3a70f4444d507eed5026d1841e3","execution":{"iopub.status.busy":"2022-07-14T06:33:33.629706Z","iopub.execute_input":"2022-07-14T06:33:33.629934Z","iopub.status.idle":"2022-07-14T06:33:33.651269Z","shell.execute_reply.started":"2022-07-14T06:33:33.629899Z","shell.execute_reply":"2022-07-14T06:33:33.650654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()  # implies no null values and hence no imputation needed ::).","metadata":{"_uuid":"ede9438401bf029bfddbddee29116b0c399b6c3e","execution":{"iopub.status.busy":"2022-07-14T06:33:33.652573Z","iopub.execute_input":"2022-07-14T06:33:33.653106Z","iopub.status.idle":"2022-07-14T06:33:33.666962Z","shell.execute_reply.started":"2022-07-14T06:33:33.652844Z","shell.execute_reply":"2022-07-14T06:33:33.665700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df)  # just to visualize. no missing value.","metadata":{"_uuid":"8819819cf5c172ef6f02b72ca1f430da12eca583","execution":{"iopub.status.busy":"2022-07-14T06:33:33.668506Z","iopub.execute_input":"2022-07-14T06:33:33.669031Z","iopub.status.idle":"2022-07-14T06:33:34.574950Z","shell.execute_reply.started":"2022-07-14T06:33:33.668973Z","shell.execute_reply":"2022-07-14T06:33:34.574098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  NOW WE CAN EXPLORE OUR FEATURES. FIRST LETS EXPLORE THE DISTRIBUTION OF VARIOUS DISCRETE FEATURES LIKE weather , season etc... .","metadata":{"_uuid":"4e7b5435d670440f6928c0c35343a8c440416893"}},{"cell_type":"code","source":"# let us consider season.\ndf.season.value_counts()","metadata":{"_uuid":"80c0b0eeb79267103d7e0f47e11cd0dd787a8393","execution":{"iopub.status.busy":"2022-07-14T06:33:34.578279Z","iopub.execute_input":"2022-07-14T06:33:34.578551Z","iopub.status.idle":"2022-07-14T06:33:34.588713Z","shell.execute_reply.started":"2022-07-14T06:33:34.578505Z","shell.execute_reply":"2022-07-14T06:33:34.587365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.factorplot(x='season',data=df,kind='count',size=5,aspect=1)\nsns.factorplot(x='season',data=df,kind='count',size=5,aspect=1.5)","metadata":{"_uuid":"07a59826be47b0c10e8aa41955907262d0e0178c","execution":{"iopub.status.busy":"2022-07-14T06:33:34.590519Z","iopub.execute_input":"2022-07-14T06:33:34.590832Z","iopub.status.idle":"2022-07-14T06:33:34.855702Z","shell.execute_reply.started":"2022-07-14T06:33:34.590776Z","shell.execute_reply":"2022-07-14T06:33:34.854807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#holiday\ndf.holiday.value_counts()\nsns.factorplot(x='holiday',data=df,kind='count',size=5,aspect=1) # majority of data is for non holiday days.","metadata":{"_uuid":"a84974a01956d6e8ae5492f88d763211e1b4aaee","execution":{"iopub.status.busy":"2022-07-14T06:33:34.857201Z","iopub.execute_input":"2022-07-14T06:33:34.857796Z","iopub.status.idle":"2022-07-14T06:33:35.103168Z","shell.execute_reply.started":"2022-07-14T06:33:34.857737Z","shell.execute_reply":"2022-07-14T06:33:35.102247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#holiday\ndf.workingday.value_counts()\nsns.factorplot(x='workingday',data=df,kind='count',size=5,aspect=1) # majority of data is for working days.","metadata":{"_uuid":"51a7a3ebfbaefafa24b3515afb93fd1a3a008632","execution":{"iopub.status.busy":"2022-07-14T06:33:35.104579Z","iopub.execute_input":"2022-07-14T06:33:35.105133Z","iopub.status.idle":"2022-07-14T06:33:35.362399Z","shell.execute_reply.started":"2022-07-14T06:33:35.105064Z","shell.execute_reply":"2022-07-14T06:33:35.361456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weather\ndf.weather.value_counts()","metadata":{"_uuid":"4469180f5a72e11927ca8969c83672259bf824fc","execution":{"iopub.status.busy":"2022-07-14T06:33:35.363991Z","iopub.execute_input":"2022-07-14T06:33:35.364632Z","iopub.status.idle":"2022-07-14T06:33:35.375007Z","shell.execute_reply.started":"2022-07-14T06:33:35.364564Z","shell.execute_reply":"2022-07-14T06:33:35.374128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.factorplot(x='weather',data=df,kind='count',size=5,aspect=1)  \n# 1-> spring\n# 2-> summer\n# 3-> fall\n# 4-> winter","metadata":{"_uuid":"693e609795fbec7681820985bccaec9fb629fecf","execution":{"iopub.status.busy":"2022-07-14T06:33:35.376672Z","iopub.execute_input":"2022-07-14T06:33:35.377308Z","iopub.status.idle":"2022-07-14T06:33:35.642008Z","shell.execute_reply.started":"2022-07-14T06:33:35.377229Z","shell.execute_reply":"2022-07-14T06:33:35.641050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  NOW WE CAN  ALSO SEE DISTRIBUTION OF CONTINOUS VARIABLES.","metadata":{"_uuid":"102aced93acb7479348a0c7754c06c892e5f8041"}},{"cell_type":"code","source":"df.describe()","metadata":{"_uuid":"7438c866b1e8dbb04055d59759b50cb5077142da","execution":{"iopub.status.busy":"2022-07-14T06:33:35.643549Z","iopub.execute_input":"2022-07-14T06:33:35.644093Z","iopub.status.idle":"2022-07-14T06:33:35.741193Z","shell.execute_reply.started":"2022-07-14T06:33:35.644030Z","shell.execute_reply":"2022-07-14T06:33:35.740098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just to visualize.\nsns.boxplot(data=df[['temp',\n       'atemp', 'humidity', 'windspeed', 'casual', 'registered', 'count']])\nfig=plt.gcf()\nfig.set_size_inches(10,10)","metadata":{"_uuid":"726401270678c29ac7a4c4523feb297b8fc4fec6","execution":{"iopub.status.busy":"2022-07-14T06:33:35.742563Z","iopub.execute_input":"2022-07-14T06:33:35.742918Z","iopub.status.idle":"2022-07-14T06:33:36.128889Z","shell.execute_reply.started":"2022-07-14T06:33:35.742870Z","shell.execute_reply":"2022-07-14T06:33:36.127856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# can also be visulaized using histograms for all the continuous variables.\ndf.temp.unique()\nfig,axes=plt.subplots(2,2)\naxes[0,0].hist(x=\"temp\",data=df,edgecolor=\"black\",linewidth=2,color='#ff4125')\naxes[0,0].set_title(\"Variation of temp\")\naxes[0,1].hist(x=\"atemp\",data=df,edgecolor=\"black\",linewidth=2,color='#ff4125')\naxes[0,1].set_title(\"Variation of atemp\")\naxes[1,0].hist(x=\"windspeed\",data=df,edgecolor=\"black\",linewidth=2,color='#ff4125')\naxes[1,0].set_title(\"Variation of windspeed\")\naxes[1,1].hist(x=\"humidity\",data=df,edgecolor=\"black\",linewidth=2,color='#ff4125')\naxes[1,1].set_title(\"Variation of humidity\")\nfig.set_size_inches(10,10)","metadata":{"_uuid":"6be3e1df01d7fe136efc717480e08f2d05c4e3dd","execution":{"iopub.status.busy":"2022-07-14T06:33:36.130486Z","iopub.execute_input":"2022-07-14T06:33:36.131056Z","iopub.status.idle":"2022-07-14T06:33:37.074126Z","shell.execute_reply.started":"2022-07-14T06:33:36.130991Z","shell.execute_reply":"2022-07-14T06:33:37.073181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  NOW AFTER SEEING THE DISTRIBUTION OF VARIOUS DISCRETE AS WELL AS CONTINUOUS VARIABLES WE CAN SEE THE INTERREALTION B/W THEM USING A HEAT MAP.","metadata":{"_uuid":"3ce33b106ecd580890d903ce4951ad1c39fc3e18"}},{"cell_type":"code","source":"#corelation matrix.\ncor_mat= df[:].corr()\nmask = np.array(cor_mat)\nmask[np.tril_indices_from(mask)] = False\nfig=plt.gcf()\nfig.set_size_inches(30,12)\nsns.heatmap(data=cor_mat,mask=mask,square=True,annot=True,cbar=True)","metadata":{"_uuid":"52d92f76809b32f84e936a57ba582c9188aba311","execution":{"iopub.status.busy":"2022-07-14T06:33:37.075630Z","iopub.execute_input":"2022-07-14T06:33:37.076161Z","iopub.status.idle":"2022-07-14T06:33:38.023066Z","shell.execute_reply.started":"2022-07-14T06:33:37.076101Z","shell.execute_reply":"2022-07-14T06:33:38.016360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  INFERENCES FROM THE ABOVE HEATMAP--\n\n1. self realtion i.e. of a feature to itself is equal to 1 as expected.\n\n2. temp and atemp are highly related as expected.\n \n3. humidity is inversely related to count as expected as the weather is humid people will not like to travel on a bike.\n\n4. also note that casual and working day are highly inversely related as you would expect.\n\n5. Also note that count and holiday are highly inversely related as you would expect.\n\n6. Also note that temp(or atemp) highly effects the count. \n\n7. Also note that weather and count are highly inversely related. This is bcoz for uour data as weather increases from (1 to 4) implies that  weather is getting more worse and so lesser people will rent bikes.\n\n8. registered/casual and count are highly related which indicates that most of the bikes that are rented are registered.\n\n9. similarly we can draw some more inferences like weather and humidity and so on... .\n","metadata":{"_uuid":"62d325256592003e5fc438367a9e4cd023396c34"}},{"cell_type":"markdown","source":"######  NOW WE  CAN DO SOME FEATURE ENGINEERING AND GET SOME NEW FEATURES AND DROP SOME USELESS OR LESS RELEVANT FEATURES.","metadata":{"_uuid":"6d55facfabff7b9b685dca09bac2b6607df12d8e"}},{"cell_type":"code","source":"# # seperating season as per values. this is bcoz this will enhance features.\nseason=pd.get_dummies(df['season'],prefix='season')\ndf=pd.concat([df,season],axis=1)\ndf.head()\nseason=pd.get_dummies(test_df['season'],prefix='season')\ntest_df=pd.concat([test_df,season],axis=1)\ntest_df.head()","metadata":{"_uuid":"3d5d88ba071ae70848b9d5649ff1b29d5afaffc5","execution":{"iopub.status.busy":"2022-07-14T06:33:38.024714Z","iopub.execute_input":"2022-07-14T06:33:38.024981Z","iopub.status.idle":"2022-07-14T06:33:38.066607Z","shell.execute_reply.started":"2022-07-14T06:33:38.024929Z","shell.execute_reply":"2022-07-14T06:33:38.065851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # # same for weather. this is bcoz this will enhance features.\nweather=pd.get_dummies(df['weather'],prefix='weather')\ndf=pd.concat([df,weather],axis=1)\ndf.head()\nweather=pd.get_dummies(test_df['weather'],prefix='weather')\ntest_df=pd.concat([test_df,weather],axis=1)\ntest_df.head()","metadata":{"_uuid":"41cd4105b035ed61f789686bc1bba906f092496b","execution":{"iopub.status.busy":"2022-07-14T06:33:38.067694Z","iopub.execute_input":"2022-07-14T06:33:38.067913Z","iopub.status.idle":"2022-07-14T06:33:38.112974Z","shell.execute_reply.started":"2022-07-14T06:33:38.067873Z","shell.execute_reply":"2022-07-14T06:33:38.111929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # # now can drop weather and season.\ndf.drop(['season','weather'],inplace=True,axis=1)\ndf.head()\ntest_df.drop(['season','weather'],inplace=True,axis=1)\ntest_df.head()\n\n\n# # # also I dont prefer both registered and casual but for ow just let them both.","metadata":{"_uuid":"b98082ac0dd6d032e8b0094b024ceb2d00097b8a","execution":{"iopub.status.busy":"2022-07-14T06:33:38.114095Z","iopub.execute_input":"2022-07-14T06:33:38.114320Z","iopub.status.idle":"2022-07-14T06:33:38.153823Z","shell.execute_reply.started":"2022-07-14T06:33:38.114285Z","shell.execute_reply":"2022-07-14T06:33:38.153168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  now most importantly split the date and time as the time of day is expected to effect the no of bikes. for eg at office hours like early mornning or evening one would expect a greater demand of rental bikes.","metadata":{"_uuid":"7bfc03e0568857a40a2561b0566913103c0752c4"}},{"cell_type":"code","source":"df[\"hour\"] = [t.hour for t in pd.DatetimeIndex(df.datetime)]\ndf[\"day\"] = [t.dayofweek for t in pd.DatetimeIndex(df.datetime)]\ndf[\"month\"] = [t.month for t in pd.DatetimeIndex(df.datetime)]\ndf['year'] = [t.year for t in pd.DatetimeIndex(df.datetime)]\ndf['year'] = df['year'].map({2011:0, 2012:1})\ndf.head()","metadata":{"_uuid":"8e997d136e4b03fc9dd5eee8667da39ac6cfd9e6","execution":{"iopub.status.busy":"2022-07-14T06:33:38.155000Z","iopub.execute_input":"2022-07-14T06:33:38.155365Z","iopub.status.idle":"2022-07-14T06:33:38.297787Z","shell.execute_reply.started":"2022-07-14T06:33:38.155309Z","shell.execute_reply":"2022-07-14T06:33:38.296909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"hour\"] = [t.hour for t in pd.DatetimeIndex(test_df.datetime)]\ntest_df[\"day\"] = [t.dayofweek for t in pd.DatetimeIndex(test_df.datetime)]\ntest_df[\"month\"] = [t.month for t in pd.DatetimeIndex(test_df.datetime)]\ntest_df['year'] = [t.year for t in pd.DatetimeIndex(test_df.datetime)]\ntest_df['year'] = test_df['year'].map({2011:0, 2012:1})\ntest_df.head()","metadata":{"_uuid":"950fcc8ea72778c7f1963ee6106f4a68649974d4","execution":{"iopub.status.busy":"2022-07-14T06:33:38.298991Z","iopub.execute_input":"2022-07-14T06:33:38.299222Z","iopub.status.idle":"2022-07-14T06:33:38.398750Z","shell.execute_reply.started":"2022-07-14T06:33:38.299179Z","shell.execute_reply":"2022-07-14T06:33:38.397655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now can drop datetime column.\ndf.drop('datetime',axis=1,inplace=True)\ndf.head()","metadata":{"_uuid":"d04fa7dc6c3d03f70ba6b53abcb74b492f9c0739","execution":{"iopub.status.busy":"2022-07-14T06:33:38.400130Z","iopub.execute_input":"2022-07-14T06:33:38.400490Z","iopub.status.idle":"2022-07-14T06:33:38.443388Z","shell.execute_reply.started":"2022-07-14T06:33:38.400439Z","shell.execute_reply":"2022-07-14T06:33:38.442601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### NOW LETS HAVE A LOOK AT OUR NEW FEATURES.","metadata":{"_uuid":"33a931615968a6aee0fdf1381d968de1d9844720"}},{"cell_type":"code","source":"cor_mat= df[:].corr()\nmask = np.array(cor_mat)\nmask[np.tril_indices_from(mask)] = False\nfig=plt.gcf()\nfig.set_size_inches(30,12)\nsns.heatmap(data=cor_mat,mask=mask,square=True,annot=True,cbar=True)","metadata":{"_uuid":"cfaec3adda96ef23da53953b518d6af02d62865f","execution":{"iopub.status.busy":"2022-07-14T06:33:38.444833Z","iopub.execute_input":"2022-07-14T06:33:38.445074Z","iopub.status.idle":"2022-07-14T06:33:39.917770Z","shell.execute_reply.started":"2022-07-14T06:33:38.445033Z","shell.execute_reply":"2022-07-14T06:33:39.916786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['casual','registered'],axis=1,inplace=True)","metadata":{"_uuid":"525418035b60f42107117baceed4c217f9ac90ef","execution":{"iopub.status.busy":"2022-07-14T06:33:39.919023Z","iopub.execute_input":"2022-07-14T06:33:39.919319Z","iopub.status.idle":"2022-07-14T06:33:39.925253Z","shell.execute_reply.started":"2022-07-14T06:33:39.919259Z","shell.execute_reply":"2022-07-14T06:33:39.924372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"_uuid":"bbd11afa67d731e3749229861f79f0de5ea7729e","execution":{"iopub.status.busy":"2022-07-14T06:33:39.926736Z","iopub.execute_input":"2022-07-14T06:33:39.927067Z","iopub.status.idle":"2022-07-14T06:33:39.972557Z","shell.execute_reply.started":"2022-07-14T06:33:39.927003Z","shell.execute_reply":"2022-07-14T06:33:39.971399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### NOW LET SEE HOW COUNT VARIES WITH DIFFERENT FEATURES.","metadata":{"_uuid":"9553c175415db75c0d867b1d59bcc99d4a7c0d9b"}},{"cell_type":"markdown","source":"######  note that the highest demand is in hours from say 7-10 and the from 15-19. this is bcoz in most of the metroploitan cities this is the        peak office time and so more people would be renting bikes. this is just one of the plausible reason.","metadata":{"_uuid":"975c0eddd8c0330115685c15e3f96c2ff18d9425"}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\n## 1단계 : 전체 그래프 기본 설정\n# 그래프 사이 간격\nfig.tight_layout()\n\n# 전체 그래프 사이즈 관리\nfig.set_size_inches(10, 9)\n\n## 2단계 :  각 개별 그래프 입력\nsns.barplot(x = 'year', y = 'count', data = df, ax=ax[0,0])\nsns.barplot(x = 'month',y = 'count', data = df, ax=ax[0,1])\nsns.barplot(x = 'day', y = 'count', data = df, ax=ax[1,0])\nsns.barplot(x = 'hour', y = 'count', data = df, ax=ax[1,1])\n\n## 3단계 : 디테일 옵션\nax[0, 0].set_title(\"Rental Amounts by Year\")\nax[0, 1].set_title(\"Rental Amounts by month\")\nax[1, 0].set_title(\"Rental Amounts by day\")\nax[1, 1].set_title(\"Rental Amounts by hour\")\n\nax[0, 0].tick_params(axis = 'x', labelrotation=90)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:33:39.974070Z","iopub.execute_input":"2022-07-14T06:33:39.974412Z","iopub.status.idle":"2022-07-14T06:33:42.598789Z","shell.execute_reply.started":"2022-07-14T06:33:39.974343Z","shell.execute_reply":"2022-07-14T06:33:42.598186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- boxplot for rental amounts over season, weather, holiday, workingday","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\n## 1단계 : 전체 그래프 기본 설정\n# 그래프 사이 간격\nfig.tight_layout()\n\n# 전체 그래프 사이즈 관리\nfig.set_size_inches(10, 9)\n\n## 2단계 :  각 개별 그래프 입력\nsns.boxplot(x = 'season', y = 'count', data = train, ax=ax[0,0])\nsns.boxplot(x = 'weather',y = 'count', data = train, ax=ax[0,1])\nsns.boxplot(x = 'holiday', y = 'count', data = train, ax=ax[1,0])\nsns.boxplot(x = 'workingday', y = 'count', data = train, ax=ax[1,1])\n\n## 3단계 : 디테일 옵션\nax[0, 0].set_title(\"Box Plot on count across season\")\nax[0, 1].set_title(\"Rental Amounts by month\")\nax[1, 0].set_title(\"Rental Amounts by day\")\nax[1, 1].set_title(\"Rental Amounts by hour\")\n\nax[0, 1].tick_params(axis = 'x', labelrotation=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:33:42.599900Z","iopub.execute_input":"2022-07-14T06:33:42.600328Z","iopub.status.idle":"2022-07-14T06:33:43.418965Z","shell.execute_reply.started":"2022-07-14T06:33:42.600284Z","shell.execute_reply":"2022-07-14T06:33:43.417941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2, ncols =2)\n\nfig.tight_layout()\nfig.set_size_inches(10, 9)\n\n\nsns.regplot(x ='temp', y ='count', data = df, scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : '#ff4125'}, ax = ax[0, 0])\nsns.regplot(x ='atemp', y ='count', data = df, scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : '#ff4125'}, ax = ax[0, 1])\nsns.regplot(x ='humidity', y ='count', data = df, scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : '#ff4125'}, ax = ax[1, 0])\nsns.regplot(x ='windspeed', y ='count', data = df, scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : '#ff4125'}, ax = ax[1, 1])\nplt.show()","metadata":{"_uuid":"3cf03370927cf8de90934f29a1be02008e5fa8d4","execution":{"iopub.status.busy":"2022-07-14T06:33:43.424216Z","iopub.execute_input":"2022-07-14T06:33:43.426835Z","iopub.status.idle":"2022-07-14T06:33:48.038191Z","shell.execute_reply.started":"2022-07-14T06:33:43.426765Z","shell.execute_reply":"2022-07-14T06:33:48.037319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### note that this way this is hard to visualze. a better way is to convert the 'temp' variable into intervals or so called bins and then treat it like a discrete variable.","metadata":{"_uuid":"ce8e333b760088b04db3ebf63d3ca0ccf80a469a"}},{"cell_type":"code","source":"new_df=df.copy()\nnew_df.temp.describe()\nnew_df['temp_bin']=np.floor(new_df['temp'])//5\nnew_df['temp_bin'].unique()\n# now we can visualize as follows\nsns.factorplot(x=\"temp_bin\",y=\"count\",data=new_df,kind='bar')","metadata":{"_uuid":"611d77d4880383d051fd70efa7f1080b287b912e","execution":{"iopub.status.busy":"2022-07-14T06:33:48.039506Z","iopub.execute_input":"2022-07-14T06:33:48.039731Z","iopub.status.idle":"2022-07-14T06:33:48.819825Z","shell.execute_reply.started":"2022-07-14T06:33:48.039689Z","shell.execute_reply":"2022-07-14T06:33:48.818700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  now the demand is highest for bins 6 and 7 which is about tempearure  30-35(bin 6) and 35-40 (bin 7).","metadata":{"_uuid":"4b5a4df796e9a4c8baa462928a5dda605de80111"}},{"cell_type":"code","source":"# and similarly we can do for other continous variables and see how it effect the target variable.","metadata":{"_uuid":"b5d2e93aab1c35721fc6ca47876feac210ed195a","execution":{"iopub.status.busy":"2022-07-14T06:33:48.821639Z","iopub.execute_input":"2022-07-14T06:33:48.821987Z","iopub.status.idle":"2022-07-14T06:33:48.827891Z","shell.execute_reply.started":"2022-07-14T06:33:48.821931Z","shell.execute_reply":"2022-07-14T06:33:48.826889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### NOW THE DATA EXPLORATION ,ANALYSIS AND VISUALIZATION  AND PREPROCESSING HAS BEEN DONE AND NOW WE CAN MOVE TO MODELLING PART.","metadata":{"_uuid":"afa64373781b68e043c502d20514433ef8dfb2c6"}},{"cell_type":"code","source":"df.head()","metadata":{"_uuid":"1c0c1f7cac8789094276dc9eb07a86d62447398c","execution":{"iopub.status.busy":"2022-07-14T06:33:48.829645Z","iopub.execute_input":"2022-07-14T06:33:48.830438Z","iopub.status.idle":"2022-07-14T06:33:48.890595Z","shell.execute_reply.started":"2022-07-14T06:33:48.830363Z","shell.execute_reply":"2022-07-14T06:33:48.889876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns.to_series().groupby(df.dtypes).groups","metadata":{"_uuid":"24cd5ce3a44e8cd0095f2dcf98c25bf42d7e7da8","execution":{"iopub.status.busy":"2022-07-14T06:33:48.892212Z","iopub.execute_input":"2022-07-14T06:33:48.892646Z","iopub.status.idle":"2022-07-14T06:33:48.901864Z","shell.execute_reply.started":"2022-07-14T06:33:48.892505Z","shell.execute_reply":"2022-07-14T06:33:48.900940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_test,y_train,y_test=train_test_split(df.drop('count',axis=1),df['count'],test_size=0.25,random_state=42)","metadata":{"_uuid":"baf073b956710cb64c10422bcf33644fb6f71794","execution":{"iopub.status.busy":"2022-07-14T06:33:48.903132Z","iopub.execute_input":"2022-07-14T06:33:48.903712Z","iopub.status.idle":"2022-07-14T06:33:48.917462Z","shell.execute_reply.started":"2022-07-14T06:33:48.903665Z","shell.execute_reply":"2022-07-14T06:33:48.916353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=[RandomForestRegressor(),AdaBoostRegressor(),BaggingRegressor(),SVR(),KNeighborsRegressor()]\nmodel_names=['RandomForestRegressor','AdaBoostRegressor','BaggingRegressor','SVR','KNeighborsRegressor']\nrmsle=[]\nd={}\nfor model in range (len(models)):\n    clf=models[model]\n    clf.fit(x_train,y_train)\n    test_pred=clf.predict(x_test)\n    rmsle.append(np.sqrt(mean_squared_log_error(test_pred,y_test)))\nd={'Modelling Algo':model_names,'RMSLE':rmsle}   \nd\n    ","metadata":{"_uuid":"1dae4e766e42560879e3b5d471d74d42880312a7","execution":{"iopub.status.busy":"2022-07-14T06:33:48.918992Z","iopub.execute_input":"2022-07-14T06:33:48.919547Z","iopub.status.idle":"2022-07-14T06:33:57.310927Z","shell.execute_reply.started":"2022-07-14T06:33:48.919502Z","shell.execute_reply":"2022-07-14T06:33:57.310085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmsle_frame=pd.DataFrame(d)\nrmsle_frame","metadata":{"_uuid":"6dabf1935c3d529e977407b8f424097057d71ab2","execution":{"iopub.status.busy":"2022-07-14T06:33:57.311919Z","iopub.execute_input":"2022-07-14T06:33:57.312348Z","iopub.status.idle":"2022-07-14T06:33:57.326902Z","shell.execute_reply.started":"2022-07-14T06:33:57.312277Z","shell.execute_reply":"2022-07-14T06:33:57.326093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.factorplot(y='Modelling Algo',x='RMSLE',data=rmsle_frame,kind='bar',size=5,aspect=2)","metadata":{"_uuid":"fb4017692bd3ae58846d968a3356cf691e1b7c10","execution":{"iopub.status.busy":"2022-07-14T06:33:57.328193Z","iopub.execute_input":"2022-07-14T06:33:57.328504Z","iopub.status.idle":"2022-07-14T06:33:57.650378Z","shell.execute_reply.started":"2022-07-14T06:33:57.328453Z","shell.execute_reply":"2022-07-14T06:33:57.649015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.factorplot(x='Modelling Algo',y='RMSLE',data=rmsle_frame,kind='point',size=5,aspect=2)","metadata":{"_uuid":"b1e29191394aad3b67a3de0516582ed459735245","execution":{"iopub.status.busy":"2022-07-14T06:33:57.652439Z","iopub.execute_input":"2022-07-14T06:33:57.652997Z","iopub.status.idle":"2022-07-14T06:33:57.982297Z","shell.execute_reply.started":"2022-07-14T06:33:57.652742Z","shell.execute_reply":"2022-07-14T06:33:57.981102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  NOTE THAT THERE ARE OTHER MODELLING ALGOS LIKE LINEAR REGRESSION ,RIDGE AND RIDGECV BUT THE PROBLEM IS THAT THOSE MODELS ARE PREDICTING NEGATIVE VALUES FOR THE COUNT TARGET WHICH IS NOT POSSIBLE.                                                                                                                                                                                                                                                                                                                  NOW I DONT KNOW WHAT TO DO IN THOSE CASES :::) !!!!!!!!!!!!!!!","metadata":{"_uuid":"dc69054aef137b48a8e7f396822866049e946cf1"}},{"cell_type":"markdown","source":"######  NOW LET'S TUNE A BIT...","metadata":{"_uuid":"08fe66f89d40e95e70593ddd7541de96039c57af"}},{"cell_type":"code","source":"#for random forest regresion.\nno_of_test=[500]\nparams_dict={'n_estimators':no_of_test,'n_jobs':[-1],'max_features':[\"auto\",'sqrt','log2']}\nclf_rf=GridSearchCV(estimator=RandomForestRegressor(),param_grid=params_dict,scoring='neg_mean_squared_log_error')\nclf_rf.fit(x_train,y_train)\npred=clf_rf.predict(x_test)\nprint((np.sqrt(mean_squared_log_error(pred,y_test))))","metadata":{"_uuid":"59e510a4b96c1c314e23d8a1a193d97fb09ef99e","execution":{"iopub.status.busy":"2022-07-14T06:33:57.983995Z","iopub.execute_input":"2022-07-14T06:33:57.984619Z","iopub.status.idle":"2022-07-14T06:35:20.452097Z","shell.execute_reply.started":"2022-07-14T06:33:57.984534Z","shell.execute_reply":"2022-07-14T06:35:20.450959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_rf.best_params_","metadata":{"_uuid":"519e7fc88dbb4872136fd3eb8eca61e2a75941a2","execution":{"iopub.status.busy":"2022-07-14T06:35:20.453543Z","iopub.execute_input":"2022-07-14T06:35:20.454085Z","iopub.status.idle":"2022-07-14T06:35:20.460786Z","shell.execute_reply.started":"2022-07-14T06:35:20.453972Z","shell.execute_reply":"2022-07-14T06:35:20.459969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for KNN\nn_neighbors=[]\nfor i in range (0,50,5):\n    if(i!=0):\n        n_neighbors.append(i)\nparams_dict={'n_neighbors':n_neighbors,'n_jobs':[-1]}\nclf_knn=GridSearchCV(estimator=KNeighborsRegressor(),param_grid=params_dict,scoring='neg_mean_squared_log_error')\nclf_knn.fit(x_train,y_train)\npred=clf_knn.predict(x_test)\nprint((np.sqrt(mean_squared_log_error(pred,y_test))))\n","metadata":{"_uuid":"a963bd9247ce535dbc357c0a43c142095fbb1994","execution":{"iopub.status.busy":"2022-07-14T06:35:20.462589Z","iopub.execute_input":"2022-07-14T06:35:20.463385Z","iopub.status.idle":"2022-07-14T06:35:28.857023Z","shell.execute_reply.started":"2022-07-14T06:35:20.463286Z","shell.execute_reply":"2022-07-14T06:35:28.856057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_knn.best_params_","metadata":{"_uuid":"e64bf3c33e01f5d4d3fa037e5fd95957506af3cf","execution":{"iopub.status.busy":"2022-07-14T06:35:28.858454Z","iopub.execute_input":"2022-07-14T06:35:28.858720Z","iopub.status.idle":"2022-07-14T06:35:28.863933Z","shell.execute_reply.started":"2022-07-14T06:35:28.858667Z","shell.execute_reply":"2022-07-14T06:35:28.863122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:28.864839Z","iopub.execute_input":"2022-07-14T06:35:28.865178Z","iopub.status.idle":"2022-07-14T06:35:28.948345Z","shell.execute_reply.started":"2022-07-14T06:35:28.865140Z","shell.execute_reply":"2022-07-14T06:35:28.947111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:28.949731Z","iopub.execute_input":"2022-07-14T06:35:28.949991Z","iopub.status.idle":"2022-07-14T06:35:29.033845Z","shell.execute_reply.started":"2022-07-14T06:35:28.949940Z","shell.execute_reply":"2022-07-14T06:35:29.032915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#XGBoost \n\n# fit a final xgboost model on the housing dataset and make a prediction\nfrom numpy import asarray\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\n\ndef cv_rmse(model, n_folds=5):\n    cv = KFold(n_splits=n_folds, random_state=42, shuffle=True)\n    rmse_list = np.sqrt(-cross_val_score(model, x_train, y_train, scoring='neg_mean_squared_error', cv=cv))\n    print('CV RMSE value list:', np.round(rmse_list, 4))\n    print('CV RMSE mean value:', np.round(np.mean(rmse_list), 4))\n    return (rmse_list)\n\nn_folds = 5\nrmse_scores = {}\nlr_model = LinearRegression()\nlgb_model = LGBMRegressor()\nxgb_model = XGBRegressor()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:29.035110Z","iopub.execute_input":"2022-07-14T06:35:29.035425Z","iopub.status.idle":"2022-07-14T06:35:29.042033Z","shell.execute_reply.started":"2022-07-14T06:35:29.035369Z","shell.execute_reply":"2022-07-14T06:35:29.041327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(lgb_model, n_folds)\nprint(\"linear regression - mean: {:.4f} (std: {:.4f})\".format(score.mean(), score.std()))\nrmse_scores['linear regression'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:29.043136Z","iopub.execute_input":"2022-07-14T06:35:29.043604Z","iopub.status.idle":"2022-07-14T06:35:29.685289Z","shell.execute_reply.started":"2022-07-14T06:35:29.043557Z","shell.execute_reply":"2022-07-14T06:35:29.684574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(xgb_model, n_folds)\nprint(\"linear regression - mean: {:.4f} (std: {:.4f})\".format(score.mean(), score.std()))\nrmse_scores['linear regression'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:29.686617Z","iopub.execute_input":"2022-07-14T06:35:29.686921Z","iopub.status.idle":"2022-07-14T06:35:31.337212Z","shell.execute_reply.started":"2022-07-14T06:35:29.686856Z","shell.execute_reply":"2022-07-14T06:35:31.336442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######  NOW RANDOM FORETS REGRESSOR GIVES THE LEAST RMSLE. HENCE WE USE IT TO MAKE PREDICTIONS ON KAGGLE.","metadata":{"_uuid":"69d75cb3de8df42e9e67d62d1dd99c597b832e3c"}},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:35:31.338450Z","iopub.execute_input":"2022-07-14T06:35:31.338676Z","iopub.status.idle":"2022-07-14T06:35:31.419021Z","shell.execute_reply.started":"2022-07-14T06:35:31.338628Z","shell.execute_reply":"2022-07-14T06:35:31.418153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_predict\n\n# X = all_df.iloc[:len(y), :]\n# X_test = all_df.iloc[len(y):, :]\n# X.shape, y.shape, X_test.shape\n\nlr_model_fit = lgb_model.fit(x_train, y_train)\nfinal_preds = np.floor(np.expm1(lr_model_fit.predict(x_test)))\nprint(final_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:36:24.217010Z","iopub.execute_input":"2022-07-14T06:36:24.217353Z","iopub.status.idle":"2022-07-14T06:36:24.400764Z","shell.execute_reply.started":"2022-07-14T06:36:24.217259Z","shell.execute_reply":"2022-07-14T06:36:24.399910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=clf_rf.predict(test_df.drop('datetime',axis=1))\nd={'datetime':test['datetime'],'count':pred}\nsubmission=pd.DataFrame(d)\nsubmission.to_csv('submission.csv',index=False) # saving to a csv file for predictions on kaggle.\n","metadata":{"_uuid":"e73d62fca06d7300383d441cec9e45635805b951","execution":{"iopub.status.busy":"2022-07-14T06:51:37.125437Z","iopub.execute_input":"2022-07-14T06:51:37.125858Z","iopub.status.idle":"2022-07-14T06:51:37.866664Z","shell.execute_reply.started":"2022-07-14T06:51:37.125795Z","shell.execute_reply":"2022-07-14T06:51:37.865790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"7dfc73617eced31fb9a7eb2dac50cd8bf195e000"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}