{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics librarabsies installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nfrom sklearn.linear_model import LinearRegression,LogisticRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error,r2_score,accuracy_score,mean_absolute_error\nfrom sklearn.model_selection import GridSearchCV\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.preprocessing import LabelEncoder\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfrom sklearn.preprocessing import QuantileTransformer\nimport matplotlib.pyplot as plt\nimport warnings\nimport random\nfrom sklearn.model_selection import train_test_split\nwarnings.filterwarnings('ignore')\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n        \nrandom.seed(42)\nnp.random.seed(42)\nimport matplotlib as mpl\nmpl.style.use('seaborn')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-03T08:02:40.819413Z","iopub.execute_input":"2022-09-03T08:02:40.819715Z","iopub.status.idle":"2022-09-03T08:02:40.833854Z","shell.execute_reply.started":"2022-09-03T08:02:40.819665Z","shell.execute_reply":"2022-09-03T08:02:40.832916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://media.istockphoto.com/photos/portrait-of-a-funny-dog-jack-russell-terrier-in-sunglasses-behind-the-picture-id1180219878?k=20&m=1180219878&s=612x612&w=0&h=NB14R3M-_7K7q6KCa6OqwdHaKjxk4MTStFHI1vI9CZI=)","metadata":{}},{"cell_type":"markdown","source":"# **This is a Beginner's note**\n\n1. Why should you read this?\n\n -> This notebook is made based on what I have learnt, and I am writing very concisely. If you want to know broadly about any topic you can find a website link attached with it.\n\nI'm trying my best >(-.-)<\n","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/price-prediction-multiple-linear-regression/scrap price.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:40.909646Z","iopub.execute_input":"2022-09-03T08:02:40.911432Z","iopub.status.idle":"2022-09-03T08:02:40.941616Z","shell.execute_reply.started":"2022-09-03T08:02:40.911403Z","shell.execute_reply":"2022-09-03T08:02:40.940711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Basic EDA**\n- Data Shape\n- Basic Visualization\n- Check Null value\n1. [know more](https://www.analyticsvidhya.com/blog/2021/05/exploratory-data-analysis-eda-a-step-by-step-guide/)\n2. [more](https://towardsdatascience.com/exploratory-data-analysis-8fc1cb20fd15)\n3. [and more](https://www.kaggle.com/code/dejavu23/house-prices-eda-to-ml-beginner)","metadata":{}},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:40.970414Z","iopub.execute_input":"2022-09-03T08:02:40.970974Z","iopub.status.idle":"2022-09-03T08:02:40.977955Z","shell.execute_reply.started":"2022-09-03T08:02:40.970937Z","shell.execute_reply":"2022-09-03T08:02:40.976644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.051705Z","iopub.execute_input":"2022-09-03T08:02:41.052246Z","iopub.status.idle":"2022-09-03T08:02:41.067407Z","shell.execute_reply.started":"2022-09-03T08:02:41.052210Z","shell.execute_reply":"2022-09-03T08:02:41.066345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This data contains 26 columns and 205 rows. \n\nIn 26 Columns, There are 10 columns which are objects, 8 columns are float, and rest of them are integer.","metadata":{}},{"cell_type":"code","source":"data = data.drop(columns=['ID','symboling'])","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.075592Z","iopub.execute_input":"2022-09-03T08:02:41.076398Z","iopub.status.idle":"2022-09-03T08:02:41.082912Z","shell.execute_reply.started":"2022-09-03T08:02:41.076372Z","shell.execute_reply":"2022-09-03T08:02:41.081974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ID and symboling colums are just useless, so I just dropped them","metadata":{}},{"cell_type":"code","source":"data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.089918Z","iopub.execute_input":"2022-09-03T08:02:41.090820Z","iopub.status.idle":"2022-09-03T08:02:41.099261Z","shell.execute_reply.started":"2022-09-03T08:02:41.090775Z","shell.execute_reply":"2022-09-03T08:02:41.098332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Simple checking if there any null value.. But Data is so Clean**","metadata":{}},{"cell_type":"markdown","source":"# ****Data wrangling**** : \n\nClean Data and prepare for train","metadata":{}},{"cell_type":"markdown","source":"Looking for unique values in non numeric columns, thus I can analyze or change or clear data.","metadata":{}},{"cell_type":"code","source":"for i in data.columns:\n    if data[i].dtype == 'object':\n        print(f'{i} has {data[i].nunique()} values ->{data[i].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.200235Z","iopub.execute_input":"2022-09-03T08:02:41.200781Z","iopub.status.idle":"2022-09-03T08:02:41.211732Z","shell.execute_reply.started":"2022-09-03T08:02:41.200745Z","shell.execute_reply":"2022-09-03T08:02:41.210579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 147 unique car model names. That is a huge unique name. Instead of a model name I can use only brand name. That will be more obvious.","metadata":{}},{"cell_type":"code","source":"for i in range(len(data['name'])):\n    data['name'][i] = data['name'][i].split()[0]","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.258784Z","iopub.execute_input":"2022-09-03T08:02:41.259465Z","iopub.status.idle":"2022-09-03T08:02:41.316819Z","shell.execute_reply.started":"2022-09-03T08:02:41.259438Z","shell.execute_reply":"2022-09-03T08:02:41.315829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(data['name'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.318760Z","iopub.execute_input":"2022-09-03T08:02:41.319188Z","iopub.status.idle":"2022-09-03T08:02:41.327841Z","shell.execute_reply.started":"2022-09-03T08:02:41.319150Z","shell.execute_reply":"2022-09-03T08:02:41.326752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After Cleaning model names, Only left 28 brand names. \n","metadata":{}},{"cell_type":"markdown","source":"*Let's do some Bar visualization which category is much popular*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,35))\nj=1\nfor i in data.columns:\n    if data[i].dtype == 'object':\n        plt.subplot(4,3,j)\n        x = pd.DataFrame(data[i].value_counts())\n        plt.title(f'{i} most popular category is {x.index[0]}')\n        data[i].value_counts().plot.bar()\n        j += 1\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:41.329409Z","iopub.execute_input":"2022-09-03T08:02:41.330054Z","iopub.status.idle":"2022-09-03T08:02:42.743401Z","shell.execute_reply.started":"2022-09-03T08:02:41.330018Z","shell.execute_reply":"2022-09-03T08:02:42.742254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in data.columns:\n    if data[i].dtype == 'object':\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:42.750002Z","iopub.execute_input":"2022-09-03T08:02:42.752743Z","iopub.status.idle":"2022-09-03T08:02:42.763245Z","shell.execute_reply.started":"2022-09-03T08:02:42.752667Z","shell.execute_reply":"2022-09-03T08:02:42.762175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,35))\nj = 1\nfor i in data.columns:\n    if data[i].dtype == 'object':\n        plt.subplot(4,4,j)\n        x = data.groupby([i])['price'].mean().nlargest()\n        x.plot.bar(colormap='Paired')\n        plt.xticks(rotation=45)\n        j += 1\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:42.767058Z","iopub.execute_input":"2022-09-03T08:02:42.768701Z","iopub.status.idle":"2022-09-03T08:02:44.217552Z","shell.execute_reply.started":"2022-09-03T08:02:42.768653Z","shell.execute_reply":"2022-09-03T08:02:44.216591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **A story rises from this graphs is**\n\n\"**Jaguar**\" car has highest price among all of cars, \"**Diesel**\" is costly, \"**turbo**\" engine, \"**4 door**\" car is little bit expensive, in carbody category '**hardtop**', '**convertible**' are almost same price, '**rwd**' is expensive, those car is much expensive which has engine in '**front**', '**eight**' cylindered car and '**mpfi**' fuel system is preferably high price.","metadata":{}},{"cell_type":"code","source":"numerical_val =[]\nfor i in data.columns:\n    if data[i].dtype != 'object':\n        numerical_val.append(i)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.219206Z","iopub.execute_input":"2022-09-03T08:02:44.220175Z","iopub.status.idle":"2022-09-03T08:02:44.226194Z","shell.execute_reply.started":"2022-09-03T08:02:44.220137Z","shell.execute_reply":"2022-09-03T08:02:44.225084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save Numerical data for further use. ","metadata":{}},{"cell_type":"markdown","source":"# ***Let's change object variable to numerical variable*** \n\nBecause No ML model can be trained using object or string variables. So 1st convert string variable to numeric value.. There are several techniques to convert string to numeric value :-\n- [Label Encoder](https://towardsdatascience.com/categorical-encoding-using-label-encoding-and-one-hot-encoder-911ef77fb5bd)\n- [One hot encoder](https://machinelearningmastery.com/how-to-one-hot-encode-sequence-data-in-python/)\n- [Manually](https://pandas.pydata.org/docs/reference/api/pandas.Series.map.html) \n\n> I choose manually(Mapping) and LabelEncoder  for this dataset\n","metadata":{}},{"cell_type":"markdown","source":"1st of all Mapping","metadata":{}},{"cell_type":"code","source":"data['doornumbers'] = data['doornumbers'].map({'two':2,'four':4})","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.229346Z","iopub.execute_input":"2022-09-03T08:02:44.230097Z","iopub.status.idle":"2022-09-03T08:02:44.237881Z","shell.execute_reply.started":"2022-09-03T08:02:44.230060Z","shell.execute_reply":"2022-09-03T08:02:44.236786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['cylindernumber'] = data['cylindernumber'].map({'four':4, 'six':6, 'five':5, 'three':3, 'twelve':12, 'two':2, 'eight':8})","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.239500Z","iopub.execute_input":"2022-09-03T08:02:44.239894Z","iopub.status.idle":"2022-09-03T08:02:44.248202Z","shell.execute_reply.started":"2022-09-03T08:02:44.239861Z","shell.execute_reply":"2022-09-03T08:02:44.247130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.249777Z","iopub.execute_input":"2022-09-03T08:02:44.250125Z","iopub.status.idle":"2022-09-03T08:02:44.262769Z","shell.execute_reply.started":"2022-09-03T08:02:44.250091Z","shell.execute_reply":"2022-09-03T08:02:44.261846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Using Label Encoder**\nLabelEncoder takes columns as input and converts strings as alphabetical order. \n\nsuppose some input values are: a,aab,b,c,bd\n\nLabelEncoder will return as: 1,2,3,5,4\n","metadata":{}},{"cell_type":"code","source":"\nfor i in data.columns:\n    if data[i].dtype == 'object':\n        data[i] = LabelEncoder().fit_transform(data[i])","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.268843Z","iopub.execute_input":"2022-09-03T08:02:44.269097Z","iopub.status.idle":"2022-09-03T08:02:44.279125Z","shell.execute_reply.started":"2022-09-03T08:02:44.269073Z","shell.execute_reply":"2022-09-03T08:02:44.278236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.280444Z","iopub.execute_input":"2022-09-03T08:02:44.281058Z","iopub.status.idle":"2022-09-03T08:02:44.292419Z","shell.execute_reply.started":"2022-09-03T08:02:44.281021Z","shell.execute_reply":"2022-09-03T08:02:44.291364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Some Visualizations**\n - [Histogram](https://towardsdatascience.com/histograms-why-how-431a5cfbfcd5)\n - [Distplot](https://stackoverflow.com/questions/56707800/what-are-the-arguments-of-seaborns-distplot-used-for)\n - [Boxplot](https://www.storytellingwithdata.com/blog/what-is-a-boxplot)","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,15))\nfor i in enumerate(numerical_val):\n    plt.subplot(4,4,i[0]+1)\n    plt.hist(data[i[1]])\n    plt.xlabel(i[1])\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:44.294798Z","iopub.execute_input":"2022-09-03T08:02:44.295864Z","iopub.status.idle":"2022-09-03T08:02:45.879028Z","shell.execute_reply.started":"2022-09-03T08:02:44.295836Z","shell.execute_reply":"2022-09-03T08:02:45.878028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,15))\nfor i in enumerate(numerical_val):\n    plt.subplot(4,4,i[0]+1)\n    plt.boxplot(data[i[1]])\n    plt.xlabel(i[1])\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:45.880538Z","iopub.execute_input":"2022-09-03T08:02:45.881165Z","iopub.status.idle":"2022-09-03T08:02:47.240586Z","shell.execute_reply.started":"2022-09-03T08:02:45.881124Z","shell.execute_reply":"2022-09-03T08:02:47.239593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Outliers**\n\nOutliers is uncommon data in a particular column. [know more](https://machinelearningmastery.com/how-to-use-statistics-to-identify-outliers-in-data/)\n\nRoadblock for train a model better. Generally, Outliers should be removed carefully or transform/ Scale data.\n\nBut as This dataset is tiny..I won't drop any data but transform. \n","metadata":{}},{"cell_type":"markdown","source":"# **Next target is Transform Data** \n\nBecause the Numerical continuous data is not well distributed as we can see from upper hist diagrams. And well distributed data should be bell curved. Scattered data can be an asset for Worst model or a model won't learn properly.\n\n*When should I use a transformer?*\n\n     If my data is not finely distributed. Like Carlength, Carwidth, Carheight, Cubweight. These Columns are not in well shaped \n     \n     If my data has more than one peak point. boreratio,peakrpm, citympg these columns have multiple peak values \n     \n     If my data is skewed or the column has a tail. wheelbase, enginesize, horsepower, price, these columns are right skewed and have tails. \n     \n\n*Kind of Transformers*  \n- Log transform : when data is skewed. \n- Scaler Transform: when data has large value and data values are spreaded\n- MinMax Transform: when I want to shrink the data range to [0,1] .\n- Quantile Transform :  When data is uniformly distributed, multiple peak value, (In fact every kind of data).\n\n[So on](https://machinelearningmastery.com/power-transforms-with-scikit-learn/)\n\n\n**Importance**\n - Helps in multiple Linear regression\n - Helpful for to remove Multicollinearity(Highly correlated)\n - Helpful for remove Homoscedasticity(correlation is not linear)\n","metadata":{}},{"cell_type":"code","source":"\n\nplt.figure(figsize = (20,20))\nscaler = QuantileTransformer(output_distribution = 'normal')\n\nfor i in enumerate(numerical_val):\n    plt.subplot(6,6,i[0]+1)\n    data[i[1]] = scaler.fit_transform(data[[i[1]]])\n    plt.hist(data[i[1]])\n    plt.xlabel(i[1])\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:47.242230Z","iopub.execute_input":"2022-09-03T08:02:47.242629Z","iopub.status.idle":"2022-09-03T08:02:48.569119Z","shell.execute_reply.started":"2022-09-03T08:02:47.242590Z","shell.execute_reply":"2022-09-03T08:02:48.568171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,15))\nfor i in enumerate(numerical_val):\n    plt.subplot(4,4,i[0]+1)\n    plt.boxplot(data[i[1]])\n    plt.xlabel(i[1])\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:48.570368Z","iopub.execute_input":"2022-09-03T08:02:48.570974Z","iopub.status.idle":"2022-09-03T08:02:49.527631Z","shell.execute_reply.started":"2022-09-03T08:02:48.570933Z","shell.execute_reply":"2022-09-03T08:02:49.526561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nplt.figure(figsize = (20,20))\n\nfor i in enumerate(data.columns):\n    plt.subplot(6,6,i[0]+1)\n    plt.hist(data[i[1]])\n    plt.xlabel(i[1])\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:49.529038Z","iopub.execute_input":"2022-09-03T08:02:49.530049Z","iopub.status.idle":"2022-09-03T08:02:51.997762Z","shell.execute_reply.started":"2022-09-03T08:02:49.530012Z","shell.execute_reply":"2022-09-03T08:02:51.996628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Relation between every column with each other**  \nTo see correlation between 2 columns generally use .corr() pre-built function, which is basically use pearson formula.\\\nAnd for graphical representation I use a heat map. \n\nThe .corr() function returns a value between -1 to 1. \n\nif value is -1 then 2 columns are highly correlated with each other in negative way,\n0 means no relation \n1 means again highly related  in a positive way \n\n\nGenerally  corr> .7 or -.7  and 0 correlation should be removed for best result or best prediction.\n\n*One example* \n- if someone is a 'chain smoker' for a long time definitely the person must have 'lungs issues'. One is co-related to another. So I can just use chain smoker column or lungs issues column.\n\n[More](https://medium.com/analytics-vidhya/correlation-and-machine-learning-fee0ffc5faac)\n","metadata":{}},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:51.999411Z","iopub.execute_input":"2022-09-03T08:02:51.999810Z","iopub.status.idle":"2022-09-03T08:02:52.005012Z","shell.execute_reply.started":"2022-09-03T08:02:51.999769Z","shell.execute_reply":"2022-09-03T08:02:52.003738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncor = data.corr()\nplt.figure(figsize = (20,5))\nsns.heatmap(cor,fmt='.1f',annot=True,mask = np.triu(cor))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:52.006710Z","iopub.execute_input":"2022-09-03T08:02:52.007074Z","iopub.status.idle":"2022-09-03T08:02:53.501610Z","shell.execute_reply.started":"2022-09-03T08:02:52.007034Z","shell.execute_reply":"2022-09-03T08:02:53.500741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfor i in data.columns[:-1]:\n    if data[i].dtype == 'float':\n        plt.scatter(data[i],data['price'])\n        plt.title(f\"{i} vs price {data[i].corr(data['price'])}\")\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:53.502926Z","iopub.execute_input":"2022-09-03T08:02:53.503944Z","iopub.status.idle":"2022-09-03T08:02:55.894539Z","shell.execute_reply.started":"2022-09-03T08:02:53.503904Z","shell.execute_reply":"2022-09-03T08:02:55.893620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plots show relation between 'Price' with others. I can remove columns who have correlation >.7 . But I will select Features based on Feature importance.\n\n**Why should you choose some features instead of all?** \\\nBecause..\n- I choose important columns in the dataset which will enrich my training process.\n- To reduce memory size\n- To reduce time on training a model\n\n**How can I choose features?**\n- Choose according to feature importance in a model. \n- use feature selection library(wrapper method -> time consuming :) )\n\n* Based on Feature importance\n - If I use Linear Regression model, feature importance will be coef_ values\n - Otherwise call the feature_importance function .. Those will return continuous values. Higher value indicates higher importance.\n \n* Wrapper Method\n- There are huge methods in xtend feature selection collection. These are time consuming\n\n\n\n\nI won't drop any column. You may help your self [by](https://machinelearningmastery.com/calculate-feature-importance-with-python/)\n","metadata":{}},{"cell_type":"code","source":"features = data.drop(columns = ['price'])\nlabel = data[['price']]","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:55.895744Z","iopub.execute_input":"2022-09-03T08:02:55.896122Z","iopub.status.idle":"2022-09-03T08:02:55.903027Z","shell.execute_reply.started":"2022-09-03T08:02:55.896093Z","shell.execute_reply":"2022-09-03T08:02:55.901883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# **Time to train model**\n\nBut before that should split data between train and test. Train data is 80% of the entire data and the rest of them for test data.\n\nTrain data is for training models, a model can learn tricks and tricks from data. \n\nTest data is for Examination how well a model learns in the training phase.\n\n[More](https://machinelearningmastery.com/train-test-split-for-evaluating-machine-learning-algorithms/)\n","metadata":{}},{"cell_type":"code","source":"\n# features = data[columns]\n# label = data['price']\n\nxtrain,xtest,ytrain,ytest = train_test_split(features,label,train_size=.8)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:55.904830Z","iopub.execute_input":"2022-09-03T08:02:55.905533Z","iopub.status.idle":"2022-09-03T08:02:55.914055Z","shell.execute_reply.started":"2022-09-03T08:02:55.905498Z","shell.execute_reply":"2022-09-03T08:02:55.913048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I don't know which model will be the best for this data. \n\nSo I use several models and train them. [here is details](https://www.projectpro.io/article/common-machine-learning-algorithms-for-beginners/202)\n\n\n\nAnd For evaluation I choose [mean_absolute_error](https://datagy.io/mae-python/). And as I did transform data before so I need to reverse that process thus I can get original value for evaluation.\n","metadata":{}},{"cell_type":"code","source":"lir = LinearRegression()\ndt = DecisionTreeRegressor()\nrf = RandomForestRegressor()\ngb = GradientBoostingRegressor()\nlgb = LGBMRegressor()\ncat = CatBoostRegressor()\nxgb = XGBRegressor()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:55.915722Z","iopub.execute_input":"2022-09-03T08:02:55.916203Z","iopub.status.idle":"2022-09-03T08:02:55.925143Z","shell.execute_reply.started":"2022-09-03T08:02:55.916163Z","shell.execute_reply":"2022-09-03T08:02:55.923962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Can I train the model better or can I train mulptiple time and find best one?**\n\nYES. Using GridSearchCV I can do Hyperparameter tuning.\n\n*Hyperparameter Tuning* \n\nHyperparameters are those parameters I can change parameters of a model's value in so many ways to find the best score. This changing process is called Tuning and overall Hyperparameter Tuning. My Motivation is to find a better score.\n\n\n[For more about GridSearchCV](https://towardsdatascience.com/gridsearchcv-for-beginners-db48a90114ee)\n","metadata":{}},{"cell_type":"code","source":"params= {\n    'n_jobs':[-1,1],\n}\n\nlr_grid = GridSearchCV(lir,param_grid = params,cv=5,scoring = 'r2')\nlr_grid.fit(features,label)\nprint(lr_grid.best_params_)\nprint(lr_grid.best_score_)\nlr_pred=scaler.inverse_transform(lr_grid.predict(xtest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:55.927097Z","iopub.execute_input":"2022-09-03T08:02:55.927619Z","iopub.status.idle":"2022-09-03T08:02:56.004279Z","shell.execute_reply.started":"2022-09-03T08:02:55.927584Z","shell.execute_reply":"2022-09-03T08:02:56.003217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params= {\n    'criterion':['squared_error', 'friedman_mse', 'absolute_error', 'poisson'],\n    'max_depth':[None,5,10,15,20],\n}\n\ndt_grid = GridSearchCV(dt,param_grid = params,cv=5,scoring = 'r2')\ndt_grid.fit(features,label)\nprint(dt_grid.best_params_)\nprint(dt_grid.best_score_)\ndt_pred=scaler.inverse_transform(dt_grid.predict(xtest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:56.005889Z","iopub.execute_input":"2022-09-03T08:02:56.006229Z","iopub.status.idle":"2022-09-03T08:02:56.704822Z","shell.execute_reply.started":"2022-09-03T08:02:56.006196Z","shell.execute_reply":"2022-09-03T08:02:56.703722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n 'max_depth': [10, 20, 30, None],\n 'max_features': ['auto', 'sqrt'],\n 'n_estimators': [200, 400]}\n\nrf_grid = GridSearchCV(rf,param_grid = params,cv=5,scoring = 'r2',n_jobs = -1)\nrf_grid.fit(features,label)\nprint(rf_grid.best_params_)\nprint(rf_grid.best_score_)\nrf_pred=scaler.inverse_transform(rf_grid.predict(xtest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:02:56.706113Z","iopub.execute_input":"2022-09-03T08:02:56.706570Z","iopub.status.idle":"2022-09-03T08:03:29.034353Z","shell.execute_reply.started":"2022-09-03T08:02:56.706531Z","shell.execute_reply":"2022-09-03T08:03:29.032408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n 'n_estimators':[500,1000,1500],\n 'learning_rate':[.001,0.01,.1],\n'max_depth': [10, 20, 30, None],\n 'max_features': ['auto', 'sqrt'],\n\n}\n\ngb_grid = GridSearchCV(gb,param_grid = params,cv=5,scoring = 'r2',n_jobs = -1)\ngb_grid.fit(features,label)\nprint(gb_grid.best_params_)\nprint(gb_grid.best_score_)\ngb_pred=scaler.inverse_transform(gb_grid.predict(xtest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:03:29.036201Z","iopub.execute_input":"2022-09-03T08:03:29.036693Z","iopub.status.idle":"2022-09-03T08:07:41.831293Z","shell.execute_reply.started":"2022-09-03T08:03:29.036633Z","shell.execute_reply":"2022-09-03T08:07:41.830092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    'iterations':[1000,1500,2000],\n    'learning_rate':[.1,.01],\n    'task_type':['GPU']\n}\n\ncat_grid = GridSearchCV(cat,param_grid = params,cv=3,scoring = 'r2',n_jobs = -1)\ncat_grid.fit(features,label,silent=True)\nprint(cat_grid.best_params_)\nprint(cat_grid.best_score_)\ncat_pred=scaler.inverse_transform(cat_grid.predict(xtest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:07:41.832852Z","iopub.execute_input":"2022-09-03T08:07:41.834237Z","iopub.status.idle":"2022-09-03T08:16:07.706609Z","shell.execute_reply.started":"2022-09-03T08:07:41.834196Z","shell.execute_reply":"2022-09-03T08:16:07.705657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nUsing Bar graph I can find which model gave less error.. less error means better model.","metadata":{}},{"cell_type":"code","source":"actual = scaler.inverse_transform(np.array(ytest).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:16:07.712360Z","iopub.execute_input":"2022-09-03T08:16:07.713154Z","iopub.status.idle":"2022-09-03T08:16:07.720031Z","shell.execute_reply.started":"2022-09-03T08:16:07.713115Z","shell.execute_reply":"2022-09-03T08:16:07.719047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"li_result = mean_absolute_error(actual,lr_pred)\ndt_result = mean_absolute_error(actual,dt_pred)\nrf_result = mean_absolute_error(actual,rf_pred)\ncat_result = mean_absolute_error(actual,cat_pred)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:16:07.722713Z","iopub.execute_input":"2022-09-03T08:16:07.724253Z","iopub.status.idle":"2022-09-03T08:16:07.732698Z","shell.execute_reply.started":"2022-09-03T08:16:07.724215Z","shell.execute_reply":"2022-09-03T08:16:07.731819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = [\n    (li_result,'linear'),\n    (dt_result,'tree'),\n    (rf_result,'forest'),\n    (cat_result,'cat'),\n]\nresult.sort()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:16:07.735396Z","iopub.execute_input":"2022-09-03T08:16:07.737090Z","iopub.status.idle":"2022-09-03T08:16:07.742508Z","shell.execute_reply.started":"2022-09-03T08:16:07.737057Z","shell.execute_reply":"2022-09-03T08:16:07.741726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame(result)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:16:07.745014Z","iopub.execute_input":"2022-09-03T08:16:07.746648Z","iopub.status.idle":"2022-09-03T08:16:07.752637Z","shell.execute_reply.started":"2022-09-03T08:16:07.746608Z","shell.execute_reply":"2022-09-03T08:16:07.751844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (10, 5))\n\n# creating the bar plot\nplt.bar(result[1], result[0], color ='maroon',\n        width = 0.4)\nplt.xticks(rotation = 45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T08:16:07.755291Z","iopub.execute_input":"2022-09-03T08:16:07.756924Z","iopub.status.idle":"2022-09-03T08:16:07.937167Z","shell.execute_reply.started":"2022-09-03T08:16:07.756891Z","shell.execute_reply":"2022-09-03T08:16:07.936400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**So After all of Works for this dataset RandomForestRegressor did a great job**:\n\nReason is: maximum data type is categorical. And RandomForest works best at categorical data. So For this dataset it gave less error than others.","metadata":{}},{"cell_type":"markdown","source":"\n![](https://thumbs.dreamstime.com/b/under-construction-sign-7718446.jpg)","metadata":{}},{"cell_type":"markdown","source":"# **If you learned something from this kernel please upvote it. If you think there should be more, the author needs to learn ... ,.... topics. Be my guest. Waiting for your suggestion.**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}