{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T03:31:59.265543Z","iopub.execute_input":"2022-07-11T03:31:59.266418Z","iopub.status.idle":"2022-07-11T03:31:59.294143Z","shell.execute_reply.started":"2022-07-11T03:31:59.266280Z","shell.execute_reply":"2022-07-11T03:31:59.293256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"original notebook = https://www.kaggle.com/code/artgor/eda-feature-engineering-and-model-interpretation","metadata":{}},{"cell_type":"markdown","source":"**Importing modules**","metadata":{}},{"cell_type":"code","source":"!pip install scikeras","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:46:09.143091Z","iopub.execute_input":"2022-07-11T03:46:09.143541Z","iopub.status.idle":"2022-07-11T03:46:20.288661Z","shell.execute_reply.started":"2022-07-11T03:46:09.143499Z","shell.execute_reply":"2022-07-11T03:46:20.287148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install --upgrade tensorflow ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:48:54.040501Z","iopub.execute_input":"2022-07-11T03:48:54.041799Z","iopub.status.idle":"2022-07-11T03:50:13.769270Z","shell.execute_reply.started":"2022-07-11T03:48:54.041734Z","shell.execute_reply":"2022-07-11T03:50:13.766963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport ast\nimport eli5\nimport lightgbm as lgb\n\npd.set_option('display.max_columns', None)\nfrom collections import Counter\nfrom wordcloud import WordCloud\n%matplotlib inline\nplt.style.use('ggplot')\nfrom sklearn.feature_extraction.text import TfidfVectorizer \nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.metrics import accuracy_score, mean_squared_error, mean_absolute_error\nfrom sklearn.model_selection import train_test_split, KFold\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import AdaBoostRegressor, GradientBoostingRegressor\nimport xgboost\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import RepeatedKFold\nfrom numpy import absolute\nfrom lightgbm import LGBMRegressor\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Normalization\n# from scikeras.wrappers import KerasRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\n\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:58:52.780718Z","iopub.execute_input":"2022-07-11T03:58:52.781318Z","iopub.status.idle":"2022-07-11T03:58:52.805600Z","shell.execute_reply.started":"2022-07-11T03:58:52.781268Z","shell.execute_reply":"2022-07-11T03:58:52.804450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Read train and test data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/tmdb-box-office-prediction/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/tmdb-box-office-prediction/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:12.312485Z","iopub.execute_input":"2022-07-11T03:32:12.313759Z","iopub.status.idle":"2022-07-11T03:32:14.113008Z","shell.execute_reply.started":"2022-07-11T03:32:12.313712Z","shell.execute_reply":"2022-07-11T03:32:14.111700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train data : \", train.shape)\nprint(\"Shape of test data : \", test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:14.114575Z","iopub.execute_input":"2022-07-11T03:32:14.115022Z","iopub.status.idle":"2022-07-11T03:32:14.121041Z","shell.execute_reply.started":"2022-07-11T03:32:14.114973Z","shell.execute_reply":"2022-07-11T03:32:14.120249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:14.123579Z","iopub.execute_input":"2022-07-11T03:32:14.123890Z","iopub.status.idle":"2022-07-11T03:32:14.190563Z","shell.execute_reply.started":"2022-07-11T03:32:14.123861Z","shell.execute_reply":"2022-07-11T03:32:14.189218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have given data in string format for accessing value by key it's important to convert string into dictionary","metadata":{}},{"cell_type":"code","source":"# Convert string columns to dictionary\njson_columns = ['belongs_to_collection', 'genres', 'production_companies', 'production_countries', 'spoken_languages', 'Keywords', 'cast', 'crew']\n\ndef text_to_dict(df, json_columns = json_columns):\n    for col in json_columns:\n        df[col] = df[col].apply(lambda x: {} if pd.isna(x) else ast.literal_eval(x))\n    return df\n\ntrain = text_to_dict(train)\ntest = text_to_dict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:14.192073Z","iopub.execute_input":"2022-07-11T03:32:14.192548Z","iopub.status.idle":"2022-07-11T03:32:25.887118Z","shell.execute_reply.started":"2022-07-11T03:32:14.192514Z","shell.execute_reply":"2022-07-11T03:32:25.885806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**belongs_to_collection**","metadata":{}},{"cell_type":"code","source":"train['belongs_to_collection'][:3]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:25.888914Z","iopub.execute_input":"2022-07-11T03:32:25.889656Z","iopub.status.idle":"2022-07-11T03:32:25.899398Z","shell.execute_reply.started":"2022-07-11T03:32:25.889605Z","shell.execute_reply":"2022-07-11T03:32:25.898246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['belongs_to_collection'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:25.901046Z","iopub.execute_input":"2022-07-11T03:32:25.902188Z","iopub.status.idle":"2022-07-11T03:32:25.933297Z","shell.execute_reply.started":"2022-07-11T03:32:25.902146Z","shell.execute_reply":"2022-07-11T03:32:25.931597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['collection'] = train['belongs_to_collection'].apply(lambda x: x[0]['name'] if x!={} else 0)\ntrain['has_collection'] = train['collection'].apply(lambda x: 1 if x!=0 else 0)\n\ntest['collection'] = test['belongs_to_collection'].apply(lambda x: x[0]['name'] if x!={} else 0)\ntest['has_collection'] = test['collection'].apply(lambda x: 1 if x!=0 else 0)\n\ntrain.drop(['belongs_to_collection'], axis=1, inplace=True)\ntest.drop(['belongs_to_collection'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:25.934848Z","iopub.execute_input":"2022-07-11T03:32:25.935306Z","iopub.status.idle":"2022-07-11T03:32:25.971639Z","shell.execute_reply.started":"2022-07-11T03:32:25.935260Z","shell.execute_reply":"2022-07-11T03:32:25.970236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Genres**","metadata":{}},{"cell_type":"code","source":"train['genres'][:3]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:25.973714Z","iopub.execute_input":"2022-07-11T03:32:25.974184Z","iopub.status.idle":"2022-07-11T03:32:25.984127Z","shell.execute_reply.started":"2022-07-11T03:32:25.974139Z","shell.execute_reply":"2022-07-11T03:32:25.983133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['genres'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:25.988247Z","iopub.execute_input":"2022-07-11T03:32:25.988912Z","iopub.status.idle":"2022-07-11T03:32:26.000487Z","shell.execute_reply.started":"2022-07-11T03:32:25.988876Z","shell.execute_reply":"2022-07-11T03:32:25.999212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"most of movie have 1 to 5 genre but there is also exist some movie that have more then 5 genre so may be thay can be outlier.","metadata":{}},{"cell_type":"code","source":"train['num_genres'] = train['genres'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_genres = list(train['genres'].apply(lambda x: [j['name'] for j in x] if x!={} else []).values)\ntrain['all_genres'] = train['genres'].apply(lambda x: ' '.join([j['name'] for j in x]) if x!={} else \"\" )\n\ntop_genres = [m[0] for m in Counter([i for j in list_of_genres for i in j]).most_common(15)]\nfor g in top_genres:\n    train['genres_'+g] = train['all_genres'].apply(lambda x: 1 if g in x else 0)\n\n\ntest['num_genres'] = test['genres'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_genres'] = test['genres'].apply(lambda x: ' '.join([j['name'] for j in x]) if x!={} else \"\" )\n\nfor g in top_genres:\n    test['genres_'+g] = test['all_genres'].apply(lambda x: 1 if g in x else 0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:26.002018Z","iopub.execute_input":"2022-07-11T03:32:26.002913Z","iopub.status.idle":"2022-07-11T03:32:26.103668Z","shell.execute_reply.started":"2022-07-11T03:32:26.002877Z","shell.execute_reply":"2022-07-11T03:32:26.102577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we made special category for most common 15 genre and assign 1 if movie belongs to that genre else 0","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\ntext = ' '.join(i for j in list_of_genres for i in j)\nimg = WordCloud(max_font_size=None, background_color='white', collocations=False,\n                      width=1200, height=1000).generate(text)\nplt.title('Top genres wordcloud')\nplt.axis(\"off\")\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:26.105033Z","iopub.execute_input":"2022-07-11T03:32:26.105568Z","iopub.status.idle":"2022-07-11T03:32:27.231754Z","shell.execute_reply.started":"2022-07-11T03:32:26.105534Z","shell.execute_reply":"2022-07-11T03:32:27.230768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on wordcloud we can say that most of the movie belongs to Drama, comedy and thriller genre.","metadata":{}},{"cell_type":"markdown","source":"**production_companies**","metadata":{}},{"cell_type":"markdown","source":"Number of production companies involved for making one movie","metadata":{}},{"cell_type":"code","source":"train['production_companies'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:27.232880Z","iopub.execute_input":"2022-07-11T03:32:27.233762Z","iopub.status.idle":"2022-07-11T03:32:27.244244Z","shell.execute_reply.started":"2022-07-11T03:32:27.233722Z","shell.execute_reply":"2022-07-11T03:32:27.243453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['production_companies'].head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:27.245518Z","iopub.execute_input":"2022-07-11T03:32:27.247700Z","iopub.status.idle":"2022-07-11T03:32:27.257536Z","shell.execute_reply.started":"2022-07-11T03:32:27.247651Z","shell.execute_reply":"2022-07-11T03:32:27.256497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_production_companies'] = train['production_companies'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_pc = train['production_companies'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_production_companies'] = train['production_companies'].apply(lambda x : ' '.join( i['name'] for i in x ) if x!={} else \"\" )\n\ntop_pc = [m[0] for m in Counter(i for j in list_of_pc for i in j).most_common(15)]\nfor g in top_pc:\n    train['production_companies_' + g] = train['all_production_companies'].apply(lambda x:1 if g in x else 0)\n\n\n\ntest['num_production_companies'] = test['production_companies'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_production_companies'] = test['production_companies'].apply(lambda x : ' '.join( i['name'] for i in x ) if x!={} else \"\" )\n\nfor g in top_pc:\n    test['production_companies_' + g] = test['all_production_companies'].apply(lambda x:1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:27.259041Z","iopub.execute_input":"2022-07-11T03:32:27.259989Z","iopub.status.idle":"2022-07-11T03:32:27.365175Z","shell.execute_reply.started":"2022-07-11T03:32:27.259946Z","shell.execute_reply":"2022-07-11T03:32:27.363764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\ntext = ' '.join(i for j in list_of_pc for i in j)\nimg = WordCloud(max_font_size=None, background_color='white', collocations=False,\n                      width=1200, height=1000).generate(text)\nplt.axis('off')\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:27.366481Z","iopub.execute_input":"2022-07-11T03:32:27.366819Z","iopub.status.idle":"2022-07-11T03:32:29.942115Z","shell.execute_reply.started":"2022-07-11T03:32:27.366787Z","shell.execute_reply":"2022-07-11T03:32:29.940645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Production Countries**","metadata":{}},{"cell_type":"code","source":"train['production_countries'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:29.943663Z","iopub.execute_input":"2022-07-11T03:32:29.944584Z","iopub.status.idle":"2022-07-11T03:32:29.952692Z","shell.execute_reply.started":"2022-07-11T03:32:29.944545Z","shell.execute_reply":"2022-07-11T03:32:29.951717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['production_countries'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:29.953868Z","iopub.execute_input":"2022-07-11T03:32:29.954789Z","iopub.status.idle":"2022-07-11T03:32:29.971037Z","shell.execute_reply.started":"2022-07-11T03:32:29.954735Z","shell.execute_reply":"2022-07-11T03:32:29.969893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In data there exist some movie that does not belongs to any country so we can say those are outliers.","metadata":{}},{"cell_type":"code","source":"train['num_production_countries'] = train['production_countries'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_prod_countries = train['production_countries'].apply(lambda x: [i['name'] for i in x]  if x!={} else []).values\ntrain['all_production_countries'] = train['production_countries'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\ntop_production_countries = [m[0] for m in Counter([i for j in list_of_prod_countries for i in j]).most_common(15)]\n\nfor g in top_production_countries:\n    train['production_countries_' + g] = train['all_production_countries'].apply(lambda x: 1 if g in x else 0)\n    \n\ntest['num_production_countries'] = test['production_countries'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_production_countries'] = test['production_countries'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_production_countries:\n    test['production_countries_' + g] = test['all_production_countries'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:29.972862Z","iopub.execute_input":"2022-07-11T03:32:29.973254Z","iopub.status.idle":"2022-07-11T03:32:30.067734Z","shell.execute_reply.started":"2022-07-11T03:32:29.973217Z","shell.execute_reply":"2022-07-11T03:32:30.066659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in Counter([i for j in list_of_prod_countries for i in j]).most_common(10):\n    print(i,(30-len(i))*\" \", j/len(list_of_prod_countries)*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.069131Z","iopub.execute_input":"2022-07-11T03:32:30.070012Z","iopub.status.idle":"2022-07-11T03:32:30.076957Z","shell.execute_reply.started":"2022-07-11T03:32:30.069979Z","shell.execute_reply":"2022-07-11T03:32:30.075861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"US involved as production country in more than 76% movie","metadata":{}},{"cell_type":"markdown","source":"**Spoken language**","metadata":{}},{"cell_type":"code","source":"train['spoken_languages'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.078124Z","iopub.execute_input":"2022-07-11T03:32:30.079210Z","iopub.status.idle":"2022-07-11T03:32:30.094314Z","shell.execute_reply.started":"2022-07-11T03:32:30.079161Z","shell.execute_reply":"2022-07-11T03:32:30.093192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_spoken_languages'] = train['spoken_languages'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_spoken_languages = train['spoken_languages'].apply(lambda x: [i['name'] for i in x] if x!={} else \"\").values\ntrain['all_spoken_languages'] = train['spoken_languages'].apply(lambda x: [' '.join(i['name'] for i in x)] if x!={} else [] )\ntop_spoken_languages = [m[0] for m in Counter(i for j in list_of_spoken_languages for i in j).most_common(15)]\n\nfor g in top_spoken_languages:\n    train['spoken_language_'+g] = train['all_spoken_languages'].apply(lambda x: 1 if g in x else 0)\n    \ntest['num_spoken_languages'] = test['spoken_languages'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_spoken_languages'] = test['spoken_languages'].apply(lambda x: [' '.join(i['name'] for i in x)] if x!={} else [] )\n\nfor g in top_spoken_languages:\n    test['spoken_language_'+g] = test['all_spoken_languages'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.095529Z","iopub.execute_input":"2022-07-11T03:32:30.095836Z","iopub.status.idle":"2022-07-11T03:32:30.435861Z","shell.execute_reply.started":"2022-07-11T03:32:30.095810Z","shell.execute_reply":"2022-07-11T03:32:30.434612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in Counter([i for j in list_of_spoken_languages for i in j]).most_common(10):\n    print(i, (70-len(i))*\" \", round(j*100/len(list_of_spoken_languages), 2))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.437388Z","iopub.execute_input":"2022-07-11T03:32:30.437873Z","iopub.status.idle":"2022-07-11T03:32:30.446436Z","shell.execute_reply.started":"2022-07-11T03:32:30.437826Z","shell.execute_reply":"2022-07-11T03:32:30.445222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"in 87% of the movie spoken language is english","metadata":{}},{"cell_type":"markdown","source":"**Keywords**","metadata":{}},{"cell_type":"code","source":"train['Keywords'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.448160Z","iopub.execute_input":"2022-07-11T03:32:30.449248Z","iopub.status.idle":"2022-07-11T03:32:30.461458Z","shell.execute_reply.started":"2022-07-11T03:32:30.449210Z","shell.execute_reply":"2022-07-11T03:32:30.460165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Keywords'].apply(lambda x: len(x) if x!={} else 0).value_counts().head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.462791Z","iopub.execute_input":"2022-07-11T03:32:30.463593Z","iopub.status.idle":"2022-07-11T03:32:30.474243Z","shell.execute_reply.started":"2022-07-11T03:32:30.463561Z","shell.execute_reply":"2022-07-11T03:32:30.473428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_Keywords'] = train['Keywords'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_Keywords = train['Keywords'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_Keywords'] = train['Keywords'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\ntop_Keywords = [m[0] for m in Counter(i for j in list_of_Keywords for i in j).most_common(30)]\n\nfor g in top_Keywords:\n    train['Keywords_'+g] = train['all_Keywords'].apply(lambda x: 1 if g in x else 0)\n    \n    \ntest['num_Keywords'] = test['Keywords'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_Keywords'] = test['Keywords'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_Keywords:\n    test['Keywords_'+g] = test['all_Keywords'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.475319Z","iopub.execute_input":"2022-07-11T03:32:30.476216Z","iopub.status.idle":"2022-07-11T03:32:30.658188Z","shell.execute_reply.started":"2022-07-11T03:32:30.476184Z","shell.execute_reply":"2022-07-11T03:32:30.657215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in Counter([i for j in list_of_Keywords for i in j]).most_common(20):\n    print(i, (50 - len(i))*\" \", j)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.659733Z","iopub.execute_input":"2022-07-11T03:32:30.660152Z","iopub.status.idle":"2022-07-11T03:32:30.674467Z","shell.execute_reply.started":"2022-07-11T03:32:30.660112Z","shell.execute_reply":"2022-07-11T03:32:30.673246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most occuring keywords are woman director, independent film, duringcreditsstinger, murder and based on novel","metadata":{}},{"cell_type":"markdown","source":"**Cast**","metadata":{}},{"cell_type":"code","source":"train['cast'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.682115Z","iopub.execute_input":"2022-07-11T03:32:30.682692Z","iopub.status.idle":"2022-07-11T03:32:30.699707Z","shell.execute_reply.started":"2022-07-11T03:32:30.682648Z","shell.execute_reply":"2022-07-11T03:32:30.698401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cast'].apply(lambda x:len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.701292Z","iopub.execute_input":"2022-07-11T03:32:30.702161Z","iopub.status.idle":"2022-07-11T03:32:30.714443Z","shell.execute_reply.started":"2022-07-11T03:32:30.702114Z","shell.execute_reply":"2022-07-11T03:32:30.713461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_cast'] = train['cast'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_cast = train['cast'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_cast'] = train['cast'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\ntop_cast = [m[0] for m in Counter([i for j in list_of_cast for i in j]).most_common(20)]\n\nfor g in top_cast:\n    train['cast_name_'+g] = train['all_cast'].apply(lambda x: 1 if g in x else 0)\n    \n\ntest['num_cast'] = test['cast'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_cast'] = test['cast'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_cast:\n    test['cast_name_'+g] = test['all_cast'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.715969Z","iopub.execute_input":"2022-07-11T03:32:30.716756Z","iopub.status.idle":"2022-07-11T03:32:30.963481Z","shell.execute_reply.started":"2022-07-11T03:32:30.716721Z","shell.execute_reply":"2022-07-11T03:32:30.962191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most popular popular cast and number of movies they are in","metadata":{}},{"cell_type":"code","source":"for i, j in Counter([i for j in list_of_cast for i in j]).most_common(15):\n    print(i, (50-len(i))*\" \", j)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:30.964714Z","iopub.execute_input":"2022-07-11T03:32:30.965048Z","iopub.status.idle":"2022-07-11T03:32:31.002476Z","shell.execute_reply.started":"2022-07-11T03:32:30.965017Z","shell.execute_reply":"2022-07-11T03:32:31.000870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_of_gender = train['cast'].apply(lambda x: [i['gender'] for i in x] if x!={} else [])\nCounter([i for j in list_of_gender for i in j]).most_common()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.004071Z","iopub.execute_input":"2022-07-11T03:32:31.004559Z","iopub.status.idle":"2022-07-11T03:32:31.042070Z","shell.execute_reply.started":"2022-07-11T03:32:31.004501Z","shell.execute_reply":"2022-07-11T03:32:31.040870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['gender_0_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 0]))\ntrain['gender_1_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 1]))\ntrain['gender_2_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 2]))\n\ntest['gender_0_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 0]))\ntest['gender_1_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 1]))\ntest['gender_2_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 2]))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.043650Z","iopub.execute_input":"2022-07-11T03:32:31.044035Z","iopub.status.idle":"2022-07-11T03:32:31.163963Z","shell.execute_reply.started":"2022-07-11T03:32:31.044003Z","shell.execute_reply":"2022-07-11T03:32:31.162814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we just assigned number of gender in total cast for particular one movie like number of male for x movie from all x cast","metadata":{}},{"cell_type":"code","source":"list_of_characters = train['cast'].apply(lambda x: [i['character'] for i in x] if x!={} else [])\nCounter(i for j in list_of_characters for i in j).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.165518Z","iopub.execute_input":"2022-07-11T03:32:31.166255Z","iopub.status.idle":"2022-07-11T03:32:31.221376Z","shell.execute_reply.started":"2022-07-11T03:32:31.166211Z","shell.execute_reply":"2022-07-11T03:32:31.220273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For cast character there are lot of people who play himself/ herself in movie","metadata":{}},{"cell_type":"code","source":"train['num_cast_characters'] = train['cast'].apply(lambda x: len(x) if x!={} else 0)\ntrain['all_cast_character'] = train['cast'].apply(lambda x: \" \".join(i['character'] for i in x) if x!={} else \"\")\ntop_cast_characters = [m[0] for m in Counter(i for j in list_of_characters for i in j).most_common(20)]\nfor g in top_cast_characters:\n    train['cast_character_'+g] = train['all_cast_character'].apply(lambda x: 1 if g in x else 0)\n\ntest['num_cast_characters'] = test['cast'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_cast_character'] = test['cast'].apply(lambda x: \" \".join(i['character'] for i in x) if x!={} else \"\")\nfor g in top_cast_characters:\n    test['cast_character_'+g] = test['all_cast_character'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.222679Z","iopub.execute_input":"2022-07-11T03:32:31.222955Z","iopub.status.idle":"2022-07-11T03:32:31.440085Z","shell.execute_reply.started":"2022-07-11T03:32:31.222930Z","shell.execute_reply":"2022-07-11T03:32:31.438904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Crew**","metadata":{}},{"cell_type":"code","source":"train['crew'].apply(lambda x: len(x) if x!={} else 0).value_counts().head(10), train['crew'].apply(lambda x: len(x) if x!={} else 0).value_counts().tail(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.441555Z","iopub.execute_input":"2022-07-11T03:32:31.442261Z","iopub.status.idle":"2022-07-11T03:32:31.457978Z","shell.execute_reply.started":"2022-07-11T03:32:31.442218Z","shell.execute_reply":"2022-07-11T03:32:31.456811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Usually there is 2 to 15 crew but there are also exist some data that have more then 100 crew","metadata":{}},{"cell_type":"code","source":"list_of_crew_names = train['crew'].apply(lambda x: [i['name'] for i in x] if x!={} else [])\nCounter([i for j in list_of_crew_names for i in j]).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.459918Z","iopub.execute_input":"2022-07-11T03:32:31.460709Z","iopub.status.idle":"2022-07-11T03:32:31.527913Z","shell.execute_reply.started":"2022-07-11T03:32:31.460662Z","shell.execute_reply":"2022-07-11T03:32:31.527064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_crew'] = train['crew'].apply(lambda x: len(x) if x!={} else 0)\ntrain['all_crew_name'] = train['crew'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else [])\ntop_crew_names = [m[0] for m in Counter([i for j in list_of_crew_names for i in j]).most_common(15)]\n\nfor g in top_crew_names:\n    train['crew_name_' + g] = train['all_crew_name'].apply(lambda x: 1 if g in x else 0)\n\n\ntest['num_crew'] = test['crew'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_crew_name'] = test['crew'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else [])\n\nfor g in top_crew_names:\n    test['crew_name_' + g] = test['all_crew_name'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.529223Z","iopub.execute_input":"2022-07-11T03:32:31.529732Z","iopub.status.idle":"2022-07-11T03:32:31.742984Z","shell.execute_reply.started":"2022-07-11T03:32:31.529697Z","shell.execute_reply":"2022-07-11T03:32:31.741870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Genders for crew**","metadata":{}},{"cell_type":"code","source":"list_of_gender = train['crew'].apply(lambda x: [i['gender'] for i in x] if x!={} else [])\nCounter(i for j in list_of_gender for i in j).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.744416Z","iopub.execute_input":"2022-07-11T03:32:31.745247Z","iopub.status.idle":"2022-07-11T03:32:31.777998Z","shell.execute_reply.started":"2022-07-11T03:32:31.745208Z","shell.execute_reply":"2022-07-11T03:32:31.776858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['crew_gender_0'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntrain['crew_gender_1'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntrain['crew_gender_2'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 2))\n\ntest['crew_gender_0'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntest['crew_gender_1'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntest['crew_gender_2'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 2))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.779606Z","iopub.execute_input":"2022-07-11T03:32:31.780108Z","iopub.status.idle":"2022-07-11T03:32:31.902618Z","shell.execute_reply.started":"2022-07-11T03:32:31.780070Z","shell.execute_reply":"2022-07-11T03:32:31.901604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we just assigned number of gender in total crew for particular one movie like number of male for x movie from all x cast","metadata":{}},{"cell_type":"markdown","source":"Job of crew and number of jobs","metadata":{}},{"cell_type":"code","source":"list_of_jobs = train['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\nCounter([i for j in list_of_jobs for i in j]).most_common(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.904007Z","iopub.execute_input":"2022-07-11T03:32:31.904321Z","iopub.status.idle":"2022-07-11T03:32:31.956421Z","shell.execute_reply.started":"2022-07-11T03:32:31.904294Z","shell.execute_reply":"2022-07-11T03:32:31.955115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['all_job'] = train['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\ntop_job = [m[0] for m in Counter([i for j in list_of_jobs for i in j]).most_common(20)]\n\nfor g in top_job:\n    train['job_'+g] = train['all_job'].apply(lambda x: 1 if g in x else 0)\n\ntest['all_job'] = test['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\n\nfor g in top_job:\n    test['job_'+g] = test['all_job'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:31.957435Z","iopub.execute_input":"2022-07-11T03:32:31.958283Z","iopub.status.idle":"2022-07-11T03:32:32.246656Z","shell.execute_reply.started":"2022-07-11T03:32:31.958246Z","shell.execute_reply":"2022-07-11T03:32:32.245423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Visualization**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(16, 6))\n\nplt.subplot(1, 2, 1)\nplt.hist(train['revenue'])\nplt.title(\"Revenue distribution\")\n\nplt.subplot(1, 2, 2)\nplt.hist(np.log1p(train['revenue']))\nplt.title(\"Log1 revenue distribution\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:32.248322Z","iopub.execute_input":"2022-07-11T03:32:32.249174Z","iopub.status.idle":"2022-07-11T03:32:32.591496Z","shell.execute_reply.started":"2022-07-11T03:32:32.249119Z","shell.execute_reply":"2022-07-11T03:32:32.587345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When we apply log on revenue then it becomes less skewed and it will be better for our model","metadata":{}},{"cell_type":"code","source":"train['log_revenue'] = np.log1p(train['revenue'])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:32.593124Z","iopub.execute_input":"2022-07-11T03:32:32.593476Z","iopub.status.idle":"2022-07-11T03:32:32.599680Z","shell.execute_reply.started":"2022-07-11T03:32:32.593444Z","shell.execute_reply":"2022-07-11T03:32:32.598510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['budget'].skew(), np.log1p(train['budget']).skew()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:32.601134Z","iopub.execute_input":"2022-07-11T03:32:32.601934Z","iopub.status.idle":"2022-07-11T03:32:32.613933Z","shell.execute_reply.started":"2022-07-11T03:32:32.601898Z","shell.execute_reply":"2022-07-11T03:32:32.612830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(16, 6))\n\nplt.subplot(1,2,1)\nplt.hist(train['budget'])\nplt.title(\"Budget distribution\")\n\nplt.subplot(1,2,2)\nplt.hist(np.log1p(train['budget']))\nplt.title(\"Log1p Budget distribution\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:32.615311Z","iopub.execute_input":"2022-07-11T03:32:32.616348Z","iopub.status.idle":"2022-07-11T03:32:32.967095Z","shell.execute_reply.started":"2022-07-11T03:32:32.616305Z","shell.execute_reply":"2022-07-11T03:32:32.966286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\nplt.subplot(1, 2, 1)\nsns.scatterplot(x = np.log(train['budget']), y=train['log_revenue'])\nplt.xlabel('log_budget')\nplt.title(\"log Revenue vs log budget\")\n\nplt.subplot(1,2,2)\nsns.scatterplot(x='budget', y='log_revenue', data=train)\nplt.title(\"log revenue vs budget\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:32.968160Z","iopub.execute_input":"2022-07-11T03:32:32.968813Z","iopub.status.idle":"2022-07-11T03:32:33.386065Z","shell.execute_reply.started":"2022-07-11T03:32:32.968778Z","shell.execute_reply":"2022-07-11T03:32:33.384962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scatter plot of log_revenue vs log_budget is very good as compared to log_revenue vs budget","metadata":{}},{"cell_type":"code","source":"train['log_budget'] = np.log1p(train['budget'])\ntest['log_budget'] = np.log1p(test['budget'])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:33.387983Z","iopub.execute_input":"2022-07-11T03:32:33.388469Z","iopub.status.idle":"2022-07-11T03:32:33.396751Z","shell.execute_reply.started":"2022-07-11T03:32:33.388434Z","shell.execute_reply":"2022-07-11T03:32:33.394742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['homepage'].apply(lambda x: 1 if x!=np.nan else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:33.397976Z","iopub.execute_input":"2022-07-11T03:32:33.398310Z","iopub.status.idle":"2022-07-11T03:32:33.412668Z","shell.execute_reply.started":"2022-07-11T03:32:33.398280Z","shell.execute_reply":"2022-07-11T03:32:33.411466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(1,2,figsize=(16, 6))\nplt.subplot(1,2,1)\nsns.boxenplot(x='original_language', y='revenue', data= train.loc[train['original_language'].isin(train['original_language'].value_counts().head(10).index)] )\nplt.title(\"Revenue vs Original language\")\n\nplt.subplot(1,2,2)\nsns.boxenplot(x='original_language', y='log_revenue', data= train.loc[train['original_language'].isin(train['original_language'].value_counts().head(10).index)] )\nplt.title(\"Log Revenue vs Original language\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:33.414614Z","iopub.execute_input":"2022-07-11T03:32:33.415086Z","iopub.status.idle":"2022-07-11T03:32:33.974925Z","shell.execute_reply.started":"2022-07-11T03:32:33.415040Z","shell.execute_reply":"2022-07-11T03:32:33.973827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on previous three graph we can completely say that log_revenue is really good choice for our model as compared to revenue","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(9, 9))\ntitle = ' '.join(train['original_title'].values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(title)\nplt.title(\"Original title wordcloud\")\nplt.imshow(img)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:33.977045Z","iopub.execute_input":"2022-07-11T03:32:33.977519Z","iopub.status.idle":"2022-07-11T03:32:36.957648Z","shell.execute_reply.started":"2022-07-11T03:32:33.977476Z","shell.execute_reply":"2022-07-11T03:32:36.956152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most occuring words in movie titles are Man and Last.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(9,9))\nOverview = ' '.join(train['overview'].fillna(' ').values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(Overview)\nplt.imshow(img)\nplt.title(\"Overview wordcloud\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:36.959685Z","iopub.execute_input":"2022-07-11T03:32:36.960289Z","iopub.status.idle":"2022-07-11T03:32:40.160706Z","shell.execute_reply.started":"2022-07-11T03:32:36.960241Z","shell.execute_reply":"2022-07-11T03:32:40.159193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see impact of words on revenue","metadata":{}},{"cell_type":"code","source":"tfidf = TfidfVectorizer()\n\nx = tfidf.fit_transform(train['overview'].fillna(''))\ny = train['log_revenue']\n\nX_train, X_valid, y_train, y_valid = train_test_split(x, y, test_size=0.1)\n\nlr = LinearRegression()\nlr.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.162744Z","iopub.execute_input":"2022-07-11T03:32:40.163201Z","iopub.status.idle":"2022-07-11T03:32:40.448284Z","shell.execute_reply.started":"2022-07-11T03:32:40.163160Z","shell.execute_reply":"2022-07-11T03:32:40.446981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_squared_error(lr.predict(X_valid), y_valid), mean_absolute_error(lr.predict(X_valid), y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.450266Z","iopub.execute_input":"2022-07-11T03:32:40.451105Z","iopub.status.idle":"2022-07-11T03:32:40.462645Z","shell.execute_reply.started":"2022-07-11T03:32:40.451050Z","shell.execute_reply":"2022-07-11T03:32:40.461205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eli5.show_weights(lr, vec=tfidf, top=20, feature_filter=lambda x: x != '<BIAS>')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.464889Z","iopub.execute_input":"2022-07-11T03:32:40.466032Z","iopub.status.idle":"2022-07-11T03:32:40.604390Z","shell.execute_reply.started":"2022-07-11T03:32:40.465966Z","shell.execute_reply":"2022-07-11T03:32:40.603479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Target value : \", train['log_revenue'][500])\neli5.show_prediction(lr, doc=train['overview'].values[500], vec = tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.606358Z","iopub.execute_input":"2022-07-11T03:32:40.607522Z","iopub.status.idle":"2022-07-11T03:32:40.653552Z","shell.execute_reply.started":"2022-07-11T03:32:40.607478Z","shell.execute_reply":"2022-07-11T03:32:40.652433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Popularity**","metadata":{}},{"cell_type":"code","source":"plt.subplots(1,2,figsize=(16, 6))\n\nplt.subplot(1,2,1)\nplt.scatter(x='popularity', y='revenue', data=train)\nplt.title(\"Revenue vs Popularity\")\n\nplt.subplot(1,2,2)\nplt.scatter(x='popularity', y='log_revenue', data=train)\nplt.title(\"Log revenue vs popularity\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.655461Z","iopub.execute_input":"2022-07-11T03:32:40.656229Z","iopub.status.idle":"2022-07-11T03:32:40.986543Z","shell.execute_reply.started":"2022-07-11T03:32:40.656184Z","shell.execute_reply":"2022-07-11T03:32:40.985417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on scatter plot we can say revenue is not much dependent on popularity","metadata":{}},{"cell_type":"markdown","source":"**Release Date**","metadata":{}},{"cell_type":"code","source":"test[test.release_date.isnull()].index","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:40.988214Z","iopub.execute_input":"2022-07-11T03:32:40.988907Z","iopub.status.idle":"2022-07-11T03:32:41.010562Z","shell.execute_reply.started":"2022-07-11T03:32:40.988861Z","shell.execute_reply":"2022-07-11T03:32:41.009553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.at[828, 'release_date'] = '1/5/2000'","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:41.012059Z","iopub.execute_input":"2022-07-11T03:32:41.012424Z","iopub.status.idle":"2022-07-11T03:32:41.018024Z","shell.execute_reply.started":"2022-07-11T03:32:41.012355Z","shell.execute_reply":"2022-07-11T03:32:41.016615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['release_date_year'] = train['release_date'].apply(lambda x: 19*100 + int(x.split('/')[-1]) if int(x.split('/')[-1]) > 22 else 20*100 + int(x.split('/')[-1]))\ntest['release_date_year'] = test['release_date'].apply(lambda x: 19*100 + int(x.split('/')[-1]) if int(x.split('/')[-1]) > 22 else 20*100 + int(x.split('/')[-1]))\n\ntrain['release_date_month'] = pd.to_datetime(train['release_date']).dt.month\ntest['release_date_month'] = pd.to_datetime(test['release_date']).dt.month\n\ntrain['release_date_weekday'] = pd.to_datetime(train['release_date']).dt.weekday\ntest['release_date_weekday'] = pd.to_datetime(test['release_date']).dt.weekday\n\ntrain['release_date_weekofyear'] = pd.to_datetime(train['release_date']).dt.weekofyear\ntest['release_date_weekofyear'] = pd.to_datetime(test['release_date']).dt.weekofyear\n\ntrain['release_date_day'] = pd.to_datetime(train['release_date']).dt.day\ntest['release_date_day'] = pd.to_datetime(test['release_date']).dt.day","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:41.019382Z","iopub.execute_input":"2022-07-11T03:32:41.020158Z","iopub.status.idle":"2022-07-11T03:32:42.907811Z","shell.execute_reply.started":"2022-07-11T03:32:41.020124Z","shell.execute_reply":"2022-07-11T03:32:42.906544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = test['release_date_year'].value_counts().sort_index()\n\ndata = [go.Scatter(x = d1.index, y=d1.values, name='train'), go.Scatter(x = d2.index, y = d2.values, name='test')]\nlayout = go.Layout(dict(title=\"Number of movies per year\",\n                  xaxis = dict(title=\"year\"),\n                  yaxis = dict(title=\"Count\")),legend = dict(orientation='v'))\npy.iplot(dict(data = data, layout=layout))\n\n# d1 = train['release_date_year'].value_counts().sort_index()\n# d2 = test['release_date_year'].value_counts().sort_index()\n# data = [go.Scatter(x=d1.index, y=d1.values, name='train'), go.Scatter(x=d2.index, y=d2.values, name='test')]\n# layout = go.Layout(dict(title = \"Number of films per year\",\n#                   xaxis = dict(title = 'Year'),\n#                   yaxis = dict(title = 'Count'),\n#                   ),legend=dict(\n#                 orientation=\"v\"))\n# py.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:42.909186Z","iopub.execute_input":"2022-07-11T03:32:42.909593Z","iopub.status.idle":"2022-07-11T03:32:43.820912Z","shell.execute_reply.started":"2022-07-11T03:32:42.909560Z","shell.execute_reply":"2022-07-11T03:32:43.819684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = test['release_date_year'].value_counts().sort_index()\ndata = [go.Scatter(x=d1.index, y=d1.values, name='train'), go.Scatter(x=d2.index, y=d2.values, name='test')]\nlayout = go.Layout(dict(title = \"Number of films per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:43.822399Z","iopub.execute_input":"2022-07-11T03:32:43.822729Z","iopub.status.idle":"2022-07-11T03:32:43.858511Z","shell.execute_reply.started":"2022-07-11T03:32:43.822700Z","shell.execute_reply":"2022-07-11T03:32:43.857312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = train.groupby('release_date_year')['revenue'].sum()\n\ndata = [go.Scatter(x=d1.index, y=d1.values, name='film count'), go.Scatter(x=d2.index, y=d2.values, name='total revenue', yaxis='y2')]\nlayout = go.Layout(dict(title = \"Number of films and total revenue per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  yaxis2=dict(title='Total revenue', overlaying='y', side='right')\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:43.859799Z","iopub.execute_input":"2022-07-11T03:32:43.860662Z","iopub.status.idle":"2022-07-11T03:32:43.904917Z","shell.execute_reply.started":"2022-07-11T03:32:43.860624Z","shell.execute_reply":"2022-07-11T03:32:43.903788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = train.groupby(['release_date_year'])['revenue'].mean()\ndata = [go.Scatter(x=d1.index, y=d1.values, name='film count'), go.Scatter(x=d2.index, y=d2.values, name='mean revenue', yaxis='y2')]\nlayout = go.Layout(dict(title = \"Number of films and average revenue per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  yaxis2=dict(title='Average revenue', overlaying='y', side='right')\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:43.906612Z","iopub.execute_input":"2022-07-11T03:32:43.907536Z","iopub.status.idle":"2022-07-11T03:32:43.948475Z","shell.execute_reply.started":"2022-07-11T03:32:43.907488Z","shell.execute_reply":"2022-07-11T03:32:43.947585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x='release_date_weekday', y='revenue', data=train);\nplt.title('Revenue on different days of week of release');","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:43.949710Z","iopub.execute_input":"2022-07-11T03:32:43.950195Z","iopub.status.idle":"2022-07-11T03:32:44.210767Z","shell.execute_reply.started":"2022-07-11T03:32:43.950165Z","shell.execute_reply":"2022-07-11T03:32:44.209565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Films releases on Wednesdays and on Thursdays tend to have a higher revenue.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(20, 6))\n\nplt.subplot(1,3,1)\nplt.hist(train['runtime'].fillna(0)/60, bins=20)\nplt.title(\"Distribution of runtime\")\n\nplt.subplot(1,3,2)\nplt.scatter(x = 'runtime', y='revenue', data=train)\nplt.title(\"Revenue vs runtime\")\n\nplt.subplot(1,3,3)\nplt.scatter(x = 'runtime', y='popularity', data=train)\nplt.title(\"Revenue vs popularity\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:44.212647Z","iopub.execute_input":"2022-07-11T03:32:44.213339Z","iopub.status.idle":"2022-07-11T03:32:44.648550Z","shell.execute_reply.started":"2022-07-11T03:32:44.213283Z","shell.execute_reply":"2022-07-11T03:32:44.647392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['status'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:44.650355Z","iopub.execute_input":"2022-07-11T03:32:44.650864Z","iopub.status.idle":"2022-07-11T03:32:44.661265Z","shell.execute_reply.started":"2022-07-11T03:32:44.650807Z","shell.execute_reply":"2022-07-11T03:32:44.660110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['status'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:44.662417Z","iopub.execute_input":"2022-07-11T03:32:44.663407Z","iopub.status.idle":"2022-07-11T03:32:44.672906Z","shell.execute_reply.started":"2022-07-11T03:32:44.663347Z","shell.execute_reply":"2022-07-11T03:32:44.671684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tagline**","metadata":{}},{"cell_type":"code","source":"text = ' '.join(train['tagline'].fillna(\"\").values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(text)\nplt.figure(figsize=(9, 9))\nplt.imshow(img)\nplt.title(\"Tagline wordcloud\")\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:44.674009Z","iopub.execute_input":"2022-07-11T03:32:44.674791Z","iopub.status.idle":"2022-07-11T03:32:47.325771Z","shell.execute_reply.started":"2022-07-11T03:32:44.674721Z","shell.execute_reply":"2022-07-11T03:32:47.324857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(3, 5, figsize=(24, 12))\nplt.suptitle('Violinplot of revenue vs genres')\nfor i, e in enumerate([col for col in train.columns if 'genres_' in col]):\n    sns.violinplot(x=e, y='revenue', data=train, ax=axes[i // 5][i % 5]);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:47.326992Z","iopub.execute_input":"2022-07-11T03:32:47.327528Z","iopub.status.idle":"2022-07-11T03:32:49.249919Z","shell.execute_reply.started":"2022-07-11T03:32:47.327482Z","shell.execute_reply":"2022-07-11T03:32:49.248658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Family, adventure, fantasy and animation have high revenue","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(4, 5, figsize=(21, 18))\nfig.tight_layout()\nplt.subplots_adjust(wspace=None, hspace=None)\nplt.axis('off')\nplt.suptitle(\"Violin plot of revenue vs production companies\")\nfor i,e in enumerate([col for col in train.columns if 'production_companies_' in col]):\n    sns.violinplot(x = e, y='revenue', data = train, ax = ax[i // 5][i % 5])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:49.251317Z","iopub.execute_input":"2022-07-11T03:32:49.251701Z","iopub.status.idle":"2022-07-11T03:32:51.785316Z","shell.execute_reply.started":"2022-07-11T03:32:49.251668Z","shell.execute_reply":"2022-07-11T03:32:51.784175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(4, 5, figsize=(24, 24))\nplt.suptitle('Violinplot of revenue vs production country')\nfor i, e in enumerate([col for col in train.columns if 'production_countries_' in col]):\n    sns.violinplot(x=e, y='revenue', data=train, ax=axes[i // 5][i % 5]);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:51.786825Z","iopub.execute_input":"2022-07-11T03:32:51.787209Z","iopub.status.idle":"2022-07-11T03:32:54.570174Z","shell.execute_reply.started":"2022-07-11T03:32:51.787175Z","shell.execute_reply":"2022-07-11T03:32:54.569085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\n\nplt.subplot(1, 2, 1)\nplt.scatter(train['num_cast'], train['revenue'])\nplt.title('Number of cast members vs revenue');\nplt.xlabel(\"Number of cast\")\nplt.ylabel(\"Revenue\")\n\nplt.subplot(1, 2, 2)\nplt.scatter(train['num_cast'], train['log_revenue'])\nplt.title('Number of cast members vs log revenue');\nplt.xlabel(\"Number of cast\")\nplt.ylabel(\"Log Revenue\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:54.571599Z","iopub.execute_input":"2022-07-11T03:32:54.572746Z","iopub.status.idle":"2022-07-11T03:32:54.958396Z","shell.execute_reply.started":"2022-07-11T03:32:54.572707Z","shell.execute_reply":"2022-07-11T03:32:54.957384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"number of cast and log_revenue is positively correlated with eachother","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 4, ncols = 5, figsize=(24, 18))\n\nfor i,e in enumerate([col for col in train.columns if 'cast_name_' in col]):\n    sns.violinplot(x = e, y = 'revenue', data= train, ax = ax[i//5][i%5])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:54.959714Z","iopub.execute_input":"2022-07-11T03:32:54.960048Z","iopub.status.idle":"2022-07-11T03:32:57.548484Z","shell.execute_reply.started":"2022-07-11T03:32:54.960018Z","shell.execute_reply":"2022-07-11T03:32:57.547351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len([col for col in train.columns if 'cast_character_' in col])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:57.549826Z","iopub.execute_input":"2022-07-11T03:32:57.550173Z","iopub.status.idle":"2022-07-11T03:32:57.557342Z","shell.execute_reply.started":"2022-07-11T03:32:57.550142Z","shell.execute_reply":"2022-07-11T03:32:57.556281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(4, 5, figsize=(24, 18))\nfor i,e in enumerate([col for col in train.columns if 'cast_character_' in col]):\n    sns.violinplot(x = e, y = 'revenue', data=train, ax = ax[i//5][i%5])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:32:57.558947Z","iopub.execute_input":"2022-07-11T03:32:57.559546Z","iopub.status.idle":"2022-07-11T03:33:00.199737Z","shell.execute_reply.started":"2022-07-11T03:32:57.559513Z","shell.execute_reply":"2022-07-11T03:33:00.198693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(1, 2, figsize=(18, 6))\n\nplt.subplot(1,2,1)\nsns.scatterplot(x = 'num_crew', y = 'revenue', data=train)\nplt.title(\"Revenue vs number of crew\")\nplt.xlabel(\"Number of crew\")\nplt.ylabel(\"Revenue\")\n\nplt.subplot(1,2, 2)\nsns.scatterplot(x = 'num_crew', y = 'log_revenue', data=train)\nplt.title(\"log_revenue vs number of crew\")\nplt.xlabel(\"Number of crew\")\nplt.ylabel(\"log_revenue\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:00.201236Z","iopub.execute_input":"2022-07-11T03:33:00.201892Z","iopub.status.idle":"2022-07-11T03:33:01.047127Z","shell.execute_reply.started":"2022-07-11T03:33:00.201852Z","shell.execute_reply":"2022-07-11T03:33:01.045810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.048828Z","iopub.execute_input":"2022-07-11T03:33:01.049906Z","iopub.status.idle":"2022-07-11T03:33:01.172004Z","shell.execute_reply.started":"2022-07-11T03:33:01.049847Z","shell.execute_reply":"2022-07-11T03:33:01.168629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here i deleted column that contains unique values and columns that i already took data from them","metadata":{}},{"cell_type":"code","source":"train.drop(['genres', 'homepage', 'imdb_id', 'poster_path', 'production_companies','production_countries','release_date', 'spoken_languages','status', 'Keywords','cast','crew', 'id'], axis=1, inplace=True)\ntest.drop(['genres', 'homepage', 'imdb_id', 'poster_path', 'production_companies','production_countries','release_date', 'spoken_languages','status', 'Keywords','cast','crew', 'id'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.174082Z","iopub.execute_input":"2022-07-11T03:33:01.174560Z","iopub.status.idle":"2022-07-11T03:33:01.285922Z","shell.execute_reply.started":"2022-07-11T03:33:01.174515Z","shell.execute_reply":"2022-07-11T03:33:01.284857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['collection'] = train['collection'].apply(lambda x: str(x))\ntest['collection'] = test['collection'].apply(lambda x: str(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.287427Z","iopub.execute_input":"2022-07-11T03:33:01.287749Z","iopub.status.idle":"2022-07-11T03:33:01.297080Z","shell.execute_reply.started":"2022-07-11T03:33:01.287721Z","shell.execute_reply":"2022-07-11T03:33:01.296042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using label encoder assign unique number to all unique values","metadata":{}},{"cell_type":"code","source":"for col in ['original_language', 'all_genres', 'collection']:\n    lb = LabelEncoder()\n    lb.fit(list(train[col].fillna(\"\")) + list(test[col].fillna(\"\")))\n    train[col] = lb.transform(train[col].fillna(\"\"))\n    test[col] = lb.transform(test[col].fillna(\"\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.298774Z","iopub.execute_input":"2022-07-11T03:33:01.299107Z","iopub.status.idle":"2022-07-11T03:33:01.351934Z","shell.execute_reply.started":"2022-07-11T03:33:01.299071Z","shell.execute_reply":"2022-07-11T03:33:01.350869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['title', 'tagline', 'overview', 'original_title']:\n    train['len_' + col] = train[col].fillna('').apply(lambda x: len(str(x)))\n    train['words_' + col] = train[col].fillna('').apply(lambda x: len(str(x.split(' '))))\n    train = train.drop(col, axis=1)\n    test['len_' + col] = test[col].fillna('').apply(lambda x: len(str(x)))\n    test['words_' + col] = test[col].fillna('').apply(lambda x: len(str(x.split(' '))))\n    test = test.drop(col, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.362272Z","iopub.execute_input":"2022-07-11T03:33:01.362667Z","iopub.status.idle":"2022-07-11T03:33:01.560586Z","shell.execute_reply.started":"2022-07-11T03:33:01.362637Z","shell.execute_reply":"2022-07-11T03:33:01.559564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again dropped unnecessary columns","metadata":{}},{"cell_type":"code","source":"train.drop(['all_production_companies', 'all_production_countries', 'all_spoken_languages', 'all_Keywords', 'all_cast', 'all_cast_character', 'all_crew_name', 'all_job'], axis=1, inplace=True)\ntest.drop(['all_production_companies', 'all_production_countries', 'all_spoken_languages', 'all_Keywords', 'all_cast', 'all_cast_character', 'all_crew_name', 'all_job'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.562055Z","iopub.execute_input":"2022-07-11T03:33:01.562527Z","iopub.status.idle":"2022-07-11T03:33:01.591892Z","shell.execute_reply.started":"2022-07-11T03:33:01.562485Z","shell.execute_reply":"2022-07-11T03:33:01.590457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.594069Z","iopub.execute_input":"2022-07-11T03:33:01.594574Z","iopub.status.idle":"2022-07-11T03:33:01.667642Z","shell.execute_reply.started":"2022-07-11T03:33:01.594527Z","shell.execute_reply":"2022-07-11T03:33:01.666808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.668454Z","iopub.execute_input":"2022-07-11T03:33:01.668745Z","iopub.status.idle":"2022-07-11T03:33:01.741646Z","shell.execute_reply.started":"2022-07-11T03:33:01.668720Z","shell.execute_reply":"2022-07-11T03:33:01.740460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.743140Z","iopub.execute_input":"2022-07-11T03:33:01.743595Z","iopub.status.idle":"2022-07-11T03:33:01.750793Z","shell.execute_reply.started":"2022-07-11T03:33:01.743551Z","shell.execute_reply":"2022-07-11T03:33:01.749969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Columns that have null values\")\nfor col in test.columns:\n    if train[col].isnull().sum() > 0:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.751872Z","iopub.execute_input":"2022-07-11T03:33:01.752738Z","iopub.status.idle":"2022-07-11T03:33:01.823015Z","shell.execute_reply.started":"2022-07-11T03:33:01.752707Z","shell.execute_reply":"2022-07-11T03:33:01.821834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Columns that are not in test but in train\")\nfor col in train.columns:\n    if col not in test.columns:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.825283Z","iopub.execute_input":"2022-07-11T03:33:01.826506Z","iopub.status.idle":"2022-07-11T03:33:01.835417Z","shell.execute_reply.started":"2022-07-11T03:33:01.826457Z","shell.execute_reply":"2022-07-11T03:33:01.834309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['budget'], axis=1, inplace=True)\ntest.drop(['budget'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.836774Z","iopub.execute_input":"2022-07-11T03:33:01.837940Z","iopub.status.idle":"2022-07-11T03:33:01.852963Z","shell.execute_reply.started":"2022-07-11T03:33:01.837893Z","shell.execute_reply":"2022-07-11T03:33:01.852067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['runtime'].fillna(train['runtime'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.854268Z","iopub.execute_input":"2022-07-11T03:33:01.855026Z","iopub.status.idle":"2022-07-11T03:33:01.861615Z","shell.execute_reply.started":"2022-07-11T03:33:01.854989Z","shell.execute_reply":"2022-07-11T03:33:01.860220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drop columns that have only one unique value","metadata":{}},{"cell_type":"code","source":"for col in train.columns:\n    if(train[col].var() == 0):\n        print(col)\n        train.drop(col, axis=1, inplace=True)\n        test.drop(col, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.862889Z","iopub.execute_input":"2022-07-11T03:33:01.864181Z","iopub.status.idle":"2022-07-11T03:33:01.908295Z","shell.execute_reply.started":"2022-07-11T03:33:01.864142Z","shell.execute_reply":"2022-07-11T03:33:01.907103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Y as log_revenue our dependent variable\n\nX our input features","metadata":{}},{"cell_type":"code","source":"X = train.drop(['log_revenue', 'revenue'], axis=1)\nY = train['log_revenue']","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.909487Z","iopub.execute_input":"2022-07-11T03:33:01.909795Z","iopub.status.idle":"2022-07-11T03:33:01.916874Z","shell.execute_reply.started":"2022-07-11T03:33:01.909767Z","shell.execute_reply":"2022-07-11T03:33:01.916130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divide dataset into train and valid for validation purpose","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, Y_train, Y_valid = train_test_split(X, Y, test_size=0.1, random_state = 42)\nX_train.shape, X_valid.shape, Y_train.shape, Y_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.918066Z","iopub.execute_input":"2022-07-11T03:33:01.918571Z","iopub.status.idle":"2022-07-11T03:33:01.933962Z","shell.execute_reply.started":"2022-07-11T03:33:01.918541Z","shell.execute_reply":"2022-07-11T03:33:01.932664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Linear Regression**","metadata":{}},{"cell_type":"code","source":"lr = LinearRegression()\nlr.fit(X_train, Y_train)\nmean_squared_error(lr.predict(X_valid), Y_valid), mean_absolute_error(lr.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:01.935457Z","iopub.execute_input":"2022-07-11T03:33:01.935849Z","iopub.status.idle":"2022-07-11T03:33:02.023242Z","shell.execute_reply.started":"2022-07-11T03:33:01.935807Z","shell.execute_reply":"2022-07-11T03:33:02.020449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Random Forest**","metadata":{}},{"cell_type":"code","source":"rf = RandomForestRegressor()\nrf.fit(X_train, Y_train)\nmean_squared_error(rf.predict(X_valid), Y_valid), mean_absolute_error(rf.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:02.025563Z","iopub.execute_input":"2022-07-11T03:33:02.028646Z","iopub.status.idle":"2022-07-11T03:33:08.975744Z","shell.execute_reply.started":"2022-07-11T03:33:02.028595Z","shell.execute_reply":"2022-07-11T03:33:08.974458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Importance","metadata":{}},{"cell_type":"code","source":"ax = plt.figure(figsize=(12, 10))\nfeat_importances = pd.Series(rf.feature_importances_, index=X_train.columns)\nfeat_importances.nlargest(30).plot(kind='barh')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:08.977278Z","iopub.execute_input":"2022-07-11T03:33:08.977732Z","iopub.status.idle":"2022-07-11T03:33:09.383846Z","shell.execute_reply.started":"2022-07-11T03:33:08.977689Z","shell.execute_reply":"2022-07-11T03:33:09.382287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"output feature is highly dependent on log_budget, populrity and release_date_year","metadata":{}},{"cell_type":"markdown","source":"**Ada boost**","metadata":{}},{"cell_type":"code","source":"ada = AdaBoostRegressor()\nada.fit(X_train, Y_train)\nmean_squared_error(ada.predict(X_valid), Y_valid), mean_absolute_error(ada.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:09.385786Z","iopub.execute_input":"2022-07-11T03:33:09.386846Z","iopub.status.idle":"2022-07-11T03:33:10.501070Z","shell.execute_reply.started":"2022-07-11T03:33:09.386790Z","shell.execute_reply":"2022-07-11T03:33:10.499827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Gradient Boosting regression**","metadata":{}},{"cell_type":"code","source":"params = {\n    \"n_estimators\": 500,\n    \"max_depth\": 4,\n    \"min_samples_split\": 5,\n    \"learning_rate\": 0.01,\n    \"loss\": \"squared_error\",\n}\n\nreg = GradientBoostingRegressor(**params)\nreg.fit(X_train, y_train)\nmean_squared_error(reg.predict(X_valid), Y_valid), mean_absolute_error(reg.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:10.502636Z","iopub.execute_input":"2022-07-11T03:33:10.503012Z","iopub.status.idle":"2022-07-11T03:33:21.898419Z","shell.execute_reply.started":"2022-07-11T03:33:10.502971Z","shell.execute_reply":"2022-07-11T03:33:21.897248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**XGB Regressor**","metadata":{}},{"cell_type":"code","source":"xgb = XGBRegressor()\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\nscores = cross_val_score(xgb, X, Y, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\nscores = absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (scores.mean(), scores.std()) )","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:33:21.899761Z","iopub.execute_input":"2022-07-11T03:33:21.900096Z","iopub.status.idle":"2022-07-11T03:34:00.024652Z","shell.execute_reply.started":"2022-07-11T03:33:21.900065Z","shell.execute_reply":"2022-07-11T03:34:00.023051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LGBM Regressor**","metadata":{}},{"cell_type":"code","source":"params = {'num_leaves': 30,\n         'min_data_in_leaf': 20,\n         'objective': 'regression',\n         'max_depth': 5,\n         'learning_rate': 0.01,\n         \"boosting\": \"gbdt\",\n         \"feature_fraction\": 0.9,\n         \"bagging_freq\": 1,\n         \"bagging_fraction\": 0.9,\n         \"bagging_seed\": 11,\n         \"metric\": 'rmse',\n         \"lambda_l1\": 0.2,\n         \"verbosity\": -1}\n\nlgb = LGBMRegressor(**params, n_estimators = 20000, nthread = 4, n_jobs = -1)\nlgb.fit(X_train, Y_train)\nmean_squared_error(lgb.predict(X_valid), Y_valid), mean_absolute_error(lgb.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:34:00.026656Z","iopub.execute_input":"2022-07-11T03:34:00.027513Z","iopub.status.idle":"2022-07-11T03:34:31.926124Z","shell.execute_reply.started":"2022-07-11T03:34:00.027463Z","shell.execute_reply":"2022-07-11T03:34:31.925211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eli5.show_weights(lgb, feature_filter=lambda x: x != '<BIAS>')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:34:31.927339Z","iopub.execute_input":"2022-07-11T03:34:31.927827Z","iopub.status.idle":"2022-07-11T03:34:31.953502Z","shell.execute_reply.started":"2022-07-11T03:34:31.927799Z","shell.execute_reply":"2022-07-11T03:34:31.952565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RandomForest and LGBMRegressor perform best as per there mean absolute error","metadata":{}},{"cell_type":"markdown","source":"**Regression neural network**","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Dense(198, input_shape = (198, ), kernel_initializer = 'normal', activation='elu'))\nmodel.add(Normalization())\nmodel.add(Dense(396, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Normalization())\nmodel.add(Dense(1028, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Dense(396, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Dense(1, kernel_initializer='normal'))\n\nmodel.compile(loss='mean_absolute_error', optimizer='adam')\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T04:04:16.977970Z","iopub.execute_input":"2022-07-11T04:04:16.978358Z","iopub.status.idle":"2022-07-11T04:04:17.076464Z","shell.execute_reply.started":"2022-07-11T04:04:16.978328Z","shell.execute_reply":"2022-07-11T04:04:17.075221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X, Y, batch_size=20, epochs = 40, validation_split=0.1)\nhistory","metadata":{"execution":{"iopub.status.busy":"2022-07-11T04:06:06.016998Z","iopub.execute_input":"2022-07-11T04:06:06.017831Z","iopub.status.idle":"2022-07-11T04:07:14.728630Z","shell.execute_reply.started":"2022-07-11T04:06:06.017780Z","shell.execute_reply":"2022-07-11T04:07:14.727155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_loss(history):\n  plt.figure(figsize=(14, 6))\n  plt.plot(history.history['loss'], label='loss')\n  plt.plot(history.history['val_loss'], label='val_loss')\n  plt.ylim([0, 10])\n  plt.xlabel('Epoch')\n  plt.ylabel('Error [MPG]')\n  plt.legend()\n  plt.grid(True)\n\nplot_loss(history)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T04:11:08.984929Z","iopub.execute_input":"2022-07-11T04:11:08.985673Z","iopub.status.idle":"2022-07-11T04:11:09.214295Z","shell.execute_reply.started":"2022-07-11T04:11:08.985619Z","shell.execute_reply":"2022-07-11T04:11:09.213094Z"},"trusted":true},"execution_count":null,"outputs":[]}]}