{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# We can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when we create a version using \"Save & Run All\" \n# We can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T10:33:54.887371Z","iopub.execute_input":"2022-07-24T10:33:54.887871Z","iopub.status.idle":"2022-07-24T10:33:54.903306Z","shell.execute_reply.started":"2022-07-24T10:33:54.887820Z","shell.execute_reply":"2022-07-24T10:33:54.902125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**Importing modules**","metadata":{}},{"cell_type":"code","source":"# !pip3 install scikeras==0.9.\n!pip3 install scikeras","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:33:54.906040Z","iopub.execute_input":"2022-07-24T10:33:54.906872Z","iopub.status.idle":"2022-07-24T10:36:24.573189Z","shell.execute_reply.started":"2022-07-24T10:33:54.906823Z","shell.execute_reply":"2022-07-24T10:36:24.571379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install --upgrade tensorflow ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:36:24.575356Z","iopub.execute_input":"2022-07-24T10:36:24.576392Z","iopub.status.idle":"2022-07-24T10:41:01.830168Z","shell.execute_reply.started":"2022-07-24T10:36:24.576345Z","shell.execute_reply":"2022-07-24T10:41:01.828893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport ast\nimport eli5\nimport lightgbm as lgb\n\npd.set_option('display.max_columns', None)\nfrom collections import Counter\nfrom wordcloud import WordCloud\n%matplotlib inline\nplt.style.use('ggplot')\nfrom sklearn.feature_extraction.text import TfidfVectorizer \nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.metrics import accuracy_score, mean_squared_error, mean_absolute_error\nfrom sklearn.model_selection import train_test_split, KFold\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import AdaBoostRegressor, GradientBoostingRegressor\nimport xgboost\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import RepeatedKFold\nfrom numpy import absolute\nfrom lightgbm import LGBMRegressor\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Normalization\n# from scikeras.wrappers import KerasRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\n\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:01.832912Z","iopub.execute_input":"2022-07-24T10:41:01.833313Z","iopub.status.idle":"2022-07-24T10:41:01.910382Z","shell.execute_reply.started":"2022-07-24T10:41:01.833275Z","shell.execute_reply":"2022-07-24T10:41:01.909129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Read train and test data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/movie-collectiondata/box_office_train.csv\")\ntest = pd.read_csv(\"../input/movie-collectiondata/box_office_test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:01.911840Z","iopub.execute_input":"2022-07-24T10:41:01.912216Z","iopub.status.idle":"2022-07-24T10:41:02.289077Z","shell.execute_reply.started":"2022-07-24T10:41:01.912182Z","shell.execute_reply":"2022-07-24T10:41:02.287582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train data : \", train.shape)\nprint(\"Shape of test data : \", test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:02.291009Z","iopub.execute_input":"2022-07-24T10:41:02.291485Z","iopub.status.idle":"2022-07-24T10:41:02.298323Z","shell.execute_reply.started":"2022-07-24T10:41:02.291439Z","shell.execute_reply":"2022-07-24T10:41:02.297003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:02.300098Z","iopub.execute_input":"2022-07-24T10:41:02.300558Z","iopub.status.idle":"2022-07-24T10:41:02.334983Z","shell.execute_reply.started":"2022-07-24T10:41:02.300515Z","shell.execute_reply":"2022-07-24T10:41:02.334062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have given data in string format for accessing value by key it's important to convert string into dictionary","metadata":{}},{"cell_type":"code","source":"# Convert string columns to dictionary\njson_columns = ['belongs_to_collection', 'genres', 'production_companies', 'production_countries', 'spoken_languages', 'Keywords', 'cast', 'crew']\n\ndef text_to_dict(df, json_columns = json_columns):\n    for col in json_columns:\n        df[col] = df[col].apply(lambda x: {} if pd.isna(x) else ast.literal_eval(x))\n    return df\n\ntrain = text_to_dict(train)\ntest = text_to_dict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:02.335902Z","iopub.execute_input":"2022-07-24T10:41:02.336251Z","iopub.status.idle":"2022-07-24T10:41:08.295281Z","shell.execute_reply.started":"2022-07-24T10:41:02.336221Z","shell.execute_reply":"2022-07-24T10:41:08.294141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**belongs_to_collection**","metadata":{}},{"cell_type":"code","source":"train['belongs_to_collection'][:3]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.300547Z","iopub.execute_input":"2022-07-24T10:41:08.301001Z","iopub.status.idle":"2022-07-24T10:41:08.311021Z","shell.execute_reply.started":"2022-07-24T10:41:08.300964Z","shell.execute_reply":"2022-07-24T10:41:08.309608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['belongs_to_collection'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.313253Z","iopub.execute_input":"2022-07-24T10:41:08.313616Z","iopub.status.idle":"2022-07-24T10:41:08.337126Z","shell.execute_reply.started":"2022-07-24T10:41:08.313577Z","shell.execute_reply":"2022-07-24T10:41:08.336118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['collection'] = train['belongs_to_collection'].apply(lambda x: x[0]['name'] if x!={} else 0)\ntrain['has_collection'] = train['collection'].apply(lambda x: 1 if x!=0 else 0)\n\ntest['collection'] = test['belongs_to_collection'].apply(lambda x: x[0]['name'] if x!={} else 0)\ntest['has_collection'] = test['collection'].apply(lambda x: 1 if x!=0 else 0)\n\ntrain.drop(['belongs_to_collection'], axis=1, inplace=True)\ntest.drop(['belongs_to_collection'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.338728Z","iopub.execute_input":"2022-07-24T10:41:08.339728Z","iopub.status.idle":"2022-07-24T10:41:08.362373Z","shell.execute_reply.started":"2022-07-24T10:41:08.339683Z","shell.execute_reply":"2022-07-24T10:41:08.361157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Genres**","metadata":{}},{"cell_type":"code","source":"train['genres'][:3]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.363910Z","iopub.execute_input":"2022-07-24T10:41:08.364318Z","iopub.status.idle":"2022-07-24T10:41:08.374435Z","shell.execute_reply.started":"2022-07-24T10:41:08.364285Z","shell.execute_reply":"2022-07-24T10:41:08.373249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['genres'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.375754Z","iopub.execute_input":"2022-07-24T10:41:08.376096Z","iopub.status.idle":"2022-07-24T10:41:08.395700Z","shell.execute_reply.started":"2022-07-24T10:41:08.376060Z","shell.execute_reply":"2022-07-24T10:41:08.394449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"most of movie have 1 to 5 genre but there is also exist some movie that have more then 5 genre so may be thay can be outlier.","metadata":{}},{"cell_type":"code","source":"train['num_genres'] = train['genres'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_genres = list(train['genres'].apply(lambda x: [j['name'] for j in x] if x!={} else []).values)\ntrain['all_genres'] = train['genres'].apply(lambda x: ' '.join([j['name'] for j in x]) if x!={} else \"\" )\n\ntop_genres = [m[0] for m in Counter([i for j in list_of_genres for i in j]).most_common(15)]\nfor g in top_genres:\n    train['genres_'+g] = train['all_genres'].apply(lambda x: 1 if g in x else 0)\n\n\ntest['num_genres'] = test['genres'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_genres'] = test['genres'].apply(lambda x: ' '.join([j['name'] for j in x]) if x!={} else \"\" )\n\nfor g in top_genres:\n    test['genres_'+g] = test['all_genres'].apply(lambda x: 1 if g in x else 0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.397492Z","iopub.execute_input":"2022-07-24T10:41:08.398792Z","iopub.status.idle":"2022-07-24T10:41:08.480320Z","shell.execute_reply.started":"2022-07-24T10:41:08.398749Z","shell.execute_reply":"2022-07-24T10:41:08.479020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we made special category for most common 15 genre and assign 1 if movie belongs to that genre else 0","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\ntext = ' '.join(i for j in list_of_genres for i in j)\nimg = WordCloud(max_font_size=None, background_color='white', collocations=False,\n                      width=1200, height=1000).generate(text)\nplt.title('Top genres wordcloud')\nplt.axis(\"off\")\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:08.482424Z","iopub.execute_input":"2022-07-24T10:41:08.482822Z","iopub.status.idle":"2022-07-24T10:41:09.617968Z","shell.execute_reply.started":"2022-07-24T10:41:08.482785Z","shell.execute_reply":"2022-07-24T10:41:09.616659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on wordcloud we can say that most of the movie belongs to Drama, comedy and thriller genre.","metadata":{}},{"cell_type":"markdown","source":"**production_companies**","metadata":{}},{"cell_type":"markdown","source":"Number of production companies involved for making one movie","metadata":{}},{"cell_type":"code","source":"train['production_companies'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:09.620060Z","iopub.execute_input":"2022-07-24T10:41:09.620554Z","iopub.status.idle":"2022-07-24T10:41:09.634440Z","shell.execute_reply.started":"2022-07-24T10:41:09.620505Z","shell.execute_reply":"2022-07-24T10:41:09.633120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['production_companies'].head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:09.636065Z","iopub.execute_input":"2022-07-24T10:41:09.636445Z","iopub.status.idle":"2022-07-24T10:41:09.649831Z","shell.execute_reply.started":"2022-07-24T10:41:09.636413Z","shell.execute_reply":"2022-07-24T10:41:09.648993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_production_companies'] = train['production_companies'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_pc = train['production_companies'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_production_companies'] = train['production_companies'].apply(lambda x : ' '.join( i['name'] for i in x ) if x!={} else \"\" )\n\ntop_pc = [m[0] for m in Counter(i for j in list_of_pc for i in j).most_common(15)]\nfor g in top_pc:\n    train['production_companies_' + g] = train['all_production_companies'].apply(lambda x:1 if g in x else 0)\n\n\n\ntest['num_production_companies'] = test['production_companies'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_production_companies'] = test['production_companies'].apply(lambda x : ' '.join( i['name'] for i in x ) if x!={} else \"\" )\n\nfor g in top_pc:\n    test['production_companies_' + g] = test['all_production_companies'].apply(lambda x:1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:09.651405Z","iopub.execute_input":"2022-07-24T10:41:09.651796Z","iopub.status.idle":"2022-07-24T10:41:09.738406Z","shell.execute_reply.started":"2022-07-24T10:41:09.651755Z","shell.execute_reply":"2022-07-24T10:41:09.737291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\ntext = ' '.join(i for j in list_of_pc for i in j)\nimg = WordCloud(max_font_size=None, background_color='white', collocations=False,\n                      width=1200, height=1000).generate(text)\nplt.axis('off')\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:09.739797Z","iopub.execute_input":"2022-07-24T10:41:09.740168Z","iopub.status.idle":"2022-07-24T10:41:12.650222Z","shell.execute_reply.started":"2022-07-24T10:41:09.740136Z","shell.execute_reply":"2022-07-24T10:41:12.649067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Production Countries**","metadata":{}},{"cell_type":"code","source":"train['production_countries'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.651614Z","iopub.execute_input":"2022-07-24T10:41:12.652234Z","iopub.status.idle":"2022-07-24T10:41:12.662668Z","shell.execute_reply.started":"2022-07-24T10:41:12.652194Z","shell.execute_reply":"2022-07-24T10:41:12.661206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['production_countries'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.664695Z","iopub.execute_input":"2022-07-24T10:41:12.665815Z","iopub.status.idle":"2022-07-24T10:41:12.683587Z","shell.execute_reply.started":"2022-07-24T10:41:12.665762Z","shell.execute_reply":"2022-07-24T10:41:12.682153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In data there exist some movie that does not belongs to any country so we can say those are outliers.","metadata":{}},{"cell_type":"code","source":"train['num_production_countries'] = train['production_countries'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_prod_countries = train['production_countries'].apply(lambda x: [i['name'] for i in x]  if x!={} else []).values\ntrain['all_production_countries'] = train['production_countries'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\ntop_production_countries = [m[0] for m in Counter([i for j in list_of_prod_countries for i in j]).most_common(15)]\n\nfor g in top_production_countries:\n    train['production_countries_' + g] = train['all_production_countries'].apply(lambda x: 1 if g in x else 0)\n    \n\ntest['num_production_countries'] = test['production_countries'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_production_countries'] = test['production_countries'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_production_countries:\n    test['production_countries_' + g] = test['all_production_countries'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.686273Z","iopub.execute_input":"2022-07-24T10:41:12.686760Z","iopub.status.idle":"2022-07-24T10:41:12.763934Z","shell.execute_reply.started":"2022-07-24T10:41:12.686712Z","shell.execute_reply":"2022-07-24T10:41:12.762790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in Counter([i for j in list_of_prod_countries for i in j]).most_common(10):\n    print(i,(30-len(i))*\" \", j/len(list_of_prod_countries)*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.767164Z","iopub.execute_input":"2022-07-24T10:41:12.768428Z","iopub.status.idle":"2022-07-24T10:41:12.776571Z","shell.execute_reply.started":"2022-07-24T10:41:12.768386Z","shell.execute_reply":"2022-07-24T10:41:12.775637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Spoken language**","metadata":{}},{"cell_type":"code","source":"train['spoken_languages'].apply(lambda x: len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.784215Z","iopub.execute_input":"2022-07-24T10:41:12.785331Z","iopub.status.idle":"2022-07-24T10:41:12.799147Z","shell.execute_reply.started":"2022-07-24T10:41:12.785282Z","shell.execute_reply":"2022-07-24T10:41:12.797904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_spoken_languages'] = train['spoken_languages'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_spoken_languages = train['spoken_languages'].apply(lambda x: [i['name'] for i in x] if x!={} else \"\").values\ntrain['all_spoken_languages'] = train['spoken_languages'].apply(lambda x: [' '.join(i['name'] for i in x)] if x!={} else [] )\ntop_spoken_languages = [m[0] for m in Counter(i for j in list_of_spoken_languages for i in j).most_common(15)]\n\nfor g in top_spoken_languages:\n    train['spoken_language_'+g] = train['all_spoken_languages'].apply(lambda x: 1 if g in x else 0)\n    \ntest['num_spoken_languages'] = test['spoken_languages'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_spoken_languages'] = test['spoken_languages'].apply(lambda x: [' '.join(i['name'] for i in x)] if x!={} else [] )\n\nfor g in top_spoken_languages:\n    test['spoken_language_'+g] = test['all_spoken_languages'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.800666Z","iopub.execute_input":"2022-07-24T10:41:12.801501Z","iopub.status.idle":"2022-07-24T10:41:12.882965Z","shell.execute_reply.started":"2022-07-24T10:41:12.801462Z","shell.execute_reply":"2022-07-24T10:41:12.881893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in Counter([i for j in list_of_spoken_languages for i in j]).most_common(10):\n    print(i, (70-len(i))*\" \", round(j*100/len(list_of_spoken_languages), 2))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.884800Z","iopub.execute_input":"2022-07-24T10:41:12.885200Z","iopub.status.idle":"2022-07-24T10:41:12.893735Z","shell.execute_reply.started":"2022-07-24T10:41:12.885162Z","shell.execute_reply":"2022-07-24T10:41:12.892426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Keywords**","metadata":{}},{"cell_type":"code","source":"train['Keywords'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.895348Z","iopub.execute_input":"2022-07-24T10:41:12.896398Z","iopub.status.idle":"2022-07-24T10:41:12.912656Z","shell.execute_reply.started":"2022-07-24T10:41:12.896359Z","shell.execute_reply":"2022-07-24T10:41:12.911306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Keywords'].apply(lambda x: len(x) if x!={} else 0).value_counts().head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.914255Z","iopub.execute_input":"2022-07-24T10:41:12.914625Z","iopub.status.idle":"2022-07-24T10:41:12.930548Z","shell.execute_reply.started":"2022-07-24T10:41:12.914581Z","shell.execute_reply":"2022-07-24T10:41:12.928827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_Keywords'] = train['Keywords'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_Keywords = train['Keywords'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_Keywords'] = train['Keywords'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\ntop_Keywords = [m[0] for m in Counter(i for j in list_of_Keywords for i in j).most_common(30)]\n\nfor g in top_Keywords:\n    train['Keywords_'+g] = train['all_Keywords'].apply(lambda x: 1 if g in x else 0)\n    \n    \ntest['num_Keywords'] = test['Keywords'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_Keywords'] = test['Keywords'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_Keywords:\n    test['Keywords_'+g] = test['all_Keywords'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:12.932792Z","iopub.execute_input":"2022-07-24T10:41:12.933797Z","iopub.status.idle":"2022-07-24T10:41:13.082206Z","shell.execute_reply.started":"2022-07-24T10:41:12.933744Z","shell.execute_reply":"2022-07-24T10:41:13.080949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in Counter([i for j in list_of_Keywords for i in j]).most_common(20):\n    print(i, (50 - len(i))*\" \", j)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.084239Z","iopub.execute_input":"2022-07-24T10:41:13.084629Z","iopub.status.idle":"2022-07-24T10:41:13.098807Z","shell.execute_reply.started":"2022-07-24T10:41:13.084588Z","shell.execute_reply":"2022-07-24T10:41:13.097325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most occuring keywords are woman director, independent film, duringcreditsstinger, murder and based on novel","metadata":{}},{"cell_type":"markdown","source":"**Cast**","metadata":{}},{"cell_type":"code","source":"train['cast'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.100490Z","iopub.execute_input":"2022-07-24T10:41:13.100908Z","iopub.status.idle":"2022-07-24T10:41:13.124267Z","shell.execute_reply.started":"2022-07-24T10:41:13.100866Z","shell.execute_reply":"2022-07-24T10:41:13.123224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cast'].apply(lambda x:len(x) if x!={} else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.125770Z","iopub.execute_input":"2022-07-24T10:41:13.126135Z","iopub.status.idle":"2022-07-24T10:41:13.142335Z","shell.execute_reply.started":"2022-07-24T10:41:13.126102Z","shell.execute_reply":"2022-07-24T10:41:13.141190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_cast'] = train['cast'].apply(lambda x: len(x) if x!={} else 0)\nlist_of_cast = train['cast'].apply(lambda x: [i['name'] for i in x] if x!={} else []).values\ntrain['all_cast'] = train['cast'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\ntop_cast = [m[0] for m in Counter([i for j in list_of_cast for i in j]).most_common(20)]\n\nfor g in top_cast:\n    train['cast_name_'+g] = train['all_cast'].apply(lambda x: 1 if g in x else 0)\n    \n\ntest['num_cast'] = test['cast'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_cast'] = test['cast'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else \"\")\n\nfor g in top_cast:\n    test['cast_name_'+g] = test['all_cast'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.143832Z","iopub.execute_input":"2022-07-24T10:41:13.144986Z","iopub.status.idle":"2022-07-24T10:41:13.324317Z","shell.execute_reply.started":"2022-07-24T10:41:13.144946Z","shell.execute_reply":"2022-07-24T10:41:13.323120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most popular popular cast and number of movies they are in","metadata":{}},{"cell_type":"code","source":"for i, j in Counter([i for j in list_of_cast for i in j]).most_common(15):\n    print(i, (50-len(i))*\" \", j)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.326216Z","iopub.execute_input":"2022-07-24T10:41:13.326575Z","iopub.status.idle":"2022-07-24T10:41:13.363297Z","shell.execute_reply.started":"2022-07-24T10:41:13.326545Z","shell.execute_reply":"2022-07-24T10:41:13.362050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_of_gender = train['cast'].apply(lambda x: [i['gender'] for i in x] if x!={} else [])\nCounter([i for j in list_of_gender for i in j]).most_common()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.365971Z","iopub.execute_input":"2022-07-24T10:41:13.367147Z","iopub.status.idle":"2022-07-24T10:41:13.399965Z","shell.execute_reply.started":"2022-07-24T10:41:13.367107Z","shell.execute_reply":"2022-07-24T10:41:13.399002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['gender_0_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 0]))\ntrain['gender_1_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 1]))\ntrain['gender_2_cast'] = train['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 2]))\n\ntest['gender_0_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 0]))\ntest['gender_1_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 1]))\ntest['gender_2_cast'] = test['cast'].apply(lambda x: sum([1 for i in x if i['gender'] == 2]))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.401704Z","iopub.execute_input":"2022-07-24T10:41:13.402430Z","iopub.status.idle":"2022-07-24T10:41:13.473230Z","shell.execute_reply.started":"2022-07-24T10:41:13.402394Z","shell.execute_reply":"2022-07-24T10:41:13.472170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we just assigned number of gender in total cast for particular one movie like number of male for x movie from all x cast","metadata":{}},{"cell_type":"code","source":"list_of_characters = train['cast'].apply(lambda x: [i['character'] for i in x] if x!={} else [])\nCounter(i for j in list_of_characters for i in j).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.474585Z","iopub.execute_input":"2022-07-24T10:41:13.475610Z","iopub.status.idle":"2022-07-24T10:41:13.531353Z","shell.execute_reply.started":"2022-07-24T10:41:13.475572Z","shell.execute_reply":"2022-07-24T10:41:13.529957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For cast character there are lot of people who play himself/ herself in movie","metadata":{}},{"cell_type":"code","source":"train['num_cast_characters'] = train['cast'].apply(lambda x: len(x) if x!={} else 0)\ntrain['all_cast_character'] = train['cast'].apply(lambda x: \" \".join(i['character'] for i in x) if x!={} else \"\")\ntop_cast_characters = [m[0] for m in Counter(i for j in list_of_characters for i in j).most_common(20)]\nfor g in top_cast_characters:\n    train['cast_character_'+g] = train['all_cast_character'].apply(lambda x: 1 if g in x else 0)\n\ntest['num_cast_characters'] = test['cast'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_cast_character'] = test['cast'].apply(lambda x: \" \".join(i['character'] for i in x) if x!={} else \"\")\nfor g in top_cast_characters:\n    test['cast_character_'+g] = test['all_cast_character'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.533471Z","iopub.execute_input":"2022-07-24T10:41:13.533956Z","iopub.status.idle":"2022-07-24T10:41:13.689319Z","shell.execute_reply.started":"2022-07-24T10:41:13.533892Z","shell.execute_reply":"2022-07-24T10:41:13.687471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Crew**","metadata":{}},{"cell_type":"code","source":"train['crew'].apply(lambda x: len(x) if x!={} else 0).value_counts().head(10), train['crew'].apply(lambda x: len(x) if x!={} else 0).value_counts().tail(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.691511Z","iopub.execute_input":"2022-07-24T10:41:13.692164Z","iopub.status.idle":"2022-07-24T10:41:13.713234Z","shell.execute_reply.started":"2022-07-24T10:41:13.692110Z","shell.execute_reply":"2022-07-24T10:41:13.711981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Usually there is 2 to 15 crew but there are also exist some data that have more then 100 crew","metadata":{}},{"cell_type":"code","source":"list_of_crew_names = train['crew'].apply(lambda x: [i['name'] for i in x] if x!={} else [])\nCounter([i for j in list_of_crew_names for i in j]).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.714626Z","iopub.execute_input":"2022-07-24T10:41:13.715318Z","iopub.status.idle":"2022-07-24T10:41:13.797603Z","shell.execute_reply.started":"2022-07-24T10:41:13.715272Z","shell.execute_reply":"2022-07-24T10:41:13.796438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['num_crew'] = train['crew'].apply(lambda x: len(x) if x!={} else 0)\ntrain['all_crew_name'] = train['crew'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else [])\ntop_crew_names = [m[0] for m in Counter([i for j in list_of_crew_names for i in j]).most_common(15)]\n\nfor g in top_crew_names:\n    train['crew_name_' + g] = train['all_crew_name'].apply(lambda x: 1 if g in x else 0)\n\n\ntest['num_crew'] = test['crew'].apply(lambda x: len(x) if x!={} else 0)\ntest['all_crew_name'] = test['crew'].apply(lambda x: ' '.join(i['name'] for i in x) if x!={} else [])\n\nfor g in top_crew_names:\n    test['crew_name_' + g] = test['all_crew_name'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.799146Z","iopub.execute_input":"2022-07-24T10:41:13.799492Z","iopub.status.idle":"2022-07-24T10:41:13.956404Z","shell.execute_reply.started":"2022-07-24T10:41:13.799462Z","shell.execute_reply":"2022-07-24T10:41:13.954717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Genders for crew**","metadata":{}},{"cell_type":"code","source":"list_of_gender = train['crew'].apply(lambda x: [i['gender'] for i in x] if x!={} else [])\nCounter(i for j in list_of_gender for i in j).most_common(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.958341Z","iopub.execute_input":"2022-07-24T10:41:13.958734Z","iopub.status.idle":"2022-07-24T10:41:13.994054Z","shell.execute_reply.started":"2022-07-24T10:41:13.958699Z","shell.execute_reply":"2022-07-24T10:41:13.992610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['crew_gender_0'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntrain['crew_gender_1'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntrain['crew_gender_2'] = train['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 2))\n\ntest['crew_gender_0'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntest['crew_gender_1'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 0))\ntest['crew_gender_2'] = test['cast'].apply(lambda x: sum(1 for i in x if i['gender'] == 2))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:13.995857Z","iopub.execute_input":"2022-07-24T10:41:13.996645Z","iopub.status.idle":"2022-07-24T10:41:14.071712Z","shell.execute_reply.started":"2022-07-24T10:41:13.996591Z","shell.execute_reply":"2022-07-24T10:41:14.070413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we just assigned number of gender in total crew for particular one movie like number of male for x movie from all x cast","metadata":{}},{"cell_type":"markdown","source":"Job of crew and number of jobs","metadata":{}},{"cell_type":"code","source":"list_of_jobs = train['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\nCounter([i for j in list_of_jobs for i in j]).most_common(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:14.073599Z","iopub.execute_input":"2022-07-24T10:41:14.074014Z","iopub.status.idle":"2022-07-24T10:41:14.135835Z","shell.execute_reply.started":"2022-07-24T10:41:14.073976Z","shell.execute_reply":"2022-07-24T10:41:14.134488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['all_job'] = train['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\ntop_job = [m[0] for m in Counter([i for j in list_of_jobs for i in j]).most_common(20)]\n\nfor g in top_job:\n    train['job_'+g] = train['all_job'].apply(lambda x: 1 if g in x else 0)\n\ntest['all_job'] = test['crew'].apply(lambda x: [i['job'] for i in x] if x!={} else [])\n\nfor g in top_job:\n    test['job_'+g] = test['all_job'].apply(lambda x: 1 if g in x else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:14.138254Z","iopub.execute_input":"2022-07-24T10:41:14.139152Z","iopub.status.idle":"2022-07-24T10:41:14.665825Z","shell.execute_reply.started":"2022-07-24T10:41:14.139108Z","shell.execute_reply":"2022-07-24T10:41:14.664492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Visualization**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(16, 6))\n\nplt.subplot(1, 2, 1)\nplt.hist(train['revenue'])\nplt.title(\"Revenue distribution\")\n\nplt.subplot(1, 2, 2)\nplt.hist(np.log1p(train['revenue']))\nplt.title(\"Log1 revenue distribution\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:14.667527Z","iopub.execute_input":"2022-07-24T10:41:14.668259Z","iopub.status.idle":"2022-07-24T10:41:15.078437Z","shell.execute_reply.started":"2022-07-24T10:41:14.668197Z","shell.execute_reply":"2022-07-24T10:41:15.077100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When we apply log on revenue then it becomes less skewed and it will be better for our model","metadata":{}},{"cell_type":"code","source":"train['log_revenue'] = np.log1p(train['revenue'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.080481Z","iopub.execute_input":"2022-07-24T10:41:15.080851Z","iopub.status.idle":"2022-07-24T10:41:15.088149Z","shell.execute_reply.started":"2022-07-24T10:41:15.080819Z","shell.execute_reply":"2022-07-24T10:41:15.086836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['budget'].skew(), np.log1p(train['budget']).skew()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.089828Z","iopub.execute_input":"2022-07-24T10:41:15.090192Z","iopub.status.idle":"2022-07-24T10:41:15.103205Z","shell.execute_reply.started":"2022-07-24T10:41:15.090163Z","shell.execute_reply":"2022-07-24T10:41:15.101959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(16, 6))\n\nplt.subplot(1,2,1)\nplt.hist(train['budget'])\nplt.title(\"Budget distribution\")\n\nplt.subplot(1,2,2)\nplt.hist(np.log1p(train['budget']))\nplt.title(\"Log1p Budget distribution\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.104747Z","iopub.execute_input":"2022-07-24T10:41:15.105270Z","iopub.status.idle":"2022-07-24T10:41:15.487703Z","shell.execute_reply.started":"2022-07-24T10:41:15.105233Z","shell.execute_reply":"2022-07-24T10:41:15.486324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\nplt.subplot(1, 2, 1)\nsns.scatterplot(x = np.log(train['budget']), y=train['log_revenue'])\nplt.xlabel('log_budget')\nplt.title(\"log Revenue vs log budget\")\n\nplt.subplot(1,2,2)\nsns.scatterplot(x='budget', y='log_revenue', data=train)\nplt.title(\"log revenue vs budget\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.489261Z","iopub.execute_input":"2022-07-24T10:41:15.489600Z","iopub.status.idle":"2022-07-24T10:41:15.876631Z","shell.execute_reply.started":"2022-07-24T10:41:15.489570Z","shell.execute_reply":"2022-07-24T10:41:15.875653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scatter plot of log_revenue vs log_budget is very good as compared to log_revenue vs budget","metadata":{}},{"cell_type":"code","source":"train['log_budget'] = np.log1p(train['budget'])\ntest['log_budget'] = np.log1p(test['budget'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.878573Z","iopub.execute_input":"2022-07-24T10:41:15.879267Z","iopub.status.idle":"2022-07-24T10:41:15.887027Z","shell.execute_reply.started":"2022-07-24T10:41:15.879226Z","shell.execute_reply":"2022-07-24T10:41:15.885636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['homepage'].apply(lambda x: 1 if x!=np.nan else 0).value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.888544Z","iopub.execute_input":"2022-07-24T10:41:15.889576Z","iopub.status.idle":"2022-07-24T10:41:15.908950Z","shell.execute_reply.started":"2022-07-24T10:41:15.889540Z","shell.execute_reply":"2022-07-24T10:41:15.907713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(1,2,figsize=(16, 6))\nplt.subplot(1,2,1)\nsns.boxenplot(x='original_language', y='revenue', data= train.loc[train['original_language'].isin(train['original_language'].value_counts().head(10).index)] )\nplt.title(\"Revenue vs Original language\")\n\nplt.subplot(1,2,2)\nsns.boxenplot(x='original_language', y='log_revenue', data= train.loc[train['original_language'].isin(train['original_language'].value_counts().head(10).index)] )\nplt.title(\"Log Revenue vs Original language\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:15.910529Z","iopub.execute_input":"2022-07-24T10:41:15.911905Z","iopub.status.idle":"2022-07-24T10:41:16.469664Z","shell.execute_reply.started":"2022-07-24T10:41:15.911865Z","shell.execute_reply":"2022-07-24T10:41:16.468505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on previous three graph we can completely say that log_revenue is really good choice for our model as compared to revenue","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(9, 9))\ntitle = ' '.join(train['original_title'].values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(title)\nplt.title(\"Original title wordcloud\")\nplt.imshow(img)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:16.471161Z","iopub.execute_input":"2022-07-24T10:41:16.471515Z","iopub.status.idle":"2022-07-24T10:41:19.209105Z","shell.execute_reply.started":"2022-07-24T10:41:16.471482Z","shell.execute_reply":"2022-07-24T10:41:19.207996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most occuring words in movie titles are Man and Last.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(9,9))\nOverview = ' '.join(train['overview'].fillna(' ').values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(Overview)\nplt.imshow(img)\nplt.title(\"Overview wordcloud\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:19.210436Z","iopub.execute_input":"2022-07-24T10:41:19.211069Z","iopub.status.idle":"2022-07-24T10:41:22.444611Z","shell.execute_reply.started":"2022-07-24T10:41:19.211029Z","shell.execute_reply":"2022-07-24T10:41:22.443155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see impact of words on revenue","metadata":{}},{"cell_type":"code","source":"tfidf = TfidfVectorizer()\n\nx = tfidf.fit_transform(train['overview'].fillna(''))\ny = train['log_revenue']\n\nX_train, X_valid, y_train, y_valid = train_test_split(x, y, test_size=0.1)\n\nlr = LinearRegression()\nlr.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:22.446268Z","iopub.execute_input":"2022-07-24T10:41:22.446622Z","iopub.status.idle":"2022-07-24T10:41:22.745874Z","shell.execute_reply.started":"2022-07-24T10:41:22.446591Z","shell.execute_reply":"2022-07-24T10:41:22.744623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_squared_error(lr.predict(X_valid), y_valid), mean_absolute_error(lr.predict(X_valid), y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:22.747856Z","iopub.execute_input":"2022-07-24T10:41:22.748734Z","iopub.status.idle":"2022-07-24T10:41:22.763315Z","shell.execute_reply.started":"2022-07-24T10:41:22.748661Z","shell.execute_reply":"2022-07-24T10:41:22.762017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eli5.show_weights(lr, vec=tfidf, top=20, feature_filter=lambda x: x != '<BIAS>')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:22.768524Z","iopub.execute_input":"2022-07-24T10:41:22.773104Z","iopub.status.idle":"2022-07-24T10:41:22.871622Z","shell.execute_reply.started":"2022-07-24T10:41:22.773038Z","shell.execute_reply":"2022-07-24T10:41:22.870799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Target value : \", train['log_revenue'][500])\neli5.show_prediction(lr, doc=train['overview'].values[500], vec = tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:22.873084Z","iopub.execute_input":"2022-07-24T10:41:22.874322Z","iopub.status.idle":"2022-07-24T10:41:22.919966Z","shell.execute_reply.started":"2022-07-24T10:41:22.874281Z","shell.execute_reply":"2022-07-24T10:41:22.916183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Popularity**","metadata":{}},{"cell_type":"code","source":"plt.subplots(1,2,figsize=(16, 6))\n\nplt.subplot(1,2,1)\nplt.scatter(x='popularity', y='revenue', data=train)\nplt.title(\"Revenue vs Popularity\")\n\nplt.subplot(1,2,2)\nplt.scatter(x='popularity', y='log_revenue', data=train)\nplt.title(\"Log revenue vs popularity\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:22.921886Z","iopub.execute_input":"2022-07-24T10:41:22.922399Z","iopub.status.idle":"2022-07-24T10:41:23.319773Z","shell.execute_reply.started":"2022-07-24T10:41:22.922349Z","shell.execute_reply":"2022-07-24T10:41:23.318868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on scatter plot we can say revenue is not much dependent on popularity","metadata":{}},{"cell_type":"markdown","source":"**Release Date**","metadata":{}},{"cell_type":"code","source":"test[test.release_date.isnull()].index","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:23.321332Z","iopub.execute_input":"2022-07-24T10:41:23.321934Z","iopub.status.idle":"2022-07-24T10:41:23.334732Z","shell.execute_reply.started":"2022-07-24T10:41:23.321885Z","shell.execute_reply":"2022-07-24T10:41:23.333406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.at[828, 'release_date'] = '1/5/2000'","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:23.336269Z","iopub.execute_input":"2022-07-24T10:41:23.337330Z","iopub.status.idle":"2022-07-24T10:41:23.345829Z","shell.execute_reply.started":"2022-07-24T10:41:23.337283Z","shell.execute_reply":"2022-07-24T10:41:23.344961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['release_date_year'] = train['release_date'].apply(lambda x: 19*100 + int(x.split('/')[-1]) if int(x.split('/')[-1]) > 22 else 20*100 + int(x.split('/')[-1]))\ntest['release_date_year'] = test['release_date'].apply(lambda x: 19*100 + int(x.split('/')[-1]) if int(x.split('/')[-1]) > 22 else 20*100 + int(x.split('/')[-1]))\n\ntrain['release_date_month'] = pd.to_datetime(train['release_date']).dt.month\ntest['release_date_month'] = pd.to_datetime(test['release_date']).dt.month\n\ntrain['release_date_weekday'] = pd.to_datetime(train['release_date']).dt.weekday\ntest['release_date_weekday'] = pd.to_datetime(test['release_date']).dt.weekday\n\ntrain['release_date_weekofyear'] = pd.to_datetime(train['release_date']).dt.weekofyear\ntest['release_date_weekofyear'] = pd.to_datetime(test['release_date']).dt.weekofyear\n\ntrain['release_date_day'] = pd.to_datetime(train['release_date']).dt.day\ntest['release_date_day'] = pd.to_datetime(test['release_date']).dt.day","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:23.347565Z","iopub.execute_input":"2022-07-24T10:41:23.348159Z","iopub.status.idle":"2022-07-24T10:41:24.266954Z","shell.execute_reply.started":"2022-07-24T10:41:23.348124Z","shell.execute_reply":"2022-07-24T10:41:24.265741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = test['release_date_year'].value_counts().sort_index()\n\ndata = [go.Scatter(x = d1.index, y=d1.values, name='train'), go.Scatter(x = d2.index, y = d2.values, name='test')]\nlayout = go.Layout(dict(title=\"Number of movies per year\",\n                  xaxis = dict(title=\"year\"),\n                  yaxis = dict(title=\"Count\")),legend = dict(orientation='v'))\npy.iplot(dict(data = data, layout=layout))\n\n# d1 = train['release_date_year'].value_counts().sort_index()\n# d2 = test['release_date_year'].value_counts().sort_index()\n# data = [go.Scatter(x=d1.index, y=d1.values, name='train'), go.Scatter(x=d2.index, y=d2.values, name='test')]\n# layout = go.Layout(dict(title = \"Number of films per year\",\n#                   xaxis = dict(title = 'Year'),\n#                   yaxis = dict(title = 'Count'),\n#                   ),legend=dict(\n#                 orientation=\"v\"))\n# py.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.268786Z","iopub.execute_input":"2022-07-24T10:41:24.269445Z","iopub.status.idle":"2022-07-24T10:41:24.316846Z","shell.execute_reply.started":"2022-07-24T10:41:24.269394Z","shell.execute_reply":"2022-07-24T10:41:24.315658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = test['release_date_year'].value_counts().sort_index()\ndata = [go.Scatter(x=d1.index, y=d1.values, name='train'), go.Scatter(x=d2.index, y=d2.values, name='test')]\nlayout = go.Layout(dict(title = \"Number of films per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.318529Z","iopub.execute_input":"2022-07-24T10:41:24.319707Z","iopub.status.idle":"2022-07-24T10:41:24.364266Z","shell.execute_reply.started":"2022-07-24T10:41:24.319654Z","shell.execute_reply":"2022-07-24T10:41:24.362933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = train.groupby('release_date_year')['revenue'].sum()\n\ndata = [go.Scatter(x=d1.index, y=d1.values, name='film count'), go.Scatter(x=d2.index, y=d2.values, name='total revenue', yaxis='y2')]\nlayout = go.Layout(dict(title = \"Number of films and total revenue per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  yaxis2=dict(title='Total revenue', overlaying='y', side='right')\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.366313Z","iopub.execute_input":"2022-07-24T10:41:24.366812Z","iopub.status.idle":"2022-07-24T10:41:24.411222Z","shell.execute_reply.started":"2022-07-24T10:41:24.366765Z","shell.execute_reply":"2022-07-24T10:41:24.410280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = train['release_date_year'].value_counts().sort_index()\nd2 = train.groupby(['release_date_year'])['revenue'].mean()\ndata = [go.Scatter(x=d1.index, y=d1.values, name='film count'), go.Scatter(x=d2.index, y=d2.values, name='mean revenue', yaxis='y2')]\nlayout = go.Layout(dict(title = \"Number of films and average revenue per year\",\n                  xaxis = dict(title = 'Year'),\n                  yaxis = dict(title = 'Count'),\n                  yaxis2=dict(title='Average revenue', overlaying='y', side='right')\n                  ),legend=dict(\n                orientation=\"v\"))\npy.iplot(dict(data=data, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.412526Z","iopub.execute_input":"2022-07-24T10:41:24.413174Z","iopub.status.idle":"2022-07-24T10:41:24.457672Z","shell.execute_reply.started":"2022-07-24T10:41:24.413141Z","shell.execute_reply":"2022-07-24T10:41:24.456620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x='release_date_weekday', y='revenue', data=train);\nplt.title('Revenue on different days of week of release');","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.459268Z","iopub.execute_input":"2022-07-24T10:41:24.459588Z","iopub.status.idle":"2022-07-24T10:41:24.805002Z","shell.execute_reply.started":"2022-07-24T10:41:24.459560Z","shell.execute_reply":"2022-07-24T10:41:24.804021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Films releases on Wednesdays and on Thursdays tend to have a higher revenue.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(20, 6))\n\nplt.subplot(1,3,1)\nplt.hist(train['runtime'].fillna(0)/60, bins=20)\nplt.title(\"Distribution of runtime\")\n\nplt.subplot(1,3,2)\nplt.scatter(x = 'runtime', y='revenue', data=train)\nplt.title(\"Revenue vs runtime\")\n\nplt.subplot(1,3,3)\nplt.scatter(x = 'runtime', y='popularity', data=train)\nplt.title(\"Revenue vs popularity\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:24.816330Z","iopub.execute_input":"2022-07-24T10:41:24.817638Z","iopub.status.idle":"2022-07-24T10:41:25.335229Z","shell.execute_reply.started":"2022-07-24T10:41:24.817586Z","shell.execute_reply":"2022-07-24T10:41:25.333672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['status'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:25.336804Z","iopub.execute_input":"2022-07-24T10:41:25.337832Z","iopub.status.idle":"2022-07-24T10:41:25.347186Z","shell.execute_reply.started":"2022-07-24T10:41:25.337791Z","shell.execute_reply":"2022-07-24T10:41:25.345888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['status'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:25.348721Z","iopub.execute_input":"2022-07-24T10:41:25.349826Z","iopub.status.idle":"2022-07-24T10:41:25.365312Z","shell.execute_reply.started":"2022-07-24T10:41:25.349774Z","shell.execute_reply":"2022-07-24T10:41:25.364158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tagline**","metadata":{}},{"cell_type":"code","source":"text = ' '.join(train['tagline'].fillna(\"\").values)\nimg = WordCloud(width=1000, height=1000, background_color='white').generate(text)\nplt.figure(figsize=(9, 9))\nplt.imshow(img)\nplt.title(\"Tagline wordcloud\")\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:25.366722Z","iopub.execute_input":"2022-07-24T10:41:25.367457Z","iopub.status.idle":"2022-07-24T10:41:27.909408Z","shell.execute_reply.started":"2022-07-24T10:41:25.367421Z","shell.execute_reply":"2022-07-24T10:41:27.907809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(3, 5, figsize=(24, 12))\nplt.suptitle('Violinplot of revenue vs genres')\nfor i, e in enumerate([col for col in train.columns if 'genres_' in col]):\n    sns.violinplot(x=e, y='revenue', data=train, ax=axes[i // 5][i % 5]);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:27.911244Z","iopub.execute_input":"2022-07-24T10:41:27.911749Z","iopub.status.idle":"2022-07-24T10:41:29.994996Z","shell.execute_reply.started":"2022-07-24T10:41:27.911699Z","shell.execute_reply":"2022-07-24T10:41:29.993991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Family, adventure, fantasy and animation have high revenue","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(4, 5, figsize=(21, 18))\nfig.tight_layout()\nplt.subplots_adjust(wspace=None, hspace=None)\nplt.axis('off')\nplt.suptitle(\"Violin plot of revenue vs production companies\")\nfor i,e in enumerate([col for col in train.columns if 'production_companies_' in col]):\n    sns.violinplot(x = e, y='revenue', data = train, ax = ax[i // 5][i % 5])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:29.996442Z","iopub.execute_input":"2022-07-24T10:41:29.997039Z","iopub.status.idle":"2022-07-24T10:41:33.512000Z","shell.execute_reply.started":"2022-07-24T10:41:29.997003Z","shell.execute_reply":"2022-07-24T10:41:33.510849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(4, 5, figsize=(24, 24))\nplt.suptitle('Violinplot of revenue vs production country')\nfor i, e in enumerate([col for col in train.columns if 'production_countries_' in col]):\n    sns.violinplot(x=e, y='revenue', data=train, ax=axes[i // 5][i % 5]);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:33.513573Z","iopub.execute_input":"2022-07-24T10:41:33.513997Z","iopub.status.idle":"2022-07-24T10:41:36.366874Z","shell.execute_reply.started":"2022-07-24T10:41:33.513959Z","shell.execute_reply":"2022-07-24T10:41:36.365535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\n\nplt.subplot(1, 2, 1)\nplt.scatter(train['num_cast'], train['revenue'])\nplt.title('Number of cast members vs revenue');\nplt.xlabel(\"Number of cast\")\nplt.ylabel(\"Revenue\")\n\nplt.subplot(1, 2, 2)\nplt.scatter(train['num_cast'], train['log_revenue'])\nplt.title('Number of cast members vs log revenue');\nplt.xlabel(\"Number of cast\")\nplt.ylabel(\"Log Revenue\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:36.368857Z","iopub.execute_input":"2022-07-24T10:41:36.369961Z","iopub.status.idle":"2022-07-24T10:41:36.782530Z","shell.execute_reply.started":"2022-07-24T10:41:36.369892Z","shell.execute_reply":"2022-07-24T10:41:36.781152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"number of cast and log_revenue is positively correlated with eachother","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 4, ncols = 5, figsize=(24, 18))\n\nfor i,e in enumerate([col for col in train.columns if 'cast_name_' in col]):\n    sns.violinplot(x = e, y = 'revenue', data= train, ax = ax[i//5][i%5])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:36.784449Z","iopub.execute_input":"2022-07-24T10:41:36.785304Z","iopub.status.idle":"2022-07-24T10:41:39.819289Z","shell.execute_reply.started":"2022-07-24T10:41:36.785251Z","shell.execute_reply":"2022-07-24T10:41:39.818128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len([col for col in train.columns if 'cast_character_' in col])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:39.820858Z","iopub.execute_input":"2022-07-24T10:41:39.821244Z","iopub.status.idle":"2022-07-24T10:41:39.829541Z","shell.execute_reply.started":"2022-07-24T10:41:39.821210Z","shell.execute_reply":"2022-07-24T10:41:39.828017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(4, 5, figsize=(24, 18))\nfor i,e in enumerate([col for col in train.columns if 'cast_character_' in col]):\n    sns.violinplot(x = e, y = 'revenue', data=train, ax = ax[i//5][i%5])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:39.831078Z","iopub.execute_input":"2022-07-24T10:41:39.831455Z","iopub.status.idle":"2022-07-24T10:41:43.124542Z","shell.execute_reply.started":"2022-07-24T10:41:39.831421Z","shell.execute_reply":"2022-07-24T10:41:43.123148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(1, 2, figsize=(18, 6))\n\nplt.subplot(1,2,1)\nsns.scatterplot(x = 'num_crew', y = 'revenue', data=train)\nplt.title(\"Revenue vs number of crew\")\nplt.xlabel(\"Number of crew\")\nplt.ylabel(\"Revenue\")\n\nplt.subplot(1,2, 2)\nsns.scatterplot(x = 'num_crew', y = 'log_revenue', data=train)\nplt.title(\"log_revenue vs number of crew\")\nplt.xlabel(\"Number of crew\")\nplt.ylabel(\"log_revenue\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.126487Z","iopub.execute_input":"2022-07-24T10:41:43.127044Z","iopub.status.idle":"2022-07-24T10:41:43.548914Z","shell.execute_reply.started":"2022-07-24T10:41:43.126969Z","shell.execute_reply":"2022-07-24T10:41:43.547788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.550372Z","iopub.execute_input":"2022-07-24T10:41:43.550727Z","iopub.status.idle":"2022-07-24T10:41:43.663511Z","shell.execute_reply.started":"2022-07-24T10:41:43.550696Z","shell.execute_reply":"2022-07-24T10:41:43.661848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here i deleted column that contains unique values and columns that i already took data from them","metadata":{}},{"cell_type":"code","source":"train.drop(['genres', 'homepage', 'imdb_id', 'poster_path', 'production_companies','production_countries','release_date', 'spoken_languages','status', 'Keywords','cast','crew', 'id'], axis=1, inplace=True)\ntest.drop(['genres', 'homepage', 'imdb_id', 'poster_path', 'production_companies','production_countries','release_date', 'spoken_languages','status', 'Keywords','cast','crew', 'id'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.665200Z","iopub.execute_input":"2022-07-24T10:41:43.665572Z","iopub.status.idle":"2022-07-24T10:41:43.693338Z","shell.execute_reply.started":"2022-07-24T10:41:43.665536Z","shell.execute_reply":"2022-07-24T10:41:43.692314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['collection'] = train['collection'].apply(lambda x: str(x))\ntest['collection'] = test['collection'].apply(lambda x: str(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.694387Z","iopub.execute_input":"2022-07-24T10:41:43.694705Z","iopub.status.idle":"2022-07-24T10:41:43.711193Z","shell.execute_reply.started":"2022-07-24T10:41:43.694676Z","shell.execute_reply":"2022-07-24T10:41:43.710097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using label encoder assign unique number to all unique values","metadata":{}},{"cell_type":"code","source":"for col in ['original_language', 'all_genres', 'collection']:\n    lb = LabelEncoder()\n    lb.fit(list(train[col].fillna(\"\")) + list(test[col].fillna(\"\")))\n    train[col] = lb.transform(train[col].fillna(\"\"))\n    test[col] = lb.transform(test[col].fillna(\"\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.713763Z","iopub.execute_input":"2022-07-24T10:41:43.714385Z","iopub.status.idle":"2022-07-24T10:41:43.756762Z","shell.execute_reply.started":"2022-07-24T10:41:43.714317Z","shell.execute_reply":"2022-07-24T10:41:43.755872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['title', 'tagline', 'overview', 'original_title']:\n    train['len_' + col] = train[col].fillna('').apply(lambda x: len(str(x)))\n    train['words_' + col] = train[col].fillna('').apply(lambda x: len(str(x.split(' '))))\n    train = train.drop(col, axis=1)\n    test['len_' + col] = test[col].fillna('').apply(lambda x: len(str(x)))\n    test['words_' + col] = test[col].fillna('').apply(lambda x: len(str(x.split(' '))))\n    test = test.drop(col, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.758552Z","iopub.execute_input":"2022-07-24T10:41:43.759297Z","iopub.status.idle":"2022-07-24T10:41:43.879519Z","shell.execute_reply.started":"2022-07-24T10:41:43.759248Z","shell.execute_reply":"2022-07-24T10:41:43.878499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again dropped unnecessary columns","metadata":{}},{"cell_type":"code","source":"train.drop(['all_production_companies', 'all_production_countries', 'all_spoken_languages', 'all_Keywords', 'all_cast', 'all_cast_character', 'all_crew_name', 'all_job'], axis=1, inplace=True)\ntest.drop(['all_production_companies', 'all_production_countries', 'all_spoken_languages', 'all_Keywords', 'all_cast', 'all_cast_character', 'all_crew_name', 'all_job'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.880798Z","iopub.execute_input":"2022-07-24T10:41:43.881343Z","iopub.status.idle":"2022-07-24T10:41:43.893521Z","shell.execute_reply.started":"2022-07-24T10:41:43.881310Z","shell.execute_reply":"2022-07-24T10:41:43.892372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.895440Z","iopub.execute_input":"2022-07-24T10:41:43.895846Z","iopub.status.idle":"2022-07-24T10:41:43.985304Z","shell.execute_reply.started":"2022-07-24T10:41:43.895799Z","shell.execute_reply":"2022-07-24T10:41:43.984153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:43.987123Z","iopub.execute_input":"2022-07-24T10:41:43.987571Z","iopub.status.idle":"2022-07-24T10:41:44.154274Z","shell.execute_reply.started":"2022-07-24T10:41:43.987533Z","shell.execute_reply":"2022-07-24T10:41:44.152780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.156253Z","iopub.execute_input":"2022-07-24T10:41:44.156733Z","iopub.status.idle":"2022-07-24T10:41:44.164336Z","shell.execute_reply.started":"2022-07-24T10:41:44.156684Z","shell.execute_reply":"2022-07-24T10:41:44.163084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Columns that have null values\")\nfor col in test.columns:\n    if train[col].isnull().sum() > 0:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.166188Z","iopub.execute_input":"2022-07-24T10:41:44.166523Z","iopub.status.idle":"2022-07-24T10:41:44.223313Z","shell.execute_reply.started":"2022-07-24T10:41:44.166495Z","shell.execute_reply":"2022-07-24T10:41:44.222008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Columns that are not in test but in train\")\nfor col in train.columns:\n    if col not in test.columns:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.224795Z","iopub.execute_input":"2022-07-24T10:41:44.225275Z","iopub.status.idle":"2022-07-24T10:41:44.232945Z","shell.execute_reply.started":"2022-07-24T10:41:44.225228Z","shell.execute_reply":"2022-07-24T10:41:44.231384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['budget'], axis=1, inplace=True)\ntest.drop(['budget'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.234378Z","iopub.execute_input":"2022-07-24T10:41:44.234715Z","iopub.status.idle":"2022-07-24T10:41:44.249986Z","shell.execute_reply.started":"2022-07-24T10:41:44.234687Z","shell.execute_reply":"2022-07-24T10:41:44.248587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['runtime'].fillna(train['runtime'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.252840Z","iopub.execute_input":"2022-07-24T10:41:44.253382Z","iopub.status.idle":"2022-07-24T10:41:44.265598Z","shell.execute_reply.started":"2022-07-24T10:41:44.253311Z","shell.execute_reply":"2022-07-24T10:41:44.264378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drop columns that have only one unique value","metadata":{}},{"cell_type":"code","source":"for col in train.columns:\n    if(train[col].var() == 0):\n        print(col)\n        train.drop(col, axis=1, inplace=True)\n        test.drop(col, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.267255Z","iopub.execute_input":"2022-07-24T10:41:44.268374Z","iopub.status.idle":"2022-07-24T10:41:44.324990Z","shell.execute_reply.started":"2022-07-24T10:41:44.268268Z","shell.execute_reply":"2022-07-24T10:41:44.323679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Y as log_revenue our dependent variable\n\nX our input features","metadata":{}},{"cell_type":"code","source":"X = train.drop(['log_revenue', 'revenue'], axis=1)\nY = train['log_revenue']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.326221Z","iopub.execute_input":"2022-07-24T10:41:44.326680Z","iopub.status.idle":"2022-07-24T10:41:44.335375Z","shell.execute_reply.started":"2022-07-24T10:41:44.326632Z","shell.execute_reply":"2022-07-24T10:41:44.334007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divide dataset into train and valid for validation purpose","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, Y_train, Y_valid = train_test_split(X, Y, test_size=0.1, random_state = 42)\nX_train.shape, X_valid.shape, Y_train.shape, Y_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.337445Z","iopub.execute_input":"2022-07-24T10:41:44.337916Z","iopub.status.idle":"2022-07-24T10:41:44.356493Z","shell.execute_reply.started":"2022-07-24T10:41:44.337868Z","shell.execute_reply":"2022-07-24T10:41:44.355443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Linear Regression**","metadata":{}},{"cell_type":"code","source":"lr = LinearRegression()\nlr.fit(X_train, Y_train)\nmean_squared_error(lr.predict(X_valid), Y_valid), mean_absolute_error(lr.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.357786Z","iopub.execute_input":"2022-07-24T10:41:44.358189Z","iopub.status.idle":"2022-07-24T10:41:44.420168Z","shell.execute_reply.started":"2022-07-24T10:41:44.358158Z","shell.execute_reply":"2022-07-24T10:41:44.418904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Random Forest**","metadata":{}},{"cell_type":"code","source":"rf = RandomForestRegressor()\nrf.fit(X_train, Y_train)\nmean_squared_error(rf.predict(X_valid), Y_valid), mean_absolute_error(rf.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:44.422104Z","iopub.execute_input":"2022-07-24T10:41:44.422991Z","iopub.status.idle":"2022-07-24T10:41:49.875539Z","shell.execute_reply.started":"2022-07-24T10:41:44.422941Z","shell.execute_reply":"2022-07-24T10:41:49.874154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Importance","metadata":{}},{"cell_type":"code","source":"ax = plt.figure(figsize=(12, 10))\nfeat_importances = pd.Series(rf.feature_importances_, index=X_train.columns)\nfeat_importances.nlargest(30).plot(kind='barh')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:49.877044Z","iopub.execute_input":"2022-07-24T10:41:49.878140Z","iopub.status.idle":"2022-07-24T10:41:50.287804Z","shell.execute_reply.started":"2022-07-24T10:41:49.878090Z","shell.execute_reply":"2022-07-24T10:41:50.286807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"output feature is highly dependent on log_budget, populrity and release_date_year","metadata":{}},{"cell_type":"markdown","source":"**Ada boost**","metadata":{}},{"cell_type":"code","source":"ada = AdaBoostRegressor()\nada.fit(X_train, Y_train)\nmean_squared_error(ada.predict(X_valid), Y_valid), mean_absolute_error(ada.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:50.289302Z","iopub.execute_input":"2022-07-24T10:41:50.290349Z","iopub.status.idle":"2022-07-24T10:41:50.557994Z","shell.execute_reply.started":"2022-07-24T10:41:50.290301Z","shell.execute_reply":"2022-07-24T10:41:50.556592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Gradient Boosting regression**","metadata":{}},{"cell_type":"code","source":"params = {\n    \"n_estimators\": 500,\n    \"max_depth\": 4,\n    \"min_samples_split\": 5,\n    \"learning_rate\": 0.01,\n    \"loss\": \"squared_error\",\n}\n\nreg = GradientBoostingRegressor(**params)\nreg.fit(X_train, y_train)\nmean_squared_error(reg.predict(X_valid), Y_valid), mean_absolute_error(reg.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:50.559463Z","iopub.execute_input":"2022-07-24T10:41:50.559789Z","iopub.status.idle":"2022-07-24T10:41:59.699281Z","shell.execute_reply.started":"2022-07-24T10:41:50.559757Z","shell.execute_reply":"2022-07-24T10:41:59.698070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**XGB Regressor**","metadata":{}},{"cell_type":"code","source":"xgb = XGBRegressor()\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\nscores = cross_val_score(xgb, X, Y, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\nscores = absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (scores.mean(), scores.std()) )","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:41:59.701122Z","iopub.execute_input":"2022-07-24T10:41:59.701977Z","iopub.status.idle":"2022-07-24T10:42:31.214072Z","shell.execute_reply.started":"2022-07-24T10:41:59.701904Z","shell.execute_reply":"2022-07-24T10:42:31.212206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LGBM Regressor**","metadata":{}},{"cell_type":"code","source":"params = {'num_leaves': 30,\n         'min_data_in_leaf': 20,\n         'objective': 'regression',\n         'max_depth': 5,\n         'learning_rate': 0.01,\n         \"boosting\": \"gbdt\",\n         \"feature_fraction\": 0.9,\n         \"bagging_freq\": 1,\n         \"bagging_fraction\": 0.9,\n         \"bagging_seed\": 11,\n         \"metric\": 'rmse',\n         \"lambda_l1\": 0.2,\n         \"verbosity\": -1}\n\nlgb = LGBMRegressor(**params, n_estimators = 20000, nthread = 4, n_jobs = -1)\nlgb.fit(X_train, Y_train)\nmean_squared_error(lgb.predict(X_valid), Y_valid), mean_absolute_error(lgb.predict(X_valid), Y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:31.216563Z","iopub.execute_input":"2022-07-24T10:42:31.217163Z","iopub.status.idle":"2022-07-24T10:42:59.514451Z","shell.execute_reply.started":"2022-07-24T10:42:31.217107Z","shell.execute_reply":"2022-07-24T10:42:59.513384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eli5.show_weights(lgb, feature_filter=lambda x: x != '<BIAS>')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:59.518736Z","iopub.execute_input":"2022-07-24T10:42:59.521020Z","iopub.status.idle":"2022-07-24T10:42:59.538448Z","shell.execute_reply.started":"2022-07-24T10:42:59.520970Z","shell.execute_reply":"2022-07-24T10:42:59.537370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RandomForest and LGBMRegressor perform best as per there mean absolute error","metadata":{}},{"cell_type":"markdown","source":"**Regression neural network**","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Dense(198, input_shape = (198, ), kernel_initializer = 'normal', activation='elu'))\nmodel.add(Normalization())\nmodel.add(Dense(396, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Normalization())\nmodel.add(Dense(1028, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Dense(396, kernel_initializer = 'normal', activation='elu'))\nmodel.add(Dense(1, kernel_initializer='normal'))\n\nmodel.compile(loss='mean_absolute_error', optimizer='adam')\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:59.540089Z","iopub.execute_input":"2022-07-24T10:42:59.540716Z","iopub.status.idle":"2022-07-24T10:42:59.655064Z","shell.execute_reply.started":"2022-07-24T10:42:59.540679Z","shell.execute_reply":"2022-07-24T10:42:59.653757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}