{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Ensembling by finding the most frequent label for each sample from public notebooks\nThis notebook presents an automated ensemble model using predicted results from the most relevant public notebooks. The goal is to show the power of a simple ensembling technique on the final score.","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport plotly.express as px","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targetName = 'label'\nidName = 'id'\ncompetitionDir = '/kaggle/input/tpu-getting-started'\nsubmission = pd.read_csv('../input/tpu-getting-started/sample_submission.csv')\nsubmission.sort_values(by=[idName], inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import any number of public notebooks to update the ensemble prediction¶","metadata":{}},{"cell_type":"code","source":"preds = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if (dirname != competitionDir) & ('.csv' in filename):\n            df = pd.read_csv(os.path.join(dirname, filename))\n            if len(df) == len(submission):\n                try:\n                    df.sort_values(by=[idName], inplace=True)\n                    preds.append(df[targetName])\n                except Exception:\n                    pass","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save ensemble prediction to csv¶","metadata":{}},{"cell_type":"code","source":"submission[targetName] = stats.mode(np.array(preds), axis=0)[0].transpose()\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of the predicted classes¶","metadata":{}},{"cell_type":"code","source":"target_df = pd.DataFrame(np.log(submission[targetName].value_counts())).reset_index()\ntarget_df.columns = [targetName, 'Count']\nfig = px.bar(data_frame = target_df, \n             x = targetName,\n             y = 'Count' , \n             color = \"Count\",\n             color_continuous_scale=\"Emrld\") \nfig.show()","metadata":{},"execution_count":null,"outputs":[]}]}