{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# math libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pandas.plotting import scatter_matrix\n\n# Training model\n# Check out 'https://scikit-learn.org/stable/tutorial/machine_learning_map/index.html' for which model to use\nfrom sklearn.linear_model import LinearRegression, LogisticRegression # Linear -> regression, Logistic -> classification\nfrom xgboost import XGBRegressor, XGBClassifier # Regression and Classification\nfrom sklearn.impute import SimpleImputer # for imputing Nan values\nfrom sklearn.utils import shuffle\n\n# validating model\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, log_loss, accuracy_score\n\n# misc.\nfrom IPython import display\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# display data sources\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ee332ea2fd6a356ec1a23530eef9fd9c812a3002"},"cell_type":"markdown","source":"# The Data"},{"metadata":{"_uuid":"72cbf4168988e02fedd275297e233c1cec24105e"},"cell_type":"markdown","source":"## Load Data"},{"metadata":{"trusted":true,"_uuid":"16eb054ee64699019a817e9b25580c8a4cfd85a3"},"cell_type":"code","source":"data_dir = '../input/'\n# df = pd.read_csv(data_dir + '')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"## Discover Data"},{"metadata":{"trusted":true,"_uuid":"dd8ffec82417bd6548f8c6ce428941b089e0c8c1"},"cell_type":"code","source":"def discover_data(df):\n    print(\"Head\\n\")\n    display.display(df.head())\n    print(\"\\nDescription\\n\")\n    display.display(df.describe())\n    print(\"\\nInfo\\n\")\n    display.display(df.info())\n\n# df.some_feature.value_counts() shows you all types of values for that feature","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ae86f4a69d65421fca939ca1da694baf6eebefa"},"cell_type":"markdown","source":"## Visualize Data"},{"metadata":{"trusted":true,"_uuid":"4e53a219e03abd82d485367804cf6d5a1ae508e8"},"cell_type":"code","source":"# _ = df.hist(figsize=(20, 15), bins=50)\n\n# _ = df.plot(kind=\"scatter\", x=\"longitude\", y=\"latitude\", alpha=0.4, \n#                 s=housing[\"population\"]/100, label=\"population\", # s defines radius of each circle, based on population in this case\n#                 c=\"median_house_value\", cmap=plt.get_cmap(\"jet\"), colorbar=True) # c defines color of circle, based on median house value in this case","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6505758a076b7a164807c59e5d75ba1baaa12b76"},"cell_type":"markdown","source":"## Understand Correlations"},{"metadata":{"trusted":true,"_uuid":"bb8f805340c2e30078e273d79c4bccbab8856edf"},"cell_type":"code","source":"# corr_matrix = df.corr()\n# corr_matrix.median_house_value.sort_values(ascending=False)\n# attributes = [\"median_house_value\", \"median_income\", \"total_rooms\", \"housing_median_age\"]\n# _ = scatter_matrix(housing[attributes], figsize=(12, 8))\n# _ = housing.plot(kind=\"scatter\", x=\"median_income\", y=\"median_house_value\", alpha=0.2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"69e6c2df9d8050d1f821aee3df87a2cc0e4cd3c4"},"cell_type":"markdown","source":"## Prepare Data"},{"metadata":{"_uuid":"4a47227d9481c49559e635e33cdf1025769df251"},"cell_type":"markdown","source":"### Handle Missing Values"},{"metadata":{"trusted":true,"_uuid":"4cf7e23d9b444b39d38225f0933bfb018ebdc72a"},"cell_type":"code","source":"def drop_missing_values(df):\n    new_df = df.copy()\n    cols_with_missing = [col for col in new_df.columns if new_df[col].isnull().any()]\n    return new_df.drop(cols_with_missing, axis=1)\n\ndef impute_missing_values(df):\n    my_imputer = SimpleImputer()\n    new_df = pd.DataFrame(my_imputer.fit_transform(df))\n    new_df.columns = df.columns\n    return new_df\n\ndef impute_with_categorical_indications(df):\n    # make copy to avoid changing original data (when Imputing)\n    new_data = df.copy()\n    # make new columns indicating what will be imputed\n    cols_with_missing = (col for col in new_data.columns if new_data[col].isnull().any())\n    for col in cols_with_missing:\n        new_data[col + '_was_missing'] = new_data[col].isnull()\n    # Imputation\n    new_data = impute_missing_values(new_data)\n    new_data.columns = df.columns\n    return new_data","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d0dca014defec60e368f43273e5fd88254d2f693"},"cell_type":"markdown","source":"### Handling text and categorical data"},{"metadata":{"trusted":true,"_uuid":"63c7f4e00b39f6b8bd0691b2d03b320addd8f228"},"cell_type":"code","source":"def drop_categoricals(df):\n    return df.select_dtypes(exclude=['object'])\n\ndef one_hot_encode(df):\n    # drop missing values\n    clean_df = drop_missing_values(df)\n    # get numeric cols and unique object cols and merge\n    numeric_cols = [col for col in clean_df.columns if clean_df[col].dtype in ['int64', 'float64']]\n    low_cardinality_cols = [col for col in clean_df.columns if\n                                   clean_df[col].nunique() < 10 and\n                                   clean_df[col].dtype == \"object\"]\n    # merge cols and return one hot encoded version of df\n    my_cols = numeric_cols + low_cardinality_cols\n    df_with_my_cols = clean_df[my_cols]\n    return pd.get_dummies(df_with_my_cols)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"098da9b7ebd10fe663e492c75051f1a3a06a4a70"},"cell_type":"markdown","source":"### The actual processing function used in training and testing"},{"metadata":{"trusted":true,"_uuid":"d9b91fb80946d45b1ea96eff75561c247d0f2b09"},"cell_type":"code","source":"def preprocess_data(df):\n    # Impute for missing values, create synthetic features, etc\n    return df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21aadf6dff6228a2ada17f4da0f8f7b2cc5b4958"},"cell_type":"markdown","source":"# Train, Validate, Submit Model"},{"metadata":{"trusted":true,"_uuid":"c1a21fd8f2ed79aa64a0f7572f1d9700d1f0c674"},"cell_type":"code","source":"### Important Globals\n\n# features = []\n# target = \"\"\n# data_dir = '../input/'\n# df = pd.read_csv(data_dir + 'train.csv')\n# x = preprocess_data(df[features])\n# y = df[target]\n# discover_data(df)\n\n### Choose a model\n\n# model = XGBClassifier(n_estimators=1000, learning_rate=0.05)\n# model = XGBRegressor(n_estimators=1000, learning_rate=0.05)\n# model = LogisticRegression()\n# model = LinearRegression()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d736c7331b1ab00e050882750313708d9eae995"},"cell_type":"code","source":"def validate_model(x, y):\n    # split test and train data\n    train_x, val_x, train_y, val_y = train_test_split(x, y, random_state=42, test_size=0.2)\n\n    ### Fit linear model\n#     model.fit(train_x, train_y)\n    ### Fit XGB model\n#     model.fit(train_x, train_y, early_stopping_rounds=5, \n#              eval_set=[(val_x, val_y)], verbose=False)\n\n    # Calculate the mean absolute error of your Random Forest model on the validation data\n    val_predictions = model.predict(val_x)\n    print(\"First 5 predictions:\", val_predictions[:5])\n    print(\"First 5 actual values:\", val_y[:5])\n#     print(\"Accuracy: \", accuracy_score(val_predictions, val_y))  # Classification\n#     print(\"Validation MAE for Model: \", mean_absolute_error(val_predictions, val_y))  # Regression\n    \n# validate_model(x, y)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf7a2bc06f2037ad60e28f74010e7ff5ef82f691"},"cell_type":"markdown","source":"# Submit Predictions (after validating model)"},{"metadata":{"trusted":true,"_uuid":"60094a5661a33c98fd0357b4999132eea814c69f"},"cell_type":"code","source":"def submit_predictions(model, x, y, features):\n    model.fit(x, y)\n\n    # read test data file using pandas\n    test_data_path = '../input/test.csv'\n    test_data = pd.read_csv(test_data_path)\n\n    # create test features and make predictions\n    test_x = test_data[features]\n    processed_test_x = preprocess_data(test_x)\n    test_preds = model.predict(processed_test_x)\n\n    # Save data in format necessary to score in competition\n    ids = test_data[ID]\n    output_dummy = list(zip(ids, test_preds))\n    output = pd.DataFrame()\n    \n    for my_id, pred in output_dummy:\n        dummy_df = pd.Series()\n        dummy_df['PassengerId'] = str(my_id)\n        dummy_df['Survived'] = str(pred)\n        output = output.append(dummy_df, ignore_index=True)\n\n    output.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74e05a25a99e2c6a78fef276530effe016c9db2c"},"cell_type":"code","source":"# submit_predictions(model, x, y, features)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}