{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install requests\n!pip install tabulate\n!pip install \"colorama>=0.3.8\"\n!pip install future","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -f http://h2o-release.s3.amazonaws.com/h2o/latest_stable_Py.html h2o","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import h2o\nfrom h2o.automl import H2OAutoML\nh2o.init()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\npath = \"/kaggle/input/talkingdata-adtracking-fraud-detection/train.csv\"\ns = !wc -l {path}\n\nprint(s)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_tmp = pd.read_csv(path, nrows=5)\ndf_tmp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_tmp.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"chunksize = 100000\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df1 = pd.read_csv(path, chunksize = chunksize, usecols = ['is_attributed'] )\ndf1 = pd.DataFrame(df1.get_chunk(1))\nprint(type(df1))\n#print(len(ch1))\n##import dask.dataframe as dd\n##path = \"/kaggle/input/talkingdata-adtracking-fraud-detection/train.csv\"\n##df = dd.read_csv(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(type(df1.info()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ndf_list = [] # list to hold the batch dataframe\ni=0\nfor df_chunk in tqdm(pd.read_csv(path, chunksize = chunksize, usecols = ['is_attributed'] )):\n     \n    # Neat trick from https://www.kaggle.com/btyuhas/bayesian-optimization-with-xgboost\n    # Using parse_dates would be much slower!\n    #df_chunk['pickup_datetime'] = df_chunk['pickup_datetime'].str.slice(0, 16)\n    #df_chunk['pickup_datetime'] = pd.to_datetime(df_chunk['pickup_datetime'], utc=True, format='%Y-%m-%d %H:%M')\n    \n    # Can process each chunk of dataframe here\n    # clean_data(), feature_engineer(),fit()\n    df1 = pd.DataFrame(df_chunk)\n    if i<10:\n        print(df1.shape)\n    # Alternatively, append the chunk to list and merge all\n    df_list.append(df_chunk) \n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(df_list))\nprint(df_list[0][1500:1600])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sys import getsizeof\ngetsizeof(df_list)\ntype(df_list[0])\n#получаем в чанке 0 элемент с индексом 1504 (строка) и колонкой 0 ( у нас одна колонка)\ndf_list[0].iloc[1504,0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s=0\nfor i in range(len(df_list)):\n    s += df_list[i].sum()\n\n#во всех чанках у нас столько единиц\nprint(s)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# а всего записей у нас у нас:\ndf_list[-1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# то есть 184903889+1\n#итого: доля единиц:\nprint(456846/184903890*100) #в процентах доля единиц среди всех данных 0.247%","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Merge all dataframes into one dataframe\ntrain_df = pd.concat(df_list)\n\n# Delete the dataframe list to release memory\ndel df_list\n\n# See what we have loaded\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from tqdm import tqdm\n#from progressbar import progressbar as pb\n\n#!ls -l /kaggle/input/talkingdata-adtracking-fraud-detection\n\n# Load data into H2O\npath = \"/kaggle/input/talkingdata-adtracking-fraud-detection/train.csv\"\n#i = 0\n\nchunks = []\nfor chunk in tqdm(pd.read_csv(path, chunksize=1000)):\n    #i +=1\n    #print(i)\n    chunks.append(chunk)\n    \n#df = pd.read_csv(path)\n#df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#возьмем временно другой датасет train_sample\n# Load data into H2O\npath = \"/kaggle/input/talkingdata-adtracking-fraud-detection/train_sample.csv\"\ndf = pd.read_csv(path)\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#не заполнены колонки со временем: нужно их заполнить\n#выведем заполненные\ndf_samp = df[pd.notnull(df['attributed_time'])]\ndf_samp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(df_samp.iloc[0,6])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.iloc[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#надо заполнить колонку attributed_time так:\n#для тех строк у которых известно attributed_time вычисляем среднюю разницу между click_time и attributed_time\n#это и будет добавка к колонке click_time, далее просто заполняем незаполненные путем добавления времени к click_time","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.drop('attributed_time', axis = 1, inplace=True)\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labelEncoder = LabelEncoder()\nfor col in df.columns:\n    df[col] = labelEncoder.fit_transform(df[col])\n\ndf = h2o.H2OFrame(df)\ndf = df.asfactor()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.describe ()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"смущает что время было преобразовано в большой набор чисел, может ли это повлиять на результаты автообучения???\n\nНиже в коде: \nмогу ли я не расщеплять исходный датасет на тестовый и тренировочный? (скорее всего да)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#настройка и  тренировка\n\ntrain = df\n\ny = \"is_attributed\"\nx_train = train.columns\nx_train.remove(y)\n\naml = H2OAutoML(max_runtime_secs=300, seed = 1)\naml.train(x = x_train, y = y, training_frame = train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#просмотреть результаты работы\nlb = aml.leaderboard\nlb.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#preds = aml.predict(test)\naml.training_info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#тестовые данные\npath = \"/kaggle/input/talkingdata-adtracking-fraud-detection/test.csv\"\ndf_test = pd.read_csv(path)\ndf_test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#дропаем лишнюю колонку, чтобы была структура такая же как и у train но без target колонки\ndf_test.drop('click_id', axis = 1, inplace=True)\ndf_test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labelEncoder = LabelEncoder()\n\nfor col in df_test.columns:\n    df_test[col] = labelEncoder.fit_transform(df_test[col])\n\ndf_test = h2o.H2OFrame(df_test)\ndf_test = df_test.asfactor()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = aml.predict(df_test)\npreds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_preds = h2o.as_list(preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sum(df_preds['predict']==1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#то есть для стольких элементов 97634 из 18790469 предсказана 1, для остальных 0 по лучшей модели\n#доля:\nprint(97634/18790469*100) # в процентах, то есть больше чем исходно единиц в основном тренировочном \n#датасете (было 0.24707214109989792% )","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}