{"cells":[{"metadata":{"_cell_guid":"f69f22bb-7766-4fe6-b585-24c69d7d92f0","collapsed":true,"_uuid":"f061969fc0011850217a9f88bd48edd67bbbcb65","trusted":false},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport csv\nimport tensorflow as tf\nimport nltk\nimport gc\nfrom gensim.models import Word2Vec\nfrom keras.preprocessing import text, sequence\nfrom sklearn.model_selection import train_test_split\nfrom collections import Counter\nimport math","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"31a9cb58-4cb9-4647-9b44-05a3a4d6c858","collapsed":true,"_uuid":"c1dfc32c6041e87fa7d672a4c102945fc14f3d01","trusted":false},"cell_type":"code","source":"df_train = pd.read_csv('../input/avito-demand-prediction/train.csv') \ntrain_input = df_train['title']\ntrain_input_des = df_train['description'].astype(str)\ny_train = df_train['deal_probability']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9fb35e25-efb1-4fdb-829f-0698e24ece85","collapsed":true,"_uuid":"e6b3a07a871194e087d0bfdde2dcdad208e7be20","trusted":false},"cell_type":"code","source":"df_test = pd.read_csv('../input/avito-demand-prediction/test.csv')\ntest_input = df_test['title']\ntest_input_des = df_test['description'].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d8f8f404-34eb-42a3-b338-ae9cce352f30","collapsed":true,"_uuid":"943203f58572010d6ac39b1bffba646e3cbb9b6b","trusted":false},"cell_type":"code","source":"df_train['date'] = pd.to_datetime(df_train['activation_date']).dt.day.astype('int')\n#train_input_other = df_train[['region','parent_category_name','category_name','date','user_type','price']]\ndf_test['date'] = pd.to_datetime(df_test['activation_date']).dt.day.astype('int')\n#test_input_other = df_test[['region','parent_category_name','category_name','date','user_type','price']]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ff2b2d13-55d1-43e4-b8b3-891f8ad8439d","collapsed":true,"_uuid":"07f605e2447cee69b7567b53514c37c6c995bbb5","trusted":false},"cell_type":"code","source":"del df_train, df_test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"41daaca3-710a-402d-a548-706f4ea01273","collapsed":true,"_uuid":"79dc5807186bdc0c3a58cfbfc32dacc1f6339020","trusted":false},"cell_type":"code","source":"#l_region = dict(zip(list(set(train_input_other['region'])),range(28)))\n#l_parent_category_name = dict(zip(list(set(train_input_other['parent_category_name'])),range(9)))\n#l_category_name = dict(zip(list(set(train_input_other['category_name'])),range(47)))\n#l_user_type = dict(zip(list(set(train_input_other['user_type'])),range(3)))\n#l_date = dict(zip(list(set(train_input_other['date'])),range(21)))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"48ed7b4d-9117-41cb-a64c-76fc04cf6b1b","collapsed":true,"_uuid":"80776a08078b8638e8075a247792034bd3ef1ea6","trusted":false},"cell_type":"code","source":"#train_input_other['region'] = train_input_other['region'].replace(l_region)\n#train_input_other['parent_category_name'] = train_input_other['parent_category_name'].replace(l_parent_category_name)\n#train_input_other['category_name'] = train_input_other['category_name'].replace(l_category_name)\n#train_input_other['user_type'] = train_input_other['user_type'].replace(l_user_type)\n#train_input_other['date'] = train_input_other['date'].replace(l_date)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d5c2f62f-137b-4386-933d-cd50fe212192","collapsed":true,"_uuid":"9dae25373558b6a4b1c26082e25e35792fd2bfc6","trusted":false},"cell_type":"code","source":"#test_input_other['region'] = test_input_other['region'].replace(l_region)\n#test_input_other['parent_category_name'] = test_input_other['parent_category_name'].replace(l_parent_category_name)\n#test_input_other['category_name'] = test_input_other['category_name'].replace(l_category_name)\n#test_input_other['user_type'] = test_input_other['user_type'].replace(l_user_type)\n#test_input_other['date'] = test_input_other['date'].replace(l_date)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"94e68ca1-5030-4c70-999f-91ee2c53c120","collapsed":true,"_uuid":"6dbaa02caad4c5c6e874fa4222075dce21fe9edb","trusted":false},"cell_type":"code","source":"#del l_region, l_parent_category_name, l_category_name, l_user_type, l_date\n#gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f2e7084b-26ab-4333-bf54-b24992a74d1a","collapsed":true,"_uuid":"f2dd5f849189ab537ddc2485350b858869941a7f","trusted":false},"cell_type":"code","source":"def get_coefs(word, *arr): \n    return word, np.asarray(arr, dtype='float32')\n\nembeddings_index = dict(get_coefs(*o.rstrip().rsplit(' ')) for o in open('../input/fasttext-russian-2m/wiki.ru.vec'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"1b19ca6c-2546-471f-983d-ecbe29a8ebf0","collapsed":true,"_uuid":"ad187e0b4d60698381289577a9bd4a382c7a9578","trusted":false},"cell_type":"code","source":"len(embeddings_index) ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"55359b13-4400-40f0-b11f-d8d3cd163eed","collapsed":true,"_uuid":"1851f75467558d3ecb30602003c35e6ba8af5b82","trusted":false},"cell_type":"code","source":"del embeddings_index['1888423']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5dd5dba6-1445-4523-83b6-46bc381b6a3f","collapsed":true,"_uuid":"2a7534893c3c9a15f061584f5fb2a01b409c31a7","trusted":false},"cell_type":"code","source":"def clean(string):\n    string = re.sub(r'\\n', ' ', string)\n    string = re.sub(r'\\t', ' ', string)\n    string = re.sub('[\\W]', ' ', string)\n    string = re.sub(r'\\s{2,}', ' ', string.lower())\n    return string","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6a82a393-1bb8-4cb7-82d8-e8a2756fa9d9","collapsed":true,"_uuid":"a1ff929775b481c6b50fba5d5d9d04a91c496c31","trusted":false},"cell_type":"code","source":"x_train = train_input.apply(clean)\nx_train_des = train_input_des.apply(clean)\nx_test = test_input.apply(clean)\nx_test_des = test_input_des.apply(clean)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6e89bf79-dd16-417b-bf28-4be9b167e6f5","collapsed":true,"_uuid":"af2fc5e183268127d7dcbab65098ee274b641622","trusted":false},"cell_type":"code","source":"x_train = x_train.fillna('fillna')\nx_train_des = x_train_des.fillna('fillna')\nx_test = x_test.fillna('fillna')\nx_test_des = x_test_des.fillna('fillna')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ec4501af-3ddf-4230-a468-d0c269641635","collapsed":true,"_uuid":"17d9271cf28091a2532804427f1c8bac712170f5","trusted":false},"cell_type":"code","source":"lst = []\nfor line in x_train:\n    lst += line.split()\n    \ncount = Counter(lst)\nfor k in list(count.keys()):\n    if k not in embeddings_index:\n        del count[k]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5f9c6291-8238-42df-bb74-b00ff9559552","collapsed":true,"_uuid":"5a11e8b40d9167328544e9e6d5ab2ac1f35d9aaf","trusted":false},"cell_type":"code","source":"len(count)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"453beccb-1870-4b34-8e15-3eb5bb40bd9e","collapsed":true,"_uuid":"c19eeed7280496d0ede6257166b99750c373c54c","trusted":false},"cell_type":"code","source":"count = dict(sorted(count.items(), key=lambda x: -x[1]))\ncount = {k:v for (k,v) in count.items() if v >= 3}","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"620cd226-4691-4b81-a1ca-663f4126f9e7","collapsed":true,"_uuid":"596cb84acad7b4549aec26ef9667f148d33ed00f","trusted":false},"cell_type":"code","source":"len(count)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fc258364-c6b0-4364-a4d0-e268f857f8a1","collapsed":true,"_uuid":"69273e3157c6734eb9d8c03020e3d4968cef6828","trusted":false},"cell_type":"code","source":"count = dict(zip(list(count.keys()),range(1,42843 + 1)))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"417d2760-7cdf-49c2-9e04-e1b4d3a554d3","collapsed":true,"_uuid":"100cc447a896d7c6c67a9c9d565b933a202fd92b","trusted":false},"cell_type":"code","source":"embedding_matrix = {}\nfor key in count:\n    embedding_matrix[key] = embeddings_index[key]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"2faad2d4-df3a-46e4-ab13-4511e3de6ea8","collapsed":true,"_uuid":"b19b311bf6b434df7151ac7de4bc2671f837caff","trusted":false},"cell_type":"code","source":"lst = []\nfor line in x_test:\n    lst += line.split()\n    \ncount_test = Counter(lst)\nfor k in list(count_test.keys()):\n    if k not in embedding_matrix:\n        del count_test[k]\n    else:\n        count_test[k] = count[k]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"cd325507-28a1-43cb-bab0-bd3be72a86d5","collapsed":true,"_uuid":"ac3e683b93bdd7dce7b207c6bc736f7490b644ec","trusted":false},"cell_type":"code","source":"for i in range(len(x_train)):\n    temp = x_train[i].split()\n    for word in temp[:]:\n        if word not in count:\n            temp.remove(word)\n    for j in range(len(temp)):\n        temp[j] = count[temp[j]]\n    x_train[i] = temp\n\nfor i in range(len(x_test)):\n    temp = x_test[i].split()\n    for word in temp[:]:\n        if word not in count_test:\n            temp.remove(word)\n    for j in range(len(temp)):\n        temp[j] = count_test[temp[j]]\n    x_test[i] = temp","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c99be76e-9482-419e-88db-46eb6ad48a8b","collapsed":true,"_uuid":"ba027dbe402c48771d54ad4a80715d5300944de2","trusted":false},"cell_type":"code","source":"lst = []\nfor line in x_train_des:\n    lst += line.split()\n    \ncount = Counter(lst)\nfor k in list(count.keys()):\n    if k not in embeddings_index:\n        del count[k]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7d230c59-05d4-4710-af37-52a536215bc0","collapsed":true,"_uuid":"bef75f55eb46cdca9f2d8d2c7e951c8a54a09eb2","trusted":false},"cell_type":"code","source":"len(count)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fac57584-2f29-46b3-87c7-a31315b7e908","collapsed":true,"_uuid":"b944586e23867530f096192d0de2cb92c93b5b8b","trusted":false},"cell_type":"code","source":"count = dict(sorted(count.items(), key=lambda x: -x[1]))\ncount = {k:v for (k,v) in count.items() if v >= 10}","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8f085294-3266-4d80-85d1-596b4ad69ea9","collapsed":true,"_uuid":"3e640419408e32b52f7b7945cbb1b7167ca8a7c2","trusted":false},"cell_type":"code","source":"len(count)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f1539204-90f7-4b9d-90e5-ab6bfd0b6e9f","collapsed":true,"_uuid":"9954cce649d25ecdbd0831a624972d4d274e7bf2","trusted":false},"cell_type":"code","source":"count = dict(zip(list(count.keys()),range(1, 84411 + 1)))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"da913e07-fb98-452b-a377-66e83cfbf968","collapsed":true,"_uuid":"57307651ec2c2b2c8ab4c95bf5c605a1182a577f","trusted":false},"cell_type":"code","source":"embedding_matrix = {}\nfor key in count:\n    embedding_matrix[key] = embeddings_index[key]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4b619352-8ca7-4074-8133-1dc6a1699e38","collapsed":true,"_uuid":"ea967910fc86d37c14a7ff7d6166a254f3b24858","trusted":false},"cell_type":"code","source":"lst = []\nfor line in x_test_des:\n    lst += line.split()\n    \ncount_test = Counter(lst)\nfor k in list(count_test.keys()):\n    if k not in embedding_matrix:\n        del count_test[k]\n    else:\n        count_test[k] = count[k]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fd24a085-f299-4ea9-ac80-501cb48f09c6","collapsed":true,"_uuid":"077d87564c3f10a1bf82551d7a1065f895e01d5f","trusted":false},"cell_type":"code","source":"for i in range(len(x_train_des)):\n    temp = x_train_des[i].split()\n    for word in temp[:]:\n        if word not in count:\n            temp.remove(word)\n    for j in range(len(temp)):\n        temp[j] = count[temp[j]]\n    x_train_des[i] = temp\n\nfor i in range(len(x_test_des)):\n    temp = x_test_des[i].split()\n    for word in temp[:]:\n        if word not in count_test:\n            temp.remove(word)\n    for j in range(len(temp)):\n        temp[j] = count_test[temp[j]]\n    x_test_des[i] = temp","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a208afc7-5eea-4518-886f-a15378776a41","collapsed":true,"_uuid":"ff8436298ed6ee4275d76f114303e7c6d1a7cf02","trusted":false},"cell_type":"code","source":"W = np.zeros((1,300))\nW = np.append(W, np.array(list(embedding_matrix.values())),axis=0)\nW = W.astype(np.float32, copy=False)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d6c47ab7-60cf-4ff5-9c6e-c4f50a0f6307","collapsed":true,"_uuid":"c37e08d1648a8c36d05c6db7e5ddb94036c96dc9","trusted":false},"cell_type":"code","source":"del lst, embeddings_index, embedding_matrix, temp, count, count_test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"90d41794-adc6-48cd-8310-a3096f5a6563","collapsed":true,"_uuid":"f029d1fa6fce0b5ba2d5292a78a3248f3b37f428","trusted":false},"cell_type":"code","source":"#Xtrain, Xval, ytrain, yval, Xtrain_des, Xval_des = train_test_split(x_train, y_train, x_train_des, train_size=0.80, random_state=123)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c54a89d4-cb42-4bed-bcd4-3ef785741f78","_uuid":"1bd096aefb8015c8ca8eec4aef5b86f0bd422899"},"cell_type":"markdown","source":"# For test"},{"metadata":{"_cell_guid":"ba81d7da-242d-4308-8926-1065a0eaaa5f","collapsed":true,"_uuid":"c117c3ad3ef96504b99c2fbfbbc8e40bd1d755e8","trusted":false},"cell_type":"code","source":"#Xtrain = sequence.pad_sequences(list(Xtrain), maxlen = 10)\n#Xval = sequence.pad_sequences(list(Xval), maxlen = 10)\n#Xtrain_des = sequence.pad_sequences(list(Xtrain_des), maxlen = 60)\n#Xval_des = sequence.pad_sequences(list(Xval_des), maxlen = 60)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d0c8ed9e-bd00-4d03-9ce6-a4710efb9db2","_uuid":"b6e3f836faf2fbfd6a9e252f2353acbb42d371c4"},"cell_type":"markdown","source":"# Formal"},{"metadata":{"_cell_guid":"97238efa-392b-472f-8db2-607814cf3070","collapsed":true,"_uuid":"165383aac3d0267a68cebbdac80d3c5818cfb546","trusted":false},"cell_type":"code","source":"x_train = sequence.pad_sequences(list(x_train), maxlen = 10)\nx_train_des = sequence.pad_sequences(list(x_train_des), maxlen = 60)\nx_test = sequence.pad_sequences(list(x_test), maxlen = 10)\nx_test_des = sequence.pad_sequences(list(x_test_des), maxlen = 60)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3f2f8b36-a4d6-43ef-9232-d877e9547c52","_uuid":"d45e952722129f6b84b928eeaeb8d2c187860c2e"},"cell_type":"markdown","source":"# CNN"},{"metadata":{"_cell_guid":"fa742810-ce38-4fa5-8872-1a1827c666d4","collapsed":true,"_uuid":"698d83c1db8ae676321a86d961088eaee6539cf8","trusted":false},"cell_type":"code","source":"filter_sizes = [1,2,3,4,5]\nnum_filters = 32\nbatch_size = 256\n#This large batch_size is specially for this case. Usually it is between 64-128.\nnum_filters_total = num_filters * len(filter_sizes)\nembedding_size = 300\nsequence_length = 10\nsequence_length_des = 60\nnum_epochs = 3 #Depends on your choice.\ndropout_keep_prob = 0.8","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"336ce86b-e3e5-4ea0-bb95-49cc6dd17dae","collapsed":true,"_uuid":"1da7e4c5aed7cd62c1935668918accc8fbb77b65","trusted":false},"cell_type":"code","source":"input_x = tf.placeholder(tf.int32, [None, sequence_length], name = \"input_x\")\ninput_y = tf.placeholder(tf.float32, [None,2], name = \"input_y\")\ninput_x_des = tf.placeholder(tf.int32, [None, sequence_length_des], name = \"input_x_des\")\n#input_x_other = tf.placeholder(tf.int32, [None, 5], name = \"input_x_other_des\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"97f5f655-ed58-4245-9559-1a048697a332","collapsed":true,"_uuid":"e5c8faf68b35aa103687dc4e9fcd41f4ce0e1bab","trusted":false},"cell_type":"code","source":"embedded_chars = tf.nn.embedding_lookup(W, input_x)\nembedded_chars_expanded = tf.expand_dims(embedded_chars, -1)\nembedded_chars_des = tf.nn.embedding_lookup(W, input_x_des)\nembedded_chars_expanded_des = tf.expand_dims(embedded_chars_des, -1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f50f2df4-4c07-4909-b9ad-e22d148e8a5e","collapsed":true,"_uuid":"1dd671727a4ffe07641bb7a22c63edc1332cd2a5","trusted":false},"cell_type":"code","source":"def CNN(data, length):\n    pooled_outputs = []\n    \n    for i, filter_size in enumerate(filter_sizes):\n        \n        filter_shape = [filter_size, embedding_size, 1, num_filters]\n        \n        w = tf.Variable(tf.truncated_normal(filter_shape,stddev = 0.05), name = \"w\")\n        b = tf.Variable(tf.truncated_normal([num_filters], stddev = 0.05), name = \"b\")\n            \n        conv = tf.nn.conv2d(\n            data,\n            w,\n            strides = [1,1,1,1],\n            padding = \"VALID\",\n            name = \"conv\"\n        )\n        h = tf.nn.relu(tf.nn.bias_add(conv, b), name = \"relu\")\n        pooled = tf.nn.max_pool(\n            h,\n            ksize = [1, length - filter_size + 1, 1, 1],\n            strides = [1,1,1,1],\n            padding = \"VALID\",\n            name = \"pool\"\n        )\n        \n        pooled_outputs.append(pooled)\n    \n    #return pooled_outputs\n    h_pool = tf.concat(pooled_outputs, 3)\n    h_pool_flat = tf.reshape(h_pool, [-1, num_filters_total])\n    return h_pool_flat","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"1e5bdc7e-a115-4079-acc9-ddfd744020db","collapsed":true,"_uuid":"34676b9f72bde2df57e85f77d27ac995b9e67318","trusted":false},"cell_type":"code","source":"h_pool_flat = CNN(embedded_chars_expanded, sequence_length)\nh_pool_flat_des = CNN(embedded_chars_expanded_des, sequence_length_des)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"93df597d-55c7-449f-85cc-88c9fa06109a","collapsed":true,"_uuid":"bbc158098fbac354a37ec8b004730e41e9974792","trusted":false},"cell_type":"code","source":"h_pool_flat_all = tf.concat([h_pool_flat, h_pool_flat_des],1)#, input_x_other)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5f58bd3e-879d-47b2-865a-42289f7ae31c","collapsed":true,"_uuid":"605d98ab267634d4e0acacb42fa95e61ecaa2fb9","trusted":false},"cell_type":"code","source":"h_drop = tf.nn.dropout(h_pool_flat_all, dropout_keep_prob)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ea90f2ae-435c-4ac5-8e26-e62d95ef95f1","collapsed":true,"_uuid":"3f9b7a896a8c937b49621737b0be64a4f4a56224","trusted":false},"cell_type":"code","source":"wd1 = tf.Variable(tf.truncated_normal([num_filters_total*2, num_filters_total], stddev=0.05), name = \"wd1\")\nbd1 = tf.Variable(tf.truncated_normal([num_filters_total], stddev = 0.05), name = \"bd1\")\nlayer1 = tf.nn.xw_plus_b(h_drop, wd1, bd1, name = 'layer1') # Do wd1*h_drop + bd1\nlayer1 = tf.nn.relu(layer1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"2efceeb4-db2c-4591-8ad6-50b97e1ff17b","collapsed":true,"_uuid":"b5f9ae811a073a5e2105a4e4234615915c09113a","trusted":false},"cell_type":"code","source":"wd2 = tf.Variable(tf.truncated_normal([num_filters_total,2], stddev = 0.05), name = 'wd2')\nbd2 = tf.Variable(tf.truncated_normal([2], stddev = 0.05), name = \"bd2\")\nlayer2 = tf.nn.xw_plus_b(layer1, wd2, bd2, name = 'layer2') \nprediction = tf.nn.sigmoid(layer2)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"bc3b2bfb-dbd8-4961-80cb-c5070530179c","collapsed":true,"_uuid":"717a656d3828826c2f01c31b1ac6bf47b3c49ad4","trusted":false},"cell_type":"code","source":"rmse = tf.reduce_mean(tf.losses.mean_squared_error(predictions = prediction, labels = input_y))\noptimizer = tf.train.AdamOptimizer(learning_rate = 0.0001).minimize(rmse)\n#accuracy = tf.reduce_mean(tf.cast(tf.equal(prediction, input_y), tf.float32))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e7cfee45-72ee-47de-a551-7d936dba7a60","collapsed":true,"_uuid":"05329005522265b88cf7c500bf98b92f90e263e2","trusted":false},"cell_type":"code","source":"def generate_batch(data, batch_size, num_epochs, shuffle=True):\n    data = np.array(data)\n    data_size = len(data)\n    num_batches_per_epoch = int((len(data)-1)/batch_size) + 1\n    l = 0\n    for epoch in range(num_epochs):\n        l += 1\n        if shuffle:\n            shuffle_indices = np.random.permutation(np.arange(data_size))\n            shuffled_data = data[shuffle_indices]\n        else:\n            shuffled_data = data\n        for batch_num in range(num_batches_per_epoch):\n            start_index = batch_num * batch_size\n            end_index = min((batch_num + 1) * batch_size, data_size)\n            yield shuffled_data[start_index:end_index]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ce94f642-0473-4423-bc9a-414baea42820","collapsed":true,"_uuid":"978cc47e2351a63581bab4e7f9cf066d2ba6c1f9","trusted":false},"cell_type":"code","source":"def blocks(data, block_size):\n    data = np.array(data)\n    data_size = len(data)\n    nums = int((data_size-1)/block_size) + 1\n    for block_num in range(nums):\n        if block_num == 0:\n            print(\"prediction start!\")\n        start_index = block_num * block_size\n        end_index = min((block_num + 1) * block_size, data_size)\n        print(end_index)\n        yield data[start_index:end_index]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9a3b5969-c477-4922-b4e3-394bd2c1cad5","collapsed":true,"_uuid":"c525082ae53fe44dbb056a713a05f3b9592b5e7e","trusted":false},"cell_type":"code","source":"batch1 = generate_batch(list(zip(np.array(x_train), np.array(x_train_des), y_train)), batch_size, 1)\nbatch2 = generate_batch(list(zip(np.array(x_train), np.array(x_train_des), y_train)), batch_size, 1)\nbatch3 = generate_batch(list(zip(np.array(x_train), np.array(x_train_des), y_train)), batch_size, 1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7d1a6efe-f91a-4c3b-b2b7-86f160b69be9","collapsed":true,"_uuid":"6da6cb64689b558c290b95bd2d49a03aee0b67de","trusted":false},"cell_type":"code","source":"batch_bag = [batch1,batch2,batch3]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fac3533c-b0ee-46bd-9eb8-532c301f6a2b","collapsed":true,"_uuid":"3f2f0fd9f8c4fa27ce9bdc38336eb928c77d30ea","trusted":false},"cell_type":"code","source":"#batches = generate_batch(list(zip(np.array(train_x), ytrain)), batch_size, 1, shuffle = False)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"99ee01e5-57b2-4b43-9f1c-52aa3e80aefc","collapsed":true,"_uuid":"65fff0778ef520f4817a35eeed60114cd00e3e1e","trusted":false},"cell_type":"code","source":"test_blocks = blocks(list(zip(np.array(x_test), np.array(x_test_des))), 2000)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8c5a62a8-0931-4a5e-a6ef-ce0dedc27d77","collapsed":true,"_uuid":"9a0f8398e6b05bdd76a93dfeda1f492184e26f36","trusted":false},"cell_type":"code","source":"int((len(x_train)-1)/256) + 1","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"139ec0f8-1ca1-4d01-b98e-742afbbe551e","collapsed":true,"_uuid":"eb9da9094d415b6e19ae6d5691a571104bf2084c","trusted":false},"cell_type":"code","source":"int((len(x_test)-1)/2000) + 1","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"76c21c47-a6cd-411f-b2e4-379235e6e445","scrolled":true,"collapsed":true,"_uuid":"78f3ec5904426a2b507e785baed7bd451afb59d0","trusted":false},"cell_type":"code","source":"init_op = tf.global_variables_initializer()\n\nwith tf.Session() as sess:\n    \n    sess.run(init_op)\n    i = 0\n    for batches in batch_bag:\n        i += 1\n        print('Epoch: ' + str(i) + ' start!')\n        avg_rmse = 0\n        for batch in batches:\n            batch = pd.DataFrame(batch, columns = ['a','b','1'])\n            x_batch = pd.DataFrame(list(batch['a']))\n            x_des_batch = pd.DataFrame(list(batch['b']))\n            y_batch = batch.loc[:, batch.columns == '1']\n            y_batch['0'] = 1 - y_batch['1']\n            _,m = sess.run([optimizer, rmse], feed_dict = {input_x: x_batch, input_y: y_batch, input_x_des: x_des_batch})\n            avg_rmse += m\n            #print('pred_train')\n            #print(prediction.eval({input_x: x_batch, input_y: y_batch}))\n        avg_rmse = math.sqrt(avg_rmse/(int((len(x_train)-1)/256) + 1))\n        print('Epoch:' + str(i) + ' rmse is ' + str(avg_rmse))\n    \n    print('Prediction Start!')\n    \n    df = pd.DataFrame()\n    for block in test_blocks:\n        block = pd.DataFrame(block, columns = ['a','b'])\n        x_block = pd.DataFrame(list(block['a']))\n        x_des_block = pd.DataFrame(list(block['b']))\n        pred = sess.run(prediction, feed_dict = {input_x: x_block, input_x_des: x_des_block})\n        df = df.append(pd.DataFrame(pred))\n    \n    print('Finish!') ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5a4a7ca8-9ab3-422f-bb53-0eff89b9f686","collapsed":true,"_uuid":"fb149835f6ab251bd69524418d641a3236beea28","trusted":false},"cell_type":"code","source":"df.round().mean()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"90d5b919-e7e8-423e-9226-eed98dc52d02","collapsed":true,"_uuid":"2fc00311501f73dce0c877913746d59d8f4cf358","trusted":false},"cell_type":"code","source":"df.columns = ['0','1']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5e4333cf-1a2d-49dc-8297-ae790888991f","collapsed":true,"_uuid":"eba8bf592a470fac01440b817ae1f8e207c48293","trusted":false},"cell_type":"code","source":"submission = pd.read_csv('../input/avito-demand-prediction/sample_submission.csv')\nsubmission['deal_probability'] = np.array(df['0'])\nsubmission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}