{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport math\nfrom zipfile import ZipFile\nfrom urllib.request import urlretrieve\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import StringLookup","metadata":{"execution":{"iopub.status.busy":"2022-04-07T08:38:16.631368Z","iopub.execute_input":"2022-04-07T08:38:16.631836Z","iopub.status.idle":"2022-04-07T08:38:22.114905Z","shell.execute_reply.started":"2022-04-07T08:38:16.631796Z","shell.execute_reply":"2022-04-07T08:38:22.114137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',parse_dates=['t_dat'])\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\ncustomers = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-07T08:38:22.11631Z","iopub.execute_input":"2022-04-07T08:38:22.116814Z","iopub.status.idle":"2022-04-07T08:39:33.763447Z","shell.execute_reply.started":"2022-04-07T08:38:22.116781Z","shell.execute_reply":"2022-04-07T08:39:33.762703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['t_dat'] = transactions['t_dat'].astype(int) / 10**9\ntransactions['t_dat']","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:20:22.161443Z","iopub.execute_input":"2022-03-30T06:20:22.161688Z","iopub.status.idle":"2022-03-30T06:20:22.694749Z","shell.execute_reply.started":"2022-03-30T06:20:22.161654Z","shell.execute_reply":"2022-03-30T06:20:22.694046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:20:22.696768Z","iopub.execute_input":"2022-03-30T06:20:22.69705Z","iopub.status.idle":"2022-03-30T06:20:22.716641Z","shell.execute_reply.started":"2022-03-30T06:20:22.697015Z","shell.execute_reply":"2022-03-30T06:20:22.716029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_group = transactions.sort_values(by=[\"t_dat\"]).groupby(\"customer_id\")","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:20:22.717718Z","iopub.execute_input":"2022-03-30T06:20:22.718018Z","iopub.status.idle":"2022-03-30T06:20:25.731258Z","shell.execute_reply.started":"2022-03-30T06:20:22.717983Z","shell.execute_reply":"2022-03-30T06:20:25.730498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data = pd.DataFrame(\n    data={\n        \"customer_id\": list(transactions_group.groups.keys()),\n        \"item_ids\": list(transactions_group.article_id.apply(list)),\n        \"timestamps\": list(transactions_group.t_dat.apply(list)),\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:20:25.733122Z","iopub.execute_input":"2022-03-30T06:20:25.73365Z","iopub.status.idle":"2022-03-30T06:22:00.607355Z","shell.execute_reply.started":"2022-03-30T06:20:25.733611Z","shell.execute_reply":"2022-03-30T06:22:00.606593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transaction_data.to_csv('transaction_ids.csv')\ntransaction_data = transaction_data[:1000]","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:00.60853Z","iopub.execute_input":"2022-03-30T06:22:00.608784Z","iopub.status.idle":"2022-03-30T06:22:00.615991Z","shell.execute_reply.started":"2022-03-30T06:22:00.608753Z","shell.execute_reply":"2022-03-30T06:22:00.615098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\nsequence_length = 4\nstep_size = 2\nitem_list = list(articles.article_id.unique())\ntarget = []\n\ndef create_sequences(values, window_size, step_size):\n    sequences = []\n    temp_target = []\n    start_index = 0\n    i = 0\n    while True:\n        end_index = start_index + window_size\n        seq = values[start_index:end_index]\n        if len(seq) < window_size:\n            seq = values[-window_size:]\n            if len(seq) == window_size:\n                sequences.append(seq)\n                temp_target.append(1)\n            break\n        rand = random.randint(0,1)\n        if rand == 0:\n            seq[-1] = np.random.choice(item_list)\n        temp_target.append(rand)\n        seq = [str(x) for x in seq]\n        sequences.append(seq)\n        start_index += step_size\n    target.append(temp_target)\n    return sequences\n\n\ntransaction_data['item_id'] = transaction_data.item_ids.apply(\n    lambda ids: create_sequences(ids, sequence_length, step_size)\n)\n\ntransaction_data['target'] = target\n\n\ndel transaction_data[\"timestamps\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:00.617351Z","iopub.execute_input":"2022-03-30T06:22:00.617606Z","iopub.status.idle":"2022-03-30T06:22:29.0654Z","shell.execute_reply.started":"2022-03-30T06:22:00.617571Z","shell.execute_reply":"2022-03-30T06:22:29.064637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:29.066526Z","iopub.execute_input":"2022-03-30T06:22:29.068565Z","iopub.status.idle":"2022-03-30T06:22:29.137701Z","shell.execute_reply.started":"2022-03-30T06:22:29.068533Z","shell.execute_reply":"2022-03-30T06:22:29.137009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data_item = transaction_data[[\"customer_id\", \"item_id\",\"target\"]].explode(\n    [\"item_id\",\"target\"], ignore_index=True\n)\ntransaction_data_item","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:29.140668Z","iopub.execute_input":"2022-03-30T06:22:29.140857Z","iopub.status.idle":"2022-03-30T06:22:30.391217Z","shell.execute_reply.started":"2022-03-30T06:22:29.140834Z","shell.execute_reply":"2022-03-30T06:22:30.39047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data_transformed = transaction_data_item.join(\n    customers.set_index(\"customer_id\"), on=\"customer_id\"\n)\ntransaction_data_transformed = transaction_data_transformed.dropna(subset=['item_id'])","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:30.392654Z","iopub.execute_input":"2022-03-30T06:22:30.392923Z","iopub.status.idle":"2022-03-30T06:22:30.951902Z","shell.execute_reply.started":"2022-03-30T06:22:30.392888Z","shell.execute_reply":"2022-03-30T06:22:30.951196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data_transformed.item_id = transaction_data_transformed.item_id.apply(\n    lambda x: \",\".join(str(v) for v in x)\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:30.953299Z","iopub.execute_input":"2022-03-30T06:22:30.953545Z","iopub.status.idle":"2022-03-30T06:22:30.974472Z","shell.execute_reply.started":"2022-03-30T06:22:30.953511Z","shell.execute_reply":"2022-03-30T06:22:30.973247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data_transformed","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:30.97622Z","iopub.execute_input":"2022-03-30T06:22:30.97667Z","iopub.status.idle":"2022-03-30T06:22:31.000137Z","shell.execute_reply.started":"2022-03-30T06:22:30.976512Z","shell.execute_reply":"2022-03-30T06:22:30.999465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del transaction_data_transformed['FN'] \ndel transaction_data_transformed['Active'] \ndel transaction_data_transformed['club_member_status'] \ndel transaction_data_transformed['fashion_news_frequency'] \ndel transaction_data_transformed['postal_code']\n\ntransaction_data_transformed.rename(\n    columns={\"item_id\": \"sequence_item_ids\"},\n    inplace=True,\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.001624Z","iopub.execute_input":"2022-03-30T06:22:31.001951Z","iopub.status.idle":"2022-03-30T06:22:31.011221Z","shell.execute_reply.started":"2022-03-30T06:22:31.001918Z","shell.execute_reply":"2022-03-30T06:22:31.010413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transaction_data_transformed = pd.read_csv('./transaction_data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.012534Z","iopub.execute_input":"2022-03-30T06:22:31.012832Z","iopub.status.idle":"2022-03-30T06:22:31.021082Z","shell.execute_reply.started":"2022-03-30T06:22:31.012798Z","shell.execute_reply":"2022-03-30T06:22:31.020268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_selection = np.random.rand(len(transaction_data_transformed.index)) <= 0.8\ntransaction_data_transformed.to_csv(\"transaction_data.csv\", index=False)\ntransaction_data_transformed = transaction_data_transformed.drop(columns=['customer_id', 'age'])\ntrain_data = transaction_data_transformed[random_selection]\ntest_data = transaction_data_transformed[~random_selection]\n\ntrain_data.to_csv(\"train_data.csv\", index=False, sep=\"|\", header=False)\ntest_data.to_csv(\"test_data.csv\", index=False, sep=\"|\", header=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.022268Z","iopub.execute_input":"2022-03-30T06:22:31.022599Z","iopub.status.idle":"2022-03-30T06:22:31.120742Z","shell.execute_reply.started":"2022-03-30T06:22:31.022563Z","shell.execute_reply":"2022-03-30T06:22:31.120119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transaction_data = pd.read_csv('transaction_ids.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.12186Z","iopub.execute_input":"2022-03-30T06:22:31.122091Z","iopub.status.idle":"2022-03-30T06:22:31.125635Z","shell.execute_reply.started":"2022-03-30T06:22:31.122058Z","shell.execute_reply":"2022-03-30T06:22:31.125005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CSV_HEADER = list(transaction_data_transformed.columns)\nCSV_HEADER","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.126956Z","iopub.execute_input":"2022-03-30T06:22:31.127449Z","iopub.status.idle":"2022-03-30T06:22:31.137147Z","shell.execute_reply.started":"2022-03-30T06:22:31.127415Z","shell.execute_reply":"2022-03-30T06:22:31.136443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CSV_HEADER = list(transaction_data_transformed.columns)\n\nCATEGORICAL_FEATURES_WITH_VOCABULARY = {\n    \"user_id\": list(customers.customer_id.unique()),\n    \"item_id\": list(articles.article_id.unique()),\n    \"age\": list(customers.age.unique()),\n    \"colour_group_name\": list(articles.colour_group_name.unique()),\n    \"product_type_name\": list(articles.product_type_name.unique())\n}\n\nUSER_FEATURES = [\"age\"]\n\nITEM_FEATURES = [\"colour_group_name\",\"product_type_name\"]","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.138403Z","iopub.execute_input":"2022-03-30T06:22:31.138708Z","iopub.status.idle":"2022-03-30T06:22:31.655272Z","shell.execute_reply.started":"2022-03-30T06:22:31.138673Z","shell.execute_reply":"2022-03-30T06:22:31.654508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dataset_from_csv(csv_file_path, shuffle=False, batch_size=128):\n    def process(features):\n        item_ids_string = features[\"sequence_item_ids\"]\n        sequence_item_ids = tf.strings.split(item_ids_string, \",\").to_tensor()\n        target_item_id = []\n        target = features[\"target\"]\n        \n        # The last movie id in the sequence is the target movie.\n            \n        features[\"target_item_id\"] = sequence_item_ids[:, -1]\n        features[\"sequence_item_ids\"] = sequence_item_ids[:, :-1]\n        \n\n        return features,target\n\n    dataset = tf.data.experimental.make_csv_dataset(\n        csv_file_path,\n        batch_size=batch_size,\n        column_names=CSV_HEADER,\n        num_epochs=1,\n        header=False,\n        field_delim=\"|\",\n        shuffle=shuffle,\n    ).map(process)\n\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.656609Z","iopub.execute_input":"2022-03-30T06:22:31.656885Z","iopub.status.idle":"2022-03-30T06:22:31.698061Z","shell.execute_reply.started":"2022-03-30T06:22:31.656848Z","shell.execute_reply":"2022-03-30T06:22:31.697253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model_inputs():\n    return {\n#         \"customer_id\": layers.Input(name=\"user_id\", shape=(1,), dtype=tf.string),\n        \"sequence_item_ids\": layers.Input(\n            name=\"sequence_item_ids\", shape=(sequence_length - 1,), dtype=tf.string\n        ),\n        \"target_item_id\": layers.Input(\n            name=\"target_item_id\", shape=(1,), dtype=tf.string\n        ),\n#         \"age\": layers.Input(name=\"age\", shape=(1,), dtype=tf.string),\n#         \"colour_group_name_target\": layers.Input(name=\"colour_group_name_target\", shape=(1,), dtype=tf.string),\n#         \"product_type_name_target\": layers.Input(name=\"product_type_name_target\", shape=(1,), dtype=tf.string),\n#         \"colour_group_name_sequence\": layers.Input(name=\"colour_group_name_sequence\", shape=(sequence_length - 1,), dtype=tf.string),\n#         \"product_type_name_sequence\": layers.Input(name=\"product_type_name_sequence\", shape=(sequence_length - 1,), dtype=tf.string),\n    }","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.699476Z","iopub.execute_input":"2022-03-30T06:22:31.700338Z","iopub.status.idle":"2022-03-30T06:22:31.708336Z","shell.execute_reply.started":"2022-03-30T06:22:31.700289Z","shell.execute_reply":"2022-03-30T06:22:31.70763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_input_features(\n    inputs,\n    include_user_id=True,\n    include_user_features=True,\n    include_movie_features=True,\n):\n\n    encoded_transformer_features = []\n    encoded_other_features = []\n\n    other_feature_names = []\n    if include_user_id:\n        other_feature_names.append(\"user_id\")\n    if include_user_features:\n        other_feature_names.extend(USER_FEATURES)\n\n    ## Encode user features\n    for feature_name in other_feature_names:\n        # Convert the string input values into integer indices.\n        vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[feature_name]\n        idx = StringLookup(vocabulary=vocabulary, mask_token=None, num_oov_indices=0)(\n            inputs[feature_name]\n        )\n        # Compute embedding dimensions\n        embedding_dims = int(math.sqrt(len(vocabulary)))\n        # Create an embedding layer with the specified dimensions.\n        embedding_encoder = layers.Embedding(\n            input_dim=len(vocabulary),\n            output_dim=embedding_dims,\n            name=f\"{feature_name}_embedding\",\n        )\n        # Convert the index values to embedding representations.\n        encoded_other_features.append(embedding_encoder(idx))\n\n    ## Create a single embedding vector for the user features\n    if len(encoded_other_features) > 1:\n        encoded_other_features = layers.concatenate(encoded_other_features)\n    elif len(encoded_other_features) == 1:\n        encoded_other_features = encoded_other_features[0]\n    else:\n        encoded_other_features = None\n\n    ## Create a movie embedding encoder\n    item_vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[\"item_id\"]\n    item_embedding_dims = int(math.sqrt(len(item_vocabulary)))\n    # Create a lookup to convert string values to integer indices.\n    item_index_lookup = StringLookup(\n        vocabulary=[str(x) for x in item_vocabulary],\n        mask_token=None,\n        num_oov_indices=0,\n        name=\"item_index_lookup\",\n    )\n    # Create an embedding layer with the specified dimensions.\n    item_embedding_encoder = layers.Embedding(\n        input_dim=len(item_vocabulary),\n        output_dim=item_embedding_dims,\n        name=\"item_embedding\",\n    )\n    \n    \n    item_colour_vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[\"colour_group_name\"]\n    item_colour_embedding_dims = int(math.sqrt(len(item_colour_vocabulary)))\n    # Create a lookup to convert string values to integer indices.\n    item_colour_index_lookup = StringLookup(\n        vocabulary=item_colour_vocabulary,\n        mask_token=None,\n        num_oov_indices=0,\n        name=\"item_colour_index_lookup\",\n    )\n    # Create an embedding layer with the specified dimensions.\n    item_colour_embedding_encoder = layers.Embedding(\n        input_dim=len(item_colour_vocabulary),\n        output_dim=item_colour_embedding_dims,\n        name=\"item_colour_embedding\",\n    )\n    \n    item_type_vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[\"product_type_name\"]\n    item_type_embedding_dims = int(math.sqrt(len(item_type_vocabulary)))\n    # Create a lookup to convert string values to integer indices.\n    item_type_index_lookup = StringLookup(\n        vocabulary=item_type_vocabulary,\n        mask_token=None,\n        num_oov_indices=0,\n        name=\"item_type_index_lookup\",\n    )\n    # Create an embedding layer with the specified dimensions.\n    item_type_embedding_encoder = layers.Embedding(\n        input_dim=len(item_type_vocabulary),\n        output_dim=item_type_embedding_dims,\n        name=\"item_type_embedding\",\n    )\n    \n    # Create a processing layer for genres.\n    item_embedding_processor = layers.Dense(\n        units=item_embedding_dims,\n        activation=\"relu\",\n        name=\"process_item_embedding_with_other_feature\",\n    )\n\n    ## Define a function to encode a given movie id.\n    def encode_item(item_id,colour_id=None,type_id=None):\n        # Convert the string input values into integer indices.\n        item_idx = item_index_lookup(item_id)\n        item_embedding = item_embedding_encoder(item_idx)\n        encoded_item = item_embedding\n        if include_movie_features:\n            idx_colour = item_colour_index_lookup(colour_id)\n            item_colour_vector = item_colour_embedding_encoder(idx_colour)\n            idx_type = item_type_index_lookup(type_id)\n            item_type_vector = item_type_embedding_encoder(idx_type)\n            encoded_item = item_embedding_processor(\n                layers.concatenate([item_embedding, item_type_vector, item_colour_vector])\n            )\n        return encoded_item\n\n    ## Encoding target_movie_id\n    target_item_id = inputs[\"target_item_id\"]\n#     target_item_colour = inputs[\"colour_group_name_target\"]\n#     target_item_type = inputs[\"product_type_name_target\"]\n#     encoded_target_item = encode_item(target_item_id, target_item_colour, target_item_type)\n    encoded_target_item = encode_item(target_item_id)\n\n    ## Encoding sequence movie_ids.\n    sequence_item_ids = inputs[\"sequence_item_ids\"]\n#     sequence_item_colour = inputs[\"colour_group_name_sequence\"]\n#     sequence_item_type = inputs[\"product_type_name_sequence\"]\n#     encoded_sequence_items = encode_item(sequence_item_ids, sequence_item_colour, sequence_item_type)\n    encoded_sequence_items = encode_item(sequence_item_ids)\n    # Create positional embedding.\n    position_embedding_encoder = layers.Embedding(\n        input_dim=sequence_length,\n        output_dim=item_embedding_dims,\n        name=\"position_embedding\",\n    )\n    positions = tf.range(start=0, limit=sequence_length - 1, delta=1)\n    encodded_positions = position_embedding_encoder(positions)\n    # Add the positional encoding to the movie encodings and multiply them by rating.\n    encoded_sequence_items_with_poistion = encoded_sequence_items + encodded_positions\n\n    # Construct the transformer inputs.\n    for encoded_item in tf.unstack(\n        encoded_sequence_items_with_poistion, axis=1\n    ):\n        encoded_transformer_features.append(tf.expand_dims(encoded_item, 1))\n    encoded_transformer_features.append(encoded_target_item)\n\n    encoded_transformer_features = layers.concatenate(\n        encoded_transformer_features, axis=1\n    )\n\n    return encoded_transformer_features, encoded_other_features","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.711304Z","iopub.execute_input":"2022-03-30T06:22:31.711544Z","iopub.status.idle":"2022-03-30T06:22:31.733779Z","shell.execute_reply.started":"2022-03-30T06:22:31.711512Z","shell.execute_reply":"2022-03-30T06:22:31.733026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"include_user_id = False\ninclude_user_features = False\ninclude_movie_features = False\n\nhidden_units = [512, 256]\ndropout_rate = 0.1\nnum_heads = 3\nsequence_length = 4\n\ndef create_model():\n    inputs = create_model_inputs()\n    transformer_features, other_features = encode_input_features(\n        inputs, include_user_id, include_user_features, include_movie_features\n    )\n\n    # Create a multi-headed attention layer.\n    attention_output = layers.MultiHeadAttention(\n        num_heads=num_heads, key_dim=transformer_features.shape[2], dropout=dropout_rate\n    )(transformer_features, transformer_features)\n\n    # Transformer block.\n    attention_output = layers.Dropout(dropout_rate)(attention_output)\n    x1 = layers.Add()([transformer_features, attention_output])\n    x1 = layers.LayerNormalization()(x1)\n    x2 = layers.LeakyReLU()(x1)\n    x2 = layers.Dense(units=x2.shape[-1])(x2)\n    x2 = layers.Dropout(dropout_rate)(x2)\n    transformer_features = layers.Add()([x1, x2])\n    transformer_features = layers.LayerNormalization()(transformer_features)\n    features = layers.Flatten()(transformer_features)\n\n    # Included the other features.\n    if other_features is not None:\n        features = layers.concatenate(\n            [features, layers.Reshape([other_features.shape[-1]])(other_features)]\n        )\n\n    # Fully-connected layers.\n    for num_units in hidden_units:\n        features = layers.Dense(num_units)(features)\n        features = layers.BatchNormalization()(features)\n        features = layers.LeakyReLU()(features)\n        features = layers.Dropout(dropout_rate)(features)\n\n    outputs = layers.Dense(units=1,activation='sigmoid')(features)\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model\n\n\nmodel = create_model()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:31.735268Z","iopub.execute_input":"2022-03-30T06:22:31.735572Z","iopub.status.idle":"2022-03-30T06:22:35.244643Z","shell.execute_reply.started":"2022-03-30T06:22:31.735532Z","shell.execute_reply":"2022-03-30T06:22:35.243951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model.\nmodel.compile(\n    optimizer=keras.optimizers.Adagrad(learning_rate=0.01),\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    metrics=['accuracy'],\n)\n\n# Read the training data.\ntrain_dataset = get_dataset_from_csv(\"train_data.csv\", shuffle=True, batch_size=256)\n\n# Fit the model with the training data.\nmodel.fit(train_dataset, epochs=5)\n\n# Read the test data.\ntest_dataset = get_dataset_from_csv(\"test_data.csv\", batch_size=265)\n\n# Evaluate the model on the test data.\n_, acc = model.evaluate(test_dataset, verbose=0)\nprint(f\"Test acc: {round(acc, 3)}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-30T06:22:35.245722Z","iopub.execute_input":"2022-03-30T06:22:35.24597Z","iopub.status.idle":"2022-03-30T06:23:11.13904Z","shell.execute_reply.started":"2022-03-30T06:22:35.245937Z","shell.execute_reply":"2022-03-30T06:23:11.138249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_item_list = transactions.article_id.value_counts().index[:1000]","metadata":{"execution":{"iopub.status.busy":"2022-04-07T08:47:24.00678Z","iopub.execute_input":"2022-04-07T08:47:24.007282Z","iopub.status.idle":"2022-04-07T08:47:24.644087Z","shell.execute_reply.started":"2022-04-07T08:47:24.007237Z","shell.execute_reply":"2022-04-07T08:47:24.643347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2022-04-07T08:48:23.458716Z","iopub.execute_input":"2022-04-07T08:48:23.459388Z","iopub.status.idle":"2022-04-07T08:48:27.844394Z","shell.execute_reply.started":"2022-04-07T08:48:23.459349Z","shell.execute_reply":"2022-04-07T08:48:27.843717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}