{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 3D convolution for tabular data ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\n\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import Input, Activation, Dense, concatenate\nfrom tensorflow.keras.layers import Conv3D, Flatten, BatchNormalization,LeakyReLU\n\nglobalSeed=768\n\nfrom numpy.random import seed \nseed(globalSeed)\n\ntf.compat.v1.set_random_seed(globalSeed)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-25T08:23:02.582153Z","iopub.execute_input":"2022-08-25T08:23:02.582517Z","iopub.status.idle":"2022-08-25T08:23:02.610588Z","shell.execute_reply.started":"2022-08-25T08:23:02.582486Z","shell.execute_reply":"2022-08-25T08:23:02.609442Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############################################################################\n# Metrics\n# from https://www.kaggle.com/code/rohanrao/amex-competition-metric-implementations\n###############################################################################\n\ndef amex_metric_tensorflow(y_true, y_pred):\n\n    # convert dtypes to float64\n    y_true = tf.cast(y_true, dtype=tf.float64)\n    y_pred = tf.cast(y_pred, dtype=tf.float64)\n\n    # count of positives and negatives\n    n_pos = tf.math.reduce_sum(y_true)\n    n_neg = tf.cast(tf.shape(y_true)[0], dtype=tf.float64) - n_pos\n\n    # sorting by descring prediction values\n    indices = tf.argsort(y_pred, axis=0, direction='DESCENDING')\n    preds, target = tf.gather(y_pred, indices), tf.gather(y_true, indices)\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = tf.cumsum(weight / tf.reduce_sum(weight))\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = tf.reduce_sum(target[four_pct_filter]) / n_pos\n\n    # weighted gini coefficient\n    lorentz = tf.cumsum(target / n_pos)\n    gini = tf.reduce_sum((lorentz - cum_norm_weight) * weight)\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n\n###############################################################################\n# Loading packages \n###############################################################################\n\n#Wrapper function to make the basic convolutional block \ndef Make3DConvolutionBlock(X, Convolutions):\n    \n    X = Conv3D(Convolutions, (1,1,1), \n                padding='same',\n                use_bias=False)(X)\n    X = BatchNormalization()(X)\n    X = LeakyReLU()(X)\n    \n    return X\n\n#Wrapper function to make the dense convolutional block\ndef MakeDenseBlock(x, Convolutions,Depth,MakeBlock):\n\n    concat_feat= x\n    for i in range(Depth):\n        x = MakeBlock(concat_feat,Convolutions)\n        concat_feat=concatenate([concat_feat,x])\n\n    return concat_feat\n\n#Wraper function creates a dense convolutional block and resamples the data \ndef SamplingBlock(X,Units,Depth,BlockFunction,strids):\n    \n    X = MakeDenseBlock(X,Units,Depth,BlockFunction)\n    X = Conv3D(Units,(2,2,2),\n               strides=strids,\n               padding='same',\n               use_bias=False)(X)    \n    X = BatchNormalization()(X)\n    X = LeakyReLU()(X)\n    \n    return X \n\ndef MakeClassifier(InputShape,Convolutions,Depth,BlockFunction):\n    \n    InputFunction = Input(shape=InputShape)\n    \n    X = SamplingBlock(InputFunction,Convolutions[0],2*Depth,BlockFunction,(2,1,1))\n    X = SamplingBlock(X,Convolutions[0],Depth,BlockFunction,(1,2,2))\n    X = SamplingBlock(X,Convolutions[1],Depth,BlockFunction,(2,1,1))\n    X = SamplingBlock(X,Convolutions[1],Depth//2,BlockFunction,(1,2,2))\n    X = SamplingBlock(X,Convolutions[2],Depth//2,BlockFunction,(2,1,1))\n    X = SamplingBlock(X,Convolutions[2],Depth//4,BlockFunction,(1,2,2))\n    \n    Intermediate = Flatten()(X)\n\n    Output = Dense(1,use_bias=False)(Intermediate)\n    Output = BatchNormalization()(Output)\n    Output = Activation('sigmoid')(Output)\n    \n    intermediateModel = Model(inputs=InputFunction,outputs=Intermediate)\n    outputModel = Model(inputs=InputFunction,outputs=Output)\n    \n    return intermediateModel,outputModel","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-25T08:21:07.434020Z","iopub.execute_input":"2022-08-25T08:21:07.434378Z","iopub.status.idle":"2022-08-25T08:21:07.458348Z","shell.execute_reply.started":"2022-08-25T08:21:07.434344Z","shell.execute_reply":"2022-08-25T08:21:07.457453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labelsData = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\nlabelsData = labelsData.set_index('customer_ID')\ntrain_index,test_index,train_labels,test_labels = train_test_split(labelsData.index,labelsData['target'],test_size=0.05,random_state=23)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:07.460017Z","iopub.execute_input":"2022-08-25T08:21:07.460804Z","iopub.status.idle":"2022-08-25T08:21:08.338976Z","shell.execute_reply.started":"2022-08-25T08:21:07.460751Z","shell.execute_reply":"2022-08-25T08:21:08.337926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matrixDataDir = '../input/americanexpresstraindata/train'\n\ntraindata = [np.load(matrixDataDir+'/'+val+'.npy') for val in train_index]\ntraindata = np.array(traindata)\n\ntestdata = [np.load(matrixDataDir+'/'+val+'.npy') for val in test_index]\ntestdata = np.array(testdata)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:08.344973Z","iopub.execute_input":"2022-08-25T08:21:08.347610Z","iopub.status.idle":"2022-08-25T08:21:10.263416Z","shell.execute_reply.started":"2022-08-25T08:21:08.347569Z","shell.execute_reply":"2022-08-25T08:21:10.262315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = 0.005\nminlr = 0.0005\nepochs = 50\nbatch_size = 64\ndecay = (lr-minlr)/epochs\ninputShape = (13, 8, 8, 3)\nconvolutions = [4,8,1]\ndepth = 4","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:10.265390Z","iopub.execute_input":"2022-08-25T08:21:10.266208Z","iopub.status.idle":"2022-08-25T08:21:10.273021Z","shell.execute_reply.started":"2022-08-25T08:21:10.266168Z","shell.execute_reply":"2022-08-25T08:21:10.271837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datasetTrain = tf.data.Dataset.from_tensor_slices((traindata, train_labels))\ndatasetTrain = datasetTrain.batch(batch_size)\ndatasetTrain = datasetTrain.shuffle(buffer_size=100,seed=125)\ndatasetTrain = datasetTrain.prefetch(tf.data.experimental.AUTOTUNE)\n    \ndatasetTest = tf.data.Dataset.from_tensor_slices((testdata, test_labels))\ndatasetTest = datasetTest.batch(batch_size)\ndatasetTest = datasetTest.shuffle(buffer_size=100,seed=125)\ndatasetTest = datasetTest.prefetch(tf.data.experimental.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:47.847144Z","iopub.execute_input":"2022-08-25T08:21:47.847528Z","iopub.status.idle":"2022-08-25T08:21:47.937403Z","shell.execute_reply.started":"2022-08-25T08:21:47.847495Z","shell.execute_reply":"2022-08-25T08:21:47.936497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"intermediateModel, Classifier = MakeClassifier(inputShape,convolutions,depth,Make3DConvolutionBlock)\nClassifier.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:51.257353Z","iopub.execute_input":"2022-08-25T08:21:51.257907Z","iopub.status.idle":"2022-08-25T08:21:51.850154Z","shell.execute_reply.started":"2022-08-25T08:21:51.257872Z","shell.execute_reply":"2022-08-25T08:21:51.849210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Classifier.compile(Adam(learning_rate=lr,decay=decay),loss='binary_crossentropy',metrics=['accuracy',amex_metric_tensorflow])\nClassifier.fit(datasetTrain,batch_size=batch_size,epochs=epochs,validation_data=datasetTest)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:21:59.776112Z","iopub.execute_input":"2022-08-25T08:21:59.776685Z","iopub.status.idle":"2022-08-25T08:22:16.903386Z","shell.execute_reply.started":"2022-08-25T08:21:59.776648Z","shell.execute_reply":"2022-08-25T08:22:16.902508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rep = intermediateModel.predict(testdata)\nplt.figure()\nplt.scatter(rep[:,0],rep[:,1],c=test_labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T08:23:38.578382Z","iopub.execute_input":"2022-08-25T08:23:38.578785Z","iopub.status.idle":"2022-08-25T08:23:38.945873Z","shell.execute_reply.started":"2022-08-25T08:23:38.578732Z","shell.execute_reply":"2022-08-25T08:23:38.944992Z"},"trusted":true},"execution_count":null,"outputs":[]}]}