{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/35887/logos/header.png?t=2022-05-09-22-33-02\">\n\n","metadata":{}},{"cell_type":"code","source":"!pip install jieba --no-index --find-links=file:///kaggle/input/jiebads","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:23.028261Z","iopub.execute_input":"2022-08-12T07:46:23.028892Z","iopub.status.idle":"2022-08-12T07:46:34.939498Z","shell.execute_reply.started":"2022-08-12T07:46:23.028785Z","shell.execute_reply":"2022-08-12T07:46:34.938295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\nfrom typing import List\nimport jieba\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport transformers\nfrom tqdm.notebook import tqdm\n\nfrom matplotlib import pyplot as plt\n\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\n\ndebug = True\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:34.942777Z","iopub.execute_input":"2022-08-12T07:46:34.944028Z","iopub.status.idle":"2022-08-12T07:46:41.838071Z","shell.execute_reply.started":"2022-08-12T07:46:34.943992Z","shell.execute_reply":"2022-08-12T07:46:41.837127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_notebook(path: str) -> pd.DataFrame:\n    return (\n        pd.read_json(path, dtype={\"cell_type\": \"category\", \"source\": \"str\"})\n        .assign(id=os.path.basename(path).split(\".\")[0])\n        .rename_axis(\"cell_id\")\n    )\n\n\n\ndef get_ranks(base: pd.Series, derived: List[str]) -> List[str]:\n    return [base.index(d) for d in derived]\n\n\ndef get_dataset(\n    input_ids: np.array,\n    attention_mask: np.array,\n    feature: np.array,\n) -> tf.data.Dataset:\n    dataset = tf.data.Dataset.from_tensor_slices(\n        {\"input_ids\": input_ids, \"attention_mask\": attention_mask, \"feature\": feature}\n    )\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset.prefetch(tf.data.AUTOTUNE)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-12T07:46:41.839334Z","iopub.execute_input":"2022-08-12T07:46:41.839965Z","iopub.status.idle":"2022-08-12T07:46:41.852846Z","shell.execute_reply.started":"2022-08-12T07:46:41.839926Z","shell.execute_reply":"2022-08-12T07:46:41.851904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SparseDense(tf.keras.layers.Layer):\n  def __init__(self, dense_shape,  odim):\n    super(SparseDense, self).__init__()\n    self.odim = odim\n    self.dense_shape = dense_shape\n\n  def build(self, input_shape):\n    self.kernel = self.add_weight(\"kernel\",\n                                  shape=[ self.dense_shape[1],\n                                         self.odim])\n\n  def call(self, inputs):\n\n    I, V = inputs\n\n\n    S = tf.sparse.SparseTensor(I, V,  dense_shape=[2*4096, TFID_DIM])\n\n    X = tf.sparse.sparse_dense_matmul(S, self.kernel)\n\n\n    return X\n\n  def get_config(self):\n    return {\"odim\": self.odim, 'dense_shape': self.dense_shape}\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:41.856240Z","iopub.execute_input":"2022-08-12T07:46:41.856775Z","iopub.status.idle":"2022-08-12T07:46:42.783119Z","shell.execute_reply.started":"2022-08-12T07:46:41.856737Z","shell.execute_reply":"2022-08-12T07:46:42.782066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass PosEmbeddingLayer(tf.keras.layers.Layer):\n  def __init__(self,  odim):\n    super(PosEmbeddingLayer, self).__init__()\n    self.odim = odim\n \n    self.dense = layers.Dense(odim)\n\n    self.t = tf.constant( tf.range(1,1024, delta=1, dtype=tf.float32)[tf.newaxis,tf.newaxis, :] )\n\n\n  def build(self, input_shape):\n    pass\n\n  def call(self, inputs):\n    cr, cm = inputs\n\n    ncode = tf.math.reduce_sum(cm, 1, keepdims=True)\n\n    xpos= tf.concat(  [\n            tf.math.cos( self.t*cr*np.pi/2 )/(self.t)**1*( ncode**2 / ( self.t**2 + ncode**2 ) ),\n            tf.math.sin( self.t*cr*np.pi/2 )/(self.t)**1*( ncode**2 / ( self.t**2 + ncode**2 ) )\n              ], -1 )\n    \n    xpos = self.dense(xpos)\n\n\n    return xpos\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:42.784624Z","iopub.execute_input":"2022-08-12T07:46:42.785034Z","iopub.status.idle":"2022-08-12T07:46:42.796393Z","shell.execute_reply.started":"2022-08-12T07:46:42.784994Z","shell.execute_reply":"2022-08-12T07:46:42.795247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_sparse(V, W, B, D, dim=256, nw=200):\n\n\n    o1 = tf.ones_like(W)\n\n    R = tf.cast(tf.shape(W)[1], dtype=tf.int64)*tf.cast(tf.cumsum(o1, 0, exclusive=True), dtype=tf.int64) +  tf.cast(tf.cumsum(o1, 1, exclusive=True), dtype=tf.int64)\n\n    V = tf.reshape(V, [-1])\n    W = tf.reshape(W, [-1,1])\n\n    R = R\n    R = tf.reshape(R, [B,D,nw])\n\n\n    R = tf.reshape(R, [-1,1])\n\n    I = tf.concat((R,W), axis=1)\n\n    return I, V\n\n\ndef do_sparse(I, V, B, D, dim=256, nw=200):\n\n\n    sd = SparseDense([ 1, TFID_DIM], dim)\n\n\n    x = sd([I, V])\n\n    x = x[:B*D, :]\n\n\n    x = tf.reshape(x, [B,D,dim])\n\n    return x\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:42.798105Z","iopub.execute_input":"2022-08-12T07:46:42.798756Z","iopub.status.idle":"2022-08-12T07:46:42.815346Z","shell.execute_reply.started":"2022-08-12T07:46:42.798705Z","shell.execute_reply":"2022-08-12T07:46:42.814275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.python.framework import sparse_tensor\nfrom tensorflow.keras import layers\n\n\nval_inp = tf.keras.layers.Input(\n    shape=( None, 200 ),\n    dtype=tf.float32,\n    name=\"val_input\",\n)\n\nword_inp = tf.keras.layers.Input(\n    shape=( None, 200 ),\n    dtype=tf.int64,\n    name=\"word_input\",\n)\n\ncode_rank_inp = tf.keras.layers.Input(\n    shape=( None, 1 ),\n    dtype=tf.float32,\n    name=\"code_rank_input\",\n)\n\nis_code_inp = tf.keras.layers.Input(\n    shape=( None, 1 ),\n    dtype=tf.float32,\n    name=\"code_is_input\",\n)\n\nattention_mask = tf.keras.layers.Input(\n    shape=(None,),\n    dtype=tf.float32,\n    name=\"attention_mask\",\n)\n\n\nsims_inp = tf.keras.layers.Input(\n    shape=(None,None,2),\n    dtype=tf.float32,\n    name=\"sims_inp\",\n)\n\n\nsims = sims_inp[:, tf.newaxis, :, :]\n\n\nE = tf.eye(tf.shape(sims)[2], dtype=tf.float32)[tf.newaxis,:,:]\n\nE = tf.tile(E, (tf.shape(sims)[0], 1, 1) )[:, tf.newaxis, :,:, tf.newaxis]\n\n\nsims = tf.concat( [sims, E ], -1)\n\nprint('sims' , sims)\n\nV = val_inp\nW = word_inp\ncr = code_rank_inp\ncm = is_code_inp\n\n\nB = tf.cast( tf.shape(V)[0], dtype=tf.int64)\nD = tf.cast( tf.shape(V)[1], dtype=tf.int64)\n\nV = layers.Dropout(.4)(V)\n\n\nI,V = prepare_sparse(V, W, B, D, dim=768, nw=200)\n\nx = do_sparse(I, V, B, D, dim=768, nw=200)\n\n\n\n\nx = cm * layers.LayerNormalization()(x) + ( 1 - cm ) * layers.LayerNormalization()(x)\n\nx = (    cm *layers.Dense(768)(x)\n    + (1-cm)*layers.Dense(768)(x)\n    +        layers.Dense(768)(x) )\n\n\nxpos = PosEmbeddingLayer(768)([cr,cm])\nxpos = tf.transpose(xpos, [0,2,1])\nxpos = layers.SpatialDropout1D(.25)(xpos)\nxpos = tf.transpose(xpos, [0,2,1])\nxpos = layers.LayerNormalization()(xpos)\n\n#x = layers.Dropout(.3)(x)\n\n\nx = x + cm*xpos\nx = cm * layers.LayerNormalization()(x) + ( 1 - cm ) * layers.LayerNormalization()(x)\n\n\ncell_cell_mask = 1000*(  attention_mask[ :, tf.newaxis, tf.newaxis, : ]\n                       * attention_mask[ :, tf.newaxis, :, tf.newaxis ] )\n\n\n\ndef augattn3(xq, xk, xa=None, msk=None, heads=16, dr=.1, augdim=8):\n    B = tf.shape(xq)[ 0]\n    C = xq.shape[-1]\n\n    S = tf.shape(xa)[ 2]\n\n\n    dim = C//heads\n\n\n\n\n    qq = layers.Dense(dim*heads)(xq)\n    kk = layers.Dense(dim*heads)(xk)\n    vv = layers.Dense(dim*heads)(xk)\n    aq = layers.Dense(    heads*xa.shape[-1]*augdim)(xq)\n    ak = layers.Dense(    heads*xa.shape[-1]*augdim)(xk)\n\n  \n    qq = tf.reshape(qq, [B, -1, heads, dim])\n    kk = tf.reshape(kk, [B, -1, heads, dim])\n    vv = tf.reshape(vv, [B, -1, heads, dim])\n\n  \n    aq = tf.reshape(aq, [B, -1, heads*xa.shape[-1], augdim])\n    ak = tf.reshape(ak, [B, -1, heads*xa.shape[-1], augdim])\n\n\n\n    qq = tf.transpose(qq, [0,2,1,3])\n    kk = tf.transpose(kk, [0,2,1,3])\n    vv = tf.transpose(vv, [0,2,1,3])\n    aq = tf.transpose(aq, [0,2,1,3])\n    ak = tf.transpose(ak, [0,2,1,3])\n\n\n    scores = tf.matmul(qq, kk, transpose_b=True)\n    scores = scores/dim**0.5\n\n\n    aug = tf.matmul(aq, ak, transpose_b=True)\n    aug = aug/augdim**0.5\n\n    aug = tf.reshape(aug, [B, heads, xa.shape[-1], S, S ])\n\n    aug = tf.transpose(aug, [0, 1,3,4, 2,])\n\n\n\n    scores = scores + msk + tf.math.reduce_sum( aug*xa, -1)\n\n    scores = tf.nn.softmax(scores, -1)\n\n    print('scores', scores)\n\n    scores = layers.Dropout(dr)(scores)\n\n    rr = tf.matmul(scores, vv)\n    rr = tf.transpose(rr, [0, 2, 1, 3])\n    rr = tf.reshape(rr, [B,-1,C])\n\n    return rr\n\n\n\n\ndef post_attn(x):\n\n  x1 = x\n\n  x1 = layers.Dense(3072)(x1)\n  x1 = layers.Activation('gelu')(x1)\n  x1 = layers.Dropout(.1)(x1)\n  x1 = layers.Dense(768)(x1)\n\n  x = layers.LayerNormalization()(x+x1)\n\n  return x\n\n\nx = layers.LayerNormalization()(x)\n\nN_Layers = 12\n\nxc = x\nfor k in range(N_Layers):\n\n\n  xca1 = augattn3(xc, xc, sims, cell_cell_mask, 12)\n\n  xc = xc +  layers.Dense(768)( xca1 )\n  xc = layers.LayerNormalization()( xc )\n\n  xc = post_attn( xc )\n\nx = xc\n\nx = layers.Dense(2)(x)\n\nx = layers.Concatenate()([x, attention_mask[:,:,tf.newaxis], is_code_inp])\n\nmodel = tf.keras.Model(\n    inputs=[val_inp, word_inp, code_rank_inp,\n            attention_mask, is_code_inp,\n            sims_inp],\n    outputs=x,\n)\n\nmodel.summary()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:47:46.591726Z","iopub.execute_input":"2022-08-12T07:47:46.592142Z","iopub.status.idle":"2022-08-12T07:47:46.770327Z","shell.execute_reply.started":"2022-08-12T07:47:46.592100Z","shell.execute_reply":"2022-08-12T07:47:46.768851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Collect Data","metadata":{}},{"cell_type":"code","source":"paths = glob.glob(os.path.join('../input/AI4Code', \"test\", \"*.json\"))\ndf = (\n    pd.concat([read_notebook(x) for x in tqdm(paths, desc=\"Concat\")])\n    .set_index(\"id\", append=True)\n    .swaplevel()\n    .sort_index(level=\"id\", sort_remaining=False)\n).reset_index()\ndf[\"source\"] = df[\"source\"]\ndf[\"rank\"] = df.groupby([\"id\", \"cell_type\"]).cumcount()\ndf[\"pct_rank\"] = df.groupby([\"id\", \"cell_type\"])[\"rank\"].rank(pct=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:48:55.783061Z","iopub.execute_input":"2022-08-12T07:48:55.785273Z","iopub.status.idle":"2022-08-12T07:48:55.882758Z","shell.execute_reply.started":"2022-08-12T07:48:55.785227Z","shell.execute_reply":"2022-08-12T07:48:55.881875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run Inference","metadata":{}},{"cell_type":"code","source":"import pickle as pkl\n\n\n\nmodel = tf.keras.models.load_model('../input/full-model', custom_objects={\"hnmaeM\": None, \"hloss\":None})\n#model.summary()\n\nvectorizers = pkl.load(open('../input/ai4csmallvectorized/vectorizers.pkl', 'rb'))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:49:12.976593Z","iopub.execute_input":"2022-08-12T07:49:12.977324Z","iopub.status.idle":"2022-08-12T07:49:53.659659Z","shell.execute_reply.started":"2022-08-12T07:49:12.977283Z","shell.execute_reply":"2022-08-12T07:49:53.658553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def centered_ngram_fun(subdf):\n    \n    \n\n    ic = subdf['cell_type'].values == 'code'\n\n    tfidf = CountVectorizer( min_df=2, max_features=8000, ngram_range=(2,5), analyzer='char', \n                                token_pattern=r\".\")\n\n    try:\n        X_vec = tfidf.fit_transform(subdf['source'])\n    except:\n        X_vec = sp.csr_matrix(( subdf.shape[0],0))\n\n    X_vec.data =  X_vec.data**0.5\n\n    im= 1-ic\n\n    X_vec = np.array(X_vec.todense())\n\n\n    sbc = np.sum( X_vec*ic[:, np.newaxis], 0, keepdims=True )/np.sum(ic)\n    sbm = np.sum( X_vec*im[:, np.newaxis], 0, keepdims=True )/np.sum(im)\n\n    X_vec = X_vec - ( ic[:,np.newaxis]*sbc + im[:,np.newaxis]*sbm)\n\n\n    sbc = np.sum( X_vec**2*ic[:, np.newaxis], 0, keepdims=True )/np.sum(ic)\n    sbm = np.sum( X_vec**2*im[:, np.newaxis], 0, keepdims=True )/np.sum(im)\n\n    X_vec = X_vec / ( ic[:,np.newaxis]*sbc + im[:,np.newaxis]*sbm + 1e-8)**0.5\n\n\n\n    ss = (sbc+2)*(sbm+2)\n    ss = sbc+sbm\n\n    X_vec = X_vec - np.mean(X_vec,1,keepdims=True)\n    X_vec = X_vec/(np.sum(X_vec**2, 1,keepdims=True) + .0001)**0.5\n\n\n    Z = np.matmul(X_vec,X_vec.T)\n    Z = np.array(Z)\n    Z = Z - Z*np.eye(Z.shape[0])\n\n    \n    return Z\n\n\ndef ngram_fun(subdf):\n\n\n    tfidf = TfidfVectorizer( min_df=2, max_features=8000, ngram_range=(3,6), analyzer='char', \n                                token_pattern=r\".\")\n\n    try:\n        X_vec = tfidf.fit_transform(subdf['source'])\n    except:\n        X_vec = sp.csr_matrix(( subdf.shape[0],0))\n\n\n\n    Z = np.dot(X_vec,X_vec.T).todense()\n\n    Z = np.array(Z)\n\n    Z = Z - Z*np.eye(Z.shape[0])\n\n    return Z\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:49:53.662285Z","iopub.execute_input":"2022-08-12T07:49:53.662758Z","iopub.status.idle":"2022-08-12T07:49:53.683998Z","shell.execute_reply.started":"2022-08-12T07:49:53.662716Z","shell.execute_reply":"2022-08-12T07:49:53.682774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse as sp\n\n\n\nsamp = pd.read_csv('../input/AI4Code/sample_submission.csv')\nsamp = samp.set_index('id')\n\n\n\nfor chunk in tqdm(df.groupby('id'), total=len(df.groupby('id'))\n):\n\n    id, subdf = chunk\n    \n    \n    if debug:\n        print(id, subdf.shape)\n        print()\n\n        print(subdf)\n\n\n        print()\n        print()\n\n    X = sp.hstack([vectorizer.transform(subdf['source']) for vectorizer in vectorizers ])\n\n        \n    if debug:\n        plt.imshow(X.todense()[:128,:100])\n        plt.show()\n        plt.imshow(X.todense()[:128,200:300])\n        plt.show()\n\n        \n    cng = centered_ngram_fun(subdf)\n\n\n\n    ng = ngram_fun(subdf)\n\n    \n\n    \n    X = np.array(X.todense())\n\n    if debug:\n        \n        print('Centered Ngram Similarity')\n        plt.imshow(cng)\n        plt.show()\n\n        print('Ngram Similarity')\n        plt.imshow(ng)\n        plt.show()\n        \n        print('TFIDF Outer Product')\n        D =  np.matmul(X[:, :], X[:,:].T)\n        \n        plt.imshow(D)\n        plt.show()\n\n        \n        \n        \n    b = subdf['cell_type'] == 'code'\n    \n    ng = np.dstack( [ cng[:,:,np.newaxis], ng[:,:,np.newaxis]])[np.newaxis,:,:,:]\n   \n    \n    Cr = (subdf['rank']).values[:,np.newaxis]\n\n    Cr = (Cr+1)/(np.sum(b)+1)\n    \n    L = sp.lil_matrix(X)\n\n    ZI = np.array(  [ np.array( x[:200] + [0]*max(0,200-len(x)) ).astype(int)  for x in L.rows ]  )\n    ZV = np.array(  [ np.array( x[:200] + [0]*max(0,200-len(x)) ).astype(np.float16) for x in L.data]  )\n\n    \n    \n    p = model.predict([ZV[np.newaxis], ZI[np.newaxis],\n                       Cr[np.newaxis], np.ones([1,ZI.shape[0]]), \n                       (subdf['cell_type']=='code').values[np.newaxis,:,np.newaxis], ng])\n    \n    p = p[0]\n    \n    \n    x = p[:,0]\n    y = p[:,1]\n    w = ( x**2 + y**2 + 1.0 )**0.5\n    \n    \n    x = x/(w-y)\n    \n    \n    cids = subdf['cell_id'].values\n    \n    cids = cids[np.argsort(x)]\n    \n    cids = ' '.join(cids)\n    \n    \n    samp.loc[id] = cids\n    \n    \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:49:53.685857Z","iopub.execute_input":"2022-08-12T07:49:53.686250Z","iopub.status.idle":"2022-08-12T07:50:16.893184Z","shell.execute_reply.started":"2022-08-12T07:49:53.686213Z","shell.execute_reply":"2022-08-12T07:50:16.892153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Submission","metadata":{}},{"cell_type":"code","source":"\nsamp.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T07:46:48.688729Z","iopub.status.idle":"2022-08-12T07:46:48.689432Z","shell.execute_reply.started":"2022-08-12T07:46:48.689175Z","shell.execute_reply":"2022-08-12T07:46:48.689198Z"},"trusted":true},"execution_count":null,"outputs":[]}]}