{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":7902037,"sourceType":"datasetVersion","datasetId":4640957},{"sourceId":7996468,"sourceType":"datasetVersion","datasetId":4708215}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is word embedding?\nWord embedding or word vector is an approach with which we represent documents and words. It is defined as a numeric vector input that allows words with similar meanings to have the same representation. \n\nFor instance, a word embedding with 50 values holds the capability of representing 50 unique features. Many people choose pre-trained word embedding models like Flair, fastText, SpaCy, and others.","metadata":{}},{"cell_type":"code","source":"def get_stats(ids):\n    counts = {}\n    for pair in zip(ids, ids[1:]):\n        counts[pair] = counts.get(pair, 0) + 1\n    return counts\n\ndef merge(ids, pair, idx):\n    new_ids = []\n    \n    i = 0\n    while i < len(ids):\n        if i < len(ids) - 1 and ids[i] == pair[0] and ids[i+1] == pair[1]:\n            new_ids.append(idx)\n            i += 2\n        else:\n            new_ids.append(ids[i])\n            i += 1\n    return new_ids\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.461732Z","iopub.execute_input":"2024-04-02T03:07:58.462223Z","iopub.status.idle":"2024-04-02T03:07:58.513596Z","shell.execute_reply.started":"2024-04-02T03:07:58.462188Z","shell.execute_reply":"2024-04-02T03:07:58.512434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\nwith open('/kaggle/input/bpe-tokenizer/merges.txt', 'r') as file:\n    data_string = file.read()\n    merges = ast.literal_eval(data_string)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.516536Z","iopub.execute_input":"2024-04-02T03:07:58.517159Z","iopub.status.idle":"2024-04-02T03:07:58.750223Z","shell.execute_reply.started":"2024-04-02T03:07:58.517106Z","shell.execute_reply":"2024-04-02T03:07:58.748774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode(text):\n    tokens = list(text.encode('utf-8'))\n    \n    while len(tokens)>=2:\n        stats = get_stats(tokens)\n        pair = min(stats, key=lambda p: merges.get(p, float('inf')))\n        \n        if pair not in merges:\n            break\n        \n        idx = merges[pair]\n        tokens = merge(tokens, pair, idx)\n        \n        \n    return tokens\n\nencode('a')","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.752666Z","iopub.execute_input":"2024-04-02T03:07:58.753191Z","iopub.status.idle":"2024-04-02T03:07:58.769193Z","shell.execute_reply.started":"2024-04-02T03:07:58.753147Z","shell.execute_reply":"2024-04-02T03:07:58.767694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\nwith open('/kaggle/input/bpe-tokenizer/vocab.txt', 'r') as file:\n    data_string = file.read()\n    vocab = ast.literal_eval(data_string)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.773322Z","iopub.execute_input":"2024-04-02T03:07:58.773966Z","iopub.status.idle":"2024-04-02T03:07:58.849387Z","shell.execute_reply.started":"2024-04-02T03:07:58.773907Z","shell.execute_reply":"2024-04-02T03:07:58.848059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef decode(ids):\n    tokens = b''.join(vocab[idx] for idx in ids)\n    text = tokens.decode('utf-8', errors='replace')\n    \n    return text\n\n\nprint(decode([97]))","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.852793Z","iopub.execute_input":"2024-04-02T03:07:58.853288Z","iopub.status.idle":"2024-04-02T03:07:58.860566Z","shell.execute_reply.started":"2024-04-02T03:07:58.853241Z","shell.execute_reply":"2024-04-02T03:07:58.859493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/romeo-and-juliet-tokenization/romeo-and-juliet_tokenization.txt', 'r') as file:\n    text = file.read()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.862063Z","iopub.execute_input":"2024-04-02T03:07:58.864102Z","iopub.status.idle":"2024-04-02T03:07:58.881786Z","shell.execute_reply.started":"2024-04-02T03:07:58.864060Z","shell.execute_reply":"2024-04-02T03:07:58.880200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenized_text = text.split('.')\ntokenized_text[:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.883796Z","iopub.execute_input":"2024-04-02T03:07:58.884638Z","iopub.status.idle":"2024-04-02T03:07:58.893377Z","shell.execute_reply.started":"2024-04-02T03:07:58.884593Z","shell.execute_reply":"2024-04-02T03:07:58.892140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenized = []\n\nfor sentence in tokenized_text:\n    tokenized.append(encode(sentence))\n\nlen(tokenized)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:07:58.897266Z","iopub.execute_input":"2024-04-02T03:07:58.897646Z","iopub.status.idle":"2024-04-02T03:08:07.661644Z","shell.execute_reply.started":"2024-04-02T03:07:58.897607Z","shell.execute_reply":"2024-04-02T03:08:07.660134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenized[:3]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:08:07.663470Z","iopub.execute_input":"2024-04-02T03:08:07.663830Z","iopub.status.idle":"2024-04-02T03:08:07.674397Z","shell.execute_reply.started":"2024-04-02T03:08:07.663785Z","shell.execute_reply":"2024-04-02T03:08:07.672750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Word embeddings from scratch","metadata":{}},{"cell_type":"code","source":"pip install --upgrade tensorflow\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:08:07.678571Z","iopub.execute_input":"2024-04-02T03:08:07.679492Z","iopub.status.idle":"2024-04-02T03:09:37.757980Z","shell.execute_reply.started":"2024-04-02T03:08:07.679447Z","shell.execute_reply":"2024-04-02T03:09:37.753342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Embedding, GlobalAveragePooling1D, Dense\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split\n\n# Simulated tokenized data (replace with actual tokenized data using your BPE tokenizer)\ntexts_tokenized = tokenized\n\n# Simulated labels (replace with your actual labels, e.g., for classification)\nlabels = np.random.randint(0, 2, size=(len(texts_tokenized),))\n\n# Parameters\nvocab_size = 27795  # Assuming 1000 tokens in your BPE vocab\nmax_length = 255  # Maximum length of sequences\nembedding_dim = 64  # Size of embedding vectors\n\n# Pad sequences\ntexts_padded = pad_sequences(texts_tokenized, maxlen=max_length, padding='post', truncating='post')\n\n# Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(texts_padded, labels, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:32:11.574363Z","iopub.execute_input":"2024-04-02T03:32:11.575667Z","iopub.status.idle":"2024-04-02T03:32:11.600001Z","shell.execute_reply.started":"2024-04-02T03:32:11.575610Z","shell.execute_reply":"2024-04-02T03:32:11.598857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[0]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:32:12.495455Z","iopub.execute_input":"2024-04-02T03:32:12.496259Z","iopub.status.idle":"2024-04-02T03:32:12.504239Z","shell.execute_reply.started":"2024-04-02T03:32:12.496219Z","shell.execute_reply":"2024-04-02T03:32:12.503060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the model\nmodel = Sequential([\n    Embedding(input_dim=vocab_size, output_dim=embedding_dim),\n    GlobalAveragePooling1D(),\n    Dense(24, activation='relu'),\n    Dense(1, activation='sigmoid')  # Assuming binary classification\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Model summary\nmodel.summary()\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:32:13.496303Z","iopub.execute_input":"2024-04-02T03:32:13.496733Z","iopub.status.idle":"2024-04-02T03:32:26.566752Z","shell.execute_reply.started":"2024-04-02T03:32:13.496702Z","shell.execute_reply":"2024-04-02T03:32:26.565336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# After training, you can extract the learned embeddings like this:\nembeddings = model.layers[0].get_weights()[0]\nembeddings","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-02T03:32:28.591850Z","iopub.execute_input":"2024-04-02T03:32:28.592579Z","iopub.status.idle":"2024-04-02T03:32:28.605638Z","shell.execute_reply.started":"2024-04-02T03:32:28.592536Z","shell.execute_reply":"2024-04-02T03:32:28.604377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:32:30.943923Z","iopub.execute_input":"2024-04-02T03:32:30.944482Z","iopub.status.idle":"2024-04-02T03:32:30.953636Z","shell.execute_reply.started":"2024-04-02T03:32:30.944430Z","shell.execute_reply":"2024-04-02T03:32:30.952184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:32:34.511110Z","iopub.execute_input":"2024-04-02T03:32:34.511590Z","iopub.status.idle":"2024-04-02T03:32:34.520314Z","shell.execute_reply.started":"2024-04-02T03:32:34.511556Z","shell.execute_reply":"2024-04-02T03:32:34.519109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.layers[0].get_weights()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T03:37:46.043369Z","iopub.execute_input":"2024-04-02T03:37:46.043934Z","iopub.status.idle":"2024-04-02T03:37:46.057269Z","shell.execute_reply.started":"2024-04-02T03:37:46.043894Z","shell.execute_reply":"2024-04-02T03:37:46.055656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}