{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"modelInstanceVersion","sourceId":4576,"databundleVersionId":6401209,"modelInstanceId":3368,"modelId":1103,"isSourceIdPinned":false},{"sourceType":"kernelVersion","sourceId":52130328,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:05:18.612133Z","iopub.execute_input":"2026-03-24T09:05:18.612734Z","iopub.status.idle":"2026-03-24T09:05:18.894570Z","shell.execute_reply.started":"2026-03-24T09:05:18.612701Z","shell.execute_reply":"2026-03-24T09:05:18.893955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport torch\nprint(\"GPU available:\", torch.cuda.is_available())\nprint(\"GPU name:\", torch.cuda.get_device_name(0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:05:18.895771Z","iopub.execute_input":"2026-03-24T09:05:18.896166Z","iopub.status.idle":"2026-03-24T09:05:27.259744Z","shell.execute_reply.started":"2026-03-24T09:05:18.896142Z","shell.execute_reply":"2026-03-24T09:05:27.259086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install transformers datasets tokenizers accelerate -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:05:27.260590Z","iopub.execute_input":"2026-03-24T09:05:27.261034Z","iopub.status.idle":"2026-03-24T09:05:32.893767Z","shell.execute_reply.started":"2026-03-24T09:05:27.260988Z","shell.execute_reply":"2026-03-24T09:05:32.892764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import GPT2Tokenizer, GPT2LMHeadModel, GPT2Config\nfrom datasets import load_dataset\nfrom transformers import Trainer, TrainingArguments, DataCollatorForLanguageModeling\nimport torch\n\nprint(\"All imports successful!\")\nprint(\"PyTorch version:\", torch.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:05:32.895285Z","iopub.execute_input":"2026-03-24T09:05:32.895875Z","iopub.status.idle":"2026-03-24T09:05:59.365266Z","shell.execute_reply.started":"2026-03-24T09:05:32.895843Z","shell.execute_reply":"2026-03-24T09:05:59.364445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = load_dataset(\"wikitext\", \"wikitext-103-raw-v1\")\nprint(\"Dataset loaded!\")\nprint(dataset)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:05:59.367002Z","iopub.execute_input":"2026-03-24T09:05:59.367498Z","iopub.status.idle":"2026-03-24T09:06:07.920562Z","shell.execute_reply.started":"2026-03-24T09:05:59.367471Z","shell.execute_reply":"2026-03-24T09:06:07.919935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = GPT2Tokenizer.from_pretrained(\"gpt2\")\ntokenizer.pad_token = tokenizer.eos_token\nprint(\"Tokenizer loaded!\")\nprint(\"Vocabulary size:\", tokenizer.vocab_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:06:07.921395Z","iopub.execute_input":"2026-03-24T09:06:07.921699Z","iopub.status.idle":"2026-03-24T09:06:09.378494Z","shell.execute_reply.started":"2026-03-24T09:06:07.921664Z","shell.execute_reply":"2026-03-24T09:06:09.377757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tokenize_function(examples):\n    return tokenizer(\n        examples[\"text\"],\n        truncation=True,\n        max_length=512,\n        padding=\"max_length\"\n    )\n\ntokenized_dataset = dataset.map(\n    tokenize_function,\n    batched=True,\n    remove_columns=[\"text\"]\n)\n\nprint(\"Tokenization done!\")\nprint(tokenized_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:06:09.379312Z","iopub.execute_input":"2026-03-24T09:06:09.379602Z","iopub.status.idle":"2026-03-24T09:13:02.452521Z","shell.execute_reply.started":"2026-03-24T09:06:09.379570Z","shell.execute_reply":"2026-03-24T09:13:02.451892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"config = GPT2Config(\n    vocab_size=tokenizer.vocab_size,\n    n_positions=512,\n    n_embd=256,\n    n_layer=4,\n    n_head=4,\n    resid_pdrop=0.1,\n    embd_pdrop=0.1,\n    attn_pdrop=0.1,\n)\n\nmodel = GPT2LMHeadModel(config)\n\ntotal_params = sum(p.numel() for p in model.parameters())\nprint(f\"Model created!\")\nprint(f\"Total parameters: {total_params/1e6:.1f}M\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:13:02.453510Z","iopub.execute_input":"2026-03-24T09:13:02.453769Z","iopub.status.idle":"2026-03-24T09:13:02.956405Z","shell.execute_reply.started":"2026-03-24T09:13:02.453745Z","shell.execute_reply":"2026-03-24T09:13:02.955650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"/kaggle/working/my-ai-model\",\n    num_train_epochs=1,\n    per_device_train_batch_size=8,\n    per_device_eval_batch_size=8,\n    gradient_accumulation_steps=4,\n    learning_rate=5e-4,\n    warmup_steps=200,\n    weight_decay=0.01,\n    fp16=True,\n    logging_steps=50,\n    save_steps=200,\n    save_total_limit=1,\n    prediction_loss_only=True,\n    report_to=\"none\"\n)\n\nprint(\"Training settings ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:13:02.957473Z","iopub.execute_input":"2026-03-24T09:13:02.957864Z","iopub.status.idle":"2026-03-24T09:13:03.788949Z","shell.execute_reply.started":"2026-03-24T09:13:02.957838Z","shell.execute_reply":"2026-03-24T09:13:03.788294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import DataCollatorForLanguageModeling, Trainer, TrainingArguments\nfrom transformers import GPT2Tokenizer, GPT2LMHeadModel, GPT2Config\nfrom datasets import load_dataset\nimport torch\nprint(\"Re-imported successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:13:03.789849Z","iopub.execute_input":"2026-03-24T09:13:03.790140Z","iopub.status.idle":"2026-03-24T09:13:03.794305Z","shell.execute_reply.started":"2026-03-24T09:13:03.790116Z","shell.execute_reply":"2026-03-24T09:13:03.793531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_collator = DataCollatorForLanguageModeling(\n    tokenizer=tokenizer,\n    mlm=False\n)\n\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_dataset[\"train\"],\n    eval_dataset=tokenized_dataset[\"validation\"],\n    data_collator=data_collator,\n)\n\nprint(\"Starting training!\")\ntrainer.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:13:03.795169Z","iopub.execute_input":"2026-03-24T09:13:03.795390Z","execution_failed":"2026-03-24T09:21:45.682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save_pretrained(\"/kaggle/working/my-ai-model\")\ntokenizer.save_pretrained(\"/kaggle/working/my-ai-model\")\nprint(\"Model saved successfully!\")","metadata":{"trusted":true,"execution":{"execution_failed":"2026-03-24T09:21:45.683Z"}},"outputs":[],"execution_count":null}]}