{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":130287,"databundleVersionId":15633993,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-11T14:11:29.874512Z","iopub.execute_input":"2026-02-11T14:11:29.874785Z","execution_failed":"2026-02-11T14:38:39.394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n# --- 1. ROBUST DATA LOADING ---\n# This function searches for the files in the most common Kaggle directories\ndef load_data(filename):\n    potential_paths = [\n        './',                                                                            # Current directory\n        '/kaggle/input/motion-s-hierarchical-text-to-motion-generation-for-sign-language/', # Kaggle Default\n        '/kaggle/input/motion-s/',                                                       # Alternate Kaggle\n        '../input/motion-s-hierarchical-text-to-motion-generation-for-sign-language/'   # Jupyter parent\n    ]\n    \n    for path in potential_paths:\n        full_path = os.path.join(path, filename)\n        if os.path.exists(full_path):\n            print(f\"Found {filename} at: {full_path}\")\n            return pd.read_csv(full_path)\n    \n    # If code reaches here, the file wasn't found\n    print(f\"\\nCRITICAL ERROR: Could not find {filename}.\")\n    print(f\"Current Working Directory: {os.getcwd()}\")\n    print(\"Files in Current Directory:\", os.listdir(os.getcwd()))\n    raise FileNotFoundError(f\"Please make sure the dataset is attached to the notebook.\")\n\n# Load the datasets\ntry:\n    print(\"--- Loading Data ---\")\n    train_df = load_data('train.csv')\n    test_df = load_data('test.csv')\nexcept FileNotFoundError as e:\n    print(e)\n    # Stop execution if files are missing\n    raise\n\n# --- 2. PREPROCESSING & FILTERING ---\nprint(\"\\n--- Filtering Training Data ---\")\n# Drop rows with missing values in critical columns\ntrain_df = train_df.dropna(subset=['gloss', 'base_tokens', 'residual_1', 'residual_2', 'residual_3', 'residual_4', 'residual_5'])\n\n# Helper function to check token length (must be 40-800)\ndef get_token_len(s):\n    try:\n        return len(str(s).split())\n    except:\n        return 0\n\ntrain_df['token_len'] = train_df['base_tokens'].apply(get_token_len)\n\n# Keep only valid training samples\nvalid_train = train_df[(train_df['token_len'] >= 40) & (train_df['token_len'] <= 800)].copy().reset_index(drop=True)\nprint(f\"Original training samples: {len(train_df)}\")\nprint(f\"Valid training samples used: {len(valid_train)}\")\n\n# --- 3. RETRIEVAL (TF-IDF) ---\nprint(\"\\n--- Vectorizing Glosses ---\")\n# Combine train and test glosses to fit the vectorizer (ensures vocabulary covers both)\nall_glosses = pd.concat([valid_train['gloss'], test_df['gloss']]).fillna(\"\")\n\n# Use word-level n-grams to capture phrases (e.g., \"GO STORE\")\nvectorizer = TfidfVectorizer(analyzer='word', token_pattern=r'\\b\\w+\\b', ngram_range=(1, 2))\nvectorizer.fit(all_glosses)\n\ntrain_tfidf = vectorizer.transform(valid_train['gloss'].fillna(\"\"))\ntest_tfidf = vectorizer.transform(test_df['gloss'].fillna(\"\"))\n\n# --- 4. SIMILARITY SEARCH ---\nprint(\"\\n--- Matching Test Samples to Training Data ---\")\n# Compute cosine similarity between every Test sample and every Train sample\nsimilarities = cosine_similarity(test_tfidf, train_tfidf)\n\n# Find the index of the best match for each test sample\nbest_match_indices = np.argmax(similarities, axis=1)\n\n# --- 5. CREATE SUBMISSION ---\nprint(\"\\n--- constructing submission.csv ---\")\nsubmission = pd.DataFrame()\nsubmission['id'] = test_df['id']\n\n# Retrieve tokens from the matched training rows\nmatched_rows = valid_train.iloc[best_match_indices]\ntoken_cols = ['base_tokens', 'residual_1', 'residual_2', 'residual_3', 'residual_4', 'residual_5']\n\nfor col in token_cols:\n    submission[col] = matched_rows[col].values\n\n# --- 6. SAVE OUTPUT ---\noutput_file = 'submission.csv'\nsubmission.to_csv(output_file, index=False)\nprint(f\"Success! {output_file} generated with {len(submission)} rows.\")\nprint(\"\\nFirst 5 rows of submission:\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-11T14:38:29.775842Z","iopub.execute_input":"2026-02-11T14:38:29.776649Z","iopub.status.idle":"2026-02-11T14:38:31.585453Z","shell.execute_reply.started":"2026-02-11T14:38:29.776620Z","shell.execute_reply":"2026-02-11T14:38:31.584870Z"}},"outputs":[],"execution_count":null}]}