{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-30T17:29:41.254888Z","iopub.execute_input":"2023-08-30T17:29:41.255575Z","iopub.status.idle":"2023-08-30T17:29:41.291037Z","shell.execute_reply.started":"2023-08-30T17:29:41.255538Z","shell.execute_reply":"2023-08-30T17:29:41.289867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv('/kaggle/input/commonlit-evaluate-student-summaries/prompts_test.csv')\ndata","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:32:16.527328Z","iopub.execute_input":"2023-08-30T17:32:16.527745Z","iopub.status.idle":"2023-08-30T17:32:16.563406Z","shell.execute_reply.started":"2023-08-30T17:32:16.527716Z","shell.execute_reply":"2023-08-30T17:32:16.562539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\n\n# Load the training data\ntrain_data = pd.read_csv('/kaggle/input/commonlit-evaluate-student-summaries/summaries_train.csv')\ntrain_prompts = pd.read_csv('/kaggle/input/commonlit-evaluate-student-summaries/prompts_train.csv')\n\n# Merge training data with prompts\ntrain_merged = pd.merge(train_data, train_prompts, on='prompt_id', how='left')\n\n# Separate features and target variables\nX = train_merged['text']\ny_content = train_merged['content']\ny_wording = train_merged['wording']\n\n# Split the data into training and validation sets\nX_train, X_val, y_content_train, y_content_val, y_wording_train, y_wording_val = train_test_split(\n    X, y_content, y_wording, test_size=0.2, random_state=42\n)\n\n# Preprocessing for text data\ntext_vectorizer = CountVectorizer(max_features=5000)  # Adjust max_features as needed\nX_train_vec = text_vectorizer.fit_transform(X_train)\nX_val_vec = text_vectorizer.transform(X_val)\n\n# Model 1: Linear Regression for Content Score\ncontent_model = LinearRegression()\ncontent_model.fit(X_train_vec, y_content_train)\ncontent_predictions = content_model.predict(X_val_vec)\ncontent_rmse = np.sqrt(mean_squared_error(y_content_val, content_predictions))\nprint(\"Content RMSE:\", content_rmse)\n\n# Model 2: Random Forest for Wording Score\nwording_model = RandomForestRegressor(n_estimators=100, random_state=42)\nwording_model.fit(X_train_vec, y_wording_train)\nwording_predictions = wording_model.predict(X_val_vec)\nwording_rmse = np.sqrt(mean_squared_error(y_wording_val, wording_predictions))\nprint(\"Wording RMSE:\", wording_rmse)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:49:17.489871Z","iopub.execute_input":"2023-08-30T17:49:17.490353Z","iopub.status.idle":"2023-08-30T17:52:42.732271Z","shell.execute_reply.started":"2023-08-30T17:49:17.490313Z","shell.execute_reply":"2023-08-30T17:52:42.731108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the test data\ntest_data = pd.read_csv('/kaggle/input/commonlit-evaluate-student-summaries/summaries_test.csv')\n\n# Extract text from test data\nX_test = test_data['text']\n\n# Preprocess the test data using the trained text_vectorizer\nX_test_vec = text_vectorizer.transform(X_test)\n\n# Generate predictions using the trained models\ntest_content_predictions = content_model.predict(X_test_vec)\ntest_wording_predictions = wording_model.predict(X_test_vec)\n\n# Create a DataFrame for the submission file\nsubmission_df = pd.DataFrame({\n    'student_id': test_data['student_id'],\n    'content': test_content_predictions,\n    'wording': test_wording_predictions\n})\n\n# Save the submission DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:53:38.284270Z","iopub.execute_input":"2023-08-30T17:53:38.285281Z","iopub.status.idle":"2023-08-30T17:53:38.308798Z","shell.execute_reply.started":"2023-08-30T17:53:38.285239Z","shell.execute_reply":"2023-08-30T17:53:38.307830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv('/kaggle/working/submission.csv')\ndata","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:54:35.488818Z","iopub.execute_input":"2023-08-30T17:54:35.489370Z","iopub.status.idle":"2023-08-30T17:54:35.503543Z","shell.execute_reply.started":"2023-08-30T17:54:35.489327Z","shell.execute_reply":"2023-08-30T17:54:35.502416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the test data\ntest_data = pd.read_csv('/kaggle/input/commonlit-evaluate-student-summaries/summaries_test.csv')\n\n# Extract text from test data\nX_test = test_data['text']\n\n# Preprocess the test data using the trained text_vectorizer\nX_test_vec = text_vectorizer.transform(X_test)\n\n# Generate predictions using the trained models\ntest_content_predictions = content_model.predict(X_test_vec)\ntest_wording_predictions = wording_model.predict(X_test_vec)\n\n# Create a DataFrame for the submission file\nsubmission_df = pd.DataFrame({\n    'student_id': test_data['student_id'],\n    'content': test_content_predictions,\n    'wording': test_wording_predictions\n})\n\n# Save the submission DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:57:42.351875Z","iopub.execute_input":"2023-08-30T17:57:42.352314Z","iopub.status.idle":"2023-08-30T17:57:42.375918Z","shell.execute_reply.started":"2023-08-30T17:57:42.352280Z","shell.execute_reply":"2023-08-30T17:57:42.374853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv('/kaggle/working/submission.csv')\ndata","metadata":{"execution":{"iopub.status.busy":"2023-08-30T17:58:11.043856Z","iopub.execute_input":"2023-08-30T17:58:11.044280Z","iopub.status.idle":"2023-08-30T17:58:11.058366Z","shell.execute_reply.started":"2023-08-30T17:58:11.044248Z","shell.execute_reply":"2023-08-30T17:58:11.057399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}