{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom pathlib import Path\n\n\nnum_lines = sum(1 for line in open(\"/kaggle/input/otto-recommender-system/train.jsonl\"))\nprint(f'number of lines in train: {num_lines:_}')\n\nchunksize = 100_000\nnum_chunks = int(np.ceil(num_lines / 100_000))\nprint(f'number of chunks: {num_chunks:_}')\n\nn = 2\ntrain_sessions = pd.DataFrame()\nchunks = pd.read_json(\"/kaggle/input/otto-recommender-system/train.jsonl\", lines=True, chunksize=chunksize)\n\nfor e, chunk in enumerate(chunks):\n    if e < 2:\n        train_sessions = pd.concat([train_sessions, chunk])\n    else:\n        break\ntrain_sessions = train_sessions.set_index('session', drop=True).sort_index()\n\ntrain_sessions","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:01.957099Z","iopub.execute_input":"2023-12-22T15:26:01.957669Z","iopub.status.idle":"2023-12-22T15:26:29.436179Z","shell.execute_reply.started":"2023-12-22T15:26:01.957626Z","shell.execute_reply":"2023-12-22T15:26:29.434247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Task1\n# transform the dataframe train_sessions to a dataframe train_sessions_detailed : columns ['aid', 'ts', 'type' , 'session']\ndf_session = []\nfor session_id in range(1000):\n    df_detailed = pd.DataFrame(train_sessions.iloc[session_id][0])\n    df_detailed['session'] = session_id\n    df_session.append(df_detailed)\n\ndf_result = pd.concat(df_session, ignore_index=True)\ndf_result\n\n#Task2 Explore the data","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:29.439256Z","iopub.execute_input":"2023-12-22T15:26:29.439704Z","iopub.status.idle":"2023-12-22T15:26:30.383313Z","shell.execute_reply.started":"2023-12-22T15:26:29.439673Z","shell.execute_reply":"2023-12-22T15:26:30.381317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.385411Z","iopub.execute_input":"2023-12-22T15:26:30.385966Z","iopub.status.idle":"2023-12-22T15:26:30.40117Z","shell.execute_reply.started":"2023-12-22T15:26:30.385925Z","shell.execute_reply":"2023-12-22T15:26:30.399621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.404275Z","iopub.execute_input":"2023-12-22T15:26:30.404767Z","iopub.status.idle":"2023-12-22T15:26:30.418187Z","shell.execute_reply.started":"2023-12-22T15:26:30.404736Z","shell.execute_reply":"2023-12-22T15:26:30.415874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.columns","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.420829Z","iopub.execute_input":"2023-12-22T15:26:30.421354Z","iopub.status.idle":"2023-12-22T15:26:30.433603Z","shell.execute_reply.started":"2023-12-22T15:26:30.421315Z","shell.execute_reply":"2023-12-22T15:26:30.431691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.info()","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.435647Z","iopub.execute_input":"2023-12-22T15:26:30.436333Z","iopub.status.idle":"2023-12-22T15:26:30.47372Z","shell.execute_reply.started":"2023-12-22T15:26:30.436291Z","shell.execute_reply":"2023-12-22T15:26:30.472029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.476226Z","iopub.execute_input":"2023-12-22T15:26:30.476711Z","iopub.status.idle":"2023-12-22T15:26:30.504142Z","shell.execute_reply.started":"2023-12-22T15:26:30.476672Z","shell.execute_reply":"2023-12-22T15:26:30.502921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result.describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.505832Z","iopub.execute_input":"2023-12-22T15:26:30.506296Z","iopub.status.idle":"2023-12-22T15:26:30.53778Z","shell.execute_reply.started":"2023-12-22T15:26:30.506262Z","shell.execute_reply":"2023-12-22T15:26:30.535956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count events by type\nevent_counts = df_result[\"type\"].value_counts()\nprint(event_counts)\n\n# Visualize event frequencies\nsns.barplot(x=event_counts.index, y=event_counts.values)\nplt.title(\"Frequency of Event Types\")\nplt.show()\n\n# Explore session lengths\nsession_lengths = df_result.groupby(\"session\")[\"ts\"].nunique()\nplt.hist(session_lengths)\nplt.title(\"Distribution of Session Lengths\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.539188Z","iopub.execute_input":"2023-12-22T15:26:30.5397Z","iopub.status.idle":"2023-12-22T15:26:30.955853Z","shell.execute_reply.started":"2023-12-22T15:26:30.53966Z","shell.execute_reply":"2023-12-22T15:26:30.953944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Analyze popular products\npopular_products = df_result[\"aid\"].value_counts().head(20)\nprint(popular_products)\n\n# Visualize product popularity by event type\nsns.countplot(x=\"aid\", hue=\"type\", data=df_result[df_result[\"aid\"].isin(popular_products.index)])\nplt.title(\"Popular Product Interactions by Event Type\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:30.960334Z","iopub.execute_input":"2023-12-22T15:26:30.960762Z","iopub.status.idle":"2023-12-22T15:26:31.419257Z","shell.execute_reply.started":"2023-12-22T15:26:30.960731Z","shell.execute_reply":"2023-12-22T15:26:31.417147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:31.421065Z","iopub.execute_input":"2023-12-22T15:26:31.421425Z","iopub.status.idle":"2023-12-22T15:26:31.428816Z","shell.execute_reply.started":"2023-12-22T15:26:31.421398Z","shell.execute_reply":"2023-12-22T15:26:31.427079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Descriptive statistics\nprint(df_result.describe())\n\n# Visualizations\nsns.histplot(df_result[\"session_length\"])  # Example visualization\n\n# Correlation analysis\nplt.matshow(df_result.corr())\n\n# Feature importance (example using a decision tree)\nfrom sklearn.tree import DecisionTreeClassifier\nmodel = DecisionTreeClassifier()\nmodel.fit(X_train, y_train)\nprint(model.feature_importances_)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:31.432102Z","iopub.execute_input":"2023-12-22T15:26:31.432709Z","iopub.status.idle":"2023-12-22T15:26:31.572775Z","shell.execute_reply.started":"2023-12-22T15:26:31.432667Z","shell.execute_reply":"2023-12-22T15:26:31.57057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_result.columns)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:31.574062Z","iopub.status.idle":"2023-12-22T15:26:31.574591Z","shell.execute_reply.started":"2023-12-22T15:26:31.574326Z","shell.execute_reply":"2023-12-22T15:26:31.574346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Extract product descriptions\ndescriptions = df_result[\"product_descriptions\"]\n\n# Preprocess text\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer\n\nstop_words = stopwords.words(\"english\")\nstemmer = PorterStemmer()\n\npreprocessed_text = []\nfor description in descriptions:\n    # Lowercase, remove punctuation, tokenize\n    words = [word.lower() for word in description.split() if word.isalnum()]\n    # Remove stop words and stem\n    words = [stemmer.stem(word) for word in words if word not in stop_words]\n    preprocessed_text.append(\" \".join(words))\nfrom gensim.models import Word2Vec\n\nmodel = Word2Vec(preprocessed_text, vector_size=100, window=5, min_count=1)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:31.575961Z","iopub.status.idle":"2023-12-22T15:26:31.576349Z","shell.execute_reply.started":"2023-12-22T15:26:31.576178Z","shell.execute_reply":"2023-12-22T15:26:31.576196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"product_embedding = model.wv[\"product_name\"]","metadata":{"execution":{"iopub.status.busy":"2023-12-22T15:26:31.580254Z","iopub.status.idle":"2023-12-22T15:26:31.580796Z","shell.execute_reply.started":"2023-12-22T15:26:31.580576Z","shell.execute_reply":"2023-12-22T15:26:31.5806Z"},"trusted":true},"execution_count":null,"outputs":[]}]}