{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":3043,"databundleVersionId":46668,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Fast prototype: ~ tiny sample + hashing + SGD (probabilities)\nimport pandas as pd\nimport numpy as np\nfrom sklearn.feature_extraction.text import HashingVectorizer\nfrom sklearn.linear_model import SGDClassifier\n\n# ---- 1) Load (adjust paths if needed) ----\ntrain = pd.read_csv(\"/kaggle/input/predict-closed-questions-on-stack-overflow/train.csv\")\ntest  = pd.read_csv(\"/kaggle/input/predict-closed-questions-on-stack-overflow/public_leaderboard.csv\")\n\n# ---- 2) Quick stratified sample (keeps all classes) ----\nclasses = [\"not a real question\",\"not constructive\",\"off topic\",\"open\",\"too localized\"]\nper_class = 4000  # reduce if you need even faster (e.g., 2000)\nparts = []\nfor c in classes:\n    parts.append(train[train[\"OpenStatus\"] == c].sample(\n        n=min(per_class, (train[\"OpenStatus\"] == c).sum()),\n        random_state=42\n    ))\ntrain_small = pd.concat(parts, axis=0, ignore_index=True)\n\n# ---- 3) Text fields ----\nXtr_text = (train_small[\"Title\"].fillna(\"\") + \" \" + train_small[\"BodyMarkdown\"].fillna(\"\"))\nXte_text = (test[\"Title\"].fillna(\"\") + \" \" + test[\"BodyMarkdown\"].fillna(\"\"))\n\n# ---- 4) Very fast features: HashingVectorizer (no fit) ----\nhv = HashingVectorizer(n_features=2**18, ngram_range=(1,2), alternate_sign=False)  # ~262k dims, fast\nXtr = hv.transform(Xtr_text)\nXte = hv.transform(Xte_text)\n\n# ---- 5) Fast linear model with probabilities ----\nclf = SGDClassifier(loss=\"log_loss\", max_iter=5, tol=1e-3, n_jobs=-1, random_state=42)\nclf.fit(Xtr, train_small[\"OpenStatus\"])\n\n# ---- 6) Probabilities (ensure correct class order) ----\ndef softmax(z):\n    z = z - z.max(axis=1, keepdims=True)\n    ez = np.exp(z)\n    return ez / ez.sum(axis=1, keepdims=True)\n\nif hasattr(clf, \"predict_proba\"):\n    proba_all = clf.predict_proba(Xte)\nelse:\n    # Fallback if sklearn build doesn’t expose predict_proba for SGD\n    scores = clf.decision_function(Xte)\n    if scores.ndim == 1:  # rare binary-shape case, expand to 2 classes then map\n        scores = np.column_stack([-scores, scores])\n    proba_all = softmax(scores)\n\n# Map columns to the required order\n# clf.classes_ may be a subset ordering; reindex to expected classes\nproba_df = pd.DataFrame(proba_all, columns=clf.classes_)\nfor c in classes:\n    if c not in proba_df.columns:\n        proba_df[c] = 0.0  # if a class never appeared in the sample\nsub = proba_df.reindex(columns=classes, fill_value=0)\n\n# ---- 7) Optional id column ----\nif \"id\" in test.columns:\n    sub.insert(0, \"id\", test[\"id\"].values)\n\n# ---- 8) Save submission ----\nsub.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv written (fast prototype).\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-19T13:47:57.432292Z","iopub.execute_input":"2025-09-19T13:47:57.432591Z","iopub.status.idle":"2025-09-19T13:49:44.814631Z","shell.execute_reply.started":"2025-09-19T13:47:57.432564Z","shell.execute_reply":"2025-09-19T13:49:44.813769Z"}},"outputs":[],"execution_count":null}]}