{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12392767,"sourceType":"datasetVersion","datasetId":7814713}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n\n# data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:19:17.232781Z","iopub.execute_input":"2025-07-26T15:19:17.233033Z","iopub.status.idle":"2025-07-26T15:19:23.549427Z","shell.execute_reply.started":"2025-07-26T15:19:17.233008Z","shell.execute_reply":"2025-07-26T15:19:23.543615Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Describe and Learn Data","metadata":{}},{"cell_type":"markdown","source":"train.parquet The training dataset containing all historical market data along with the corresponding labels.\n\ntimestamp: The timestamp index representing the minute associated with each row. bid_qty: The total quantity buyers are willing to purchase at the best (highest) bid price at the given timestamp. ask_qty: The total quantity sellers are offering to sell at the best (lowest) ask price at the given timestamp. buy_qty: The total trading quantity executed at the best ask price during the given minute. sellqty: The total trading quantity executed at the best bid price during the given minute. volume: The total traded volume during the minute. X{1,...,890}: A set of anonymized market features derived from proprietary data sources. label: The target variable representing the anonymized market price movement to be predicted.","metadata":{}},{"cell_type":"markdown","source":"timestamp: The timestamp index representing the minute associated with each row.\nbid_qty: The total quantity buyers are willing to purchase at the best (highest) bid price at the given timestamp.\nask_qty: The total quantity sellers are offering to sell at the best (lowest) ask price at the given timestamp.\nbuy_qty: The total trading quantity executed at the best ask price during the given minute.\nsell_qty: The total trading quantity executed at the best bid price during the given minute.\nvolume: The total traded volume during the minute.\nX_{1,...,780}: A set of anonymized market features derived from proprietary data sources.\nlabel: The target variable representing the anonymized market price movement to be predicted.","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/mark23/train.parquet')\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:19:28.310663Z","iopub.execute_input":"2025-07-26T15:19:28.310948Z","iopub.status.idle":"2025-07-26T15:19:47.488153Z","shell.execute_reply.started":"2025-07-26T15:19:28.310923Z","shell.execute_reply":"2025-07-26T15:19:47.481771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef optimize_dataframe_memory(df):\n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type).startswith('float'):\n            if df[col].isnull().any(): # NaN varsa float'ta kalmalı\n                continue # NaN içeren float sütunları şimdilik dönüştürmeyelim\n            min_val = df[col].min()\n            max_val = df[col].max()\n            if min_val > np.finfo(np.float32).min and max_val < np.finfo(np.float32).max:\n                df[col] = df[col].astype(np.float32)\n        elif str(col_type).startswith('int'):\n            min_val = df[col].min()\n            max_val = df[col].max()\n            if min_val > np.iinfo(np.int8).min and max_val < np.iinfo(np.int8).max:\n                df[col] = df[col].astype(np.int8)\n            elif min_val > np.iinfo(np.int16).min and max_val < np.iinfo(np.int16).max:\n                df[col] = df[col].astype(np.int16)\n            elif min_val > np.iinfo(np.int32).min and max_val < np.iinfo(np.int32).max:\n                df[col] = df[col].astype(np.int32)\n    return df\n\n# Örneğin train.parquet dosyasını yüklerken","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.614200Z","iopub.status.idle":"2025-07-26T15:16:41.614465Z","shell.execute_reply.started":"2025-07-26T15:16:41.614340Z","shell.execute_reply":"2025-07-26T15:16:41.614349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.615425Z","iopub.status.idle":"2025-07-26T15:16:41.615795Z","shell.execute_reply.started":"2025-07-26T15:16:41.615598Z","shell.execute_reply":"2025-07-26T15:16:41.615612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df0= df.copy()\ndf1 = df.copy()\ndf2 = df.copy()\ndf3 = df.copy()\ndf0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:19:59.096584Z","iopub.execute_input":"2025-07-26T15:19:59.096879Z","iopub.status.idle":"2025-07-26T15:20:04.568753Z","shell.execute_reply.started":"2025-07-26T15:19:59.096853Z","shell.execute_reply":"2025-07-26T15:20:04.563475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.618467Z","iopub.status.idle":"2025-07-26T15:16:41.618806Z","shell.execute_reply.started":"2025-07-26T15:16:41.618626Z","shell.execute_reply":"2025-07-26T15:16:41.618649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.620265Z","iopub.status.idle":"2025-07-26T15:16:41.620590Z","shell.execute_reply.started":"2025-07-26T15:16:41.620429Z","shell.execute_reply":"2025-07-26T15:16:41.620445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.tail(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.621388Z","iopub.status.idle":"2025-07-26T15:16:41.621696Z","shell.execute_reply.started":"2025-07-26T15:16:41.621542Z","shell.execute_reply":"2025-07-26T15:16:41.621555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.sample(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.622434Z","iopub.status.idle":"2025-07-26T15:16:41.622735Z","shell.execute_reply.started":"2025-07-26T15:16:41.622572Z","shell.execute_reply":"2025-07-26T15:16:41.622587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"bid_qty\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.623603Z","iopub.status.idle":"2025-07-26T15:16:41.623826Z","shell.execute_reply.started":"2025-07-26T15:16:41.623724Z","shell.execute_reply":"2025-07-26T15:16:41.623733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"ask_qty\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.625237Z","iopub.status.idle":"2025-07-26T15:16:41.625515Z","shell.execute_reply.started":"2025-07-26T15:16:41.625396Z","shell.execute_reply":"2025-07-26T15:16:41.625409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"buy_qty\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.626535Z","iopub.status.idle":"2025-07-26T15:16:41.626828Z","shell.execute_reply.started":"2025-07-26T15:16:41.626699Z","shell.execute_reply":"2025-07-26T15:16:41.626712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"sell_qty\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.628054Z","iopub.status.idle":"2025-07-26T15:16:41.628394Z","shell.execute_reply.started":"2025-07-26T15:16:41.628215Z","shell.execute_reply":"2025-07-26T15:16:41.628230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"volume\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.629241Z","iopub.status.idle":"2025-07-26T15:16:41.629455Z","shell.execute_reply.started":"2025-07-26T15:16:41.629349Z","shell.execute_reply":"2025-07-26T15:16:41.629357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"X1\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.630680Z","iopub.status.idle":"2025-07-26T15:16:41.630928Z","shell.execute_reply.started":"2025-07-26T15:16:41.630825Z","shell.execute_reply":"2025-07-26T15:16:41.630835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def unique_values(df, columns):\n    for column_name in columns:\n        print(f\"Column: {column_name}\\n{'-'*30}\")\n\n        value_counts_result = df[column_name].value_counts() # Doğrudan bu değişkeni kullanabiliriz\n\n        print(f\"Unique Values ({len(value_counts_result)})\\n\")\n        print(f\"Value Counts:\\n{value_counts_result}\\n{'='*40}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.631603Z","iopub.status.idle":"2025-07-26T15:16:41.631834Z","shell.execute_reply.started":"2025-07-26T15:16:41.631736Z","shell.execute_reply":"2025-07-26T15:16:41.631744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def unique_values(df,columns):\n    for column_name in columns:\n        print(f\"Column:{column_name}\\n{'-'*30}\")\n        unique_vals = df[column_name].value_counts() \n        value_counts = df[column_name].value_counts()\n        print(f\"unique Values({len(unique_vals)}\\n)\")\n        print(f\"Value Counts:\\n{value_counts}\\n{'='*40}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.632774Z","iopub.status.idle":"2025-07-26T15:16:41.633082Z","shell.execute_reply.started":"2025-07-26T15:16:41.632907Z","shell.execute_reply":"2025-07-26T15:16:41.632923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feature in df.columns:\n    if df[feature].dtype==\"object\":\n        print(feature,df[feature].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.634105Z","iopub.status.idle":"2025-07-26T15:16:41.634472Z","shell.execute_reply.started":"2025-07-26T15:16:41.634255Z","shell.execute_reply":"2025-07-26T15:16:41.634268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.635197Z","iopub.status.idle":"2025-07-26T15:16:41.635483Z","shell.execute_reply.started":"2025-07-26T15:16:41.635352Z","shell.execute_reply":"2025-07-26T15:16:41.635365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.637048Z","iopub.status.idle":"2025-07-26T15:16:41.637278Z","shell.execute_reply.started":"2025-07-26T15:16:41.637163Z","shell.execute_reply":"2025-07-26T15:16:41.637171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.638201Z","iopub.status.idle":"2025-07-26T15:16:41.638645Z","shell.execute_reply.started":"2025-07-26T15:16:41.638358Z","shell.execute_reply":"2025-07-26T15:16:41.638370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:20:15.508038Z","iopub.execute_input":"2025-07-26T15:20:15.508360Z","iopub.status.idle":"2025-07-26T15:20:16.317575Z","shell.execute_reply.started":"2025-07-26T15:20:15.508335Z","shell.execute_reply":"2025-07-26T15:20:16.313170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:21:22.788553Z","iopub.execute_input":"2025-07-26T15:21:22.788839Z","iopub.status.idle":"2025-07-26T15:21:47.763637Z","shell.execute_reply.started":"2025-07-26T15:21:22.788814Z","shell.execute_reply":"2025-07-26T15:21:47.758398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.fillna(df.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:21:50.092898Z","iopub.execute_input":"2025-07-26T15:21:50.093270Z","iopub.status.idle":"2025-07-26T15:21:56.137215Z","shell.execute_reply.started":"2025-07-26T15:21:50.093239Z","shell.execute_reply":"2025-07-26T15:21:56.132014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.resample('h').mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:22:01.203764Z","iopub.execute_input":"2025-07-26T15:22:01.204075Z","iopub.status.idle":"2025-07-26T15:22:10.391696Z","shell.execute_reply.started":"2025-07-26T15:22:01.204049Z","shell.execute_reply":"2025-07-26T15:22:10.387684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['X884'].shift(1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.645999Z","iopub.status.idle":"2025-07-26T15:16:41.646284Z","shell.execute_reply.started":"2025-07-26T15:16:41.646156Z","shell.execute_reply":"2025-07-26T15:16:41.646169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['bid_qty'].rolling(window=5).mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.647031Z","iopub.status.idle":"2025-07-26T15:16:41.647272Z","shell.execute_reply.started":"2025-07-26T15:16:41.647159Z","shell.execute_reply":"2025-07-26T15:16:41.647170Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Veri Tipini Küçültme: Özellikle float64 olan tüm sütunları float32'ye dönüştürmek bellek kullanımını yarı yarıya düşürmesi****","metadata":{}},{"cell_type":"markdown","source":"Dikkat: test.parquet'teki timestamp'lerin maskelenmiş olması, gecikmeli özelliklerin oluşturulmasında sıralamanın doğru yapılmasını zorlaştıracaktır. Eğitim verisinde bunu yaparken dikkatli olun ve test verisi için bu tip özellikleri oluştururken ID'ye güvenmeyin, zira ID'ler sıralı zamanı temsil etmiyor.","metadata":{}},{"cell_type":"markdown","source":"bid_qty / ask_qty gibi oranlar (alım-satım dengesi).\n\nbuy_qty - sell_qty gibi farklar (net emir akışı).\n\nbid_qty + ask_qty (derinlik toplamı).","metadata":{}},{"cell_type":"markdown","source":"Anonimleştirilmiş X_ özellikleri arasındaki potansiyel etkileşimleri keşfedin. Bazı özelliklerin birbiriyle çarpımı veya oranları anlamlı olabilir.\n\nBoyut indirgeme teknikleri (PCA gibi) kullanarak X_ özelliklerinin özetini çıkarabilirsiniz, ancak bu, orijinal özelliklerin yorumlanabilirliğini azaltabilir.","metadata":{}},{"cell_type":"markdown","source":"Gradient Boosting Modelleri: LightGBM, XGBoost, CatBoost gibi modeller zaman serisi verilerinde ve yüksek boyutlu verilerde genellikle çok iyi performans gösterirler. Hızlı olmaları ve kategorik/sayısal veriyi iyi işlemeleri avantajdır.\n\nGeleneksel Regresyon Modelleri: Ridge, Lasso gibi doğrusal modeller de başlangıç noktası olarak kullanılabilir, ancak karmaşık piyasa dinamiklerini yakalamakta yetersiz kalabilirler.\n\nDerin Öğrenme Modelleri: LSTM (Uzun Kısa Vadeli Bellek) veya Transformer tabanlı modeller, zaman serisi verileri için güçlü olabilir. Ancak, bu modeller daha fazla veri ve hesaplama gücü gerektirir ve genellikle daha karmaşıktır. Bu seviyedeki bir problem için gradient boosting modelleri genellikle iyi bir başlangıç noktasıdır.","metadata":{}},{"cell_type":"markdown","source":"Eğitim Stratejisi (Time Series Split):\n\nVerinin zaman serisi yapısından dolayı, çapraz doğrulama (cross-validation) yaparken standart K-Fold yerine zaman serisi tabanlı çapraz doğrulama (Time Series Split) kullanmalısınız. Bu, eğitim setinin her zaman test setinden önceki zaman dilimlerinden oluşmasını sağlar ve \"geleceği görmeyi\" engeller. sklearn.model_selection.TimeSeriesSplit bu iş için idealdir.\n\nUnutmayın: \"En son aylara odaklanın\" ipucunu göz önünde bulundurarak, eğitim setinizin sadece son kısmını kullanmayı düşünebilirsiniz.\n\nHiperparametre Optimizasyonu:\n\nModelinizin performansını artırmak için hiperparametrelerini optimize edin. GridSearchCV, RandomizedSearchCV veya Optuna, Hyperopt gibi kütüphaneler kullanılabilir.","metadata":{}},{"cell_type":"markdown","source":"Böyle dinamik ve karmaşık bir kripto piyasası veri setiyle çalışırken, başarılı bir model geliştirmek için sistematik bir yaklaşım izlemek önemlidir. İşte adım adım izleyebileceğiniz bir yol haritası:\n\n1. Veri Keşfi ve Ön İşleme (EDA & Preprocessing)\nBu ilk adım, veriyi anlamak ve modelleme için hazırlamak için hayati öneme sahiptir.\n\nVeri Yükleme ve İlk İnceleme:\n\ntrain.parquet ve test.parquet dosyalarını yükleyin (pandas.read_parquet).\n\nVeri setlerinin boyutlarına (shape), sütun isimlerine (columns), veri tiplerine (dtypes) ve bellek kullanımına (info()) bakın.\n\nİlk birkaç satırını (head()) ve son birkaç satırını (tail()) inceleyerek verinin genel yapısını gözlemleyin.\n\nEksik Değer Analizi:\n\nHer sütunda ne kadar NaN (Not a Number) değeri olduğunu kontrol edin (isnull().sum()).\n\nNaN değerlerinin dağılımını ve yoğunluğunu anlamak için görselleştirmeler kullanabilirsiniz (ısı haritaları vb.).\n\nEksik Değerleri Ele Alma:\n\nKaldırma: Çok fazla NaN içeren sütunları veya satırları tamamen atmayı düşünebilirsiniz (ancak dikkatli olun, bilgi kaybına yol açabilir).\n\nDoldurma (Imputation): NaN değerleri için uygun stratejiler belirleyin.\n\nSayısal özellikler için ortalama (mean), medyan (median) veya sıfır (0) ile doldurma.\n\nÖzelliklerin dağılımına bakarak en uygun yöntemi seçin.\n\nÖrneğin, bid_qty, ask_qty, buy_qty, sell_qty, volume gibi özellikler için 0 ile doldurmak mantıklı olabilir (işlem veya teklif yok). X_ özelliklerinde ise ortalama veya medyan daha uygun olabilir.\n\nnp.inf (sonsuz) değerleri de kontrol edin ve uygun şekilde (örneğin np.nan_to_num kullanarak) ele alın.\n\nAykırı Değer (Outlier) Tespiti ve Yönetimi:\n\nKutu grafikleri (boxplot) veya dağılım grafikleri (histogram) kullanarak aykırı değerleri görselleştirin.\n\nAykırı değerlerin nedenlerini anlamaya çalışın (veri girişi hatası mı, gerçek piyasa olayı mı?).\n\nAykırı değerleri kırpma (clipping) (belirli bir aralıkta sınırlama) veya dönüştürme (transformation) (logaritmik dönüşüm gibi) yöntemleriyle ele alabilirsiniz.\n\nVeri Tiplerini Optimize Etme:\n\nBüyük veri setlerinde bellek kullanımını azaltmak için sayısal sütunların veri tiplerini optimize edin (örn. float64 yerine float32, int64 yerine int32 veya daha küçük tipler).\n\n2. Özellik Mühendisliği (Feature Engineering)\nBu kısım, modelinizin performansını doğrudan etkileyecek yeni ve anlamlı özellikler oluşturmakla ilgilidir.\n\nZaman Serisi Özellikleri:\n\ntimestamp sütunundan zaman tabanlı özellikler türetin:\n\nGün, hafta içi/hafta sonu, ay, yıl, günün saati, dakikanın başlangıcı/sonu. Kripto piyasası 7/24 çalıştığı için, hafta içi/hafta sonu etkisi veya belirli saatlerdeki volatilite önemli olabilir.\n\nÖzellikle test veri setindeki zaman damgaları karıştırılmış ve maskelenmiş olsa da, eğitim verisindeki zaman damgalarını kullanarak döngüsel (cyclical) özellikler (sinüs/kosinüs dönüşümü) oluşturabilirsiniz.\n\nGecikmeli Özellikler (Lag Features):\n\nGeçmiş zaman adımlarındaki değerleri kullanarak yeni özellikler oluşturun. Örneğin, son 1, 5, 10 dakikadaki volume, bid_qty, ask_qty değerleri.\n\nlabel için de gecikmeli değerler kullanmayı düşünebilirsiniz (ancak test setinde bu mümkün olmayacaktır).\n\nDikkat: test.parquet'teki timestamp'lerin maskelenmiş olması, gecikmeli özelliklerin oluşturulmasında sıralamanın doğru yapılmasını zorlaştıracaktır. Eğitim verisinde bunu yaparken dikkatli olun ve test verisi için bu tip özellikleri oluştururken ID'ye güvenmeyin, zira ID'ler sıralı zamanı temsil etmiyor.\n\nHareketli Ortalamalar (Moving Averages):\n\nBelirli bir pencere üzerindeki volume, bid_qty, ask_qty gibi özelliklerin hareketli ortalamalarını hesaplayın. Bu, kısa vadeli gürültüyü azaltarak trendleri yakalamaya yardımcı olabilir.\n\nVolatilite Özellikleri:\n\nBelirli bir pencere üzerindeki bid_qty, ask_qty veya işlem hacimlerinin standart sapması gibi volatilite ölçümleri.\n\nOranlar ve Farklar:\n\nbid_qty / ask_qty gibi oranlar (alım-satım dengesi).\n\nbuy_qty - sell_qty gibi farklar (net emir akışı).\n\nbid_qty + ask_qty (derinlik toplamı).\n\nÖzel X_ Özellikleri İçin Etkileşimler:\n\nAnonimleştirilmiş X_ özellikleri arasındaki potansiyel etkileşimleri keşfedin. Bazı özelliklerin birbiriyle çarpımı veya oranları anlamlı olabilir.\n\nBoyut indirgeme teknikleri (PCA gibi) kullanarak X_ özelliklerinin özetini çıkarabilirsiniz, ancak bu, orijinal özelliklerin yorumlanabilirliğini azaltabilir.\n\n3. Model Seçimi ve Eğitimi\nVeri setinizin yapısı ve hedeflenen çıktı (sürekli bir değer tahmini, Pearson korelasyonu ile değerlendirme) göz önüne alındığında, regresyon modelleri uygun olacaktır.\n\nModel Seçimi:\n\nGradient Boosting Modelleri: LightGBM, XGBoost, CatBoost gibi modeller zaman serisi verilerinde ve yüksek boyutlu verilerde genellikle çok iyi performans gösterirler. Hızlı olmaları ve kategorik/sayısal veriyi iyi işlemeleri avantajdır.\n\nGeleneksel Regresyon Modelleri: Ridge, Lasso gibi doğrusal modeller de başlangıç noktası olarak kullanılabilir, ancak karmaşık piyasa dinamiklerini yakalamakta yetersiz kalabilirler.\n\nDerin Öğrenme Modelleri: LSTM (Uzun Kısa Vadeli Bellek) veya Transformer tabanlı modeller, zaman serisi verileri için güçlü olabilir. Ancak, bu modeller daha fazla veri ve hesaplama gücü gerektirir ve genellikle daha karmaşıktır. Bu seviyedeki bir problem için gradient boosting modelleri genellikle iyi bir başlangıç noktasıdır.\n\nEğitim Stratejisi (Time Series Split):\n\nVerinin zaman serisi yapısından dolayı, çapraz doğrulama (cross-validation) yaparken standart K-Fold yerine zaman serisi tabanlı çapraz doğrulama (Time Series Split) kullanmalısınız. Bu, eğitim setinin her zaman test setinden önceki zaman dilimlerinden oluşmasını sağlar ve \"geleceği görmeyi\" engeller. sklearn.model_selection.TimeSeriesSplit bu iş için idealdir.\n\nUnutmayın: \"En son aylara odaklanın\" ipucunu göz önünde bulundurarak, eğitim setinizin sadece son kısmını kullanmayı düşünebilirsiniz.\n\nHiperparametre Optimizasyonu:\n\nModelinizin performansını artırmak için hiperparametrelerini optimize edin. GridSearchCV, RandomizedSearchCV veya Optuna, Hyperopt gibi kütüphaneler kullanılabilir.\n\n4. Değerlendirme ve İyileştirme\nModelinizi değerlendirmek ve performansını artırmak için sürekli bir döngü izleyin.\n\nDeğerlendirme Metriği:\n\nModelinizin performansını Pearson korelasyon katsayısı (Sklearn'de pearsonr veya corr fonksiyonları ile hesaplanabilir) kullanarak izleyin.\n\nModelin Yorumlanabilirliği:\n\nLightGBM veya XGBoost gibi modeller için özellik önemini (feature importance) inceleyerek hangi özelliklerin tahminlerinizde en etkili olduğunu anlayın. Bu, yeni özellik mühendisliği fikirleri için yol gösterebilir.\n\nHata Analizi:\n\nModelinizin en çok hangi durumlarda hatalı tahminler yaptığını analiz edin. Hataların belirli piyasa koşullarıyla (yüksek volatilite, düşük hacim vb.) ilişkili olup olmadığını inceleyin.\n\n5. Tahmin ve Gönderim\nSon adımlar, modelinizi test verisi üzerinde çalıştırmak ve sonuçları göndermektir.\n\nTest Verisi Tahmini:\n\nEğittiğiniz modeli test.parquet üzerindeki özellikler (tabii ki label sütunu hariç) kullanarak tahminler yapın.\n\nTest verisindeki timestamp'lerin maskeli olduğunu unutmayın. Gecikmeli özellikler gibi zaman bağımlı özelliklerin eğitimde oluşturulduğu mantıkla, test veri setinin ID yapısı kullanılarak uygun şekilde oluşturulması gerekir. Bu, genellikle test verisinin her satırı için sadece o ana kadar mevcut olan bilgiyi kullanmanız gerektiği anlamına gelir.\n\nGönderim Dosyası Oluşturma:\n\nTahminlerinizi sample_submission.csv formatına uygun olarak hazırlayın. Genellikle bu, bir ID sütunu ve bir label sütunu içerir.\n\nGenel İpuçları ve Ek Hususlar\nBellek Yönetimi: Kripto verileri büyük olabilir. Parquet formatının kullanılması bellek dostudur. Ancak yine de büyük veri setleriyle çalışırken bellek tüketimini izleyin ve gerekirse veri tiplerini küçültün veya veriyi parça parça işleyin.\n\nHesaplama Kaynakları: Özellikle derin öğrenme veya yoğun hiperparametre optimizasyonu yapacaksanız, yeterli CPU/GPU kaynaklarına sahip olduğunuzdan emin olun (Kaggle Notebook'ları GPU/TPU desteği sunar).\n\nFuture Peeking'den Kaçınma: Bu yarışmanın en kritik kurallarından biridir. Modelinizi eğitirken veya özellik mühendisliği yaparken, asla test setindeki (veya gelecekteki harici veri setlerindeki) gelecekteki bilgilere bakmayın.\n\nKod Organizasyonu: Kodunuzu temiz, modüler ve okunabilir tutun. Fonksiyonlar ve sınıflar kullanmak, karmaşık bir projeyi yönetmeyi kolaylaştırır.\n\nSürüm Kontrolü: Çalışmalarınızı takip etmek için (örneğin Git) kullanın veya Kaggle Notebook'larında farklı sürümleri kaydedin.","metadata":{}},{"cell_type":"code","source":"df['hour'] = df.index.hour\ndf['dayofweek'] = df.index.dayofweek\n# One-hot encoding for categorical time features if necessary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.648698Z","iopub.status.idle":"2025-07-26T15:16:41.649003Z","shell.execute_reply.started":"2025-07-26T15:16:41.648862Z","shell.execute_reply":"2025-07-26T15:16:41.648874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Örnek bir DataFrame oluşturalım (gerçek veri setinizi bununla değiştireceksiniz)\n# train.parquet veya test.parquet dosyanızı yükleyin\n# df = pd.read_parquet('train.parquet')\n# Veya örnek olması için rastgele NaN içeren bir DataFrame oluşturalım:\n\n\ndf = pd.DataFrame(df)\n\nprint(\"DataFrame'in İlk 5 Satırı:\")\nprint(df.head())\nprint(\"\\nDataFrame'deki NaN Sayıları:\")\nprint(df.isnull().sum())\n\n# --- NaN Değerleri İçin Isı Haritası (Heatmap) ---\nplt.figure(figsize=(10, 6))\n# df.isnull() DataFrame'deki her hücre için True (NaN ise) veya False döndürür.\n# Isı haritası bu True/False değerlerini renk olarak yorumlar (varsayılan olarak True için daha sıcak renk).\nsns.heatmap(df.isnull(), cbar=False, cmap='viridis') # 'viridis' renk haritası, NaN'ları belirginleştirir\nplt.title('DataFrame\\'deki NaN Değerlerin Isı Haritası')\nplt.xlabel('Özellikler')\nplt.ylabel('Satır İndeksi')\nplt.show()\n\n# --- NaN Oranları İçin Çubuk Grafik (Bar Plot) ---\n# Her sütundaki NaN oranını hesaplayalım\nnan_percentages = df.isnull().sum() / len(df) * 100\nnan_percentages = nan_percentages.sort_values(ascending=False) # Azalan sıraya göre sırala\n\nplt.figure(figsize=(12, 7))\nsns.barplot(x=nan_percentages.index, y=nan_percentages.values, palette='plasma')\nplt.title('Her Sütundaki NaN Oranı (%)')\nplt.xlabel('Özellikler')\nplt.ylabel('NaN Oranı (%)')\nplt.xticks(rotation=45, ha='right') # Sütun isimlerini okunur yapmak için döndürme\nplt.tight_layout() # Grafiğin düzenini iyileştirir\nplt.show()\n\n# --- Özelliklerin Yüzde Kaçının NaN Olduğunu Gösteren Metinsel Çıktı ---\nprint(\"\\nHer Sütundaki NaN Oranları (Yüzde):\")\nprint(nan_percentages[nan_percentages > 0]) # Sadece NaN içeren sütunları göster","metadata":{"execution":{"iopub.status.busy":"2025-07-26T14:17:01.795710Z","iopub.execute_input":"2025-07-26T14:17:01.796035Z","execution_failed":"2025-07-26T14:17:10.696Z"}}},{"cell_type":"markdown","source":"sns.pairplot(df,hue=\"volume\");","metadata":{"execution":{"iopub.status.busy":"2025-07-26T14:55:07.698820Z","iopub.execute_input":"2025-07-26T14:55:07.699139Z","iopub.status.idle":"2025-07-26T14:55:07.785125Z","shell.execute_reply.started":"2025-07-26T14:55:07.699115Z","shell.execute_reply":"2025-07-26T14:55:07.783952Z"}}},{"cell_type":"markdown","source":"sns.pairplot(df,hue=\"label\");","metadata":{}},{"cell_type":"code","source":"df_handled_manual = df.copy() # Orijinal DataFrame'i korumak için bir kopyasını alalım\n\n# Maksimum/minimum eşik değerleri belirleyelim (veri dağılımına göre ayarlanabilir)\n# Örneğin, 1e10 (10 milyar) veya verinin 99. yüzdelik dilimi gibi\nMAX_VALUE = 1e10\nMIN_VALUE = -1e10\n\n# Sonsuzlukları belirli bir değerle değiştirme\nfor col in df_handled_manual.select_dtypes(include=np.number).columns:\n    df_handled_manual[col] = df_handled_manual[col].replace(np.inf, MAX_VALUE)\n    df_handled_manual[col] = df_handled_manual[col].replace(-np.inf, MIN_VALUE)\n\n# NaN değerleri için ek olarak doldurma yapılabilir (eğer henüz yapılmadıysa)\n# df_handled_manual = df_handled_manual.fillna(df_handled_manual.mean(numeric_only=True))\n\nprint(\"\\nManuel Değer Atama Kullanılarak Inf Değerleri İşlendikten Sonra:\")\nprint(df_handled_manual.head())\n\n# İşlem sonrası sonsuz değer kontrolü\nprint(\"\\nİşlem Sonrası Sonsuz Değer Kontrolü:\")\nprint((df_handled_manual == np.inf).sum().sum())\nprint((df_handled_manual == -np.inf).sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.650948Z","iopub.status.idle":"2025-07-26T15:16:41.651308Z","shell.execute_reply.started":"2025-07-26T15:16:41.651140Z","shell.execute_reply":"2025-07-26T15:16:41.651155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef optimize_dataframe_memory(df):\n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type).startswith('float'):\n            if df[col].isnull().any(): # NaN varsa float'ta kalmalı\n                continue # NaN içeren float sütunları şimdilik dönüştürmeyelim\n            min_val = df[col].min()\n            max_val = df[col].max()\n            if min_val > np.finfo(np.float32).min and max_val < np.finfo(np.float32).max:\n                df[col] = df[col].astype(np.float32)\n        elif str(col_type).startswith('int'):\n            min_val = df[col].min()\n            max_val = df[col].max()\n            if min_val > np.iinfo(np.int8).min and max_val < np.iinfo(np.int8).max:\n                df[col] = df[col].astype(np.int8)\n            elif min_val > np.iinfo(np.int16).min and max_val < np.iinfo(np.int16).max:\n                df[col] = df[col].astype(np.int16)\n            elif min_val > np.iinfo(np.int32).min and max_val < np.iinfo(np.int32).max:\n                df[col] = df[col].astype(np.int32)\n    return df\n\n# Örneğin train.parquet dosyasını yüklerken:\n  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.652072Z","iopub.status.idle":"2025-07-26T15:16:41.652374Z","shell.execute_reply.started":"2025-07-26T15:16:41.652220Z","shell.execute_reply":"2025-07-26T15:16:41.652233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:38:34.396148Z","iopub.execute_input":"2025-07-26T15:38:34.396530Z","iopub.status.idle":"2025-07-26T15:38:34.408043Z","shell.execute_reply.started":"2025-07-26T15:38:34.396501Z","shell.execute_reply":"2025-07-26T15:38:34.403081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.654155Z","iopub.status.idle":"2025-07-26T15:16:41.654406Z","shell.execute_reply.started":"2025-07-26T15:16:41.654298Z","shell.execute_reply":"2025-07-26T15:16:41.654309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X=df2.drop(\"label\", axis=1)\ny=df2.label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:38:41.081226Z","iopub.execute_input":"2025-07-26T15:38:41.081633Z","iopub.status.idle":"2025-07-26T15:38:42.169809Z","shell.execute_reply.started":"2025-07-26T15:38:41.081599Z","shell.execute_reply":"2025-07-26T15:38:42.165201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(X,y,test_size=0.2, random_state=101)\n\nprint(\"Train features shape : \", X_train.shape)\nprint(\"Train target shape   : \", y_train.shape)\nprint(\"Test features shape  : \", X_test.shape)\nprint(\"Test target shape    : \", y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:38:49.938676Z","iopub.execute_input":"2025-07-26T15:38:49.939000Z","iopub.status.idle":"2025-07-26T15:38:53.404735Z","shell.execute_reply.started":"2025-07-26T15:38:49.938976Z","shell.execute_reply":"2025-07-26T15:38:53.400254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:03.150983Z","iopub.execute_input":"2025-07-26T15:39:03.151291Z","iopub.status.idle":"2025-07-26T15:39:03.161783Z","shell.execute_reply.started":"2025-07-26T15:39:03.151266Z","shell.execute_reply":"2025-07-26T15:39:03.157176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor # This is the boosting model\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:06.446069Z","iopub.execute_input":"2025-07-26T15:39:06.446391Z","iopub.status.idle":"2025-07-26T15:39:06.457058Z","shell.execute_reply.started":"2025-07-26T15:39:06.446357Z","shell.execute_reply":"2025-07-26T15:39:06.452566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Infinite values in X_train:\", np.isinf(X_train).sum())\nprint(\"Infinite values in X_test:\", np.isinf(X_test).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:10.809875Z","iopub.execute_input":"2025-07-26T15:39:10.810243Z","iopub.status.idle":"2025-07-26T15:39:11.514886Z","shell.execute_reply.started":"2025-07-26T15:39:10.810211Z","shell.execute_reply":"2025-07-26T15:39:11.510427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Max value in X_train:\", X_train.max())\nprint(\"Min value in X_train:\", X_train.min())\nprint(\"Max value in X_test:\", X_test.max())\nprint(\"Min value in X_test:\", X_test.min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:16.904336Z","iopub.execute_input":"2025-07-26T15:39:16.904652Z","iopub.status.idle":"2025-07-26T15:39:18.958712Z","shell.execute_reply.started":"2025-07-26T15:39:16.904625Z","shell.execute_reply":"2025-07-26T15:39:18.954131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Max values per column in X_train:\\n\", X_train.max())\nprint(\"Min values per column in X_train:\\n\", X_train.min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:28:28.417523Z","iopub.execute_input":"2025-07-26T15:28:28.417816Z","iopub.status.idle":"2025-07-26T15:28:30.078860Z","shell.execute_reply.started":"2025-07-26T15:28:28.417791Z","shell.execute_reply":"2025-07-26T15:28:30.073429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = GradientBoostingRegressor(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:28.897357Z","iopub.execute_input":"2025-07-26T15:39:28.897776Z","iopub.status.idle":"2025-07-26T15:39:28.908827Z","shell.execute_reply.started":"2025-07-26T15:39:28.897744Z","shell.execute_reply":"2025-07-26T15:39:28.904257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Shape of X_train BEFORE imputation: {X_train.shape}\")\nprint(f\"Number of NaN values in X_train BEFORE imputation:\\n{X_train.isna().sum()}\")\nprint(f\"Number of Infinite values in X_train BEFORE imputation:\\n{(X_train == np.inf).sum() + (X_train == -np.inf).sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:34.714949Z","iopub.execute_input":"2025-07-26T15:39:34.715253Z","iopub.status.idle":"2025-07-26T15:39:36.326514Z","shell.execute_reply.started":"2025-07-26T15:39:34.715229Z","shell.execute_reply":"2025-07-26T15:39:36.320719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For X_train\nfinite_mask_train = np.isfinite(X_train).all(axis=1)\nX_train = X_train[finite_mask_train]\ny_train = y_train[finite_mask_train] # Remember to apply to y_train as well!\n\n# For X_test\nfinite_mask_test = np.isfinite(X_test).all(axis=1)\nX_test = X_test[finite_mask_test]\ny_test = y_test[finite_mask_test] # Remember to apply to y_test as well!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:45.257104Z","iopub.execute_input":"2025-07-26T15:39:45.257440Z","iopub.status.idle":"2025-07-26T15:39:45.750037Z","shell.execute_reply.started":"2025-07-26T15:39:45.257413Z","shell.execute_reply":"2025-07-26T15:39:45.743495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# After X_train = X_train.astype(float) and X_train.replace()\n# Find columns that are all NaN in the training set\nall_nan_cols = X_train.columns[X_train.isnull().all()].tolist()\nif all_nan_cols:\n    print(f\"Dropping columns that are entirely NaN in X_train: {all_nan_cols}\")\n    X_train.drop(columns=all_nan_cols, inplace=True)\n    X_test.drop(columns=all_nan_cols, inplace=True) # Apply to test set as well!\n    # Update original_cols for later DataFrame conversion if needed\n    # original_cols = [col for col in original_cols if col not in all_nan_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:39:53.390370Z","iopub.execute_input":"2025-07-26T15:39:53.390672Z","iopub.status.idle":"2025-07-26T15:39:53.404436Z","shell.execute_reply.started":"2025-07-26T15:39:53.390646Z","shell.execute_reply":"2025-07-26T15:39:53.398817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\n# --- 1. Load your actual data here instead of generating dummy data ---\n# Example: df = pd.read_csv('your_data.csv')\n# X = df.drop('target_column', axis=1)\n# y = df['target_column']\n\n# For demonstration, let's create data that will cause the overflow\nnp.random.seed(42)\nnum_samples = 100\nnum_features = 5\nX = pd.DataFrame(np.random.rand(num_samples, num_features), columns=[f'feature_{i}' for i in range(num_features)])\ny = 2 * X['feature_0'] + 3 * X['feature_1'] - 0.5 * X['feature_2'] + np.random.randn(num_samples) * 0.5\n\n# Intentionally introduce some very large numbers and infinities\nX.iloc[5, 0] = 1e300\nX.iloc[10, 2] = -1e250\nX.iloc[15, 4] = np.inf\nX.iloc[20, 1] = -np.inf\n\n# New: Introduce a column that will become entirely NaN in X_train\n# Let's say feature_0 is problematic in the first 80 rows (mostly in train)\n# We make it so that for some split, it could be entirely inf/nan in the train set.\nX.iloc[np.random.choice(X.index, 20, replace=False), 0] = np.inf\n\n\n# --- 2. Split data into training and testing sets ---\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# --- DEBUGGING: Initial Check ---\nprint(f\"--- Initial Data State (before any processing) ---\")\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_train contains NaN: {X_train.isnull().any().any()}\")\nprint(f\"X_train contains Inf: {np.isinf(X_train).any().any()}\")\nif X_train.shape[0] > 0:\n    print(f\"Max value in X_train: {X_train.max().max()}\")\n    print(f\"Min value in X_train: {X_train.min().min()}\")\nprint(\"-\" * 40)\n\n\n# --- 3. Handle infinite and very large values (Replacing with NaN then Imputing) ---\n\n# IMPORTANT: Ensure data is float type for proper NaN/inf handling\nX_train = X_train.astype(float)\nX_test = X_test.astype(float)\n\n# Replace explicit infinite values with NaN\nX_train.replace([np.inf, -np.inf], np.nan, inplace=True)\nX_test.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# --- DEBUGGING: After NaN replacement ---\nprint(f\"--- After Replacing Inf with NaN ---\")\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_train contains NaN: {X_train.isnull().any().any()}\")\nprint(f\"X_train contains Inf (should be False): {np.isinf(X_train).any().any()}\")\nif X_train.shape[0] > 0:\n    print(f\"Max value in X_train: {X_train.max().max()}\")\n    print(f\"Min value in X_train: {X_train.min().min()}\")\nprint(\"-\" * 40)\n\n# --- FIX: Drop columns that are entirely NaN in the training set ---\n# This is crucial if some columns became all NaNs after the replacement step.\nall_nan_cols = X_train.columns[X_train.isnull().all()].tolist()\nif all_nan_cols:\n    print(f\"Dropping columns that are entirely NaN in X_train: {all_nan_cols}\")\n    X_train.drop(columns=all_nan_cols, inplace=True)\n    # Ensure to drop the same columns from the test set to maintain consistency\n    X_test.drop(columns=all_nan_cols, inplace=True)\n    print(f\"New X_train shape after dropping all-NaN columns: {X_train.shape}\")\n    print(f\"New X_test shape after dropping all-NaN columns: {X_test.shape}\")\nprint(\"-\" * 40)\n\n\n# Define an imputer (e.g., mean imputation)\nimputer = SimpleImputer(strategy='mean')\n\n# Fit imputer ONLY on X_train to prevent data leakage\nX_train_processed = imputer.fit_transform(X_train)\nX_test_processed = imputer.transform(X_test)\n\n# --- DEBUGGING: After Imputation ---\nprint(f\"--- After Imputation (before float64 cast) ---\")\nprint(f\"X_train_processed shape: {X_train_processed.shape}\")\nprint(f\"X_train_processed contains NaN: {np.isnan(X_train_processed).any()}\") # Should now be False\nprint(f\"X_train_processed contains Inf: {np.isinf(X_train_processed).any()}\")\nif X_train_processed.shape[0] > 0:\n    print(f\"Max value in X_train_processed: {X_train_processed.max()}\")\n    print(f\"Min value in X_train_processed: {X_train_processed.min()}\")\nprint(\"-\" * 40)\n\n# --- Ensure processed arrays are float64 ---\nX_train_processed = X_train_processed.astype(np.float64)\nX_test_processed = X_test_processed.astype(np.float64)\n\n# --- DEBUGGING: After float64 cast ---\nprint(f\"--- After Explicit float64 Cast ---\")\nprint(f\"X_train_processed dtype: {X_train_processed.dtype}\")\nprint(f\"X_train_processed contains NaN: {np.isnan(X_train_processed).any()}\") # Should be False\nprint(f\"X_train_processed contains Inf: {np.isinf(X_train_processed).any()}\") # Should be False\nif X_train_processed.shape[0] > 0:\n    print(f\"Max value in X_train_processed: {X_train_processed.max()}\")\n    print(f\"Min value in X_train_processed: {X_train_processed.min()}\")\nprint(\"-\" * 40)\n\n\n# --- Apply Scaling ---\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_processed)\nX_test_scaled = scaler.transform(X_test_processed)\n\n# --- DEBUGGING: After Scaling ---\nprint(f\"--- After Scaling ---\")\nprint(f\"X_train_scaled shape: {X_train_scaled.shape}\")\nprint(f\"X_train_scaled contains NaN: {np.isnan(X_train_scaled).any()}\") # Should be False\nprint(f\"X_train_scaled contains Inf: {np.isinf(X_train_scaled).any()}\") # Should be False\nif X_train_scaled.shape[0] > 0:\n    print(f\"Max value in X_train_scaled: {X_train_scaled.max()}\")\n    print(f\"Min value in X_train_scaled: {X_train_scaled.min()}\")\nprint(\"-\" * 40)\n\n\n# --- 4. Instantiate and train the boosting model ---\nmodel = GradientBoostingRegressor(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\nmodel.fit(X_train_scaled, y_train)\n\n# --- 5. Make predictions ---\ny_pred = model.predict(X_test_scaled)\ny_train_pred = model.predict(X_train_scaled)\n\n# --- 6. Calculate and store scores ---\nscores = {\n    \"train\": {\n        \"R2\": r2_score(y_train, y_train_pred),\n        \"mae\": mean_absolute_error(y_train, y_train_pred),\n        \"mse\": mean_squared_error(y_train, y_train_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_train, y_train_pred))\n    },\n    \"test\": {\n        \"R2\": r2_score(y_test, y_pred),\n        \"mae\": mean_absolute_error(y_test, y_pred),\n        \"mse\": mean_squared_error(y_test, y_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_test, y_pred))\n    }\n}\n\n# --- 7. Return as DataFrame ---\nscores_df = pd.DataFrame(scores)\nprint(scores_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:40:02.374088Z","iopub.execute_input":"2025-07-26T15:40:02.374447Z","iopub.status.idle":"2025-07-26T15:40:02.547436Z","shell.execute_reply.started":"2025-07-26T15:40:02.374418Z","shell.execute_reply":"2025-07-26T15:40:02.542866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\n# --- 1. Load your actual data here instead of generating dummy data ---\n# Example: df = pd.read_csv('your_data.csv')\n# X = df.drop('target_column', axis=1)\n# y = df['target_column']\n\n# For demonstration, let's create data that will cause the overflow\nnp.random.seed(42)\nnum_samples = 100\nnum_features = 5\nX = pd.DataFrame(np.random.rand(num_samples, num_features), columns=[f'feature_{i}' for i in range(num_features)])\ny = 2 * X['feature_0'] + 3 * X['feature_1'] - 0.5 * X['feature_2'] + np.random.randn(num_samples) * 0.5\n\n# Intentionally introduce some very large numbers and infinities\nX.iloc[5, 0] = 1e300\nX.iloc[10, 2] = -1e250\nX.iloc[15, 4] = np.inf\nX.iloc[20, 1] = -np.inf\n\n# New: Introduce a column that will become entirely NaN in X_train\n# Let's say feature_0 is problematic in the first 80 rows (mostly in train)\n# We make it so that for some split, it could be entirely inf/nan in the train set.\nX.iloc[np.random.choice(X.index, 20, replace=False), 0] = np.inf\n\n\n# --- 2. Split data into training and testing sets ---\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# --- DEBUGGING: Initial Check ---\nprint(f\"--- Initial Data State (before any processing) ---\")\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_train contains NaN: {X_train.isnull().any().any()}\")\nprint(f\"X_train contains Inf: {np.isinf(X_train).any().any()}\")\nif X_train.shape[0] > 0:\n    print(f\"Max value in X_train: {X_train.max().max()}\")\n    print(f\"Min value in X_train: {X_train.min().min()}\")\nprint(\"-\" * 40)\n\n\n# --- 3. Handle infinite and very large values (Replacing with NaN then Imputing) ---\n\n# IMPORTANT: Ensure data is float type for proper NaN/inf handling\nX_train = X_train.astype(float)\nX_test = X_test.astype(float)\n\n# Replace explicit infinite values with NaN\nX_train.replace([np.inf, -np.inf], np.nan, inplace=True)\nX_test.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# --- DEBUGGING: After NaN replacement ---\nprint(f\"--- After Replacing Inf with NaN ---\")\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_train contains NaN: {X_train.isnull().any().any()}\")\nprint(f\"X_train contains Inf (should be False): {np.isinf(X_train).any().any()}\")\nif X_train.shape[0] > 0:\n    print(f\"Max value in X_train: {X_train.max().max()}\")\n    print(f\"Min value in X_train: {X_train.min().min()}\")\nprint(\"-\" * 40)\n\n# --- FIX: Drop columns that are entirely NaN in the training set ---\n# This is crucial if some columns became all NaNs after the replacement step.\nall_nan_cols = X_train.columns[X_train.isnull().all()].tolist()\nif all_nan_cols:\n    print(f\"Dropping columns that are entirely NaN in X_train: {all_nan_cols}\")\n    X_train.drop(columns=all_nan_cols, inplace=True)\n    # Ensure to drop the same columns from the test set to maintain consistency\n    X_test.drop(columns=all_nan_cols, inplace=True)\n    print(f\"New X_train shape after dropping all-NaN columns: {X_train.shape}\")\n    print(f\"New X_test shape after dropping all-NaN columns: {X_test.shape}\")\nprint(\"-\" * 40)\n\n\n# Define an imputer (e.g., mean imputation)\nimputer = SimpleImputer(strategy='mean')\n\n# Fit imputer ONLY on X_train to prevent data leakage\nX_train_processed = imputer.fit_transform(X_train)\nX_test_processed = imputer.transform(X_test)\n\n# --- DEBUGGING: After Imputation ---\nprint(f\"--- After Imputation (before float64 cast) ---\")\nprint(f\"X_train_processed shape: {X_train_processed.shape}\")\nprint(f\"X_train_processed contains NaN: {np.isnan(X_train_processed).any()}\") # Should now be False\nprint(f\"X_train_processed contains Inf: {np.isinf(X_train_processed).any()}\")\nif X_train_processed.shape[0] > 0:\n    print(f\"Max value in X_train_processed: {X_train_processed.max()}\")\n    print(f\"Min value in X_train_processed: {X_train_processed.min()}\")\nprint(\"-\" * 40)\n\n# --- Ensure processed arrays are float64 ---\nX_train_processed = X_train_processed.astype(np.float64)\nX_test_processed = X_test_processed.astype(np.float64)\n\n# --- DEBUGGING: After float64 cast ---\nprint(f\"--- After Explicit float64 Cast ---\")\nprint(f\"X_train_processed dtype: {X_train_processed.dtype}\")\nprint(f\"X_train_processed contains NaN: {np.isnan(X_train_processed).any()}\") # Should be False\nprint(f\"X_train_processed contains Inf: {np.isinf(X_train_processed).any()}\") # Should be False\nif X_train_processed.shape[0] > 0:\n    print(f\"Max value in X_train_processed: {X_train_processed.max()}\")\n    print(f\"Min value in X_train_processed: {X_train_processed.min()}\")\nprint(\"-\" * 40)\n\n\n# --- Apply Scaling ---\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_processed)\nX_test_scaled = scaler.transform(X_test_processed)\n\n# --- DEBUGGING: After Scaling ---\nprint(f\"--- After Scaling ---\")\nprint(f\"X_train_scaled shape: {X_train_scaled.shape}\")\nprint(f\"X_train_scaled contains NaN: {np.isnan(X_train_scaled).any()}\") # Should be False\nprint(f\"X_train_scaled contains Inf: {np.isinf(X_train_scaled).any()}\") # Should be False\nif X_train_scaled.shape[0] > 0:\n    print(f\"Max value in X_train_scaled: {X_train_scaled.max()}\")\n    print(f\"Min value in X_train_scaled: {X_train_scaled.min()}\")\nprint(\"-\" * 40)\n\n\n# --- 4. Instantiate and train the boosting model ---\nmodel = GradientBoostingRegressor(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\nmodel.fit(X_train_scaled, y_train)\n\n# --- 5. Make predictions ---\ny_pred = model.predict(X_test_scaled)\ny_train_pred = model.predict(X_train_scaled)\n\n# --- 6. Calculate and store scores ---\nscores = {\n    \"train\": {\n        \"R2\": r2_score(y_train, y_train_pred),\n        \"mae\": mean_absolute_error(y_train, y_train_pred),\n        \"mse\": mean_squared_error(y_train, y_train_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_train, y_train_pred))\n    },\n    \"test\": {\n        \"R2\": r2_score(y_test, y_pred),\n        \"mae\": mean_absolute_error(y_test, y_pred),\n        \"mse\": mean_squared_error(y_test, y_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_test, y_pred))\n    }\n}\n\n# --- 7. Return as DataFrame ---\nscores_df = pd.DataFrame(scores)\nprint(scores_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:40:17.538022Z","iopub.execute_input":"2025-07-26T15:40:17.538361Z","iopub.status.idle":"2025-07-26T15:40:17.704286Z","shell.execute_reply.started":"2025-07-26T15:40:17.538333Z","shell.execute_reply":"2025-07-26T15:40:17.699153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\nfrom sklearn.preprocessing import StandardScaler\n\n# --- 1. Verinizi buraya yükleyin ---\n# Örnek: df = pd.read_csv('veriniz.csv')\n# X = df.drop('hedef_sütun', axis=1)\n# y = df['hedef_sütun']\n\n# Örnek veri oluşturma (kendi verinizi buraya yapıştırın)\nnp.random.seed(42)\nnum_samples = 100\nnum_features = 5\nX = pd.DataFrame(np.random.rand(num_samples, num_features), columns=[f'feature_{i}' for i in range(num_features)])\ny = 2 * X['feature_0'] + 3 * X['feature_1'] - 0.5 * X['feature_2'] + np.random.randn(num_samples) * 0.5\n\n# Test için kasıtlı olarak NaN ve Inf değerler ekleyelim\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = np.inf\nX.iloc[15, 4] = -np.inf\nX.iloc[20, 1] = 1e40 # float32 overflow yapabilecek çok büyük bir sayı\ny.iloc[7] = np.nan # y'ye de NaN ekleyelim\n\n# --- DEBUGGING: Başlangıç Veri Durumu ---\nprint(f\"--- Başlangıç Veri Durumu (İşleme Öncesi) ---\")\nprint(f\"X boyutu: {X.shape}\")\nprint(f\"y boyutu: {y.shape}\")\nprint(f\"X'teki toplam NaN sayısı: {X.isnull().sum().sum()}\")\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X).sum().sum()}\")\nprint(f\"y'deki toplam NaN sayısı: {y.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 2. Tüm NaN ve Inf değerleri temizle (ana FIX) ---\n\n# Adım 1: Tüm sonsuz değerleri NaN'a dönüştür (eğer hala varsa)\n# Bu, .dropna() ile birlikte çalışmak için önemlidir.\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# Adım 2: X ve y'yi birleştirerek birlikte NaN içeren satırları bul\n# Bu, X'teki veya y'deki NaN'ları içeren tüm satırları kaldırmamızı sağlar.\ndf_combined = pd.concat([X, y.rename('target')], axis=1) # y'yi DataFrame'e ekle\ndf_cleaned = df_combined.dropna() # Tüm NaN içeren satırları sil\n\n# Temizlenmiş veriyi X ve y'ye geri ayır\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\n# --- DEBUGGING: Temizleme Sonrası Durum ---\nprint(f\"--- Temizleme Sonrası Veri Durumu ---\")\nprint(f\"X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki toplam NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\nprint(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned).sum().sum()}\")\nprint(f\"y_cleaned'daki toplam NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 3. Veriyi eğitim ve test setlerine ayır ---\n# Temizlenmiş veriyi kullanıyoruz\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. Veriyi ölçekle (StandardScaler) ---\n# Modelin sayısal hassasiyet sorunları yaşamaması için hala iyi bir uygulamadır.\n# Standard Scaler, NaN içermeyen NumPy dizileri bekler.\nscaler = StandardScaler()\n\n# NumPy dizilerine dönüştürerek scikit-learn'ün uyarılarını önle\nX_train_scaled = scaler.fit_transform(X_train.values)\nX_test_scaled = scaler.transform(X_test.values)\n\n# --- 5. Boosting modelini başlat ve eğit ---\nmodel = GradientBoostingRegressor(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\nmodel.fit(X_train_scaled, y_train)\n\n# --- 6. Tahminler yap ---\ny_pred = model.predict(X_test_scaled)\ny_train_pred = model.predict(X_train_scaled)\n\n# --- 7. Skorları hesapla ve depola ---\nscores = {\n    \"train\": {\n        \"R2\": r2_score(y_train, y_train_pred),\n        \"mae\": mean_absolute_error(y_train, y_train_pred),\n        \"mse\": mean_squared_error(y_train, y_train_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_train, y_train_pred))\n    },\n    \"test\": {\n        \"R2\": r2_score(y_test, y_pred),\n        \"mae\": mean_absolute_error(y_test, y_pred),\n        \"mse\": mean_squared_error(y_test, y_pred),\n        \"rmse\": np.sqrt(mean_squared_error(y_test, y_pred))\n    }\n}\n\n# --- 8. DataFrame olarak döndür ---\nscores_df = pd.DataFrame(scores)\nprint(\"\\n--- Model Performans Skorları ---\")\nprint(scores_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:45:22.695234Z","iopub.execute_input":"2025-07-26T15:45:22.695570Z","iopub.status.idle":"2025-07-26T15:45:22.813591Z","shell.execute_reply.started":"2025-07-26T15:45:22.695545Z","shell.execute_reply":"2025-07-26T15:45:22.807529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"   \n    y_pred = model.predict(X_test)\n    y_train_pred = model.predict(X_train)\n    \n    scores = {\"train\": {\"R2\" : r2_score(y_train, y_train_pred),\n    \"mae\" : mean_absolute_error(y_train, y_train_pred),\n    \"mse\" : mean_squared_error(y_train, y_train_pred),                          \n    \"rmse\" : np.sqrt(mean_squared_error(y_train, y_train_pred))},\n    \n    \"test\": {\"R2\" : r2_score(y_test, y_pred),\n    \"mae\" : mean_absolute_error(y_test, y_pred),\n    \"mse\" : mean_squared_error(y_test, y_pred),\n    \"rmse\" : np.sqrt(mean_squared_error(y_test, y_pred))}}\n    \n    return pd.DataFrame(scores)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:27:01.115418Z","iopub.execute_input":"2025-07-26T15:27:01.115754Z","iopub.status.idle":"2025-07-26T15:27:01.791190Z","shell.execute_reply.started":"2025-07-26T15:27:01.115727Z","shell.execute_reply":"2025-07-26T15:27:01.786885Z"}}},{"cell_type":"code","source":"cat_features = X.select_dtypes(\"object\").columns\ncat_features ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:46:38.056428Z","iopub.execute_input":"2025-07-26T15:46:38.056750Z","iopub.status.idle":"2025-07-26T15:46:38.071425Z","shell.execute_reply.started":"2025-07-26T15:46:38.056726Z","shell.execute_reply":"2025-07-26T15:46:38.065930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import make_column_transformer\nfrom sklearn.preprocessing import OrdinalEncoder\n\n\nord_enc = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n\ncolumn_trans = make_column_transformer((ord_enc, cat_features), remainder='passthrough')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:46:45.727856Z","iopub.execute_input":"2025-07-26T15:46:45.728375Z","iopub.status.idle":"2025-07-26T15:46:45.740213Z","shell.execute_reply.started":"2025-07-26T15:46:45.728334Z","shell.execute_reply":"2025-07-26T15:46:45.734319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import AdaBoostRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:46:49.766731Z","iopub.execute_input":"2025-07-26T15:46:49.767060Z","iopub.status.idle":"2025-07-26T15:46:49.778948Z","shell.execute_reply.started":"2025-07-26T15:46:49.767032Z","shell.execute_reply":"2025-07-26T15:46:49.772552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import AdaBoostRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder # İhtiyacınıza göre diğer encoder'lar\n\n# --- Örnek Veri Oluşturma (Gerçek verinizle değiştirin) ---\n# Hem sayısal hem de kategorik sütunlar içeren bir DataFrame oluşturalım\ndata = {\n    'numerical_feature_1': np.random.rand(100) * 100,\n    'numerical_feature_2': np.random.rand(100) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high'], 100),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D'], 100),\n    'target': np.random.rand(100) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Eksik değer ekleyelim ki temizleme adımını test edebilelim\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = 'medium' # Ordinal feature'a örnek\nX.iloc[15, 3] = 'C' # Nominal feature'a örnek\ny.iloc[7] = np.nan\n\n# --- Veri Temizleme (Önceki yanıttan alınmıştır) ---\n# NaN ve sonsuz değerleri temizleme\n# Bu adımı, ColumnTransformer'dan önce yapmalısınız çünkü transformer'lar genellikle NaN'ları kendi başlarına işlemezler\n# Ancak, OrdinalEncoder varsayılan olarak NaN'ları geçirir. Eğer NaN'ları encode etmek isterseniz,\n# handle_missing='use_encoded_value' ve encoded_missing_value parametrelerini kullanmalısınız.\n# Basitlik adına, burada yine tüm NaN içeren satırları siliyoruz.\n\n# Adım 1: Tüm sonsuz değerleri NaN'a dönüştür (eğer hala varsa)\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# Adım 2: X ve y'yi birleştirerek birlikte NaN içeren satırları bul ve sil\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\n# Temizlenmiş veriyi X ve y'ye geri ayır\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\nprint(f\"Temizleme sonrası X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"Temizleme sonrası y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\nprint(f\"y_cleaned'daki NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n\n# --- Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- Eksik Kodu Buraya Tanımlayalım: column_trans ---\n\n# Sütunları tanımlayın\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature'] # Eğer nominal kategorik sütunlarınız varsa\n\n# OrdinalEncoder için kategorilerin sırasını belirtmek önemlidir\n# Eğer belirli bir sıra varsa, onu burada tanımlayın\n# Aksi takdirde, OrdinalEncoder varsayılan olarak kategorileri alfabetik sıraya göre atar.\n# Örnek: categories=[['low', 'medium', 'high']]\nordinal_categories_order = [['low', 'medium', 'high']] # Kendi kategorilerinize göre ayarlayın\n\n# Her bir sütun tipine uygulanacak transformer'ları tanımlayın\n# Sayısal sütunlar için bir StandardScaler\nnumerical_transformer = StandardScaler()\n\n# Sıralı kategorik sütunlar için bir OrdinalEncoder\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order) # handle_unknown='use_encoded_value', unknown_value=-1 gibi ayarlar eklenebilir\n\n# Nominal kategorik sütunlar için bir OneHotEncoder\nnominal_transformer = OneHotEncoder(handle_unknown='ignore') # handle_unknown='ignore' bilinmeyen kategorileri sıfırlarla kodlar\n\n# ColumnTransformer'ı oluşturun\n# 'remainder='passthrough'' belirtmezseniz, belirtilmeyen sütunlar otomatik olarak düşürülür.\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols) # Eğer nominal sütunlarınız varsa ekleyin\n    ],\n    remainder='passthrough' # İşlenmeyen diğer tüm sütunları olduğu gibi bırakır\n)\n\n# --- Pipeline Tanımı ---\noperations = [\n    (\"preprocessor\", column_trans), # column_trans'ı burada kullanıyoruz\n    (\"Ada_model\", AdaBoostRegressor(random_state=101))\n]\n\npipe_model = Pipeline(steps=operations)\n\n# --- Modeli Eğitme ---\npipe_model.fit(X_train, y_train)\n\n# --- Tahmin ve Değerlendirme ---\ny_pred_train = pipe_model.predict(X_train)\ny_pred_test = pipe_model.predict(X_test)\n\nprint(\"\\n--- Eğitim Seti Performansı ---\")\nprint(f\"R2 Skoru: {r2_score(y_train, y_pred_train):.4f}\")\nprint(f\"MAE: {mean_absolute_error(y_train, y_pred_train):.4f}\")\nprint(f\"MSE: {mean_squared_error(y_train, y_pred_train):.4f}\")\nprint(f\"RMSE: {np.sqrt(mean_squared_error(y_train, y_pred_train)):.4f}\")\n\nprint(\"\\n--- Test Seti Performansı ---\")\nprint(f\"R2 Skoru: {r2_score(y_test, y_pred_test):.4f}\")\nprint(f\"MAE: {mean_absolute_error(y_test, y_pred_test):.4f}\")\nprint(f\"MSE: {mean_squared_error(y_test, y_pred_test):.4f}\")\nprint(f\"RMSE: {np.sqrt(mean_squared_error(y_test, y_pred_test)):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:47:55.102352Z","iopub.execute_input":"2025-07-26T15:47:55.102681Z","iopub.status.idle":"2025-07-26T15:47:55.209911Z","shell.execute_reply.started":"2025-07-26T15:47:55.102654Z","shell.execute_reply":"2025-07-26T15:47:55.204332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\noperations = [(\"OrdinalEncoder\", column_trans),\n              (\"Ada_model\", AdaBoostRegressor(random_state=101))]\n\npipe_model = Pipeline(steps=operations)\n\npipe_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:46:57.103489Z","iopub.execute_input":"2025-07-26T15:46:57.103795Z","iopub.status.idle":"2025-07-26T15:46:57.440410Z","shell.execute_reply.started":"2025-07-26T15:46:57.103769Z","shell.execute_reply":"2025-07-26T15:46:57.435586Z"}}},{"cell_type":"markdown","source":"train_val(pipe_model, X_train, y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:48:31.713221Z","iopub.execute_input":"2025-07-26T15:48:31.713528Z","iopub.status.idle":"2025-07-26T15:48:31.746577Z","shell.execute_reply.started":"2025-07-26T15:48:31.713502Z","shell.execute_reply":"2025-07-26T15:48:31.740352Z"}}},{"cell_type":"markdown","source":"y_pred_ada = pipe_model.predict(X_test)\ny_pred_ada","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:16:41.668184Z","iopub.status.idle":"2025-07-26T15:16:41.668488Z","shell.execute_reply.started":"2025-07-26T15:16:41.668336Z","shell.execute_reply":"2025-07-26T15:16:41.668351Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import cross_validate, cross_val_score\n\noperations = [(\"OrdinalEncoder\", column_trans),\n              (\"Ada_model\", AdaBoostRegressor(random_state=101))]\n\nmodel = Pipeline(steps=operations)\n\nscores = cross_validate(model,\n                        X_train,\n                        y_train,\n                        scoring=[\n                            'r2', 'neg_mean_absolute_error',\n                            'neg_mean_squared_error',\n                            'neg_root_mean_squared_error'\n                        ],\n                        cv=10,\n                        return_train_score=True)\npd.DataFrame(scores)\npd.DataFrame(scores).mean()[2:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:49:11.584808Z","iopub.execute_input":"2025-07-26T15:49:11.585129Z","iopub.status.idle":"2025-07-26T15:49:12.624623Z","shell.execute_reply.started":"2025-07-26T15:49:11.585102Z","shell.execute_reply":"2025-07-26T15:49:12.619816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:49:26.225851Z","iopub.execute_input":"2025-07-26T15:49:26.226175Z","iopub.status.idle":"2025-07-26T15:49:26.238743Z","shell.execute_reply.started":"2025-07-26T15:49:26.226121Z","shell.execute_reply":"2025-07-26T15:49:26.231451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\noperations = [(\"OrdinalEncoder\", column_trans), (\"GB_model\", GradientBoostingRegressor(random_state=101))]\n\npipe_model = Pipeline(steps=operations)\n\npipe_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:49:30.815117Z","iopub.execute_input":"2025-07-26T15:49:30.815433Z","iopub.status.idle":"2025-07-26T15:49:30.937058Z","shell.execute_reply.started":"2025-07-26T15:49:30.815407Z","shell.execute_reply":"2025-07-26T15:49:30.932706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"train_val(pipe_model, X_train, y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:49:37.877939Z","iopub.execute_input":"2025-07-26T15:49:37.878248Z","iopub.status.idle":"2025-07-26T15:49:37.906670Z","shell.execute_reply.started":"2025-07-26T15:49:37.878221Z","shell.execute_reply":"2025-07-26T15:49:37.903182Z"}}},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans), (\"GB_model\", GradientBoostingRegressor(random_state=101))]\n\nmodel = Pipeline(steps=operations)\nscores = cross_validate(model, X_train, y_train, scoring=['r2', \n            'neg_mean_absolute_error','neg_mean_squared_error','neg_root_mean_squared_error'], cv =10,\n                       return_train_score=True)\n\npd.DataFrame(scores).mean()[2:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:49:51.391281Z","iopub.execute_input":"2025-07-26T15:49:51.391600Z","iopub.status.idle":"2025-07-26T15:49:52.387047Z","shell.execute_reply.started":"2025-07-26T15:49:51.391575Z","shell.execute_reply":"2025-07-26T15:49:52.380622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\"GB_model__n_estimators\":[35,50], \n              \"GB_model__subsample\":[0.7, 0.8, 1], \n              \"GB_model__max_features\" : [4,5,6],\n              \"GB_model__learning_rate\": [0.02, 0.03,0.05], \n              'GB_model__max_depth':[1,2],\n              'GB_model__min_samples_split':[1,2],\n              'GB_model__min_samples_leaf':[1,2]}\n\n# classificationdan en önemli farkı loss='squared_error'dür. Classifciationda bu logloss'tu.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:50:11.894771Z","iopub.execute_input":"2025-07-26T15:50:11.895134Z","iopub.status.idle":"2025-07-26T15:50:11.908719Z","shell.execute_reply.started":"2025-07-26T15:50:11.895094Z","shell.execute_reply":"2025-07-26T15:50:11.903625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer # Eğer imputation'a ihtiyacınız olursa\n\n# --- Örnek Veri Oluşturma (Gerçek verinizle değiştirin) ---\n# Hem sayısal hem de kategorik sütunlar içeren bir DataFrame oluşturalım\ndata = {\n    'numerical_feature_1': np.random.rand(100) * 100,\n    'numerical_feature_2': np.random.rand(100) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high'], 100),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D'], 100),\n    'target': np.random.rand(100) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Eksik değer ekleyelim ki temizleme adımını test edebilelim\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = 'medium' # Ordinal feature'a örnek\nX.iloc[15, 3] = 'C' # Nominal feature'a örnek\ny.iloc[7] = np.nan\n\n# --- Veri Temizleme (Önceki yanıttan alınmıştır - NaN ve sonsuz değerleri temizleme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\n# --- Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- ColumnTransformer Tanımı (Önceki yanıttan alınmıştır) ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\n\nordinal_categories_order = [['low', 'medium', 'high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- Pipeline Tanımı ---\n# Not: Pipeline adını 'OrdinalEncoder' yerine 'preprocessor' olarak değiştirdim,\n# bu daha açıklayıcı ve daha önceki örnekle tutarlı.\n# Eğer ordinal encoder'a özel parametreleri optimize etmek isterseniz,\n# 'preprocessor__ord__...' şeklinde parametre ızgarasına eklemeniz gerekir.\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\n\nmodel = Pipeline(steps=operations)\n\n# --- EKSİK KOD: param_grid Tanımı ---\n# GradientBoostingRegressor için denemek istediğiniz hiperparametreleri buraya ekleyin.\n# Pipeline'daki model adımının adı \"GB_model\" olduğu için,\n# parametre adları \"GB_model__\" ile başlamalıdır.\n\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 200],  # Denenecek ağaç sayısı\n    'GB_model__learning_rate': [0.01, 0.1, 0.2], # Her ağacın katkısı\n    'GB_model__max_depth': [3, 4, 5],          # Her ağacın maksimum derinliği\n    # 'GB_model__subsample': [0.8, 1.0],         # Her ağacı eğitmek için kullanılan örneklerin oranı\n    # 'GB_model__min_samples_split': [2, 5],     # Bir düğümü bölmek için gereken minimum örnek sayısı\n    # Eğer ColumnTransformer içindeki bir transformer'ın parametresini optimize etmek isterseniz:\n    # 'preprocessor__num__with_mean': [True, False], # StandardScaler için\n    # 'preprocessor__ord__handle_unknown': ['use_encoded_value'], # OrdinalEncoder için\n    # 'preprocessor__ord__unknown_value': [-1] # OrdinalEncoder için, handle_unknown 'use_encoded_value' ise\n}\n\n# --- GridSearchCV Tanımı ve Eğitimi ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error', # Daha yüksek skor daha iyidir, RMSE'yi minimize etmek için negatifini kullanırız\n                          cv=5,                                 # 5 katlı çapraz doğrulama\n                          n_jobs=-1,                            # Tüm işlemcileri kullan\n                          return_train_score=True)\n\n# Modeli eğitin\ngrid_model.fit(X_train, y_train)\n\n# --- En İyi Parametreler ve Skor ---\nprint(\"\\n--- GridSearchCV Sonuçları ---\")\nprint(f\"En İyi Parametreler: {grid_model.best_params_}\")\nprint(f\"En İyi RMSE Skoru (Negatif): {grid_model.best_score_:.4f}\") # 'neg_root_mean_squared_error' olduğu için negatif olacak\nprint(f\"En İyi RMSE Skoru (Pozitif): {-grid_model.best_score_:.4f}\")\n\n# En iyi modeli al\nbest_model = grid_model.best_estimator_\n\n# Test seti üzerinde tahminler yap\ny_pred_test = best_model.predict(X_test)\n\n# Test seti performansını değerlendir\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\nprint(\"\\n--- Test Seti Performansı (En İyi Model) ---\")\nprint(f\"R2 Skoru: {r2_score(y_test, y_pred_test):.4f}\")\nprint(f\"MAE: {mean_absolute_error(y_test, y_pred_test):.4f}\")\nprint(f\"MSE: {mean_squared_error(y_test, y_pred_test):.4f}\")\nprint(f\"RMSE: {np.sqrt(mean_squared_error(y_test, y_pred_test)):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:51:01.514952Z","iopub.execute_input":"2025-07-26T15:51:01.515287Z","iopub.status.idle":"2025-07-26T15:51:05.667897Z","shell.execute_reply.started":"2025-07-26T15:51:01.515262Z","shell.execute_reply":"2025-07-26T15:51:05.663737Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"operations = [(\"OrdinalEncoder\", column_trans), (\"GB_model\", GradientBoostingRegressor(random_state=101))]\n\nmodel = Pipeline(steps=operations)\n\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs = -1,\n                          return_train_score=True).fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:50:16.750023Z","iopub.execute_input":"2025-07-26T15:50:16.750367Z","iopub.status.idle":"2025-07-26T15:50:16.786781Z","shell.execute_reply.started":"2025-07-26T15:50:16.750336Z","shell.execute_reply":"2025-07-26T15:50:16.781174Z"}}},{"cell_type":"code","source":"grid_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:51:57.234366Z","iopub.execute_input":"2025-07-26T15:51:57.234718Z","iopub.status.idle":"2025-07-26T15:51:57.251202Z","shell.execute_reply.started":"2025-07-26T15:51:57.234691Z","shell.execute_reply":"2025-07-26T15:51:57.244675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_estimator_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:52:02.914612Z","iopub.execute_input":"2025-07-26T15:52:02.914967Z","iopub.status.idle":"2025-07-26T15:52:02.956848Z","shell.execute_reply.started":"2025-07-26T15:52:02.914938Z","shell.execute_reply":"2025-07-26T15:52:02.950347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:52:11.858292Z","iopub.execute_input":"2025-07-26T15:52:11.858637Z","iopub.status.idle":"2025-07-26T15:52:11.873797Z","shell.execute_reply.started":"2025-07-26T15:52:11.858607Z","shell.execute_reply":"2025-07-26T15:52:11.867567Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"train_val(grid_model, X_train, y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:52:16.978056Z","iopub.execute_input":"2025-07-26T15:52:16.978466Z","iopub.status.idle":"2025-07-26T15:52:17.016250Z","shell.execute_reply.started":"2025-07-26T15:52:16.978434Z","shell.execute_reply":"2025-07-26T15:52:17.010274Z"}}},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans),\n              (\"GB_model\",\n               GradientBoostingRegressor(learning_rate=0.05, max_depth=2, max_features=6,\n                          n_estimators=50, random_state=101, subsample=0.7))]\n\nmodel = Pipeline(steps=operations)\n\nscores = cross_validate(model,\n                        X_train,\n                        y_train,\n                        scoring=[\n                            'r2', 'neg_mean_absolute_error',\n                            'neg_mean_squared_error',\n                            'neg_root_mean_squared_error'\n                        ],\n                        cv=10,\n                        return_train_score=True)\npd.DataFrame(scores).mean()[2:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:52:38.322675Z","iopub.execute_input":"2025-07-26T15:52:38.323032Z","iopub.status.idle":"2025-07-26T15:52:38.933706Z","shell.execute_reply.started":"2025-07-26T15:52:38.323005Z","shell.execute_reply":"2025-07-26T15:52:38.926615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install --upgrade scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:53:19.141857Z","iopub.execute_input":"2025-07-26T15:53:19.142261Z","iopub.status.idle":"2025-07-26T15:53:29.425791Z","shell.execute_reply.started":"2025-07-26T15:53:19.142231Z","shell.execute_reply":"2025-07-26T15:53:29.419614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer # Eğer imputation'a ihtiyacınız olursa\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error # mean_squared_error'ı import etmeyi unutmayın\n\n# --- Örnek Veri Oluşturma (Gerçek verinizle değiştirin) ---\n# Hem sayısal hem de kategorik sütunlar içeren bir DataFrame oluşturalım\ndata = {\n    'numerical_feature_1': np.random.rand(100) * 100,\n    'numerical_feature_2': np.random.rand(100) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high'], 100),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D'], 100),\n    'target': np.random.rand(100) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Eksik değer ekleyelim ki temizleme adımını test edebilelim\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = 'medium' # Ordinal feature'a örnek\nX.iloc[15, 3] = 'C' # Nominal feature'a örnek\ny.iloc[7] = np.nan\n\n# --- Veri Temizleme (NaN ve sonsuz değerleri temizleme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\n# --- Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- ColumnTransformer Tanımı ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\n\nordinal_categories_order = [['low', 'medium', 'high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- Pipeline Tanımı ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\n\nmodel = Pipeline(steps=operations)\n\n# --- param_grid Tanımı ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 200],\n    'GB_model__learning_rate': [0.01, 0.1, 0.2],\n    'GB_model__max_depth': [3, 4, 5],\n}\n\n# --- GridSearchCV Tanımı ve Eğitimi ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\ngrid_model.fit(X_train, y_train) # Eğitimi burada yapıyoruz.\n\n# --- En İyi Modeli Kullanarak Tahminler ---\nbest_model = grid_model.best_estimator_\ny_pred_test = best_model.predict(X_test)\ny_pred_train = best_model.predict(X_train) # Eğitim seti için de tahmin yapalım\n\n# --- Metriklerin Hesaplanması ---\nprint(\"\\n--- Model Performans Skorları ---\")\n\n# Eğitim seti metrikleri\ngrad_r2_train = r2_score(y_train, y_pred_train)\ngrad_mae_train = mean_absolute_error(y_train, y_pred_train)\ngrad_mse_train = mean_squared_error(y_train, y_pred_train)\ngrad_rmse_train = np.sqrt(mean_squared_error(y_train, y_pred_train)) # FIX: np.sqrt kullanıldı\n\n# Test seti metrikleri\ngrad_r2_test = r2_score(y_test, y_pred_test)\ngrad_mae_test = mean_absolute_error(y_test, y_pred_test)\ngrad_mse_test = mean_squared_error(y_test, y_pred_test)\ngrad_rmse_test = np.sqrt(mean_squared_error(y_test, y_pred_test)) # FIX: np.sqrt kullanıldı\n\nscores = {\n    \"train\": {\n        \"R2\": grad_r2_train,\n        \"mae\": grad_mae_train,\n        \"mse\": grad_mse_train,\n        \"rmse\": grad_rmse_train\n    },\n    \"test\": {\n        \"R2\": grad_r2_test,\n        \"mae\": grad_mae_test,\n        \"mse\": grad_mse_test,\n        \"rmse\": grad_rmse_test\n    }\n}\nscores_df = pd.DataFrame(scores)\nprint(scores_df)\n\n# --- train_val fonksiyonu eğer ayrı bir fonksiyon ise ---\n# train_val fonksiyonunuzu da kontrol edin ve orada da `squared=False` kullanılıyorsa,\n# onu da `np.sqrt()` ile değiştirmeniz gerekecektir.\n# Örnek train_val fonksiyonu (varsayım):\ndef train_val(model_fitted_by_grid, X_train, y_train, X_test, y_test):\n    y_train_pred = model_fitted_by_grid.predict(X_train)\n    y_test_pred = model_fitted_by_grid.predict(X_test)\n\n    train_r2 = r2_score(y_train, y_train_pred)\n    train_mae = mean_absolute_error(y_train, y_train_pred)\n    train_mse = mean_squared_error(y_train, y_train_pred)\n    train_rmse = np.sqrt(mean_squared_error(y_train, y_train_pred)) # FIX: np.sqrt\n\n    test_r2 = r2_score(y_test, y_test_pred)\n    test_mae = mean_absolute_error(y_test, y_test_pred)\n    test_mse = mean_squared_error(y_test, y_test_pred)\n    test_rmse = np.sqrt(mean_squared_error(y_test, y_test_pred)) # FIX: np.sqrt\n\n    print(\"\\n--- train_val Fonksiyonu Sonuçları ---\")\n    results = pd.DataFrame({\n        'Metric': ['R2', 'MAE', 'MSE', 'RMSE'],\n        'Train': [train_r2, train_mae, train_mse, train_rmse],\n        'Test': [test_r2, test_mae, test_mse, test_rmse]\n    })\n    print(results)\n\n# train_val fonksiyonunu çağır (eğer tanımlıysa ve kullanmak istiyorsanız)\n# train_val(grid_model, X_train, y_train, X_test, y_test) # grid_model en iyi modeli içerir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:54:00.026018Z","iopub.execute_input":"2025-07-26T15:54:00.026429Z","iopub.status.idle":"2025-07-26T15:54:00.685173Z","shell.execute_reply.started":"2025-07-26T15:54:00.026396Z","shell.execute_reply":"2025-07-26T15:54:00.680537Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"y_pred = grid_model.predict(X_test)\ngrad_R2 = r2_score(y_test, y_pred)\ngrad_mae = mean_absolute_error(y_test, y_pred)\ngrad_mse = mean_squared_error(y_test, y_pred)\ngrad_rmse = mean_squared_error(y_test, y_pred, squared=False)\ntrain_val(grid_model, X_train, y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:53:36.050726Z","iopub.execute_input":"2025-07-26T15:53:36.051031Z","iopub.status.idle":"2025-07-26T15:53:36.122544Z","shell.execute_reply.started":"2025-07-26T15:53:36.051005Z","shell.execute_reply":"2025-07-26T15:53:36.116947Z"}}},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans),\n              (\"GB_model\",\n               GradientBoostingRegressor(learning_rate=0.05,\n                                         max_depth=2,\n                                         max_features=6,\n                                         n_estimators=50,\n                                         random_state=101,\n                                         subsample=0.7))]\n\npipe_model = Pipeline(steps=operations)\n\npipe_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:54:29.188472Z","iopub.execute_input":"2025-07-26T15:54:29.188857Z","iopub.status.idle":"2025-07-26T15:54:29.282414Z","shell.execute_reply.started":"2025-07-26T15:54:29.188825Z","shell.execute_reply":"2025-07-26T15:54:29.276985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe_model[\"GB_model\"].feature_importances_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:54:39.293108Z","iopub.execute_input":"2025-07-26T15:54:39.293485Z","iopub.status.idle":"2025-07-26T15:54:39.311241Z","shell.execute_reply.started":"2025-07-26T15:54:39.293454Z","shell.execute_reply":"2025-07-26T15:54:39.305486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer # If you need imputation\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Create Sample Data (Replace with your actual data) ---\ndata = {\n    'numerical_feature_1': np.random.rand(100) * 100,\n    'numerical_feature_2': np.random.rand(100) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high'], 100),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D'], 100),\n    'target': np.random.rand(100) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Add some missing/infinite values for demonstration\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = 'medium'\nX.iloc[15, 3] = 'C'\ny.iloc[7] = np.nan\n\n# --- 2. Data Cleaning (Handling NaNs and Infs) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\n# --- 3. Split Data into Training and Testing Sets ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Definition ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\n\nordinal_categories_order = [['low', 'medium', 'high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- 5. Pipeline Definition ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# --- 6. param_grid Definition ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 200],\n    'GB_model__learning_rate': [0.01, 0.1, 0.2],\n    'GB_model__max_depth': [3, 4, 5],\n}\n\n# --- 7. GridSearchCV Definition and Training ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\ngrid_model.fit(X_train, y_train)\n\n# --- 8. Get Feature Importances and Feature Names ---\n# The best_estimator_ from GridSearchCV is the fitted pipeline\nfitted_pipeline = grid_model.best_estimator_\n\n# FIX: Get the feature names AFTER the ColumnTransformer has been fitted\n# You can access the 'preprocessor' step of the fitted pipeline\nnew_features = fitted_pipeline.named_steps['preprocessor'].get","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:55:25.069180Z","iopub.execute_input":"2025-07-26T15:55:25.069554Z","iopub.status.idle":"2025-07-26T15:55:25.732092Z","shell.execute_reply.started":"2025-07-26T15:55:25.069525Z","shell.execute_reply":"2025-07-26T15:55:25.725795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Create Sample Data (Replace with your actual data) ---\ndata = {\n    'numerical_feature_1': np.random.rand(100) * 100,\n    'numerical_feature_2': np.random.rand(100) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high'], 100),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D'], 100),\n    'target': np.random.rand(100) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Add some missing/infinite values for demonstration\nX.iloc[5, 0] = np.nan\nX.iloc[10, 2] = 'medium'\nX.iloc[15, 3] = 'C'\ny.iloc[7] = np.nan\n\n# --- 2. Data Cleaning (Handling NaNs and Infs) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\n# --- 3. Split Data into Training and Testing Sets ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Definition ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\n\nordinal_categories_order = [['low', 'medium', 'high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- 5. Pipeline Definition ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# --- 6. param_grid Definition ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 200],\n    'GB_model__learning_rate': [0.01, 0.1, 0.2],\n    'GB_model__max_depth': [3, 4, 5],\n}\n\n# --- 7. GridSearchCV Definition and Training ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\ngrid_model.fit(X_train, y_train)\n\n# --- 8. Get Feature Importances and Feature Names ---\nfitted_pipeline = grid_model.best_estimator_\n\n# FIX: Correct the method name from .get to .get_feature_names_out()\nnew_features = fitted_pipeline.named_steps['preprocessor'].get_feature_names_out()\n\n# Now, use these feature names for your DataFrame index\nimp_feats = pd.DataFrame(data=fitted_pipeline[\"GB_model\"].feature_importances_,\n                         columns=['Grad_Importance'],\n                         index=new_features)\n\ngrad_imp_feats = imp_feats.sort_values('Grad_Importance', ascending=False)\nprint(\"\\n--- Feature Importances ---\")\nprint(grad_imp_feats)\n\n# --- 9. Predictions and Evaluation ---\ny_pred_test = fitted_pipeline.predict(X_test)\ny_pred_train = fitted_pipeline.predict(X_train)\n\nprint(\"\\n--- Model Performance Scores ---\")\n\ngrad_r2_train = r2_score(y_train, y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:56:03.294093Z","iopub.execute_input":"2025-07-26T15:56:03.294445Z","iopub.status.idle":"2025-07-26T15:56:03.315406Z","shell.execute_reply.started":"2025-07-26T15:56:03.294418Z","shell.execute_reply":"2025-07-26T15:56:03.311417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 9. Predictions and Evaluation ---\ny_pred_test = fitted_pipeline.predict(X_test)\ny_pred_train = fitted_pipeline.predict(X_train) # This line predicts for the training set\n\nprint(\"\\n--- Model Performance Scores ---\")\n\ngrad_r2_train = r2_score(y_train, y_pred_train) # Corrected: used y_pred_train\ngrad_mae_train = mean_absolute_error(y_train, y_pred_train)\ngrad_mse_train = mean_squared_error(y_train, y_pred_train)\ngrad_rmse_train = np.sqrt(mean_squared_error(y_train, y_pred_train)) # Ensure y_pred_train here too\n\ngrad_r2_test = r2_score(y_test, y_pred_test)\ngrad_mae_test = mean_absolute_error(y_test, y_pred_test)\ngrad_mse_test = mean_squared_error(y_test, y_pred_test)\ngrad_rmse_test = np.sqrt(mean_squared_error(y_test, y_pred_test))\n\nscores = {\n    \"train\": {\n        \"R2\": grad_r2_train,\n        \"mae\": grad_mae_train,\n        \"mse\": grad_mse_train,\n        \"rmse\": grad_rmse_train\n    },\n    \"test\": {\n        \"R2\": grad_r2_test,\n        \"mae\": grad_mae_test,\n        \"mse\": grad_mse_test,\n        \"rmse\": grad_rmse_test\n    }\n}\nscores_df = pd.DataFrame(scores)\nprint(scores_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:56:39.125824Z","iopub.execute_input":"2025-07-26T15:56:39.126158Z","iopub.status.idle":"2025-07-26T15:56:39.157749Z","shell.execute_reply.started":"2025-07-26T15:56:39.126117Z","shell.execute_reply":"2025-07-26T15:56:39.152552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"imp_feats = pd.DataFrame(data=pipe_model[\"GB_model\"].feature_importances_,columns=['Grad_Importance'], index=new_features)\ngrad_imp_feats = imp_feats.sort_values('Grad_Importance', ascending=False)\ngrad_imp_feats","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:54:43.207563Z","iopub.execute_input":"2025-07-26T15:54:43.207897Z","iopub.status.idle":"2025-07-26T15:54:43.245520Z","shell.execute_reply.started":"2025-07-26T15:54:43.207869Z","shell.execute_reply":"2025-07-26T15:54:43.240074Z"}}},{"cell_type":"code","source":"ax = sns.barplot(data=grad_imp_feats, x=grad_imp_feats.index, y='Grad_Importance')\nax.bar_label(ax.containers[0],fmt=\"%.3f\")\nplt.xticks(rotation=90);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:16:41.697781Z","iopub.status.idle":"2025-07-26T15:16:41.698026Z","shell.execute_reply.started":"2025-07-26T15:16:41.697885Z","shell.execute_reply":"2025-07-26T15:16:41.697893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:57:03.248280Z","iopub.execute_input":"2025-07-26T15:57:03.248589Z","iopub.status.idle":"2025-07-26T15:57:17.780769Z","shell.execute_reply.started":"2025-07-26T15:57:03.248565Z","shell.execute_reply":"2025-07-26T15:57:17.776400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:57:24.464404Z","iopub.execute_input":"2025-07-26T15:57:24.464759Z","iopub.status.idle":"2025-07-26T15:57:26.359093Z","shell.execute_reply.started":"2025-07-26T15:57:24.464723Z","shell.execute_reply":"2025-07-26T15:57:26.354318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans), (\"XGB_model\", XGBRegressor(random_state=101))]\n\npipe_model = Pipeline(steps=operations)\n\npipe_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:57:52.623950Z","iopub.execute_input":"2025-07-26T15:57:52.624281Z","iopub.status.idle":"2025-07-26T15:57:52.902492Z","shell.execute_reply.started":"2025-07-26T15:57:52.624254Z","shell.execute_reply":"2025-07-26T15:57:52.897275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_val(pipe_model, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:57:57.737410Z","iopub.execute_input":"2025-07-26T15:57:57.737732Z","iopub.status.idle":"2025-07-26T15:57:57.781251Z","shell.execute_reply.started":"2025-07-26T15:57:57.737704Z","shell.execute_reply":"2025-07-26T15:57:57.774585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans), (\"XGB_model\", XGBRegressor(random_state=101))]\n\nmodel = Pipeline(steps=operations)\n\nscores = cross_validate(model, X_train, y_train, scoring=['r2', \n            'neg_mean_absolute_error','neg_mean_squared_error','neg_root_mean_squared_error'], cv =10,\n                       return_train_score=True)\npd.DataFrame(scores).iloc[:, 2:].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:58:02.122779Z","iopub.execute_input":"2025-07-26T15:58:02.123132Z","iopub.status.idle":"2025-07-26T15:58:05.958909Z","shell.execute_reply.started":"2025-07-26T15:58:02.123105Z","shell.execute_reply":"2025-07-26T15:58:05.954879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- param_grid Tanımı ---\n# XGBoost modeliniz için denemek istediğiniz hiperparametreleri buraya ekleyin.\n# Pipeline'daki model adımının adı \"XGB_model\" (veya \"GB_model\") olduğu için,\n# parametre adları \"XGB_model__\" (veya \"GB_model__\") ile başlamalıdır.\n\nparam_grid = {\n    'XGB_model__n_estimators': [50, 100, 200],\n    'XGB_model__learning_rate': [0.01, 0.1, 0.2],\n    'XGB_model__max_depth': [3, 4, 5],\n    'XGB_model__colsample_bytree': [0.5, 0.8, 1]\n} # <-- Make sure you have this closing curly brace here!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:58:54.756061Z","iopub.execute_input":"2025-07-26T15:58:54.756459Z","iopub.status.idle":"2025-07-26T15:58:54.767272Z","shell.execute_reply.started":"2025-07-26T15:58:54.756430Z","shell.execute_reply":"2025-07-26T15:58:54.763662Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"param_grid = {\n    \"XGB_model__n_estimators\": [40, 50, 100],\n    \"XGB_model__max_depth\": [2, 3],\n    \"XGB_model__learning_rate\": [0.01, 0.05, 0.06],\n    \"XGB_model__subsample\": [0.5, 0.8, 1],\n    \"XGB_model__colsample_bytree\": [0.5, 0.8, 1]","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:59:01.261767Z","iopub.execute_input":"2025-07-26T15:59:01.262169Z","iopub.status.idle":"2025-07-26T15:59:01.275027Z","shell.execute_reply.started":"2025-07-26T15:59:01.262121Z","shell.execute_reply":"2025-07-26T15:59:01.271071Z"}}},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans), (\"XGB_model\", XGBRegressor(random_state=101))]\n\nmodel = Pipeline(steps=operations)\n\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=10,\n                          n_jobs = -1,\n                          return_train_score=True).fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:59:41.163405Z","iopub.execute_input":"2025-07-26T15:59:41.163716Z","iopub.status.idle":"2025-07-26T15:59:43.495775Z","shell.execute_reply.started":"2025-07-26T15:59:41.163688Z","shell.execute_reply":"2025-07-26T15:59:43.489188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:59:48.388498Z","iopub.execute_input":"2025-07-26T15:59:48.388832Z","iopub.status.idle":"2025-07-26T15:59:48.402379Z","shell.execute_reply.started":"2025-07-26T15:59:48.388803Z","shell.execute_reply":"2025-07-26T15:59:48.397460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T15:59:52.967627Z","iopub.execute_input":"2025-07-26T15:59:52.967989Z","iopub.status.idle":"2025-07-26T15:59:52.983441Z","shell.execute_reply.started":"2025-07-26T15:59:52.967957Z","shell.execute_reply":"2025-07-26T15:59:52.976508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n# Assuming y_test and y_pred are already defined from your model's predictions\n\n# --- Calculate Metrics for XGBoost (or whichever model you're using here) ---\nXGB_r2 = r2_score(y_test, y_pred)\nXGB_mae = mean_absolute_error(y_test, y_pred)\nXGB_mse = mean_squared_error(y_test, y_pred)\nXGB_rmse = np.sqrt(mean_squared_error(y_test, y_pred)) # Corrected line\n\n# And if you have a `train_val` function that also uses `squared=False`,\n# make sure to update it as well:\ndef train_val(model_fitted_by_grid, X_train, y_train, X_test, y_test):\n    y_train_pred = model_fitted_by_grid.predict(X_train)\n    y_test_pred = model_fitted_by_grid.predict(X_test)\n\n    train_r2 = r2_score(y_train, y_train_pred)\n    train_mae = mean_absolute_error(y_train, y_train_pred)\n    train_mse = mean_squared_error(y_train, y_train_pred)\n    train_rmse = np.sqrt(mean_squared_error(y_train, y_train_pred)) # Corrected here\n\n    test_r2 = r2_score(y_test, y_test_pred)\n    test_mae = mean_absolute_error(y_test, y_test_pred)\n    test_mse = mean_squared_error(y_test, y_test_pred)\n    test_rmse = np.sqrt(mean_squared_error(y_test, y_test_pred)) # Corrected here\n\n    print(\"\\n--- train_val Fonksiyonu Sonuçları ---\")\n    results = pd.DataFrame({\n        'Metric': ['R2', 'MAE', 'MSE', 'RMSE'],\n        'Train': [train_r2, train_mae, train_mse, train_rmse],\n        'Test': [test_r2, test_mae, test_mse, test_rmse]\n    })\n    print(results)\n\n# Call train_val (if applicable)\n# train_val(grid_model, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:04:20.748883Z","iopub.execute_input":"2025-07-26T16:04:20.749301Z","iopub.status.idle":"2025-07-26T16:04:20.768771Z","shell.execute_reply.started":"2025-07-26T16:04:20.749259Z","shell.execute_reply":"2025-07-26T16:04:20.762366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n# Diğer import'lar (Pipeline, GridSearchCV, vs.) burada olmalı\n\n# ... (Önceki kodunuz: veri yükleme, temizleme, split, ColumnTransformer, Pipeline, GridSearchCV tanımı ve eğitimi) ...\n\n# Farz edelim ki grid_model.fit(X_train, y_train) zaten çalıştırıldı\n# ve en iyi modeli grid_model.best_estimator_ olarak alabiliriz.\n\n# --- Metrik Hesaplamaları (Ana Modelin Test Seti Performansı) ---\n# grid_model.predict, en iyi modeli kullanarak tahmin yapar.\ny_pred = grid_model.predict(X_test) # Test seti tahminleri\n\nXGB_R2 = r2_score(y_test, y_pred)\nXGB_mae = mean_absolute_error(y_test, y_pred)\nXGB_mse = mean_squared_error(y_test, y_pred)\n# Hata düzeltmesi: squared=False yerine np.sqrt() kullanıyoruz\nXGB_rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\nprint(f\"XGBoost Test R2: {XGB_R2:.4f}\")\nprint(f\"XGBoost Test MAE: {XGB_mae:.4f}\")\nprint(f\"XGBoost Test MSE: {XGB_mse:.4f}\")\nprint(f\"XGBoost Test RMSE: {XGB_rmse:.4f}\")\n\n# --- EKSİK KOD: train_val fonksiyonunun tanımı ---\ndef train_val(model_or_grid_model, X_train, y_train, X_test, y_test):\n    \"\"\"\n    Eğitim ve test setleri üzerindeki model performansını değerlendirir.\n\n    Parametreler:\n    model_or_grid_model: Eğitilmiş bir scikit-learn modeli veya GridSearchCV objesi.\n                         GridSearchCV ise en iyi modelini kullanır.\n    X_train, y_train: Eğitim verisi\n    X_test, y_test: Test verisi\n    \"\"\"\n    # Eğer model_or_grid_model bir GridSearchCV objesi ise, en iyi tahminciyi kullan\n    if hasattr(model_or_grid_model, 'best_estimator_'):\n        model_to_predict = model_or_grid_model.best_estimator_\n    else: # Aksi takdirde, doğrudan verilen modeli kullan\n        model_to_predict = model_or_grid_model\n\n    y_train_pred = model_to_predict.predict(X_train)\n    y_test_pred = model_to_predict.predict(X_test)\n\n    # Eğitim Seti Metrikleri\n    train_r2 = r2_score(y_train, y_train_pred)\n    train_mae = mean_absolute_error(y_train, y_train_pred)\n    train_mse = mean_squared_error(y_train, y_train_pred)\n    train_rmse = np.sqrt(mean_squared_error(y_train, y_train_pred)) # FIX: np.sqrt\n\n    # Test Seti Metrikleri\n    test_r2 = r2_score(y_test, y_test_pred)\n    test_mae = mean_absolute_error(y_test, y_test_pred)\n    test_mse = mean_squared_error(y_test, y_test_pred)\n    test_rmse = np.sqrt(mean_squared_error(y_test, y_test_pred)) # FIX: np.sqrt\n\n    print(\"\\n--- Model Performans Özeti ---\")\n    results = pd.DataFrame({\n        'Metrik': ['R2', 'MAE', 'MSE', 'RMSE'],\n        'Eğitim Seti': [train_r2, train_mae, train_mse, train_rmse],\n        'Test Seti': [test_r2, test_mae, test_mse, test_rmse]\n    })\n    print(results)\n\n# --- train_val fonksiyonunu çağır ---\ntrain_val(grid_model, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:05:23.487752Z","iopub.execute_input":"2025-07-26T16:05:23.488127Z","iopub.status.idle":"2025-07-26T16:05:23.681792Z","shell.execute_reply.started":"2025-07-26T16:05:23.488096Z","shell.execute_reply":"2025-07-26T16:05:23.677039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"y_pred = grid_model.predict(X_test)\nXGB_R2 = r2_score(y_test, y_pred)\nXGB_mae = mean_absolute_error(y_test, y_pred)\nXGB_mse = mean_squared_error(y_test, y_pred)\nXGB_rmse = mean_squared_error(y_test, y_pred, squared=False)\ntrain_val(grid_model, X_train, y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T15:59:56.981316Z","iopub.execute_input":"2025-07-26T15:59:56.981642Z","iopub.status.idle":"2025-07-26T15:59:57.100223Z","shell.execute_reply.started":"2025-07-26T15:59:56.981614Z","shell.execute_reply":"2025-07-26T15:59:57.095340Z"}}},{"cell_type":"code","source":"operations = [(\"OrdinalEncoder\", column_trans),\n              (\"XGB_model\",\n               XGBRegressor(n_estimators=100,\n                            learning_rate=0.06,\n                            max_depth=3,\n                            random_state=101,\n                            subsample=0.5))]\n\npipe_model = Pipeline(steps=operations)\n\npipe_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:05:41.952247Z","iopub.execute_input":"2025-07-26T16:05:41.952584Z","iopub.status.idle":"2025-07-26T16:05:42.575004Z","shell.execute_reply.started":"2025-07-26T16:05:41.952556Z","shell.execute_reply":"2025-07-26T16:05:42.569452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe_model[\"XGB_model\"].feature_importances_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:05:47.473021Z","iopub.execute_input":"2025-07-26T16:05:47.473336Z","iopub.status.idle":"2025-07-26T16:05:47.487694Z","shell.execute_reply.started":"2025-07-26T16:05:47.473311Z","shell.execute_reply":"2025-07-26T16:05:47.483174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe_model[\"OrdinalEncoder\"].get_feature_names_out()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:05:51.679438Z","iopub.execute_input":"2025-07-26T16:05:51.679732Z","iopub.status.idle":"2025-07-26T16:05:51.692935Z","shell.execute_reply.started":"2025-07-26T16:05:51.679708Z","shell.execute_reply":"2025-07-26T16:05:51.689602Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"new_features","metadata":{"execution":{"iopub.status.busy":"2025-07-26T16:05:56.769586Z","iopub.execute_input":"2025-07-26T16:05:56.769874Z","iopub.status.idle":"2025-07-26T16:05:56.801401Z","shell.execute_reply.started":"2025-07-26T16:05:56.769850Z","shell.execute_reply":"2025-07-26T16:05:56.796395Z"}}},{"cell_type":"markdown","source":"imp_feats = pd.DataFrame(data=pipe_model[\"XGB_model\"].feature_importances_, columns=['XGB_Importance'], index=new_features)\nxgb_imp_feats = imp_feats.sort_values('XGB_Importance', ascending=False)\nxgb_imp_feats","metadata":{"execution":{"iopub.status.busy":"2025-07-26T16:06:15.778678Z","iopub.execute_input":"2025-07-26T16:06:15.778982Z","iopub.status.idle":"2025-07-26T16:06:15.811201Z","shell.execute_reply.started":"2025-07-26T16:06:15.778956Z","shell.execute_reply":"2025-07-26T16:06:15.806255Z"}}},{"cell_type":"markdown","source":"ax = sns.barplot(data=xgb_imp_feats, x=xgb_imp_feats.index, y='XGB_Importance')\nax.bar_label(ax.containers[0],fmt=\"%.3f\")\nplt.xticks(rotation=90);","metadata":{"execution":{"iopub.status.busy":"2025-07-26T16:08:30.530683Z","iopub.execute_input":"2025-07-26T16:08:30.531052Z","iopub.status.idle":"2025-07-26T16:08:30.565362Z","shell.execute_reply.started":"2025-07-26T16:08:30.531024Z","shell.execute_reply":"2025-07-26T16:08:30.560154Z"}}},{"cell_type":"markdown","source":"pd.concat([ grad_imp_feats, xgb_imp_feats], axis=1)","metadata":{"execution":{"iopub.status.busy":"2025-07-26T16:06:49.755712Z","iopub.execute_input":"2025-07-26T16:06:49.756025Z","iopub.status.idle":"2025-07-26T16:06:49.785384Z","shell.execute_reply.started":"2025-07-26T16:06:49.756001Z","shell.execute_reply":"2025-07-26T16:06:49.780789Z"}}},{"cell_type":"code","source":"# --- 9. En İyi Model ile Tahmin ve Değerlendirme ---\nprint(\"\\n--- En İyi GridSearchCV Modeli ---\")\nprint(f\"En İyi Parametreler: {grid_model.best_params_}\")\nprint(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:10:40.032257Z","iopub.execute_input":"2025-07-26T16:10:40.032625Z","iopub.status.idle":"2025-07-26T16:10:40.043166Z","shell.execute_reply.started":"2025-07-26T16:10:40.032597Z","shell.execute_reply":"2025-07-26T16:10:40.038864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_features = fitted_pipeline.named_steps['OrdinalEncoder'].get_feature_names_out()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:13:53.520800Z","iopub.execute_input":"2025-07-26T16:13:53.521110Z","iopub.status.idle":"2025-07-26T16:13:53.533296Z","shell.execute_reply.started":"2025-07-26T16:13:53.521086Z","shell.execute_reply":"2025-07-26T16:13:53.527128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. Pipeline Tanımı ---\n# Make sure the name of your ColumnTransformer step here is 'preprocessor'\noperations = [\n    (\"preprocessor\", column_trans), # <--- THIS NAME MUST MATCH!\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# ... (rest of your code, including GridSearchCV fit) ...\n\n# --- 10. Özellik Önem Dereceleri (Feature Importances) ---\nfitted_pipeline = grid_model.best_estimator_\n\n# Get the feature names AFTER the ColumnTransformer has been fitted\n# The name 'preprocessor' must match the name given in the Pipeline's operations list.\nnew_features = fitted_pipeline.named_steps['preprocessor'].get_feature_names_out()\n\n# ... (rest of the feature importances calculation) ...","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:14:13.657370Z","iopub.execute_input":"2025-07-26T16:14:13.657668Z","iopub.status.idle":"2025-07-26T16:14:13.700517Z","shell.execute_reply.started":"2025-07-26T16:14:13.657643Z","shell.execute_reply":"2025-07-26T16:14:13.694409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"operations = [\n    (\"preprocessor\", column_trans), # <--- Here it is!\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:13:17.926087Z","iopub.execute_input":"2025-07-26T16:13:17.926467Z","iopub.status.idle":"2025-07-26T16:13:17.937358Z","shell.execute_reply.started":"2025-07-26T16:13:17.926438Z","shell.execute_reply":"2025-07-26T16:13:17.931867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_val fonksiyonunu çağırarak eğitim ve test setindeki performansı gösterelim\ntrain_val(grid_model, X_train, y_train, X_test, y_test)\n\n# --- 10. Özellik Önem Dereceleri (Feature Importances) ---\nfitted_pipeline = grid_model.best_estimator_\n\n# ColumnTransformer sonrası özellik adlarını al\n# StandardScaler, OrdinalEncoder ve OneHotEncoder'ın çıktı isimlerini birleştiriyoruz\n# OneHotEncoder'ın sütun isimleri 'prefix__category' formatında gelir.\nnew_features = fitted_pipeline.named_steps['preprocessor'].get_feature_names_out()\n\n# Özellik önem derecelerini bir DataFrame'e dönüştür\nimp_feats = pd.DataFrame(data=fitted_pipeline[\"GB_model\"].feature_importances_,\n                         columns=['Önem Derecesi'],\n                         index=new_features)\n\ngrad_imp_feats = imp_feats.sort_values('Önem Derecesi', ascending=False)\nprint(\"\\n--- Özellik Önem Dereceleri (En Önemliden Başlayarak) ---\")\nprint(grad_imp_feats.to_markdown()) # Markdown tablo formatında çıktı\n\n\n# --- 11. Sonuç Dosyası Oluşturma (İsteğe Bağlı) ---\n# Genellikle test seti tahminlerini veya model metriklerini kaydetmek istenir\nresults_df = pd.DataFrame({\n    'Gerçek Değerler': y_test,\n    'Tahmin Edilen Değerler': fitted_pipeline.predict(X_test)\n})\n\n# CSV olarak kaydet\nresults_filename = \"model_tahmin_sonuclari.csv\"\nresults_df.to_csv(results_filename, index=False)\nprint(f\"\\nTahmin sonuçları '{results_filename}' dosyasına kaydedildi.\")\n\n# Metrikleri de bir metin dosyasına kaydetmek isteyebilirsiniz\nwith open(\"model_metrikleri.txt\", \"w\") as f:\n    f.write(f\"En İyi Parametreler: {grid_model.best_params_}\\n\")\n    f.write(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\\n\")\n    f.write(\"\\nModel Performans Özeti:\\n\")\n    # train_val fonksiyonunun çıktısını dosyaya yazmak için\n    # geçici olarak stdout'u yakalamak daha gelişmiş bir yöntem olur,\n    # burada basitçe aynı değerleri tekrar yazdırıyoruz.\n    f.write(f\"Eğitim R2: {r2_score(y_train, fitted_pipeline.predict(X_train)):.4f}\\n\")\n    f.write(f\"Test R2: {r2_score(y_test, fitted_pipeline.predict(X_test)):.4f}\\n\")\n    f.write(f\"Eğitim RMSE: {np.sqrt(mean_squared_error(y_train, fitted_pipeline.predict(X_train))):.4f}\\n\")\n    f.write(f\"Test RMSE: {np.sqrt(mean_squared_error(y_test, fitted_pipeline.predict(X_test))):.4f}\\n\")\nprint(f\"Model metrikleri 'model_metrikleri.txt' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:13:25.012896Z","iopub.execute_input":"2025-07-26T16:13:25.013208Z","iopub.status.idle":"2025-07-26T16:13:25.167523Z","shell.execute_reply.started":"2025-07-26T16:13:25.013183Z","shell.execute_reply":"2025-07-26T16:13:25.162853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 11. Sonuç Dosyası Oluşturma (İsteğe Bağlı) ---\nresults_df = pd.DataFrame({\n    'Gerçek Değerler': y_test,\n    'Tahmin Edilen Değerler': fitted_pipeline.predict(X_test)\n})\n\n# CSV olarak kaydet\nresults_filename = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\" # <-- Burası dosya adını belirliyor\nresults_df.to_csv(results_filename, index=False)\nprint(f\"\\nTahmin sonuçları '{results_filename}' dosyasına kaydedildi.\")\n\n# ...\nwith open(\"model_metrikleri.txt\", \"w\") as f: # <-- Burası da metrik dosyasının adını belirliyor\n    # ...\nprint(f\"Model metrikleri 'model_metrikleri.txt' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:19:58.606233Z","iopub.execute_input":"2025-07-26T16:19:58.606590Z","iopub.status.idle":"2025-07-26T16:19:58.621536Z","shell.execute_reply.started":"2025-07-26T16:19:58.606563Z","shell.execute_reply":"2025-07-26T16:19:58.616699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 11. Sonuç Dosyası Oluşturma (İsteğe Bağlı) ---\n# ... (results_df and results_filename creation) ...\n\nresults_df.to_csv(results_filename, index=False)\nprint(f\"\\nTahmin sonuçları '{results_filename}' dosyasına kaydedildi.\")\n\n# Metrikleri de bir metin dosyasına kaydetmek isteyebilirsiniz\nwith open(\"model_metrikleri.txt\", \"w\") as f:\n    # Everything inside this 'with' block MUST be indented\n    f.write(f\"En İyi Parametreler: {grid_model.best_params_}\\n\")\n    f.write(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\\n\")\n    f.write(\"\\nModel Performans Özeti:\\n\")\n    f.write(f\"Eğitim R2: {r2_score(y_train, fitted_pipeline.predict(X_train)):.4f}\\n\")\n    f.write(f\"Test R2: {r2_score(y_test, fitted_pipeline.predict(X_test)):.4f}\\n\")\n    f.write(f\"Eğitim RMSE: {np.sqrt(mean_squared_error(y_train, fitted_pipeline.predict(X_train))):.4f}\\n\")\n    f.write(f\"Test RMSE: {np.sqrt(mean_squared_error(y_test, fitted_pipeline.predict(X_test))):.4f}\\n\")\n\n# This print statement is OUTSIDE the 'with' block, so it should be at the same level as 'with'\nprint(f\"Model metrikleri 'model_metrikleri.txt' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:20:41.349699Z","iopub.execute_input":"2025-07-26T16:20:41.350062Z","iopub.status.idle":"2025-07-26T16:20:41.387134Z","shell.execute_reply.started":"2025-07-26T16:20:41.350035Z","shell.execute_reply":"2025-07-26T16:20:41.382358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 11. Sonuç Dosyası Oluşturma (İsteğe Bağlı) ---\nresults_df = pd.DataFrame({ # <--- This line defines results_df\n    'Gerçek Değerler': y_test,\n    'Tahmin Edilen Değerler': fitted_pipeline.predict(X_test)\n})\n\n# CSV olarak kaydet\nresults_filename = \"model_tahmin_sonuclari.csv\" # <--- This line defines results_filename\nresults_df.to_csv(results_filename, index=False)\n# ... rest of the code","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:21:21.044798Z","iopub.execute_input":"2025-07-26T16:21:21.045115Z","iopub.status.idle":"2025-07-26T16:21:21.112700Z","shell.execute_reply.started":"2025-07-26T16:21:21.045089Z","shell.execute_reply":"2025-07-26T16:21:21.106914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 11. Sonuç Dosyası Oluşturma (İsteğe Bağlı) ---\n\n# Make sure these lines are present and uncommented!\nresults_df = pd.DataFrame({\n    'Gerçek Değerler': y_test,\n    'Tahmin Edilen Değerler': fitted_pipeline.predict(X_test)\n})\n\nresults_filename = \"model_tahmin_sonuclari.csv\" # Or \"submission.csv\" if required by a platform\n\nresults_df.to_csv(results_filename, index=False)\nprint(f\"\\nTahmin sonuçları '{results_filename}' dosyasına kaydedildi.\")\n\n# Metrikleri de bir metin dosyasına kaydetmek isteyebilirsiniz\nwith open(\"model_metrikleri.txt\", \"w\") as f:\n    f.write(f\"En İyi Parametreler: {grid_model.best_params_}\\n\")\n    f.write(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\\n\")\n    f.write(\"\\nModel Performans Özeti:\\n\")\n    f.write(f\"Eğitim R2: {r2_score(y_train, fitted_pipeline.predict(X_train)):.4f}\\n\")\n    f.write(f\"Test R2: {r2_score(y_test, fitted_pipeline.predict(X_test)):.4f}\\n\")\n    f.write(f\"Eğitim RMSE: {np.sqrt(mean_squared_error(y_train, fitted_pipeline.predict(X_train))):.4f}\\n\")\n    f.write(f\"Test RMSE: {np.sqrt(mean_squared_error(y_test, fitted_pipeline.predict(X_test))):.4f}\\n\")\n\n# This print statement is OUTSIDE the 'with' block, so it should be at the same level as 'with'\nprint(f\"Model metrikleri 'model_metrikleri.txt' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:21:38.516483Z","iopub.execute_input":"2025-07-26T16:21:38.516775Z","iopub.status.idle":"2025-07-26T16:21:38.581637Z","shell.execute_reply.started":"2025-07-26T16:21:38.516750Z","shell.execute_reply":"2025-07-26T16:21:38.575190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# ... (rest of your imports and previous code for model training) ...\n\n# --- Generate your predictions for the test set ---\n# Assuming 'X_test_final' is your preprocessed test data ready for prediction\n# If you used a pipeline, it would look like this:\n# y_test_predictions = fitted_pipeline.predict(X_test_competition)\n# Make sure X_test_competition is the *unseen* test data from the competition,\n# not your split X_test from cross-validation.\n\n# If you need to load the competition's test data separately:\n# test_data_for_submission = pd.read_csv('path/to/competition_test_data.csv')\n# # Apply the same preprocessing pipeline to this unseen test data\n# y_test_predictions = fitted_pipeline.predict(test_data_for_submission)\n\n\n# Create the DataFrame for submission\n# This assumes y_test_predictions contains your model's outputs for the competition's test set\nsubmission_df = pd.DataFrame({\n    'ID': test_data_for_submission['ID'], # Replace 'ID' with your actual ID column name from competition test data\n    'Target': y_test_predictions # Replace 'Target' with your actual target column name for submission\n})\n\n# --- Save the submission file ---\nsubmission_filename = \"submission.csv\" # <--- THIS IS CRUCIAL!\n\nsubmission_df.to_csv(submission_filename, index=False) # index=False is important to avoid writing DataFrame index\n\nprint(f\"\\nSubmission file '{submission_filename}' successfully created!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:26:59.237712Z","iopub.execute_input":"2025-07-26T16:26:59.238044Z","iopub.status.idle":"2025-07-26T16:26:59.275967Z","shell.execute_reply.started":"2025-07-26T16:26:59.238017Z","shell.execute_reply":"2025-07-26T16:26:59.269517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Örnek Veri Oluşturma (Bu kısım model eğitimi için kendi veriniz olmalı) ---\n# Gerçek bir yarışmada bu, genellikle \"train.csv\" dosyasını yüklemektir.\nnp.random.seed(42)\nnum_samples = 150\nnum_features = 5\ndata = {\n    'numerical_feature_1': np.random.rand(num_samples) * 100,\n    'numerical_feature_2': np.random.rand(num_samples) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], num_samples),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], num_samples),\n    'another_numerical_feature': np.random.randint(0, 100, num_samples),\n    'target': np.random.rand(num_samples) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\nnan_indices_X = np.random.choice(X.index, 10, replace=False)\ninf_indices_X = np.random.choice(X.index, 5, replace=False)\nnan_indices_y = np.random.choice(y.index, 3, replace=False)\n\nfor idx in nan_indices_X:\n    col = np.random.choice(X.columns)\n    X.loc[idx, col] = np.nan\nfor idx in inf_indices_X:\n    col = np.random.choice(X.select_dtypes(include=np.number).columns)\n    X.loc[idx, col] = np.inf\ny.loc[nan_indices_y] = np.nan\n\nprint(f\"--- Başlangıç Veri Durumu ---\")\nprint(f\"X boyutu: {X.shape}\")\nprint(f\"y boyutu: {y.shape}\")\nprint(f\"X'teki toplam NaN sayısı: {X.isnull().sum().sum()}\")\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X).sum().sum()}\")\nprint(f\"y'deki toplam NaN sayısı: {y.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 2. Veri Temizleme (NaN ve Sonsuz Değerleri Silme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\nprint(f\"--- Temizleme Sonrası Veri Durumu ---\")\nprint(f\"X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki toplam NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\nprint(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned).sum().sum()}\")\nprint(f\"y_cleaned'daki toplam NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 3. Veriyi Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Tanımı (Ön İşleme) ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2', 'another_numerical_feature']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\nordinal_categories_order = [['low', 'medium', 'high', 'very_high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order, handle_unknown='use_encoded_value', unknown_value=-1)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- 5. Pipeline Tanımı ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# --- 6. param_grid Tanımı (GridSearchCV için Hiperparametreler) ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 150],\n    'GB_model__learning_rate': [0.05, 0.1, 0.15],\n    'GB_model__max_depth': [3, 4],\n}\n\n# --- 7. GridSearchCV Tanımı ve Eğitimi ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\nprint(\"\\n--- GridSearchCV Başlıyor... Bu biraz zaman alabilir ---\")\ngrid_model.fit(X_train, y_train)\nprint(\"--- GridSearchCV Tamamlandı ---\")\n\n# --- 8. Model Performansını Değerlendirme Fonksiyonu ---\ndef train_val(model_or_grid_model, X_train, y_train, X_test, y_test):\n    if hasattr(model_or_grid_model, 'best_estimator_'):\n        model_to_predict = model_or_grid_model.best_estimator_\n    else:\n        model_to_predict = model_or_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:27:58.969309Z","iopub.execute_input":"2025-07-26T16:27:58.969632Z","iopub.status.idle":"2025-07-26T16:27:59.888609Z","shell.execute_reply.started":"2025-07-26T16:27:58.969606Z","shell.execute_reply":"2025-07-26T16:27:59.882981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Örnek Veri Oluşturma (Bu kısım model eğitimi için kendi veriniz olmalı) ---\n# Gerçek bir yarışmada bu, genellikle \"train.csv\" dosyasını yüklemektir.\nnp.random.seed(42)\nnum_samples = 150\nnum_features = 5\ndata = {\n    'numerical_feature_1': np.random.rand(num_samples) * 100,\n    'numerical_feature_2': np.random.rand(num_samples) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], num_samples),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], num_samples),\n    'another_numerical_feature': np.random.randint(0, 100, num_samples),\n    'target': np.random.rand(num_samples) * 200\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\nnan_indices_X = np.random.choice(X.index, 10, replace=False)\ninf_indices_X = np.random.choice(X.index, 5, replace=False)\nnan_indices_y = np.random.choice(y.index, 3, replace=False)\n\nfor idx in nan_indices_X:\n    col = np.random.choice(X.columns)\n    X.loc[idx, col] = np.nan\nfor idx in inf_indices_X:\n    col = np.random.choice(X.select_dtypes(include=np.number).columns)\n    X.loc[idx, col] = np.inf\ny.loc[nan_indices_y] = np.nan\n\nprint(f\"--- Başlangıç Veri Durumu ---\")\nprint(f\"X boyutu: {X.shape}\")\nprint(f\"y boyutu: {y.shape}\")\nprint(f\"X'teki toplam NaN sayısı: {X.isnull().sum().sum()}\")\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X).sum().sum()}\")\nprint(f\"y'deki toplam NaN sayısı: {y.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 2. Veri Temizleme (NaN ve Sonsuz Değerleri Silme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\nprint(f\"--- Temizleme Sonrası Veri Durumu ---\")\nprint(f\"X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki toplam NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\nprint(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned).sum().sum()}\")\nprint(f\"y_cleaned'daki toplam NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 3. Veriyi Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Tanımı (Ön İşleme) ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2', 'another_numerical_feature']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\nordinal_categories_order = [['low', 'medium', 'high', 'very_high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order, handle_unknown='use_encoded_value', unknown_value=-1)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- 5. Pipeline Tanımı ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# --- 6. param_grid Tanımı (GridSearchCV için Hiperparametreler) ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 150],\n    'GB_model__learning_rate': [0.05, 0.1, 0.15],\n    'GB_model__max_depth': [3, 4],\n}\n\n# --- 7. GridSearchCV Tanımı ve Eğitimi ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\nprint(\"\\n--- GridSearchCV Başlıyor... Bu biraz zaman alabilir ---\")\ngrid_model.fit(X_train, y_train)\nprint(\"--- GridSearchCV Tamamlandı ---\")\n\n# --- 8. Model Performansını Değerlendirme Fonksiyonu ---\ndef train_val(model_or_grid_model, X_train, y_train, X_test, y_test):\n    if hasattr(model_or_grid_model, 'best_estimator_'):\n        model_to_predict = model_or_grid_model.best_estimator_\n    else:\n        model_to_predict = model_or_grid_model\n\n    y_train_pred = model_to_predict.predict(X_train)\n    y_test_pred = model_to_predict.predict(X_test)\n\n    train_r2 = r2_score(y_train, y_train_pred)\n    train_mae = mean_absolute_error(y_train, y_train_pred)\n    train_mse = mean_squared_error(y_train, y_train_pred)\n    train_rmse = np.sqrt(mean_squared_error(y_train, y_train_pred))\n\n    test_r2 = r2_score(y_test, y_test_pred)\n    test_mae = mean_absolute_error(y_test, y_test_pred)\n    test_mse = mean_squared_error(y_test, y_test_pred)\n    test_rmse = np.sqrt(mean_squared_error(y_test, y_test_pred))\n\n    print(\"\\n--- Model Performans Özeti ---\")\n    results = pd.DataFrame({\n        'Metrik': ['R2', 'MAE', 'MSE', 'RMSE'],\n        'Eğitim Seti': [f\"{train_r2:.4f}\", f\"{train_mae:.4f}\", f\"{train_mse:.4f}\", f\"{train_rmse:.4f}\"],\n        'Test Seti': [f\"{test_r2:.4f}\", f\"{test_mae:.4f}\", f\"{test_mse:.4f}\", f\"{test_rmse:.4f}\"]\n    })\n    print(results.to_markdown(index=False))\n\n# --- 9. En İyi Model ile Tahmin ve Değerlendirme ---\nprint(\"\\n--- En İyi GridSearchCV Modeli ---\")\nprint(f\"En İyi Parametreler: {grid_model.best_params_}\")\nprint(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\")\n\ntrain_val(grid_model, X_train, y_train, X_test, y_test)\n\n# --- 10. Özellik Önem Dereceleri (Feature Importances) ---\nfitted_pipeline = grid_model.best_estimator_\nnew_features = fitted_pipeline.named_steps['preprocessor'].get_feature_names_out()\n\nimp_feats = pd.DataFrame(data=fitted_pipeline[\"GB_model\"].feature_importances_,\n                         columns=['Önem Derecesi'],\n                         index=new_features)\n\ngrad_imp_feats = imp_feats.sort_values('Önem Derecesi', ascending=False)\nprint(\"\\n--- Özellik Önem Dereceleri (En Önemliden Başlayarak) ---\")\nprint(grad_imp_feats.to_markdown())\n\n\n# --- 11. Sonuç Dosyası Oluşturma (Yarışma İçin) ---\n\n# --- YARIŞMA TEST VERİSİNİ BURADA YÜKLEYİN ---\n# Bu, modelinizi eğitmek için kullandığınız X_cleaned'den farklıdır.\n# Bu dosya, yarışma platformu tarafından size sağlanır ve genellikle \"test.csv\" veya \"sample_submission.csv\" ile birlikte gelir.\n# Örneğin:\ntry:\n    # Gerçek yarışma verisi yerine geçici bir örnek test verisi oluşturalım\n    # Gerçek senaryoda bu satırı:\n    # test_data_for_submission = pd.read_csv('path/to/your/competition_test_data.csv')\n    # olarak değiştirmelisiniz. ID sütununa sahip olduğundan emin olun.\n    test_data_for_submission = pd.DataFrame({\n        'ID': np.arange(200, 200 + 50), # Örnek ID'ler\n        'numerical_feature_1': np.random.rand(50) * 100,\n        'numerical_feature_2': np.random.rand(50) * 50,\n        'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], 50),\n        'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], 50),\n        'another_numerical_feature': np.random.randint(0, 100, 50),\n        # Yarışma test verisinde hedef sütun olmaz\n    })\n    # Eğer test_data_for_submission'da NaN veya Inf varsa, bunları da temizlemeniz gerekebilir\n    test_data_for_submission.replace([np.inf, -np.inf], np.nan, inplace=True)\n    # İmputation (doldurma) kullanmıyorsak, NaN içeren satırları silmeliyiz (dikkatli olun)\n    # Veya test_data_for_submission'da eksik veriyi ele almak için bir imputation stratejisi uygulayın.\n    # Örneğin: test_data_for_submission.dropna(inplace=True)\n\nexcept FileNotFoundError:\n    print(\"\\nUYARI: 'test_data_for_submission' için örnek veri kullanılıyor çünkü dosya bulunamadı.\")\n    print(\"Yarışma için kendi test veri setinizin dosya yolunu 'pd.read_csv()' içinde belirttiğinizden emin olun.\")\n    # Örnek oluşturma kodunu yukarıda bıraktık.\n\n# Modelinizi kullanarak yarışma test verileri üzerinde tahminler yapın\n# Pipeline, ön işleme adımlarını (ölçekleme, kodlama) otomatik olarak uygulayacaktır.\ny_competition_predictions = fitted_pipeline.predict(test_data_for_submission)\n\n# Submission DataFrame'ini oluşturun\n# 'ID' ve 'Target' sütun adları yarışmanın beklentisine göre değişebilir.\nsubmission_df = pd.DataFrame({\n    'ID': test_data_for_submission['ID'], # Yarışma test verisindeki ID sütunu\n    'Target': y_competition_predictions    # Modelin tahminleri\n})\n\n# --- Submission dosyasını kaydedin ---\nsubmission_filename = \"submission.csv\" # <--- Yarışma platformunun beklediği dosya adı bu OLMALIDIR!\n\nsubmission_df.to_csv(submission_filename, index=False) # index=False çok önemli!\n\nprint(f\"\\nSubmission dosyası '{submission_filename}' başarıyla oluşturuldu ve kaydedildi.\")\nprint(f\"Submission dosyasının ilk 5 satırı:\\n{submission_df.head().to_markdown(index=False)}\")\n\n\n# Metrikleri de bir metin dosyasına kaydetmek isteyebilirsiniz\nmetrics_filename = \"model_metrikleri.txt\"\nwith open(metrics_filename, \"w\") as f:\n    f.write(f\"En İyi Parametreler: {grid_model.best_params_}\\n\")\n    f.write(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\\n\")\n    f.write(\"\\nModel Performans Özeti:\\n\")\n    f.write(f\"Eğitim R2: {r2_score(y_train, fitted_pipeline.predict(X_train)):.4f}\\n\")\n    f.write(f\"Test R2: {r2_score(y_test, fitted_pipeline.predict(X_test)):.4f}\\n\")\n    f.write(f\"Eğitim RMSE: {np.sqrt(mean_squared_error(y_train, fitted_pipeline.predict(X_train))):.4f}\\n\")\n    f.write(f\"Test RMSE: {np.sqrt(mean_squared_error(y_test, fitted_pipeline.predict(X_test))):.4f}\\n\")\nprint(f\"Model metrikleri '{metrics_filename}' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:29:10.726031Z","iopub.execute_input":"2025-07-26T16:29:10.726368Z","iopub.status.idle":"2025-07-26T16:29:10.882132Z","shell.execute_reply.started":"2025-07-26T16:29:10.726342Z","shell.execute_reply":"2025-07-26T16:29:10.875570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n# ... (rest of your imports) ...\n\n# ... (rest of your data creation and initial prints) ...\n\n# Corrected line for checking infinite values:\n# Only check for inf in numeric columns of X\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X.select_dtypes(include=np.number)).sum().sum()}\")\n# X.select_dtypes(include=np.number) creates a view of the DataFrame containing only numeric columns.\n# Then, np.isinf() can be safely applied.\n\n# ... (rest of your code) ...\n\n# You might also want to apply a similar check to X_cleaned if you ever print its inf count\n# print(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned.select_dtypes(include=np.number)).sum().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:29:56.648961Z","iopub.execute_input":"2025-07-26T16:29:56.649306Z","iopub.status.idle":"2025-07-26T16:29:56.661469Z","shell.execute_reply.started":"2025-07-26T16:29:56.649280Z","shell.execute_reply":"2025-07-26T16:29:56.655869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Örnek Veri Oluşturma ---\nnp.random.seed(42) # Tekrarlanabilirlik için\nnum_samples = 150 # Daha fazla örnek\nnum_features = 5\ndata = {\n    'numerical_feature_1': np.random.rand(num_samples) * 100,\n    'numerical_feature_2': np.random.rand(num_samples) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], num_samples),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], num_samples),\n    'another_numerical_feature': np.random.randint(0, 100, num_samples), # Yeni bir sayısal özellik\n    'target': np.random.rand(num_samples) * 200 # Tahmin edilecek hedef\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Test için kasıtlı olarak NaN ve Inf değerler ekleyelim\nnan_indices_X = np.random.choice(X.index, 10, replace=False)\ninf_indices_X = np.random.choice(X.index, 5, replace=False)\nnan_indices_y = np.random.choice(y.index, 3, replace=False)\n\nfor idx in nan_indices_X:\n    col = np.random.choice(X.columns)\n    X.loc[idx, col] = np.nan\nfor idx in inf_indices_X:\n    col = np.random.choice(X.select_dtypes(include=np.number).columns) # Sadece sayısal sütunlara inf ekle\n    X.loc[idx, col] = np.inf\ny.loc[nan_indices_y] = np.nan\n\n\nprint(f\"--- Başlangıç Veri Durumu ---\")\nprint(f\"X boyutu: {X.shape}\")\nprint(f\"y boyutu: {y.shape}\")\nprint(f\"X'teki toplam NaN sayısı: {X.isnull().sum().sum()}\")\n# --- DÜZELTİLMİŞ SATIR ---\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X.select_dtypes(include=np.number)).sum().sum()}\")\n# -------------------------\nprint(f\"y'deki toplam NaN sayısı: {y.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 2. Veri Temizleme (NaN ve Sonsuz Değerleri Silme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\nprint(f\"--- Temizleme Sonrası Veri Durumu ---\")\nprint(f\"X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki toplam NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\n# --- DÜZELTİLMİŞ SATIR ---\n# Artık temizlediğimiz için 0 olmalı, ama yine de hata vermemesi için select_dtypes kullanırız.\nprint(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned.select_dtypes(include=np.number)).sum().sum()}\")\n# -------------------------\nprint(f\"y_cleaned'daki toplam NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 3. Veriyi Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Tanımı (Ön İşleme) ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2', 'another_numerical_feature']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:30:54.499181Z","iopub.execute_input":"2025-07-26T16:30:54.499477Z","iopub.status.idle":"2025-07-26T16:30:54.538744Z","shell.execute_reply.started":"2025-07-26T16:30:54.499452Z","shell.execute_reply":"2025-07-26T16:30:54.533172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n# --- 1. Örnek Veri Oluşturma ---\nnp.random.seed(42) # Tekrarlanabilirlik için\nnum_samples = 150 # Daha fazla örnek\nnum_features = 5\ndata = {\n    'numerical_feature_1': np.random.rand(num_samples) * 100,\n    'numerical_feature_2': np.random.rand(num_samples) * 50,\n    'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], num_samples),\n    'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], num_samples),\n    'another_numerical_feature': np.random.randint(0, 100, num_samples), # Yeni bir sayısal özellik\n    'target': np.random.rand(num_samples) * 200 # Tahmin edilecek hedef\n}\nX = pd.DataFrame(data)\ny = X['target']\nX = X.drop('target', axis=1)\n\n# Test için kasıtlı olarak NaN ve Inf değerler ekleyelim\nnan_indices_X = np.random.choice(X.index, 10, replace=False)\ninf_indices_X = np.random.choice(X.index, 5, replace=False)\nnan_indices_y = np.random.choice(y.index, 3, replace=False)\n\nfor idx in nan_indices_X:\n    col = np.random.choice(X.columns)\n    X.loc[idx, col] = np.nan\nfor idx in inf_indices_X:\n    col = np.random.choice(X.select_dtypes(include=np.number).columns) # Sadece sayısal sütunlara inf ekle\n    X.loc[idx, col] = np.inf\ny.loc[nan_indices_y] = np.nan\n\n\nprint(f\"--- Başlangıç Veri Durumu ---\")\nprint(f\"X boyutu: {X.shape}\")\nprint(f\"y boyutu: {y.shape}\")\nprint(f\"X'teki toplam NaN sayısı: {X.isnull().sum().sum()}\")\n# --- DÜZELTİLMİŞ SATIR ---\nprint(f\"X'teki toplam Inf sayısı: {np.isinf(X.select_dtypes(include=np.number)).sum().sum()}\")\n# -------------------------\nprint(f\"y'deki toplam NaN sayısı: {y.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 2. Veri Temizleme (NaN ve Sonsuz Değerleri Silme) ---\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_combined = pd.concat([X, y.rename('target')], axis=1)\ndf_cleaned = df_combined.dropna()\n\nX_cleaned = df_cleaned.drop('target', axis=1)\ny_cleaned = df_cleaned['target']\n\nprint(f\"--- Temizleme Sonrası Veri Durumu ---\")\nprint(f\"X_cleaned boyutu: {X_cleaned.shape}\")\nprint(f\"y_cleaned boyutu: {y_cleaned.shape}\")\nprint(f\"X_cleaned'daki toplam NaN sayısı: {X_cleaned.isnull().sum().sum()}\")\n# --- DÜZELTİLMİŞ SATIR ---\n# Artık temizlediğimiz için 0 olmalı, ama yine de hata vermemesi için select_dtypes kullanırız.\nprint(f\"X_cleaned'daki toplam Inf sayısı: {np.isinf(X_cleaned.select_dtypes(include=np.number)).sum().sum()}\")\n# -------------------------\nprint(f\"y_cleaned'daki toplam NaN sayısı: {y_cleaned.isnull().sum()}\")\nprint(\"-\" * 40)\n\n# --- 3. Veriyi Eğitim ve Test Setlerine Ayırma ---\nX_train, X_test, y_train, y_test = train_test_split(X_cleaned, y_cleaned, test_size=0.2, random_state=42)\n\n# --- 4. ColumnTransformer Tanımı (Ön İşleme) ---\nnumerical_cols = ['numerical_feature_1', 'numerical_feature_2', 'another_numerical_feature']\nordinal_cols = ['ordinal_feature']\nnominal_cols = ['nominal_feature']\n\nordinal_categories_order = [['low', 'medium', 'high', 'very_high']]\n\nnumerical_transformer = StandardScaler()\nordinal_transformer = OrdinalEncoder(categories=ordinal_categories_order, handle_unknown='use_encoded_value', unknown_value=-1)\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')\n\ncolumn_trans = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('ord', ordinal_transformer, ordinal_cols),\n        ('nom', nominal_transformer, nominal_cols)\n    ],\n    remainder='passthrough'\n)\n\n# --- 5. Pipeline Tanımı ---\noperations = [\n    (\"preprocessor\", column_trans),\n    (\"GB_model\", GradientBoostingRegressor(random_state=101))\n]\nmodel = Pipeline(steps=operations)\n\n# --- 6. param_grid Tanımı (GridSearchCV için Hiperparametreler) ---\nparam_grid = {\n    'GB_model__n_estimators': [50, 100, 150],\n    'GB_model__learning_rate': [0.05, 0.1, 0.15],\n    'GB_model__max_depth': [3, 4],\n}\n\n# --- 7. GridSearchCV Tanımı ve Eğitimi ---\ngrid_model = GridSearchCV(estimator=model,\n                          param_grid=param_grid,\n                          scoring='neg_root_mean_squared_error',\n                          cv=5,\n                          n_jobs=-1,\n                          return_train_score=True)\n\nprint(\"\\n--- GridSearchCV Başlıyor... Bu biraz zaman alabilir ---\")\ngrid_model.fit(X_train, y_train)\nprint(\"--- GridSearchCV Tamamlandı ---\")\n\n# --- 8. Model Performansını Değerlendirme Fonksiyonu ---\ndef train_val(model_or_grid_model, X_train, y_train, X_test, y_test):\n    if hasattr(model_or_grid_model, 'best_estimator_'):\n        model_to_predict = model_or_grid_model.best_estimator_\n    else:\n        model_to_predict = model_or_grid_model\n\n    y_train_pred = model_to_predict.predict(X_train)\n    y_test_pred = model_to_predict.predict(X_test)\n\n    train_r2 = r2_score(y_train, y_train_pred)\n    train_mae = mean_absolute_error(y_train, y_train_pred)\n    train_mse = mean_squared_error(y_train, y_train_pred)\n    train_rmse = np.sqrt(mean_squared_error(y_train, y_train_pred))\n\n    test_r2 = r2_score(y_test, y_test_pred)\n    test_mae = mean_absolute_error(y_test, y_test_pred)\n    test_mse = mean_squared_error(y_test, y_test_pred)\n    test_rmse = np.sqrt(mean_squared_error(y_test, y_test_pred))\n\n    print(\"\\n--- Model Performans Özeti ---\")\n    results = pd.DataFrame({\n        'Metrik': ['R2', 'MAE', 'MSE', 'RMSE'],\n        'Eğitim Seti': [f\"{train_r2:.4f}\", f\"{train_mae:.4f}\", f\"{train_mse:.4f}\", f\"{train_rmse:.4f}\"],\n        'Test Seti': [f\"{test_r2:.4f}\", f\"{test_mae:.4f}\", f\"{test_mse:.4f}\", f\"{test_rmse:.4f}\"]\n    })\n    print(results.to_markdown(index=False))\n\n# --- 9. En İyi Model ile Tahmin ve Değerlendirme ---\nprint(\"\\n--- En İyi GridSearchCV Modeli ---\")\nprint(f\"En İyi Parametreler: {grid_model.best_params_}\")\nprint(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\")\n\ntrain_val(grid_model, X_train, y_train, X_test, y_test)\n\n# --- 10. Özellik Önem Dereceleri (Feature Importances) ---\nfitted_pipeline = grid_model.best_estimator_\nnew_features = fitted_pipeline.named_steps['preprocessor'].get_feature_names_out()\n\nimp_feats = pd.DataFrame(data=fitted_pipeline[\"GB_model\"].feature_importances_,\n                         columns=['Önem Derecesi'],\n                         index=new_features)\n\ngrad_imp_feats = imp_feats.sort_values('Önem Derecesi', ascending=False)\nprint(\"\\n--- Özellik Önem Dereceleri (En Önemliden Başlayarak) ---\")\nprint(grad_imp_feats.to_markdown())\n\n\n# --- 11. Sonuç Dosyası Oluşturma (Yarışma İçin) ---\n\ntry:\n    test_data_for_submission = pd.DataFrame({\n        'ID': np.arange(200, 200 + 50),\n        'numerical_feature_1': np.random.rand(50) * 100,\n        'numerical_feature_2': np.random.rand(50) * 50,\n        'ordinal_feature': np.random.choice(['low', 'medium', 'high', 'very_high'], 50),\n        'nominal_feature': np.random.choice(['A', 'B', 'C', 'D', 'E'], 50),\n        'another_numerical_feature': np.random.randint(0, 100, 50),\n    })\n    test_data_for_submission.replace([np.inf, -np.inf], np.nan, inplace=True)\n    # Eğer test_data_for_submission'da NaN varsa ve ColumnTransformer içinde SimpleImputer yoksa,\n    # burada .dropna() veya başka bir imputation stratejisi uygulamanız gerekebilir.\n    # Aksi takdirde, predict() çağrıldığında NaN'lar hata verebilir.\n\nexcept FileNotFoundError:\n    print(\"\\nUYARI: 'test_data_for_submission' için örnek veri kullanılıyor çünkü dosya bulunamadı.\")\n    print(\"Yarışma için kendi test veri setinizin dosya yolunu 'pd.read_csv()' içinde belirttiğinizden emin olun.\")\n\ny_competition_predictions = fitted_pipeline.predict(test_data_for_submission)\n\nsubmission_df = pd.DataFrame({\n    'ID': test_data_for_submission['ID'],\n    'Target': y_competition_predictions\n})\n\nsubmission_filename = \"submission.csv\"\n\nsubmission_df.to_csv(submission_filename, index=False)\n\nprint(f\"\\nSubmission dosyası '{submission_filename}' başarıyla oluşturuldu ve kaydedildi.\")\nprint(f\"Submission dosyasının ilk 5 satırı:\\n{submission_df.head().to_markdown(index=False)}\")\n\n\nmetrics_filename = \"model_metrikleri.txt\"\nwith open(metrics_filename, \"w\") as f:\n    f.write(f\"En İyi Parametreler: {grid_model.best_params_}\\n\")\n    f.write(f\"En İyi Çapraz Doğrulama RMSE Skoru: {-grid_model.best_score_:.4f}\\n\")\n    f.write(\"\\nModel Performans Özeti:\\n\")\n    f.write(f\"Eğitim R2: {r2_score(y_train, fitted_pipeline.predict(X_train)):.4f}\\n\")\n    f.write(f\"Test R2: {r2_score(y_test, fitted_pipeline.predict(X_test)):.4f}\\n\")\n    f.write(f\"Eğitim RMSE: {np.sqrt(mean_squared_error(y_train, fitted_pipeline.predict(X_train))):.4f}\\n\")\n    f.write(f\"Test RMSE: {np.sqrt(mean_squared_error(y_test, fitted_pipeline.predict(X_test))):.4f}\\n\")\nprint(f\"Model metrikleri '{metrics_filename}' dosyasına kaydedildi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:31:37.714895Z","iopub.execute_input":"2025-07-26T16:31:37.715223Z","iopub.status.idle":"2025-07-26T16:31:41.615919Z","shell.execute_reply.started":"2025-07-26T16:31:37.715195Z","shell.execute_reply":"2025-07-26T16:31:41.611624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Before (causing the error):\n# submission_df = pd.DataFrame({\n#     'ID': test_data_for_submission['ID'],\n#     'Target': y_competition_predictions # <--- This is 'Target'\n# })\n\n# After (Corrected):\nsubmission_df = pd.DataFrame({\n    'ID': test_data_for_submission['ID'],\n    'prediction': y_competition_predictions # <--- CHANGE THIS TO 'prediction'\n})\n\n# Make sure to run your entire notebook again after this change,\n# and then try submitting the newly generated submission.csv file.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:45:16.142223Z","iopub.execute_input":"2025-07-26T16:45:16.142568Z","iopub.status.idle":"2025-07-26T16:45:16.152803Z","shell.execute_reply.started":"2025-07-26T16:45:16.142540Z","shell.execute_reply":"2025-07-26T16:45:16.148450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_filename = \"submission.csv\" # NOT \"Submission.csv\" or \"mysubmission.csv\"\nsubmission_df.to_csv(submission_filename, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-26T16:45:21.576712Z","iopub.execute_input":"2025-07-26T16:45:21.577002Z","iopub.status.idle":"2025-07-26T16:45:21.587938Z","shell.execute_reply.started":"2025-07-26T16:45:21.576977Z","shell.execute_reply":"2025-07-26T16:45:21.583530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}