{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<p style=\"text-align: center;\">\n  <img src=\"https://www.aimtechnologies.co/wp-content/uploads/2023/12/Data-Analysis-in-Social-Media.png\" alt=\"Analiz\">\n</p>\n\n\n\n# 🎯 Analysis and Lightgbm Model Project \n\n\n\n\n## 📋 1. Problem Tanımı (Giriş) \n- **Hedefler:**\n  * Bu yarışmanın amacı çeşitli faktörlere bağlı olarak sigorta primlerini tahmin etmektir.\n\n## 📊 2. Veri Seti Bilgisi\n- **Veri Kaynağı:** [Regression with an Insurance Dataset](https://www.kaggle.com/competitions/playground-series-s4e12/data)\n- **Değişkenler:**\n  * id : Kullanıcı id\n  * Age: Yaş\n  * Gender: Cinsiyet\n  * Annual Income : Yıllık Gelir\n  * Marital Status : Medeni Durum\n  * Number of Dependents : Bakmakla yükümlü olduğu kişi sayısı\n  * Education Level : Eğitim Seviyesi\n  * Occupation : Meslek\n  * Health Score : Sağlık Skoru\n  * Location : Lokasyonu\n  * Policy Type : Poliçe Türü\n  * Previous Claims : Önceki Talepleri\n  * Vehicle Age : Araç yaşı\n  * Credit Score : Kredi Skoru\n  * Insurance Duration : Sigorta Süresi\n  * Policy Start Date : Poliçe Başlama Tarihi\n  * Customer Feedback : Dönüşü\n  * Smoking Status : Sigara İçme Durumu\n  * Exercise Frequency : Egzersiz Sıklığı \n  * Property Type : Emlak Tipi\n  * Premium Amount : Prim Tutarı\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.exceptions import ConvergenceWarning\nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso, ElasticNet\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.tree import DecisionTreeRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split, cross_val_score,GridSearchCV\nfrom sklearn.preprocessing import StandardScaler\n\nwarnings.simplefilter(action='ignore', category=FutureWarning)\nwarnings.simplefilter(\"ignore\", category=ConvergenceWarning)\nwarnings.filterwarnings(\"ignore\")\n\n\npd.set_option('display.max_columns', None)\n#pd.set_option('display.max_rows', None)\npd.set_option('display.width', None)\npd.set_option('display.float_format', lambda x: '%.3f' % x)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:14:42.725185Z","iopub.execute_input":"2025-01-08T17:14:42.725528Z","iopub.status.idle":"2025-01-08T17:14:47.420334Z","shell.execute_reply.started":"2025-01-08T17:14:42.7255Z","shell.execute_reply":"2025-01-08T17:14:47.418925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1.Veri Keşfi ve Anlama\n\ntrain_df=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n\ntest_df=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\nsample_df=pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\ndf = pd.concat([train_df, test_df], axis=0, ignore_index=True)\n\n# Bu verisetinde hedef değişken Premium Amount \n# df değişkenin de train ve test i birleştirip tek seferde işlemler yapacağız","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:14:51.092647Z","iopub.execute_input":"2025-01-08T17:14:51.093498Z","iopub.status.idle":"2025-01-08T17:15:04.999502Z","shell.execute_reply.started":"2025-01-08T17:14:51.093461Z","shell.execute_reply":"2025-01-08T17:15:04.99827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape\n# train satır ve sütunlarını kontrol ediyoruz","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:15:16.104413Z","iopub.execute_input":"2025-01-08T17:15:16.10486Z","iopub.status.idle":"2025-01-08T17:15:16.11348Z","shell.execute_reply.started":"2025-01-08T17:15:16.104826Z","shell.execute_reply":"2025-01-08T17:15:16.112036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()\n\n# boş değer sayısı","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:15:44.160227Z","iopub.execute_input":"2025-01-08T17:15:44.160642Z","iopub.status.idle":"2025-01-08T17:15:44.803498Z","shell.execute_reply.started":"2025-01-08T17:15:44.160611Z","shell.execute_reply":"2025-01-08T17:15:44.80227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape\n\n# test in satır ve sütun bilgisine bakıyoruz","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:15:57.523962Z","iopub.execute_input":"2025-01-08T17:15:57.52438Z","iopub.status.idle":"2025-01-08T17:15:57.531259Z","shell.execute_reply.started":"2025-01-08T17:15:57.524349Z","shell.execute_reply":"2025-01-08T17:15:57.529873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()\n# veri setine ilk bakış","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:16:06.677091Z","iopub.execute_input":"2025-01-08T17:16:06.677462Z","iopub.status.idle":"2025-01-08T17:16:06.700522Z","shell.execute_reply.started":"2025-01-08T17:16:06.677434Z","shell.execute_reply":"2025-01-08T17:16:06.699498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n\n\ndf['Year'] =df['Policy Start Date'].dt.year\ndf['Quarter'] = df['Policy Start Date'].dt.quarter\ndf['Month'] = df['Policy Start Date'].dt.month\ndf['Day'] = df['Policy Start Date'].dt.day\ndf['day_of_week'] = df['Policy Start Date'].dt.day_name()\ndf['week_of_year'] = df['Policy Start Date'].dt.isocalendar().week\n\ndf['day_sin'] = np.sin(2 * np.pi * df['Day'] / 365.0)\ndf['day_cos'] = np.cos(2 * np.pi * df['Day'] / 365.0)\ndf['month_sin'] = np.sin(2 * np.pi * df['Month'] / 12.0)\ndf['month_cos'] = np.cos(2 * np.pi * df['Month'] / 12.0)\ndf['year_sin'] = np.sin(2 * np.pi * df['Year'] / 7.0)\ndf['year_cos'] = np.cos(2 * np.pi * df['Year'] / 7.0)\ndf['Group']=(df['Year']-2010)*48+df['Month']*4+df['Day']//7\n\n\ndf['Quarter'] = df['Quarter'].astype('str')\ndf['Month'] = df['Month'].astype('str')\ndf['day_of_week'] = df['day_of_week'].astype('str')\ndf['week_of_year'] = df['week_of_year'].astype('str')\n\n# Policy Start Date değişkenini ayırdık","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:16:21.47521Z","iopub.execute_input":"2025-01-08T17:16:21.475603Z","iopub.status.idle":"2025-01-08T17:16:26.619626Z","shell.execute_reply.started":"2025-01-08T17:16:21.475573Z","shell.execute_reply":"2025-01-08T17:16:26.618484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop('Policy Start Date', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:16:37.45648Z","iopub.execute_input":"2025-01-08T17:16:37.456905Z","iopub.status.idle":"2025-01-08T17:16:38.111569Z","shell.execute_reply.started":"2025-01-08T17:16:37.456876Z","shell.execute_reply":"2025-01-08T17:16:38.110424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()\n# df in ilk 5 satırı","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:16:47.611923Z","iopub.execute_input":"2025-01-08T17:16:47.612334Z","iopub.status.idle":"2025-01-08T17:16:47.636103Z","shell.execute_reply.started":"2025-01-08T17:16:47.612304Z","shell.execute_reply":"2025-01-08T17:16:47.63484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:16:57.646201Z","iopub.execute_input":"2025-01-08T17:16:57.646597Z","iopub.status.idle":"2025-01-08T17:16:57.666613Z","shell.execute_reply.started":"2025-01-08T17:16:57.646566Z","shell.execute_reply":"2025-01-08T17:16:57.664866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Veri Kalitesi Kontrolü \n\ndef check_df(dataframe):\n    print(\"##################### Shape #####################\")\n    print(dataframe.shape)\n    print(\"##################### Types #####################\")\n    print(dataframe.dtypes)\n    print(\"##################### Head #####################\")\n    print(dataframe.head(3))\n    print(\"##################### Tail #####################\")\n    print(dataframe.tail(3))\n    print(\"##################### NA #####################\")\n    print(dataframe.isnull().sum())\n    \n    \ncheck_df(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:17:08.602485Z","iopub.execute_input":"2025-01-08T17:17:08.602895Z","iopub.status.idle":"2025-01-08T17:17:09.970531Z","shell.execute_reply.started":"2025-01-08T17:17:08.602866Z","shell.execute_reply":"2025-01-08T17:17:09.969071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Numerik ve Kategorik Değişken Analizi\n\ndef grab_col_names(dataframe, cat_th=10, car_th=20):\n    cat_cols = [col for col in dataframe.columns if dataframe[col].dtypes == \"O\"]\n\n    num_but_cat = [col for col in dataframe.columns if dataframe[col].nunique() < cat_th and\n                   dataframe[col].dtypes != \"O\"]\n\n    cat_but_car = [col for col in dataframe.columns if dataframe[col].nunique() > car_th and\n                   dataframe[col].dtypes == \"O\"]\n\n    cat_cols = cat_cols + num_but_cat\n    cat_cols = [col for col in cat_cols if col not in cat_but_car]\n\n    num_cols = [col for col in dataframe.columns if dataframe[col].dtypes != \"O\"]\n    num_cols = [col for col in num_cols if col not in num_but_cat]\n    num_cols = [col for col in num_cols if col not in ['Premium Amount', 'id']]\n\n    print(f\"Observations: {dataframe.shape[0]}\")\n    print(f\"Variables: {dataframe.shape[1]}\")\n    print(f'cat_cols: {len(cat_cols)}')\n    print(f'num_cols: {len(num_cols)}')\n    print(f'cat_but_car: {len(cat_but_car)}')\n    print(f'num_but_cat: {len(num_but_cat)}')\n\n\n    return cat_cols, cat_but_car, num_cols\n\ncat_cols, cat_but_car, num_cols = grab_col_names(df)\n\n# grab_col_names fonksiyonu ile kategorik ve numerik değişkenleri belirledik \n# hedef değişken ve id yi etkilenmemesi için çıkardık","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:17:33.054285Z","iopub.execute_input":"2025-01-08T17:17:33.054689Z","iopub.status.idle":"2025-01-08T17:17:37.043488Z","shell.execute_reply.started":"2025-01-08T17:17:33.054658Z","shell.execute_reply":"2025-01-08T17:17:37.042142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\ndf[num_cols] = scaler.fit_transform(df[num_cols])\n\ndf['Premium Amount'] = np.log1p(df['Premium Amount'])\n\n# standartlaşma ve logaritmik dönüşüm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:17:51.199327Z","iopub.execute_input":"2025-01-08T17:17:51.199756Z","iopub.status.idle":"2025-01-08T17:17:52.804651Z","shell.execute_reply.started":"2025-01-08T17:17:51.199719Z","shell.execute_reply":"2025-01-08T17:17:52.80353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kategorik Değişken Analizi (Analysis of Categorical Variables)\n\ndef cat_summary(dataframe,col_name):\n  print(pd.DataFrame({col_name:dataframe[col_name].value_counts(),\n                      'Ratio':100*dataframe[col_name].value_counts()/len(dataframe)}))\n  print('#################################################################')\n\n# burada cat_summary adında fonksiyon a dataframe ve col_name parametleri verdik\n# ve fonksiyon içinde dataframe oluşturup col adlarını ve özelliklerini aldık\n\n# aynı zamanda ratio ile değişkenin verisetindeki yüzdesini görebiliriz\n\n\nfor col in cat_cols:\n  cat_summary(df,col)\n\n# for döngüsü kullanarak oluşturduğumuz cat_cols içerisinde gezindik\n# ve tek tek işlem yaptık\n\ndef cat_summary(dataframe,col_name,plot=False):\n  print(pd.DataFrame({col_name:dataframe[col_name].value_counts(),\n                      'Ratio':100*dataframe[col_name].value_counts()/len(dataframe)}))\n  print('#################################################################')\n\n# burada bir fark ile kategorik değişkenlerimizi görselleştirdik\n\n# grafik kullabilmek için plot parametresi ekledik\n\n  if plot:\n    sns.countplot(data=dataframe,x=dataframe[col_name])\n    plt.show(block=True)\n\n# grafikleştirebilmek için seaborn ve matplotlib kullandık\n\n\nfor col in cat_cols:\n  cat_summary(df,col,plot=True)\n\n# for döngüsü ile cat_cols içerisinde gezindik\n\n# kategorik değişkenlerimizi görselleştirdik\n\nfor col in cat_cols:\n    if df[col].dtypes ==\"bool\":\n        df[col] = df[col].astype(int)\n        cat_summary(df,col,plot=True)\n    else:\n        cat_summary(df,col,plot=True)\n\n# burada ise cat_cols içerisinde gezin ve tipi bool ise int e çevir dedik","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:18:08.350824Z","iopub.execute_input":"2025-01-08T17:18:08.351229Z","iopub.status.idle":"2025-01-08T17:19:24.464057Z","shell.execute_reply.started":"2025-01-08T17:18:08.351191Z","shell.execute_reply":"2025-01-08T17:19:24.462824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Numerik Değişken Analizi\n\ndef num_summary(dataframe,numerical_col):\n    quantiles = [0.05, 0.10, 0.20, 0.30, 0.40, 0.50, 0.60, 0.70, 0.80, 0.90, 0.95, 0.99]\n    print(dataframe[numerical_col].describe(quantiles).T)\n    print(\"################################################\")\n\n\nnum_summary(df,\"Premium Amount\")\n\nfor col in num_cols:\n    num_summary(df,col)\n\ndef num_summary(dataframe,numerical_col,plot=False):\n    quantiles = [0.05, 0.10, 0.20, 0.30, 0.40, 0.50, 0.60, 0.70, 0.80, 0.90, 0.95, 0.99]\n    print(dataframe[numerical_col].describe(quantiles).T)\n    print(\"################################################\")\n\n    if plot:\n      sns.histplot(data=dataframe, x=numerical_col)\n      plt.show(block=True)\n\nfor col in num_cols:\n  num_summary(df,col,plot=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:20:28.587302Z","iopub.execute_input":"2025-01-08T17:20:28.587769Z","iopub.status.idle":"2025-01-08T17:20:45.594091Z","shell.execute_reply.started":"2025-01-08T17:20:28.587733Z","shell.execute_reply":"2025-01-08T17:20:45.592725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hedef Değişkenin kategorik değişkenler ile   analizi\n\n\ndef target_summary_with_cat(dataframe,target,categorical_col):\n  print(pd.DataFrame({'TARGET_MEAN':dataframe.groupby(categorical_col,observed=True)[target].mean()}), end=\"\\n\\n\\n\")\n\nfor col in cat_cols:\n  target_summary_with_cat(df,'Premium Amount',col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:21:18.510568Z","iopub.execute_input":"2025-01-08T17:21:18.510996Z","iopub.status.idle":"2025-01-08T17:21:20.674344Z","shell.execute_reply.started":"2025-01-08T17:21:18.510968Z","shell.execute_reply":"2025-01-08T17:21:20.673252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hedef Değişkenin numerik değişkenler ile   analizi\n\ndef target_summary_with_num(dataframe,target,numerical_col):\n  print(dataframe.groupby(target).agg({numerical_col:'mean'}), end=\"\\n\\n\\n\")\n\nfor col in num_cols:\n  target_summary_with_num(df,'Premium Amount',col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:21:38.481288Z","iopub.execute_input":"2025-01-08T17:21:38.481721Z","iopub.status.idle":"2025-01-08T17:21:38.960216Z","shell.execute_reply.started":"2025-01-08T17:21:38.48167Z","shell.execute_reply":"2025-01-08T17:21:38.958883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRANSFORMATION\n\n# Bağımlı değişkenin incelenmesi (Hedef Değişken)\n\ndf['Premium Amount'].hist(bins=100)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:21:50.866334Z","iopub.execute_input":"2025-01-08T17:21:50.866753Z","iopub.status.idle":"2025-01-08T17:21:51.252794Z","shell.execute_reply.started":"2025-01-08T17:21:50.866719Z","shell.execute_reply":"2025-01-08T17:21:51.251689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bağımlı değişkenin logaritmasının incelenmesi\nnp.log1p(df['Premium Amount']).hist(bins=50)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:21:59.986406Z","iopub.execute_input":"2025-01-08T17:21:59.986852Z","iopub.status.idle":"2025-01-08T17:22:00.268244Z","shell.execute_reply.started":"2025-01-08T17:21:59.986817Z","shell.execute_reply":"2025-01-08T17:22:00.267028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Feature Engineering","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Eksik Veri Analizi\n\ndf.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:22:15.320583Z","iopub.execute_input":"2025-01-08T17:22:15.321017Z","iopub.status.idle":"2025-01-08T17:22:16.744369Z","shell.execute_reply.started":"2025-01-08T17:22:15.320986Z","shell.execute_reply":"2025-01-08T17:22:16.743019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def missing_values_table(dataframe, na_name=False):\n    na_columns = [col for col in dataframe.columns if dataframe[col].isnull().sum() > 0]\n\n    n_miss = dataframe[na_columns].isnull().sum().sort_values(ascending=False)\n\n    ratio = (dataframe[na_columns].isnull().sum() / dataframe.shape[0] * 100).sort_values(ascending=False)\n\n    missing_df = pd.concat([n_miss, np.round(ratio, 2)], axis=1, keys=['n_miss', 'ratio'])\n\n    print(missing_df, end=\"\\n\")\n\n    if na_name:\n        return na_columns\n\nmissing_values_table(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:22:27.460688Z","iopub.execute_input":"2025-01-08T17:22:27.461151Z","iopub.status.idle":"2025-01-08T17:22:29.677979Z","shell.execute_reply.started":"2025-01-08T17:22:27.461116Z","shell.execute_reply":"2025-01-08T17:22:29.676814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quick_missing_imp(data, num_method=\"median\", cat_length=20, target=\"Premium Amount\"):\n    variables_with_na = [col for col in data.columns if data[col].isnull().sum() > 0]  # Eksik değere sahip olan değişkenler listelenir\n\n    temp_target = data[target]\n\n    print(\"# BEFORE\")\n    print(data[variables_with_na].isnull().sum(), \"\\n\\n\")  # Uygulama öncesi değişkenlerin eksik değerlerinin sayısı\n\n    # değişken object ve sınıf sayısı cat_lengthe eşit veya altındaysa boş değerleri mode ile doldur\n    data = data.apply(lambda x: x.fillna(x.mode()[0]) if (x.dtype == \"O\" and len(x.unique()) <= cat_length) else x, axis=0)\n\n    # num_method mean ise tipi object olmayan değişkenlerin boş değerleri ortalama ile dolduruluyor\n    if num_method == \"mean\":\n        data = data.apply(lambda x: x.fillna(x.mean()) if x.dtype != \"O\" else x, axis=0)\n    # num_method median ise tipi object olmayan değişkenlerin boş değerleri ortalama ile dolduruluyor\n    elif num_method == \"median\":\n        data = data.apply(lambda x: x.fillna(x.median()) if x.dtype != \"O\" else x, axis=0)\n\n    data[target] = temp_target\n\n    print(\"# AFTER \\n Imputation method is 'MODE' for categorical variables!\")\n    print(\" Imputation method is '\" + num_method.upper() + \"' for numeric variables! \\n\")\n    print(data[variables_with_na].isnull().sum(), \"\\n\\n\")\n\n    return data\n\n\ndf = quick_missing_imp(df, num_method=\"median\", cat_length=17)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:22:42.504349Z","iopub.execute_input":"2025-01-08T17:22:42.504796Z","iopub.status.idle":"2025-01-08T17:22:55.611883Z","shell.execute_reply.started":"2025-01-08T17:22:42.504764Z","shell.execute_reply":"2025-01-08T17:22:55.610501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:23:06.449638Z","iopub.execute_input":"2025-01-08T17:23:06.45006Z","iopub.status.idle":"2025-01-08T17:23:07.795894Z","shell.execute_reply.started":"2025-01-08T17:23:06.450022Z","shell.execute_reply":"2025-01-08T17:23:07.794875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aykırı Değer Analizi\n\n# Aykırı değerlerin baskılanması\ndef outlier_thresholds(dataframe, variable, low_quantile=0.10, up_quantile=0.90):\n    quantile_one = dataframe[variable].quantile(low_quantile)\n    quantile_three = dataframe[variable].quantile(up_quantile)\n    interquantile_range = quantile_three - quantile_one\n    up_limit = quantile_three + 1.5 * interquantile_range\n    low_limit = quantile_one - 1.5 * interquantile_range\n    return low_limit, up_limit\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:23:23.953382Z","iopub.execute_input":"2025-01-08T17:23:23.953796Z","iopub.status.idle":"2025-01-08T17:23:23.959997Z","shell.execute_reply.started":"2025-01-08T17:23:23.953765Z","shell.execute_reply":"2025-01-08T17:23:23.958493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aykırı değer kontrolü\ndef check_outlier(dataframe, col_name):\n    low_limit, up_limit = outlier_thresholds(dataframe, col_name)\n    if dataframe[(dataframe[col_name] > up_limit) | (dataframe[col_name] < low_limit)].any(axis=None):\n        return True\n    else:\n        return False\n\n\nfor col in num_cols:\n    if col != \"Premium Amount\":\n      print(col, check_outlier(df, col))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:23:57.321157Z","iopub.execute_input":"2025-01-08T17:23:57.321511Z","iopub.status.idle":"2025-01-08T17:23:58.221015Z","shell.execute_reply.started":"2025-01-08T17:23:57.321484Z","shell.execute_reply":"2025-01-08T17:23:58.219713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aykırı değerlerin baskılanması\ndef replace_with_thresholds(dataframe, variable):\n    low_limit, up_limit = outlier_thresholds(dataframe, variable)\n    dataframe.loc[(dataframe[variable] < low_limit), variable] = low_limit\n    dataframe.loc[(dataframe[variable] > up_limit), variable] = up_limit\n\n\nfor col in num_cols:\n    if col != \"Premium Amount\":\n        replace_with_thresholds(df,col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:24:10.394431Z","iopub.execute_input":"2025-01-08T17:24:10.39484Z","iopub.status.idle":"2025-01-08T17:24:11.35578Z","shell.execute_reply.started":"2025-01-08T17:24:10.394809Z","shell.execute_reply":"2025-01-08T17:24:11.354338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in num_cols:\n    if col != \"Premium Amoont\":\n      print(col, check_outlier(df, col))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:24:21.466819Z","iopub.execute_input":"2025-01-08T17:24:21.467287Z","iopub.status.idle":"2025-01-08T17:24:22.433954Z","shell.execute_reply.started":"2025-01-08T17:24:21.46725Z","shell.execute_reply":"2025-01-08T17:24:22.432685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label Encoding \n\nlabel_encoders = {col: LabelEncoder() for col in cat_cols}\n\n\nfor col in cat_cols:\n    le = label_encoders[col]\n    le.fit(df[col])\n    df[col] = le.transform(df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:24:30.444993Z","iopub.execute_input":"2025-01-08T17:24:30.4454Z","iopub.status.idle":"2025-01-08T17:24:36.23746Z","shell.execute_reply.started":"2025-01-08T17:24:30.44537Z","shell.execute_reply":"2025-01-08T17:24:36.236284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:24:43.27066Z","iopub.execute_input":"2025-01-08T17:24:43.271076Z","iopub.status.idle":"2025-01-08T17:24:43.293076Z","shell.execute_reply.started":"2025-01-08T17:24:43.271044Z","shell.execute_reply":"2025-01-08T17:24:43.291755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['week_of_year'] = df['week_of_year'].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:25:02.43147Z","iopub.execute_input":"2025-01-08T17:25:02.431928Z","iopub.status.idle":"2025-01-08T17:25:02.442019Z","shell.execute_reply.started":"2025-01-08T17:25:02.431891Z","shell.execute_reply":"2025-01-08T17:25:02.440821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# MODELLEME\n\nfrom lightgbm import LGBMClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:27:23.753016Z","iopub.execute_input":"2025-01-08T17:27:23.753458Z","iopub.status.idle":"2025-01-08T17:27:23.758671Z","shell.execute_reply.started":"2025-01-08T17:27:23.753427Z","shell.execute_reply":"2025-01-08T17:27:23.757129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:27:28.328161Z","iopub.execute_input":"2025-01-08T17:27:28.328566Z","iopub.status.idle":"2025-01-08T17:27:28.335665Z","shell.execute_reply.started":"2025-01-08T17:27:28.328538Z","shell.execute_reply":"2025-01-08T17:27:28.334466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = df[df['Premium Amount'].notna()].copy()\ntest_df = df[df['Premium Amount'].isna()].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:27:37.634035Z","iopub.execute_input":"2025-01-08T17:27:37.634384Z","iopub.status.idle":"2025-01-08T17:27:38.69017Z","shell.execute_reply.started":"2025-01-08T17:27:37.634349Z","shell.execute_reply":"2025-01-08T17:27:38.688825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:27:46.174334Z","iopub.execute_input":"2025-01-08T17:27:46.174719Z","iopub.status.idle":"2025-01-08T17:27:46.183005Z","shell.execute_reply.started":"2025-01-08T17:27:46.17467Z","shell.execute_reply":"2025-01-08T17:27:46.181664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.drop('Premium Amount', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:27:52.989723Z","iopub.execute_input":"2025-01-08T17:27:52.990275Z","iopub.status.idle":"2025-01-08T17:27:53.07128Z","shell.execute_reply.started":"2025-01-08T17:27:52.990228Z","shell.execute_reply":"2025-01-08T17:27:53.069981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:28:00.437683Z","iopub.execute_input":"2025-01-08T17:28:00.438137Z","iopub.status.idle":"2025-01-08T17:28:00.445826Z","shell.execute_reply.started":"2025-01-08T17:28:00.438105Z","shell.execute_reply":"2025-01-08T17:28:00.444623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.drop(['Premium Amount'], axis=1)\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:28:07.552964Z","iopub.execute_input":"2025-01-08T17:28:07.553349Z","iopub.status.idle":"2025-01-08T17:28:07.669388Z","shell.execute_reply.started":"2025-01-08T17:28:07.55331Z","shell.execute_reply":"2025-01-08T17:28:07.668244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:28:17.836936Z","iopub.execute_input":"2025-01-08T17:28:17.837294Z","iopub.status.idle":"2025-01-08T17:28:18.460136Z","shell.execute_reply.started":"2025-01-08T17:28:17.837266Z","shell.execute_reply":"2025-01-08T17:28:18.458844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_percentage_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:28:28.419936Z","iopub.execute_input":"2025-01-08T17:28:28.420375Z","iopub.status.idle":"2025-01-08T17:28:28.425845Z","shell.execute_reply.started":"2025-01-08T17:28:28.420343Z","shell.execute_reply":"2025-01-08T17:28:28.42425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define MAPE metric\ndef mape(y_true, y_pred):\n    return mean_absolute_percentage_error(y_true, y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:28:39.297931Z","iopub.execute_input":"2025-01-08T17:28:39.298368Z","iopub.status.idle":"2025-01-08T17:28:39.303758Z","shell.execute_reply.started":"2025-01-08T17:28:39.298336Z","shell.execute_reply":"2025-01-08T17:28:39.302264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LightGBM parametreleri\nlgbm_params = {\n    'num_leaves': 71,\n    'learning_rate': 0.05412467152424433,\n    'n_estimators': 595,\n    'max_depth': 12,\n    'min_data_in_leaf': 97,\n    'bagging_fraction': 0.5200288825838669,\n    'feature_fraction': 0.9881738491942492,\n    'n_jobs': -1,\n    'verbose': -1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:31:07.169403Z","iopub.execute_input":"2025-01-08T17:31:07.169826Z","iopub.status.idle":"2025-01-08T17:31:07.175515Z","shell.execute_reply.started":"2025-01-08T17:31:07.169794Z","shell.execute_reply":"2025-01-08T17:31:07.173883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_model = LGBMRegressor(**lgbm_params)\nlgbm_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:31:11.004106Z","iopub.execute_input":"2025-01-08T17:31:11.004493Z","iopub.status.idle":"2025-01-08T17:31:44.92208Z","shell.execute_reply.started":"2025-01-08T17:31:11.004464Z","shell.execute_reply":"2025-01-08T17:31:44.920939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = lgbm_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:32:19.194285Z","iopub.execute_input":"2025-01-08T17:32:19.194688Z","iopub.status.idle":"2025-01-08T17:32:21.735619Z","shell.execute_reply.started":"2025-01-08T17:32:19.194653Z","shell.execute_reply":"2025-01-08T17:32:21.734809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_mape = mean_absolute_percentage_error(y_test, y_preds)\nprint(f\"LightGBM MAPE: {lgbm_mape:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:32:38.824427Z","iopub.execute_input":"2025-01-08T17:32:38.82485Z","iopub.status.idle":"2025-01-08T17:32:38.834682Z","shell.execute_reply.started":"2025-01-08T17:32:38.824819Z","shell.execute_reply":"2025-01-08T17:32:38.833457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test Verisi Üzerinde Çalıştırma\n\ntest_preds = lgbm_model.predict(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:36:27.436474Z","iopub.execute_input":"2025-01-08T17:36:27.436928Z","iopub.status.idle":"2025-01-08T17:36:35.646184Z","shell.execute_reply.started":"2025-01-08T17:36:27.436894Z","shell.execute_reply":"2025-01-08T17:36:35.645353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission\n\nsubmission = pd.DataFrame({\n    'id': test_df['id'], \n    'Premium Amount': test_preds \n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:36:46.067834Z","iopub.execute_input":"2025-01-08T17:36:46.06828Z","iopub.status.idle":"2025-01-08T17:36:46.076075Z","shell.execute_reply.started":"2025-01-08T17:36:46.068245Z","shell.execute_reply":"2025-01-08T17:36:46.074898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_filename = \"submission_lgbm.csv\"\nsubmission.to_csv(submission_filename, index=False)\nprint(f\"Submission file saved as {submission_filename}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:36:54.836535Z","iopub.execute_input":"2025-01-08T17:36:54.836899Z","iopub.status.idle":"2025-01-08T17:36:56.436561Z","shell.execute_reply.started":"2025-01-08T17:36:54.836873Z","shell.execute_reply":"2025-01-08T17:36:56.435372Z"}},"outputs":[],"execution_count":null}]}