{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns \n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer  # Eksik değerleri doldurmak için\nfrom sklearn.experimental import enable_iterative_imputer  # Enable IterativeImputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.preprocessing import StandardScaler  # Sayısal veriyi ölçeklemek için\nfrom sklearn.preprocessing import OneHotEncoder  # Kategorik veriyi kodlamak için\nfrom sklearn.compose import ColumnTransformer\n\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:18.118449Z","iopub.execute_input":"2024-12-10T11:43:18.118896Z","iopub.status.idle":"2024-12-10T11:43:19.43933Z","shell.execute_reply.started":"2024-12-10T11:43:18.11885Z","shell.execute_reply":"2024-12-10T11:43:19.438408Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n> ### ***I would like to mention that I am newbie for this field, I am still learning.***\n> ### ***I prepared this notebook with the help of other developer's code and ai.***\n> ###  ***Please feel free to comment and share your suggestions.***\n","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\ntrain = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntrain_df = train.copy()\ntest_df = test.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:19.44083Z","iopub.execute_input":"2024-12-10T11:43:19.441231Z","iopub.status.idle":"2024-12-10T11:43:28.263674Z","shell.execute_reply.started":"2024-12-10T11:43:19.441205Z","shell.execute_reply":"2024-12-10T11:43:28.262961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA)\nExploratory Data Analysis (EDA) is a critical step in any data project. It allows us to understand, summarize, and visualize the dataset effectively, paving the way for further analysis or modeling.\n\n***🎯 Objectives***  \n🕵️‍♀️ Gain insights into the data.  \n📊 Visualize distributions, relationships, and patterns.  \n🧹 Identify missing values, outliers, and data inconsistencies.etection.  \n\n**Let’s dive into the EDA! 🚀**\n\n","metadata":{}},{"cell_type":"markdown","source":"## 📜 Dataset Overview\n","metadata":{}},{"cell_type":"code","source":"# Check datasets dimensions \nprint(f\"Train Dataset contains {train_df.shape[0]} rows and {train_df.shape[1]} columns.\")\nprint(f\"\\nTest Dataset contains {test_df.shape[0]} rows and {test_df.shape[1]} columns.\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:28.264639Z","iopub.execute_input":"2024-12-10T11:43:28.264898Z","iopub.status.idle":"2024-12-10T11:43:28.298904Z","shell.execute_reply.started":"2024-12-10T11:43:28.264873Z","shell.execute_reply":"2024-12-10T11:43:28.298164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:28.300562Z","iopub.execute_input":"2024-12-10T11:43:28.300802Z","iopub.status.idle":"2024-12-10T11:43:28.857454Z","shell.execute_reply.started":"2024-12-10T11:43:28.300778Z","shell.execute_reply":"2024-12-10T11:43:28.856609Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Checking Missing values**","metadata":{}},{"cell_type":"code","source":"print(f\"---Train data missing values: \\n{train_df.isnull().sum()*100 / len(train_df)}\") # percentage of missing values\nprint(\"--\"*50)\nprint(f\"---Test data missing values: \\n{test_df.isnull().sum()*100 / len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:28.858728Z","iopub.execute_input":"2024-12-10T11:43:28.859104Z","iopub.status.idle":"2024-12-10T11:43:29.744655Z","shell.execute_reply.started":"2024-12-10T11:43:28.859064Z","shell.execute_reply":"2024-12-10T11:43:29.743921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧹 Data Cleaning Insights\nplt.figure(figsize=(10,7))\nplt.title(\"Visualizing Missing Values\")\nsns.heatmap(train_df.isnull(), cbar=False, cmap=sns.color_palette('magma'), yticklabels=False);\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:29.745716Z","iopub.execute_input":"2024-12-10T11:43:29.74607Z","iopub.status.idle":"2024-12-10T11:43:50.043988Z","shell.execute_reply.started":"2024-12-10T11:43:29.746037Z","shell.execute_reply":"2024-12-10T11:43:50.04306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":">**From the above, we see that Occupation and Previous Claims are the features with the highest percentage of missing values.  \n> Next, we check for potential duplicates.**","metadata":{}},{"cell_type":"code","source":"#checking for potential duplicates\nprint(f\"There are {sum(train_df.duplicated())} duplicated rows in the train data frame.\")\nprint(f\"After dropping the 'Premium Amount' column, there are {sum(train_df.drop(columns=['Premium Amount']).duplicated())} duplicated rows in the train data frame.\")\nprint(f\"There are {sum(test_df.duplicated())} duplicated rows in the test data frame.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:50.044961Z","iopub.execute_input":"2024-12-10T11:43:50.045214Z","iopub.status.idle":"2024-12-10T11:43:54.019246Z","shell.execute_reply.started":"2024-12-10T11:43:50.045189Z","shell.execute_reply":"2024-12-10T11:43:54.018449Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Finally, we check if there are any observations that appear in both the train and test data frames.**","metadata":{}},{"cell_type":"code","source":"temp_train = train_df.drop(columns=['Premium Amount'], axis=1)\ntemp_test = test_df\n\ninner_join = pd.merge(temp_train, temp_test)\nprint(f\"There are {len(inner_join)} observations that appear in both the train and test data frames\")\n#used to identify and count the overlapping observations (e.g., customers or product information) between the train and test DataFrames.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:54.02073Z","iopub.execute_input":"2024-12-10T11:43:54.02098Z","iopub.status.idle":"2024-12-10T11:43:57.290196Z","shell.execute_reply.started":"2024-12-10T11:43:54.020955Z","shell.execute_reply":"2024-12-10T11:43:57.289282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select categorical and numerical columns (initial)\ntarget_column = \"Premium Amount\"\ncategorical_columns = train_df.select_dtypes(include=['object']).columns\nnumerical_columns = train_df.select_dtypes(exclude=['object']).columns\n\n\n# Print out column information\nprint(\"Target Column:\", target_column)\nprint(\"\\nCategorical Columns:\\n\",categorical_columns.tolist())\nprint(\"\\nNumerical Columns:\\n\",numerical_columns.tolist())\nnumerical_columns= numerical_columns.tolist()\ncategorical_columns = categorical_columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:57.291164Z","iopub.execute_input":"2024-12-10T11:43:57.291438Z","iopub.status.idle":"2024-12-10T11:43:57.475003Z","shell.execute_reply.started":"2024-12-10T11:43:57.291413Z","shell.execute_reply":"2024-12-10T11:43:57.474169Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📊 Descriptive Statistics","metadata":{}},{"cell_type":"code","source":"train_df.describe().round(2).T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:57.477531Z","iopub.execute_input":"2024-12-10T11:43:57.477781Z","iopub.status.idle":"2024-12-10T11:43:58.038181Z","shell.execute_reply.started":"2024-12-10T11:43:57.477756Z","shell.execute_reply":"2024-12-10T11:43:58.037314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in categorical_columns:\n    num_unique = train_df[column].nunique()\n    print(f\"'{column}' has {num_unique} unique categories.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:58.03918Z","iopub.execute_input":"2024-12-10T11:43:58.039516Z","iopub.status.idle":"2024-12-10T11:43:58.760415Z","shell.execute_reply.started":"2024-12-10T11:43:58.039479Z","shell.execute_reply":"2024-12-10T11:43:58.759485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print top 5 unique value counts for each categorical column\nfor column in categorical_columns:\n    print(f\"\\nTop value counts in '{column}':\\n{train_df[column].value_counts().head(5)}\")","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:58.761426Z","iopub.execute_input":"2024-12-10T11:43:58.761683Z","iopub.status.idle":"2024-12-10T11:43:59.758964Z","shell.execute_reply.started":"2024-12-10T11:43:58.761657Z","shell.execute_reply":"2024-12-10T11:43:59.758006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"The mean of columns:\")\nprint(train_df[numerical_columns].mean())\n\nprint(\"\\nThe std dev of columns:\")\nprint(train_df[numerical_columns].std())\n\nprint(\"\\nThe skewness of columns:\")\nprint(train_df[numerical_columns].skew().round(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:43:59.760254Z","iopub.execute_input":"2024-12-10T11:43:59.760628Z","iopub.status.idle":"2024-12-10T11:44:00.233936Z","shell.execute_reply.started":"2024-12-10T11:43:59.760588Z","shell.execute_reply":"2024-12-10T11:44:00.232917Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n**Verilerden Çıkarılabilecek Sonuçlar:**\n- ***Gelir ve Sigorta Primi Değişkenliği:*** Yıllık gelir ve sigorta primi ücretleri hem yüksek standart sapmaya hem de pozitif çarpıklığa sahip, bu da birkaç yüksek değer nedeniyle dağılımın dengesiz olduğunu gösterir.\n- ***Yaş Verileri:*** Yaş dağılımı oldukça dengeli ve standart sapması da nispeten düşük. Çarpıklığı(-0,20) oldukça düşük yani dengeli.\n- ***Önceki Talepler:*** Çoğu kişi az talepte bulunmuş, ancak az sayıda kişi çok fazla talepte bulunmuş olabilir (pozitif çarpıklık).\n","metadata":{}},{"cell_type":"markdown","source":"## 🖼️ Visual Exploration\n","metadata":{}},{"cell_type":"code","source":"# First we start exploration with the \"Premium amount\" (target) column\nsns.histplot(x=train_df[\"Premium Amount\"],kde=True, line_kws={\"linewidth\" : 3})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:00.235285Z","iopub.execute_input":"2024-12-10T11:44:00.236338Z","iopub.status.idle":"2024-12-10T11:44:05.046915Z","shell.execute_reply.started":"2024-12-10T11:44:00.236288Z","shell.execute_reply":"2024-12-10T11:44:05.046079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above, we see that the distribution is tri-modal and right-skewed.  \nNext, we explore potential relationships between the input features and Premium Amount","metadata":{}},{"cell_type":"markdown","source":"#### **Numerical Feature Analysis**","metadata":{}},{"cell_type":"code","source":"numerical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:05.048053Z","iopub.execute_input":"2024-12-10T11:44:05.048421Z","iopub.status.idle":"2024-12-10T11:44:05.054029Z","shell.execute_reply.started":"2024-12-10T11:44:05.048383Z","shell.execute_reply":"2024-12-10T11:44:05.053227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We were determined the \"numerical_columns\" before\nsamp_col = [col for col in numerical_columns if col != 'id']\npalette = sns.color_palette('dark', len(numerical_columns))\n\nfig, axes = plt.subplots(len(numerical_columns)//3 +1 ,3,figsize=(20, 18))\naxes = axes.flatten()  # Subplot'ları tek boyutlu bir liste haline getir\n\n\n# histogram plot for all numeric columns\nfor i, col in enumerate(samp_col): \n    sns.histplot(x=train_df[col], kde=True, bins=40, ax=axes[i],color=palette[i])  # ax ile subplot'a ekle\n    axes[i].set_title(f'Distribution of {col} in Train Dataset')\n    axes[i].set_xlabel(col)\n\n# remove unused axes if necessary \nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])\n    \nplt.tight_layout()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:05.055055Z","iopub.execute_input":"2024-12-10T11:44:05.055424Z","iopub.status.idle":"2024-12-10T11:44:41.737554Z","shell.execute_reply.started":"2024-12-10T11:44:05.055397Z","shell.execute_reply":"2024-12-10T11:44:41.736725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = train_df[numerical_columns].corr()\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True)\nplt.title(\"Correlation Heatmap of Numerical Variables\", fontsize=16)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:41.738924Z","iopub.execute_input":"2024-12-10T11:44:41.739293Z","iopub.status.idle":"2024-12-10T11:44:42.539994Z","shell.execute_reply.started":"2024-12-10T11:44:41.739256Z","shell.execute_reply":"2024-12-10T11:44:42.539292Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### **Categorical Feature Analysis**","metadata":{}},{"cell_type":"code","source":"categorical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:42.541258Z","iopub.execute_input":"2024-12-10T11:44:42.541602Z","iopub.status.idle":"2024-12-10T11:44:42.547822Z","shell.execute_reply.started":"2024-12-10T11:44:42.541567Z","shell.execute_reply":"2024-12-10T11:44:42.54691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_columns = [col for col in categorical_columns if col != 'Policy Start Date']\npalette = sns.color_palette(\"tab10\",len(filtered_columns))\n\nfig, axes = plt.subplots(len(filtered_columns) ,2,figsize=(10,40))\n#axes = axes.flatten()  # Subplot'ları tek boyutlu bir liste haline getir\n\nfor i,col in enumerate(filtered_columns):\n    #countplot\n    sns.countplot(data=train_df,x=col, ax=axes[i,0], palette=\"viridis\")\n    axes[i, 0].set_title(f'Distribution of {col}', fontsize=14)\n    axes[i, 0].set_xlabel(col, fontsize=10)\n    axes[i, 0].set_ylabel('Count', fontsize=12)\n    \n    #boxplot\n    sns.boxplot(data=train_df, x=col, y=target_column, ax=axes[i,1] ,palette=palette)\n    axes[i, 1].set_title(f'Distribution of {col} vs Premium Amount', fontsize=14)\n    axes[i, 1].set_xlabel(col, fontsize=10)\n    axes[i, 1].set_ylabel(target_column, fontsize=12)\n    \nplt.tight_layout()  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:42.548891Z","iopub.execute_input":"2024-12-10T11:44:42.549189Z","iopub.status.idle":"2024-12-10T11:44:56.085828Z","shell.execute_reply.started":"2024-12-10T11:44:42.549159Z","shell.execute_reply":"2024-12-10T11:44:56.084952Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes.  \nBased on the different considered chart, the data seems pretty random.","metadata":{}},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"Data preprocessing is a crucial step in preparing the dataset for analysis and modeling.   \nIt ensures the data is clean, consistent, and ready for machine learning algorithms.","metadata":{}},{"cell_type":"markdown","source":"🔍 ***Objectives***\n- Handle missing values.\n- Encode categorical features.\n- Standardize or normalize numerical features.\n- Create new features or transform existing ones if necessary.","metadata":{}},{"cell_type":"markdown","source":"#### Feature Transformation: Date Handling","metadata":{}},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    #df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n\n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n\n    return df\n\n\ntrain_df= date(train_df)\ntest_df= date(test_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:56.087073Z","iopub.execute_input":"2024-12-10T11:44:56.087375Z","iopub.status.idle":"2024-12-10T11:44:57.732945Z","shell.execute_reply.started":"2024-12-10T11:44:56.087346Z","shell.execute_reply":"2024-12-10T11:44:57.731978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:57.734005Z","iopub.execute_input":"2024-12-10T11:44:57.734274Z","iopub.status.idle":"2024-12-10T11:44:58.269824Z","shell.execute_reply.started":"2024-12-10T11:44:57.734248Z","shell.execute_reply":"2024-12-10T11:44:58.268929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_processing(df):\n\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)\n    return df\n    \ntrain_df= feature_processing(train_df)\ntest_df= feature_processing(test_df)\n\ncat_cols = ['Marital Status', 'Education Level',\n 'Occupation', 'Location', 'Policy Type',\n 'Customer Feedback', 'Exercise Frequency', 'Property Type']\n\nfor col in cat_cols:\n    train_df[col] = train_df[col].astype('category')\n    test_df[col] = test_df[col].astype('category')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:58.271002Z","iopub.execute_input":"2024-12-10T11:44:58.271618Z","iopub.status.idle":"2024-12-10T11:44:59.442091Z","shell.execute_reply.started":"2024-12-10T11:44:58.271584Z","shell.execute_reply":"2024-12-10T11:44:59.441109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> In this step, we split the training dataset into:\n\n>**Features (X):** The independent variables used to predict the target.  \n>**Target (y):** The dependent variable that the model will learn to predict.","metadata":{}},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.443315Z","iopub.execute_input":"2024-12-10T11:44:59.443673Z","iopub.status.idle":"2024-12-10T11:44:59.467164Z","shell.execute_reply.started":"2024-12-10T11:44:59.443637Z","shell.execute_reply":"2024-12-10T11:44:59.466304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = train_df.drop(columns=[target_column, 'id',],axis=1)\ny = train_df[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.468168Z","iopub.execute_input":"2024-12-10T11:44:59.468413Z","iopub.status.idle":"2024-12-10T11:44:59.517663Z","shell.execute_reply.started":"2024-12-10T11:44:59.468389Z","shell.execute_reply":"2024-12-10T11:44:59.517005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp_cat_col = train_df.select_dtypes(include=['object','category']).columns\ntmp_num_col = train_df.select_dtypes(exclude=['object','category']).columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.518494Z","iopub.execute_input":"2024-12-10T11:44:59.518704Z","iopub.status.idle":"2024-12-10T11:44:59.682544Z","shell.execute_reply.started":"2024-12-10T11:44:59.518681Z","shell.execute_reply":"2024-12-10T11:44:59.681855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adjust_column(column):\n    for i in column:\n        if i not in X.columns:\n            column= column.drop(i)\n    return column\n    \n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.683521Z","iopub.execute_input":"2024-12-10T11:44:59.683768Z","iopub.status.idle":"2024-12-10T11:44:59.688079Z","shell.execute_reply.started":"2024-12-10T11:44:59.683743Z","shell.execute_reply":"2024-12-10T11:44:59.687165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_col = adjust_column(tmp_cat_col)\nnum_col= adjust_column(tmp_num_col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.688952Z","iopub.execute_input":"2024-12-10T11:44:59.689215Z","iopub.status.idle":"2024-12-10T11:44:59.698163Z","shell.execute_reply.started":"2024-12-10T11:44:59.68919Z","shell.execute_reply":"2024-12-10T11:44:59.697409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Handle Missing Values & Preprocessing Pipeline","metadata":{}},{"cell_type":"markdown","source":"In this step, we handle missing values and set up a preprocessing pipeline to prepare the data for machine learning.\n\n🔍 What We Did\nMissing Values Imputation:\n\nNumerical Features:\nReplaced missing values with the mean of the respective columns.\nCategorical Features:\nReplaced missing values with the constant value \"Unknown\".\nFeature Scaling & Encoding:\n\nNumerical Features:\nStandardized using StandardScaler to normalize the values.\nCategorical Features:\nOne-Hot Encoded to handle categorical variables as numerical inputs.\nCombined Using a ColumnTransformer:\n\nApplied preprocessing selectively to numerical and categorical features using a single, unified pipeline.","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer  # Eksik değerleri doldurmak için\nfrom sklearn.preprocessing import StandardScaler  # Sayısal veriyi ölçeklemek için\nfrom sklearn.preprocessing import OneHotEncoder  # Kategorik veriyi kodlamak için\nfrom sklearn.compose import ColumnTransformer\n\n\n# Preprocessing pipeline for numerical features\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())                       # Scale numerical features\n])\n\n# Preprocessing pipeline for categorical features\ncat_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant',fill_value='unknown')),  # Handle missing valuescons\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))                      # Encode categorical features\n])\n\n# Combine pipelines into a ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_pipeline, num_col),\n        ('cat', cat_pipeline, cat_col)\n    ]\n)\n\n# Preprocess train and test data\nX_processed = preprocessor.fit_transform(X)\ntest_processed = preprocessor.transform(test_df.drop(columns=['id']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:44:59.702562Z","iopub.execute_input":"2024-12-10T11:44:59.702798Z","iopub.status.idle":"2024-12-10T11:45:10.122189Z","shell.execute_reply.started":"2024-12-10T11:44:59.702777Z","shell.execute_reply":"2024-12-10T11:45:10.121471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=37)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:45:10.123327Z","iopub.execute_input":"2024-12-10T11:45:10.123619Z","iopub.status.idle":"2024-12-10T11:45:10.385581Z","shell.execute_reply.started":"2024-12-10T11:45:10.123584Z","shell.execute_reply":"2024-12-10T11:45:10.384879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"markdown","source":"Training the model is the core step in any machine learning pipeline. Here, we use the processed features and target variable to fit a predictive model and evaluate its performance on a validation set.","metadata":{}},{"cell_type":"markdown","source":"***🔍 Objectives***    \n\n- Train the model using the training dataset (X_train, y_train).  \n- Evaluate the model on the validation set (X_val, y_val).   \n- Optimize the model's parameters to improve its performance.   ","metadata":{}},{"cell_type":"markdown","source":"## 🔧 Hyperparameter Optimization with Optuna","metadata":{}},{"cell_type":"markdown","source":"Optuna is a powerful library for hyperparameter optimization.   \nIn this step, we use Optuna to fine-tune the hyperparameters of a LightGBM model to achieve optimal performance.","metadata":{}},{"cell_type":"code","source":"import optuna\nimport lightgbm as lgb\nfrom sklearn.datasets import fetch_california_housing\nfrom sklearn.metrics import mean_squared_log_error\n\n\n\n\n# Define Optuna optimization function\ndef objective(trial):\n    # Define parameter search space\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\":trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 200, 512),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 1e-4, 1e-1),\n        \"feature_fraction\": trial.suggest_uniform(\"feature_fraction\", 0.6, 1.0),\n        \"bagging_fraction\": trial.suggest_uniform(\"bagging_fraction\", 0.6, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 5, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 100),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 16),  # -1 means no limit\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1e-4, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1e-4, 10.0),\n        \"device_type\": \"gpu\",  # Enable GPU support\n        \"seed\" : 37\n\n    }\n\n    # Create a LightGBM dataset\n    dtrain = lgb.Dataset(X_train, label=y_train)\n    dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)\n\n    # Train LightGBM model\n    #model = lgb.train(param, dtrain, valid_sets=[dval], num_boost_round=1000, early_stopping_rounds=50, verbose_eval=False)\n\n\n    model = lgb.train(\n    param,\n    dtrain,\n    valid_sets=[dval],\n    num_boost_round=500,\n    callbacks=[lgb.early_stopping(stopping_rounds=50), lgb.log_evaluation(20)])\n    \n    # Calculate Rmsle and predict\n    y_pred = model.predict(X_val, num_iteration=model.best_iteration)\n    rmsle = np.sqrt(mean_squared_log_error(y_val, np.maximum(y_pred, 0)))\n\n\n    return rmsle\n\n# Run Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=1)","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:46:35.756557Z","iopub.execute_input":"2024-12-10T11:46:35.75738Z","iopub.status.idle":"2024-12-10T11:46:51.662252Z","shell.execute_reply.started":"2024-12-10T11:46:35.757332Z","shell.execute_reply":"2024-12-10T11:46:51.661514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# En iyi hiperparametreler ve sonuç\nprint(\"Best trial:\")\nprint(f\"  Value (RMSLE): {study.best_trial.value}\")\nprint(\"  Params: \")\nfor key, value in study.best_trial.params.items():\n    print(f\"    {key}: {value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:45:10.396184Z","iopub.status.idle":"2024-12-10T11:45:10.396492Z","shell.execute_reply.started":"2024-12-10T11:45:10.396325Z","shell.execute_reply":"2024-12-10T11:45:10.396339Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"***Best Hyperparameters***  \nAfter running the hyperparameter optimization process with Optuna, the best combination of parameters was identified. These parameters will be used to train the final LightGBM model.\n\n","metadata":{}},{"cell_type":"code","source":"# Initialize or update the best_params dictionary\n## this params gives 1.09 rmle score\nbest_params = {\n    \"boosting_type\": \"dart\",\n    \"num_leaves\": 248,\n    \"learning_rate\": 0.05815161496914569,\n    \"feature_fraction\": 0.9613540521120864,\n    \"bagging_fraction\": 0.8749803355299814,\n    \"bagging_freq\": 6,\n    \"min_data_in_leaf\": 48,\n    \"max_depth\": 5,\n    \"lambda_l1\": 4.277976438716461,\n    \"lambda_l2\": 0.0006468938326306755,\n    \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:47:36.411521Z","iopub.execute_input":"2024-12-10T11:47:36.412337Z","iopub.status.idle":"2024-12-10T11:47:36.418259Z","shell.execute_reply.started":"2024-12-10T11:47:36.412286Z","shell.execute_reply":"2024-12-10T11:47:36.417453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🚀 Train Final Model with Best Parameters\n","metadata":{}},{"cell_type":"code","source":"# Train final model with best parameters\n# best_params = study.best_params\n\nfinal_model = lgb.train(\n    best_params,\n    lgb.Dataset(X_processed, label=y),\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:47:38.670711Z","iopub.execute_input":"2024-12-10T11:47:38.671611Z","iopub.status.idle":"2024-12-10T11:47:58.403228Z","shell.execute_reply.started":"2024-12-10T11:47:38.671553Z","shell.execute_reply":"2024-12-10T11:47:58.402378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generate Predictions & Prepare Submission\n","metadata":{}},{"cell_type":"markdown","source":"Finally, we use the trained model to predict outcomes on the test set and format the results into a submission file for the competition.","metadata":{}},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = final_model.predict(test_processed, num_iteration=final_model.best_iteration)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': test_df['id'], 'Premium Amount': test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T11:48:05.224817Z","iopub.execute_input":"2024-12-10T11:48:05.225184Z","iopub.status.idle":"2024-12-10T11:48:08.452214Z","shell.execute_reply.started":"2024-12-10T11:48:05.225144Z","shell.execute_reply":"2024-12-10T11:48:08.451529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}