{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:05.872043Z","iopub.execute_input":"2026-01-15T14:44:05.872459Z","iopub.status.idle":"2026-01-15T14:44:05.881237Z","shell.execute_reply.started":"2026-01-15T14:44:05.872433Z","shell.execute_reply":"2026-01-15T14:44:05.88061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Train.csv","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:05.882574Z","iopub.execute_input":"2026-01-15T14:44:05.882889Z","iopub.status.idle":"2026-01-15T14:44:05.894995Z","shell.execute_reply.started":"2026-01-15T14:44:05.882869Z","shell.execute_reply":"2026-01-15T14:44:05.894282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:05.896348Z","iopub.execute_input":"2026-01-15T14:44:05.896933Z","iopub.status.idle":"2026-01-15T14:44:05.915311Z","shell.execute_reply.started":"2026-01-15T14:44:05.896896Z","shell.execute_reply":"2026-01-15T14:44:05.914299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:05.917171Z","iopub.execute_input":"2026-01-15T14:44:05.917606Z","iopub.status.idle":"2026-01-15T14:44:10.470142Z","shell.execute_reply.started":"2026-01-15T14:44:05.917582Z","shell.execute_reply":"2026-01-15T14:44:10.469203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:10.471282Z","iopub.execute_input":"2026-01-15T14:44:10.471628Z","iopub.status.idle":"2026-01-15T14:44:11.145548Z","shell.execute_reply.started":"2026-01-15T14:44:10.471598Z","shell.execute_reply":"2026-01-15T14:44:11.144823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **\" TASK \" **\n      1- Calculate the percentaage of missing values ?\n      2- Handle data types\n      3- Plotting the data ( EDA ) to explore potential handling techniques\n      4- Detect the outlier ","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:11.146325Z","iopub.execute_input":"2026-01-15T14:44:11.146583Z","iopub.status.idle":"2026-01-15T14:44:11.811537Z","shell.execute_reply.started":"2026-01-15T14:44:11.146564Z","shell.execute_reply":"2026-01-15T14:44:11.810724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat = df.select_dtypes(\"object\").columns\ncat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:11.812386Z","iopub.execute_input":"2026-01-15T14:44:11.812621Z","iopub.status.idle":"2026-01-15T14:44:11.961481Z","shell.execute_reply.started":"2026-01-15T14:44:11.812594Z","shell.execute_reply":"2026-01-15T14:44:11.960817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat:\n    df[i]= df[i].astype(\"category\")\n\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:11.962199Z","iopub.execute_input":"2026-01-15T14:44:11.962452Z","iopub.status.idle":"2026-01-15T14:44:13.677831Z","shell.execute_reply.started":"2026-01-15T14:44:11.962434Z","shell.execute_reply":"2026-01-15T14:44:13.677068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num = df.select_dtypes([\"int64\" , \"float64\"]).columns\nnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:13.680446Z","iopub.execute_input":"2026-01-15T14:44:13.680933Z","iopub.status.idle":"2026-01-15T14:44:13.722575Z","shell.execute_reply.started":"2026-01-15T14:44:13.680912Z","shell.execute_reply":"2026-01-15T14:44:13.72188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df.columns:\n    print(df[col].value_counts())\n    print(\"************************************+**\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:13.723226Z","iopub.execute_input":"2026-01-15T14:44:13.723477Z","iopub.status.idle":"2026-01-15T14:44:14.340924Z","shell.execute_reply.started":"2026-01-15T14:44:13.723459Z","shell.execute_reply":"2026-01-15T14:44:14.340079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.express as px","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.341778Z","iopub.execute_input":"2026-01-15T14:44:14.342074Z","iopub.status.idle":"2026-01-15T14:44:14.345946Z","shell.execute_reply.started":"2026-01-15T14:44:14.342043Z","shell.execute_reply":"2026-01-15T14:44:14.345289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.346958Z","iopub.execute_input":"2026-01-15T14:44:14.347232Z","iopub.status.idle":"2026-01-15T14:44:14.367625Z","shell.execute_reply.started":"2026-01-15T14:44:14.347208Z","shell.execute_reply":"2026-01-15T14:44:14.366743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.368677Z","iopub.execute_input":"2026-01-15T14:44:14.36898Z","iopub.status.idle":"2026-01-15T14:44:14.386218Z","shell.execute_reply.started":"2026-01-15T14:44:14.368954Z","shell.execute_reply":"2026-01-15T14:44:14.385583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"px.pie(df , names= \"Premium Amount\" , title= 'Annual Income' )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.386972Z","iopub.execute_input":"2026-01-15T14:44:14.387263Z","iopub.status.idle":"2026-01-15T14:44:14.631448Z","shell.execute_reply.started":"2026-01-15T14:44:14.387222Z","shell.execute_reply":"2026-01-15T14:44:14.630319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def null_percentage(col):\n    col_nun = df[col].isnull().sum()\n    na_per = (col_nun / df.shape[0]) * 100\n    return na_per\n\nfor i in df.columns:\n    if null_percentage(i) > 0 :\n        print(f\"{i} Null : {null_percentage(i).round(2)} % | {df[i].dtype}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.632604Z","iopub.execute_input":"2026-01-15T14:44:14.632969Z","iopub.status.idle":"2026-01-15T14:44:14.743818Z","shell.execute_reply.started":"2026-01-15T14:44:14.632935Z","shell.execute_reply":"2026-01-15T14:44:14.743071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df['Age'] = df['Age'].fillna(0)\n#df['Age'] = df['Age'].fillna(df['Age'].mean())\ndf['Age'] = df['Age'].fillna(df['Age'].median())\ndf['Marital Status'] = df['Marital Status'].fillna(df['Marital Status'].mode()[0])\n#df.dropna(subset = ['Vehicle Age'] , inplace= True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.744564Z","iopub.execute_input":"2026-01-15T14:44:14.744777Z","iopub.status.idle":"2026-01-15T14:44:14.783422Z","shell.execute_reply.started":"2026-01-15T14:44:14.74476Z","shell.execute_reply":"2026-01-15T14:44:14.782792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#px.histogram(df , x = 'Age' , color = 'Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.784108Z","iopub.execute_input":"2026-01-15T14:44:14.7843Z","iopub.status.idle":"2026-01-15T14:44:14.787827Z","shell.execute_reply.started":"2026-01-15T14:44:14.784286Z","shell.execute_reply":"2026-01-15T14:44:14.787155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outlier(col):\n    Q1 = np.quantile(col , 0.25)\n    Q3 = np.quantile(col , 0.75)\n    IQR = Q3 - Q1 \n    lower = Q1 - 1.5 * IQR\n    upper = Q3 + 1.5*IQR\n    k = df[i]\n    outliers = k[ (( k < lower ) | (k > upper )  ) ]\n    return outliers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:14.788518Z","iopub.execute_input":"2026-01-15T14:44:14.788755Z","iopub.status.idle":"2026-01-15T14:44:14.811025Z","shell.execute_reply.started":"2026-01-15T14:44:14.78873Z","shell.execute_reply":"2026-01-15T14:44:14.810462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outliers_dic = {}\nfor i in num:\n   outliers_dic[i] = detect_outlier(df[i])\n   \n   print(outliers_dic)\n   print(\"****************************************************************************\")    #???????????","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:47:36.672904Z","iopub.execute_input":"2026-01-15T14:47:36.673253Z","iopub.status.idle":"2026-01-15T14:47:37.100805Z","shell.execute_reply.started":"2026-01-15T14:47:36.673215Z","shell.execute_reply":"2026-01-15T14:47:37.099925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in num:\n    Q1 = df[i].quantile(0.25)\n    Q3 = df[i].quantile(0.75)\n    IQR = Q3 - Q1\n    \n    lower = Q1 - 1.5 * IQR\n    upper = Q3 + 1.5 * IQR\n    K = df[i]\n    outliers = K[ ((K > upper) | (K < lower))]\n    print(outliers)\n    print(\"********************************************************\")    #?????????","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T14:44:15.247084Z","iopub.execute_input":"2026-01-15T14:44:15.247433Z","iopub.status.idle":"2026-01-15T14:44:15.78672Z","shell.execute_reply.started":"2026-01-15T14:44:15.247412Z","shell.execute_reply":"2026-01-15T14:44:15.785961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}