{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:44:10.864367Z","iopub.execute_input":"2024-12-23T09:44:10.864822Z","iopub.status.idle":"2024-12-23T09:44:11.309687Z","shell.execute_reply.started":"2024-12-23T09:44:10.864783Z","shell.execute_reply":"2024-12-23T09:44:11.3084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df= pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:45:03.621308Z","iopub.execute_input":"2024-12-23T09:45:03.621707Z","iopub.status.idle":"2024-12-23T09:45:10.736069Z","shell.execute_reply.started":"2024-12-23T09:45:03.621675Z","shell.execute_reply":"2024-12-23T09:45:10.735031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:45:17.134758Z","iopub.execute_input":"2024-12-23T09:45:17.135177Z","iopub.status.idle":"2024-12-23T09:45:17.175159Z","shell.execute_reply.started":"2024-12-23T09:45:17.135143Z","shell.execute_reply":"2024-12-23T09:45:17.174167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:45:25.593319Z","iopub.execute_input":"2024-12-23T09:45:25.593647Z","iopub.status.idle":"2024-12-23T09:45:26.236842Z","shell.execute_reply.started":"2024-12-23T09:45:25.593621Z","shell.execute_reply":"2024-12-23T09:45:26.235915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Their are **9 float64** attributes, **11 object** attributes and **1 int64** attribute. ","metadata":{}},{"cell_type":"code","source":"#Defining a function to store the ATTRIBUTES as per their Data type\ndef data_type_of_attribute(df):\n    cat=[]\n    num=[]\n    for i in df.columns:\n        if df[i].dtypes == 'object':\n            cat.append(i)\n        else:\n            num.append(i)\n    return cat,num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:51:02.848698Z","iopub.execute_input":"2024-12-23T09:51:02.84906Z","iopub.status.idle":"2024-12-23T09:51:02.854609Z","shell.execute_reply.started":"2024-12-23T09:51:02.849036Z","shell.execute_reply":"2024-12-23T09:51:02.85318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat, df_num = data_type_of_attribute(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:51:31.93775Z","iopub.execute_input":"2024-12-23T09:51:31.938131Z","iopub.status.idle":"2024-12-23T09:51:31.942757Z","shell.execute_reply.started":"2024-12-23T09:51:31.938105Z","shell.execute_reply":"2024-12-23T09:51:31.941575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Value Count for Categorical attributes\nfor i in df_cat:\n    print(df[i].value_counts(),'\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:53:49.693216Z","iopub.execute_input":"2024-12-23T09:53:49.69354Z","iopub.status.idle":"2024-12-23T09:53:51.099029Z","shell.execute_reply.started":"2024-12-23T09:53:49.693508Z","shell.execute_reply":"2024-12-23T09:53:51.09778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Insights from Value Count of Categorical Attribute\n1. All Categorical Attributes have a **balanced** number of **observations** for each **category**.\n2. The attribute **Policy Start Date** do not contribute in any type of category it only consist of **date & time**.","metadata":{}},{"cell_type":"code","source":"#Summary of Numerical Attributes of data\npd.options.display.float_format = '{:.2f}'.format #to display values in float format\ndf.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T09:59:02.036221Z","iopub.execute_input":"2024-12-23T09:59:02.036573Z","iopub.status.idle":"2024-12-23T09:59:02.76914Z","shell.execute_reply.started":"2024-12-23T09:59:02.036548Z","shell.execute_reply":"2024-12-23T09:59:02.76775Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary of Numerical Atrributes of Data\n1. The total data consist of **1200000** observation in each column(neglect null/NaN Values).\n2. The 25% of observations in **Premium Amount** lies under **514.00** value.\n3. And the 75% of observations in **Premium Amount** lies under **1509.00** value.\n4. The point to be note from **Premium Amount** attribute is that their is a huge difference in**75% Value** and **Maximum value 4999.00**.\n5. It shows that the data consist of outliers in its **Target Attribute**.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:21:33.697361Z","iopub.execute_input":"2024-12-23T10:21:33.697745Z","iopub.status.idle":"2024-12-23T10:21:33.70263Z","shell.execute_reply.started":"2024-12-23T10:21:33.697711Z","shell.execute_reply":"2024-12-23T10:21:33.701191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.hist(bins=50, figsize=(10,12))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:27:18.19383Z","iopub.execute_input":"2024-12-23T10:27:18.194227Z","iopub.status.idle":"2024-12-23T10:27:20.728561Z","shell.execute_reply.started":"2024-12-23T10:27:18.194194Z","shell.execute_reply":"2024-12-23T10:27:20.727486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Number of instances on vertical axis.\n* Range of values on horizontal axis","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}