{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-10T14:07:35.064846Z","iopub.execute_input":"2022-07-10T14:07:35.06523Z","iopub.status.idle":"2022-07-10T14:07:35.090391Z","shell.execute_reply.started":"2022-07-10T14:07:35.065194Z","shell.execute_reply":"2022-07-10T14:07:35.089121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset=pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:07:35.092153Z","iopub.execute_input":"2022-07-10T14:07:35.092517Z","iopub.status.idle":"2022-07-10T14:07:35.100613Z","shell.execute_reply.started":"2022-07-10T14:07:35.092484Z","shell.execute_reply":"2022-07-10T14:07:35.09944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:07:35.242715Z","iopub.execute_input":"2022-07-10T14:07:35.243759Z","iopub.status.idle":"2022-07-10T14:07:35.261263Z","shell.execute_reply.started":"2022-07-10T14:07:35.24372Z","shell.execute_reply":"2022-07-10T14:07:35.260346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:07:35.476371Z","iopub.execute_input":"2022-07-10T14:07:35.47711Z","iopub.status.idle":"2022-07-10T14:07:35.492933Z","shell.execute_reply.started":"2022-07-10T14:07:35.477063Z","shell.execute_reply":"2022-07-10T14:07:35.491819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset=pd.concat([text_dataset,pd.get_dummies(text_dataset['label'])], axis=1)\nfrom sklearn.preprocessing import LabelEncoder\nbi=LabelEncoder()\npoint=bi.fit_transform(text_dataset['label'])\ndataset=dataset.drop(['label', 'CE', 'LAA'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:08:06.770834Z","iopub.execute_input":"2022-07-10T15:08:06.771349Z","iopub.status.idle":"2022-07-10T15:08:06.78569Z","shell.execute_reply.started":"2022-07-10T15:08:06.77123Z","shell.execute_reply":"2022-07-10T15:08:06.784141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset=dataset.assign(label=point)\ndataset","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:08:09.086714Z","iopub.execute_input":"2022-07-10T15:08:09.088227Z","iopub.status.idle":"2022-07-10T15:08:09.109167Z","shell.execute_reply.started":"2022-07-10T15:08:09.088158Z","shell.execute_reply":"2022-07-10T15:08:09.107558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.title('Labels')\nplt.pie(text_dataset['label'].value_counts(), labels=text_dataset['label'].unique())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:07:36.154685Z","iopub.execute_input":"2022-07-10T14:07:36.155091Z","iopub.status.idle":"2022-07-10T14:07:36.392972Z","shell.execute_reply.started":"2022-07-10T14:07:36.155058Z","shell.execute_reply":"2022-07-10T14:07:36.391546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(text_dataset['center_id'], kind='kde')\nsns.pairplot(text_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:05:45.892159Z","iopub.execute_input":"2022-07-10T15:05:45.892717Z","iopub.status.idle":"2022-07-10T15:05:47.093286Z","shell.execute_reply.started":"2022-07-10T15:05:45.892665Z","shell.execute_reply":"2022-07-10T15:05:47.091917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax=[]\nfor i in range(1, 12):\n    ax.append('ax'+str(i))\nfig, ax=plt.subplots(1,5, sharey=True)\nfor i in range(0, 5):\n    ax[i].set_title('center_id'+str(i+1))\n    ax[i].bar(text_dataset[text_dataset['center_id']==i+1]['label'].unique(), text_dataset[text_dataset['center_id']==i+1]['label'].value_counts())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:06:18.576468Z","iopub.execute_input":"2022-07-10T15:06:18.576894Z","iopub.status.idle":"2022-07-10T15:06:18.961312Z","shell.execute_reply.started":"2022-07-10T15:06:18.576845Z","shell.execute_reply":"2022-07-10T15:06:18.960082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(text_dataset['center_id'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:09:06.358991Z","iopub.execute_input":"2022-07-10T15:09:06.359376Z","iopub.status.idle":"2022-07-10T15:09:06.614967Z","shell.execute_reply.started":"2022-07-10T15:09:06.359346Z","shell.execute_reply":"2022-07-10T15:09:06.61401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(text_dataset['label'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T15:19:45.142624Z","iopub.execute_input":"2022-07-10T15:19:45.143102Z","iopub.status.idle":"2022-07-10T15:19:45.327937Z","shell.execute_reply.started":"2022-07-10T15:19:45.143066Z","shell.execute_reply":"2022-07-10T15:19:45.326918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}