{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:27.196836Z","iopub.execute_input":"2023-10-20T06:58:27.198019Z","iopub.status.idle":"2023-10-20T06:58:27.207514Z","shell.execute_reply.started":"2023-10-20T06:58:27.197956Z","shell.execute_reply":"2023-10-20T06:58:27.204876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:27.222297Z","iopub.execute_input":"2023-10-20T06:58:27.2228Z","iopub.status.idle":"2023-10-20T06:58:27.248433Z","shell.execute_reply.started":"2023-10-20T06:58:27.222767Z","shell.execute_reply":"2023-10-20T06:58:27.247291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for label,color in zip(train.label.unique(),['r','b','g','orange','k']):\n    train_sub = train[train.label==label]\n    plt.plot(train_sub.image_width,train_sub.image_height,'.',c=color,label=label)\nplt.xlabel('Image width')\nplt.ylabel('Image height')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:27.25058Z","iopub.execute_input":"2023-10-20T06:58:27.250891Z","iopub.status.idle":"2023-10-20T06:58:27.618581Z","shell.execute_reply.started":"2023-10-20T06:58:27.250863Z","shell.execute_reply":"2023-10-20T06:58:27.617178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_names = []\nchannels = []\n\nfor file_name in os.listdir('/kaggle/input/UBC-OCEAN/train_images'):\n    file_path = os.path.join('/kaggle/input/UBC-OCEAN/train_images',file_name)\n    img = cv2.imread(file_path)\n    channels_mean = img.mean(axis=(0,1))\n    channels_std = img.std(axis=(0,1))\n    \n    file_names.append(file_name)\n    channels.append(list(channels_mean)+list(channels_std))","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:27.620445Z","iopub.execute_input":"2023-10-20T06:58:27.620761Z","iopub.status.idle":"2023-10-20T06:58:39.291003Z","shell.execute_reply.started":"2023-10-20T06:58:27.620734Z","shell.execute_reply":"2023-10-20T06:58:39.289968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(np.array(channels), columns=['channel_mean_1','channel_mean_2','channel_mean_3','channel_std_1','channel_std_2','channel_std_3'])\ndf['file_name'] = file_names\ndf['image_id'] = df.file_name.apply(lambda x: int(x.split('_')[0]))\ntrain = pd.merge(df,train,on='image_id',how='inner')\ntrain.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:39.293837Z","iopub.execute_input":"2023-10-20T06:58:39.295058Z","iopub.status.idle":"2023-10-20T06:58:39.320665Z","shell.execute_reply.started":"2023-10-20T06:58:39.295017Z","shell.execute_reply":"2023-10-20T06:58:39.319037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_csv('train_extended.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nplt.subplot(2,2,1)\nfor label,color in zip(train.label.unique(),['r','b','g','orange','k']):\n    train_sub = train[train.label==label]\n    plt.plot(train_sub.channel_mean_1,train_sub.channel_mean_2,'.',c=color,label=label)\nplt.xlim(40,200)\nplt.ylim(40,200)\nplt.xlabel('Mean of the first channel')\nplt.ylabel('Mean of the second channel')\nplt.legend();\n\nplt.subplot(2,2,2)\nfor label,color in zip(train.label.unique(),['r','b','g','orange','k']):\n    train_sub = train[train.label==label]\n    plt.plot(train_sub.channel_mean_1,train_sub.channel_mean_3,'.',c=color,label=label)\nplt.xlim(40,200)\nplt.ylim(40,200)\nplt.xlabel('Mean of the first channel')\nplt.ylabel('Mean of the third channel')\nplt.legend();\n\nplt.subplot(2,2,3)\nfor label,color in zip(train.label.unique(),['r','b','g','orange','k']):\n    train_sub = train[train.label==label]\n    plt.plot(train_sub.channel_std_1,train_sub.channel_std_2,'.',c=color,label=label)\n#plt.xlim(40,200)\n#plt.ylim(40,200)\nplt.xlabel('Std of the first channel')\nplt.ylabel('Std of the second channel')\nplt.legend();\n\nplt.subplot(2,2,4)\nfor label,color in zip(train.label.unique(),['r','b','g','orange','k']):\n    train_sub = train[train.label==label]\n    plt.plot(train_sub.channel_std_1,train_sub.channel_std_3,'.',c=color,label=label)\n#plt.xlim(40,200)\n#plt.ylim(40,200)\nplt.xlabel('Std of the first channel')\nplt.ylabel('Std of the third channel')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2023-10-20T06:58:39.322781Z","iopub.execute_input":"2023-10-20T06:58:39.323251Z","iopub.status.idle":"2023-10-20T06:58:40.339921Z","shell.execute_reply.started":"2023-10-20T06:58:39.323215Z","shell.execute_reply":"2023-10-20T06:58:40.338278Z"},"trusted":true},"execution_count":null,"outputs":[]}]}