{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.utils import to_categorical","metadata":{"execution":{"iopub.status.busy":"2024-06-04T00:46:41.285715Z","iopub.execute_input":"2024-06-04T00:46:41.286199Z","iopub.status.idle":"2024-06-04T00:46:41.292705Z","shell.execute_reply.started":"2024-06-04T00:46:41.286143Z","shell.execute_reply":"2024-06-04T00:46:41.291503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images'\nlabels_path = '/kaggle/input/prostate-cancer-grade-assessment/train.csv'\nlabels_df = pd.read_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T00:46:44.013534Z","iopub.execute_input":"2024-06-04T00:46:44.014679Z","iopub.status.idle":"2024-06-04T00:46:44.039031Z","shell.execute_reply.started":"2024-06-04T00:46:44.014638Z","shell.execute_reply":"2024-06-04T00:46:44.037901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_size = 10\nsubset_df = labels_df.sample(n=subset_size, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T01:11:28.269826Z","iopub.execute_input":"2024-06-04T01:11:28.270337Z","iopub.status.idle":"2024-06-04T01:11:28.278213Z","shell.execute_reply.started":"2024-06-04T01:11:28.270298Z","shell.execute_reply":"2024-06-04T01:11:28.276914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = 128\nX = []\ny = []\n\nImage.MAX_IMAGE_PIXELS = None\n\nfor idx, row in subset_df.iterrows():\n    img_path = os.path.join(data_dir, row['image_id'] + '.tiff')\n    if os.path.exists(img_path):\n        img = Image.open(img_path)\n        img = img.resize((img_size, img_size))\n        img = np.array(img)\n        X.append(img)\n        y.append(row['isup_grade'])\n\nX = np.array(X)\ny = np.array(y)\n\nnum_classes = 6 \ny = to_categorical(y, num_classes=num_classes)\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=42)\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(img_size, img_size, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(num_classes, activation='softmax') \n])\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_val, y_val))\n\nplt.plot(history.history['accuracy'], label='accuracy')\nplt.plot(history.history['val_accuracy'], label='val_accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend(loc='lower right')\nplt.show()\n\nmodel.save('prostate_cancer_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T01:11:30.695055Z","iopub.execute_input":"2024-06-04T01:11:30.696384Z","iopub.status.idle":"2024-06-04T01:13:30.870584Z","shell.execute_reply.started":"2024-06-04T01:11:30.696336Z","shell.execute_reply":"2024-06-04T01:13:30.869387Z"},"trusted":true},"execution_count":null,"outputs":[]}]}