{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = 25000000000\n# Configuration\n\nclass Config:\n    internet=True\n    seed = 42\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 20\n    is_backbone_trainable = True\n    res_x=256\n    res_y=256\n    train_size=1\n    val_size=1-train_size\n    # Training\n    train_csv_path = \"/home/mar/Downloads/train.csv\"\n    train_thumbnail_paths = \"/home/mar/Downloads/train_thumbnails\"\n    train_paths = \"/home/mar/Downloads/train_images\"\n        \n    # Inference\n    test_csv_path = \"/home/mar/Downloads/test.csv\"\n    test_thumbnail_paths = \"/home/mar/Downloads/test_thumbnails\"\n    test_paths = \"/home/mar/Downloads/test_images\"\n    \n    if internet==True:\n        # Training\n        train_csv_path = \"../input/UBC-OCEAN/train.csv\"\n        train_thumbnail_paths = \"../input/UBC-OCEAN/train_thumbnails\"\n        train_paths = \"../input/UBC-OCEAN/train_images\"\n        # Inference\n        test_csv_path = \"../input/UBC-OCEAN/test.csv\"\n        test_thumbnail_paths = \"../input/UBC-OCEAN/test_thumbnails\"\n        test_paths = \"../input/UBC-OCEAN/test_images\"\n    \nconfig = Config()\n#Reproducibility\n#keras.utils.set_random_seed(seed=config.seed)\n\n#Basic statistics\ndf = pd.read_csv(config.train_csv_path)\n\n\n# Get basic statistics about the dataset\nnum_rows = df.shape[0]\nnum_unique_images = df['image_id'].nunique()\nnum_unique_labels = df['label'].nunique()\nunique_labels = df['label'].unique()\n\n# Perform one-hot encoding\n\n\n# Perform one-hot encoding of the 'label' column and explicitly convert to integer type\ndf_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\n# Concatenate the original DataFrame with the one-hot encoded labels\ntrain_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\n# Get the thumbnail image paths\n#train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\n#train_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_paths}/{x}.png\")\n\ntrain_df.loc[df[\"is_tma\"] == False,\"image_path\"]= train_df.loc[df[\"is_tma\"] == False,\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\ntrain_df.loc[df[\"is_tma\"] == True,\"image_path\"] = train_df.loc[df[\"is_tma\"] == True,\"image_id\"].apply(lambda x: f\"{config.train_paths}/{x}.png\")\n#print(train_df.head())\nimage_paths = train_df[\"image_path\"].values\nlabels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\nlabel_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nname_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nid_to_name = {key:value for value, key in name_to_id.items()}\n#Weights class weights to normalize model¶           \nclass_weights = np.sum(labels) - np.sum(labels, axis=0)\nclass_weights = class_weights / np.sum(class_weights) # Normalize the weights\nclass_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\n\n#Creating the tf.data.Dataset pipeline\n\ndef read_image(path):        \n    image = tf.keras.utils.img_to_array(Image.open(path).resize((config.res_x,config.res_y)))    \n    image = tf.image.per_image_standardization(image)\n    image = tf.reshape(image, (3, config.res_x, config.res_y))   \n    return image\n\nx=[read_image(itp) for itp in image_paths]\nx=tf.data.Dataset.from_tensor_slices(tf.convert_to_tensor(x)[None,:,:,:])\ny=tf.data.Dataset.from_tensor_slices(np.transpose(labels))\nds = tf.data.Dataset.zip((x, y))\ntrain_size=int(config.train_size*len(df)/config.batch_size)*config.batch_size\nval_size=int(config.val_size*len(df)/config.batch_size)*config.batch_size    \ntrain_ds=ds.take(train_size)\nval_ds=ds.skip(train_size).take(val_size)\n\n# Build the model\n    \ndef build_model_with_functional():\n    input_layer=tf.keras.Input(shape=(3,config.res_x,config.res_y))\n    fl=tf.keras.layers.Flatten()(input_layer)\n    first_dense=tf.keras.layers.Dense(128,activation=tf.nn.relu)(fl)\n    output_layer=tf.keras.layers.Dense(5,activation=tf.nn.softmax)(first_dense)\n    func_model=tf.keras.Model(inputs=input_layer,outputs=output_layer)\n    return func_model\nmodel=build_model_with_functional()\n#model.summary()\n     \nopt=tf.optimizers.Adam()\nopt.learning_rate.assign(config.learning_rate)\nmodel.compile(optimizer=opt,loss='sparse_categorical_crossentropy',metrics=['accuracy'])\nmodel.fit(train_ds,epochs=config.epochs,batch_size=config.batch_size)\n#results=model.evaluate(val_ds,batch_size=config.batch_size)\n#print(results)\n\n  \n    \n\ndft = pd.read_csv(config.test_csv_path)\ndft[\"image_path\"] = dft[\"image_id\"].apply(lambda x: f\"{config.test_paths}/{x}.png\")    \n        \n#Inference\npredicted_labels = []\n\nfor index, row in dft.iterrows():\n    # Get the image path\n    image_path = row[\"image_path\"]\n\n    # Get the image\n    image = read_image(image_path)[None, ...]\n\n    # Predict the label\n    logits = model.predict(image)\n    #print(logits)\n    pred = np.argmax(logits, axis=-1).tolist()[0]\n\n    # Map the pred to the name\n    label = id_to_name[pred]\n\n    predicted_labels.append(label)\n\n# Add the predicted labels to the csv\ndft[\"label\"] = predicted_labels\n#print(dft[\"label\"])\n\n# Create the submission\nsubmission_df = dft[[\"image_id\", \"label\"]]\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}