{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"},{"sourceId":8210320,"sourceType":"datasetVersion","datasetId":4865495}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Atmospheric Physics\n* I tried to implement a simple EDA and baseline modelling for this challanges on a small sample of dataset\n* I tried to convert time series sequences to list to make dataframe more readeable\n* For modelling i used a simple lstm model\n\n**Steps**\n1. Reducing Mem\n2. EDA\n3. Tensoflow Dense Modelling\n5. Submission","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport polars as pl\nfrom warnings import simplefilter\nimport warnings\nimport gc\npd.set_option('display.max_columns', None)\nsimplefilter(action=\"ignore\", category=pd.errors.PerformanceWarning)\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols = [\"ptend_t\",\n\"ptend_q0001\",\n\"ptend_q0002\",\n\"ptend_q0003\",\n\"ptend_u\",\n\"ptend_v\",\n\"cam_out_NETSW\",\n\"cam_out_FLWDS\",\n\"cam_out_PRECSC\",\n\"cam_out_PRECC\",\n\"cam_out_SOLS\",\n\"cam_out_SOLL\",\n\"cam_out_SOLSD\",\n\"cam_out_SOLLD\"]\n\nmy_random_seed = 111","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_cols = {\n    'state_t': 'air_temperature',\n    'state_q0001': 'specific_humidity',\n    'state_q0002': 'cloud_liquid mixing_ratio',\n    'state_q0003': 'cloud_ice_mixing_ratio',\n    'state_u': 'zonal_wind_speed',\n    'state_v': 'meridional_wind_speed',\n    'state_ps': 'surface_pressure',\n    'pbuf_SOLIN': 'solar_insolation',\n    'pbuf_LHFLX': 'surface_latent_heat_flux',\n    'pbuf_SHFLX': 'surface_sensible_heat_flux',\n    'pbuf_TAUX': 'zonal_surface_stress',\n    'pbuf_TAUY': 'meridional_surface_stress',\n    'pbuf_COSZRS': 'cosine_of_solar_zenith_angle',\n    'cam_in_ALDIF': 'albedo_for_diffuse_longwave_radiation',\n    'cam_in_ALDIR': 'albedo_for_direct_longwave_radiation',\n    'cam_in_ASDIF': 'albedo_for_diffuse_shortwave_radiation',\n    'cam_in_ASDIR': 'albedo_for_direct_shortwave_radiation',\n    'cam_in_LWUP': 'upward_longwave_flux',\n    'cam_in_ICEFRAC': 'sea-ice_areal_fraction',\n    'cam_in_LANDFRAC': 'land_areal_fraction',\n    'cam_in_OCNFRAC': 'ocean_areal_fraction',\n    'cam_in_SNOWHLAND': 'snow_depth_over_land',\n    'pbuf_ozone': 'ozone_volume_mixing_ratio',\n    'pbuf_CH4': 'methane_volume_mixing_ratio',\n    'pbuf_N2O': 'nitrous_oxide_volume_mixing_ratio'\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import SUBSET of data\nn_rows = 100000\nt1 = time.time()\n#df = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', n_rows=n_rows).to_pandas()\nt2 = time.time()\nprint('Elapsed time [s]: ', np.round(t2-t1,4))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glob import glob\nimport cudf\ntrain_files = sorted(glob(\"/kaggle/input/leap-dataset-giba/train_batch/*.parquet\"))\ntest_files = glob(\"/kaggle/input/leap-dataset-giba/test_batch/*.parquet\")\nlen(train_files), len(test_files)\n\n# Train on 2/17 of the full dataset\nt1 = time.time()\ndf = pd.read_parquet(train_files[:1]).astype('float32')\n#df = cudf.from_pandas(df) # Send to GPU for speedup\ngc.collect()\nt2 = time.time()\nprint('Elapsed time [s]: ', np.round(t2-t1,4))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns[491:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_similar_value_cols(df, percent=90):\n    count = 0\n    sim_val_cols = []\n    for col in df.columns:\n        percent_vals = (df[col].value_counts()/len(df)*100).values\n        # filter columns where more than 90% values are same and leave out binary encoded columns\n        if percent_vals[0] > percent and len(percent_vals) > 2:\n            sim_val_cols.append(col)\n            count += 1\n    print(\"Total columns with majority singular value shares: \", count)\n    return sim_val_cols","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_dataframe_size(df):\n    # Make a copy of the DataFrame to avoid modifying the original\n    df_copy = df.copy()\n\n    # Define a mapping of column names to more memory-efficient data types\n    column_type_mapping = {}\n\n    for column in df_copy.columns:\n        dtype = df_copy[column].dtype\n\n        if dtype == 'object':\n            # Handle object columns (e.g., strings)\n            unique_values = df_copy[column].nunique()\n\n            if unique_values == 1:\n                # If a column has only one unique value, convert it to the appropriate data type\n                if 'int' in dtype.name:\n                    column_type_mapping[column] = 'int64'\n                elif 'float' in dtype.name:\n                    column_type_mapping[column] = 'float64'\n                elif 'datetime' in dtype.name:\n                    column_type_mapping[column] = 'datetime64'\n                else:\n                    column_type_mapping[column] = 'category'\n            elif unique_values < len(df_copy) / 2:\n                # If a column has less than half unique values, convert it to category\n                column_type_mapping[column] = 'category'\n        elif dtype == 'float64':\n            # Reduce float64 columns to float32 or float16\n            column_type_mapping[column] = 'float32'\n        elif dtype == 'int64':\n            # Reduce int64 columns to int32 or int16\n            column_type_mapping[column] = 'int32'\n        elif dtype == 'datetime64':\n            # Reduce datetime64 columns to datetime32\n            column_type_mapping[column] = 'datetime32'\n    \n    # Apply the data type conversions based on the mapping\n    df_copy = df_copy.astype(column_type_mapping)\n\n    return df_copy\ndf = reduce_dataframe_size(df)\ndf.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_columns = df.iloc[:,1:491].columns\ntarget_columns = df.iloc[:,491:].columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.rename(columns=input_cols)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analysis","metadata":{}},{"cell_type":"code","source":"#df_f = drop_columns_with_high_null_rate(df_train, null_threshold=0.3)\nget_similar_value_cols(df, percent=90)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_cols = ['surface_pressure', 'solar_insolation', 'surface_latent_heat_flux',\n       'surface_sensible_heat_flux', 'zonal_surface_stress',\n       'meridional_surface_stress', 'cosine_of_solar_zenith_angle',\n       'albedo_for_diffuse_longwave_radiation',\n       'albedo_for_direct_longwave_radiation',\n       'albedo_for_diffuse_shortwave_radiation',\n       'albedo_for_direct_shortwave_radiation', 'upward_longwave_flux',\n       'sea-ice_areal_fraction', 'land_areal_fraction', 'ocean_areal_fraction',\n       'snow_depth_over_land', 'cam_out_NETSW', 'cam_out_FLWDS',\n       'cam_out_PRECSC', 'cam_out_PRECC', 'cam_out_SOLS', 'cam_out_SOLL',\n       'cam_out_SOLSD', 'cam_out_SOLLD']\ndf[corr_cols].describe().T.style.background_gradient(axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr =df[corr_cols].corr()\ncorr.style.background_gradient(cmap='Reds')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select a random row index\nrandom_index = np.random.randint(0, len(df))\nlike_list = ['state_t_', 'state_q0001_', 'state_q0002_', 'state_q0003_','state_u_','state_v_',\n             'pbuf_ozone_', 'pbuf_CH4_', 'pbuf_N2O_', 'ptend_t_', 'ptend_q0001_', 'ptend_q0002_'\n             ,'ptend_q0003_','ptend_v_','ptend_u_']\n\n# Number of plots\nnum_plots = len(like_list)\n\n# Create subplots with a 4x3 grid\nnrows, ncols = 4, 3\nfig, axes = plt.subplots(nrows=nrows, ncols=ncols, figsize=(15, 10))\naxes = axes.flatten()\nfor i, ax in enumerate(axes):\n    if i < num_plots:\n        column = like_list[i]\n        pd.DataFrame(corr.to_dict())\n        data_to_plot = df.loc[df.index==random_index].filter(like=column).values.tolist()[0]#df.loc[random_index, column]\n        ax.plot(data_to_plot, marker='o', linestyle='-')\n        ax.set_title(f'Plot of {column} at row {random_index}')\n        ax.set_xlabel('Index')\n        ax.set_ylabel('Value')\n        ax.grid(True)\n    else:\n        ax.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[corr_cols].select_dtypes(include='number').hist(figsize=(20,12))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df = cudf.from_pandas(df) # Send to GPU for speedup","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### modelling","metadata":{}},{"cell_type":"markdown","source":"#### NN","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MinMaxScaler\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nscaler = MinMaxScaler()\n\nX = scaler.fit_transform(df.iloc[:,1:491].to_numpy().astype(np.float32))\ny = df.iloc[:,491:].to_numpy().astype(np.float32)\ndel df\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape,y.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\ndel X\ndel y\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def r_squared(y_true, y_pred):\n    residual = tf.reduce_sum(tf.square(tf.subtract(y_true, y_pred)))\n    total = tf.reduce_sum(tf.square(tf.subtract(y_true, tf.reduce_mean(y_true))))\n    r2 = tf.subtract(1.0, tf.divide(residual, total))\n    return r2\n\n# simple model\nmodel_dense = Sequential([\n    Dense(512, activation='relu', input_shape=(490,)),#number of input cols-features\n    Dropout(0.3),\n    Dense(256, activation='relu'),\n    Dropout(0.3),\n    Dense(128, activation='relu'),\n    Dropout(0.3),\n    Dense(368, activation='linear')#number of targets\n])\n\n# Compiling the model\nmodel_dense.compile(optimizer='adam',\n              loss='mse',\n              metrics=[r_squared])  # Using the custom R-squared metric\n\n# Model summary to see the full architecture\nmodel_dense.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nhistory = model_dense.fit(X_train, y_train, validation_data=(X_test, y_test), epochs=10, batch_size=32)\ntest_loss, test_r2 = model_dense.evaluate(X_test, y_test)\nprint('Test loss:', test_loss, 'Test R²:', test_r2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.reshape((X_train.shape[0], X_train.shape[1], 1))  # Add the features dimension\ny_train = y_train.reshape((y_train.shape[0], y_train.shape[1], 1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape, y_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train\ndel X_test\ndel y_train\ndel y_test\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"test = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv', n_rows=1000).to_pandas()[input_columns]\ntest.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_scaled = scaler.fit_transform(test.to_numpy().astype(np.float32))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_nn = model.predict(test_scaled)\npred_nn.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub =pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv').to_pandas()\nsub.sample(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[target_columns] = pred_nn\ntest.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}