{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv\nimport os \n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import accuracy_score, confusion_matrix, mean_squared_error\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.linear_model import LinearRegression\nfrom datetime import datetime\n\n\nimport warnings\nwarnings.filterwarnings('ignore')\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:30:27.530154Z","iopub.execute_input":"2024-12-24T21:30:27.530616Z","iopub.status.idle":"2024-12-24T21:30:28.840265Z","shell.execute_reply.started":"2024-12-24T21:30:27.530576Z","shell.execute_reply":"2024-12-24T21:30:28.83903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Większość początkowego kodu skopiowałem z naszych notebooków z zajęć na YouTube**.\n\nDostosowałem je do potrzeb konkursu i tak na prawdę dopiero wtedy zrozumiałem co do tej pory robiliśmy.\n\n**Z pomocą chataGPT pouzupełniałem brakujące dane w DF'ach.**\n\nPóźniej próbowałem przeprowadzić trenig taki jak na zajęciach ale zżerało to całe dostępne zasoby. \n\n\n**Z pomocą chatGPT próbowałem zmniejszyć wielkości DF konwertując typy kolumn, które to miały znacznie zmniejszyć pliki ale nic to nie dało**\n\nKolejne godziny spędziłem również z chatGPT ucząc się jak można ternowac tak duże dane.\n\nDostarczone odpowiedzi próbowałem przerobić na potrzeby konkursu i ostatecznie miałem dwie metody trenowania:\n\n\n* Lightgbm z odpowiednimi parametrami dla tak dużych zbiorów\n* Lightgbm z Optuna, które też miało swoje paramerty\n\nOstatecznie nie udało mi się poprawić wyników trenowania (zmieniając parametry modelu: gałęzie liście joby itp)\n\n**Podejrzewam, że problem, polegał w wyborze złej metody trenowania regresji linowej, albo w złym podejściu do wypełnienia brakujących danych.**\n\n","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:30:34.502294Z","iopub.execute_input":"2024-12-24T21:30:34.503396Z","iopub.status.idle":"2024-12-24T21:30:43.766963Z","shell.execute_reply.started":"2024-12-24T21:30:34.503354Z","shell.execute_reply":"2024-12-24T21:30:43.765609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(20)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#y_train kolumna z wynikiem \ny_train = df_train['Premium Amount']\n\n#usunięcie wyniku z df_train\ndf_train.drop('Premium Amount', axis=1, inplace=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:30:43.769141Z","iopub.execute_input":"2024-12-24T21:30:43.769516Z","iopub.status.idle":"2024-12-24T21:30:43.979715Z","shell.execute_reply.started":"2024-12-24T21:30:43.769479Z","shell.execute_reply":"2024-12-24T21:30:43.978508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#teraz df_train i df_test powinny być identyczne\n\ndf_train.info()\ndf_test.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#scalanie obu df\ndf_all = pd.concat([df_train, df_test], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:30:48.176014Z","iopub.execute_input":"2024-12-24T21:30:48.176452Z","iopub.status.idle":"2024-12-24T21:30:48.589779Z","shell.execute_reply.started":"2024-12-24T21:30:48.176415Z","shell.execute_reply":"2024-12-24T21:30:48.588869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sprawdzanie unikalnych wartości w każdej kolumnie\nunique_values = {col: df_all[col].unique() for col in df_all.columns}\n\n# Wyświetlenie wyników\nfor column, values in unique_values.items():\n    print(f\"Kolumna '{column}': {len(values)} unikalnych wartości\")\n    print(f\"Przykładowe wartości: {values[:10]}\\n\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#wypełniam brakujące dane\n\ndf_all['Age'].fillna(df_all['Age'].median(), inplace=True)\ndf_all['Annual Income'].fillna(df_all['Annual Income'].median(), inplace=True)\ndf_all['Marital Status'].fillna('Unknown', inplace=True)\ndf_all['Number of Dependents'].fillna(0, inplace=True)\ndf_all['Occupation'].fillna('Unknown', inplace=True)\ndf_all['Health Score'].fillna(df_all['Health Score'].median(), inplace=True)\ndf_all['Previous Claims'].fillna(0, inplace=True)  # Brak roszczeń = 0\ndf_all['Vehicle Age'].fillna(df_all['Vehicle Age'].median(), inplace=True)\ndf_all['Credit Score'].fillna(df_all['Credit Score'].median(), inplace=True)\ndf_all['Insurance Duration'].fillna(0, inplace=True)\ndf_all['Customer Feedback'].fillna('Average', inplace=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:30:52.882041Z","iopub.execute_input":"2024-12-24T21:30:52.882454Z","iopub.status.idle":"2024-12-24T21:30:53.573406Z","shell.execute_reply.started":"2024-12-24T21:30:52.882419Z","shell.execute_reply":"2024-12-24T21:30:53.572164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_date_features(df, column_name):\n    df[column_name] = pd.to_datetime(df[column_name], errors='coerce')\n    df['Year'] = df[column_name].dt.year\n    df['Month'] = df[column_name].dt.month\n    df['Day'] = df[column_name].dt.day\n    df['Weekday'] = df[column_name].dt.weekday\n    df['Week'] = df[column_name].dt.isocalendar().week\n    df['Quarter'] = df[column_name].dt.quarter\n    df['Day of Year'] = df[column_name].dt.dayofyear\n    df['Is Month Start'] = df[column_name].dt.is_month_start\n    df['Is Month End'] = df[column_name].dt.is_month_end\n    df['Is Leap Year'] = df[column_name].dt.is_leap_year\n    df['Days Since Start'] = (datetime.now() - df[column_name]).dt.days\n    df.drop(column_name, inplace=True, axis = 1)\n    return df_all","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:37:19.444078Z","iopub.execute_input":"2024-12-24T21:37:19.444725Z","iopub.status.idle":"2024-12-24T21:37:19.456959Z","shell.execute_reply.started":"2024-12-24T21:37:19.444674Z","shell.execute_reply":"2024-12-24T21:37:19.455449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all = extract_date_features(df_all, 'Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:37:23.41355Z","iopub.execute_input":"2024-12-24T21:37:23.414085Z","iopub.status.idle":"2024-12-24T21:37:25.954742Z","shell.execute_reply.started":"2024-12-24T21:37:23.41404Z","shell.execute_reply":"2024-12-24T21:37:25.953753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all.drop('id', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:37:46.565375Z","iopub.execute_input":"2024-12-24T21:37:46.565819Z","iopub.status.idle":"2024-12-24T21:37:46.856615Z","shell.execute_reply.started":"2024-12-24T21:37:46.565778Z","shell.execute_reply":"2024-12-24T21:37:46.855512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:38:42.334616Z","iopub.execute_input":"2024-12-24T21:38:42.335018Z","iopub.status.idle":"2024-12-24T21:38:42.349989Z","shell.execute_reply.started":"2024-12-24T21:38:42.334984Z","shell.execute_reply":"2024-12-24T21:38:42.348469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def split_numerical_categorical(df):\n    \"\"\"\n    Splits the columns of a DataFrame into numerical and categorical features.\n\n    Parameters:\n    df (pandas.DataFrame): The DataFrame to split.\n\n    Returns:\n    tuple: A tuple containing two lists - numerical columns and categorical columns.\n    \"\"\"\n    numerical_cols = df.select_dtypes(include=['number']).columns.tolist()\n    categorical_cols = df.select_dtypes(exclude=['number']).columns.tolist()\n    return numerical_cols, categorical_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:13.432397Z","iopub.execute_input":"2024-12-24T21:39:13.433401Z","iopub.status.idle":"2024-12-24T21:39:13.439714Z","shell.execute_reply.started":"2024-12-24T21:39:13.433359Z","shell.execute_reply":"2024-12-24T21:39:13.438195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_cols, categorical_cols = split_numerical_categorical(df_all)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:31.583236Z","iopub.execute_input":"2024-12-24T21:39:31.583651Z","iopub.status.idle":"2024-12-24T21:39:32.59634Z","shell.execute_reply.started":"2024-12-24T21:39:31.583618Z","shell.execute_reply":"2024-12-24T21:39:32.59506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#zamiana danych na kategoryczne (One-Hot Encoding)\n\ndf_all = pd.get_dummies(df_all, columns=categorical_cols, drop_first=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:39.441864Z","iopub.execute_input":"2024-12-24T21:39:39.44235Z","iopub.status.idle":"2024-12-24T21:39:41.997229Z","shell.execute_reply.started":"2024-12-24T21:39:39.442307Z","shell.execute_reply":"2024-12-24T21:39:41.996172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#zmiana typu kolumn kategorycznych z bool na category\n\nfor col in df_all.columns:\n    if any(col.startswith(c) for c in categorical_cols):  # Sprawdzamy, czy kolumna pochodzi z kategorii\n        df_all[col] = df_all[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:46.453739Z","iopub.execute_input":"2024-12-24T21:39:46.454978Z","iopub.status.idle":"2024-12-24T21:39:47.015251Z","shell.execute_reply.started":"2024-12-24T21:39:46.454889Z","shell.execute_reply":"2024-12-24T21:39:47.014012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Konwertowanie kolumn typu int64 na int32 lub int16, w zależności od zakresu\nfor col in df_all.select_dtypes(include=['int64']).columns:\n    # Sprawdzamy maksymalną wartość w kolumnie i wybieramy odpowiedni typ\n    if df_all[col].min() >= -32768 and df_all[col].max() <= 32767:\n        df_all[col] = df_all[col].astype('int16')  # Mniejsze zakresy\n    else:\n        df_all[col] = df_all[col].astype('int32')  # Większe zakresy\n\n# Konwertowanie kolumn typu float64 na float32\nfor col in df_all.select_dtypes(include=['float64']).columns:\n    df_all[col] = df_all[col].astype('float32')  # Zmniejszamy precyzję\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:50.174338Z","iopub.execute_input":"2024-12-24T21:39:50.174767Z","iopub.status.idle":"2024-12-24T21:39:50.256708Z","shell.execute_reply.started":"2024-12-24T21:39:50.174714Z","shell.execute_reply":"2024-12-24T21:39:50.255594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#zamiana nazw kolumn - spacja na podkreślenie\n\ndf_all.columns = df_all.columns.str.replace(' ', '_')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:39:53.687296Z","iopub.execute_input":"2024-12-24T21:39:53.687779Z","iopub.status.idle":"2024-12-24T21:39:53.694268Z","shell.execute_reply.started":"2024-12-24T21:39:53.687734Z","shell.execute_reply":"2024-12-24T21:39:53.692908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:33:37.980067Z","iopub.execute_input":"2024-12-24T22:33:37.98073Z","iopub.status.idle":"2024-12-24T22:33:38.000835Z","shell.execute_reply.started":"2024-12-24T22:33:37.980681Z","shell.execute_reply":"2024-12-24T22:33:37.999004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_all.dtypes)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X_train df dla trenigu wycięty z df_all o długości df_train\n\nX_train = df_all[:df_train.shape[0]]\n\n# i reszta do trenowania\nX_test = df_all[df_train.shape[0]:]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:40:09.764403Z","iopub.execute_input":"2024-12-24T21:40:09.764849Z","iopub.status.idle":"2024-12-24T21:40:09.773017Z","shell.execute_reply.started":"2024-12-24T21:40:09.764811Z","shell.execute_reply":"2024-12-24T21:40:09.77123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sprawdzenie\nX_train.info()\nX_test.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:34:28.893353Z","iopub.execute_input":"2024-12-24T22:34:28.893833Z","iopub.status.idle":"2024-12-24T22:34:33.510695Z","shell.execute_reply.started":"2024-12-24T22:34:28.893786Z","shell.execute_reply":"2024-12-24T22:34:33.509425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:34:39.175102Z","iopub.execute_input":"2024-12-24T22:34:39.175543Z","iopub.status.idle":"2024-12-24T22:34:39.451826Z","shell.execute_reply.started":"2024-12-24T22:34:39.175496Z","shell.execute_reply":"2024-12-24T22:34:39.448806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:34:49.866712Z","iopub.execute_input":"2024-12-24T22:34:49.867367Z","iopub.status.idle":"2024-12-24T22:34:49.876895Z","shell.execute_reply.started":"2024-12-24T22:34:49.867306Z","shell.execute_reply":"2024-12-24T22:34:49.875503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#trenowanie wrzucone na kaggle\n\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\n\n# Split the training data for validation\nX_train_split, X_valid, y_train_split, y_valid = train_test_split(\n    X_train, y_train, test_size=0.1, random_state=42\n)\n\n# Convert to LightGBM dataset format\ntrain_data = lgb.Dataset(X_train_split, label=y_train_split, categorical_feature='auto')\nvalid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature='auto', reference=train_data)\n\n# LightGBM parameters\nparams = {\n    \"objective\": \"regression\",\n    \"boosting_type\": \"gbdt\",\n    \"metric\": \"rmse\",\n    \"learning_rate\": 0.05,\n    \"max_depth\": -1,\n    \"num_leaves\": 31,\n    \"feature_fraction\": 0.8,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"verbosity\": -1,\n}\n\n# Define callbacks\ncallbacks = [\n    lgb.early_stopping(stopping_rounds=50),  # Stops training if no improvement\n    lgb.log_evaluation(period=10)           # Logs evaluation results every 10 iterations\n]\n\n# Train model\nmodel = lgb.train(\n    params,\n    train_data,\n    valid_sets=[train_data, valid_data],\n    num_boost_round=1000,\n    callbacks=callbacks  # No need for verbose_eval as log_evaluation handles it\n)\n\n# Save the model\nmodel.save_model('lightgbm_regressor.txt')\n\n# Predict on test data\ny_pred = model.predict(X_test, num_iteration=model.best_iteration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:44:56.599568Z","iopub.execute_input":"2024-12-24T21:44:56.600109Z","iopub.status.idle":"2024-12-24T21:45:40.412536Z","shell.execute_reply.started":"2024-12-24T21:44:56.600069Z","shell.execute_reply":"2024-12-24T21:45:40.411376Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# automatyczne dostrajanie hiperparametrów\n# to jeszcze nie jest na kaggle, bo nie wiem gdzie mam wynik \nimport lightgbm as lgb\nimport optuna\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\n\n# Podział danych\nX_train_split, X_valid, y_train_split, y_valid = train_test_split(\n    X_train, y_train, test_size=0.1, random_state=42\n)\n\n# Funkcja celu dla Optuna\ndef objective(trial):\n    # Propozycja wartości hiperparametrów\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\": \"gbdt\",\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 150),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 20),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.5, 1.0),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.5, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 1, 10),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 10, 50),\n        \"verbosity\": -1,\n    }\n\n    # Callbacki do logowania i zatrzymywania wczesnego \n    callbacks = [\n        lgb.early_stopping(stopping_rounds=50),  # Stops training if no improvement\n        lgb.log_evaluation(period=25)           # Logs evaluation results every 10 iterations\n    ]\n\n        \n    model = lgb.train(\n        param,\n        lgb.Dataset(X_train_split, label=y_train_split, categorical_feature='auto'),\n        valid_sets=[lgb.Dataset(X_valid, label=y_valid, categorical_feature='auto')],\n        num_boost_round=1000,\n        callbacks=callbacks\n    )\n\n    # Predykcja i obliczenie metryki\n    y_pred = model.predict(X_valid, num_iteration=model.best_iteration)\n    rmse = mean_squared_error(y_valid, y_pred, squared=False)\n    return rmse\n\n# Inicjalizacja Optuna i optymalizacja\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50, n_jobs=5 )  # Liczba prób\n\n# Najlepsze hiperparametry\nprint(\"Najlepsze parametry:\", study.best_params)\n\n# Możesz użyć najlepszych parametrów do trenowania końcowego modelu\nbest_params = study.best_params\nbest_params.update({\"objective\": \"regression\", \"metric\": \"rmse\"})\n\nfinal_model = lgb.train(\n    best_params,\n    lgb.Dataset(X_train_split, label=y_train_split, categorical_feature='auto'),\n    valid_sets=[lgb.Dataset(X_valid, label=y_valid, categorical_feature='auto')],\n    num_boost_round=1000,\n    callbacks=[lgb.log_evaluation(period=10)]\n)\n\n# Predict on test data\ny_pred = final_model.predict(X_test, num_iteration=final_model.best_iteration)\n\n# Wyświetlenie wyników\n# print(\"Przewidywania:\", y_pred[:10])  # Przykładowe 10 wyników\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T21:47:03.516918Z","iopub.execute_input":"2024-12-24T21:47:03.517529Z","iopub.status.idle":"2024-12-24T22:21:31.117778Z","shell.execute_reply.started":"2024-12-24T21:47:03.517479Z","shell.execute_reply":"2024-12-24T22:21:31.116261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred\n","metadata":{"execution":{"iopub.status.busy":"2024-12-24T22:23:59.309532Z","iopub.execute_input":"2024-12-24T22:23:59.310087Z","iopub.status.idle":"2024-12-24T22:23:59.323127Z","shell.execute_reply.started":"2024-12-24T22:23:59.310043Z","shell.execute_reply":"2024-12-24T22:23:59.3217Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ssub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\nssub['Premium Amount'] = y_pred ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:35:26.930045Z","iopub.execute_input":"2024-12-24T22:35:26.930483Z","iopub.status.idle":"2024-12-24T22:35:27.110466Z","shell.execute_reply.started":"2024-12-24T22:35:26.930449Z","shell.execute_reply":"2024-12-24T22:35:27.109063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ssub.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T22:35:30.416473Z","iopub.execute_input":"2024-12-24T22:35:30.41695Z","iopub.status.idle":"2024-12-24T22:35:32.15766Z","shell.execute_reply.started":"2024-12-24T22:35:30.416884Z","shell.execute_reply":"2024-12-24T22:35:32.156444Z"}},"outputs":[],"execution_count":null}]}