{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":210710864,"sourceType":"kernelVersion"},{"sourceId":210804510,"sourceType":"kernelVersion"},{"sourceId":210833056,"sourceType":"kernelVersion"},{"sourceId":211282696,"sourceType":"kernelVersion"},{"sourceId":211629801,"sourceType":"kernelVersion"},{"sourceId":211653094,"sourceType":"kernelVersion"},{"sourceId":211945583,"sourceType":"kernelVersion"},{"sourceId":212378976,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🎨 Welcome to the Blending Notebook! 🧩\n\nIn this notebook, we thoughtfully blend two submissions to enhance our predictions and improve scores! 🚀\n\nIf this approach leads to better results, we’ll update the notebook to reflect the improvements. Stay tuned and let’s aim for the top! 🏆","metadata":{}},{"cell_type":"markdown","source":"Thanks for P04e12 Blended Submission 🎯, turning this contest in a blending game lol!\nhttps://www.kaggle.com/code/pawelkauf/p04e12-blended-submission","metadata":{}},{"cell_type":"markdown","source":"# Appreciate for an upvote if you find it fun to play :)","metadata":{}},{"cell_type":"markdown","source":"# Import files","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-12T00:34:27.380628Z","iopub.execute_input":"2024-12-12T00:34:27.381157Z","iopub.status.idle":"2024-12-12T00:34:27.475462Z","shell.execute_reply.started":"2024-12-12T00:34:27.38111Z","shell.execute_reply":"2024-12-12T00:34:27.474252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import open source submission","metadata":{}},{"cell_type":"code","source":"rid_train_h2o = pd.read_csv('/kaggle/input/rid-train-h2o/submission.csv') # 1.02922\n# p04e12_blended = pd.read_csv('/kaggle/input/p04e12-blended-submission/submission.csv') # 1.02903 - 0.7 Regression ESB - 0.3 H2O\nregression_ESB = pd.read_csv('/kaggle/input/regression-with-an-insurance-ensemble/submission.csv') # 1.02915\nLGBR_STACK_1 = pd.read_csv('/kaggle/input/insurance-competition-database/LGBR_STACK_1.03088.csv') # 1.03088\n\naverager_1 = pd.read_csv('/kaggle/input/insurance-competition-database/averager_1.0313127009620042.csv') # 1.03131\ndiff_ev_1 = pd.read_csv('/kaggle/input/insurance-competition-database/diff_ev_1.0313762362011574.csv') # 1.031376","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T00:34:27.477248Z","iopub.execute_input":"2024-12-12T00:34:27.477653Z","iopub.status.idle":"2024-12-12T00:34:28.603677Z","shell.execute_reply.started":"2024-12-12T00:34:27.477601Z","shell.execute_reply":"2024-12-12T00:34:28.602104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# diff_sub1_sub2 = abs(sub1['Premium Amount'] - sub2['Premium Amount'])\n# diff_sub1_sub3 = abs(sub1['Premium Amount'] - sub3['Premium Amount'])\n\n# # Calculate overall percentage differences\n# percentage_diff_sub2 = (diff_sub1_sub2.sum() / sub1['Premium Amount'].sum()) * 100\n# percentage_diff_sub3 = (diff_sub1_sub3.sum() / sub1['Premium Amount'].sum()) * 100\n\n# print(f\"Overall Percentage Difference between sub1 and sub2: {percentage_diff_sub2:.2f}%\")\n# print(f\"Overall Percentage Difference between sub1 and sub3: {percentage_diff_sub3:.2f}%\")\n\n# if not diff_sub1_sub2.isnull().all():\n#     print(f\"Rows with differences between sub1 and sub2:\\n{diff_sub1_sub2[diff_sub1_sub2 > 0]}\")\n\n# if not diff_sub1_sub3.isnull().all():\n#     print(f\"Rows with differences between sub1 and sub3:\\n{diff_sub1_sub3[diff_sub1_sub3 > 0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T00:34:28.605121Z","iopub.execute_input":"2024-12-12T00:34:28.605481Z","iopub.status.idle":"2024-12-12T00:34:28.61098Z","shell.execute_reply.started":"2024-12-12T00:34:28.605445Z","shell.execute_reply":"2024-12-12T00:34:28.609472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Start playing blending","metadata":{}},{"cell_type":"code","source":"# Create a copy for blending\nblended = rid_train_h2o.copy()\n\n# # Purely Weighted (Bad)\n# # Using weights: [0.33179983, 0.33411145, 0.33408872]\n# blended['Premium Amount'] = (\n#     (0.33179983) * rid_train_h2o['Premium Amount'] +\n#     (0.33411145) * regression_ESB['Premium Amount'] +\n#     (0.33408872) * LGBR_STACK_1['Premium Amount']\n# )\n\n# # x + x^2 + x^3 = 1 with stack\n# blended['Premium Amount'] = (\n#     (0.2956) * rid_train_h2o['Premium Amount'] +\n#     (0.5437) * regression_ESB['Premium Amount'] +\n#     (0.1607) * LGBR_STACK_1['Premium Amount']\n#     # x + x^2 + x^3 = 1, x = 0.5437, The higher score gets higher weight\n# )\n\n# # x + x^2 + x^3 = 1 with stack & diff & average\n# blended['Premium Amount'] = (\n#     (0.2956) * rid_train_h2o['Premium Amount'] +\n#     (0.5437) * regression_ESB['Premium Amount'] +\n#     (0.1607/3) * LGBR_STACK_1['Premium Amount'] +\n#     (0.1607/3) * averager_1['Premium Amount'] +\n#     (0.1607/3) * diff_ev_1['Premium Amount']\n#     # x + x^2 + x^3 = 1, x = 0.5437, The higher score gets higher weight\n# )\n\n# # x + x^2 + x^3 = 1, if stack plays the key role\n# blended['Premium Amount'] = (\n#     (0.2956) * regression_ESB['Premium Amount'] +\n#     (0.5437) * LGBR_STACK_1['Premium Amount'] +\n#     (0.1607) * rid_train_h2o['Premium Amount']\n#     # x + x^2 + x^3 = 1, x = 0.5437, The higher score gets higher weight\n# )\n\n# more stack\nblended['Premium Amount'] = (\n    (0.30) * regression_ESB['Premium Amount'] +\n    (0.545) * LGBR_STACK_1['Premium Amount'] +\n    (0.155) * rid_train_h2o['Premium Amount']\n)\n\n# Save the blended results\nblended.to_csv('submission.csv', index=False)\n\nblended","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T00:34:28.613748Z","iopub.execute_input":"2024-12-12T00:34:28.614438Z","iopub.status.idle":"2024-12-12T00:34:30.379336Z","shell.execute_reply.started":"2024-12-12T00:34:28.614381Z","shell.execute_reply":"2024-12-12T00:34:30.378097Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation-based weighting","metadata":{}},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n\n# # Simulated model prediction results (replace with actual data)\n# np.random.seed(42)\n\n# # Copy the predictions from rid_train_h2o to a new DataFrame\n# blended_premium = rid_train_h2o.copy()  # Collect predictions from all models\n\n# # Create a DataFrame containing predictions from all models\n# predictions = pd.DataFrame({\n#     'regression_ESB': regression_ESB['Premium Amount'],  # Predictions from regression_ESB model\n#     'LGBR_STACK_1': LGBR_STACK_1['Premium Amount'],      # Predictions from LGBR_STACK_1 model\n#     'rid_train_h2o': rid_train_h2o['Premium Amount']     # Predictions from rid_train_h2o model\n# })\n\n# # Calculate the correlation matrix for the predictions\n# correlation_matrix = predictions.corr()\n\n# # Compute the average correlation (each model's correlation with others)\n# avg_correlation = correlation_matrix.mean(axis=1)\n\n# # Adjust weights based on the formula (inverse of average correlation)\n# inverse_correlation = 1 / avg_correlation\n# weights = inverse_correlation / inverse_correlation.sum()\n\n# # Display the adjusted weights\n# print(\"Adjusted weights based on correlation:\")\n# for model, weight in zip(predictions.columns, weights):\n#     print(f\"{model}: {weight:.4f}\")\n\n# # Apply the adjusted weights to compute the weighted average predictions\n# blended_premium['Premium Amount'] = (\n#     weights['regression_ESB'] * regression_ESB['Premium Amount'] +\n#     weights['LGBR_STACK_1'] * LGBR_STACK_1['Premium Amount'] +\n#     weights['rid_train_h2o'] * rid_train_h2o['Premium Amount']\n# )\n\n# # Output the final blended results\n# print(\"Final blended predictions:\")\n\n# # Save the final blended predictions to a CSV file\n# blended_premium.to_csv('submission.csv', index=False)\n\n# blended_premium","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T00:34:30.380714Z","iopub.execute_input":"2024-12-12T00:34:30.38111Z","iopub.status.idle":"2024-12-12T00:34:30.387258Z","shell.execute_reply.started":"2024-12-12T00:34:30.381068Z","shell.execute_reply":"2024-12-12T00:34:30.386012Z"}},"outputs":[],"execution_count":null}]}