{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport io\nimport os\nimport shutil\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.konwinski_prize_inference_server\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-12-19T08:33:36.173561Z","iopub.execute_input":"2024-12-19T08:33:36.1741Z","iopub.status.idle":"2024-12-19T08:33:36.203648Z","shell.execute_reply.started":"2024-12-19T08:33:36.174058Z","shell.execute_reply":"2024-12-19T08:33:36.202252Z"},"papermill":{"duration":14.873526,"end_time":"2024-12-11T03:22:08.818755","exception":false,"start_time":"2024-12-11T03:21:53.945229","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The evaluation API requires that you set up a server which will respond to inference requests. We have already defined the server; you just need write the predict function. When we evaluate your submission on the hidden test set the client defined in `konwinski_prize_gateway` will run in a different container with direct access to the hidden test set and hand off the data.\n\nYour code will always have access to the published copies of the files.","metadata":{"papermill":{"duration":0.002032,"end_time":"2024-12-11T03:22:08.823897","exception":false,"start_time":"2024-12-11T03:22:08.821865","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport zipfile\n\n# Path to the ZIP file\nzip_path = '/kaggle/input/konwinski-prize/data.a_zip'\n\n# Extract the ZIP file\nextract_dir = '/kaggle/working/unzipped_data/'\nos.makedirs(extract_dir, exist_ok=True)\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_dir)\n\n# List extracted files and directories\nprint(\"Extracted Files:\", os.listdir(extract_dir))\n\n# Check inside the 'data' directory\ndata_dir = os.path.join(extract_dir, 'data')\nprint(\"Contents of 'data' directory:\", os.listdir(data_dir))\n\n# Locate a CSV or readable file inside the 'data' directory\nfor file in os.listdir(data_dir):\n    file_path = os.path.join(data_dir, file)\n    print(\"Found file:\", file_path)\n    \n    # Attempt to read the file as a CSV\n    if file.endswith('.csv'):\n        df = pd.read_csv(file_path)\n        print(\"File successfully read as CSV.\")\n        print(df.head())\n        break\nelse:\n    print(\"No CSV file found. Check the file formats.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:33:36.205529Z","iopub.execute_input":"2024-12-19T08:33:36.205934Z","iopub.status.idle":"2024-12-19T08:33:38.386822Z","shell.execute_reply.started":"2024-12-19T08:33:36.205898Z","shell.execute_reply":"2024-12-19T08:33:38.385173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path to the Parquet file\nparquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\n\n# Read the Parquet file\ndf = pd.read_parquet(parquet_file)\n\n# Display the first few rows of the DataFrame\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:33:38.388287Z","iopub.execute_input":"2024-12-19T08:33:38.388723Z","iopub.status.idle":"2024-12-19T08:33:38.411946Z","shell.execute_reply.started":"2024-12-19T08:33:38.388687Z","shell.execute_reply":"2024-12-19T08:33:38.410626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Load Data\nparquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\ndf = pd.read_parquet(parquet_file)\n\n# Step 1: Prepare Target Variable\n# Convert 'issue_numbers' to the count of issues\ndf['issue_count'] = df['issue_numbers'].apply(lambda x: len(x) if isinstance(x, list) else 0)\n\n# Step 2: Text Features Preparation\ntext_columns = ['problem_statement', 'patch', 'test_patch']\ndf['combined_text'] = df[text_columns].apply(lambda row: ' '.join(row.values.astype(str)), axis=1)\n\n# Step 3: Vectorize Text Data (TF-IDF)\nvectorizer = TfidfVectorizer(max_features=200000)\nX_text = vectorizer.fit_transform(df['combined_text'])\n\n# Step 4: Prepare Input Features and Target\nX = X_text  # TF-IDF vectors as features\ny = df['issue_count']  # Target: issue count\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 5: Train LightGBM Model\nmodel = LGBMRegressor(\n    objective='regression',\n    n_estimators=10000,\n    learning_rate=0.1,\n    boosting_type='gbdt',\n    random_state=42\n)\nmodel.fit(\n    X_train,\n    y_train,\n    eval_set=[(X_test, y_test)],\n    eval_metric='rmse',\n)\n\n# Step 6: Evaluate the Model\ny_pred = model.predict(X_test)\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\nprint(\"Root Mean Squared Error (RMSE):\", rmse)\n\n# Optional: Define utility functions\ndef instance():\n    return y  # Replace 'y' with the actual value or logic you want to return\n\ndef predict():\n    return y_pred  # Replace 'y_pred' with the actual prediction logic or value\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\n\n# Evaluate the Model on Test Set\ny_pred = model.predict(X_test)\n\n# Calculate RMSE\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\n# Calculate MAE\nmae = mean_absolute_error(y_test, y_pred)\n\n# Calculate R² Score\nr2 = r2_score(y_test, y_pred)\n\n# Calculate MAPE\nmape = np.mean(np.abs((y_test - y_pred) / y_test)) * 100\n\n# Print metrics\nprint(\"Model Performance Metrics:\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"R² Score: {r2:.4f}\")\nprint(f\"Mean Absolute Percentage Error (MAPE): {mape:.2f}%\")\n\n# Scatter plot of actual vs predicted values\nplt.figure(figsize=(20, 6))\nplt.scatter(y_test, y_pred, alpha=0.6, color=\"blue\")\nplt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], '--r', linewidth=2)  # Line of perfect prediction\nplt.title(\"Actual vs Predicted Values\")\nplt.xlabel(\"Actual Values\")\nplt.ylabel(\"Predicted Values\")\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:33:40.147693Z","iopub.execute_input":"2024-12-19T08:33:40.148296Z","iopub.status.idle":"2024-12-19T08:33:40.504189Z","shell.execute_reply.started":"2024-12-19T08:33:40.148237Z","shell.execute_reply":"2024-12-19T08:33:40.502978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"execution":{"iopub.status.busy":"2024-12-19T08:33:40.505919Z","iopub.execute_input":"2024-12-19T08:33:40.506961Z","iopub.status.idle":"2024-12-19T08:33:40.515221Z","shell.execute_reply.started":"2024-12-19T08:33:40.506906Z","shell.execute_reply":"2024-12-19T08:33:40.513983Z"},"papermill":{"duration":0.011949,"end_time":"2024-12-11T03:22:08.838279","exception":false,"start_time":"2024-12-11T03:22:08.82633","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error\nimport io\nimport shutil\n\nfirst_prediction = True\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    # Load Data\n    parquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\n    df = pd.read_parquet(parquet_file)\n\n    # Step 1: Prepare Target Variable\n    # Convert 'issue_numbers' to the count of issues\n    df['issue_count'] = df['issue_numbers'].apply(lambda x: len(x) if isinstance(x, list) else 0)\n\n    # Step 2: Text Features Preparation\n    text_columns = ['problem_statement', 'patch', 'test_patch']\n    df['combined_text'] = df[text_columns].apply(lambda row: ' '.join(row.values.astype(str)), axis=1)\n\n    # Step 3: Vectorize Text Data (TF-IDF)\n    vectorizer = TfidfVectorizer(max_features=5000)\n    X_text = vectorizer.fit_transform(df['combined_text'])\n\n    # Step 4: Prepare Input Features and Target\n    X = X_text  # TF-IDF vectors as features\n    y = df['issue_count']  # Target: issue count\n\n    # Split the data into train and test sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Step 5: Train LightGBM Model\n    model = LGBMRegressor(\n        n_estimators=100,\n        learning_rate=0.1,\n        max_depth=-1,\n        random_state=42\n    )\n    model.fit(X_train, y_train)\n\n    # Step 6: Evaluate the Model\n    y_pred = model.predict(X_test)\n    rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n    print(\"Root Mean Squared Error (RMSE):\", rmse)\n\n    global first_prediction\n    if not first_prediction:\n        return None  # Skip issue.\n\n    # Unpack repository archive\n    with open('repo_archive.tar', 'wb') as f:\n        f.write(repo_archive.read())\n    repo_path = 'repo'\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive('repo_archive.tar', extract_dir=repo_path)\n    os.remove('repo_archive.tar')\n    first_prediction = False\n\n    # Placeholder return for submission\n    return \"Hello World\"\n","metadata":{"execution":{"iopub.status.busy":"2024-12-19T08:33:40.516855Z","iopub.execute_input":"2024-12-19T08:33:40.517319Z","iopub.status.idle":"2024-12-19T08:33:40.532919Z","shell.execute_reply.started":"2024-12-19T08:33:40.517269Z","shell.execute_reply":"2024-12-19T08:33:40.531324Z"},"papermill":{"duration":0.011382,"end_time":"2024-12-11T03:22:08.852112","exception":false,"start_time":"2024-12-11T03:22:08.84073","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 30 minute response deadline.","metadata":{"papermill":{"duration":0.001889,"end_time":"2024-12-11T03:22:08.856283","exception":false,"start_time":"2024-12-11T03:22:08.854394","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Ensure instance_id refers to a valid column in the dataframe\ninstance_id = ''  # Replace 'id' with the actual column name\n\n# Initialize the inference server\ninference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,\n    predict\n)\n\n# Serve the model or run it locally based on the environment\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Scratch directory for unpacking\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-12-19T08:33:40.534881Z","iopub.execute_input":"2024-12-19T08:33:40.535946Z","iopub.status.idle":"2024-12-19T08:34:10.086052Z","shell.execute_reply.started":"2024-12-19T08:33:40.53589Z","shell.execute_reply":"2024-12-19T08:34:10.08484Z"},"papermill":{"duration":3.790202,"end_time":"2024-12-11T03:22:12.648591","exception":false,"start_time":"2024-12-11T03:22:08.858389","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}