{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"},{"sourceId":10302984,"sourceType":"datasetVersion","datasetId":6377357},{"sourceId":207931,"sourceType":"modelInstanceVersion","modelInstanceId":177269,"modelId":199575},{"sourceId":209541,"sourceType":"modelInstanceVersion","modelInstanceId":178636,"modelId":200929},{"sourceId":11371,"sourceType":"modelInstanceVersion","modelInstanceId":5171,"modelId":3533}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: EDA Process run it online and the markdown it\n\npersist the variables and files","metadata":{}},{"cell_type":"code","source":"    !cp -r /kaggle/input/andro_konwinski_llm_model_readiness_offline/other/default/5/* /kaggle/working\n    #v6 v7 v8 v9 v10 on this code will be updated soon !!\n\n    import sys \n    sys.path.append('/kaggle/working/andro_swebench_processor_offline_v5.py')\n\n    from andro_swebench_processor_offline_v5 import andro_swebench_processor_offline_v5\n\n    processor = andro_swebench_processor_offline_v5(\n                        zip_file_path='../input/konwinski-prize/data.a_zip',\n                        parquet_file_path='data/data.parquet',\n                        dataset_name='/kaggle/input/swe-bench/SWE-bench'\n                    )\n\n    print(\"Visualizing data...\")\n    processor.visualize_data()\n\n    print(\"Creating QNA dataset as list...\")\n    processor.create_QNA_dataset_as_list()\n\n    # Store the QNA dataset list\n    qna_dataset_list = processor.get_all_QNA_dataset()\n\n    print(\"QNA List Size:\")\n    processor.show_list_size()\n\n    print(\"Creating QNA dataset as DataFrame...\")\n    processor.create_QNA_dataset_as_dataframe()\n\n    print(\"DataFrame Summary:\")\n    processor.get_dataframe_summary()\n\n    print(\"Sample QNA dataset:\")\n    processor.print_colored_QNA_samples()\n\n    # Example: Print the first QNA template from the list\n    print(\"First QNA template from the list:\")\n    print(qna_dataset_list[0] if qna_dataset_list else \"No data available.\")\n    \n    # Save the QNA dataset as a CSV file \n    #qna_dataset_df = processor.QNA_dataframe \n    #csv_file_path = \"qna_dataset.csv\" \n    #qna_dataset_df.to_csv(csv_file_path, index=False) \n    #print(f\"QNA dataset saved to {csv_file_path}\")","metadata":{"execution":{"iopub.status.busy":"2024-12-29T15:21:13.241461Z","iopub.execute_input":"2024-12-29T15:21:13.241669Z","iopub.status.idle":"2024-12-29T15:24:52.667551Z","shell.execute_reply.started":"2024-12-29T15:21:13.241649Z","shell.execute_reply":"2024-12-29T15:24:52.666409Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qna_dataset_list[1:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T15:27:28.304092Z","iopub.execute_input":"2024-12-29T15:27:28.304395Z","iopub.status.idle":"2024-12-29T15:27:28.309619Z","shell.execute_reply.started":"2024-12-29T15:27:28.304373Z","shell.execute_reply":"2024-12-29T15:27:28.308516Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### install these libraries\n!pip install -q -U keras-nlp\n!pip install -q -U keras>=3","metadata":{"execution":{"iopub.status.busy":"2024-12-28T07:43:00.175118Z","iopub.execute_input":"2024-12-28T07:43:00.175481Z","iopub.status.idle":"2024-12-28T07:43:10.003772Z","shell.execute_reply.started":"2024-12-28T07:43:00.175451Z","shell.execute_reply":"2024-12-28T07:43:10.002714Z"}}},{"cell_type":"markdown","source":"# Step 2. Train the model","metadata":{}},{"cell_type":"code","source":"# Move all necessary files\n!cp -r /kaggle/input/androgemmallmpipelinemodel/other/default/1/androgemmallmpipeline-0.1/* /kaggle/working/\n# change directory\n%cd /kaggle/working/\n# packaging the path\n!mkdir -p /kaggle/working/androgemmallmpipeline\n# compiling and building the setup.py files\n!python setup.py build_ext --inplace\n\nimport sys\nsys.path.append('/kaggle/working/')\n\nfrom androgemmallmpipeline.androgemmallmpipelinemodel import androgemmallmpipelinemodel\n\n# Example usage\noptimizer_params = {\n    \"learning_rate\": 5e-5,\n    \"weight_decay\": 0.01,\n    \"beta_1\": 0.9,\n    \"beta_2\": 0.999,\n}\n\n# Assume `sample_dataset` is defined and loaded elsewhere\nmodel = androgemmallmpipelinemodel(\n    rank=4,\n    dataset=qna_dataset_list,\n    optimizer_params=optimizer_params,\n    sequence_length=256,\n    epochs=1,\n    batch_size=1\n)\n\nmodel.fit_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T15:27:36.922546Z","iopub.execute_input":"2024-12-29T15:27:36.922901Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Generate, Predict and submit the .csv file","metadata":{}},{"cell_type":"code","source":"instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances\n\nfirst_prediction = True\n\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    \"\"\"Inference function to generate a patch for a GitHub issue.\n\n    Args:\n        problem_statement (str): The text of the GitHub issue.\n        repo_archive (io.BytesIO): A BytesIO buffer containing a .tar archive of the codebase.\n\n    Returns:\n        str: The generated patch as a string.\n    \"\"\"\n    global first_prediction\n    if not first_prediction:\n        return None  # Skip the first issue.\n\n    # Unpack the repository archive\n    with open(\"repo_archive.tar\", \"wb\") as f:\n        f.write(repo_archive.read())\n    repo_path = \"repo\"\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(\"repo_archive.tar\", extract_dir=repo_path)\n    os.remove(\"repo_archive.tar\")\n    first_prediction = False\n\n    # Generate a patch\n    input_text = f\"Problem:\\n{problem_statement}\\n\\nPatch:\"\n    try:\n        # Modify the generate call to remove unsupported arguments\n        generated_patch = model.generate([input_text])[0]\n        return generated_patch\n    except Exception as e:\n        print(f\"Error during generation: {e}\")\n        return \"Error generating patch.\"\n\n\ninference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,   \n    predict\n)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data.a_zip.\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}