{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Meqsed","metadata":{}},{"cell_type":"markdown","source":"### meqsed odurki bizde studyinstanceid var ve onlarin ozlerine aid feature var. Biz sample submission fayli ucun herbir studyinstanceid ucun 12 dene feature cixarmaliyiq ve vermeliyik. Yalniz test faylinin icinde knee MRI sekilleri var","metadata":{}},{"cell_type":"markdown","source":"### .dcm - MRI sekli; train.csv - hansi xesteliklerin oldugu, hansi studylerin icinde hansi MRI seriesler var ve bu ne tip goruntudur","metadata":{}},{"cell_type":"markdown","source":"# Data analysis","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nimport seaborn as sns\nfrom transformers import TrainingArguments, Trainer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:47.721354Z","iopub.execute_input":"2026-09-20T10:17:47.722244Z","iopub.status.idle":"2026-09-20T10:17:47.72629Z","shell.execute_reply.started":"2026-09-20T10:17:47.722208Z","shell.execute_reply":"2026-09-20T10:17:47.725457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall -y torchao","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:47.740887Z","iopub.execute_input":"2026-09-20T10:17:47.741226Z","iopub.status.idle":"2026-09-20T10:17:48.695533Z","shell.execute_reply.started":"2026-09-20T10:17:47.74119Z","shell.execute_reply":"2026-09-20T10:17:48.694702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index \"/kaggle/input/datasets/thesharifzade/torchao/torchao_pkg/torchao-0.16.0-cp310-abi3-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:48.697281Z","iopub.execute_input":"2026-09-20T10:17:48.697528Z","iopub.status.idle":"2026-09-20T10:17:52.938296Z","shell.execute_reply.started":"2026-09-20T10:17:48.697501Z","shell.execute_reply":"2026-09-20T10:17:52.937481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Dataset","metadata":{}},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:52.939677Z","iopub.execute_input":"2026-09-20T10:17:52.940041Z","iopub.status.idle":"2026-09-20T10:17:53.035761Z","shell.execute_reply.started":"2026-09-20T10:17:52.939995Z","shell.execute_reply":"2026-09-20T10:17:53.035158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.037665Z","iopub.execute_input":"2026-09-20T10:17:53.037881Z","iopub.status.idle":"2026-09-20T10:17:53.077816Z","shell.execute_reply.started":"2026-09-20T10:17:53.037859Z","shell.execute_reply":"2026-09-20T10:17:53.077036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.078787Z","iopub.execute_input":"2026-09-20T10:17:53.079122Z","iopub.status.idle":"2026-09-20T10:17:53.114745Z","shell.execute_reply.started":"2026-09-20T10:17:53.079087Z","shell.execute_reply":"2026-09-20T10:17:53.113899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.115705Z","iopub.execute_input":"2026-09-20T10:17:53.116025Z","iopub.status.idle":"2026-09-20T10:17:53.121395Z","shell.execute_reply.started":"2026-09-20T10:17:53.115981Z","shell.execute_reply":"2026-09-20T10:17:53.12049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.122232Z","iopub.execute_input":"2026-09-20T10:17:53.122521Z","iopub.status.idle":"2026-09-20T10:17:53.136654Z","shell.execute_reply.started":"2026-09-20T10:17:53.122498Z","shell.execute_reply":"2026-09-20T10:17:53.136028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.137483Z","iopub.execute_input":"2026-09-20T10:17:53.137742Z","iopub.status.idle":"2026-09-20T10:17:53.175943Z","shell.execute_reply.started":"2026-09-20T10:17:53.137691Z","shell.execute_reply":"2026-09-20T10:17:53.175351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.176809Z","iopub.execute_input":"2026-09-20T10:17:53.177761Z","iopub.status.idle":"2026-09-20T10:17:53.183894Z","shell.execute_reply.started":"2026-09-20T10:17:53.177735Z","shell.execute_reply":"2026-09-20T10:17:53.183315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(train_dataset.isnull(), cbar=False)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.186117Z","iopub.execute_input":"2026-09-20T10:17:53.186407Z","iopub.status.idle":"2026-09-20T10:17:53.563262Z","shell.execute_reply.started":"2026-09-20T10:17:53.186373Z","shell.execute_reply":"2026-09-20T10:17:53.562558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.564092Z","iopub.execute_input":"2026-09-20T10:17:53.564321Z","iopub.status.idle":"2026-09-20T10:17:53.596094Z","shell.execute_reply.started":"2026-09-20T10:17:53.564297Z","shell.execute_reply":"2026-09-20T10:17:53.595231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.describe(include='all')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.597088Z","iopub.execute_input":"2026-09-20T10:17:53.597382Z","iopub.status.idle":"2026-09-20T10:17:53.638288Z","shell.execute_reply.started":"2026-09-20T10:17:53.597347Z","shell.execute_reply":"2026-09-20T10:17:53.637645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.corr(numeric_only = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.639136Z","iopub.execute_input":"2026-09-20T10:17:53.639438Z","iopub.status.idle":"2026-09-20T10:17:53.662828Z","shell.execute_reply.started":"2026-09-20T10:17:53.639413Z","shell.execute_reply":"2026-09-20T10:17:53.662045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation matrix\ncorr = train_dataset.corr(numeric_only=True)\n\n# Heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr, annot=True, cmap=\"coolwarm\", fmt=\".2f\")\n\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:53.663657Z","iopub.execute_input":"2026-09-20T10:17:53.663844Z","iopub.status.idle":"2026-09-20T10:17:54.102384Z","shell.execute_reply.started":"2026-09-20T10:17:53.663825Z","shell.execute_reply":"2026-09-20T10:17:54.101695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution \nnumeric_cols = train_dataset.select_dtypes(include=\"number\").columns\nfor i in numeric_cols:\n    sns.histplot(train_dataset[i], kde=True)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:54.103319Z","iopub.execute_input":"2026-09-20T10:17:54.103709Z","iopub.status.idle":"2026-09-20T10:17:55.913232Z","shell.execute_reply.started":"2026-09-20T10:17:54.103672Z","shell.execute_reply":"2026-09-20T10:17:55.912559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in numeric_cols:\n    sns.boxplot(x=train_dataset[i])\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:55.913932Z","iopub.execute_input":"2026-09-20T10:17:55.914195Z","iopub.status.idle":"2026-09-20T10:17:56.845361Z","shell.execute_reply.started":"2026-09-20T10:17:55.914168Z","shell.execute_reply":"2026-09-20T10:17:56.844662Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training dataset series","metadata":{}},{"cell_type":"code","source":"train_dataset_series = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/train_series.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:56.846262Z","iopub.execute_input":"2026-09-20T10:17:56.846654Z","iopub.status.idle":"2026-09-20T10:17:56.913903Z","shell.execute_reply.started":"2026-09-20T10:17:56.846609Z","shell.execute_reply":"2026-09-20T10:17:56.913026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:56.914885Z","iopub.execute_input":"2026-09-20T10:17:56.915262Z","iopub.status.idle":"2026-09-20T10:17:56.923343Z","shell.execute_reply.started":"2026-09-20T10:17:56.915224Z","shell.execute_reply":"2026-09-20T10:17:56.922523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:56.924436Z","iopub.execute_input":"2026-09-20T10:17:56.924775Z","iopub.status.idle":"2026-09-20T10:17:56.944314Z","shell.execute_reply.started":"2026-09-20T10:17:56.92475Z","shell.execute_reply":"2026-09-20T10:17:56.943632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:56.945251Z","iopub.execute_input":"2026-09-20T10:17:56.94556Z","iopub.status.idle":"2026-09-20T10:17:56.963637Z","shell.execute_reply.started":"2026-09-20T10:17:56.945523Z","shell.execute_reply":"2026-09-20T10:17:56.962844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.describe(include ='all')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:56.964561Z","iopub.execute_input":"2026-09-20T10:17:56.964855Z","iopub.status.idle":"2026-09-20T10:17:56.999264Z","shell.execute_reply.started":"2026-09-20T10:17:56.964821Z","shell.execute_reply":"2026-09-20T10:17:56.998421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:57.00012Z","iopub.execute_input":"2026-09-20T10:17:57.000361Z","iopub.status.idle":"2026-09-20T10:17:57.005688Z","shell.execute_reply.started":"2026-09-20T10:17:57.000337Z","shell.execute_reply":"2026-09-20T10:17:57.004664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_series.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:57.006714Z","iopub.execute_input":"2026-09-20T10:17:57.007672Z","iopub.status.idle":"2026-09-20T10:17:57.020373Z","shell.execute_reply.started":"2026-09-20T10:17:57.007637Z","shell.execute_reply":"2026-09-20T10:17:57.019663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Series (images)","metadata":{}},{"cell_type":"code","source":"import os\n\nfolder = \"/kaggle/input/competitions/rsna-knee-abnormality-detection/test_series/\"\n\ndcm_files_test = []\n\nfor root, dirs, files in os.walk(folder):\n    for file in files:\n        if file.endswith(\".dcm\"):\n            dcm_files_test.append(os.path.join(root, file))\n\nprint(len(dcm_files_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:57.021159Z","iopub.execute_input":"2026-09-20T10:17:57.021751Z","iopub.status.idle":"2026-09-20T10:17:57.227431Z","shell.execute_reply.started":"2026-09-20T10:17:57.021725Z","shell.execute_reply":"2026-09-20T10:17:57.226449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\nfor file in dcm_files_test[:5]:\n    dcm = pydicom.dcmread(file)\n    image = dcm.pixel_array\n\n    plt.figure(figsize=(5, 5))\n    plt.imshow(image, cmap=\"gray\")\n    plt.title(file)\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:57.228473Z","iopub.execute_input":"2026-09-20T10:17:57.228776Z","iopub.status.idle":"2026-09-20T10:17:58.304864Z","shell.execute_reply.started":"2026-09-20T10:17:57.228741Z","shell.execute_reply":"2026-09-20T10:17:58.304032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training with AI","metadata":{}},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.model_download(\"metaresearch/dinov2/pyTorch/large\")\n\nprint(\"Path to model files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:58.305875Z","iopub.execute_input":"2026-09-20T10:17:58.306224Z","iopub.status.idle":"2026-09-20T10:17:58.563209Z","shell.execute_reply.started":"2026-09-20T10:17:58.306197Z","shell.execute_reply":"2026-09-20T10:17:58.562522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ai_model = \"/kaggle/input/models/metaresearch/dinov2/pytorch/large/1\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:58.564005Z","iopub.execute_input":"2026-09-20T10:17:58.565065Z","iopub.status.idle":"2026-09-20T10:17:58.568547Z","shell.execute_reply.started":"2026-09-20T10:17:58.565021Z","shell.execute_reply":"2026-09-20T10:17:58.567912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(ai_model))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:58.573329Z","iopub.execute_input":"2026-09-20T10:17:58.573609Z","iopub.status.idle":"2026-09-20T10:17:58.585419Z","shell.execute_reply.started":"2026-09-20T10:17:58.573586Z","shell.execute_reply":"2026-09-20T10:17:58.584614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoModel\n\ndino = AutoModel.from_pretrained(ai_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:17:58.586403Z","iopub.execute_input":"2026-09-20T10:17:58.586633Z","iopub.status.idle":"2026-09-20T10:18:01.220227Z","shell.execute_reply.started":"2026-09-20T10:17:58.586588Z","shell.execute_reply":"2026-09-20T10:18:01.216038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dino = dino.to(\"cuda\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:01.221289Z","iopub.execute_input":"2026-09-20T10:18:01.221604Z","iopub.status.idle":"2026-09-20T10:18:09.797283Z","shell.execute_reply.started":"2026-09-20T10:18:01.221579Z","shell.execute_reply":"2026-09-20T10:18:09.796575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(dino)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:09.798226Z","iopub.execute_input":"2026-09-20T10:18:09.798487Z","iopub.status.idle":"2026-09-20T10:18:09.804679Z","shell.execute_reply.started":"2026-09-20T10:18:09.798463Z","shell.execute_reply":"2026-09-20T10:18:09.803926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\ndcm = pydicom.dcmread(dcm_files_test[0])\n\nimage = dcm.pixel_array\n\nprint(image.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:09.805566Z","iopub.execute_input":"2026-09-20T10:18:09.805929Z","iopub.status.idle":"2026-09-20T10:18:09.829117Z","shell.execute_reply.started":"2026-09-20T10:18:09.805902Z","shell.execute_reply":"2026-09-20T10:18:09.828465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\n\nimage = image.astype(np.float32)\n\nimage = (image - image.min()) / (image.max() - image.min())\n\nimage = (image * 255).astype(np.uint8)\n\nimage = Image.fromarray(image).convert(\"RGB\")\n\nprint(image.size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:09.830103Z","iopub.execute_input":"2026-09-20T10:18:09.830507Z","iopub.status.idle":"2026-09-20T10:18:09.843807Z","shell.execute_reply.started":"2026-09-20T10:18:09.830469Z","shell.execute_reply":"2026-09-20T10:18:09.843125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoImageProcessor\n\nprocessor = AutoImageProcessor.from_pretrained(ai_model)\n\ninputs = processor(\n    images=image,\n    return_tensors=\"pt\"\n)\n\nprint(inputs[\"pixel_values\"].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:09.844899Z","iopub.execute_input":"2026-09-20T10:18:09.84522Z","iopub.status.idle":"2026-09-20T10:18:09.908567Z","shell.execute_reply.started":"2026-09-20T10:18:09.845184Z","shell.execute_reply":"2026-09-20T10:18:09.907901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs = {k: v.to(\"cuda\") for k, v in inputs.items()}\n\nwith torch.no_grad():\n    outputs = dino(**inputs)\n\nfeatures = outputs.last_hidden_state[:, 0]\n\nprint(features.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:09.909348Z","iopub.execute_input":"2026-09-20T10:18:09.909698Z","iopub.status.idle":"2026-09-20T10:18:10.757131Z","shell.execute_reply.started":"2026-09-20T10:18:09.909673Z","shell.execute_reply":"2026-09-20T10:18:10.756349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training with Qwen","metadata":{}},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.model_download(\"qwen-lm/qwen2.5/transformers/1.5b\")\n\nprint(\"Path to model files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:10.758166Z","iopub.execute_input":"2026-09-20T10:18:10.758462Z","iopub.status.idle":"2026-09-20T10:18:11.234073Z","shell.execute_reply.started":"2026-09-20T10:18:10.758429Z","shell.execute_reply":"2026-09-20T10:18:11.23333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ai_model = \"/kaggle/input/models/qwen-lm/qwen2.5/transformers/1.5b/1\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.235097Z","iopub.execute_input":"2026-09-20T10:18:11.235429Z","iopub.status.idle":"2026-09-20T10:18:11.239925Z","shell.execute_reply.started":"2026-09-20T10:18:11.235404Z","shell.execute_reply":"2026-09-20T10:18:11.239352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# yalnız 12 label-i olan 58 sətri götürəcəyik\nlabeled_data = train_dataset.dropna()\n\nprint(labeled_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.240752Z","iopub.execute_input":"2026-09-20T10:18:11.240944Z","iopub.status.idle":"2026-09-20T10:18:11.258756Z","shell.execute_reply.started":"2026-09-20T10:18:11.240923Z","shell.execute_reply":"2026-09-20T10:18:11.258123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\nqwen_data = []\n\nfor _, row in labeled_data.iterrows():\n    qwen_data.append({\n        \"input\": row[\"Report\"],\n        \"output\": {label: int(row[label]) for label in labels}\n    })\n\nprint(qwen_data[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.259798Z","iopub.execute_input":"2026-09-20T10:18:11.26041Z","iopub.status.idle":"2026-09-20T10:18:11.278894Z","shell.execute_reply.started":"2026-09-20T10:18:11.260379Z","shell.execute_reply":"2026-09-20T10:18:11.278132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Research about tokenizers","metadata":{}},{"cell_type":"markdown","source":"## Creating my own tokenizer","metadata":{}},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/input/datasets/thesharifzade/pharmaconerdataset\"):\n    print(root)\n    print(files[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.27969Z","iopub.execute_input":"2026-09-20T10:18:11.280412Z","iopub.status.idle":"2026-09-20T10:18:11.30059Z","shell.execute_reply.started":"2026-09-20T10:18:11.280386Z","shell.execute_reply":"2026-09-20T10:18:11.299767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# so we use the sentencepiece\n# download the dataset\nfrom datasets import load_from_disk\n\ncantemistnerdataset = load_from_disk(\n    \"/kaggle/input/datasets/thesharifzade/cantemistdatasetfile/cantemist-ner\"\n)\n\nprint(cantemistnerdataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.301503Z","iopub.execute_input":"2026-09-20T10:18:11.301938Z","iopub.status.idle":"2026-09-20T10:18:11.338659Z","shell.execute_reply.started":"2026-09-20T10:18:11.301902Z","shell.execute_reply":"2026-09-20T10:18:11.338106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# upload pharmaconerdataset\nfrom datasets import load_from_disk\npharmaconerdataset = load_from_disk(\"/kaggle/input/datasets/thesharifzade/pharmaconerdataset/pharmaco-ner\")\nprint(pharmaconerdataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.339488Z","iopub.execute_input":"2026-09-20T10:18:11.339778Z","iopub.status.idle":"2026-09-20T10:18:11.369177Z","shell.execute_reply.started":"2026-09-20T10:18:11.339733Z","shell.execute_reply":"2026-09-20T10:18:11.368532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import concatenate_datasets\n\ntrain1 = cantemistnerdataset[\"train\"].remove_columns([\"ner_tags\"])\ntrain2 = pharmaconerdataset[\"train\"].remove_columns([\"ner_tags\"])\n\ntrain_ds = concatenate_datasets([\n    train1,\n    train2\n])\n\nprint(train_ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.369872Z","iopub.execute_input":"2026-09-20T10:18:11.370121Z","iopub.status.idle":"2026-09-20T10:18:11.381137Z","shell.execute_reply.started":"2026-09-20T10:18:11.370099Z","shell.execute_reply":"2026-09-20T10:18:11.380455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_ds artıq hazırdır\n\ntexts = [\n    \" \".join(example[\"tokens\"])\n    for example in train_ds\n]\n\nwith open(\"/kaggle/working/cantemist_train.txt\", \"w\", encoding=\"utf-8\") as f:\n    for text in texts:\n        f.write(text + \"\\n\")\n\nprint(\"Samples:\", len(texts))\nprint(texts[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:11.382115Z","iopub.execute_input":"2026-09-20T10:18:11.382478Z","iopub.status.idle":"2026-09-20T10:18:12.594705Z","shell.execute_reply.started":"2026-09-20T10:18:11.382442Z","shell.execute_reply":"2026-09-20T10:18:12.593784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sentencepiece as spm\n\nspm.SentencePieceTrainer.train(\n    input=\"/kaggle/working/cantemist_train.txt\",\n    model_prefix=\"/kaggle/working/cantemist_tokenizer\",\n    model_type=\"bpe\",\n    vocab_size=8000,\n    character_coverage=0.99995,\n\n    normalization_rule_name=\"identity\",\n    add_dummy_prefix=True,\n    remove_extra_whitespaces=False,\n\n    split_by_unicode_script=True,\n    split_by_whitespace=True,\n    split_by_number=True,\n    split_digits=True,\n\n    max_sentencepiece_length=16,\n    allow_whitespace_only_pieces=True,\n    byte_fallback=True,\n\n    unk_id=0,\n    bos_id=1,\n    eos_id=2,\n    pad_id=3,\n\n    unk_piece=\"<unk>\",\n    bos_piece=\"<s>\",\n    eos_piece=\"</s>\",\n    pad_piece=\"<pad>\",\n\n    hard_vocab_limit=True,\n    input_sentence_size=0,\n    shuffle_input_sentence=True,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:12.595787Z","iopub.execute_input":"2026-09-20T10:18:12.596257Z","iopub.status.idle":"2026-09-20T10:18:13.938636Z","shell.execute_reply.started":"2026-09-20T10:18:12.596197Z","shell.execute_reply":"2026-09-20T10:18:13.936844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sp = spm.SentencePieceProcessor(\n    model_file=\"/kaggle/working/cantemist_tokenizer.model\"\n)\n\ntext = texts[0]\n\nprint(sp.encode(text, out_type=str))\nprint(sp.encode(text, out_type=int))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:13.94174Z","iopub.execute_input":"2026-09-20T10:18:13.942172Z","iopub.status.idle":"2026-09-20T10:18:13.956879Z","shell.execute_reply.started":"2026-09-20T10:18:13.942131Z","shell.execute_reply":"2026-09-20T10:18:13.956002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = AutoModelForCausalLM.from_pretrained(\n    ai_model,\n    torch_dtype=torch.float16\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:13.957865Z","iopub.execute_input":"2026-09-20T10:18:13.95883Z","iopub.status.idle":"2026-09-20T10:18:15.808211Z","shell.execute_reply.started":"2026-09-20T10:18:13.958792Z","shell.execute_reply":"2026-09-20T10:18:15.807209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qwen_messages = []\n\nfor item in qwen_data:\n    qwen_messages.append({\n        \"messages\": [\n            {\n                \"role\": \"system\",\n                \"content\": (\n                    \"Read the medical report and predict the 12 labels. \"\n                    \"For each label, output only 0 or 1.\"\n                )\n            },\n            {\n                \"role\": \"user\",\n                \"content\": item[\"input\"]\n            },\n            {\n                \"role\": \"assistant\",\n                \"content\": str(item[\"output\"])\n            }\n        ]\n    })\n\nprint(qwen_messages[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:15.809289Z","iopub.execute_input":"2026-09-20T10:18:15.809589Z","iopub.status.idle":"2026-09-20T10:18:15.815471Z","shell.execute_reply.started":"2026-09-20T10:18:15.809564Z","shell.execute_reply":"2026-09-20T10:18:15.814866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import Dataset\n\ndataset = Dataset.from_list(qwen_messages)\n\nprint(dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:15.816401Z","iopub.execute_input":"2026-09-20T10:18:15.816731Z","iopub.status.idle":"2026-09-20T10:18:15.842022Z","shell.execute_reply.started":"2026-09-20T10:18:15.816707Z","shell.execute_reply":"2026-09-20T10:18:15.841245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tokenize(example):\n    text = \"\"\n\n    for message in example[\"messages\"]:\n        text += message[\"content\"] + \"\\n\"\n\n    input_ids = sp.encode(text, out_type=int)\n\n    input_ids = input_ids[:1024]\n\n    padding_length = 1024 - len(input_ids)\n\n    input_ids += [0] * padding_length\n\n    attention_mask = [1] * (1024 - padding_length) + [0] * padding_length\n\n    labels = input_ids.copy()\n\n    labels = [\n        token if mask == 1 else -100\n        for token, mask in zip(labels, attention_mask)\n    ]\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_mask\": attention_mask,\n        \"labels\": labels\n    }\n\ntokenized_dataset = dataset.map(\n    tokenize,\n    remove_columns=dataset.column_names\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:15.843125Z","iopub.execute_input":"2026-09-20T10:18:15.843463Z","iopub.status.idle":"2026-09-20T10:18:15.975055Z","shell.execute_reply.started":"2026-09-20T10:18:15.843427Z","shell.execute_reply":"2026-09-20T10:18:15.974139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"input_ids\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:15.97593Z","iopub.execute_input":"2026-09-20T10:18:15.976187Z","iopub.status.idle":"2026-09-20T10:18:15.981648Z","shell.execute_reply.started":"2026-09-20T10:18:15.976164Z","shell.execute_reply":"2026-09-20T10:18:15.980875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# input ids -> result of tokenization(with ids)\n# attention_mask -> which padding, which real token\n# label -> the target\nprint(tokenized_dataset.column_names)\nprint(tokenized_dataset[0].keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:15.982675Z","iopub.execute_input":"2026-09-20T10:18:15.982983Z","iopub.status.idle":"2026-09-20T10:18:15.999426Z","shell.execute_reply.started":"2026-09-20T10:18:15.98293Z","shell.execute_reply":"2026-09-20T10:18:15.998642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"input_ids:\", len(tokenized_dataset[0][\"input_ids\"]))\nprint(\"attention_mask:\", len(tokenized_dataset[0][\"attention_mask\"]))\nprint(\"labels:\", len(tokenized_dataset[0][\"labels\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:16.000384Z","iopub.execute_input":"2026-09-20T10:18:16.000765Z","iopub.status.idle":"2026-09-20T10:18:16.02116Z","shell.execute_reply.started":"2026-09-20T10:18:16.000731Z","shell.execute_reply":"2026-09-20T10:18:16.020497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"input_ids:\", len(tokenized_dataset[0][\"input_ids\"]))\nprint(\"attention_mask:\", len(tokenized_dataset[0][\"attention_mask\"]))\nprint(\"labels:\", len(tokenized_dataset[0][\"labels\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:16.022004Z","iopub.execute_input":"2026-09-20T10:18:16.022238Z","iopub.status.idle":"2026-09-20T10:18:16.041804Z","shell.execute_reply.started":"2026-09-20T10:18:16.022217Z","shell.execute_reply":"2026-09-20T10:18:16.040815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PEFT method : free entire LLM, add tiny layers\nfrom peft import LoraConfig, get_peft_model\n\nlora_config = LoraConfig(\n    r=8,\n    lora_alpha=16,\n    lora_dropout=0.05,\n    target_modules=[\n        \"q_proj\",\n        \"k_proj\",\n        \"v_proj\",\n        \"o_proj\"\n    ],\n    bias=\"none\",\n    task_type=\"CAUSAL_LM\"\n)\n\nmodel = get_peft_model(model, lora_config)\n\nmodel.print_trainable_parameters()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:18:16.042727Z","iopub.execute_input":"2026-09-20T10:18:16.043197Z","iopub.status.idle":"2026-09-20T10:18:16.613573Z","shell.execute_reply.started":"2026-09-20T10:18:16.043159Z","shell.execute_reply":"2026-09-20T10:18:16.612828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"./qwen_knee\",\n    num_train_epochs=2,\n    per_device_train_batch_size=1,\n    gradient_accumulation_steps=32,\n    learning_rate=2e-4,\n    logging_steps=1,\n    logging_strategy = \"steps\",\n    save_strategy=\"epoch\",\n    fp16=True,\n    report_to=\"none\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:03.73868Z","iopub.execute_input":"2026-09-20T10:20:03.739507Z","iopub.status.idle":"2026-09-20T10:20:03.776128Z","shell.execute_reply.started":"2026-09-20T10:20:03.739473Z","shell.execute_reply":"2026-09-20T10:20:03.775474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_dataset,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:04.428172Z","iopub.execute_input":"2026-09-20T10:20:04.428969Z","iopub.status.idle":"2026-09-20T10:20:04.44883Z","shell.execute_reply.started":"2026-09-20T10:20:04.428922Z","shell.execute_reply":"2026-09-20T10:20:04.44803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:04.634528Z","iopub.execute_input":"2026-09-20T10:20:04.634814Z","iopub.status.idle":"2026-09-20T10:20:05.952839Z","shell.execute_reply.started":"2026-09-20T10:20:04.634785Z","shell.execute_reply":"2026-09-20T10:20:05.951728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nlogs = trainer.state.log_history\n\nsteps = []\nlosses = []\n\nfor log in logs:\n    if \"loss\" in log:\n        steps.append(log[\"step\"])\n        losses.append(log[\"loss\"])\n\nplt.plot(steps, losses)\nplt.xlabel(\"Training Steps\")\nplt.ylabel(\"Loss\")\nplt.title(\"Training Loss\")\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:06.09828Z","iopub.execute_input":"2026-09-20T10:20:06.098504Z","iopub.status.idle":"2026-09-20T10:20:06.198644Z","shell.execute_reply.started":"2026-09-20T10:20:06.098482Z","shell.execute_reply":"2026-09-20T10:20:06.197985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(trainer.state.log_history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:06.593408Z","iopub.execute_input":"2026-09-20T10:20:06.593911Z","iopub.status.idle":"2026-09-20T10:20:06.598555Z","shell.execute_reply.started":"2026-09-20T10:20:06.593883Z","shell.execute_reply":"2026-09-20T10:20:06.5977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.save_model(\"./finetuningazesavedversion\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.009161Z","iopub.status.idle":"2026-09-20T10:20:28.009514Z","shell.execute_reply.started":"2026-09-20T10:20:28.00932Z","shell.execute_reply":"2026-09-20T10:20:28.009344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check the fine tuned version (is working yes or no?)","metadata":{}},{"cell_type":"code","source":"file_path = \"/kaggle/input/datasets/thesharifzade/finetuningazerbaijanversionrsnaknee/finetuningazesavedversion\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:06.863152Z","iopub.execute_input":"2026-09-20T10:20:06.863512Z","iopub.status.idle":"2026-09-20T10:20:06.867391Z","shell.execute_reply.started":"2026-09-20T10:20:06.863484Z","shell.execute_reply":"2026-09-20T10:20:06.866468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(filepath))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:06.98142Z","iopub.execute_input":"2026-09-20T10:20:06.981946Z","iopub.status.idle":"2026-09-20T10:20:06.987175Z","shell.execute_reply.started":"2026-09-20T10:20:06.981907Z","shell.execute_reply":"2026-09-20T10:20:06.986393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoModelForCausalLM, AutoTokenizer\nfrom peft import PeftModel\n\nbase_model = \"/kaggle/input/models/qwen-lm/qwen2.5/transformers/1.5b/1\"\n\ntokenizer = AutoTokenizer.from_pretrained(base_model)\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    base_model,\n    torch_dtype=\"auto\",\n    device_map=\"auto\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:07.089691Z","iopub.execute_input":"2026-09-20T10:20:07.090105Z","iopub.status.idle":"2026-09-20T10:20:09.120474Z","shell.execute_reply.started":"2026-09-20T10:20:07.090078Z","shell.execute_reply":"2026-09-20T10:20:09.119817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmodel = PeftModel.from_pretrained(\n    model,\n    file_path\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:09.121811Z","iopub.execute_input":"2026-09-20T10:20:09.12219Z","iopub.status.idle":"2026-09-20T10:20:09.308022Z","shell.execute_reply.started":"2026-09-20T10:20:09.122154Z","shell.execute_reply":"2026-09-20T10:20:09.307112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prompt = \"Your test prompt here\"\n\ninputs = tokenizer(prompt, return_tensors=\"pt\").to(model.device)\n\noutputs = model.generate(\n    **inputs,\n    max_new_tokens=100\n)\n\nprint(tokenizer.decode(outputs[0], skip_special_tokens=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:09.309059Z","iopub.execute_input":"2026-09-20T10:20:09.309364Z","iopub.status.idle":"2026-09-20T10:20:15.871584Z","shell.execute_reply.started":"2026-09-20T10:20:09.309333Z","shell.execute_reply":"2026-09-20T10:20:15.870879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check the fine tuned version for train labels with actual data and prediction ","metadata":{}},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:15.873463Z","iopub.execute_input":"2026-09-20T10:20:15.873778Z","iopub.status.idle":"2026-09-20T10:20:15.960602Z","shell.execute_reply.started":"2026-09-20T10:20:15.873751Z","shell.execute_reply":"2026-09-20T10:20:15.95988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# yalnız 12 label-i olan 58 sətri götürəcəyik\nlabeled_data = train_dataset.dropna()\n\nprint(labeled_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:15.961468Z","iopub.execute_input":"2026-09-20T10:20:15.961722Z","iopub.status.idle":"2026-09-20T10:20:15.968506Z","shell.execute_reply.started":"2026-09-20T10:20:15.961697Z","shell.execute_reply":"2026-09-20T10:20:15.967547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\nqwen_data = []\n\nfor _, row in labeled_data.iterrows():\n    qwen_data.append({\n        \"input\": row[\"Report\"],\n        \"output\": {label: int(row[label]) for label in labels}\n    })\n\nprint(qwen_data[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:15.970189Z","iopub.execute_input":"2026-09-20T10:20:15.970472Z","iopub.status.idle":"2026-09-20T10:20:15.987643Z","shell.execute_reply.started":"2026-09-20T10:20:15.970449Z","shell.execute_reply":"2026-09-20T10:20:15.987036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qwen_messages = []\n\nfor item in qwen_data:\n    qwen_messages.append({\n        \"messages\": [\n            {\n                \"role\": \"system\",\n                \"content\": (\n                    \"Read the medical report and predict the 12 labels. \"\n                    \"For each label, output only 0 or 1.\"\n                )\n            },\n            {\n                \"role\": \"user\",\n                \"content\": item[\"input\"]\n            },\n            {\n                \"role\": \"assistant\",\n                \"content\": str(item[\"output\"])\n            }\n        ]\n    })\n\nprint(qwen_messages[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:15.988658Z","iopub.execute_input":"2026-09-20T10:20:15.98904Z","iopub.status.idle":"2026-09-20T10:20:16.002748Z","shell.execute_reply.started":"2026-09-20T10:20:15.988994Z","shell.execute_reply":"2026-09-20T10:20:16.002005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import Dataset, concatenate_datasets\n\n# 1. Qwen dataset\ndataset = Dataset.from_list(qwen_messages)\n\nprint(dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.003744Z","iopub.execute_input":"2026-09-20T10:20:16.004065Z","iopub.status.idle":"2026-09-20T10:20:16.023158Z","shell.execute_reply.started":"2026-09-20T10:20:16.00402Z","shell.execute_reply":"2026-09-20T10:20:16.022321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tokenize(example):\n    text = \"\"\n\n    for message in example[\"messages\"]:\n        text += message[\"content\"] + \"\\n\"\n\n    input_ids = sp.encode(text, out_type=int)\n\n    input_ids = input_ids[:1024]\n\n    padding_length = 1024 - len(input_ids)\n\n    input_ids += [0] * padding_length\n\n    attention_mask = [1] * (1024 - padding_length) + [0] * padding_length\n\n    labels = input_ids.copy()\n\n    labels = [\n        token if mask == 1 else -100\n        for token, mask in zip(labels, attention_mask)\n    ]\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_mask\": attention_mask,\n        \"labels\": labels\n    }\n\ntokenized_dataset = dataset.map(\n    tokenize,\n    remove_columns=dataset.column_names\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.024197Z","iopub.execute_input":"2026-09-20T10:20:16.024492Z","iopub.status.idle":"2026-09-20T10:20:16.163316Z","shell.execute_reply.started":"2026-09-20T10:20:16.024469Z","shell.execute_reply":"2026-09-20T10:20:16.162551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# input ids -> result of tokenization(with ids)\n# attention_mask -> which padding, which real token\n# label -> the target\nprint(tokenized_dataset.column_names)\nprint(tokenized_dataset[0].keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.165478Z","iopub.execute_input":"2026-09-20T10:20:16.166157Z","iopub.status.idle":"2026-09-20T10:20:16.171672Z","shell.execute_reply.started":"2026-09-20T10:20:16.166131Z","shell.execute_reply":"2026-09-20T10:20:16.171009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"input_ids\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.172463Z","iopub.execute_input":"2026-09-20T10:20:16.172942Z","iopub.status.idle":"2026-09-20T10:20:16.189344Z","shell.execute_reply.started":"2026-09-20T10:20:16.172918Z","shell.execute_reply":"2026-09-20T10:20:16.18872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"attention_mask\"][:5])\nprint(tokenized_dataset[0][\"labels\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.190107Z","iopub.execute_input":"2026-09-20T10:20:16.190404Z","iopub.status.idle":"2026-09-20T10:20:16.207541Z","shell.execute_reply.started":"2026-09-20T10:20:16.190381Z","shell.execute_reply":"2026-09-20T10:20:16.206728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_report = qwen_data[0][\"input\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.208452Z","iopub.execute_input":"2026-09-20T10:20:16.208804Z","iopub.status.idle":"2026-09-20T10:20:16.222891Z","shell.execute_reply.started":"2026-09-20T10:20:16.208766Z","shell.execute_reply":"2026-09-20T10:20:16.22232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.223607Z","iopub.execute_input":"2026-09-20T10:20:16.223785Z","iopub.status.idle":"2026-09-20T10:20:16.239211Z","shell.execute_reply.started":"2026-09-20T10:20:16.223767Z","shell.execute_reply":"2026-09-20T10:20:16.238514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_report = qwen_data[0][\"input\"]\n\nmessages = [\n    {\n        \"role\": \"system\",\n        \"content\": (\n            \"Read the medical report and output only the 12 labels as JSON. \"\n            \"Use only 0 or 1.\"\n        )\n    },\n    {\n        \"role\": \"user\",\n        \"content\": test_report\n    }\n]\n\nprompt = tokenizer.apply_chat_template(\n    messages,\n    tokenize=False,\n    add_generation_prompt=True\n)\n\nprint(\"PROMPT:\")\nprint(prompt)\n\ninputs = tokenizer(\n    prompt,\n    return_tensors=\"pt\"\n).to(model.device)\n\nprint(\"INPUT SHAPE:\", inputs[\"input_ids\"].shape)\n\nmodel.eval()\n\nwith torch.no_grad():\n    output = model.generate(\n        **inputs,\n        max_new_tokens=150,\n        do_sample=False,\n        pad_token_id=tokenizer.eos_token_id\n    )\n\nprint(\"OUTPUT SHAPE:\", output.shape)\n\nnew_tokens = output[0][inputs[\"input_ids\"].shape[1]:]\n\nprint(\"NEW TOKENS:\", new_tokens)\n\nresponse = tokenizer.decode(\n    new_tokens,\n    skip_special_tokens=False\n)\n\nprint(\"RAW RESPONSE:\")\nprint(repr(response))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:16.240262Z","iopub.execute_input":"2026-09-20T10:20:16.241033Z","iopub.status.idle":"2026-09-20T10:20:26.285618Z","shell.execute_reply.started":"2026-09-20T10:20:16.240992Z","shell.execute_reply":"2026-09-20T10:20:26.284977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# First we use Qwen tokenizer","metadata":{}},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.model_download(\"qwen-lm/qwen2.5/transformers/1.5b\")\n\nprint(\"Path to model files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:26.286595Z","iopub.execute_input":"2026-09-20T10:20:26.287153Z","iopub.status.idle":"2026-09-20T10:20:26.65467Z","shell.execute_reply.started":"2026-09-20T10:20:26.287125Z","shell.execute_reply":"2026-09-20T10:20:26.654034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ai_model = \"/kaggle/input/models/qwen-lm/qwen2.5/transformers/1.5b/1\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:26.65557Z","iopub.execute_input":"2026-09-20T10:20:26.655828Z","iopub.status.idle":"2026-09-20T10:20:26.659536Z","shell.execute_reply.started":"2026-09-20T10:20:26.655803Z","shell.execute_reply":"2026-09-20T10:20:26.658905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(ai_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:26.660632Z","iopub.execute_input":"2026-09-20T10:20:26.661362Z","iopub.status.idle":"2026-09-20T10:20:27.471315Z","shell.execute_reply.started":"2026-09-20T10:20:26.661323Z","shell.execute_reply":"2026-09-20T10:20:27.47042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = AutoModelForCausalLM.from_pretrained(\n    ai_model,\n    torch_dtype=\"auto\",\n    device_map=\"auto\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:27.472317Z","iopub.execute_input":"2026-09-20T10:20:27.472679Z","iopub.status.idle":"2026-09-20T10:20:27.994129Z","shell.execute_reply.started":"2026-09-20T10:20:27.472625Z","shell.execute_reply":"2026-09-20T10:20:27.992881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(model.device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:27.994723Z","iopub.status.idle":"2026-09-20T10:20:27.995048Z","shell.execute_reply.started":"2026-09-20T10:20:27.994877Z","shell.execute_reply":"2026-09-20T10:20:27.994893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:27.996125Z","iopub.status.idle":"2026-09-20T10:20:27.996351Z","shell.execute_reply.started":"2026-09-20T10:20:27.996241Z","shell.execute_reply":"2026-09-20T10:20:27.996255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# yalnız 12 label-i olan 58 sətri götürəcəyik\nlabeled_data = train_dataset.dropna()\n\nprint(labeled_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:27.997853Z","iopub.status.idle":"2026-09-20T10:20:27.998217Z","shell.execute_reply.started":"2026-09-20T10:20:27.998051Z","shell.execute_reply":"2026-09-20T10:20:27.998075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\nqwen_data = []\n\nfor _, row in labeled_data.iterrows():\n    qwen_data.append({\n        \"input\": row[\"Report\"],\n        \"output\": {label: int(row[label]) for label in labels}\n    })\n\nprint(qwen_data[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:27.999261Z","iopub.status.idle":"2026-09-20T10:20:27.999627Z","shell.execute_reply.started":"2026-09-20T10:20:27.999432Z","shell.execute_reply":"2026-09-20T10:20:27.999455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qwen_messages = []\n\nfor item in qwen_data:\n    qwen_messages.append({\n        \"messages\": [\n            {\n                \"role\": \"system\",\n                \"content\": (\n                    \"Read the medical report and predict the 12 labels. \"\n                    \"For each label, output only 0 or 1.\"\n                )\n            },\n            {\n                \"role\": \"user\",\n                \"content\": item[\"input\"]\n            },\n            {\n                \"role\": \"assistant\",\n                \"content\": str(item[\"output\"])\n            }\n        ]\n    })\n\nprint(qwen_messages[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.001029Z","iopub.status.idle":"2026-09-20T10:20:28.001272Z","shell.execute_reply.started":"2026-09-20T10:20:28.001158Z","shell.execute_reply":"2026-09-20T10:20:28.001173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import Dataset, concatenate_datasets\n\n# 1. Qwen dataset\ndataset = Dataset.from_list(qwen_messages)\n\n# 2. CANTEMIST\ntrain1 = cantemistnerdataset[\"train\"].remove_columns([\"ner_tags\"])\n\n# 3. PharmaCoNER\ntrain2 = pharmaconerdataset[\"train\"].remove_columns([\"ner_tags\"])\n\n# 4. Hamısını birləşdir\ntrain_ds = concatenate_datasets([\n    dataset,\n    train1,\n    train2\n])\n\nprint(train_ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.002288Z","iopub.status.idle":"2026-09-20T10:20:28.002579Z","shell.execute_reply.started":"2026-09-20T10:20:28.00244Z","shell.execute_reply":"2026-09-20T10:20:28.002455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# For supervised fine tuning learning","metadata":{}},{"cell_type":"code","source":"def tokenize(example):\n    text = tokenizer.apply_chat_template(\n        example[\"messages\"],\n        tokenize=False,\n        add_generation_prompt=False\n    )\n\n    result = tokenizer(\n        text,\n        truncation=True,\n        max_length=1024,\n        padding=False\n    )\n\n    result[\"labels\"] = result[\"input_ids\"].copy()\n\n    return result\n\n\ntokenized_dataset = dataset.map(\n    tokenize,\n    remove_columns=dataset.column_names\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.003905Z","iopub.status.idle":"2026-09-20T10:20:28.004281Z","shell.execute_reply.started":"2026-09-20T10:20:28.004116Z","shell.execute_reply":"2026-09-20T10:20:28.004134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# input ids -> result of tokenization(with ids)\n# attention_mask -> which padding, which real token\n# label -> the target\nprint(tokenized_dataset.column_names)\nprint(tokenized_dataset[0].keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.005188Z","iopub.status.idle":"2026-09-20T10:20:28.0055Z","shell.execute_reply.started":"2026-09-20T10:20:28.005374Z","shell.execute_reply":"2026-09-20T10:20:28.005392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"input_ids\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.00637Z","iopub.status.idle":"2026-09-20T10:20:28.0066Z","shell.execute_reply.started":"2026-09-20T10:20:28.006487Z","shell.execute_reply":"2026-09-20T10:20:28.006501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"attention_mask\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:28.007632Z","iopub.status.idle":"2026-09-20T10:20:28.00805Z","shell.execute_reply.started":"2026-09-20T10:20:28.007832Z","shell.execute_reply":"2026-09-20T10:20:28.007856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tokenized_dataset[0][\"labels\"][:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.125427Z","iopub.status.idle":"2026-09-20T10:20:01.125653Z","shell.execute_reply.started":"2026-09-20T10:20:01.125545Z","shell.execute_reply":"2026-09-20T10:20:01.125559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from peft import LoraConfig, get_peft_model\n\nlora_config = LoraConfig(\n    r=8,\n    lora_alpha=16,\n    lora_dropout=0.05,\n    target_modules=[\n        \"q_proj\",\n        \"k_proj\",\n        \"v_proj\",\n        \"o_proj\"\n    ],\n    bias=\"none\",\n    task_type=\"CAUSAL_LM\"\n)\n\nmodel = get_peft_model(model, lora_config)\n\nmodel.print_trainable_parameters()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.127553Z","iopub.status.idle":"2026-09-20T10:20:01.128343Z","shell.execute_reply.started":"2026-09-20T10:20:01.128149Z","shell.execute_reply":"2026-09-20T10:20:01.128174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"./qwen_knee\",\n    num_train_epochs=4,\n    per_device_train_batch_size=1,\n    gradient_accumulation_steps=32,\n    learning_rate=2e-4,\n    logging_steps=1,\n    save_strategy=\"epoch\",\n    fp16=True,\n    report_to=\"none\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.129446Z","iopub.status.idle":"2026-09-20T10:20:01.129672Z","shell.execute_reply.started":"2026-09-20T10:20:01.129564Z","shell.execute_reply":"2026-09-20T10:20:01.129578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_dataset,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.132335Z","iopub.status.idle":"2026-09-20T10:20:01.132736Z","shell.execute_reply.started":"2026-09-20T10:20:01.132475Z","shell.execute_reply":"2026-09-20T10:20:01.132534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Qwenin neticeleri menim scratchden qurdugumdan ferqlidir\n\"\"\"\nSebebler:\n1) dataset ferqi\n2) oyrenilme alqoritmi\n\"\"\"\ntrainer.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.134648Z","iopub.status.idle":"2026-09-20T10:20:01.13524Z","shell.execute_reply.started":"2026-09-20T10:20:01.135078Z","shell.execute_reply":"2026-09-20T10:20:01.135103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"from sklearn.model_selection import train_test_split\n\ntrain_data, val_data = train_test_split(\n    labeled_data,\n    test_size=0.2,\n    random_state=42\n)\n\nprint(len(train_data))\nprint(len(val_data))\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.137Z","iopub.status.idle":"2026-09-20T10:20:01.137371Z","shell.execute_reply.started":"2026-09-20T10:20:01.137182Z","shell.execute_reply":"2026-09-20T10:20:01.137207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_report = qwen_data[0][\"input\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.138934Z","iopub.status.idle":"2026-09-20T10:20:01.139331Z","shell.execute_reply.started":"2026-09-20T10:20:01.139147Z","shell.execute_reply":"2026-09-20T10:20:01.139173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_report = qwen_data[0][\"input\"]\n\nmessages = [\n    {\n        \"role\": \"system\",\n        \"content\": (\n            \"Read the medical report and output only the 12 labels as JSON. \"\n            \"Use only 0 or 1.\"\n        )\n    },\n    {\n        \"role\": \"user\",\n        \"content\": test_report\n    }\n]\n\nprompt = tokenizer.apply_chat_template(\n    messages,\n    tokenize=False,\n    add_generation_prompt=True\n)\n\nprint(\"PROMPT:\")\nprint(prompt)\n\ninputs = tokenizer(\n    prompt,\n    return_tensors=\"pt\"\n).to(model.device)\n\nprint(\"INPUT SHAPE:\", inputs[\"input_ids\"].shape)\n\nmodel.eval()\n\nwith torch.no_grad():\n    output = model.generate(\n        **inputs,\n        max_new_tokens=150,\n        do_sample=False,\n        pad_token_id=tokenizer.eos_token_id\n    )\n\nprint(\"OUTPUT SHAPE:\", output.shape)\n\nnew_tokens = output[0][inputs[\"input_ids\"].shape[1]:]\n\nprint(\"NEW TOKENS:\", new_tokens)\n\nresponse = tokenizer.decode(\n    new_tokens,\n    skip_special_tokens=False\n)\n\nprint(\"RAW RESPONSE:\")\nprint(repr(response))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T10:20:01.140553Z","iopub.status.idle":"2026-09-20T10:20:01.140788Z","shell.execute_reply.started":"2026-09-20T10:20:01.140675Z","shell.execute_reply":"2026-09-20T10:20:01.140689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}