{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install langdetect\n!pip install -qU langchain langchain-huggingface transformers accelerate bitsandbytes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:54:34.485783Z","iopub.execute_input":"2026-08-24T04:54:34.486083Z","iopub.status.idle":"2026-08-24T04:55:07.544418Z","shell.execute_reply.started":"2026-08-24T04:54:34.486042Z","shell.execute_reply":"2026-08-24T04:55:07.543044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport kagglehub\nimport re\nfrom langdetect import detect\nimport unicodedata\nimport torch\nfrom kaggle_secrets import UserSecretsClient\nfrom transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig, pipeline\nfrom langchain_huggingface import HuggingFacePipeline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:07.546466Z","iopub.execute_input":"2026-08-24T04:55:07.546783Z","iopub.status.idle":"2026-08-24T04:55:33.84364Z","shell.execute_reply.started":"2026-08-24T04:55:07.54675Z","shell.execute_reply":"2026-08-24T04:55:33.842947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from langchain_core.prompts import PromptTemplate\nfrom langchain_core.output_parsers import JsonOutputParser\nfrom pydantic import BaseModel, Field","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:33.844677Z","iopub.execute_input":"2026-08-24T04:55:33.845332Z","iopub.status.idle":"2026-08-24T04:55:33.864614Z","shell.execute_reply.started":"2026-08-24T04:55:33.845292Z","shell.execute_reply":"2026-08-24T04:55:33.863546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from langchain_core.prompts import PromptTemplate\nfrom langchain_core.output_parsers import JsonOutputParser\nfrom pydantic import BaseModel, Field\nfrom tqdm import tqdm\nimport pandas as pd\nimport json\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T05:05:49.151918Z","iopub.execute_input":"2026-08-24T05:05:49.152747Z","iopub.status.idle":"2026-08-24T05:05:49.157417Z","shell.execute_reply.started":"2026-08-24T05:05:49.152714Z","shell.execute_reply":"2026-08-24T05:05:49.15636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = kagglehub.competition_download('rsna-knee-abnormality-detection')\n\nprint(\"Path to competition files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:33.873045Z","iopub.execute_input":"2026-08-24T04:55:33.873468Z","iopub.status.idle":"2026-08-24T04:55:34.418161Z","shell.execute_reply.started":"2026-08-24T04:55:33.873441Z","shell.execute_reply":"2026-08-24T04:55:34.417364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nisi_folder = os.listdir(path)\n\nprint(isi_folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:34.419261Z","iopub.execute_input":"2026-08-24T04:55:34.420193Z","iopub.status.idle":"2026-08-24T04:55:34.424899Z","shell.execute_reply.started":"2026-08-24T04:55:34.420157Z","shell.execute_reply":"2026-08-24T04:55:34.424207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\ntrain_df = pd.read_csv(f\"{path}/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:34.426254Z","iopub.execute_input":"2026-08-24T04:55:34.42656Z","iopub.status.idle":"2026-08-24T04:55:34.59224Z","shell.execute_reply.started":"2026-08-24T04:55:34.426534Z","shell.execute_reply":"2026-08-24T04:55:34.591174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Menghitung panjang teks (Karakter dan Kata)\ntrain_df['panjang_karakter'] = train_df['Report'].apply(len)\ntrain_df['jumlah_kata'] = train_df['Report'].apply(lambda x: len(str(x).split()))\n\nprint(\"=== 1. RINGKASAN PANJANG TEKS ===\")\ntrain_df[['panjang_karakter', 'jumlah_kata']].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:34.593559Z","iopub.execute_input":"2026-08-24T04:55:34.594019Z","iopub.status.idle":"2026-08-24T04:55:34.697965Z","shell.execute_reply.started":"2026-08-24T04:55:34.593978Z","shell.execute_reply":"2026-08-24T04:55:34.696908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Mencari Laporan Paling Pendek (Anomali: Mungkin tidak ada informasi)\nprint(\"\\n=== 2. CONTOH 3 LAPORAN PALING PENDEK ===\")\nlaporan_pendek = train_df.sort_values('jumlah_kata').head(3)\nfor idx, teks in enumerate(laporan_pendek['Report']):\n    print(f\"{idx+1}. {teks}\")\n# 3. Mengecek Duplikasi (Template Laporan)\njumlah_duplikat = train_df.duplicated(subset=['Report']).sum()\nprint(f\"\\n=== 3. CEK DUPLIKASI ===\")\nprint(f\"Ada {jumlah_duplikat} laporan yang isinya sama persis (template).\")\n# 4. Mendeteksi Karakter Aneh (Selain huruf, angka, dan tanda baca standar)\n# Regex ini mencari karakter di luar a-z, A-Z, 0-9, dan spasi/tanda baca dasar\ndef cari_karakter_aneh(teks):\n    karakter_aneh = re.findall(r'[^a-zA-Z0-9\\s.,;:()\\-]', str(teks))\n    return len(karakter_aneh) > 0\n\ntrain_df['ada_karakter_aneh'] = train_df['Report'].apply(cari_karakter_aneh)\njumlah_aneh = train_df['ada_karakter_aneh'].sum()\n\nprint(f\"\\n=== 4. CEK KARAKTER ANEH ===\")\nprint(f\"Ada {jumlah_aneh} laporan yang mengandung karakter atau simbol tidak wajar.\")\n# 5. UJI DETEKSI BAHASA (LANGUAGE DETECTION)\ndef deteksi_bahasa(teks):\n    try:\n        # Mengambil 100 karakter pertama saja agar proses deteksi lebih cepat\n        return detect(str(teks)[:100])\n    except:\n        return 'unknown'\n\nprint(\"Sedang mendeteksi bahasa... (Tunggu sebentar)\")\ntrain_df['bahasa'] = train_df['Report'].apply(deteksi_bahasa)\n\nprint(\"\\n=== 5. DISTRIBUSI BAHASA LAPORAN ===\")\ndisplay(train_df['bahasa'].value_counts())\n# 6. UJI STRUKTUR LAPORAN (MENCARI KATA KUNCI KESIMPULAN)\n# Mengecek berapa banyak laporan yang menggunakan format standar \"IMPRESSION\" atau \"FINDINGS\"\njumlah_impression = train_df['Report'].str.contains('IMPRESSION', case=False, na=False).sum()\njumlah_findings = train_df['Report'].str.contains('FINDINGS', case=False, na=False).sum()\njumlah_conclusion = train_df['Report'].str.contains('CONCLUSION', case=False, na=False).sum()\n\nprint(\"\\n=== 6. STRUKTUR LAPORAN MEDIS ===\")\nprint(f\"Mengandung kata 'IMPRESSION' (Kesimpulan): {jumlah_impression} laporan\")\nprint(f\"Mengandung kata 'CONCLUSION' (Kesimpulan): {jumlah_conclusion} laporan\")\nprint(f\"Mengandung kata 'FINDINGS' (Temuan Detail): {jumlah_findings} laporan\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:34.699279Z","iopub.execute_input":"2026-08-24T04:55:34.69978Z","iopub.status.idle":"2026-08-24T04:55:51.689141Z","shell.execute_reply.started":"2026-08-24T04:55:34.699752Z","shell.execute_reply":"2026-08-24T04:55:51.688037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Menampilkan 1 contoh laporan yang memiliki kata IMPRESSION\ncontoh_impression = train_df[train_df['Report'].str.contains('IMPRESSION', case=False, na=False)]['Report'].iloc[0]\nprint(\"\\n=== CONTOH TEKS DENGAN 'IMPRESSION' ===\")\n# Menampilkan 300 karakter terakhir dari laporan tersebut\nprint(\"...\" + contoh_impression[-300:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:51.691941Z","iopub.execute_input":"2026-08-24T04:55:51.692316Z","iopub.status.idle":"2026-08-24T04:55:51.80193Z","shell.execute_reply.started":"2026-08-24T04:55:51.692289Z","shell.execute_reply":"2026-08-24T04:55:51.800829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. UJI NILAI KOSONG TERSELUBUNG\n# Mencari teks yang isinya hanya spasi, strip, titik, atau N/A\nkata_kosong = ['', '-', '.', 'N/A', 'N / A', 'NONE', 'NAN']\nkosong_terselubung = train_df[train_df['Report'].str.strip().str.upper().isin(kata_kosong)]\n\nprint(\"=== 7. UJI KOSONG TERSELUBUNG ===\")\nprint(f\"Ditemukan {len(kosong_terselubung)} laporan yang sebenarnya kosong/tidak valid.\")\nif len(kosong_terselubung) > 0:\n    display(kosong_terselubung.head())\n# 8. UJI HURUF KAPITAL SEMUA (ALL CAPS)\n# Mengecek apakah teks laporan ditulis dengan huruf besar semua\nhuruf_kapital = train_df[train_df['Report'].apply(lambda x: str(x).isupper())]\n\nprint(\"\\n=== 8. UJI HURUF KAPITAL ===\")\nprint(f\"Ditemukan {len(huruf_kapital)} laporan dengan HURUF KAPITAL SEMUA.\")\n\n# 9. UJI FOOTER / BOILERPLATE\n# Mencari jejak tanda tangan elektronik otomatis\njumlah_ttd = train_df['Report'].str.contains('electronically signed|signed by|dictated by', case=False, na=False).sum()\n\nprint(\"\\n=== 9. UJI FOOTER RUMAH SAKIT ===\")\nprint(f\"Ditemukan {jumlah_ttd} laporan yang mengandung teks tanda tangan/footer.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:51.803047Z","iopub.execute_input":"2026-08-24T04:55:51.804222Z","iopub.status.idle":"2026-08-24T04:55:52.132201Z","shell.execute_reply.started":"2026-08-24T04:55:51.804191Z","shell.execute_reply":"2026-08-24T04:55:52.131407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Mencek Distribusi Bahasa Khusus pada 58 Data Berlabel\n\ndef deteksi_bahasa(teks):\n    try:\n        return detect(str(teks)[:100])\n    except:\n        return 'unknown'\n\ntrain_df['bahasa'] = train_df['Report'].apply(deteksi_bahasa)\n\n# Daftar kolom penyakit\nkolom_penyakit = [\n    'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', \n    'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", \n    'Contusion', 'Fracture'\n]\n\n# Filter hanya 58 data yang memiliki label (tidak NaN)\ndata_berlabel = train_df[train_df['ACL'].notnull()].copy()\n\nprint(f\"Total data yang memiliki label: {len(data_berlabel)} baris\\n\")\n\nprint(\"=== DISTRIBUSI BAHASA PADA DATA BERLABEL ===\")\ndisplay(data_berlabel['bahasa'].value_counts().reset_index().rename(columns={'index':'bahasa', 'bahasa':'jumlah_pasien'}))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:55:52.13319Z","iopub.execute_input":"2026-08-24T04:55:52.133523Z","iopub.status.idle":"2026-08-24T04:56:08.187705Z","shell.execute_reply.started":"2026-08-24T04:55:52.133498Z","shell.execute_reply":"2026-08-24T04:56:08.186486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Mencek Bahasa untuk Setiap Kasus Positif (Nilai 1.0)\nprint(\"\\n=== RINCIAN BAHASA UNTUK KASUS POSITIF (1.0) ===\")\nhasil_pos_bahasa = []\n\nfor penyakit in kolom_penyakit:\n    # Ambil baris di mana penyakit tersebut bernilai 1.0\n    pasien_positif = data_berlabel[data_berlabel[penyakit] == 1.0]\n    distribusi_bahasa = pasien_positif['bahasa'].value_counts().to_dict()\n    \n    # Format string output bahasa (misal: \"en: 15, es: 5\")\n    str_bahasa = \", \".join([f\"{k}: {v}\" for k, v in distribusi_bahasa.items()])\n    \n    hasil_pos_bahasa.append({\n        'Nama Penyakit': penyakit,\n        'Total Positif': len(pasien_positif),\n        'Sebaran Bahasa': str_bahasa if str_bahasa else 'Tidak ada kasus positif'\n    })\n\ndf_hasil_bahasa = pd.DataFrame(hasil_pos_bahasa)\ndisplay(df_hasil_bahasa)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:08.188833Z","iopub.execute_input":"2026-08-24T04:56:08.189113Z","iopub.status.idle":"2026-08-24T04:56:08.217229Z","shell.execute_reply.started":"2026-08-24T04:56:08.18904Z","shell.execute_reply":"2026-08-24T04:56:08.216515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Menyiapkan Kamus dan Fungsi Pembersih\n_PRE = str.maketrans({\n    \"ı\": \"i\", \"İ\": \"i\", \"I\": \"i\", \"ß\": \"ss\", \"đ\": \"d\", \"Đ\": \"d\",\n    \"ø\": \"o\", \"Ø\": \"o\", \"æ\": \"ae\", \"Æ\": \"ae\",\n})\n\ndef normalize(text: str) -> str:\n    if not isinstance(text, str): return \"\"\n    text = text.translate(_PRE).lower()\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(ch for ch in text if not unicodedata.combining(ch))\n    text = text.replace(\"­\", \"\")\n    return text\n\ndef unwrap(text: str) -> str:\n    if not isinstance(text, str): return \"\"\n    out = []\n    for line in text.split(\"\\n\"):\n        s = line.strip()\n        if out and out[-1] and not re.search(r\"[.;:!?>*•]$\", out[-1]) \\\n                and len(out[-1].split()) >= 4 and s and not s[:1].isupper():\n            out[-1] = out[-1] + \" \" + s\n        else:\n            out.append(s)\n    return \"\\n\".join(out)\n\ndef pipeline_pembersih_super(teks):\n    teks = normalize(teks)\n    teks = unwrap(teks)\n    teks = re.sub(r'[\\n\\t\\r]+', ' ', teks)\n    teks = re.sub(r'[^\\w\\s.,;:()/\\-]', ' ', teks)\n    teks = re.sub(r'\\s+', ' ', teks).strip()\n    return teks\n\n# 2. Memuat Data Asli\npath_asli = \"/kaggle/input/competitions/rsna-knee-abnormality-detection/train.csv\"\ntrain_df = pd.read_csv(path_asli)\n# 3. Menerapkan Pipeline Pembersih\nprint(\"Memulai proses cleaning teks... (Tunggu sebentar)\")\ntrain_df['Report_Clean'] = train_df['Report'].apply(pipeline_pembersih_super)\n# 4. Menyimpan ke File CSV Baru di Folder Working Kaggle\npath_baru = \"/kaggle/working/train_cleaned.csv\"\ntrain_df.to_csv(path_baru, index=False)\nprint(f\"Selesai! Data berhasil disimpan ke: {path_baru}\\n\")\n# 5. Menampilkan Perbandingan (Sebelum vs Sesudah)\nprint(\"=== PREVIEW PERUBAHAN ===\")\ndisplay(train_df[['StudyInstanceUID', 'Report', 'Report_Clean']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:08.218401Z","iopub.execute_input":"2026-08-24T04:56:08.219161Z","iopub.status.idle":"2026-08-24T04:56:10.191362Z","shell.execute_reply.started":"2026-08-24T04:56:08.219126Z","shell.execute_reply":"2026-08-24T04:56:10.190571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== MEMBUAT DATASET SAMPEL (116 Baris) ===\")\n# 1. Mengambil 58 data yang SUDAH memiliki label (Positif/Punya Kunci Jawaban)\n# Kita cukup mengecek salah satu kolom penyakit, misalnya 'ACL', yang tidak kosong (notnull)\ndata_berlabel = train_df[train_df['ACL'].notnull()].copy()\nprint(f\"Jumlah data berlabel (Ground Truth) : {len(data_berlabel)} baris\")\n\n# 2. Mengambil data yang BELUM memiliki label (kosong / NaN)\ndata_tanpa_label = train_df[train_df['ACL'].isnull()].copy()\n\n# 3. Mengambil tepat 58 baris secara ACAK dari data tanpa label\n# random_state=42 digunakan agar jika kode di-run ulang, hasil acaknya tetap sama\ndata_acak_tanpa_label = data_tanpa_label.sample(n=58, random_state=42)\nprint(f\"Jumlah data tanpa label (Acak)      : {len(data_acak_tanpa_label)} baris\")\n\n# 4. Menggabungkan kedua data tersebut (58 + 58 = 116 baris)\ndf_sampel_116 = pd.concat([data_berlabel, data_acak_tanpa_label], ignore_index=True)\n\n# 5. Mengacak urutannya (opsional, agar yang berlabel dan tidak berlabel tercampur)\ndf_sampel_116 = df_sampel_116.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# 6. Menyimpan ke dalam file CSV baru\npath_sampel = \"/kaggle/working/sampel_116_data.csv\"\ndf_sampel_116.to_csv(path_sampel, index=False)\n\nprint(f\"\\nTotal baris dataset baru            : {len(df_sampel_116)} baris\")\nprint(f\"✅ Selesai! Data sampel berhasil disimpan di: {path_sampel}\")\n\n# Menampilkan 5 baris pertama sebagai preview\ndisplay(df_sampel_116[['StudyInstanceUID', 'Report_Clean', 'ACL']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:10.192628Z","iopub.execute_input":"2026-08-24T04:56:10.193344Z","iopub.status.idle":"2026-08-24T04:56:10.226274Z","shell.execute_reply.started":"2026-08-24T04:56:10.193299Z","shell.execute_reply":"2026-08-24T04:56:10.225526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Buka kembali file yang baru saja disimpan (opsional, jika dataframe df_sampel_116 sudah ada di memori, langkah ini bisa di-skip)\npath_sampel = \"/kaggle/working/sampel_116_data.csv\"\ndf_sampel_116 = pd.read_csv(path_sampel)\n\n# 2. Daftar kolom penyakit sesuai dengan format asli kompetisi\nkolom_penyakit_asli = [\n    'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', \n    'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", \n    'Contusion', 'Fracture'\n]\n\n# 3. Ubah semua nilai NaN / Kosong menjadi 0.0 HANYA pada kolom-kolom penyakit tersebut\ndf_sampel_116[kolom_penyakit_asli] = df_sampel_116[kolom_penyakit_asli].fillna(0.0)\n\n# 4. Simpan dan timpa (overwrite) file CSV-nya agar perubahannya permanen\ndf_sampel_116.to_csv(path_sampel, index=False)\n\nprint(\"✅ Semua nilai kosong (NaN) berhasil diubah menjadi 0.0!\")\nprint(\"\\n=== PREVIEW DATA SETELAH UPDATE ===\")\ndisplay(df_sampel_116[['StudyInstanceUID'] + kolom_penyakit_asli].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:10.22755Z","iopub.execute_input":"2026-08-24T04:56:10.227912Z","iopub.status.idle":"2026-08-24T04:56:10.273211Z","shell.execute_reply.started":"2026-08-24T04:56:10.227884Z","shell.execute_reply":"2026-08-24T04:56:10.272455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# FASE 4 (UPDATE): MULTILINGUAL & REASONING (EXPLAINABLE AI)\n# ==========================================","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:10.2743Z","iopub.execute_input":"2026-08-24T04:56:10.274638Z","iopub.status.idle":"2026-08-24T04:56:10.278817Z","shell.execute_reply.started":"2026-08-24T04:56:10.274612Z","shell.execute_reply":"2026-08-24T04:56:10.278154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Ambil Token Rahasia\ntry:\n    user_secrets = UserSecretsClient()\n    os.environ[\"HF_TOKEN\"] = user_secrets.get_secret(\"HF_TOKEN\")\nexcept Exception as e:\n    print(\"Pastikan secret HF_TOKEN sudah menyala biru!\")\n\n# 2. Setup Kuantisasi 4-bit & Load Model Mistral\nbnb_config = BitsAndBytesConfig(\n    load_in_4bit=True,\n    bnb_4bit_use_double_quant=True,\n    bnb_4bit_quant_type=\"nf4\",\n    bnb_4bit_compute_dtype=torch.bfloat16\n)\n\nmodel_id = \"mistralai/Mistral-7B-Instruct-v0.2\"\ntokenizer = AutoTokenizer.from_pretrained(model_id)\n\nif tokenizer.pad_token_id is None:\n    tokenizer.pad_token_id = tokenizer.eos_token_id\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    model_id,\n    quantization_config=bnb_config,\n    device_map=\"auto\"\n)\n\n# 3. Membangun ulang jembatan 'llm' dengan kapasitas teks lebih besar\ntext_generation_pipeline = pipeline(\n    \"text-generation\",\n    model=model,\n    tokenizer=tokenizer,\n    max_new_tokens=1024, # <--- KAPASITAS DIPERBESAR UNTUK EXPLAINABLE AI\n    temperature=0.1,    \n    do_sample=True,\n    pad_token_id=tokenizer.eos_token_id,\n    return_full_text=False \n)\n\nllm = HuggingFacePipeline(pipeline=text_generation_pipeline)\nprint(\"✅ Variabel 'llm' lokal (Mistral) berhasil dibuat dan siap digunakan!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T04:56:10.280028Z","iopub.execute_input":"2026-08-24T04:56:10.280416Z","iopub.status.idle":"2026-08-24T04:57:34.787189Z","shell.execute_reply.started":"2026-08-24T04:56:10.280374Z","shell.execute_reply":"2026-08-24T04:57:34.786163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. UPDATE SKEMA JSON SUPER LENGKAP\n# Kita pakai underscore di sini agar tidak error saat diproses Pydantic/LangChain\nclass EkstraksiLututLengkap(BaseModel):\n    Terjemahan_Inggris: str = Field(description=\"English translation of the report.\")\n    Terjemahan_Indonesia: str = Field(description=\"Terjemahan bahasa Indonesia dari laporan MRI tersebut.\")\n    Alasan_Prediksi: str = Field(description=\"Brief explanation of why the conditions were marked 1.0 or 0.0.\")\n    ACL: float = Field(description=\"1.0 if torn/injured, 0.0 if normal\")\n    MCL: float = Field(description=\"1.0 if torn/injured, 0.0 if normal\")\n    Medial_Meniscus: float = Field(description=\"1.0 if torn/injured, 0.0 if normal\")\n    Lateral_Meniscus: float = Field(description=\"1.0 if torn/injured, 0.0 if normal\")\n    Medial_OA: float = Field(description=\"1.0 if Medial OA is present, 0.0 if normal\")\n    Lateral_OA: float = Field(description=\"1.0 if Lateral OA is present, 0.0 if normal\")\n    PF_OA: float = Field(description=\"1.0 if PF OA is present, 0.0 if normal\")\n    Effusion: float = Field(description=\"1.0 if effusion is present, 0.0 if normal\")\n    Synovitis: float = Field(description=\"1.0 if Synovitis is present, 0.0 if normal\")\n    Bakers: float = Field(description=\"1.0 if Baker's cyst is present, 0.0 if normal\")\n    Contusion: float = Field(description=\"1.0 if bone contusion is present, 0.0 if normal\")\n    Fracture: float = Field(description=\"1.0 if fracture is present, 0.0 if normal\")\n\nparser = JsonOutputParser(pydantic_object=EkstraksiLututLengkap)\n\n# 2. UPDATE PROMPT (4 TUGAS SEKALIGUS)\ntemplate_instruksi = \"\"\"You are a highly efficient expert radiologist.\nTask 1: Translate the MRI report into a concise English summary.\nTask 2: Translate the MRI report into a concise Indonesian summary.\nTask 3: Extract the 12 conditions (1.0 if present/suspected, 0.0 if normal).\nTask 4: Provide a brief reasoning explaining your diagnosis.\n\nYou MUST output ONLY a valid JSON object. \nNever write introductions. Just output the raw JSON.\n\n{format_instructions}\n\nMRI REPORT:\n{report}\n\nJSON OUTPUT:\n```json\n{{\"\"\" \n\nprompt = PromptTemplate(\n    template=template_instruksi,\n    input_variables=[\"report\"],\n    partial_variables={\"format_instructions\": parser.get_format_instructions()},\n)\n\nchain = prompt | llm | parser\n\n# 3. PERSIAPAN DATAFRAME BARU\n# Membuat dictionary untuk memetakan nama Pydantic (tanpa spasi) ke Nama Kolom Asli (pakai spasi)\npeta_kolom = {\n    'ACL': 'ACL', \n    'MCL': 'MCL', \n    'Medial_Meniscus': 'Medial Meniscus', \n    'Lateral_Meniscus': 'Lateral Meniscus', \n    'Medial_OA': 'Medial OA', \n    'Lateral_OA': 'Lateral OA', \n    'PF_OA': 'PF OA', \n    'Effusion': 'Effusion', \n    'Synovitis': 'Synovitis', \n    'Bakers': \"Baker's\", \n    'Contusion': 'Contusion', \n    'Fracture': 'Fracture'\n}\n\n# Membuat kolom teks jika belum ada\nkolom_teks = ['hasil_llm_Inggris', 'hasil_llm_Indonesia', 'hasil_llm_Alasan']\nfor k in kolom_teks:\n    if k not in train_df.columns:\n        train_df[k] = None\n\n# Membuat kolom penyakit jika belum ada\nfor pydantic_key, nama_asli in peta_kolom.items():\n    if f'hasil_llm_{nama_asli}' not in train_df.columns:\n        train_df[f'hasil_llm_{nama_asli}'] = None\n\n# 4. LOOP EKSEKUSI\nCHECKPOINT_PATH = \"/kaggle/working/train_labels_checkpoint.csv\"\nSAVE_EVERY = 10 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T05:03:27.000022Z","iopub.execute_input":"2026-08-24T05:03:27.000441Z","iopub.status.idle":"2026-08-24T05:03:27.015536Z","shell.execute_reply.started":"2026-08-24T05:03:27.00041Z","shell.execute_reply":"2026-08-24T05:03:27.014585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PEREDAM SUARA (SILENCER) ---\n# Mematikan semua peringatan agar layar bersih dan hanya menampilkan progress bar\nwarnings.filterwarnings(\"ignore\")\ntransformers.logging.set_verbosity_error()\n# --------------------------------\n\nprint(\"🚀 Memulai proses Ekstraksi, Translasi & Reasoning untuk 116 Data Sampel...\")\n\n# 1. Pastikan df_sampel_116 ada di memori\npath_sampel = \"/kaggle/working/sampel_116_data.csv\"\ndf_sampel_116 = pd.read_csv(path_sampel)\n\n# 2. PERSIAPAN KOLOM BARU DI DALAM SAMPEL\npeta_kolom = {\n    'ACL': 'ACL', \n    'MCL': 'MCL', \n    'Medial_Meniscus': 'Medial Meniscus', \n    'Lateral_Meniscus': 'Lateral Meniscus', \n    'Medial_OA': 'Medial OA', \n    'Lateral_OA': 'Lateral OA', \n    'PF_OA': 'PF OA', \n    'Effusion': 'Effusion', \n    'Synovitis': 'Synovitis', \n    'Bakers': \"Baker's\", \n    'Contusion': 'Contusion', \n    'Fracture': 'Fracture'\n}\n\nkolom_teks = ['hasil_llm_Inggris', 'hasil_llm_Indonesia', 'hasil_llm_Alasan']\nfor k in kolom_teks:\n    if k not in df_sampel_116.columns:\n        df_sampel_116[k] = None\n\nfor pydantic_key, nama_asli in peta_kolom.items():\n    if f'hasil_llm_{nama_asli}' not in df_sampel_116.columns:\n        df_sampel_116[f'hasil_llm_{nama_asli}'] = None\n\n# 3. LOOP EKSEKUSI (KHUSUS 116 BARIS)\nCHECKPOINT_PATH_116 = \"/kaggle/working/sampel_116_checkpoint.csv\"\nSAVE_EVERY = 5 \n\nfor index, row in tqdm(df_sampel_116.iterrows(), total=len(df_sampel_116), desc=\"Memproses Sampel\"):\n    # Lewati baris yang sudah berhasil diekstrak\n    if pd.notna(row['hasil_llm_ACL']):\n        continue\n        \n    teks_laporan = row['Report_Clean']\n    if pd.isna(teks_laporan) or str(teks_laporan).strip() == \"\":\n        continue\n\n    try:\n        # Kirim ke AI\n        hasil = chain.invoke({\"report\": teks_laporan})\n        \n        if isinstance(hasil, dict):\n            df_sampel_116.at[index, 'hasil_llm_Inggris'] = hasil.get('Terjemahan_Inggris', '')\n            df_sampel_116.at[index, 'hasil_llm_Indonesia'] = hasil.get('Terjemahan_Indonesia', '')\n            df_sampel_116.at[index, 'hasil_llm_Alasan'] = hasil.get('Alasan_Prediksi', '')\n            \n            for pydantic_key, nama_asli in peta_kolom.items():\n                df_sampel_116.at[index, f'hasil_llm_{nama_asli}'] = hasil.get(pydantic_key, 0.0)\n                \n    except Exception as e:\n        pass \n        \n    # Auto-Save\n    if (index + 1) % SAVE_EVERY == 0:\n        df_sampel_116.to_csv(CHECKPOINT_PATH_116, index=False)\n\n# 4. Simpan final\nFINAL_PATH_116 = \"/kaggle/working/sampel_116_FINAL.csv\"\ndf_sampel_116.to_csv(FINAL_PATH_116, index=False)\nprint(f\"\\n🎉 Selesai! Data 116 sampel tersimpan di {FINAL_PATH_116}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T05:05:34.745541Z","iopub.execute_input":"2026-08-24T05:05:34.746001Z","iopub.status.idle":"2026-08-24T05:05:34.75953Z","shell.execute_reply.started":"2026-08-24T05:05:34.745968Z","shell.execute_reply":"2026-08-24T05:05:34.758101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nimport transformers\nimport logging\nfrom tqdm import tqdm\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T05:07:12.528657Z","iopub.execute_input":"2026-08-24T05:07:12.529912Z","iopub.status.idle":"2026-08-24T05:07:12.535197Z","shell.execute_reply.started":"2026-08-24T05:07:12.529873Z","shell.execute_reply":"2026-08-24T05:07:12.534116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. PEREDAM SUARA TINGKAT DEWA ---\nwarnings.filterwarnings(\"ignore\")\ntransformers.logging.set_verbosity_error()\nlogging.getLogger(\"transformers\").setLevel(logging.ERROR)\n\n# --- 2. CABUT AKAR MASALAH WARNING ---\n# Kita hapus aturan bawaan max_length=20 dari otaknya langsung!\nllm.pipeline.model.generation_config.max_length = None\n# -------------------------------------\n\nprint(\"🚀 Memulai proses untuk 116 Data Sampel (Bukan 4407)...\")\n\n# 1. Load data 116 sampel\npath_sampel = \"/kaggle/working/sampel_116_data.csv\"\ndf_sampel_116 = pd.read_csv(path_sampel)\n\n# 2. Persiapan kolom\npeta_kolom = {\n    'ACL': 'ACL', 'MCL': 'MCL', 'Medial_Meniscus': 'Medial Meniscus', \n    'Lateral_Meniscus': 'Lateral Meniscus', 'Medial_OA': 'Medial OA', \n    'Lateral_OA': 'Lateral OA', 'PF_OA': 'PF OA', 'Effusion': 'Effusion', \n    'Synovitis': 'Synovitis', 'Bakers': \"Baker's\", 'Contusion': 'Contusion', \n    'Fracture': 'Fracture'\n}\n\nkolom_teks = ['hasil_llm_Inggris', 'hasil_llm_Indonesia', 'hasil_llm_Alasan']\nfor k in kolom_teks:\n    if k not in df_sampel_116.columns:\n        df_sampel_116[k] = None\n\nfor pydantic_key, nama_asli in peta_kolom.items():\n    if f'hasil_llm_{nama_asli}' not in df_sampel_116.columns:\n        df_sampel_116[f'hasil_llm_{nama_asli}'] = None\n\n# 3. LOOP EKSEKUSI (KHUSUS 116 BARIS)\nCHECKPOINT_PATH_116 = \"/kaggle/working/sampel_116_checkpoint.csv\"\nSAVE_EVERY = 5 \n\nfor index, row in tqdm(df_sampel_116.iterrows(), total=len(df_sampel_116), desc=\"Memproses 116 Sampel\"):\n    # Lewati baris yang sudah berhasil diekstrak\n    if pd.notna(row['hasil_llm_ACL']):\n        continue\n        \n    teks_laporan = row['Report_Clean']\n    if pd.isna(teks_laporan) or str(teks_laporan).strip() == \"\":\n        continue\n\n    try:\n        # Kirim ke AI\n        hasil = chain.invoke({\"report\": teks_laporan})\n        \n        if isinstance(hasil, dict):\n            df_sampel_116.at[index, 'hasil_llm_Inggris'] = hasil.get('Terjemahan_Inggris', '')\n            df_sampel_116.at[index, 'hasil_llm_Indonesia'] = hasil.get('Terjemahan_Indonesia', '')\n            df_sampel_116.at[index, 'hasil_llm_Alasan'] = hasil.get('Alasan_Prediksi', '')\n            \n            for pydantic_key, nama_asli in peta_kolom.items():\n                df_sampel_116.at[index, f'hasil_llm_{nama_asli}'] = hasil.get(pydantic_key, 0.0)\n                \n    except Exception as e:\n        pass \n        \n    # Auto-Save\n    if (index + 1) % SAVE_EVERY == 0:\n        df_sampel_116.to_csv(CHECKPOINT_PATH_116, index=False)\n\n# 4. Simpan final\nFINAL_PATH_116 = \"/kaggle/working/sampel_116_FINAL.csv\"\ndf_sampel_116.to_csv(FINAL_PATH_116, index=False)\nprint(f\"\\n🎉 Selesai! Data 116 sampel tersimpan di {FINAL_PATH_116}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-24T05:07:26.17904Z","iopub.execute_input":"2026-08-24T05:07:26.179844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}