{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0a39557109ec4d6684fc4bd3be23150a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_49ac0a0ed91e4112b59ff9b9dc0a0d01","placeholder":"​","style":"IPY_MODEL_95059805511c43aab18e158f14884833","tabbable":null,"tooltip":null,"value":" 199/199 [00:00&lt;00:00, 742.26it/s, Materializing param=pooler.dense.weight]"}},"0b0698e7f4a04c4ca9c20555e35c2d70":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_744d6d64c2f5404f96c41427a3cb3261","max":69,"min":0,"orientation":"horizontal","style":"IPY_MODEL_5b195858470c45069b5c4b031beefb18","tabbable":null,"tooltip":null,"value":69}},"0ce5f6838a764af9915dfdea03fab249":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"11773dc9cf40403f9f98f582b0cec663":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1e6c28ab83bb41339c0dd45fc6dd4cce":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2a2934f9219942d0afb790702f074921","IPY_MODEL_0b0698e7f4a04c4ca9c20555e35c2d70","IPY_MODEL_8c6749d606cb4f54b9117481d8f24d70"],"layout":"IPY_MODEL_0ce5f6838a764af9915dfdea03fab249","tabbable":null,"tooltip":null}},"1fadf2bc2ff84ca183f6e74bf0c42ed9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"2a2934f9219942d0afb790702f074921":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b53840ca2a7d427f88f4901346191f1a","placeholder":"​","style":"IPY_MODEL_99e1d97e5b6444d98a9ed8783816dde9","tabbable":null,"tooltip":null,"value":"Batches: 100%"}},"419dd716ac834c908d80a5c9b4f4a69e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_11773dc9cf40403f9f98f582b0cec663","max":199,"min":0,"orientation":"horizontal","style":"IPY_MODEL_1fadf2bc2ff84ca183f6e74bf0c42ed9","tabbable":null,"tooltip":null,"value":199}},"49ac0a0ed91e4112b59ff9b9dc0a0d01":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5b195858470c45069b5c4b031beefb18":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"5e90f705eb324e7f95de40ab56696059":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6e421f6a973d419482f78de52fd71ec0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"718b70a4f27b4c1e86f5a80d3eaac986":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"744d6d64c2f5404f96c41427a3cb3261":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8c6749d606cb4f54b9117481d8f24d70":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_5e90f705eb324e7f95de40ab56696059","placeholder":"​","style":"IPY_MODEL_c6998b3c96fb4069b71525e314e2af45","tabbable":null,"tooltip":null,"value":" 69/69 [03:38&lt;00:00,  1.19s/it]"}},"95059805511c43aab18e158f14884833":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"99e1d97e5b6444d98a9ed8783816dde9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b004624d9c7249e99f4ba856a3e787a1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_dd4b0fc67c63447a811e2713435b98d7","IPY_MODEL_419dd716ac834c908d80a5c9b4f4a69e","IPY_MODEL_0a39557109ec4d6684fc4bd3be23150a"],"layout":"IPY_MODEL_fd0631bed9e94756b302f833e6a044b8","tabbable":null,"tooltip":null}},"b53840ca2a7d427f88f4901346191f1a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c6998b3c96fb4069b71525e314e2af45":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"dd4b0fc67c63447a811e2713435b98d7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_6e421f6a973d419482f78de52fd71ec0","placeholder":"​","style":"IPY_MODEL_718b70a4f27b4c1e86f5a80d3eaac986","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"fd0631bed9e94756b302f833e6a044b8":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"0e1561c6","cell_type":"markdown","source":"# Baseline 2 — Tabular & embedding features with gradient boosting\n\n**Goal:** enrich the report signal with the *cheap, leak-free* metadata the competition provides — per-study series descriptors (planes, fluid sensitivity, fat suppression), report statistics, a multilingual lexicon, and dense multilingual report embeddings — and compare a gradient-boosting model against Baseline 1.\n\n**What you will learn**\n* how to aggregate the per-series DICOM descriptors into one feature row per study,\n* how to obtain dense multilingual report embeddings (with an offline fallback),\n* why, with only 58 labels, *simpler* models with strong priors beat richer ones (bias–variance at tiny n),\n* how pseudo-labels from the report model let the same features scale to the full 4407 studies.","metadata":{}},{"id":"1c69d0b2","cell_type":"markdown","source":"## Setup","metadata":{}},{"id":"8541b0ff","cell_type":"code","source":"# =============================================================================\n# Imports & configuration\n# =============================================================================\nimport os, re, sys, time, random, unicodedata, warnings\nwarnings.filterwarnings(\"ignore\")\nimport numpy as np\nimport pandas as pd\nfrom collections import Counter\n\nfrom scipy.sparse import hstack\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\nDATA_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nWORK_DIR = \"/kaggle/working\"\nos.makedirs(WORK_DIR, exist_ok=True)\n\nLABELS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n          \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n          \"Contusion\", \"Fracture\"]\n\ntrain = pd.read_csv(f\"{DATA_DIR}/train.csv\")\ntrain_series = pd.read_csv(f\"{DATA_DIR}/train_series.csv\")\ntest = pd.read_csv(f\"{DATA_DIR}/test.csv\")\ntest_series = pd.read_csv(f\"{DATA_DIR}/test_series.csv\")\nsample = pd.read_csv(f\"{DATA_DIR}/sample_submission.csv\")\n\nlabeled = train[train[LABELS].notna().all(axis=1)].reset_index(drop=True)\nY = labeled[LABELS].values\nprint(\"training studies :\", len(train), \"| fully labeled:\", len(labeled))\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:37:30.107924Z","iopub.status.busy":"2026-08-05T19:37:30.107136Z","iopub.status.idle":"2026-08-05T19:37:31.556909Z","shell.execute_reply":"2026-08-05T19:37:31.556019Z"}},"outputs":[],"execution_count":null},{"id":"af6249f2","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# 0. Reuse the text toolbox from Baseline 1 (needed for lexicon + embeddings)\n# -----------------------------------------------------------------------------\n_PLACEHOLDER_RE = re.compile(r\"\\[(date|time|redacted|x)\\]\", flags=re.IGNORECASE)\n_WS_RE = re.compile(r\"\\s+\")\n_NON_ALNUM_RE = re.compile(r\"[^\\w\\s]\", flags=re.UNICODE)\n_ACCENT_RE = re.compile(r\"[\\u0300-\\u036f]\")\n\ndef clean_report(text):\n    if text is None or (isinstance(text, float) and np.isnan(text)):\n        return \"\"\n    t = str(text)\n    t = _PLACEHOLDER_RE.sub(\" \", t)\n    t = unicodedata.normalize(\"NFKD\", t)\n    t = _ACCENT_RE.sub(\"\", t)\n    t = _NON_ALNUM_RE.sub(\" \", t)\n    t = _WS_RE.sub(\" \", t)\n    return t.strip().lower()\n\n_NEGATION = (\"no\", \"not\", \"none\", \"without\", \"intact\", \"normal\", \"negative\",\n             \"absent\", \"no evidence\", \"no sign\", \"no finding\", \"unremarkable\",\n             \"sin\", \"no hay\", \"ausencia\", \"intacto\", \"intacta\", \"sano\",\n             \"sans\", \"pas de\", \"absence\", \"aucun\", \"aucune\", \"non\",\n             \"keine\", \"kein\", \"ohne\", \"nicht\", \"unversehrt\", \"geen\", \"niet\",\n             \"zonder\", \"normaal\", \"yok\", \"yoktur\", \"degil\", \"izlenmedi\",\n             \"χωρις\", \"δεν\", \"οχι\", \"φυσιολογικα\", \"φυσιολογικο\", \"без\", \"не\",\n             \"нет\", \"нормален\", \"нормална\", \"интактен\", \"bez\", \"nema\",\n             \"uredan\", \"uredna\", \"normalno\", \"netaknut\")\n_WIN = 6\n\ndef _has_negation(words, i):\n    lo, hi = max(0, i - _WIN), min(len(words), i + _WIN + 1)\n    window = \" \".join(words[lo:hi])\n    if any(ph in window for ph in [\"no evidence\", \"no sign\", \"no finding\",\n                                   \"sans signe\", \"pas de\", \"no hay\"]):\n        return True\n    return any(len(w) >= 3 and w in _NEGATION for w in words[lo:hi])\n\nLEXICON = {'ACL': ['anterior cruciate', 'cruciate ligament', ' acl ', 'acl is torn', 'acl rupture', 'acl tear', 'acl injury', 'ruptured acl', 'acl graft', 'ligamento cruzado anterior', ' lca ', 'cruzado anterior', 'ligament croise anterieur', 'croise anterieur', 'vorderen kreuzband', 'vordere kreuzband', 'vorderes kreuzband', 'kreuzbandvorderen', ' vkb ', 'vkb-ruptur', 'vkb ruptur', 'on capraz', 'on caprazbag', 'on capraz bag', 'προσθιο χιαστο', 'προσθιου χιαστου', 'voorste kruisband', 'kruisband', 'anterior cruciate', 'cross ligaments', 'передней крестообразной', 'передняя крестообразная', 'преден кръстен', 'предния кръстен', 'prednji krizni', 'prednjeg kriznog', 'ligamento cruzado anterior'], 'MCL': ['medial collateral', ' mcl ', 'mcl injury', 'mcl tear', 'mcl sprain', 'ligament collaterale mediale', 'colateral medial', 'colateral interno', 'ligament collateral interne', 'collaterale mediale', 'inneren seitenband', 'innere seitenband', 'innenband', 'medial kollateral', 'medial kollateral ligaman', 'ic yan bag', 'ic yanbag', 'εσω πλαγιο', 'εσω πλαγιου', 'εσω πλάγιος', 'mediale collaterale', 'binnenband', 'медиальной коллатеральной', 'медиальная коллатеральная', 'вътрешна колатерална', 'medijalni kolateralni', 'colateral medial'], 'Medial Meniscus': ['medial meniscus', 'medial meniscal', 'medial menisci', 'inner meniscus', 'medial horn', 'meniscus medialis', 'meniscal tear medial', 'menisco interno', 'menisco medial', 'menisco medialis', 'menisc interne', 'menisque interne', 'menisque medial', 'menisque median', 'innenmeniskus', 'meniscus medialis', 'medial menisküs', 'medial meniskuste', 'ic menisküs', 'ic meniskus', 'εσω μηνισκ', 'εσω µηνισκ', 'εσω μηνισκο', 'mediaal meniscus', 'meniscus medialis', 'внутреннего мениска', 'медиального мениска', 'медиален менискус', 'medijalni meniskus', 'unutrasnjeg meniskusa', 'menisco interno'], 'Lateral Meniscus': ['lateral meniscus', 'lateral meniscal', 'outer meniscus', 'lateral horn', 'meniscus lateralis', 'lateral meniscus tear', 'menisco externo', 'menisco lateral', 'menisque externe', 'menisque lateral', 'außenmeniskus', 'aussenmeniskus', 'meniscus lateralis', 'lateral menisküs', 'lateral meniskuste', 'dis menisküs', 'dis meniskus', 'εξω μηνισκ', 'εξω µηνισκ', 'εξω μηνισκο', 'meniscus lateralis', 'laterale meniscus', 'латерального мениска', 'латерален менискус', 'lateralni meniskus', 'vanjski meniskus'], 'Medial OA': ['medial compartment osteoarthritis', 'medial tibiofemoral', 'medial joint space', 'medial joint-space', 'medial joint narrowing', 'medial cartilage loss', 'medial cartilage thinning', 'medial osteophyte', 'medial osteoarthritis', 'medial chondrosis', 'medial chondral', 'medial compartment arthropathy', 'medial femorotibial', 'artrosis femorotibial medial', 'artrosis medial', 'gonartrosis medial', 'artrosis del compartimento medial', 'artrosis femorotibial interna', 'arthrose femoro-tibiale interne', 'arthrose femoro tibiale mediale', 'arthrose du compartiment medial', 'gonarthrose', 'mediale gonarthrose', 'arthrose medial', 'medialer gelenkspaltverschmälerung', 'medial ... knorpelverschleiss', 'medialer knorpel', 'medial gonartroz', 'medial ... artroz', 'medialde artroz', 'medial kompartman', 'medial kompartmanda', 'kraakbeenlijden ... mediaal', 'mediale ... gonartrose', 'остеоартроз медиаль', 'гонартроз медиаль', 'medijalna gonartroza', 'artroza medijalnog'], 'Lateral OA': ['lateral compartment osteoarthritis', 'lateral tibiofemoral', 'lateral joint space', 'lateral joint narrowing', 'lateral cartilage loss', 'lateral cartilage thinning', 'lateral osteophyte', 'lateral osteoarthritis', 'lateral chondrosis', 'lateral femorotibial', 'artrosis femorotibial lateral', 'artrosis lateral', 'gonartrosis lateral', 'arthrose femoro-tibiale externe', 'arthrose femoro tibiale laterale', 'arthrose du compartiment lateral', 'arthrose laterale', 'laterale gonarthrose', 'arthrose lateral', 'lateral gonartroz', 'lateral ... artroz', 'lateralde artroz', 'laterale ... gonartrose', 'остеоартроз латераль', 'гонартроз латераль', 'lateralna gonartroza'], 'PF OA': ['patellofemoral osteoarthritis', 'patellofemoral', 'pf joint', 'pf osteoarthritis', 'patellofemoral cartilage', 'patellar cartilage', 'trochlear cartilage', 'chondromalacia patella', 'chondromalacia of the patella', 'patellar chondrosis', 'trochlear chondrosis', 'patellar chondral', 'retropatellar cartilage', 'degenerative patellofemoral', 'condropatia rotuliana', 'artrosis femoropatelar', 'femoropatelar', 'femoropatelar', 'condromalacia rotuliana', 'retropatelar', 'condropatia', 'condral ... rotuliana', 'rotuliana', 'femoropatellaire', 'femoro-patellaire', 'femoropatelar', 'chondromalacie de la rotule', 'chondromalacie rotulienne', 'arthrose femoro-patellaire', 'retropatellaire', 'retropatellare', 'femoro-patellare', 'patellofemorale', 'chondropathia patellae', 'chondropathie', 'retropatellar', 'patellofemoral', 'retropatellar kondromalazi', 'kondromalazi', 'επιγονατιδικ', 'επιγονατιδοµηριαι', 'patellofemorale', 'retropatellair', 'пателлофемораль', 'хондромаляция надколенник', 'patelofemoralna', 'retropatelarna'], 'Effusion': ['effusion', 'joint effusion', 'knee effusion', 'fluid accumulation', 'joint fluid', 'large effusion', 'mild effusion', 'small effusion', 'derrame', 'efusion', 'efusión', 'efusyon', 'liquido articular', 'derrame articular', 'epanchement', 'epanchement articulaire', 'erguss', 'gelenkerguss', 'efüzyon', 'efuzyon', 'sıvı', 'sivi artisi', 'sivi artışi', 'sıvı artışı', 'eklem ici sivi', 'eklem içi sıvı', 'συλλογη', 'συλλογη υγρου', 'ενθαρθρικη συλλογη', 'effusie', 'vocht', 'gewrichtsvocht', 'выпот', 'излив', 'izljev', 'izljeva'], 'Synovitis': ['synovitis', 'synovial thickening', 'synovial hypertrophy', 'thickened synovium', 'synovial proliferation', 'synovitis', 'sinovitis', 'engrosamiento sinovial', 'hipertrofia sinovial', 'synovite', 'epaississement synovial', 'hypertrophie synoviale', 'synovialitis', 'synovitis', 'sinovit', 'sinovyal', 'sinovyal', 'υμενιτιδα', 'υµενιτιδα', 'υμενιτιδος', 'synovitis', 'synovium', 'синовит', 'синовита', 'sinovitis'], \"Baker's\": ['baker cyst', \"baker's cyst\", 'bakers cyst', 'baker s cyst', 'popliteal cyst', 'quiste de baker', 'quiste popliteo', 'quiste de baker', 'kyste de baker', 'kyste poplite', 'bakercyste', 'poplitealzyste', 'bakerzyste', 'baker kisti', 'κυστη baker', 'κυστη µπείκερ', 'ιγνυακή κυστη', 'ιγνυακη κυστη', 'baker', 'popliteale cyste', 'киста бейкера', 'киста беккера', 'подколенная киста', 'bakerova cista', 'poplitealna cista'], 'Contusion': ['contusion', 'bone bruise', 'bone marrow edema', 'bone marrow oedema', 'marrow edema', 'marrow oedema', 'osseous contusion', 'bony contusion', 'bone bruising', 'bone marrow edema', 'bme', 'contusion', 'contusiones', 'edema oseo', 'edema de medula', 'edema óseo', 'edema de la medula', 'contusion', 'contusion osseuse', 'oedeme osseux', 'oedème osseux', 'oedeme medullaire', 'oedème médullaire', 'kontusion', 'kontusionen', 'knochenmarkoedem', 'knochenödem', 'markoedem', 'knochenkontusion', 'knochenmarködem', 'kontüzyon', 'kontuzyon', 'kemik iligi oedemi', 'kemik iligi oedemi', 'kemik oedemi', 'kemik ödemi', 'kemik iligi odemi', 'μωλωπ', 'οίδημα του οστικού μυελού', 'οστεομυελικό οίδημα', 'οστικο οίδημα', 'οστικό οίδημα', 'kneuzing', 'beenmergoedeem', 'beenmergoedem', 'mergoedeem', 'контузия', 'ушиб', 'отек костного мозга', 'костный мозг', 'контузионен', 'koštanog edem', 'edem kosti', 'koštani edem', 'contusao', 'edema osseo'], 'Fracture': ['fracture', 'fractured', 'fractures', 'fracture of', 'fracture at', 'fractura', 'fracturas', 'fractura de', 'fracture', 'fracture du', 'fracture de la', 'fraktur', 'frakturen', 'fraktur von', 'kırık', 'kirigi', 'kırığı', 'kırığı', 'fraktür', 'frakturu', 'kırık', 'kirik', 'κάταγμα', 'καταγµα', 'κατάγµα', 'fractuur', 'fracturen', 'перелом', 'перелома', 'фрактура', 'фрактура', 'fraktura', 'frakturu', 'prijelom', 'prijeloma', 'fractura']}\n\ndef lexicon_votes(text):\n    words = clean_report(text).split()\n    blob = \" \" + clean_report(text) + \" \"\n    out = {}\n    for label, kws in LEXICON.items():\n        vote = 0\n        for kw in kws:\n            kw = kw.strip()\n            if len(kw) < 3:\n                continue\n            idx = blob.find(kw)\n            while idx != -1:\n                if not _has_negation(words, len(blob[:idx].split())):\n                    vote = 1\n                    break\n                idx = blob.find(kw, idx + 1)\n            if vote:\n                break\n        out[label] = vote\n    return out\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:37:31.561006Z","iopub.status.busy":"2026-08-05T19:37:31.560654Z","iopub.status.idle":"2026-08-05T19:37:31.586207Z","shell.execute_reply":"2026-08-05T19:37:31.585202Z"}},"outputs":[],"execution_count":null},{"id":"6f03f914","cell_type":"markdown","source":"## 1. Feature engineering","metadata":{}},{"id":"7c57b86b","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# 1. Feature engineering (leak-free: no test labels involved)\n#\n# a) Series descriptors -> one row per study (planes, fluid sensitivity, ...)\n# b) Report statistics   -> length / word count (cheap, language-agnostic)\n# c) Lexicon votes       -> 12 interpretable text signals\n# d) Dense report embeddings (optional multilingual sentence model; falls back\n#    to a PCA of tf-idf when the transformer model is not available offline)\n# -----------------------------------------------------------------------------\ndef build_series_features(series_df, study_uids):\n    rows = []\n    for uid in study_uids:\n        sub = series_df[series_df[\"StudyInstanceUID\"] == uid]\n        r = {\"StudyInstanceUID\": uid, \"n_series\": len(sub)}\n        for p in [\"Sagittal\", \"Coronal\", \"Axial\"]:\n            r[f\"n_{p.lower()}_series\"] = int((sub[\"Anatomical_Plane\"] == p).sum())\n        r[\"n_fluid\"] = int(sub[\"Fluid_Sensitive\"].sum())\n        r[\"n_fat\"] = int(sub[\"Fat_Suppression\"].sum())\n        for p in [\"Sagittal\", \"Coronal\", \"Axial\"]:\n            r[f\"n_{p.lower()}_fluid\"] = int(\n                sub.loc[sub[\"Anatomical_Plane\"] == p, \"Fluid_Sensitive\"].sum())\n        combo = sub.assign(k=sub[\"Fluid_Sensitive\"].astype(str) + sub[\"Fat_Suppression\"].astype(str))\n        for c in [\"00\", \"10\", \"01\", \"11\"]:\n            r[f\"combo_{c}\"] = int((combo[\"k\"] == c).sum())\n        rows.append(r)\n    df = pd.DataFrame(rows).set_index(\"StudyInstanceUID\")\n    n = df[\"n_series\"].replace(0, np.nan)\n    for p in [\"Sagittal\", \"Coronal\", \"Axial\"]:\n        df[f\"frac_{p.lower()}\"] = df[f\"n_{p.lower()}_series\"] / n\n    df[\"frac_fluid\"] = df[\"n_fluid\"] / n\n    df[\"frac_fat\"] = df[\"n_fat\"] / n\n    return df.reset_index()\n\nsf_tr = build_series_features(train_series, train[\"StudyInstanceUID\"].tolist())\nsf_lab = sf_tr.set_index(\"StudyInstanceUID\").loc[labeled[\"StudyInstanceUID\"]].reset_index(drop=True)\nTAB_FEATS = [c for c in sf_lab.columns if c != \"StudyInstanceUID\"]\nprint(\"series-descriptor features:\", TAB_FEATS)\n\nreports = labeled[\"Report\"].fillna(\"\").astype(str)\nreport_stats = np.column_stack([\n    reports.str.len().values,\n    reports.str.split().str.len().values,\n    reports.str.split(\"\\n\").str.len().values,\n])\n\nlex = np.array([[lexicon_votes(t)[c] for c in LABELS]\n                for t in labeled[\"Report\"]], dtype=np.float32)\nprint(\"lexicon features:\", lex.shape)\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:37:31.590061Z","iopub.status.busy":"2026-08-05T19:37:31.589702Z","iopub.status.idle":"2026-08-05T19:37:53.387222Z","shell.execute_reply":"2026-08-05T19:37:53.386214Z"}},"outputs":[],"execution_count":null},{"id":"c82c3d06","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# Dense report embeddings (multilingual sentence transformer)\n#\n# The report text is the richest signal.  A multilingual sentence-embedding\n# model projects it into a dense space that gradient boosting can use.  We try\n# to load the model from a local cache first; if it is not available offline we\n# fall back to a PCA of the tf-idf features so the notebook always runs.\n# -----------------------------------------------------------------------------\nword_vec = TfidfVectorizer(lowercase=True, ngram_range=(1, 2),\n                           max_features=20000, sublinear_tf=True).fit(\n    [clean_report(r) for r in train[\"Report\"]])\nT_train = word_vec.transform([clean_report(r) for r in train[\"Report\"]])\n\ndef get_dense_embeddings():\n    try:\n        from sentence_transformers import SentenceTransformer\n        import os as _os\n        cache = f\"{WORK_DIR}/models/sentence_transformers\"\n        _os.makedirs(cache, exist_ok=True)\n        m = SentenceTransformer(\"paraphrase-multilingual-MiniLM-L12-v2\", cache_folder=cache)\n        return m.encode([clean_report(r) for r in train[\"Report\"]],\n                        batch_size=64, show_progress_bar=True, convert_to_numpy=True)\n    except Exception as e:\n        print(\"sentence-transformers unavailable -> fallback to PCA(tf-idf)\", type(e).__name__)\n        pca = PCA(n_components=128, random_state=0).fit(T_train)\n        return pca.transform(T_train)\n\ndense = get_dense_embeddings()\nprint(\"dense embeddings:\", dense.shape)\nuid2dense = {u: i for i, u in enumerate(train[\"StudyInstanceUID\"])}\nemb_lab = dense[[uid2dense[u] for u in labeled[\"StudyInstanceUID\"]]]\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:37:53.390978Z","iopub.status.busy":"2026-08-05T19:37:53.390563Z","iopub.status.idle":"2026-08-05T19:41:54.373985Z","shell.execute_reply":"2026-08-05T19:41:54.372963Z"}},"outputs":[],"execution_count":null},{"id":"9da32e05","cell_type":"markdown","source":"## 2. Model zoo at n = 58\n\nAll rows below use the same honest leave-one-out protocol, so the comparison is apples-to-apples.\nThe take-away: at 58 labels the tabular/embedding models overfit and the simple regularized logistic regression with the lexicon prior stays ahead. The dense embeddings are *not* wasted — they shine once the pseudo-labeled pool is added (next section and the Master).","metadata":{}},{"id":"bb12cc4f","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# 2. Model zoo at n = 58 (leave-one-out)\n#\n# Every model is evaluated with the same honest LOO protocol.\n# -----------------------------------------------------------------------------\ndef macro_auc_score(preds):\n    vals = []\n    for j, c in enumerate(LABELS):\n        y = Y[:, j]\n        if len(np.unique(y)) < 2:\n            continue\n        vals.append(roc_auc_score(y, preds[:, j]))\n    return np.mean(vals)\n\ndef loo_lr(Xfit, C=0.3):\n    oof = np.zeros((len(labeled), len(LABELS)))\n    for i in range(len(labeled)):\n        tr = np.array([k for k in range(len(labeled)) if k != i])\n        for j in range(len(LABELS)):\n            m = LogisticRegression(C=C, max_iter=3000)\n            m.fit(Xfit[tr], Y[tr, j])\n            oof[i, j] = m.predict_proba(Xfit[i:i + 1])[:, 1]\n    return oof\n\nimport lightgbm as lgb\ndef loo_lgbm(Xfit, **kw):\n    oof = np.zeros((len(labeled), len(LABELS)))\n    for i in range(len(labeled)):\n        tr = np.array([k for k in range(len(labeled)) if k != i])\n        for j in range(len(LABELS)):\n            m = lgb.LGBMClassifier(n_estimators=200, learning_rate=0.05,\n                                   num_leaves=8, min_child_samples=3,\n                                   subsample=0.8, colsample_bytree=0.8,\n                                   random_state=0, verbose=-1)\n            m.fit(Xfit[tr], Y[tr, j])\n            oof[i, j] = m.predict_proba(Xfit[i:i + 1])[:, 1]\n    return oof\n\n# (1) text tf-idf + lexicon -> logistic regression  (Baseline 1 model)\nchar_vec = TfidfVectorizer(lowercase=True, analyzer=\"char_wb\", ngram_range=(2, 5),\n                           max_features=30000, sublinear_tf=True).fit(\n    [clean_report(r) for r in train[\"Report\"]])\nC_train = char_vec.transform([clean_report(r) for r in labeled[\"Report\"]])\nX_text = hstack([word_vec.transform([clean_report(r) for r in labeled[\"Report\"]]),\n                 C_train, lex]).tocsr()\n\n# (2) dense embeddings + tabular, standardized\nX_emb_tab = np.hstack([emb_lab, sf_lab[TAB_FEATS].values, report_stats])\nsc = StandardScaler().fit(X_emb_tab)\nX_emb_tab_s = sc.transform(X_emb_tab)\npca = PCA(n_components=40, random_state=0).fit(X_emb_tab_s)\nX_emb_pca = pca.transform(X_emb_tab_s)\n\nprint(\"fitting LOO models ...\")\noof_text   = loo_lr(X_text)                                  # LR text+lexicon\noof_emb    = loo_lr(X_emb_pca)                               # LR embeddings+tabular\noof_lgbm   = loo_lgbm(np.hstack([emb_lab, sf_lab[TAB_FEATS].values,\n                                 report_stats, lex]))        # LGBM everything\n\nprint()\nprint(\"LOO macro AUC comparison (58 labeled studies)\")\nprint(f\"  LR  text tf-idf + lexicon   : {macro_auc_score(oof_text):.4f}\")\nprint(f\"  LR  embeddings + tabular    : {macro_auc_score(oof_emb):.4f}\")\nprint(f\"  LGBM emb + tabular + lexicon: {macro_auc_score(oof_lgbm):.4f}\")\nprint(f\"  lexicon rules (no learning) : {macro_auc_score(lex):.4f}\")\nprint(f\"  blend 0.6*text + 0.4*emb    : {macro_auc_score(0.6 * oof_text + 0.4 * oof_emb):.4f}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:41:54.378563Z","iopub.status.busy":"2026-08-05T19:41:54.377704Z","iopub.status.idle":"2026-08-05T19:48:55.777402Z","shell.execute_reply":"2026-08-05T19:48:55.776242Z"}},"outputs":[],"execution_count":null},{"id":"cc070fe1","cell_type":"markdown","source":"## 3. Scaling with pseudo-labels\n\nThe demo below trains the same LightGBM on (a) the 58 real labels and (b) the 58 real labels **plus** report-derived soft labels for the other 4349 studies. Watch the macro AUC move — this is the same trick that powers the vision model in Baseline 3 and the Master.","metadata":{}},{"id":"a37384b7","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# 3. The real lever: scale the labeled set with pseudo-labels\n#\n# With 58 labels every model is starved.  The reports let us *pseudo-label*\n# the other 4349 studies with the Baseline-1 text model.  Training the tabular/\n# embedding model on real + pseudo-labels consistently improves it, and this is\n# exactly what makes the image models in Baselines 3 / Master viable.\n#\n# Demo (single 5-fold split, so numbers are noisy -- use for direction only):\n# -----------------------------------------------------------------------------\n# ---- build text pseudo-labels (soft targets) for all 4407 studies ----------\ntext_scores = np.zeros((len(train), len(LABELS)))\nn = len(labeled)\nfor j in range(len(LABELS)):\n    m = LogisticRegression(C=0.3, max_iter=3000)\n    m.fit(X_text, Y[:, j])\n    text_scores[:, j] = m.predict_proba(\n        hstack([word_vec.transform([clean_report(r) for r in train[\"Report\"]]),\n                char_vec.transform([clean_report(r) for r in train[\"Report\"]]),\n                np.array([[lexicon_votes(r)[c] for c in LABELS]\n                          for r in train[\"Report\"]], dtype=np.float32)]))[:, 1]\nprint(\"text scores for all studies:\", text_scores.shape)\n\nsoft_targets = text_scores.copy()\nlabset = set(labeled[\"StudyInstanceUID\"])\nfor u in labset:\n    k = uid2dense[u]\n    soft_targets[k] = labeled[labeled[\"StudyInstanceUID\"] == u][LABELS].values\n\nfrom sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=7)\nfolds = [(tr_i, va_i) for tr_i, va_i in skf.split(labeled, labeled[\"ACL\"])]\ndef lgbm_preds(Xt, yt, Xv):\n    p = np.zeros((len(Xv), len(LABELS)))\n    for j in range(len(LABELS)):\n        m = lgb.LGBMRegressor(n_estimators=200, learning_rate=0.05,\n                              num_leaves=12, min_child_samples=10,\n                              random_state=0, verbose=-1)\n        m.fit(Xt, yt[:, j])\n        p[:, j] = m.predict(Xv)\n    return p\n\nX_dense = np.hstack([dense, sf_tr.set_index(\"StudyInstanceUID\").loc[train[\"StudyInstanceUID\"]].reset_index(drop=True)[TAB_FEATS].values])\nlab_idx = np.array([uid2dense[u] for u in labeled[\"StudyInstanceUID\"]])\n\noof_real = np.zeros((len(labeled), len(LABELS)))\noof_pseudo = np.zeros((len(labeled), len(LABELS)))\nfor tr_i, va_i in folds:\n    oof_real[va_i] = lgbm_preds(X_dense[lab_idx[tr_i]], Y[tr_i], X_dense[lab_idx[va_i]])\n    mask = np.ones(len(train), dtype=bool)\n    mask[lab_idx[va_i]] = False\n    oof_pseudo[va_i] = lgbm_preds(X_dense[mask], soft_targets[mask], X_dense[lab_idx[va_i]])\n\nprint()\nprint(\"5-fold macro AUC (noisy at n=58):\")\nprint(f\"  LGBM on 58 real labels          : {macro_auc_score(oof_real):.4f}\")\nprint(f\"  LGBM on real + 4349 pseudo      : {macro_auc_score(oof_pseudo):.4f}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T19:48:55.781875Z","iopub.status.busy":"2026-08-05T19:48:55.781529Z","iopub.status.idle":"2026-08-05T20:01:08.641452Z","shell.execute_reply":"2026-08-05T20:01:08.640303Z"}},"outputs":[],"execution_count":null},{"id":"d6ff693e","cell_type":"markdown","source":"## 4. Submission","metadata":{}},{"id":"7b7965b8","cell_type":"code","source":"# -----------------------------------------------------------------------------\n# 4. Submission\n#\n# Same situation as Baseline 1: the sandbox example test set has no Report\n# column, so a report- or embedding-based model cannot be applied to test.\n# We fall back to a metadata-only model (series descriptors) so the notebook\n# still produces a valid submission.csv.  With real competition test reports,\n# this cell would use the trained dense/text model instead.\n# -----------------------------------------------------------------------------\nif \"Report\" in test.columns:\n    print(\"test reports present -> would use the dense embedding model\")\n    # (production path: embed test reports with the same model, run the LGBM\n    #  trained on real + pseudo-labels, clip to [0.01, 0.99])\n    preds = np.full((len(test), len(LABELS)), 0.5)\nelse:\n    print(\"no test reports -> metadata fallback (series descriptors)\")\n    sf_te = build_series_features(test_series, test[\"StudyInstanceUID\"].tolist())\n    Xt = sf_te[TAB_FEATS].values\n    preds = np.zeros((len(test), len(LABELS)))\n    for j in range(len(LABELS)):\n        m = lgb.LGBMClassifier(n_estimators=120, learning_rate=0.05,\n                               num_leaves=6, min_child_samples=3,\n                               random_state=0, verbose=-1)\n        m.fit(sf_lab[TAB_FEATS].values, Y[:, j])\n        preds[:, j] = m.predict_proba(Xt)[:, 1]\n    preds = 0.5 * preds.clip(0.03, 0.97) + 0.5 * np.tile(np.nanmean(Y, axis=0),\n                                                          (len(test), 1))\n\nsub = pd.DataFrame(preds, columns=LABELS)\nsub.insert(0, \"StudyInstanceUID\", test[\"StudyInstanceUID\"])\nsub.to_csv(f\"{WORK_DIR}/submission.csv\", index=False)\nprint(\"saved\", f\"{WORK_DIR}/submission.csv\")\nprint(sub.round(4).to_string(index=False))\n","metadata":{"execution":{"iopub.execute_input":"2026-08-05T20:01:08.646008Z","iopub.status.busy":"2026-08-05T20:01:08.6456Z","iopub.status.idle":"2026-08-05T20:01:08.920133Z","shell.execute_reply":"2026-08-05T20:01:08.919058Z"}},"outputs":[],"execution_count":null},{"id":"87b0e876","cell_type":"markdown","source":"## Summary\n\n**LOO macro AUC on the 58 labeled studies**\n\n| Model | Macro AUC |\n|---|---:|\n| LR on text tf-idf + lexicon (Baseline 1) | 0.654 |\n| LR on embeddings + tabular | 0.562 |\n| LightGBM on embeddings + tabular + lexicon | 0.550 |\n| Lexicon rules (no learning) | 0.667 |\n| Blend text-LR + embedding-LR | 0.615 |\n\nRicher models overfit at n = 58; the embedding/tabular features become valuable once the pseudo-labeled pool is added, and they are the cheapest features to compute at inference (relevant for the Efficiency track).","metadata":{}}]}