{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd#导入csv文件的库\nimport numpy as np#进行矩阵运算的库\n#设置随机种子,保证模型可以复现\nimport random\nnp.random.seed(2023)\nrandom.seed(2023)\nimport optuna#自动超参数优化软件框架\nimport warnings#避免一些可以忽略的报错\nwarnings.filterwarnings('ignore')#filterwarnings()方法是用于设置警告过滤器的方法，它可以控制警告信息的输出方式和级别。","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.668129Z","iopub.execute_input":"2023-10-15T01:00:23.668537Z","iopub.status.idle":"2023-10-15T01:00:23.67814Z","shell.execute_reply.started":"2023-10-15T01:00:23.668505Z","shell.execute_reply":"2023-10-15T01:00:23.677049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\nprint(f\"len(train_df):{len(train_df)}\")\nlabels=train_df['label'].unique()\n\ntrain_label=train_df['label'].values\nfor i in range(len(train_label)):\n    for j in range(len(labels)):\n        if train_label[i]==labels[j]:\n            train_label[i]=j \n            break\ntrain_df['label']=train_label\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.679385Z","iopub.execute_input":"2023-10-15T01:00:23.679736Z","iopub.status.idle":"2023-10-15T01:00:23.723582Z","shell.execute_reply.started":"2023-10-15T01:00:23.679708Z","shell.execute_reply":"2023-10-15T01:00:23.721072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\nprint(f\"len(test_df):{len(test_df)}\")\n#Notebook:https://www.kaggle.com/code/isakatsuyoshi/perfect-prediction-of-is-tma\ntest_df['is_tma']=((test_df['image_width'] < 5000) & (test_df['image_height'] < 5000))\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.724267Z","iopub.status.idle":"2023-10-15T01:00:23.724673Z","shell.execute_reply.started":"2023-10-15T01:00:23.724488Z","shell.execute_reply":"2023-10-15T01:00:23.724507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_df=pd.concat((train_df,test_df),axis=0)\ntotal_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.726105Z","iopub.status.idle":"2023-10-15T01:00:23.726621Z","shell.execute_reply.started":"2023-10-15T01:00:23.726362Z","shell.execute_reply":"2023-10-15T01:00:23.726387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_df.drop(['image_id'],axis=1,inplace=True)\ntotal_df['is_tma']=total_df['is_tma'].astype(np.int64)\n\ntotal_df['image_area']=total_df['image_width']*total_df['image_height']\ntotal_df['width_height_radio']=total_df['image_width']/total_df['image_height']\n\ntotal_df['width_height_add']=total_df['image_width']+total_df['image_height']\n\ntotal_df['width_height_reduce']=total_df['image_width']-total_df['image_height']\n\ntotal_df.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-15T01:00:23.728247Z","iopub.status.idle":"2023-10-15T01:00:23.729025Z","shell.execute_reply.started":"2023-10-15T01:00:23.728823Z","shell.execute_reply":"2023-10-15T01:00:23.728846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=total_df[:len(train_df)]\ndef get_low_high_feature(df,target=\"target\",length=None):\n    \"\"\"\n    df:传入的是需要处理的df对象,需要求的是每列的25%(low),75%(high)\n    target是不用处理的列.\n    length:如果传入的total_df,需要求的是train_df的feature,故需要取出数据的前length.\n    \"\"\"\n    keys=df.keys().values#[x0,x1,x2……]\n    for key in keys:\n        if key!=target:\n            value=df[key].values.astype(np.float64)#astype是防止[True,False]\n            if length!=None:\n                value=value[:length]\n            low_value=np.percentile(value,25) \n            high_value=np.percentile(value,75)\n            df['low_'+key]=(df[key]<=low_value)\n            df['high_'+key]=(df[key]>=high_value)\n    return df\ntotal_df=get_low_high_feature(total_df,target=\"label\",length=len(train_df))","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.729997Z","iopub.status.idle":"2023-10-15T01:00:23.730542Z","shell.execute_reply.started":"2023-10-15T01:00:23.730267Z","shell.execute_reply":"2023-10-15T01:00:23.730293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=total_df[:len(train_df)]\nkeys=train_df.keys()\nfor key in keys:\n    values=np.unique(train_df[key].values)#获取每列的value\n    if len(values)<=20 and key!=\"label\":\n        print(f\"key:{key},values:{values}\") \n        key_target=train_df['label'].groupby([train_df[key]]).mean()\n        keys=key_target.keys().values\n        target=key_target.values\n        key_target=pd.DataFrame({key:keys,key+\"_target_mean\":target})\n        total_df=pd.merge(total_df,key_target,on=key,how=\"left\")\ntrain_df=total_df[:len(train_df)]\ntest_df=total_df[len(train_df):]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.731976Z","iopub.status.idle":"2023-10-15T01:00:23.732498Z","shell.execute_reply.started":"2023-10-15T01:00:23.73224Z","shell.execute_reply":"2023-10-15T01:00:23.732265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=train_df['label'].values.astype(np.int64)\nX=train_df.drop(['label'],axis=1).values\n#划分训练集和测试集的函数\ndef train_test_split(dataX,datay,shuffle=True,percentage=0.8):\n    \"\"\"\n    将训练数据X和标签y以numpy.array数组的形式传入\n    划分的比例定为训练集:测试集=8:2\n    \"\"\"\n    if shuffle:\n        random_num=[index for index in range(len(dataX))]\n        np.random.shuffle(random_num)\n        dataX=dataX[random_num]\n        datay=datay[random_num]\n    split_num=int(len(dataX)*percentage)\n    train_X=dataX[:split_num]\n    train_y=datay[:split_num]\n    test_X=dataX[split_num:]\n    test_y=datay[split_num:]\n    return train_X,train_y,test_X,test_y\ntrain_X,train_y,valid_X,valid_y=train_test_split(X,y,percentage=0.8)\nprint(f\"train_X.shape:{train_X.shape},valid_X.shape:{valid_X.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.733667Z","iopub.status.idle":"2023-10-15T01:00:23.734125Z","shell.execute_reply.started":"2023-10-15T01:00:23.733924Z","shell.execute_reply":"2023-10-15T01:00:23.73395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import  LGBMClassifier\ndef accuracy(y_true,y_pred):\n    return np.sum(y_true==y_pred)/len(y_true)\ndef objective(trial):\n    param = {\n        'metric':\"multiclass\",\n        'random_state': trial.suggest_int('random_state',42,2023),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 600),\n        'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 10.0),\n        'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 10.0),#对数分布的建议值\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1),#浮点数\n        'subsample': trial.suggest_float('subsample', 0.5, 1),\n        'learning_rate': trial.suggest_float('learning_rate', 1e-4, 0.1, log=True),\n        'num_leaves' : trial.suggest_int('num_leaves', 8, 64),#整数\n        'min_child_samples': trial.suggest_int('min_child_samples', 1, 100),\n    }\n    model = LGBMClassifier(**param)  \n    \n    model.fit(train_X,train_y,eval_set=[(valid_X,valid_y)],early_stopping_rounds=100,verbose=False)\n    \n    preds = model.predict(valid_X)\n    \n    valid_accuracy = accuracy(valid_y, preds)\n    \n    return valid_accuracy\n#创建的研究命名,找最小值.\nstudy = optuna.create_study(direction='maximize', study_name='Optimize boosting hyperparameters')\n#目标函数,尝试的次数\nstudy.optimize(objective, n_trials=20)\n#输出最佳的参数\nlgbm_params=study.best_trial.params\nprint('lgbm_params=', study.best_trial.params)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:23.885101Z","iopub.execute_input":"2023-10-15T01:00:23.885548Z","iopub.status.idle":"2023-10-15T01:00:34.999777Z","shell.execute_reply.started":"2023-10-15T01:00:23.885515Z","shell.execute_reply":"2023-10-15T01:00:34.998724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LGBMClassifier(**lgbm_params)  \nmodel.fit(X,y)\npred=model.predict(test_df.drop(['label'],axis=1).values)\ntest_pred=[]\nfor i in range(len(pred)):\n    test_pred.append(labels[pred[i]])\ntest_pred[:10]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:35.002056Z","iopub.execute_input":"2023-10-15T01:00:35.00238Z","iopub.status.idle":"2023-10-15T01:00:35.963459Z","shell.execute_reply.started":"2023-10-15T01:00:35.002352Z","shell.execute_reply":"2023-10-15T01:00:35.962455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsubmission['label']=test_pred\nsubmission.to_csv(\"submission.csv\",index=None)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T01:00:35.964401Z","iopub.execute_input":"2023-10-15T01:00:35.964714Z","iopub.status.idle":"2023-10-15T01:00:35.982855Z","shell.execute_reply.started":"2023-10-15T01:00:35.964687Z","shell.execute_reply":"2023-10-15T01:00:35.981037Z"},"trusted":true},"execution_count":null,"outputs":[]}]}