{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.4.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10307443,"sourceType":"datasetVersion","datasetId":6380537}],"dockerImageVersionId":30749,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center>\n<img\nsrc=\"https://cdn.pet-techo.com/attachments/89b29a274c35bd4ddffa35787fadde190e7d8bee/store/fit/750/750/11d8fa6d0148fa7c64b76de67e166b7e1398155c20446fed51c77bcc4705/catch_image.jpg\" width=400 height=200 />\n</center>","metadata":{}},{"cell_type":"markdown","source":"**1.Data　Load**","metadata":{}},{"cell_type":"code","source":"options(warn = -1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T01:02:37.438546Z","iopub.execute_input":"2024-12-25T01:02:37.4405Z","iopub.status.idle":"2024-12-25T01:02:37.566816Z","shell.execute_reply":"2024-12-25T01:02:37.565074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train <- read.csv(\"/kaggle/input/playground-series-s4e12/train.csv\", row.names= 1) #set id as rowname\ntest <- read.csv(\"/kaggle/input/playground-series-s4e12/test.csv\") #set id as rowname\nsubmit = read.csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T01:02:37.569474Z","iopub.execute_input":"2024-12-25T01:02:37.601635Z","iopub.status.idle":"2024-12-25T01:03:06.99122Z","shell.execute_reply":"2024-12-25T01:03:06.989426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"id<-test[,1]　","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:30.957297Z","iopub.execute_input":"2024-12-02T01:06:30.958829Z","iopub.status.idle":"2024-12-02T01:06:30.971923Z","shell.execute_reply":"2024-12-02T01:06:30.970209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T01:03:06.99404Z","iopub.execute_input":"2024-12-25T01:03:06.996251Z","iopub.status.idle":"2024-12-25T01:03:07.02826Z","shell.execute_reply":"2024-12-25T01:03:07.025352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(RColorBrewer) #custom colors\nlibrary(corrplot)\nlibrary(xgboost) # for xgboost\nlibrary(GGally)\nlibrary(summarytools)\nlibrary(inspectdf)\nlibrary(dplyr)\nlibrary(modeltime)\nlibrary(timetk)\nlibrary(forecast)\nlibrary(ggplot2)\nlibrary(tidymodels)\nlibrary(tidyverse)\nlibrary(rsample)\nlibrary(withr)\nlibrary(skimr)\nlibrary(summarytools)\nlibrary(bonsai)\nlibrary(lubridate)\nlibrary(urca)\nlibrary(tseries)\nlibrary(bonsai)\nlibrary(lubridate)\nlibrary(catboost)\nlibrary(kableExtra)\ninstall.packages(\"stacks\")\nlibrary(stacks)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T03:10:10.756918Z","iopub.execute_input":"2024-12-12T03:10:10.758819Z","iopub.status.idle":"2024-12-12T03:10:21.410846Z","shell.execute_reply":"2024-12-12T03:10:21.408343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**2let's take a look at the distribution of each class.**.","metadata":{}},{"cell_type":"code","source":"str(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.590814Z","iopub.execute_input":"2024-12-02T01:06:41.593572Z","iopub.status.idle":"2024-12-02T01:06:41.633079Z","shell.execute_reply":"2024-12-02T01:06:41.629936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.638265Z","iopub.execute_input":"2024-12-02T01:06:41.640128Z","iopub.status.idle":"2024-12-02T01:06:41.680781Z","shell.execute_reply":"2024-12-02T01:06:41.678049Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**3.Let's take a look at the distribution of categorical data.**","metadata":{}},{"cell_type":"code","source":"a <- ggplot(train, aes(x = Gender, fill = Gender)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Gender\", x = \"Gender\", y = \"Count\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.685813Z","iopub.execute_input":"2024-12-02T01:06:41.687703Z","iopub.status.idle":"2024-12-02T01:06:41.731277Z","shell.execute_reply":"2024-12-02T01:06:41.728523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"b <- ggplot(train, aes(x = Marital.Status, fill =  Marital.Status)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Marital.Status\", x = \"Marital.Status\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.736442Z","iopub.execute_input":"2024-12-02T01:06:41.738236Z","iopub.status.idle":"2024-12-02T01:06:41.760772Z","shell.execute_reply":"2024-12-02T01:06:41.758286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"c <- ggplot(train, aes(x = Education.Level, fill =  Education.Level)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Education.Level\", x = \"Education.Level\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.765966Z","iopub.execute_input":"2024-12-02T01:06:41.767862Z","iopub.status.idle":"2024-12-02T01:06:41.790546Z","shell.execute_reply":"2024-12-02T01:06:41.788214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"d <- ggplot(train, aes(x = Occupation, fill =  Occupation)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Occupation\", x = \"Occupation\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.795451Z","iopub.execute_input":"2024-12-02T01:06:41.797268Z","iopub.status.idle":"2024-12-02T01:06:41.820122Z","shell.execute_reply":"2024-12-02T01:06:41.817553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"e <- ggplot(train, aes(x = Location, fill =  Location)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Location\", x = \"Location\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.825134Z","iopub.execute_input":"2024-12-02T01:06:41.826992Z","iopub.status.idle":"2024-12-02T01:06:41.850651Z","shell.execute_reply":"2024-12-02T01:06:41.848167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f <- ggplot(train, aes(x = Policy.Type, fill =  Policy.Type)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Policy.Type\", x = \"Policy.Type\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.855774Z","iopub.execute_input":"2024-12-02T01:06:41.857525Z","iopub.status.idle":"2024-12-02T01:06:41.880277Z","shell.execute_reply":"2024-12-02T01:06:41.877851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#g <- ggplot(train, aes(x = Policy.Start.Date, fill = Policy.Start.Date)) +\n#  geom_bar(position = \"dodge\") +\n#  labs(title = \"Count of Policy.Start.Date\", x = \"Policy.Start.Date\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.885307Z","iopub.execute_input":"2024-12-02T01:06:41.887115Z","iopub.status.idle":"2024-12-02T01:06:41.903804Z","shell.execute_reply":"2024-12-02T01:06:41.901423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"h <- ggplot(train, aes(x = Customer.Feedback, fill =  Customer.Feedback)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Customer.Feedback\", x = \"Customer.Feedback\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.908636Z","iopub.execute_input":"2024-12-02T01:06:41.910328Z","iopub.status.idle":"2024-12-02T01:06:41.93522Z","shell.execute_reply":"2024-12-02T01:06:41.932794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"i <- ggplot(train, aes(x = Smoking.Status, fill =  Smoking.Status)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Smoking.Status\", x = \"Smoking.Status\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.940541Z","iopub.execute_input":"2024-12-02T01:06:41.942356Z","iopub.status.idle":"2024-12-02T01:06:41.967841Z","shell.execute_reply":"2024-12-02T01:06:41.965142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"j <- ggplot(train, aes(x = Exercise.Frequency, fill =  Exercise.Frequency)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Exercise.Frequency\", x = \"Exercise.Frequency\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:41.972941Z","iopub.execute_input":"2024-12-02T01:06:41.974971Z","iopub.status.idle":"2024-12-02T01:06:41.9981Z","shell.execute_reply":"2024-12-02T01:06:41.995722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"k <- ggplot(train, aes(x = Property.Type, fill =  Property.Type)) +\n  geom_bar(position = \"dodge\") +\n  labs(title = \"Count of Property.Type\", x = \"Property.Type\", y = \"Count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.003053Z","iopub.execute_input":"2024-12-02T01:06:42.004829Z","iopub.status.idle":"2024-12-02T01:06:42.028313Z","shell.execute_reply":"2024-12-02T01:06:42.025967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"aa<-ggplot(train, aes(x = Gender, y = Premium.Amount, fill = Gender)) +\n  geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Gender\", \n       x = \"Gender\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.033149Z","iopub.execute_input":"2024-12-02T01:06:42.035092Z","iopub.status.idle":"2024-12-02T01:06:42.065478Z","shell.execute_reply":"2024-12-02T01:06:42.06288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bb <- ggplot(train, aes(x = Marital.Status,  y = Premium.Amount,fill =  Marital.Status)) +\n  geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Marital.Status\", \n       x = \"Marital.Status\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.071252Z","iopub.execute_input":"2024-12-02T01:06:42.073152Z","iopub.status.idle":"2024-12-02T01:06:42.103291Z","shell.execute_reply":"2024-12-02T01:06:42.100721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cc <- ggplot(train, aes(x = Education.Level, y = Premium.Amount,fill =  Education.Level)) +\n   geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Education.Level\", \n       x = \"Education.Level\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.107462Z","iopub.execute_input":"2024-12-02T01:06:42.109283Z","iopub.status.idle":"2024-12-02T01:06:42.136471Z","shell.execute_reply":"2024-12-02T01:06:42.133902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dd <- ggplot(train, aes(x = Occupation,y = Premium.Amount, fill =  Occupation)) +\n    geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Occupation\", \n       x = \"Occupation\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.13991Z","iopub.execute_input":"2024-12-02T01:06:42.141566Z","iopub.status.idle":"2024-12-02T01:06:42.202833Z","shell.execute_reply":"2024-12-02T01:06:42.200314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ee <- ggplot(train, aes(x = Location,y = Premium.Amount, fill =  Location)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Location\", \n       x = \"Location\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.206395Z","iopub.execute_input":"2024-12-02T01:06:42.208092Z","iopub.status.idle":"2024-12-02T01:06:42.234658Z","shell.execute_reply":"2024-12-02T01:06:42.232201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ff <- ggplot(train, aes(x = Policy.Type,y = Premium.Amount, fill =  Policy.Type)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Policy.Type\", \n       x = \"Policy.Type\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.238839Z","iopub.execute_input":"2024-12-02T01:06:42.240818Z","iopub.status.idle":"2024-12-02T01:06:42.267017Z","shell.execute_reply":"2024-12-02T01:06:42.264193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#gg <- ggplot(train, aes(x = Policy.Start.Date,y = Premium.Amount, fill = Policy.Start.Date)) +\n#      geom_boxplot() +\n#  labs(title = \"Boxplot of Premium Amount by Policy.Start.Date\", \n#       x = \"Policy.Start.Date\", \n#       y = \"Premium Amount\") +\n#  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.270855Z","iopub.execute_input":"2024-12-02T01:06:42.272785Z","iopub.status.idle":"2024-12-02T01:06:42.290609Z","shell.execute_reply":"2024-12-02T01:06:42.288068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hh <- ggplot(train, aes(x = Customer.Feedback,y = Premium.Amount, fill =  Customer.Feedback)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Customer.Feedback\", \n       x = \"Customer.Feedback\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.294174Z","iopub.execute_input":"2024-12-02T01:06:42.295826Z","iopub.status.idle":"2024-12-02T01:06:42.31981Z","shell.execute_reply":"2024-12-02T01:06:42.317512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ii <- ggplot(train, aes(x = Smoking.Status,y = Premium.Amount,  fill =  Smoking.Status)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Smoking.Status\", \n       x = \"Smoking.Status\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.323319Z","iopub.execute_input":"2024-12-02T01:06:42.324912Z","iopub.status.idle":"2024-12-02T01:06:42.352226Z","shell.execute_reply":"2024-12-02T01:06:42.349221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"jj <- ggplot(train, aes(x = Exercise.Frequency,y = Premium.Amount, fill =  Exercise.Frequency)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Exercise.Frequency\", \n       x = \"Exercise.Frequency\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.356399Z","iopub.execute_input":"2024-12-02T01:06:42.358126Z","iopub.status.idle":"2024-12-02T01:06:42.383794Z","shell.execute_reply":"2024-12-02T01:06:42.381382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kk <- ggplot(train, aes(x = Property.Type, y = Premium.Amount,fill =  Property.Type)) +\n      geom_boxplot() +\n  labs(title = \"Boxplot of Premium Amount by Property.Type\", \n       x = \"Property.Type\", \n       y = \"Premium Amount\") +\n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.387423Z","iopub.execute_input":"2024-12-02T01:06:42.389098Z","iopub.status.idle":"2024-12-02T01:06:42.415638Z","shell.execute_reply":"2024-12-02T01:06:42.412843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"options(\n  repr.plot.width = 16,   \n  repr.plot.height = 24   \n)\ngridExtra::grid.arrange(a,b,c,d,e,f,h,i,j,k,ncol = 2,widths = c(15, 15))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:42.419482Z","iopub.execute_input":"2024-12-02T01:06:42.421164Z","iopub.status.idle":"2024-12-02T01:06:56.7159Z","shell.execute_reply":"2024-12-02T01:06:56.712915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"options(\n  repr.plot.width = 16,   \n  repr.plot.height = 24   \n)\ngridExtra::grid.arrange(aa,bb,cc,dd,ee,ff,hh,ii,jj,kk,ncol = 2,widths = c(15, 15))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:06:56.7196Z","iopub.execute_input":"2024-12-02T01:06:56.72161Z","iopub.status.idle":"2024-12-02T01:07:43.09779Z","shell.execute_reply":"2024-12-02T01:07:43.094616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**4.Let's take a look at the distribution of numerical data**","metadata":{}},{"cell_type":"code","source":"l <- ggplot(train, aes(x = Age , fill =  Premium.Amount)) +\n   geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Age with Premium Amount\",\n    x = \"Age\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.101361Z","iopub.execute_input":"2024-12-02T01:07:43.103189Z","iopub.status.idle":"2024-12-02T01:07:43.136136Z","shell.execute_reply":"2024-12-02T01:07:43.133647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m <- ggplot(train, aes(x = Annual.Income , fill =  Premium.Amount)) +\n   geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Annual.Income with Premium Amount\",\n    x = \"Annual.Income\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.13961Z","iopub.execute_input":"2024-12-02T01:07:43.141891Z","iopub.status.idle":"2024-12-02T01:07:43.171675Z","shell.execute_reply":"2024-12-02T01:07:43.169234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n <- ggplot(train, aes(x = Number.of.Dependents , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Number.of.Dependents with Premium Amount\",\n    x = \"Number.of.Dependents\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.176793Z","iopub.execute_input":"2024-12-02T01:07:43.178517Z","iopub.status.idle":"2024-12-02T01:07:43.208676Z","shell.execute_reply":"2024-12-02T01:07:43.206117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"o <- ggplot(train, aes(x = Health.Score  , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Health.Score with Premium Amount\",\n    x = \"Health.Score\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.213729Z","iopub.execute_input":"2024-12-02T01:07:43.2155Z","iopub.status.idle":"2024-12-02T01:07:43.245909Z","shell.execute_reply":"2024-12-02T01:07:43.243297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"p <- ggplot(train, aes(x = Previous.Claims  , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Previous.Claims with Premium Amount\",\n    x = \"Previous.Claims\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.251152Z","iopub.execute_input":"2024-12-02T01:07:43.252904Z","iopub.status.idle":"2024-12-02T01:07:43.285279Z","shell.execute_reply":"2024-12-02T01:07:43.282085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q <- ggplot(train, aes(x = Vehicle.Age  , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Vehicle.Age with Premium Amount\",\n    x = \"Vehicle.Age\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.291171Z","iopub.execute_input":"2024-12-02T01:07:43.293005Z","iopub.status.idle":"2024-12-02T01:07:43.324148Z","shell.execute_reply":"2024-12-02T01:07:43.321597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r <- ggplot(train, aes(x = Credit.Score  , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Credit.Score with Premium Amount\",\n    x = \"Credit.Score\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.329819Z","iopub.execute_input":"2024-12-02T01:07:43.33179Z","iopub.status.idle":"2024-12-02T01:07:43.362103Z","shell.execute_reply":"2024-12-02T01:07:43.359467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s <- ggplot(train, aes(x = Insurance.Duration  , fill =  Premium.Amount)) +\n     geom_histogram(binwidth = 5, color = \"white\") +  # 例として幅を5に設定\n  labs(\n    title = \"Histogram of Insurance.Duration with Premium Amount\",\n    x = \"Insurance.Duration\",\n    y = \"Count\"\n  ) +\n  scale_fill_gradient(low = \"blue\", high = \"red\") + \n  theme_minimal()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.367258Z","iopub.execute_input":"2024-12-02T01:07:43.36904Z","iopub.status.idle":"2024-12-02T01:07:43.39976Z","shell.execute_reply":"2024-12-02T01:07:43.397261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"options(\n  repr.plot.width = 16,   \n  repr.plot.height = 24   \n)\ngridExtra::grid.arrange(l,m,n,o,p,q,r,s,ncol = 2,widths = c(15, 15))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:07:43.404777Z","iopub.execute_input":"2024-12-02T01:07:43.406569Z","iopub.status.idle":"2024-12-02T01:08:01.260708Z","shell.execute_reply":"2024-12-02T01:08:01.257839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**5.Create a heatmap**","metadata":{}},{"cell_type":"code","source":"train2<-data.frame(lapply(train, function(x) {\n  if (all(is.na(as.numeric(as.character(x))))) {\n    return(x)\n  } else {\n    return(as.numeric(as.character(x)))\n  }\n}))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:08:01.264603Z","iopub.execute_input":"2024-12-02T01:08:01.266468Z","iopub.status.idle":"2024-12-02T01:08:28.127945Z","shell.execute_reply":"2024-12-02T01:08:28.125226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"options(\n  repr.plot.width = 10,   \n  repr.plot.height = 10   \n)\nlibrary(reshape2)\n\nnumeric_data <- train2[, sapply(train2, is.numeric)]\n\nmelted_data <- melt(cor(numeric_data))\n\nggplot(data = melted_data, aes(x = Var1, y = Var2, fill = value)) +\n  geom_tile() +\n  scale_fill_gradient2(low = \"blue\", high = \"red\", mid = \"white\", \n                       midpoint = 0, limit = c(-1, 1), space = \"Lab\", \n                       name=\"Correlation\") +\n  theme_minimal() + \n  labs(title = \"Heatmap of Numeric Variables\", x = \"Variables\", y = \"Variables\") +\n  theme(axis.text.x = element_text(angle = 45, vjust = 1, \n                                  size = 12, hjust = 1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:08:28.133176Z","iopub.execute_input":"2024-12-02T01:08:28.134915Z","iopub.status.idle":"2024-12-02T01:08:28.714052Z","shell.execute_reply":"2024-12-02T01:08:28.711585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(lubridate)\n\n# 'Policy Start Date'\ntrain <- train %>%\n  mutate(`Policy.Start.Date` = as.Date(`Policy.Start.Date`, format = \"%Y-%m-%d\")) %>%\n  \n  # 年、月、日、曜日などを抽出\n  mutate(\n    Year = year(`Policy.Start.Date`),\n    Day = day(`Policy.Start.Date`),\n    Month = month(`Policy.Start.Date`),\n    Month_name = month(`Policy.Start.Date`, label = TRUE, abbr = FALSE),\n    Day_of_week = wday(`Policy.Start.Date`, label = TRUE, abbr = FALSE, week_start = 1),\n    Week = isoweek(`Policy.Start.Date`)\n  ) %>%\n  \n  # サイン・コサイン変換 (周期的な変数のエンコーディング)\n  mutate(\n    Year_sin = sin(2 * pi * Year),\n    Year_cos = cos(2 * pi * Year),\n    Month_sin = sin(2 * pi * Month / 12),\n    Month_cos = cos(2 * pi * Month / 12),\n    Day_sin = sin(2 * pi * Day / 31),\n    Day_cos = cos(2 * pi * Day / 31),\n    \n    # Group変数を生成 (Pythonの//は整数除算)\n    Group = (Year - 2020) * 48 + Month * 4 + floor(Day / 7)\n  ) %>%\n  \n  # 'Policy Start Date'列を削除\n  select(-`Policy.Start.Date`)\n\n\ntest <- test %>%\n  mutate(`Policy.Start.Date` = as.Date(`Policy.Start.Date`, format = \"%Y-%m-%d\")) %>%\n  \n  # 年、月、日、曜日などを抽出\n  mutate(\n    Year = year(`Policy.Start.Date`),\n    Day = day(`Policy.Start.Date`),\n    Month = month(`Policy.Start.Date`),\n    Month_name = month(`Policy.Start.Date`, label = TRUE, abbr = FALSE),\n    Day_of_week = wday(`Policy.Start.Date`, label = TRUE, abbr = FALSE, week_start = 1),\n    Week = isoweek(`Policy.Start.Date`)\n  ) %>%\n  \n  # サイン・コサイン変換 (周期的な変数のエンコーディング)\n  mutate(\n    Year_sin = sin(2 * pi * Year),\n    Year_cos = cos(2 * pi * Year),\n    Month_sin = sin(2 * pi * Month / 12),\n    Month_cos = cos(2 * pi * Month / 12),\n    Day_sin = sin(2 * pi * Day / 31),\n    Day_cos = cos(2 * pi * Day / 31),\n    \n    # Group変数を生成 (Pythonの//は整数除算)\n    Group = (Year - 2020) * 48 + Month * 4 + floor(Day / 7)\n  ) %>%\n  \n  # 'Policy Start Date'列を削除\n  select(-`Policy.Start.Date`)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:08:28.71844Z","iopub.execute_input":"2024-12-02T01:08:28.720132Z","iopub.status.idle":"2024-12-02T01:08:48.113117Z","shell.execute_reply":"2024-12-02T01:08:48.110501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train$annual_income_health_score_ratio<-train$Health.Score / train$Annual.Income\n#train$annual_income_age_ratio<-train$Annual.Income/train$Age\n#train$credit_age<-train$Credit.Score/train$Age\n#train$vehicle_age_insurance_duration<-train$Vehicle.Age/train$Insurance.Duration\n\n#test$annual_income_health_score_ratio<-test$Health.Score / test$Annual.Income\n#test$annual_income_age_ratio<-test$Annual.Income/test$Age\n#test$credit_age<-test$Credit.Score/test$Age\n#test$vehicle_age_insurance_duration<-test$Vehicle.Age/test$Insurance.Duration","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**6.Creat QQ PLOT**","metadata":{}},{"cell_type":"code","source":"numeric_vars <- names(train)[sapply(train, is.numeric)]\n\noptions(\n  repr.plot.width = 4,   \n  repr.plot.height = 4   \n)\n\nfor (var in numeric_vars) {\n  p <- ggplot(train, aes(sample = .data[[var]])) +\n    stat_qq() +\n    stat_qq_line() +\n    ggtitle(paste(\"QQ Plot of\", var)) +\n    theme_minimal()\n  \n  print(p)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:08:48.118237Z","iopub.execute_input":"2024-12-02T01:08:48.120258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T03:10:58.111953Z","iopub.execute_input":"2024-12-12T03:10:58.114208Z","iopub.status.idle":"2024-12-12T03:10:58.161184Z","shell.execute_reply":"2024-12-12T03:10:58.157635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train <- train %>%\n  filter(\n    Premium.Amount > quantile(Premium.Amount, 0.25),\n    Premium.Amount < quantile(Premium.Amount, 0.75)\n  )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train <- train %>%\n  filter(\n    Annual.Income > quantile(Annual.Income, 0.25, na.rm = TRUE),\n    Annual.Income < quantile(Annual.Income, 0.75, na.rm = TRUE)\n  )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train <- train %>%\n  filter(\n    Previous.Claims < quantile(Previous.Claims, 0.75, na.rm = TRUE)\n  )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**7.Create a model &　Ensemble**","metadata":{}},{"cell_type":"code","source":"set.seed(123)\n\ndata_split <-\n  initial_split(train, strata = Premium.Amount)\n\ntrain_data <- training(data_split)\ntest_data  <- testing(data_split)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"set.seed(123)\n\nrec <- recipe(Premium.Amount　~ ., data = train_data) %>%\n　step_impute_median(all_numeric_predictors())%>%\n  step_corr(all_numeric(), threshold = .98) %>%\n　step_YeoJohnson(all_numeric_predictors()) %>% \n  step_normalize(all_numeric_predictors()) %>%\n　step_novel(all_nominal_predictors())%>%\n  step_zv()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rec2 <- recipe(Premium.Amount　~ ., data = train_data) %>%\n　step_impute_median(all_numeric_predictors())%>%\n  step_corr(all_numeric(), threshold = .98) %>%\n　step_YeoJohnson(all_numeric_predictors()) %>% \n  step_normalize(all_numeric_predictors()) %>%\n  step_dummy(all_nominal_predictors())%>%\n　step_novel(all_nominal_predictors())%>%\n  step_zv()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_spec <-\n  boost_tree(\n    trees = 727,\n    tree_depth = 31,\n    learn_rate = 0.02007938,\n    mtry = 11,\n    min_n = 11,\n    loss_reduction = 360571.3\n  ) %>%\n  set_engine(engine = \"lightgbm\",\n             lambda_l2 = 0.03977667923976626,\n             lambda_l1 = 0.08851731976964569,\n             is_unbalance = TRUE,\n             num_leaves = 94,\n             nthread  = future::availableCores()\n) %>%\n  set_mode(mode = \"regression\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T03:15:54.20071Z","iopub.execute_input":"2024-12-12T03:15:54.202891Z","iopub.status.idle":"2024-12-12T03:15:54.237416Z","shell.execute_reply":"2024-12-12T03:15:54.234687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_spec2 <-\n  boost_tree(\n    trees = 1796,\n    tree_depth = 75,\n    learn_rate = 0.02670946,\n    mtry = 19,\n    min_n = 60,\n    loss_reduction = 14.10027\n  ) %>%\n  set_engine(engine = \"lightgbm\",\n             is_unbalance = TRUE,\n             num_leaves = 10,\n             nthread  = future::availableCores()\n) %>%\n  set_mode(mode = \"regression\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xg_spec <- \n  boost_tree(\n    trees = 3000,\n    tree_depth = 9,\n    learn_rate = 0.01,\n    mtry = 12,\n    min_n = 13,\n    loss_reduction = 0.353\n  ) %>%\n  set_engine(\"xgboost\",\n             subsample = 0.895,\n             reg_alpha = 0.353,\n             reg_lambda = 0.956,\n             max_bin = 8000,\n             tree_method = \"hist\",\n             eval_metric = \"rmse\") %>%\n  set_mode(\"regression\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nn_spec <-\n  mlp(hidden_units = 21,\n      penalty = 3.798473, \n      epochs = 60) %>%\n  set_engine('nnet') %>%\n  set_mode('regression')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wf <- workflow() %>% \n  add_model(lgbm_spec) %>% \n  add_recipe(rec)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wf1 <- workflow() %>% \n  add_model(lgbm_spec2) %>% \n  add_recipe(rec)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wf2 <- workflow() %>% \n  add_model(xg_spec) %>% \n  add_recipe(rec2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wf3 <- workflow() %>% \n  add_model(nn_spec) %>% \n  add_recipe(rec2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit<- wf %>%\n  fit(data = train_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit1<- wf1 %>%\n  fit(data = train_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit2<- wf2 %>%\n  fit(data = train_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fit3<- wf3%>%\n#  fit(data = train_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions <- predict(fit, new_data = test_data, type = \"numeric\") %>%\n  bind_cols(test_data) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions1 <- predict(fit1, new_data = test_data, type = \"numeric\") %>%\n  bind_cols(test_data) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions2 <- predict(fit2, new_data = test_data, type = \"numeric\") %>%\n  bind_cols(test_data) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#predictions3 <- predict(fit3, new_data = test_data, type = \"numeric\") %>%\n#  bind_cols(test_data) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result <- rmse(predictions, truth = Premium.Amount, .pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1 <- rmse(predictions1, truth = Premium.Amount, .pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result2 <- rmse(predictions2, truth = Premium.Amount, .pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#result3 <- rmse(predictions3, truth = Premium.Amount, .pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#result3","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_fit<- wf %>%\n  fit(data = train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_fit1<- wf1 %>%\n  fit(data = train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_fit2<- wf2 %>%\n  fit(data = train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#final_fit3<- wf3 %>%\n#  fit(data = train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred <- predict(final_fit, new_data = test, type = \"numeric\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred1 <- predict(final_fit1, new_data = test, type = \"numeric\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred2 <- predict(final_fit2, new_data = test, type = \"numeric\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pred3 <- predict(final_fit3, new_data = test, type = \"numeric\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PRE <- cbind(pred, pred1,pred2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PRE$go <- (PRE[,1]*0.3+PRE[,2]*0.2+PRE[,3]*0.5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub<-cbind(id,pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1<-cbind(id,pred1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub2<-cbind(id,pred2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sub3<-cbind(id,pred3)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub10<-cbind(id,PRE[,3])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub1)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub2)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub10)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#colnames(sub3)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write.csv(sub,\"submission.csv\",quote = FALSE, row.names = FALSE)\nwrite.csv(sub1,\"submission1.csv\",quote = FALSE, row.names = FALSE)\nwrite.csv(sub2,\"submission2.csv\",quote = FALSE, row.names = FALSE)\nwrite.csv(sub10,\"submission10.csv\",quote = FALSE, row.names = FALSE)\n#write.csv(sub3,\"submission3.csv\",quote = FALSE, row.names = FALSE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(sub)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(sub1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(sub2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(sub10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**STACKING**","metadata":{}},{"cell_type":"code","source":"set.seed(123)\n\nrec <- recipe(Premium.Amount　~ ., data = train_data) %>%\n　step_impute_median(all_numeric_predictors())%>%\n　step_YeoJohnson(all_numeric_predictors()) %>% \n  step_normalize(all_numeric_predictors()) %>%\n　step_novel(all_nominal_predictors())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"set.seed(321)\n\ncv_fold <- vfold_cv (train_data, v = 6,strata = \"Premium.Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_spec <-\n  boost_tree(\n    trees = tune(),\n    tree_depth = tune(),\n    learn_rate = tune(),\n    mtry = tune(),\n    min_n = tune(),\n    loss_reduction = tune()\n  ) %>%\n  set_engine(engine = \"lightgbm\",\n             is_unbalance = TRUE,\n             num_leaves = tune(),\n             lambda_l2 = 0.03977667923976626,\n             lambda_l1 = 0.08851731976964569,\n             nthread  = future::availableCores()\n) %>%\n  set_mode(mode = \"regression\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wf <- workflow() %>% \n  add_model(lgbm_spec) %>% \n  add_recipe(rec)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmsle_impl <- function(truth, estimate, case_weights = NULL) {\n      sqrt( mean(  ((log1p(truth) - log1p(estimate))^2 )))\n    \n}\n\n\nrmsle_vec <- function(truth, estimate, na_rm = TRUE, case_weights = NULL, ...) {\n  check_numeric_metric(truth, estimate, case_weights)\n\n  if (na_rm) {\n    result <- yardstick_remove_missing(truth, estimate, case_weights)\n\n    truth <- result$truth\n    estimate <- result$estimate\n    case_weights <- result$case_weights\n  } else if (yardstick_any_missing(truth, estimate, case_weights)) {\n    return(NA_real_)\n  }\n\n  rmsle_impl(truth, estimate, case_weights = case_weights)\n}\n\nrmsle <- function(data, ...) {\n  UseMethod(\"rmsle\")\n}\n\nrmsle <- new_numeric_metric(rmsle, direction = \"minimize\")\n\nrmsle.data.frame <- function(data, truth, estimate, na_rm = TRUE, case_weights = NULL, ...) {\n\n  numeric_metric_summarizer(\n    name = \"rmsle\",\n    fn = rmsle_vec,\n    data = data,\n    truth = !!enquo(truth),\n    estimate = !!enquo(estimate),\n    na_rm = na_rm,\n    case_weights = !!enquo(case_weights)\n  )\n}\n\nmetric_set(rmsle)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params <- wf %>%\n  extract_parameter_set_dials() %>%\n  update(trees = trees(c(50,2200)),\n    mtry = mtry(range = c(3,21)),\n         min_n = min_n(range = c(25, 90)),\n         tree_depth = tree_depth(range = c(3, 40)),\n         learn_rate = learn_rate(range = c(-3.2, -1.5)),\n         loss_reduction = loss_reduction(c(1, 20)),\n         num_leaves = num_leaves(range = c(200, 300))) %>%\n  finalize(train_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"set.seed(123)\n\nlgbm_res <- tune_grid(\n  wf,\n  resamples = cv_fold,\n  grid = 50,\n  metrics = metric_set(rmsle),\n  param_info = params,\n  control = control_grid(\n    save_pred = TRUE,    # 予測値を保存\n    save_workflow = TRUE # ワークフローを保存\n  ))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"autoplot(lgbm_res) +\n  theme(\n    legend.position = \"top\",\n    strip.background = element_rect(fill = \"white\"),\n    strip.background.x = element_rect(colour = \"white\"),\n    strip.background.y = element_rect(colour = \"white\"),\n    strip.text = element_text(\n      color = \"black\",\n      face = \"bold\",\n      size = 15\n    )\n  ) +\n  labs(title = \"Autoplot | Tune Grid\",\n    y = \"rmsle\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_best(lgbm_res, metric = \"rmsle\") %>%\n  kbl() %>%\n  kable_classic(full_width = F, position = \"left\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a<-select_best(lgbm_res, metric = \"rmsle\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_fit <- wf %>%\n  finalize_workflow(select_best(lgbm_res, metric = \"rmsle\")) %>%\n  fit(train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred <- predict(final_fit, new_data = test, type = \"numeric\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub<-cbind(id,pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write.csv(sub,\"submission100.csv\",quote = FALSE, row.names = FALSE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensamble_stacking <- stacks() %>%  \n  add_candidates(lgbm_res)%>% \n  stacks::blend_predictions(\n    penalty = c(10^seq(-4,-1, 0.1)),\n    metric = metric_set(rmsle),\n    control = tune::control_grid(allow_par = TRUE)\n  )\n\nautoplot(ensamble_stacking) +\n  theme(\n    legend.position = \"top\",\n    strip.background = element_rect(fill = \"white\"),\n    strip.background.x = element_rect(colour = \"white\"),\n    strip.background.y = element_rect(colour = \"white\"),\n    strip.text = element_text(\n      color = \"black\",\n      face = \"bold\",\n      size = 7\n    )\n  ) +\n  labs(title = \"Autoplot Mean and Penalty | Ensamble Stacking\",\n    y = \"RMSLE\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble <- fit_members(ensamble_stacking)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit <- stacks::augment(ensemble, test) \n\nhead(submit)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub<-cbind(id,submit[,33])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(sub)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write.csv(sub,\"submission150.csv\",quote = FALSE, row.names = FALSE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(sub)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tt <- read.csv(\"/kaggle/input/submission5/submission.csv\", row.names= 1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T01:02:37.569474Z","iopub.execute_input":"2024-12-25T01:02:37.601635Z","iopub.status.idle":"2024-12-25T01:03:06.99122Z","shell.execute_reply":"2024-12-25T01:03:06.989426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ttsub<-cbind(sub[,2],tt)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ttsub$pred<-ttsub[,1]*0.3+ttsub[,2]*0.7","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ttsub<-cbind(id,ttsub[,2])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colnames(ttsub)<-c(\"id\",\"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write.csv(ttsub,\"submission200.csv\",quote = FALSE, row.names = FALSE)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}