{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-17T13:37:42.594435Z","iopub.execute_input":"2023-07-17T13:37:42.594875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n!pip install fastai -Uqq\n!pip install ipyplot -Uqq\n\nfrom fastai.data.all import *\nfrom fastai.vision.all import *\nfrom PIL import Image\nimport plotly.express as px\nimport os","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/rsna-intracranial-hemorrhage-detection/')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_trn = path/'stage_2_train'\nfns_trn = path_trn.ls()\nfns_trn[:5].attrgot('name')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_tst = path/'stage_2_test'\nfns_tst = path_tst.ls()\nlen(fns_trn), len(fns_tst)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = fns_trn[0]\ndcm = fn.dcmread()\ndcm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_lbls():\n    path_lbls = path/'stage_1_train.csv'\n    lbls = pd.read_csv(path_lbls)\n    lbls[[\"ID\",\"htype\"]] = lbls.ID.str.rsplit(\"_\", n=1, expand=True)\n    lbls.drop_duplicates(['ID','htype'], inplace=True)\n    pvt = lbls.pivot('ID', 'htype', 'Label')\n    pvt.reset_index(inplace=True)    \n    pvt.to_feather('labels.fth')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_lbls()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lbls = pd.read_feather('labels.fth').set_index('ID')\ndf_lbls.head(8)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lbls.mean()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There's not much RAM on these kaggle kernel instances, so we'll clean up as we go.","metadata":{}},{"cell_type":"code","source":"del(df_lbls)\nimport gc; gc.collect();","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DICOM Meta","metadata":{}},{"cell_type":"code","source":"%timeit \ndf_tst = pd.DataFrame.from_dicoms(fns_tst, px_summ=True)\ndf_tst.to_feather('df_tst.fth')\ndf_tst.head()","metadata":{},"execution_count":null,"outputs":[]}]}