{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.display import Markdown, display\n\ndef printmd(string):\n    display(Markdown(string))\n\nbase_directory = '/kaggle/input/opencontrails-mini-sample/'\n\nprintmd(f\"### **Base Directory:** {base_directory}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:08.770223Z","iopub.execute_input":"2023-07-27T03:06:08.770922Z","iopub.status.idle":"2023-07-27T03:06:08.777824Z","shell.execute_reply.started":"2023-07-27T03:06:08.770886Z","shell.execute_reply":"2023-07-27T03:06:08.77705Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Markdown, display\n\ndef printmd(string):\n    display(Markdown(string))\n\nprintmd(\"### Input Directory\")\n\nfrom utils_directory_tree_generator import get_directory_tree\n\ntree_generator = get_directory_tree(\n    start_path=base_directory,  # Updated this line\n    max_depth=5,\n    include_files=True,\n    sort_by='type',\n    reverse=False,\n    max_items=5\n)\nfor line in tree_generator:\n    print(line)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:08.779627Z","iopub.execute_input":"2023-07-27T03:06:08.780268Z","iopub.status.idle":"2023-07-27T03:06:08.842268Z","shell.execute_reply.started":"2023-07-27T03:06:08.780235Z","shell.execute_reply":"2023-07-27T03:06:08.841333Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Markdown, display\n\ndef printmd(string):\n    display(Markdown(string))\n\nprintmd(\"### Generating Stats Report and Overlaid Band Histogram ...\")\n\nimport os\nimport time\nimport collections\nimport math\nimport re\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nfrom matplotlib.patches import Rectangle\n\n\nclass DataAnalyzer:\n    @staticmethod\n    def calculate_statistics(data):\n        if data.size == 0:\n            return {\n                'shape': data.shape,\n                'size': data.size,\n                'mean': None,\n                'std_dev': None,\n                'min': None,\n                'max': None,\n                'median': None,\n                'percentiles': [None, None, None],\n                'all_zeros': True,\n                'zero_count': 0,\n                'non_zero_count': 0,\n                'has_nan': False,\n                'has_inf': False,\n                'histogram': ([], [])\n            }\n        \n        hist, bins = np.histogram(data, bins=50)\n        hist = hist / data.size\n        return {\n            'shape': data.shape,\n            'size': data.size,\n            'mean': np.mean(data),\n            'std_dev': np.std(data),\n            'min': np.min(data),\n            'max': np.max(data),\n            'median': np.median(data),\n            'percentiles': np.percentile(data, [25, 50, 75]),\n            'all_zeros': np.all(data == 0),\n            'zero_count': np.count_nonzero(data == 0),\n            'non_zero_count': np.count_nonzero(data),\n            'has_nan': np.isnan(data).any(),\n            'has_inf': np.isinf(data).any(),\n            'histogram': (hist, bins)\n        }\n\n\nclass HistogramGenerator:\n    @staticmethod\n    def group_and_plot_histograms_by_dataset_record(file_stats, output_dir, histograms_per_row):\n        results_by_dataset_record = collections.defaultdict(list)\n        \n        for filepath, stats in file_stats.items():\n            _, dataset, record_id, _ = filepath.split('/')[-4:]\n            results_by_dataset_record[(dataset, record_id)].append((filepath, stats))\n\n        dataset_records = sorted(results_by_dataset_record.keys())\n        num_histograms = len(dataset_records)\n        num_rows = math.ceil(num_histograms / histograms_per_row)\n        fig_band = plt.figure(figsize=(15, num_rows * 7))\n\n        patches_band = []\n        labels_band = []\n\n        for i, (dataset, record_id) in enumerate(dataset_records):\n            record_results = results_by_dataset_record[(dataset, record_id)]\n            band_results = [result for result in record_results if 'band_' in result[0].split('/')[-1]]\n            colors_band = cm.viridis(np.linspace(0, 1, len(band_results)))\n\n            ax_band = fig_band.add_subplot(num_rows, histograms_per_row, i + 1)\n            for color, (filepath, stats) in zip(colors_band, band_results):\n                hist, bins = stats['histogram']\n                ax_band.bar(bins[:-1], hist, width=np.diff(bins), ec=\"k\", align=\"edge\", alpha=0.5, color=color)\n\n                if os.path.basename(filepath) not in labels_band:\n                    labels_band.append(os.path.basename(filepath))\n                    patches_band.append(Rectangle((0, 0), 1, 1, fc=color))\n\n            ax_band.set_title(f'Bands for record {record_id} in {dataset}')\n            ax_band.grid(True)\n\n        fig_band.legend(patches_band, labels_band, loc='upper right')\n        output_filepath_band = os.path.join(output_dir, f'tiled_bands_histogram.png')\n        fig_band.savefig(output_filepath_band)\n        plt.close(fig_band)\n        \n    @staticmethod\n    def display_saved_image(img_path):\n        # Load and display saved image\n        img = mpimg.imread(img_path)\n        plt.figure(figsize=(10, 6))\n        plt.imshow(img, aspect='auto')\n        plt.axis('off')\n        plt.show()\n\n\nclass ReportGenerator:\n    ANCHOR_REGEX = re.compile(r'[^\\w\\- ]', re.UNICODE)\n\n    @staticmethod\n    def normalize_anchor(text):\n        text = text.lower().replace(' ', '-')\n        return ReportGenerator.ANCHOR_REGEX.sub('', text)\n\n    @staticmethod\n    def generate_markdown_for_file(filepath, stats):\n        markdown = f\"### {filepath}\\n\\n\"\n        for stat, value in stats.items():\n            markdown += f\"- **{stat.capitalize()}**: {value}\\n\"\n        markdown += f'\\n[Back to ToC](#{ReportGenerator.normalize_anchor(\"table of contents\")})\\n\\n'\n        return markdown\n\n    @staticmethod\n    def generate_markdown_section(title, items):\n        section = f'## {title}\\n\\n'\n        for item in items:\n            section += f\"- {item}\\n\"\n        section += f'\\n[Back to ToC](#{ReportGenerator.normalize_anchor(\"table of contents\")})\\n\\n'\n        return section\n\n    @staticmethod\n    def save_results(markdown_results, files_with_zeros, files_without_zeros):\n        total_files = len(markdown_results)\n        total_files_with_zeros = len(files_with_zeros)\n        total_files_without_zeros = len(files_without_zeros)\n        \n        summary = f\"## Summary\\n\\n\"\n        summary += f\"- Total files processed: {total_files}\\n\"\n        summary += f\"- Files with zeros: {total_files_with_zeros}\\n\"\n        summary += f\"- Files without zeros: {total_files_without_zeros}\\n\\n\"\n        \n        markdown = '# Table of Contents\\n\\n'\n#         markdown += '- [Summary](#summary)\\n'\n#         markdown += summary\n        \n        for md in markdown_results:\n            markdown += md\n        \n        with open('/kaggle/working/numpy_stats/zeros_dataset_report.md', 'w') as results_file:\n            results_file.write(markdown)\n\n            \nclass FileProcessor:\n    @staticmethod\n    def process_file(filepath):\n        try:\n            data = np.load(filepath)\n            stats = DataAnalyzer.calculate_statistics(data)\n            markdown = ReportGenerator.generate_markdown_for_file(filepath, stats)\n            has_zeros = stats['zero_count'] > 0\n            return filepath, markdown, has_zeros, stats\n        except Exception as e:\n            print(f\"Error processing file {filepath}: {str(e)}\")\n            return None, None, None, None\n\n    @staticmethod\n    def process_directory(directory_path):\n        markdown_results, file_stats = [], {}\n        files_with_zeros, files_without_zeros = [], []\n\n        for root, dirs, files in os.walk(directory_path):\n            dirs[:] = [d for d in dirs if d not in ['test']] #adjust to include test dir\n            for file in filter(lambda f: f.endswith('.npy'), files):\n                filepath, markdown, has_zeros, stats = FileProcessor.process_file(os.path.join(root, file))\n                if all([filepath, markdown, stats]):\n                    markdown_results.append(markdown)\n                    file_stats[filepath] = stats\n                    (files_with_zeros if has_zeros else files_without_zeros).append(filepath)\n\n        return markdown_results, files_with_zeros, files_without_zeros, file_stats\n\n\ndef main():\n    start_time = time.time()\n\n    os.makedirs('/kaggle/working/numpy_stats', exist_ok=True)\n    try:\n        markdown_results, files_with_zeros, files_without_zeros, file_stats = FileProcessor.process_directory(base_directory)\n        ReportGenerator.save_results(markdown_results, files_with_zeros, files_without_zeros)\n        HistogramGenerator.group_and_plot_histograms_by_dataset_record(file_stats, '/kaggle/working/numpy_stats', 4) # Adjust number of histograms per row here\n    except Exception as e:\n        print(f\"An error occurred: {str(e)}\")\n\n    elapsed_time = time.time() - start_time\n    processed_files_count = len(markdown_results)\n    print(f\"Total time taken: {elapsed_time:.4f} seconds\")\n    print(f\"Total number of files processed: {processed_files_count}\")\n    print(f\"Processed {processed_files_count/elapsed_time:.2f} files per second.\")\n\n\nmain()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:08.843795Z","iopub.execute_input":"2023-07-27T03:06:08.844142Z","iopub.status.idle":"2023-07-27T03:06:24.834415Z","shell.execute_reply.started":"2023-07-27T03:06:08.844107Z","shell.execute_reply":"2023-07-27T03:06:24.833297Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Usage:\n- Set `base_directory`:\n  - Entire dataset: `'/kaggle/input/google-research-identify-contrails-reduce-global-warming'`\n  - `test` folder only: `'/kaggle/input/google-research-identify-contrails-reduce-global-warming/test'`\n- Execute code.\n- Check output at `'/kaggle/working/numpy_stats'`.","metadata":{}},{"cell_type":"code","source":"from IPython.display import Markdown, display\n\ndef printmd(string):\n    display(Markdown(string))\n\nprintmd(\"### Output Directory:\")\n\nfrom utils_directory_tree_generator import get_directory_tree\n\ntree_generator = get_directory_tree(\n    start_path='/kaggle/working/numpy_stats',\n    max_depth=4,\n    include_files=True,\n    sort_by='type',\n    reverse=False,\n    max_items=4\n)\nfor line in tree_generator:\n    print(line)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:24.837114Z","iopub.execute_input":"2023-07-27T03:06:24.83746Z","iopub.status.idle":"2023-07-27T03:06:24.846257Z","shell.execute_reply.started":"2023-07-27T03:06:24.83743Z","shell.execute_reply":"2023-07-27T03:06:24.845154Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\ndef get_image_title(img_path):\n    # Get title from path\n    return img_path.split('/')[-1].split('.')[0].replace('_', ' ').title()\n\ndef display_image(ax, img_path):\n    # Load and display image\n    img = mpimg.imread(img_path)\n    ax.imshow(img, aspect='auto')\n    ax.axis('off')\n    ax.set_title(get_image_title(img_path))\n\ndef display_images(image_files):\n    # Display list of images\n    n_images = len(image_files)\n    fig = plt.figure(figsize=(10 * n_images, 6))\n    \n    for i, image_file in enumerate(image_files):\n        ax = fig.add_subplot(1, n_images, i+1)\n        img_path = f'/kaggle/working/numpy_stats/{image_file}'\n        display_image(ax, img_path)\n    \n    plt.tight_layout()\n    plt.show()\n\n# Image list\nimage_files = ['tiled_bands_histogram.png']\ndisplay_images(image_files)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:24.847615Z","iopub.execute_input":"2023-07-27T03:06:24.848012Z","iopub.status.idle":"2023-07-27T03:06:25.669207Z","shell.execute_reply.started":"2023-07-27T03:06:24.847975Z","shell.execute_reply":"2023-07-27T03:06:25.668129Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Overlayed histograms highlight pixel distributions across spectral bands, predominantly showcasing low reflectance values in satellite imagery data.*\n\n# Multi-band Image Histograms\n\n- **Multi-band Composition**: Histogram shows different bands reflecting varied spectral information typical in remote sensing imagery.\n\n## Observations:\n\n- **Distribution Patterns**: Some bands show a bimodal distribution, reflecting distinct features like terrain types or water bodies.\n\n- **Intensity and Frequency**: The x-axis represents pixel values; the y-axis shows pixel counts. Peaks indicate prevalent pixel values for bands.\n\n- **Band Variances**: Varied distribution shapes across bands suggest each band captures distinct scene information.\n\n- **Noise Indicators**: Spikes at intensity extremes might indicate noise or outliers.\n\n## Insights from Code:\n\n- **Normalization**: Histograms emphasize proportion, not count.\n\n- **Band Coloration**: Color scheme shows band order, potentially by wavelength.\n\n- **Data Integrity**: Checks for zeros and NaNs ensure accuracy.\n\n## Mini-roadmap:\n\n- **X-axis Clarification**: Enhancing labels offers clearer context.\n\n- **Annotations**: Significant peak markers improve interpretation.\n\n- **Metadata Integration**: Including band-specific details, like its purpose or wavelength, enriches understanding.","metadata":{}},{"cell_type":"code","source":"from IPython.display import Markdown, display\nimport json\nimport os\nimport matplotlib.pyplot as plt\nfrom datetime import datetime\nimport collections\n\n\nprint(os.getcwd())\n\ndef printmd(string):\n    display(Markdown(string))\n\nclass DataLoader:\n    @staticmethod\n    def load_json_file(filename):\n        # Load JSON from a file\n        with open(filename, 'r') as f:\n            return json.load(f)\n\n    @staticmethod\n    def get_timestamps(data):\n        # Extract timestamps from data\n        return [entry[\"timestamp\"] for entry in data]\n\n    @staticmethod\n    def to_datetime_format(timestamps):\n        # Convert timestamps to datetime\n        return [datetime.utcfromtimestamp(ts).strftime('%Y-%m') for ts in timestamps]\n\n\nclass PlotGenerator:\n    @staticmethod\n    def plot_and_save_histogram(train_dates, validation_dates, save_path=None):\n        # Plot histogram and save\n        plt.figure(figsize=(15, 6))\n        plt.hist(train_dates, bins=30, alpha=0.5, label=\"Training Dataset\", color=\"blue\", edgecolor=\"black\")\n        plt.hist(validation_dates, bins=30, alpha=0.5, label=\"Validation Dataset\", color=\"orange\", edgecolor=\"black\")\n        plt.title(\"Temporal Distribution of Training vs. Validation Datasets\")\n        plt.xlabel(\"Date\")\n        plt.ylabel(\"Number of Records\")\n        plt.xticks(rotation=45)\n        plt.legend(loc=\"upper left\")\n        plt.tight_layout()\n\n        if save_path:\n            plt.savefig(save_path)\n            print(f\"Plot saved to: {save_path}\")\n\n        plt.show()\n\n    @staticmethod\n    def ensure_dir_exists(directory):\n        # Create directory if not exists\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n\n\ndef main():\n    printmd(\"### Loading Data and Generating Temporal Distribution Histogram ...\")\n\n    base_directory = '/kaggle/input/opencontrails-mini-sample/opencontrails_mini_sample/'\n\n    # Paths\n    train_file_path = os.path.join(base_directory, \"train_metadata.json\")\n    validation_file_path = os.path.join(base_directory, \"validation_metadata.json\")\n\n    \n    # Load data and convert timestamps\n    train_dates = DataLoader.to_datetime_format(DataLoader.get_timestamps(DataLoader.load_json_file(train_file_path)))\n    validation_dates = DataLoader.to_datetime_format(DataLoader.get_timestamps(DataLoader.load_json_file(validation_file_path)))\n\n    if not train_dates and not validation_dates:\n        print(\"No valid months found for datasets.\")\n        return\n\n    # Directories setup\n    output_dir = \"/kaggle/working/output\"\n    script_name_dir = \"temporal_distribution_analysis\"\n    full_dir_path = os.path.join(output_dir, script_name_dir)\n    PlotGenerator.ensure_dir_exists(full_dir_path)\n\n    # Plot and save histogram\n    plot_filename = os.path.join(full_dir_path, \"train_vs_validation_distribution.png\")\n    PlotGenerator.plot_and_save_histogram(train_dates, validation_dates, save_path=plot_filename)\n\n\nmain()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:25.670649Z","iopub.execute_input":"2023-07-27T03:06:25.671328Z","iopub.status.idle":"2023-07-27T03:06:26.73092Z","shell.execute_reply.started":"2023-07-27T03:06:25.671288Z","shell.execute_reply":"2023-07-27T03:06:26.729873Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# From the temporal breakdown of the Training vs. Validation datasets:\n  - **Training Dataset:**\n    - Records span several months.\n    - Uneven distribution with peaks in specific months.\n  - **Validation Dataset:**\n    - Data concentrated in fewer months.\n    - Likely a specific collection period or subset.\n\n- **Key Insights:**\n  - Temporal distribution can explain model performance anomalies.\n  - Potential reasons: data underrepresentation or external influences.\n  - Uneven monthly distribution may impact model's handling of varying solar zenith angles.\n\n- **Next Steps:**\n  - Examine spatial distribution and its relation to solar zenith angle.\n  - Open for feedback or questions on the temporal analysis.","metadata":{}},{"cell_type":"code","source":"from IPython.display import Markdown\n\n# Display contents of zeros_dataset_report.md\nwith open('/kaggle/working/numpy_stats/zeros_dataset_report.md', 'r') as f:\n    markdown_content = f.read()\nMarkdown(markdown_content)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:06:26.732332Z","iopub.execute_input":"2023-07-27T03:06:26.73275Z","iopub.status.idle":"2023-07-27T03:06:26.744057Z","shell.execute_reply.started":"2023-07-27T03:06:26.732712Z","shell.execute_reply":"2023-07-27T03:06:26.743124Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 id=\"credit-goes-to-all-authors-and-contributors\">🥇 Credit goes to all authors and contributors ⤵︎ </h2>\n\n- This script is inspired by [Google Research - Identify Contrails to Reduce Global Warming](https://www.kaggle.com/competitions/google-research-identify-contrails-reduce-global-warming)\n\n---\n# Appendix\n\n- [GitHub: README.md](https://github.com/patmejia/contrails-vision/blob/main/): execute the Python code locally.\n\n\n<div style=\"background-color: #f2f2f2; padding: 53px; border-radius: 5px;\">\n  <h3>If you found this notebook helpful...</h3>\n  <p>\n  Please consider giving it a star. Your support helps me continue to develop high-quality code and pursue my career as a data analyst/engineer. Feedback is always welcome and appreciated. Thank you for taking the time to read my work! \n  </p> \n  <h4>\n  <p style=\"text-align: right;\">\n  <a href=\"https://github.com/patmejia\"> - pat [¬º-°]¬ </a>\n  </h4>\n  </p>\n</div>","metadata":{}}]}