{
  "$schema": "http://json-schema.org/draft-07/schema#",
  "title": "USGS GeoPackage Schema",
  "type": "object",
  "properties": {},
  "patternProperties": {
    "^USGS_[0-9a-f]+_geometry$": {
      "type": "object",
      "title": "Geometry Layer",
      "description": "A geometry layer containing X–Y data parsed from ScienceBase and converted into simplified geospatial features. These features represent the spatial coverage of the data such as polygons for coverage areas or lines for cross-sections, offering greater spatial fidelity than a simple bounding box.",
      "properties": {
        "main_sbid": {
          "type": "string",
          "description": "The 24-character alphanumeric identifier assigned by ScienceBase that uniquely references the dataset.",
          "example": "63140560d34e36012efa2b62"
        },
        "title": {
          "type": "string",
          "description": "Title of the main dataset.",
          "example": "Bathymetry data for the post-construction survey of the Emergent Sandbar Habitat project at river mile 769.8 downstream from Gavins Point Dam on the Missouri River."
        },
        "child_sbid": {
          "type": "string",
          "description": "A 24-character alphanumeric identifier assigned by ScienceBase that uniquely references a child item linked to the main dataset. Child items are populated in the GeoPackage when the main dataset includes one or more associated river survey datasets; if no such datasets exist, this field will be empty.",
          "example": "6408f20bd34e76f5f75e515e"
        },
        "child_title": {
          "type": "string",
          "description": "Title of the child item dataset. Child items are populated in the GeoPackage when the main dataset includes one or more associated river survey datasets; if no such datasets exist, this field will be empty.",
          "example": "Single-beam bathymetric survey of the French Broad River near the Interstate 26 bridge located south of Asheville, NC – December 2021, Mid-Construction #3"
        },
        "file_num": {
          "type": "integer",
          "description": "An integer value that identifies a specific data file (e.g., 1, 2, 3, …) among the total number of files associated with the ScienceBase page. The index does not imply any particular order of files and reflects only the sequence assigned during parsing.",
          "example": 1
        },
        "num_files": {
          "type": "integer",
          "description": "The total number of distinct data files containing river survey data that are referenced by the main ScienceBase item. Each file is counted once, even if it internally includes multiple datasets. For zipped folders, the count reflects the number of individual data files within the folder, not the folder itself. For landing pages with child items, the count includes all individual data files (including those within zipped folders) across both the main item and its child items.",
          "example": 12
        },
        "class": {
          "type": "string",
          "description": "A categorical value assigned by the parser that classifies each dataset as a cross-section, river_reach, or point_location. This classification determines how nominal geometries are derived from the transformed X–Y data using GeoPandas. For example, cross sections may be represented as lines, river reaches as polygons, and individual survey points as locations. The resulting geometries provide a simplified representation of spatial coverage and do not preserve full point-level fidelity.",
          "example": "river_reach"
        },
        "comid": { 
          "type": "string",
          "description": "One or more NHDPlus COMID identifiers associated with the geometry. If the geometry overlaps multiple NHD flowlines, the COMIDs are provided as a comma-separated list.",
          "example": "2496106,4296899,4296271,2496102,2496100,2496098"
        },
        "data_profile": {
          "type": "string",
          "description": "Indicates the spatial data profile that best matches the dataset contained in the ScienceBase data file. Only one profile may be assigned per file. If multiple profiles are present, the data must be separated into distinct files according to profile.",
          "enum": [
            "X-Y",
            "X-Y-Z",
            "X-Y-z",
            "geographic_locatorID-data",
            "None"
          ],
          "enumDescriptions": {
            "X-Y": "Georeferenced coordinates only, with no elevation/height/depth.",
            "X-Y-Z": "Georeferenced coordinates with elevation/height/depth referenced to an established datum.",
            "X-Y-z": "Georeferenced coordinates with height/depth referenced to an arbitrary datum.",
            "geographic_locatorID-data": "Non-georeferenced data explicitly associated with a unique identifier that implies an X-Y location (e.g., station ID).",
            "None": "Non-georeferenced data that do not fit into any of the defined data profiles."
          },
          "example": "X-Y-Z"
        }
        "addtl_attributes": {
          "type": "integer",
          "description": "Number of additional attributes, other than those defined in the data profile, that were explicitly designated to be parsed from the data file and included in the data layer of the GeoPackage.",
          "example": "1"
        },
        "data_type": {
          "type": "string",
          "description": "Specifies the type of spatial data contained in the dataset.",
          "enum": [
            "bathymetry",
            "topography",
            "topobathymetry"
          ],
          "example": "topobathymetry"
        },
        "survey_method": {
          "type": "string",
          "description": "Specifies the survey method used to collect the data. If multiple methods were used, select 'multiple combinations'.",
          "enum": [
            "sonar",
            "lidar",
            "satellite",
            "photogrametry",
            "gnss",
            "linear measurement",
            "level",
            "multiple combinations"
          ],
          "enumDescriptions": {
            "sonar": "Data collected using sonar-based instruments, typically underwater.",
            "lidar": "Data collected using Light Detection and Ranging (LiDAR) instruments.",
            "satellite": "Data collected via satellite remote sensing.",
            "photogrametry": "Data derived from aerial or ground-based photographs.",
            "gnss": "Data collected using Global Navigation Satellite Systems (e.g., GPS).",
            "linear measurement": "Data obtained through manual linear measurements in the field.",
            "level": "Data collected using surveying levels for relative height differences.",
            "multiple combinations": "Data collected using two or more of the above survey methods."
          },
          "example": "lidar"
        },
        "survey_date": {
          "type": "string",
          "description": "Survey date and or time of the data indicated in the data profile.",
          "example": "2020"
        },
        "data_src": {
          "type": "string",
          "description": "The data filename on ScienceBase.",
          "example": "SacDepthData.csv"
        },
        "data_loc": {
          "type": "string",
          "description": "ScienceBase url of the location of the data file.",
          "example": "https://www.sciencebase.gov/catalog/file/get/5aba8249e4b081f61abb4c75?f=__disk__c5%2Fc7%2F09%2Fc5c70991634c4eb9c56c0562c2b837c0af0b13e3"
        },
        "metadata_src": {
          "type": "string",
          "description": "The metadata filename on ScienceBase.",
          "example": "SacDepthMetadata.xml"
        },
        "metadata_loc": {
          "type": "string",
          "description": "ScienceBase url of the location of the metadata file.",
          "example": "https://www.sciencebase.gov/catalog/file/get/5aba8249e4b081f61abb4c75?f=__disk__1b%2Fdf%2F65%2F1bdf6527e228f86e858c11ee32d7d7c2ed90fb65"
        },
        "bibliodata_src": {
          "type": "string",
          "description": "The filename on ScienceBase containing bibliographic information.",
          "example": "SacDepthMetadata.xml"
        },
        "bibliodata_loc": {
          "type": "string",
          "description": "ScienceBase url of the location of the file containing bibliodata.",
          "example": "https://www.sciencebase.gov/catalog/file/get/5aba8249e4b081f61abb4c75?f=__disk__1b%2Fdf%2F65%2F1bdf6527e228f86e858c11ee32d7d7c2ed90fb65"
        },
        "file_partitioning": {
          "type": "string",
          "description": "Specifies the scheme used to partition the dataset into separate files or layers.",
          "enum": [
            "spatial clustering",
            "chunked_dataset",
            "none",
            "grouped attributes"
          ],
          "enumDescriptions": {
            "spatial clustering": "The dataset contains multiple spatially distinct subsets within the data file.",
            "chunked_dataset": "The data file contained many rows and was divided into smaller chunks, distributed across at least two data layers in the GeoPackage.",
            "none": "No partitioning was applied during processing of the data file.",
            "grouped attributes": "The dataset was split based on one or more attribute columns in the data file used as grouping keys."
          },
          "example": "spatial clustering"
        },
        "file_partition": {
          "type": "string",
          "description": "Identifier for the partition the geometry object is associated with in the data layer of the GeoPackage, corresponding to the chosen file_partitioning scheme. 
                        For 'spatial clustering', each spatial cluster receives an identifier like 'cluster_1', 'cluster_2', etc. 
                        For 'grouped attributes', the partition uses the actual group names from the attribute columns. 
                        For 'chunked_dataset', partitions are named 'chunk_1', 'chunk_2', etc. 
                        For 'none', the partition is 'full_dataset'. 
                        This attribute is also included in the data layer so that individual rows can be linked to the appropriate partition and its geometry.",
          "example": "cluster_1"
        },
        "origin_all": {
          "type": "string",
          "description": "Authors of the dataset.",
          "example": "Carl J. Legleiter, Lee R. Harrison"
        },
        "geometry": {
          "type": "string",
          "description": "Geometry type of the feature.",
          "example": "LineString"
        }
      },
      "examples": [
        "USGS_5abaad29e4b081f61abb4d81_geometry",
        "USGS_64419787d34ee8d4ade7c22f_geometry",
        "USGS_63140560d34e36012efa2b62_geometry"
      ]
    },
    "^USGS_[0-9a-f]+_metadata$": {
      "type": "object",
      "title": "Metadata Layer",
      "description": "Contains metadata attributes describing projection, datum, file partitioning, and other dataset-level details. Mirrors key attributes in the geometry layer to provide consistent row-level linkage.",
      "properties": {
        "file_num": {
          "type": "integer",
          "description": "File number identifier for this dataset.",
          "example": 1
        },
        "num_files": {
          "type": "integer",
          "description": "Total number of distinct data files associated with the ScienceBase item.",
          "example": 12
        },
        "data_profile": {
          "type": "string",
          "description": "Indicates the spatial data profile that best matches the dataset contained in the ScienceBase data file.",
          "enum": [
            "X-Y",
            "X-Y-Z",
            "X-Y-z",
            "geographic_locatorID-data",
            "None"
          ],
          "enumDescriptions": {
            "X-Y": "Georeferenced coordinates only, with no elevation/height/depth.",
            "X-Y-Z": "Georeferenced coordinates with elevation/height/depth referenced to an established datum.",
            "X-Y-z": "Georeferenced coordinates with height/depth referenced to an arbitrary datum.",
            "geographic_locatorID-data": "Non-georeferenced data explicitly associated with a unique identifier that implies an X-Y location (e.g., station ID).",
            "None": "Non-georeferenced data that do not fit into any of the defined data profiles."
          },
          "example": "X-Y-Z"
        },
        "proj_name": { "type": "string", "description": "Projection name.", "example": "UTM Zone 15N" },
        "state_zone": { "type": "string", "description": "State plane zone.", "example": "Minnesota North" },
        "geo_coord_format": { "type": "string", "description": "Geographic coordinate format.", "example": "Decimal Degrees" },
        "horiz_datum": { "type": "string", "description": "Horizontal datum.", "example": "NAD83" },
        "horiz_units": { "type": "string", "description": "Horizontal units.", "example": "Meters" },
        "ellipsoid": { "type": "string", "description": "Ellipsoid model.", "example": "GRS80" },
        "epsg": { "type": "string", "description": "EPSG code.", "example": "26915" },
        "vert_ref": { "type": "string", "description": "Vertical reference.", "example": "NAVD88" },
        "vert_datum": { "type": "string", "description": "Vertical datum.", "example": "NAVD88" },
        "vert_info": { "type": "string", "description": "Additional vertical information such as the geoid model or depth convention used in the data file, depending on which data profile was selected.", "example": "Geoid12B" },
        "file_partitioning": {
          "type": "string",
          "description": "Specifies the scheme used to partition the dataset into separate files or layers.",
          "enum": [
            "spatial clustering",
            "chunked_dataset",
            "none",
            "grouped attributes"
          ],
          "enumDescriptions": {
            "spatial clustering": "The dataset contains multiple spatially distinct subsets within the data file.",
            "chunked_dataset": "The data file contained many rows and was divided into smaller chunks, distributed across at least two data layers in the GeoPackage.",
            "none": "No partitioning was applied during processing of the data file.",
            "grouped attributes": "The dataset was split based on one or more attribute columns in the data file used as grouping keys."
          },
          "example": "spatial clustering"
        },
        "file_partition": {
          "type": "string",
          "description": "Identifier for the partition this metadata corresponds to, based on the file_partitioning scheme. " +
                        "For 'spatial clustering', each cluster receives an identifier like 'cluster_1', 'cluster_2', etc. " +
                        "For 'grouped attributes', the partition uses actual group names from the attribute columns. " +
                        "For 'chunked_dataset', partitions are named 'chunk_1', 'chunk_2', etc. " +
                        "For 'none', the partition is 'full_dataset'. This attribute enables linkage of rows in the corresponding data layer to the metadata for that partition.",
          "example": "cluster_1"
        },
        "num_data_chunks": { "type": "integer", "description": "Number of data chunks if the file was split due to size.", "example": 4 }
      },
      "examples": [
        "USGS_5abaad29e4b081f61abb4d81_metadata",
        "USGS_64419787d34ee8d4ade7c22f_metadata",
        "USGS_63140560d34e36012efa2b62_metadata"
      ]
    },
    "^USGS_[0-9a-f]+_file_[0-9]+_(full|chunk[0-9]+)_data$": {
      "type": "object",
      "title": "Data Layer",
      "description": "Contains measurement data (X, Y, Z) for a specific file or chunk. Horizontal coordinates (X, Y) follow the datum, units, and EPSG code specified in the linked metadata layer. The Z coordinate represents elevation, depth, or height according to the 'data_profile' and is referenced to the vertical datum ('vert_datum') and vertical reference ('vert_ref') in the metadata; it may be empty if the profile is X-Y only. Rows are partitioned according to the file_partitioning scheme in the metadata. Chunked datasets have multiple layers named with '_chunk#', while single layers use '_full_data'. Within each layer, individual rows include a 'file_partition' value indicating their group (e.g., spatial cluster, attribute-based group, or chunk). In addition to the core data profile, each layer may include optional additional attributes extracted from the ScienceBase file; these are appended sequentially as attr_1, attr_2, etc., depending on how many attributes were present. This structure enables programmatic reading of coordinates and any additional attributes, while maintaining consistent linkage to the corresponding metadata for proper interpretation.",
      "properties": {
        "X": { 
          "type": "number",
          "description": "X coordinate in the horizontal datum and units specified in the linked metadata layer (e.g., NAD83, meters).",
          "example": -93.123456
        },
        "Y": {
          "type": "number",
          "description": "Y coordinate in the horizontal datum and units specified in the linked metadata layer (e.g., NAD83, meters).",
          "example": 44.987654
        },
        "Z": {
          "type": "number",
          "description": "Z coordinate representing elevation, depth, or height depending on the 'data_profile'. Values are referenced to the vertical datum ('vert_datum') and vertical reference ('vert_ref') provided in the metadata. May be empty if the dataset contains only X-Y coordinates.",
          "example": 278.3
        },
        "file_partition": {
          "type": "string",
          "description": "Identifier for the subset of data within this layer, based on the file_partitioning scheme in the metadata. Examples include 'cluster_1', 'cluster_2', or 'chunk_1'.",
          "example": "cluster_1"
        }
      },
      "examples": [
        "USGS_5abaad29e4b081f61abb4d81_file_1_full_data",
        "USGS_64419787d34ee8d4ade7c22f_file_1_chunk1_data",
        "USGS_63140560d34e36012efa2b62_file_2_chunk2_data"
      ]
    }
  }
}
