EMIT¶
Install package with EMIT dependencies¶
EMIT products are NetCDF files. georeader reads them through xarray, with h5netcdf as the HDF5 backend. pysolar is used to convert the EMIT L1B radiances to reflectances.
pip install georeader-spaceml xarray h5netcdf pysolar
Download an EMIT image¶
from georeader.readers import emit
from georeader.readers import download_utils
import os
import warnings
from pathlib import Path
import urllib3
# download_product talks to LP DAAC with certificate verification off; silence the per-request warning.
warnings.filterwarnings("ignore", category=urllib3.exceptions.InsecureRequestWarning)
# Resolve the repo-level examples/ folder (works regardless of where the notebook runs from).
EXAMPLES_DIR = next(
(p / "examples" for p in [Path.cwd(), *Path.cwd().parents] if (p / "examples").is_dir()),
Path("examples"),
)
dir_emit_files = EXAMPLES_DIR / "EMIT"
os.makedirs(dir_emit_files, exist_ok=True)
# Downloading EMIT products requires a NASA Earthdata account. Either set the
# EARTHDATA_TOKEN environment variable (bearer token from
# https://urs.earthdata.nasa.gov/profile) or create ~/.georeader/auth_emit.json
# with {"user": "...", "password": "..."}. See examples/README.md.
if os.environ.get("EARTHDATA_TOKEN"):
emit.AUTH_METHOD = "token"
emit.TOKEN = os.environ["EARTHDATA_TOKEN"]
# link = 'https://data.lpdaac.earthdatacloud.nasa.gov/lp-prod-protected/EMITL1BRAD.001/EMIT_L1B_RAD_001_20220828T051941_2224004_006/EMIT_L1B_RAD_001_20220828T051941_2224004_006.nc'
link = 'https://data.lpdaac.earthdatacloud.nasa.gov/lp-prod-protected/EMITL1BRAD.001/EMIT_L1B_RAD_001_20220827T060753_2223904_013/EMIT_L1B_RAD_001_20220827T060753_2223904_013.nc'
file_save = os.fspath(dir_emit_files / os.path.basename(link))
emit.download_product(link, file_save)
Create and inspec emit object¶
EMIT objects let you open an EMIT nc file without loading the content of the file in memory. The object in the cell below has 285 spectral bands. For the API description of the EMIT class see the emit module. Since the object follows the API of georeader, you can read from the rasters using the functions from the read module.
# file_save = "emit_database/EMIT_L1B_RAD_001_20220827T060753_2223904_013.nc"
ei = emit.EMITImage(file_save)
ei
ei.nc_ds
ei.wavelengths
ei.time_coverage_start, ei.time_coverage_end
import geopandas as gpd
gpd.GeoDataFrame(geometry=[ei.footprint()],
crs=ei.crs).explore()
Load RGB¶
Select the RGB bands, we see that the raster has only 3 channels now (see Shape in the output of the cell below). The ei object has an attribute called wavelengths with the central wavelength of the hyperspectral band.
import numpy as np
wavelengths_read = np.array([640, 550, 460])
bands_read = np.argmin(np.abs(wavelengths_read[:, np.newaxis] - ei.wavelengths), axis=1).tolist()
ei_rgb = ei.read_from_bands(bands_read)
ei_rgb
Normalize radiance to reflectance¶
import pandas as pd
import matplotlib.pyplot as plt
from georeader import reflectance
thuiller = reflectance.load_thuillier_irradiance() # (dataframe with 8191 rows, 2 colums -> Nanometer, Radiance(mW/m2/nm)
ei_rgb.wavelengths, ei_rgb.fwhm # (K,) vectors with the center wavelength and the FWHM
response = reflectance.srf(ei_rgb.wavelengths, ei_rgb.fwhm, thuiller["Nanometer"].values) # (8191, K)
colors = ["red","green", "blue"]
fig, ax = plt.subplots(1,1,figsize=(12,4))
ax.plot(thuiller["Nanometer"].values,
thuiller["Radiance(mW/m2/nm)"].values,
label="Thuiller irradiance")
ax.set_ylabel("Solar irradiance (mW/m$^2$/nm)")
ax = ax.twinx()
for k in range(3):
ax.plot(thuiller["Nanometer"].values, response[:, k],
label=f"{wavelengths_read[k]}nm",c=colors[k])
ax.set_xlabel("Wavelength nm")
ax.set_ylabel("SRF")
ax.set_xlim(410,700)
ax.legend(loc="upper left")
# solar_irradiance_norm = thuiller["Radiance(mW/m2/nm)"].values.dot(response) # mW/m$^2$/nm
# solar_irradiance_norm/=1_000 # W/m$^2$/nm
# solar_irradiance_norm
# ei_rgb_local = ei_rgb.load(as_reflectance=False)
# ei_rgb_local = reflectance.radiance_to_reflectance(ei_rgb_local, solar_irradiance_norm,
# ei.time_coverage_start,units=units)
ei_rgb_local = ei_rgb.load(as_reflectance=True)
from georeader.plot import show
show((ei_rgb_local).clip(0,1),
mask=ei_rgb_local.values == ei_rgb_local.fill_value_default)
Reproject to UTM¶
import georeader
crs_utm = georeader.get_utm_epsg(ei.footprint("EPSG:4326"))
emit_image_utm = ei.to_crs(crs_utm)
emit_image_utm_rgb = emit_image_utm.read_from_bands(bands_read)
emit_image_utm_rgb_local = emit_image_utm_rgb.load(as_reflectance=True)
emit_image_utm_rgb_local
emit_image_utm_rgb.observation_date_correction_factor
show((emit_image_utm_rgb_local).clip(0,1),
mask=emit_image_utm_rgb_local.values == emit_image_utm_rgb_local.fill_value_default,
add_scalebar=True)
show(emit_image_utm.elevation(), add_colorbar_next_to=True, add_scalebar=True,
mask=True,title="Elevation")
Subset EMIT image¶
from georeader import read
point_tup = (61.28, 36.21)
ei_subset = read.read_from_center_coords(emit_image_utm_rgb, point_tup,
shape=(200,200),crs_center_coords="EPSG:4326")
ei_subset_local = ei_subset.load(as_reflectance=True)
ei_subset_local
from georeader.plot import add_shape_to_plot
from shapely.geometry import Point
ax = show((ei_subset_local).clip(0,1),
mask=ei_subset_local.values == ei_rgb_local.fill_value_default,
add_scalebar=True)
add_shape_to_plot(Point(*point_tup),crs_shape="EPSG:4326",
crs_plot=ei_subset_local.crs, ax=ax)
import matplotlib.pyplot as plt
ei_rgb_raw = ei_rgb.load_raw(transpose=False)
plt.imshow((ei_rgb_raw/12).clip(0,1))
Subset¶
ei_rgb_subset = ei_subset.load_raw(transpose=False)
plt.imshow((ei_rgb_subset/12).clip(0,1))
Load L2AMask¶
The L2AMask is stored in a separated file. The mask array has the following variables:
emit_image_utm.mask_bands
# mask filtering cloudy pixels
show(emit_image_utm.validmask(), add_scalebar=True,vmin=0, vmax=1,
mask=True,title="Valid Mask")
Load metadata¶
Metadata is stored in a separated file (with suffix _OBS_ instead of _RAD_). In metadata we have the following variables:
emit_image_utm.observation_bands
show(emit_image_utm.sza(), add_colorbar_next_to=True, add_scalebar=True,
mask=True,title="Solar Zenith Angle")
show(emit_image_utm.vza(), add_colorbar_next_to=True, add_scalebar=True,
mask=True,title="Viewing Zenith Angle")
show(emit_image_utm.observation('Path length (sensor-to-ground in meters)'),
add_colorbar_next_to=True, add_scalebar=True,
mask=True,title='Path length (sensor-to-ground in meters)')
show(emit_image_utm.observation('Slope (local surface slope as derived from DEM in degrees)'),
add_colorbar_next_to=True, add_scalebar=True,
mask=True,title='Slope (local surface slope as derived from DEM in degrees)')
EMIT product versions: v001 and v002¶
NASA LP DAAC closed the EMIT v001 collections on 2026-08-31. New acquisitions are published only as v002, with the L2A mask as its own v003 collection, and NASA is reprocessing the v001 archive into v002. On 2026-09-23 the reprocessed archive covered 2022-08-10 to 2022-09-12 and 2023-01-07 to 2023-01-31, so the scene used in this tutorial exists in both versions.
| v001 | v002 | |
|---|---|---|
| L1B radiance name | EMIT_L1B_RAD_001_<datetime>_<orbit>_<scene> |
EMIT_L1B_RAD_002_<datetime> |
| Observation (OBS) file | in the L1B granule | in the L1B granule |
| L2A mask | EMIT_L2A_MASK_001_…, shipped inside the L2A RFL v001 granule |
EMIT_L2A_MASK_003_…, its own collection |
| L2A mask bands | Cloud, Cirrus, Water, Spacecraft, Dilated Cloud, AOD550, H2O, Aggregate | Cloud, Cirrus, Water, Dilated Cloud, SpecTf-Cloud Probability, SpecTf-Cloud Flag, SpecTf-Buffer Distance |
| Calibration | PGE 1.x | PGE 2.0.0: updated radiometric calibration, a spectral tilt fix at the longest wavelengths, a fix for the filter seam near 1280 nm and trimmed focal-plane edges |
| Extra variables | — | flat_field_update (cross-track × band) |
| L2B CH4 / CO2 enhancement | available as CH4ENH / CO2ENH v002 (the v001 collections were emptied) | not produced yet |
The datetime in the name is the acquisition start and is the only identifier shared by every product and version of a scene. The raw sensor grid, the GLT and the variables EMITImage reads are unchanged, so georeader reads both versions the same way: the get_*_link functions pick the companion collections from the L1B version, and validmask() selects the mask flags by name. See the LP DAAC product page for the full list of changes.
# The same acquisition in v002. v002 names drop the orbit/scene suffix, so build the name from the scene id.
scene_fid, orbit, daac_scene_number, acquisition = emit.split_product_name(file_save)
name_v002 = emit.product_name_from_params(scene_fid, version="002")
link_v002 = emit.get_radiance_link(name_v002)
file_save_v002 = os.fspath(dir_emit_files / os.path.basename(link_v002))
if not os.path.exists(file_save_v002):
emit.download_product(link_v002, file_save_v002, display_progress_bar=False)
ei_v002 = emit.EMITImage(file_save_v002)
link_v002
The table compares the two files. Reading mask_bands and percentage_clear downloads each version's L2A mask next to the radiance file.
import pandas as pd
def describe(image: emit.EMITImage) -> dict:
ch4 = emit.get_ch4enhancement_link(image.filename)
return {
"radiance file": os.path.basename(image.filename),
"L2A mask file": os.path.basename(emit.get_l2amask_link(image.filename)),
"CH4 enhancement file": os.path.basename(ch4) if ch4 else "not produced",
"bands": len(image.wavelengths),
"first / last wavelength (nm)": f"{image.wavelengths[0]:.3f} / {image.wavelengths[-1]:.3f}",
"has flat_field_update": "flat_field_update" in image.nc_ds.variables,
"mask bands": ", ".join(image.mask_bands),
"clear pixels (%)": round(image.percentage_clear, 2),
}
with pd.option_context("display.max_colwidth", None):
display(pd.DataFrame({"v001": describe(ei), "v002": describe(ei_v002)}))
Radiance: same scene, new calibration¶
Both versions use the same raw sensor grid, so the same detector pixel can be compared directly. The plot shows the median v002 / v001 radiance ratio per band over a block of raw rows in the middle of the scene. Bands with very low radiance (the water-vapour absorption bands) are left out because their ratio is dominated by noise. For this scene the two versions agree closely except near the 1280 nm filter seam, around 2100 nm and in the longest wavelengths, which are the regions the v002 calibration changed.
import matplotlib.pyplot as plt
center_row = ei.nc_ds.sizes["downtrack"] // 2
rows = slice(center_row - 8, center_row + 8)
rad_v001 = ei.nc_ds["radiance"].isel(downtrack=rows).values
rad_v002 = ei_v002.nc_ds["radiance"].isel(downtrack=rows).values
valid = (rad_v001 > 0.05) & (rad_v002 > 0.05) # excludes fill values (-9999) and absorption bands
ratio = np.where(valid, rad_v002, np.nan) / np.where(valid, rad_v001, np.nan)
with warnings.catch_warnings():
warnings.simplefilter("ignore", RuntimeWarning) # absorption bands have no valid pixels
median_ratio = np.nanmedian(ratio.reshape(-1, ratio.shape[-1]), axis=0)
fig, ax = plt.subplots(figsize=(9, 3), dpi=72)
ax.plot(ei.wavelengths, median_ratio)
ax.axhline(1, color="k", linewidth=0.5)
ax.set_xlabel("Wavelength (nm)")
ax.set_ylabel("median radiance v002 / v001")
ax.set_title(f"{scene_fid}, raw rows {rows.start} to {rows.stop - 1}")
ax.grid()
Cloud masks: different band layout¶
The v003 mask drops the Spacecraft flag and adds the SpecTf machine-learning cloud probability, flag and buffer distance. validmask() selects flags by name: Cloud and Cirrus (plus Spacecraft on v001) and, with the default with_buffer=True, the Dilated Cloud flag. include_spectf=True also masks pixels flagged by the SpecTf cloud flag. This desert scene is cloud-free (100 % clear pixels in both versions), so the three masks are identical here.
emit_image_v002_utm = ei_v002.to_crs(crs_utm)
fig, axs = plt.subplots(1, 3, figsize=(12, 4), dpi=72, sharex=True, sharey=True)
show(emit_image_utm.validmask(), ax=axs[0], vmin=0, vmax=1, mask=True, title="v001 valid mask")
show(emit_image_v002_utm.validmask(), ax=axs[1], vmin=0, vmax=1, mask=True, title="v002 valid mask")
show(emit_image_v002_utm.validmask(include_spectf=True), ax=axs[2], vmin=0, vmax=1, mask=True,
title="v002 valid mask + SpecTf flag")
A whirlwind tour of the other EMIT products¶
Besides L1B radiance, LP DAAC publishes several higher-level EMIT products. The chart shows, for the newest version of each collection, which acquisition dates had granules on 2026-09-23. Blue bars are available, red bars are dates not yet reprocessed to v002, and grey bars are collections that no longer receive new granules. All scene-level products are named after the acquisition datetime of the L1B granule.
%%{init: {"gantt": {"leftPadding": 170, "barHeight": 18, "fontSize": 12}}}%%
gantt
title Newest EMIT collections at LP DAAC, granules available on 2026-09-23
dateFormat YYYY-MM-DD
axisFormat %Y
tickInterval 12month
todayMarker off
section RAD, RFL v002 + MASK v003
reprocessed :active, 2022-08-10, 2022-09-12
not yet reprocessed :crit, 2022-09-12, 2023-01-07
reprocessed :active, 2023-01-07, 2023-01-31
not yet reprocessed :crit, 2023-01-31, 2026-08-26
new acquisitions :active, 2026-08-26, 2026-09-23
section RAD, RFL, MASK v001
closed :done, 2022-08-10, 2026-08-31
section MIN v001
mineral maps, closed :done, 2022-08-10, 2026-08-31
section FRCOV v001
fractional cover, closed :done, 2022-08-10, 2026-08-31
section CH4ENH, CO2ENH v002
enhancement, v001 scenes only, closed :done, 2022-08-10, 2026-08-31
section CH4PLM v002
plume complexes, closed :done, 2022-08-10, 2025-09-22
section CO2PLM v002
plume complexes, closed :done, 2022-08-10, 2025-11-29
section L3 ASA v002
0.5° mineral abundance, one granule :done, 2022-08-10, 2024-01-20
The products open with different tools. L1B radiance, observation geometry and the L2A mask open with emit.EMITImage. L2A RFL (reflectance and its uncertainty) and L2B MIN (mineral identification and band depth) are netCDF files on the same sensor grid and GLT as the radiance and open with xarray. FRCOV (vegetation, non-photosynthetic vegetation and bare-soil fractions), CH4ENH / CO2ENH (enhancement in ppm m, with sensitivity and uncertainty) and the CH4PLM / CO2PLM plume rasters are GeoTIFFs and open with RasterioReader. Each plume also has a GeoJSON metadata file that opens with geopandas. LP DAAC also publishes per-orbit attitude files (L1B ATT v002) and Earth system model inputs (L4 ESM v001).
The next cell asks NASA's Common Metadata Repository (CMR) which of these products exist for the acquisition in this tutorial. The query only reads metadata and needs no credentials.
import requests
# Newest version of each scene-level collection (CMR concept ids, 2026-09).
SCENE_COLLECTIONS = {
"L1B RAD v002": "C4079829720-LPCLOUD",
"L2A RFL v002": "C4079844428-LPCLOUD",
"L2A MASK v003": "C4279547358-LPCLOUD",
"L2B MIN v001": "C2408034484-LPCLOUD",
"L2B FRCOV v001": "C3911089796-LPCLOUD",
"L2B CH4ENH v002": "C3242680113-LPCLOUD",
"L2B CO2ENH v002": "C3243477145-LPCLOUD",
"L2B CH4PLM v002": "C3242707413-LPCLOUD",
"L2B CO2PLM v002": "C3244277342-LPCLOUD",
}
def granules_of_acquisition(concept_id: str, pid: emit.EMITProductID) -> list:
t = pid.acquisition.strftime("%Y-%m-%dT%H:%M:%SZ")
response = requests.get("https://cmr.earthdata.nasa.gov/search/granules.umm_json",
params={"collection_concept_id": concept_id, "temporal": f"{t},{t}", "page_size": 50},
timeout=60)
response.raise_for_status()
# The temporal query also returns the neighbouring scene; keep granules named after this acquisition.
return [item["umm"] for item in response.json()["items"] if pid.dt in item["umm"]["GranuleUR"]]
pid = emit.parse_product_name(file_save)
tour = pd.DataFrame([
{"collection": name, "granule": granule["GranuleUR"], "file": url.rsplit("/", 1)[-1], "url": url}
for name, concept_id in SCENE_COLLECTIONS.items()
for granule in granules_of_acquisition(concept_id, pid)
for url in (u["URL"] for u in granule.get("RelatedUrls", []) if u.get("Type") == "GET DATA")
])
with pd.option_context("display.max_colwidth", None, "display.max_rows", None):
display(tour[["collection", "file"]])
Methane enhancement and plume complexes¶
For scenes acquired before 2026-09, JPL's L2B CH4ENH v002 product is available. emit.get_ch4enhancement_link builds its URL from the L1B name. The CH4PLM plume complexes found in this scene come with a GeoJSON metadata file each. The cell downloads the enhancement GeoTIFF (about 15 MB) and the plume metadata, and plots the plume outlines on top of the enhancement.
import geopandas as gpd
from shapely.geometry import box
from georeader import read
from georeader.rasterio_reader import RasterioReader
from georeader.plot import add_shape_to_plot
link_ch4 = emit.get_ch4enhancement_link(file_save)
file_ch4 = os.fspath(dir_emit_files / os.path.basename(link_ch4))
if not os.path.exists(file_ch4):
emit.download_product(link_ch4, file_ch4, display_progress_bar=False)
plume_files = []
for url in tour.loc[tour["file"].str.contains("CH4PLMMETA"), "url"]:
plume_file = os.fspath(dir_emit_files / os.path.basename(url))
if not os.path.exists(plume_file):
emit.download_product(url, plume_file, display_progress_bar=False)
plume_files.append(plume_file)
plumes = pd.concat([gpd.read_file(f) for f in plume_files], ignore_index=True)
plumes[["Plume ID", "Max Plume Concentration (ppm m)", "Latitude of max concentration", "Longitude of max concentration"]]
aoi = box(*plumes.total_bounds).buffer(0.05)
ch4 = read.read_from_polygon(RasterioReader(file_ch4), aoi, crs_polygon="EPSG:4326").load()
fig, ax = plt.subplots(figsize=(8, 5), dpi=72)
show(ch4, ax=ax, vmin=0, vmax=1500, cmap="plasma", add_colorbar_next_to=True,
mask=ch4.values == ch4.fill_value_default, title="EMIT L2B CH4ENH v002 (ppm m) and CH4PLM plume complexes")
add_shape_to_plot(plumes, ax=ax, crs_plot=ch4.crs, polygon_no_fill=True,
kwargs_geopandas_plot={"color": "cyan", "linewidth": 1.5})
The other products in the tour table can be downloaded the same way with emit.download_product(url, filename). The GeoTIFF products (FRCOV, CO2ENH, CH4PLM) open with RasterioReader. The netCDF products (RFL, MIN) are in the same sensor grid as the radiance and open with xarray.
Licence¶
The georeader package is published under a GNU Lesser GPL v3 licence
georeader tutorials and notebooks are released under a Creative Commons non-commercial licence.
If you find this work useful please cite:
@article{ruzicka_starcop_2023,
title = {Semantic segmentation of methane plumes with hyperspectral machine learning models},
volume = {13},
issn = {2045-2322},
url = {https://www.nature.com/articles/s41598-023-44918-6},
doi = {10.1038/s41598-023-44918-6},
number = {1},
journal = {Scientific Reports},
author = {Růžička, Vít and Mateo-Garcia, Gonzalo and Gómez-Chova, Luis and Vaughan, Anna, and Guanter, Luis and Markham, Andrew},
month = nov,
year = {2023},
pages = {19999},
}