diff --git a/src/eodash_catalog/endpoints.py b/src/eodash_catalog/endpoints.py index 070ec096..3dc9ac53 100644 --- a/src/eodash_catalog/endpoints.py +++ b/src/eodash_catalog/endpoints.py @@ -14,6 +14,7 @@ import pyarrow.parquet as pq import requests from pystac import Asset, Catalog, Collection, Item, Link, SpatialExtent, Summaries, TemporalExtent +from pystac.extensions.file import FileExtension from pystac_client import Client from shapely import wkt from shapely.geometry import mapping @@ -1586,15 +1587,15 @@ def handle_raw_source( extents = extract_extent_from_geoparquet(table) collection.extent.temporal = extents[0] collection.extent.spatial = extents[1] - collection.add_asset( - "geoparquet", - Asset( - href=parquet_source, - media_type="application/vnd.apache.parquet", - title="GeoParquet Items", - roles=["collection-mirror"], - ), + parquet_asset = Asset( + href=parquet_source, + media_type="application/vnd.apache.parquet", + title="GeoParquet Items", + roles=["collection-mirror"], ) + collection.add_asset("geoparquet", parquet_asset) + # adding this size explicitly here because github pages gzips the file + FileExtension.ext(parquet_asset, add_if_missing=True).size = len(parquet_file.content) else: LOGGER.warn(f"NO datetimes configured for collection: {collection_config['Name']}!") diff --git a/src/eodash_catalog/utils.py b/src/eodash_catalog/utils.py index dc78078b..caa04847 100644 --- a/src/eodash_catalog/utils.py +++ b/src/eodash_catalog/utils.py @@ -19,6 +19,7 @@ from owslib.wms import WebMapService from owslib.wmts import WebMapTileService from pystac import Asset, Catalog, Collection, Item, RelType, SpatialExtent, TemporalExtent +from pystac.extensions.file import FileExtension from pytz import timezone as pytztimezone from shapely import geometry as sgeom from shapely import wkb @@ -634,20 +635,21 @@ def save_items( table = record_batch_reader.read_all() output_path = f"{buildcatpath}/{colpath}" os.makedirs(output_path, exist_ok=True) - stacgp.arrow.to_parquet(table, f"{output_path}/items.parquet") + parquet_path = f"{output_path}/items.parquet" + stacgp.arrow.to_parquet(table, parquet_path) extents = extract_extent_from_geoparquet(table) collection.extent.temporal = extents[0] collection.extent.spatial = extents[1] # Make sure to also reference the geoparquet as asset - collection.add_asset( - "geoparquet", - Asset( - href="./items.parquet", - media_type="application/vnd.apache.parquet", - title="GeoParquet Items", - roles=["collection-mirror"], - ), + parquet_asset = Asset( + href="./items.parquet", + media_type="application/vnd.apache.parquet", + title="GeoParquet Items", + roles=["collection-mirror"], ) + collection.add_asset("geoparquet", parquet_asset) + # adding this size explicitly here because github pages gzips the file + FileExtension.ext(parquet_asset, add_if_missing=True).size = os.path.getsize(parquet_path) else: # go over items and add them to the collection LOGGER.info( diff --git a/tests/test_geoparquet.py b/tests/test_geoparquet.py index b5f9e999..9ef5d69a 100644 --- a/tests/test_geoparquet.py +++ b/tests/test_geoparquet.py @@ -51,6 +51,12 @@ def test_geoparquet_geojson_items(catalog_output_folder): assert parquet_asset["type"] == "application/vnd.apache.parquet" items_path = os.path.join(child_collection_path, parquet_asset["href"].split("/")[-1]) assert os.path.exists(items_path) + # size is advertised so the client does not have to probe the mirror over HTTP + assert ( + "https://stac-extensions.github.io/file/v2.1.0/schema.json" + in collection_json["stac_extensions"] + ) + assert parquet_asset["file:size"] == os.path.getsize(items_path) with open(items_path, "rb") as fp: table = pa.parquet.read_table(fp)