Skip to content
Merged
Show file tree
Hide file tree
Changes from 3 commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,12 @@ For images without bounding box columns (e.g. older CellProfiler outputs or imag
- Crop from compartment-center coordinates plus pixel offsets with `display_options={"offset_bounding_box": {"x_min": -20, "y_min": -20, "x_max": 20, "y_max": 20}}` (requires compartment center columns such as `Nuclei_Location_Center_X/Y`).
- Render the full field of view without cropping with `display_options={"render_whole_image": True}` (works even with no bounding box and no center columns).

For image formats and inline image data:

- Image filename columns may point to `.tif`/`.tiff`, `.jpg`/`.jpeg`, `.png`, `.gif`, `.webp`, or `.jxl` ([JPEG XL](https://jpeg.org/jpegxl/)) files, and mask or outline files may use the same formats. Animated GIFs and WebPs display their first frame.
- Columns holding raw encoded image bytes (for example a DuckDB `BLOB` or a parquet `binary` column) render inline as images, in both the standard table and the widget table. JPEG, PNG, GIF, and WebP bytes are embedded as-is (animations stay animated), while JPEG XL bytes are decoded and shown as PNG since browsers can't reliably display JPEG XL.
- Inline image bytes are displayed as stored, without the brightness or contrast adjustments applied to image files. Their size follows the `width` and `height` display options.
Comment thread
d33bs marked this conversation as resolved.
Outdated

For row display in notebook/widget tables:

- CytoDataFrame respects pandas display settings (`display.max_rows`, `display.min_rows`).
Expand Down
4 changes: 4 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -147,6 +147,10 @@ ini_options.addopts = "--cov=src/cytodataframe --cov-report=term-missing:skip-co
# settings to avoid errors with cv2 and coverage
# see here for more: https://github.com/nedbat/coveragepy/issues/1653
run.omit = [
# imagecodecs' compiled extensions carry Cython line-trace data pointing at
# ``.pyx`` sources that aren't shipped, which breaks ``coverage xml`` once
# tests decode JPEG XL.
"*.pyx",
"config-3.py",
"config.py",
]
Expand Down
179 changes: 157 additions & 22 deletions src/cytodataframe/frame.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,10 +45,12 @@
from .image import (
add_image_scale_bar,
adjust_with_adaptive_histogram_equalization,
decode_jpegxl,
draw_outline_on_image_from_mask,
draw_outline_on_image_from_outline,
get_pixel_bbox_from_offsets,
image_array_to_grayscale,
read_image_file,
)
from .volume import (
build_3d_html_from_path,
Expand All @@ -61,6 +63,20 @@
logger = logging.getLogger(__name__)
MIN_VOLUME_NDIM = 3
RGB_LIKE_CHANNEL_COUNTS = (MIN_VOLUME_NDIM, 4)
# (byte offset, signature, MIME type) for encoded image formats. Used to
# recognize raw image bytes (e.g. a DuckDB ``BLOB``) held in a column. JPEG XL
# has a bare-codestream and an ISO BMFF container signature, and WebP is a
# RIFF container with ``WEBP`` at offset 8.
ENCODED_IMAGE_SIGNATURES = (
(0, b"\xff\xd8\xff", "image/jpeg"),
(0, b"\x89PNG\r\n\x1a\n", "image/png"),
(0, b"GIF87a", "image/gif"),
(0, b"GIF89a", "image/gif"),
(0, b"\xff\x0a", "image/jxl"),
(0, b"\x00\x00\x00\x0cJXL \r\n\x87\n", "image/jxl"),
(8, b"WEBP", "image/webp"),
)
JPEGXL_MIME_TYPE = "image/jxl"
MIN_RGB_SPATIAL_DIM = 8
MAX_RGB_ASPECT_RATIO = 4.0
MIN_POSITION_COMPONENTS = 2
Expand Down Expand Up @@ -1968,8 +1984,8 @@ def find_image_columns(self: CytoDataFrame_type) -> List[str]:
Find columns containing image file names.

This method searches for columns in the DataFrame
that contain image file names with extensions .tif
or .tiff (case insensitive).
that contain image file names with extensions .tif, .tiff,
.jpg, .jpeg, .png, .gif, .webp, or .jxl (case insensitive).

Performance note:
Single-cell profiles typically have thousands of numeric feature
Expand All @@ -1984,8 +2000,10 @@ def find_image_columns(self: CytoDataFrame_type) -> List[str]:
image file names.

"""
# Image file names end in ``.tif``/``.tiff`` (case insensitive).
compiled_pattern = re.compile(r".*\.(tif|tiff)$", flags=re.IGNORECASE)
# Image file names end in a supported extension (case insensitive).
compiled_pattern = re.compile(
r".*\.(tif|tiff|jpg|jpeg|png|gif|webp|jxl)$", flags=re.IGNORECASE
)

def _value_is_image_name(value: Any) -> bool:
return (
Expand Down Expand Up @@ -2420,7 +2438,7 @@ def _prepare_3d_label_overlay(

if mask_array is None:
try:
mask_array = np.asarray(imageio.imread(segmentation_path))
mask_array = np.asarray(read_image_file(segmentation_path))
except (FileNotFoundError, OSError, ValueError) as exc:
logger.debug(
"Unable to read mask/outline image %s: %s",
Expand Down Expand Up @@ -2759,7 +2777,7 @@ def _load_enhanced_image_for_display( # noqa: C901
return cached.copy()

try:
orig_image_array = imageio.imread(candidate_path)
orig_image_array = read_image_file(candidate_path)
except (FileNotFoundError, ValueError) as exc:
logger.error(exc)
return None
Expand Down Expand Up @@ -2974,7 +2992,7 @@ def _prepare_cropped_image_layers( # noqa: C901, PLR0915, PLR0912, PLR0913, PLR
mask_source_array = None
if include_mask_outline and mask_source_path is not None:
try:
loaded_mask = imageio.imread(mask_source_path)
loaded_mask = read_image_file(mask_source_path)
if loaded_mask.ndim == 3: # noqa: PLR2004
mask_gray = np.max(loaded_mask[..., :3], axis=2)
else:
Expand Down Expand Up @@ -3236,18 +3254,8 @@ def _resolve_image_min_width(width: Any) -> Optional[str]:
return None
return f"{IMAGE_MIN_DISPLAY_WIDTH_PX}px"

def _image_array_to_html(self: CytoDataFrame_type, image_array: np.ndarray) -> str:
"""Encode an image array as an HTML <img> tag."""

try:
png_bytes_io = BytesIO()
with warnings.catch_warnings():
warnings.simplefilter("ignore", UserWarning)
imageio.imwrite(png_bytes_io, image_array, format="png")
png_bytes = png_bytes_io.getvalue()
except (FileNotFoundError, ValueError) as exc:
logger.error(exc)
raise
def _image_html_style(self: CytoDataFrame_type) -> str:
"""Build the inline CSS shared by every rendered ``<img>`` cell."""

display_options = self._custom_attrs.get("display_options", {}) or {}
# Normalize bare numeric widths (e.g. 300 or "300") to CSS pixel strings
Expand All @@ -3266,12 +3274,116 @@ def _image_array_to_html(self: CytoDataFrame_type, image_array: np.ndarray) -> s
if height is not None:
html_style.append(f"height:{height}")

html_style_joined = ";".join(html_style)
return ";".join(html_style)

def _image_array_to_html(self: CytoDataFrame_type, image_array: np.ndarray) -> str:
"""Encode an image array as an HTML <img> tag."""

try:
png_bytes_io = BytesIO()
with warnings.catch_warnings():
warnings.simplefilter("ignore", UserWarning)
imageio.imwrite(png_bytes_io, image_array, format="png")
png_bytes = png_bytes_io.getvalue()
except (FileNotFoundError, ValueError) as exc:
logger.error(exc)
raise

base64_image_bytes = base64.b64encode(png_bytes).decode("utf-8")

return (
'<img src="data:image/png;base64,'
f'{base64_image_bytes}" style="{html_style_joined}"/>'
f'{base64_image_bytes}" style="{self._image_html_style()}"/>'
)

@staticmethod
def _encoded_image_mime_type(value: Any) -> Optional[str]:
"""Return the MIME type of a raw encoded-image byte string, else None."""

if not isinstance(value, (bytes, bytearray, memoryview)):
return None
head = bytes(value[:12])
return next(
(
mime_type
for offset, signature, mime_type in ENCODED_IMAGE_SIGNATURES
if head[offset : offset + len(signature)] == signature
),
None,
)

@staticmethod
def _dtype_can_hold_bytes(dtype: Any) -> bool:
"""Whether a column dtype can hold raw ``bytes`` values."""

if pd.api.types.is_object_dtype(dtype):
return True
if isinstance(dtype, pd.ArrowDtype):
import pyarrow as pa
Comment thread
d33bs marked this conversation as resolved.
Outdated

arrow_type = dtype.pyarrow_dtype
return (
pa.types.is_binary(arrow_type)
or pa.types.is_large_binary(arrow_type)
or pa.types.is_fixed_size_binary(arrow_type)
)
return False

@staticmethod
def find_encoded_image_bytes_columns(data: pd.DataFrame) -> List[str]:
"""
Identify columns that contain raw encoded image bytes (for example a
DuckDB ``BLOB`` or a parquet ``binary`` column of JPEG, PNG, GIF, WebP,
or JPEG XL images).

Only columns whose dtype can hold ``bytes`` are scanned: object dtype
(what pandas uses by default) and Arrow-backed binary dtypes.
"""

image_cols: List[str] = [
column
for column, dtype in data.dtypes.items()
if CytoDataFrame._dtype_can_hold_bytes(dtype)
and data[column].apply(CytoDataFrame._encoded_image_mime_type).notna().any()
]

if image_cols:
logger.debug("Found encoded image bytes columns: %s", image_cols)

return image_cols

def process_encoded_image_bytes_as_html_display(
self: CytoDataFrame_type,
data_value: Any,
) -> Any:
"""
Render raw encoded image bytes as an HTML <img> element.

JPEG, PNG, GIF, and WebP bytes are already browser-native images, so
they are embedded directly without decoding and re-encoding (which
also keeps animations animated). Browsers cannot reliably display JPEG
XL, so those bytes are decoded and re-encoded as PNG. Values which are
not encoded images (for example nulls), or JPEG XL data which cannot
be decoded, are returned unchanged.
"""

mime_type = self._encoded_image_mime_type(data_value)
if mime_type is None:
return data_value

if mime_type == JPEGXL_MIME_TYPE:
try:
return self._image_array_to_html(
self._ensure_uint8(decode_jpegxl(bytes(data_value)))
)
except ValueError as exc:
logger.warning("Unable to render JPEG XL bytes: %s", exc)
return data_value

encoded = base64.b64encode(bytes(data_value)).decode("ascii")
return (
f'<img src="data:{mime_type};base64,{encoded}" '
f'style="{self._image_html_style()}"/>'
)

def process_ome_arrow_data_as_html_display(
Expand Down Expand Up @@ -5502,7 +5614,18 @@ def _safe_text(value: Any) -> str:
continue

value = self.loc[row_label, col]
text_value = html_lib.escape(_safe_text(value))
# Raw encoded image bytes (e.g. a BLOB column) render as an
# image; the signature check keeps text cells escaped.
image_html = (
self.process_encoded_image_bytes_as_html_display(value)
if self._encoded_image_mime_type(value) is not None
else None
)
text_value = (
image_html
if isinstance(image_html, str)
else html_lib.escape(_safe_text(value))
)
grid[row_idx, col_idx] = widgets.HTML(
value=(
f"<div style='width:100%;height:100%;background:{row_bg};"
Expand Down Expand Up @@ -5944,6 +6067,18 @@ def _render_composite(row: Any) -> str:
display_indices, ome_col
].apply(self.process_ome_arrow_data_as_html_display)

# Only the rows actually being rendered need scanning.
for bytes_col in self.find_encoded_image_bytes_columns(
data.loc[display_indices]
):
if not pd.api.types.is_object_dtype(data[bytes_col].dtype):
# e.g. Arrow-backed binary: can't hold the HTML strings
# assigned below, so convert the display copy.
data[bytes_col] = data[bytes_col].astype(object)
data.loc[display_indices, bytes_col] = data.loc[
display_indices, bytes_col
].apply(self.process_encoded_image_bytes_as_html_display)

if self._custom_attrs["is_transposed"]:
# retranspose to return the
# data in the shape expected
Expand Down
45 changes: 43 additions & 2 deletions src/cytodataframe/image.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,11 @@
Helper functions for working with images in the context of CytoDataFrames.
"""

import pathlib
from typing import Any, Dict, Optional, Tuple

import cv2
import imagecodecs
import imageio.v2 as imageio
import numpy as np
import skimage
Expand All @@ -15,6 +17,45 @@
from skimage.util import img_as_ubyte


def decode_jpegxl(data: bytes) -> np.ndarray:
"""
Decode a JPEG XL byte string into an image array.

Raises:
ValueError: If JPEG XL decoding is unavailable or the data is invalid.
"""
if not imagecodecs.JPEGXL.available:
raise ValueError("JPEG XL decoding is unavailable in this environment.")
try:
return np.asarray(imagecodecs.jpegxl_decode(data))
except Exception as exc:
raise ValueError(f"Unable to decode JPEG XL data: {exc}") from exc


def read_image_file(path: "str | pathlib.Path") -> np.ndarray:
"""
Read an image file into an array.

``imageio`` has no JPEG XL backend, so ``.jxl`` files are decoded with
``imagecodecs`` (already a dependency); every other format goes through
``imageio`` as before. Gray+alpha images are returned as RGBA.
"""
suffix = pathlib.Path(path).suffix.lower()
array = (
decode_jpegxl(pathlib.Path(path).read_bytes())
if suffix == ".jxl"
else imageio.imread(path)
)
# A 2-channel (H, W, 2) array is gray+alpha, which downstream code would
# otherwise mistake for a 2-slice 3D volume. TIFFs are left alone since
Comment thread
d33bs marked this conversation as resolved.
Outdated
# for them a 3D array genuinely may be a volume.
if suffix not in (".tif", ".tiff") and array.ndim == 3 and array.shape[-1] == 2:
array = np.concatenate(
[np.repeat(array[..., :1], 3, axis=-1), array[..., 1:]], axis=-1
)
return array


def image_array_to_grayscale(img_array: np.ndarray) -> np.ndarray:
"""
Convert an image array to a 2D grayscale array.
Expand Down Expand Up @@ -147,7 +188,7 @@ def draw_outline_on_image_from_outline(
"""

# Load the outline image
outline_image = imageio.imread(outline_image_path)
outline_image = read_image_file(outline_image_path)

# Resize if necessary
if outline_image.shape[:2] != orig_image.shape[:2]:
Expand Down Expand Up @@ -209,7 +250,7 @@ def draw_outline_on_image_from_mask(
The resulting image with the green outline applied.
"""
# Load the binary mask image
mask_image = imageio.imread(mask_image_path)
mask_image = read_image_file(mask_image_path)

# Ensure the original image is RGB
# Grayscale input
Expand Down
Loading
Loading