| |
| """Build solidified accessibility meshes and turntable GIFs from completed depth.""" |
|
|
| from __future__ import annotations |
|
|
| import argparse |
| import hashlib |
| import html |
| import json |
| import math |
| import os |
| import sys |
| from collections import Counter, defaultdict |
| from pathlib import Path |
| from typing import Any |
|
|
| import cv2 |
| import numpy as np |
| from PIL import Image, ImageDraw, ImageOps |
|
|
| PROJECT_ROOT = Path(__file__).resolve().parents[1] |
| sys.path.insert(0, str(PROJECT_ROOT)) |
|
|
| from accessibilityamodal.category_overrides import ( |
| load_reviewed_category_overrides, |
| resolve_geometry_category_for_modeling, |
| ) |
| from accessibilityamodal.reconstruct import camera_intrinsics, pixels_to_points |
|
|
|
|
| CONTINUOUS_SURFACE_CATEGORIES = { |
| "walkway", |
| "ramp", |
| "curb_cut", |
| "raised_curb", |
| "tactile_paving", |
| } |
| RAISED_CURB_CATEGORIES = {"raised_curb"} |
| _FILE_SHA256_CACHE: dict[tuple[str, int, int], str] = {} |
|
|
|
|
| def read_json(path: Path) -> dict[str, Any]: |
| return json.loads(path.read_text(encoding="utf-8")) |
|
|
|
|
| def read_jsonl(path: Path) -> list[dict[str, Any]]: |
| text = path.read_text(encoding="utf-8") |
| stripped = text.lstrip() |
| if stripped.startswith("["): |
| data = json.loads(text) |
| if not isinstance(data, list): |
| raise ValueError(f"Expected a JSON list or JSONL rows in {path}") |
| return [row for row in data if isinstance(row, dict)] |
| rows = [] |
| for line in text.splitlines(): |
| if line.strip(): |
| rows.append(json.loads(line)) |
| return rows |
|
|
|
|
| def sha256_file(path: Path) -> str: |
| stat = path.stat() |
| key = (str(path.resolve()), int(stat.st_size), int(stat.st_mtime_ns)) |
| cached = _FILE_SHA256_CACHE.get(key) |
| if cached is not None: |
| return cached |
| digest = hashlib.sha256() |
| with path.open("rb") as handle: |
| for chunk in iter(lambda: handle.read(1024 * 1024), b""): |
| digest.update(chunk) |
| value = digest.hexdigest() |
| _FILE_SHA256_CACHE[key] = value |
| return value |
|
|
|
|
| def read_mask(path: Path, shape: tuple[int, int]) -> np.ndarray: |
| mask = np.asarray(ImageOps.exif_transpose(Image.open(path)).convert("L")) |
| if mask.shape[:2] != shape: |
| raise ValueError( |
| f"Mask/geometry raster mismatch for {path}: mask={mask.shape[:2]}, geometry={shape}. " |
| "Refusing to resize a derived mask because it can hide an alignment error." |
| ) |
| return mask > 127 |
|
|
|
|
| def read_mask_nearest(path: Path, shape: tuple[int, int]) -> np.ndarray: |
| """Read an authoritative source mask with explicit nearest-neighbor mapping.""" |
|
|
| mask = np.asarray(ImageOps.exif_transpose(Image.open(path)).convert("L")) |
| if mask.shape[:2] != shape: |
| if not aspect_ratio_matches(mask.shape[:2], shape): |
| raise ValueError( |
| f"Source mask/geometry aspect mismatch for {path}: " |
| f"mask={mask.shape[:2]}, geometry={shape}." |
| ) |
| h, w = shape |
| mask = np.asarray( |
| Image.fromarray(mask, mode="L").resize( |
| (w, h), |
| Image.Resampling.NEAREST, |
| ) |
| ) |
| return mask > 127 |
|
|
|
|
| def aspect_ratio_matches(source_shape: tuple[int, int], target_shape: tuple[int, int]) -> bool: |
| source_h, source_w = source_shape |
| target_h, target_w = target_shape |
| return abs(source_w * target_h - target_w * source_h) <= max(source_w, target_w) |
|
|
|
|
| def largest_component(mask: np.ndarray) -> np.ndarray: |
| count, labels, stats, _ = cv2.connectedComponentsWithStats(mask.astype(np.uint8), connectivity=8) |
| if count <= 1: |
| return mask |
| largest = 1 + int(np.argmax(stats[1:, cv2.CC_STAT_AREA])) |
| return labels == largest |
|
|
|
|
| def clean_target_mask(mask: np.ndarray, close_kernel: int, keep_largest: bool) -> np.ndarray: |
| cleaned = mask.astype(np.uint8) |
| if close_kernel > 1: |
| kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (close_kernel, close_kernel)) |
| cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, kernel) |
| cleaned_bool = cleaned > 0 |
| if keep_largest and np.any(cleaned_bool): |
| cleaned_bool = largest_component(cleaned_bool) |
| return cleaned_bool |
|
|
|
|
| def fill_enclosed_mask_holes(mask: np.ndarray) -> np.ndarray: |
| """Fill background components that cannot reach the image boundary. |
| |
| This is used only to make a render surface topologically continuous. It |
| never replaces or writes the reviewed amodal annotation. |
| """ |
|
|
| source = mask.astype(bool) |
| if not np.any(source): |
| return source.copy() |
| inverse = (~source).astype(np.uint8) |
| component_count, labels = cv2.connectedComponents(inverse, connectivity=8) |
| if component_count <= 1: |
| return source.copy() |
| border_labels = np.unique( |
| np.concatenate( |
| (labels[0], labels[-1], labels[:, 0], labels[:, -1]) |
| ) |
| ) |
| enclosed = inverse.astype(bool) & ~np.isin(labels, border_labels) |
| return source | enclosed |
|
|
|
|
| def mask_component_count(mask: np.ndarray, *, connectivity: int = 8) -> int: |
| count, _ = cv2.connectedComponents( |
| mask.astype(np.uint8), |
| connectivity=connectivity, |
| ) |
| return max(int(count) - 1, 0) |
|
|
|
|
| def largest_axis_aligned_rectangle(mask: np.ndarray) -> tuple[int, int, int, int]: |
| """Return the largest all-true rectangle as inclusive x/y bounds.""" |
|
|
| source = mask.astype(bool) |
| if source.ndim != 2 or not np.any(source): |
| raise ValueError("Largest rectangle requires a non-empty 2D mask") |
| height, width = source.shape |
| histogram = np.zeros(width, dtype=np.int32) |
| best_area = 0 |
| best = (0, 0, 0, 0) |
| for y in range(height): |
| histogram = np.where(source[y], histogram + 1, 0) |
| stack: list[int] = [] |
| for x in range(width + 1): |
| current_height = int(histogram[x]) if x < width else 0 |
| while stack and int(histogram[stack[-1]]) > current_height: |
| column = stack.pop() |
| rectangle_height = int(histogram[column]) |
| left = stack[-1] + 1 if stack else 0 |
| rectangle_width = x - left |
| area = rectangle_height * rectangle_width |
| if area > best_area: |
| best_area = area |
| best = ( |
| left, |
| x - 1, |
| y - rectangle_height + 1, |
| y, |
| ) |
| stack.append(x) |
| if best_area <= 0: |
| raise ValueError("Largest rectangle search produced no supported cells") |
| return best |
|
|
|
|
| def build_continuous_surface_support_mask( |
| target_mask: np.ndarray, |
| *, |
| stride: int, |
| close_kernel: int = 0, |
| visible_mask: np.ndarray | None = None, |
| hidden_mask: np.ndarray | None = None, |
| obstacle_mask: np.ndarray | None = None, |
| automatic_close_added_ratio: float = 0.05, |
| maximum_close_added_ratio: float = 0.10, |
| maximum_large_hole_ratio: float = 0.05, |
| obstacle_hole_overlap_ratio: float = 0.25, |
| maximum_bridge_added_ratio: float = 0.005, |
| maximum_opening_removed_ratio: float = 0.03, |
| minimum_opening_hidden_retention: float = 0.95, |
| ) -> tuple[np.ndarray, dict[str, Any]]: |
| """Create a conservative, render-only support mask for a road surface. |
| |
| Sub-grid breaks are closed, fully enclosed holes are filled, and only very |
| short nearby components may be bridged inside a tightly bounded evidence |
| envelope. The operation does not use a row-span/convex-hull expansion and |
| therefore does not turn unrelated background into target annotation. |
| """ |
|
|
| if stride < 1: |
| raise ValueError("continuous surface stride must be positive") |
| source = target_mask.astype(bool) |
| for name, auxiliary in ( |
| ("visible_mask", visible_mask), |
| ("hidden_mask", hidden_mask), |
| ("obstacle_mask", obstacle_mask), |
| ): |
| if auxiliary is not None and auxiliary.shape != source.shape: |
| raise ValueError(f"{name} must match the continuous target raster") |
| authoritative = source.copy() |
| if visible_mask is not None: |
| authoritative |= visible_mask.astype(bool) |
| if hidden_mask is not None: |
| authoritative |= hidden_mask.astype(bool) |
| if not np.any(authoritative): |
| raise ValueError("No target pixels for continuous surface support") |
| kernel_size = int(close_kernel) if close_kernel > 0 else 4 * int(stride) + 1 |
| kernel_size = max(kernel_size, 3) |
| if kernel_size % 2 == 0: |
| kernel_size += 1 |
| kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (kernel_size, kernel_size), |
| ) |
| obstacle_envelope = None |
| if obstacle_mask is not None and np.any(obstacle_mask): |
| full_obstacle = fill_enclosed_mask_holes(obstacle_mask.astype(bool)) |
| obstacle_kernel_size = 4 * int(stride) + 1 |
| obstacle_kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (obstacle_kernel_size, obstacle_kernel_size), |
| ) |
| obstacle_envelope = cv2.dilate( |
| full_obstacle.astype(np.uint8), |
| obstacle_kernel, |
| ).astype(bool) |
|
|
| closed_candidate = cv2.morphologyEx( |
| authoritative.astype(np.uint8), |
| cv2.MORPH_CLOSE, |
| kernel, |
| ).astype(bool) |
| close_added_candidate = closed_candidate & ~authoritative |
| automatic_close_added_limit = max( |
| 1, |
| int(math.floor(float(authoritative.sum()) * automatic_close_added_ratio)), |
| ) |
| close_added_limit = max( |
| 1, |
| int(math.floor(float(authoritative.sum()) * maximum_close_added_ratio)), |
| ) |
| close_added_obstacle_overlap = ( |
| float(np.count_nonzero(close_added_candidate & obstacle_envelope)) |
| / max(int(close_added_candidate.sum()), 1) |
| if obstacle_envelope is not None |
| else 0.0 |
| ) |
| close_gate_passed = ( |
| int(close_added_candidate.sum()) <= automatic_close_added_limit |
| or ( |
| int(close_added_candidate.sum()) <= close_added_limit |
| and close_added_obstacle_overlap >= obstacle_hole_overlap_ratio |
| ) |
| ) |
| closed = closed_candidate if close_gate_passed else authoritative.copy() |
|
|
| inverse = (~closed).astype(np.uint8) |
| component_count, labels, stats, _ = cv2.connectedComponentsWithStats( |
| inverse, |
| connectivity=8, |
| ) |
| border_labels = set( |
| int(value) |
| for value in np.unique( |
| np.concatenate((labels[0], labels[-1], labels[:, 0], labels[:, -1])) |
| ) |
| ) |
| source_pixels = int(authoritative.sum()) |
| small_hole_limit = max( |
| 4 * int(stride) * int(stride), |
| int(math.floor(0.01 * source_pixels)), |
| ) |
| large_hole_limit = max( |
| small_hole_limit, |
| int(math.floor(maximum_large_hole_ratio * source_pixels)), |
| ) |
|
|
| supported = closed.copy() |
| small_hole_pixels = 0 |
| obstacle_gated_hole_pixels = 0 |
| large_render_only_hole_pixels = 0 |
| for label in range(1, component_count): |
| if label in border_labels: |
| continue |
| area = int(stats[label, cv2.CC_STAT_AREA]) |
| hole = labels == label |
| if area <= small_hole_limit: |
| supported[hole] = True |
| small_hole_pixels += area |
| continue |
| overlap_ratio = ( |
| float(np.count_nonzero(hole & obstacle_envelope)) / max(area, 1) |
| if obstacle_envelope is not None |
| else 0.0 |
| ) |
| supported[hole] = True |
| if area <= large_hole_limit and overlap_ratio >= obstacle_hole_overlap_ratio: |
| obstacle_gated_hole_pixels += area |
| else: |
| large_render_only_hole_pixels += area |
| enclosed_hole_added_pixels = int((supported & ~closed).sum()) |
|
|
| |
| |
| |
| |
| bridge_added_limit = max( |
| 1, |
| int(math.floor(maximum_bridge_added_ratio * source_pixels)), |
| ) |
| bridge_added_pixels = 0 |
| bridge_count = 0 |
| bridge_rejected_count = 0 |
| bridge_distances_pixels: list[float] = [] |
| bridge_maximum_distance = 3.0 * float(stride) * math.sqrt(2.0) |
| bridge_width = 2 * int(stride) + 1 |
| component_count_before_bridge = mask_component_count(supported) |
| if component_count_before_bridge > 1: |
| count, labels, stats, _ = cv2.connectedComponentsWithStats( |
| supported.astype(np.uint8), |
| connectivity=8, |
| ) |
| primary_label = 1 + int(np.argmax(stats[1:, cv2.CC_STAT_AREA])) |
| anchor = labels == primary_label |
| bridge_allowed_kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (6 * int(stride) + 1, 6 * int(stride) + 1), |
| ) |
| bridge_allowed = cv2.dilate( |
| authoritative.astype(np.uint8), |
| bridge_allowed_kernel, |
| ).astype(bool) |
| if obstacle_envelope is not None: |
| bridge_allowed |= obstacle_envelope |
| minimum_component_pixels = max( |
| 4 * int(stride) * int(stride), |
| int(math.ceil(0.001 * source_pixels)), |
| ) |
| component_labels = sorted( |
| range(1, count), |
| key=lambda label: int(stats[label, cv2.CC_STAT_AREA]), |
| reverse=True, |
| ) |
| for label in component_labels: |
| if label == primary_label: |
| continue |
| component = labels == label |
| carries_hidden = bool( |
| hidden_mask is not None |
| and np.any(component & hidden_mask.astype(bool)) |
| ) |
| if ( |
| int(stats[label, cv2.CC_STAT_AREA]) < minimum_component_pixels |
| and not carries_hidden |
| ): |
| continue |
| distance, nearest_label_map = cv2.distanceTransformWithLabels( |
| (~anchor).astype(np.uint8), |
| cv2.DIST_L2, |
| 5, |
| labelType=cv2.DIST_LABEL_PIXEL, |
| ) |
| component_y, component_x = np.nonzero(component) |
| nearest_index = int(np.argmin(distance[component_y, component_x])) |
| source_y = int(component_y[nearest_index]) |
| source_x = int(component_x[nearest_index]) |
| distance_pixels = float(distance[source_y, source_x]) |
| nearest_label = int(nearest_label_map[source_y, source_x]) |
| anchor_y, anchor_x = np.nonzero(anchor) |
| anchor_labels = nearest_label_map[anchor_y, anchor_x] |
| matches = np.flatnonzero(anchor_labels == nearest_label) |
| if matches.size == 0: |
| bridge_rejected_count += 1 |
| continue |
| target_index = int(matches[0]) |
| target_y = int(anchor_y[target_index]) |
| target_x = int(anchor_x[target_index]) |
| candidate = np.zeros_like(supported, dtype=np.uint8) |
| cv2.line( |
| candidate, |
| (source_x, source_y), |
| (target_x, target_y), |
| color=1, |
| thickness=bridge_width, |
| lineType=cv2.LINE_8, |
| ) |
| candidate = candidate.astype(bool) |
| added_candidate = candidate & ~supported |
| evidence_fraction = float( |
| np.count_nonzero(candidate & bridge_allowed) |
| ) / max(int(candidate.sum()), 1) |
| if ( |
| distance_pixels > bridge_maximum_distance |
| or evidence_fraction < 0.90 |
| or bridge_added_pixels + int(added_candidate.sum()) |
| > bridge_added_limit |
| ): |
| bridge_rejected_count += 1 |
| continue |
| supported |= candidate |
| anchor |= component | candidate |
| added_pixels = int(added_candidate.sum()) |
| bridge_added_pixels += added_pixels |
| bridge_count += 1 |
| bridge_distances_pixels.append(distance_pixels) |
|
|
| |
| before_post_bridge_hole_fill = supported.copy() |
| supported = fill_enclosed_mask_holes(supported) |
| post_bridge_hole_added_pixels = int( |
| np.count_nonzero(supported & ~before_post_bridge_hole_fill) |
| ) |
|
|
| |
| |
| |
| |
| opening_candidate = cv2.morphologyEx( |
| supported.astype(np.uint8), |
| cv2.MORPH_OPEN, |
| kernel, |
| ).astype(bool) |
| opening_removed = supported & ~opening_candidate |
| opening_removed_ratio = float( |
| np.count_nonzero(opening_removed & authoritative) |
| ) / max(source_pixels, 1) |
| opening_hidden_retention = ( |
| float(np.count_nonzero(opening_candidate & hidden_mask.astype(bool))) |
| / max(int(hidden_mask.sum()), 1) |
| if hidden_mask is not None and np.any(hidden_mask) |
| else 1.0 |
| ) |
| opening_gate_passed = bool( |
| opening_removed_ratio <= maximum_opening_removed_ratio |
| and opening_hidden_retention >= minimum_opening_hidden_retention |
| and np.any(opening_candidate) |
| ) |
| if opening_gate_passed: |
| |
| |
| |
| |
| supported = fill_enclosed_mask_holes( |
| opening_candidate | authoritative |
| ) |
|
|
| added = supported & ~authoritative |
| hidden_pixels = int(hidden_mask.sum()) if hidden_mask is not None else 0 |
| retained_hidden = ( |
| int(np.count_nonzero(supported & hidden_mask.astype(bool))) |
| if hidden_mask is not None |
| else 0 |
| ) |
| return supported, { |
| "role": "render_only_geometry_support_not_annotation", |
| "policy": ( |
| "authoritative_components_plus_bounded_close_all_enclosed_holes_" |
| "and_short_evidence_bridges" |
| ), |
| "row_span_or_convex_hull_used": False, |
| "generation_mask_used": False, |
| "annotation_mask_modified": False, |
| "stride": int(stride), |
| "close_kernel": kernel_size, |
| "source_pixels": int(source.sum()), |
| "authoritative_pixels": source_pixels, |
| "visible_pixels": int(visible_mask.sum()) if visible_mask is not None else None, |
| "hidden_pixels": hidden_pixels if hidden_mask is not None else None, |
| "hidden_retained_pixels": retained_hidden if hidden_mask is not None else None, |
| "hidden_retention_fraction": ( |
| float(retained_hidden / hidden_pixels) if hidden_pixels > 0 else None |
| ), |
| "close_added_candidate_pixels": int(close_added_candidate.sum()), |
| "automatic_close_added_limit_pixels": automatic_close_added_limit, |
| "close_added_limit_pixels": close_added_limit, |
| "close_added_obstacle_overlap_fraction": close_added_obstacle_overlap, |
| "close_gate_passed": close_gate_passed, |
| "close_added_pixels": int((closed & ~authoritative).sum()), |
| "small_hole_limit_pixels": small_hole_limit, |
| "large_hole_limit_pixels": large_hole_limit, |
| "small_hole_added_pixels": small_hole_pixels, |
| "obstacle_gated_hole_added_pixels": obstacle_gated_hole_pixels, |
| "large_render_only_hole_added_pixels": large_render_only_hole_pixels, |
| "rejected_large_hole_pixels": 0, |
| "all_enclosed_holes_filled": True, |
| "obstacle_hole_overlap_ratio": float(obstacle_hole_overlap_ratio), |
| "full_obstacle_used_for_hole_gate": obstacle_envelope is not None, |
| "enclosed_hole_added_pixels": enclosed_hole_added_pixels, |
| "post_bridge_hole_added_pixels": post_bridge_hole_added_pixels, |
| "support_pixels": int(supported.sum()), |
| "support_added_pixels": int(added.sum()), |
| "source_components": mask_component_count(authoritative), |
| "support_components": mask_component_count(supported), |
| "components_before_short_bridge": component_count_before_bridge, |
| "short_bridge_count": bridge_count, |
| "short_bridge_rejected_count": bridge_rejected_count, |
| "short_bridge_added_pixels": bridge_added_pixels, |
| "short_bridge_added_limit_pixels": bridge_added_limit, |
| "short_bridge_maximum_distance_pixels": bridge_maximum_distance, |
| "short_bridge_width_pixels": bridge_width, |
| "short_bridge_distances_pixels": bridge_distances_pixels, |
| "footprint_opening_kernel": kernel_size, |
| "footprint_opening_candidate_removed_pixels": int( |
| opening_removed.sum() |
| ), |
| "footprint_opening_removed_authoritative_pixels": int( |
| np.count_nonzero(opening_removed & authoritative) |
| ), |
| "footprint_opening_removed_authoritative_ratio": opening_removed_ratio, |
| "footprint_opening_hidden_retention_fraction": opening_hidden_retention, |
| "footprint_opening_gate_passed": opening_gate_passed, |
| "footprint_opening_policy": ( |
| "thin_hair_removal_only_without_hull_row_span_or_trapezoid" |
| ), |
| "detached_hidden_bridge_policy": ( |
| "only_short_bounded_bridges_near_reviewed_surface_or_obstacle_evidence" |
| ), |
| } |
|
|
|
|
| def complete_continuous_surface_texture( |
| colors: np.ndarray, |
| geometry_support: np.ndarray, |
| obstacle_mask: np.ndarray | None, |
| *, |
| stride: int, |
| ) -> tuple[np.ndarray, dict[str, Any]]: |
| """Replace obstacle appearance with sharp same-surface exemplar texels. |
| |
| This is a 3D-render-only texture adapter. It never writes back to the |
| frozen 2D completion. Every changed byte is copied from a non-obstacle |
| texel on the same connected surface; no diffusion, mean fill, Telea, or |
| semantic color tint is used, which avoids the foggy appearance of the old |
| renderer while removing poles/people/cars painted flat onto the road. |
| """ |
|
|
| if colors.ndim != 3 or colors.shape[2] != 3: |
| raise ValueError("Continuous surface texture must be an HxWx3 RGB image") |
| if geometry_support.shape != colors.shape[:2]: |
| raise ValueError("Geometry support and surface texture must share a raster") |
| if stride < 1: |
| raise ValueError("continuous surface texture stride must be positive") |
| if obstacle_mask is None or not np.any(obstacle_mask): |
| return colors.copy(), { |
| "role": "render_only_3d_texture_not_2d_completion", |
| "policy": "no_obstacle_texture_repair_needed", |
| "obstacle_texture_pixels": 0, |
| "repaired_pixels": 0, |
| "source_file_modified": False, |
| "diffusion_or_mean_fill_used": False, |
| } |
| if obstacle_mask.shape != colors.shape[:2]: |
| raise ValueError("Obstacle mask and surface texture must share a raster") |
|
|
| support = geometry_support.astype(bool) |
| obstacle = obstacle_mask.astype(bool) |
| repair = support & obstacle |
| if not np.any(repair): |
| return colors.copy(), { |
| "role": "render_only_3d_texture_not_2d_completion", |
| "policy": "obstacle_does_not_overlap_surface", |
| "obstacle_texture_pixels": 0, |
| "repaired_pixels": 0, |
| "source_file_modified": False, |
| "diffusion_or_mean_fill_used": False, |
| } |
|
|
| kernel_size = 2 * int(stride) + 1 |
| kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (kernel_size, kernel_size), |
| ) |
| expanded_obstacle = cv2.dilate( |
| obstacle.astype(np.uint8), |
| kernel, |
| ).astype(bool) |
| output = colors.copy() |
| geometry_component_count, geometry_component_labels = cv2.connectedComponents( |
| support.astype(np.uint8), |
| connectivity=8, |
| ) |
| repair_component_count, repair_component_labels, repair_stats, _ = ( |
| cv2.connectedComponentsWithStats( |
| repair.astype(np.uint8), |
| connectivity=8, |
| ) |
| ) |
| repaired_pixels = 0 |
| skipped_pixels = 0 |
| repaired_components = 0 |
| skipped_components = 0 |
| mirrored_pixels = 0 |
| translated_pixels = 0 |
| translated_components = 0 |
| translation_shifts: list[dict[str, Any]] = [] |
| donor_distances: list[np.ndarray] = [] |
| donor_linear_indices: list[np.ndarray] = [] |
| h, w = support.shape |
|
|
| for repair_label in range(1, repair_component_count): |
| component_repair = repair_component_labels == repair_label |
| repair_y, repair_x = np.nonzero(component_repair) |
| if repair_x.size == 0: |
| continue |
| geometry_values = geometry_component_labels[component_repair] |
| geometry_values = geometry_values[geometry_values > 0] |
| if geometry_values.size == 0: |
| skipped_components += 1 |
| skipped_pixels += int(repair_x.size) |
| continue |
| geometry_label = int( |
| np.argmax(np.bincount(geometry_values.astype(np.int64))) |
| ) |
| component = geometry_component_labels == geometry_label |
| donor = component & ~expanded_obstacle |
| if int(donor.sum()) < max(16, int(0.01 * component.sum())): |
| donor = component & ~obstacle |
| if not np.any(donor): |
| skipped_components += 1 |
| skipped_pixels += int(repair_x.size) |
| continue |
|
|
| original_repair_y = repair_y.copy() |
| original_repair_x = repair_x.copy() |
| filled = np.zeros(repair_x.shape[0], dtype=bool) |
| sample_y_all = np.zeros(repair_x.shape[0], dtype=np.int32) |
| sample_x_all = np.zeros(repair_x.shape[0], dtype=np.int32) |
|
|
| x0 = int(repair_stats[repair_label, cv2.CC_STAT_LEFT]) |
| y0 = int(repair_stats[repair_label, cv2.CC_STAT_TOP]) |
| box_width = int(repair_stats[repair_label, cv2.CC_STAT_WIDTH]) |
| box_height = int(repair_stats[repair_label, cv2.CC_STAT_HEIGHT]) |
| template = component_repair[ |
| y0 : y0 + box_height, |
| x0 : x0 + box_width, |
| ].astype(np.float32) |
| translation_overlap = 0.0 |
| chosen_dx = 0 |
| chosen_dy = 0 |
| if ( |
| box_height <= h |
| and box_width <= w |
| and template.size > 0 |
| and np.any(template) |
| ): |
| response = cv2.matchTemplate( |
| donor.astype(np.float32), |
| template, |
| cv2.TM_CCORR, |
| ) |
| response = response / max(float(template.sum()), 1.0) |
| candidate_count = min(128, int(response.size)) |
| flat = response.reshape(-1) |
| if candidate_count > 0: |
| candidate_indices = np.argpartition( |
| flat, |
| -candidate_count, |
| )[-candidate_count:] |
| maximum_overlap = float(flat[candidate_indices].max()) |
| minimum_candidate_overlap = max(0.70, maximum_overlap - 0.02) |
| ring = ( |
| cv2.dilate(component_repair.astype(np.uint8), kernel).astype(bool) |
| & donor |
| ) |
| ring_values = colors[ring].astype(np.float32) |
| ring_mean = ( |
| ring_values.mean(axis=0) |
| if ring_values.size |
| else None |
| ) |
| best_score = float("inf") |
| best: tuple[int, int, float] | None = None |
| response_width = response.shape[1] |
| for flat_index in candidate_indices: |
| overlap = float(flat[int(flat_index)]) |
| if overlap < minimum_candidate_overlap: |
| continue |
| source_top = int(flat_index) // response_width |
| source_left = int(flat_index) % response_width |
| dy = source_top - y0 |
| dx = source_left - x0 |
| candidate_y = original_repair_y + dy |
| candidate_x = original_repair_x + dx |
| candidate_valid = donor[candidate_y, candidate_x] |
| if not np.any(candidate_valid): |
| continue |
| source_values = colors[ |
| candidate_y[candidate_valid], |
| candidate_x[candidate_valid], |
| ].astype(np.float32) |
| color_cost = ( |
| float(np.mean(np.abs(source_values.mean(axis=0) - ring_mean))) |
| / 255.0 |
| if ring_mean is not None |
| else 0.0 |
| ) |
| shift_cost = ( |
| abs(float(dx)) / max(float(w), 1.0) |
| + 3.0 * abs(float(dy)) / max(float(h), 1.0) |
| ) |
| score = 20.0 * (1.0 - overlap) + shift_cost + color_cost |
| if score < best_score: |
| best_score = score |
| best = (dx, dy, overlap) |
| if best is not None: |
| chosen_dx, chosen_dy, translation_overlap = best |
| translated_y = original_repair_y + chosen_dy |
| translated_x = original_repair_x + chosen_dx |
| translated_valid = donor[translated_y, translated_x] |
| output[ |
| original_repair_y[translated_valid], |
| original_repair_x[translated_valid], |
| ] = colors[ |
| translated_y[translated_valid], |
| translated_x[translated_valid], |
| ] |
| filled[translated_valid] = True |
| sample_y_all[translated_valid] = translated_y[translated_valid] |
| sample_x_all[translated_valid] = translated_x[translated_valid] |
| translated_count = int(np.count_nonzero(translated_valid)) |
| translated_pixels += translated_count |
| if translated_count > 0: |
| translated_components += 1 |
| translation_shifts.append( |
| { |
| "component": int(repair_label), |
| "dx": int(chosen_dx), |
| "dy": int(chosen_dy), |
| "support_overlap_fraction": translation_overlap, |
| "translated_pixels": translated_count, |
| "component_pixels": int(original_repair_x.size), |
| } |
| ) |
|
|
| remaining = ~filled |
| repair_y = original_repair_y[remaining] |
| repair_x = original_repair_x[remaining] |
| if repair_x.size == 0: |
| repaired_components += 1 |
| repaired_pixels += int(original_repair_x.size) |
| donor_linear_indices.append( |
| (sample_y_all * w + sample_x_all).astype(np.int64) |
| ) |
| continue |
| distance, nearest_labels = cv2.distanceTransformWithLabels( |
| (~donor).astype(np.uint8), |
| cv2.DIST_L2, |
| 5, |
| labelType=cv2.DIST_LABEL_PIXEL, |
| ) |
| maximum_label = int(nearest_labels.max()) |
| nearest_y = np.zeros(maximum_label + 1, dtype=np.int32) |
| nearest_x = np.zeros(maximum_label + 1, dtype=np.int32) |
| donor_y, donor_x = np.nonzero(donor) |
| donor_labels = nearest_labels[donor_y, donor_x] |
| nearest_y[donor_labels] = donor_y |
| nearest_x[donor_labels] = donor_x |
| repair_labels = nearest_labels[repair_y, repair_x] |
| valid = (repair_labels > 0) & (repair_labels <= maximum_label) |
| if not np.all(valid): |
| skipped_pixels += int(np.count_nonzero(~valid)) |
| repair_y = repair_y[valid] |
| repair_x = repair_x[valid] |
| repair_labels = repair_labels[valid] |
| if repair_x.size == 0: |
| skipped_components += 1 |
| continue |
|
|
| source_y = nearest_y[repair_labels] |
| source_x = nearest_x[repair_labels] |
| reflected_y = np.clip(2 * source_y - repair_y, 0, h - 1) |
| reflected_x = np.clip(2 * source_x - repair_x, 0, w - 1) |
| reflected_valid = donor[reflected_y, reflected_x] |
| sample_y = np.where(reflected_valid, reflected_y, source_y) |
| sample_x = np.where(reflected_valid, reflected_x, source_x) |
| output[repair_y, repair_x] = colors[sample_y, sample_x] |
| remaining_indices = np.flatnonzero(remaining)[valid] |
| sample_y_all[remaining_indices] = sample_y |
| sample_x_all[remaining_indices] = sample_x |
|
|
| repaired_components += 1 |
| repaired_pixels += int(original_repair_x.size) |
| mirrored_pixels += int(np.count_nonzero(reflected_valid)) |
| donor_distances.append(distance[repair_y, repair_x].astype(np.float32)) |
| donor_linear_indices.append( |
| (sample_y_all * w + sample_x_all).astype(np.int64) |
| ) |
|
|
| distances = ( |
| np.concatenate(donor_distances) |
| if donor_distances |
| else np.empty((0,), dtype=np.float32) |
| ) |
| linear_indices = ( |
| np.concatenate(donor_linear_indices) |
| if donor_linear_indices |
| else np.empty((0,), dtype=np.int64) |
| ) |
| reuse_counts = ( |
| np.unique(linear_indices, return_counts=True)[1] |
| if linear_indices.size |
| else np.empty((0,), dtype=np.int64) |
| ) |
| unchanged_outside_repair = bool( |
| np.array_equal(output[~repair], colors[~repair]) |
| ) |
| copied_texel_provenance_fraction = ( |
| float(repaired_pixels / max(int(repair.sum()), 1)) |
| ) |
| return output, { |
| "role": "render_only_3d_texture_not_2d_completion", |
| "policy": ( |
| "same_connected_surface_coherent_translated_exemplar_then_" |
| "mirrored_nearest_fallback_without_diffusion_or_bbox_warp" |
| ), |
| "source_file_modified": False, |
| "frozen_2d_consumed_read_only": True, |
| "diffusion_or_mean_fill_used": False, |
| "telea_inpaint_used": False, |
| "obstacle_texture_pixels": int(repair.sum()), |
| "repaired_pixels": repaired_pixels, |
| "skipped_pixels": skipped_pixels, |
| "repaired_components": repaired_components, |
| "skipped_components": skipped_components, |
| "donor_exclusion_kernel": kernel_size, |
| "mirrored_donor_fraction": ( |
| float(mirrored_pixels / max(repaired_pixels, 1)) |
| ), |
| "translated_exemplar_pixels": translated_pixels, |
| "translated_exemplar_fraction": ( |
| float(translated_pixels / max(repaired_pixels, 1)) |
| ), |
| "translated_exemplar_components": translated_components, |
| "translation_shifts": translation_shifts, |
| "copied_texel_provenance_fraction": copied_texel_provenance_fraction, |
| "outside_repair_byte_identical": unchanged_outside_repair, |
| "donor_distance_mean_pixels": ( |
| float(np.mean(distances)) if distances.size else 0.0 |
| ), |
| "donor_distance_p95_pixels": ( |
| float(np.percentile(distances, 95)) if distances.size else 0.0 |
| ), |
| "donor_distance_max_pixels": ( |
| float(np.max(distances)) if distances.size else 0.0 |
| ), |
| "unique_donor_fraction": ( |
| float(np.unique(linear_indices).size / max(linear_indices.size, 1)) |
| ), |
| "donor_reuse_p95": ( |
| float(np.percentile(reuse_counts, 95)) if reuse_counts.size else 0.0 |
| ), |
| "donor_reuse_max": ( |
| int(np.max(reuse_counts)) if reuse_counts.size else 0 |
| ), |
| } |
|
|
|
|
| def resolve_source_image(sample_dir: Path) -> Path | None: |
| manifest_path = sample_dir / "pipeline_manifest.json" |
| if not manifest_path.is_file(): |
| return None |
| manifest = read_json(manifest_path) |
| image = manifest.get("image") |
| if image and Path(image).is_file(): |
| return Path(image) |
| source_manifest = manifest.get("source_pipeline_manifest") |
| if source_manifest and Path(source_manifest).is_file(): |
| source = read_json(Path(source_manifest)) |
| image = source.get("image") |
| if image and Path(image).is_file(): |
| return Path(image) |
| return None |
|
|
|
|
| def prepare_texture( |
| sample_dir: Path, |
| geometry_dir: Path, |
| shape: tuple[int, int], |
| *, |
| hidden_tint: float, |
| texture_override: Path | None = None, |
| ) -> tuple[np.ndarray, str]: |
| h, w = shape |
| if texture_override is not None: |
| if not texture_override.is_file(): |
| raise FileNotFoundError( |
| "The explicit frozen 2D texture is missing; refusing to fall " |
| f"back to a lower-quality or stale texture: {texture_override}" |
| ) |
| if hidden_tint > 0: |
| raise ValueError( |
| "A frozen 2D texture cannot be color-tinted in the 3D path. " |
| "Set --hidden-tint 0 to preserve its appearance." |
| ) |
| completed_candidates = [ |
| texture_override, |
| sample_dir / "03_2d_completion" / "completed_rgb_selected.png", |
| sample_dir / "amodal2d" / "completed_rgb_selected.png", |
| geometry_dir / "completed_rgb_inpaint.png", |
| ] |
| completed_rgb = next( |
| ( |
| candidate |
| for candidate in completed_candidates |
| if candidate is not None and candidate.is_file() |
| ), |
| None, |
| ) |
| if completed_rgb is not None: |
| texture = np.asarray(ImageOps.exif_transpose(Image.open(completed_rgb)).convert("RGB")) |
| source = str(completed_rgb) |
| already_completed = True |
| else: |
| source_image = resolve_source_image(sample_dir) |
| already_completed = False |
| if completed_rgb is None and source_image is not None: |
| texture = np.asarray(ImageOps.exif_transpose(Image.open(source_image)).convert("RGB")) |
| source = str(source_image) |
| elif completed_rgb is None: |
| texture = np.asarray(Image.open(geometry_dir / "completed_region_overlay.png").convert("RGB")) |
| source = str(geometry_dir / "completed_region_overlay.png") |
| if texture.shape[:2] != (h, w): |
| if already_completed: |
| if not aspect_ratio_matches(texture.shape[:2], (h, w)): |
| raise ValueError( |
| f"Completed RGB/geometry aspect mismatch: rgb={texture.shape[:2]}, geometry={(h, w)}." |
| ) |
| source_shape = texture.shape[:2] |
| texture = np.asarray( |
| Image.fromarray(texture).resize((w, h), Image.Resampling.LANCZOS) |
| ) |
| source += ( |
| f" + read_only_texture_resample_{source_shape[1]}x{source_shape[0]}" |
| f"_to_{w}x{h}" |
| ) |
| else: |
| if not aspect_ratio_matches(texture.shape[:2], (h, w)): |
| raise ValueError( |
| f"Fallback source RGB/geometry aspect mismatch: rgb={texture.shape[:2]}, geometry={(h, w)}." |
| ) |
| texture = np.asarray( |
| Image.fromarray(texture).resize((w, h), Image.Resampling.BILINEAR) |
| ) |
|
|
| hidden_path = geometry_dir / "hidden_completion_mask.png" |
| if hidden_path.is_file(): |
| hidden = read_mask(hidden_path, (h, w)) |
| if np.any(hidden): |
| if not already_completed: |
| inpaint_mask = (hidden.astype(np.uint8) * 255) |
| texture = cv2.inpaint(texture, inpaint_mask, 5, cv2.INPAINT_TELEA) |
| if hidden_tint > 0: |
| tint = np.array([255, 95, 25], dtype=np.float32) |
| texture = texture.astype(np.float32) |
| texture[hidden] = (1.0 - hidden_tint) * texture[hidden] + hidden_tint * tint |
| texture = np.clip(texture, 0, 255).astype(np.uint8) |
| source += " + clean_hidden_texture" if already_completed else " + hidden_texture_inpaint" |
| if hidden_tint > 0: |
| source += f" + hidden_tint_{hidden_tint:.3f}" |
| return texture.astype(np.uint8), source |
|
|
|
|
| def fit_plane(points: np.ndarray) -> tuple[np.ndarray, float] | None: |
| if points.shape[0] < 50: |
| return None |
| if points.shape[0] > 50000: |
| points = points[np.linspace(0, points.shape[0] - 1, 50000).astype(np.int64)] |
| sample = points |
| for _ in range(4): |
| centroid = sample.mean(axis=0) |
| centered = sample - centroid |
| try: |
| _, _, vh = np.linalg.svd(centered, full_matrices=False) |
| except np.linalg.LinAlgError: |
| return None |
| normal = vh[-1] |
| norm = float(np.linalg.norm(normal)) |
| if norm < 1e-6: |
| return None |
| normal = normal / norm |
| d = -float(np.dot(normal, centroid)) |
| residual = np.abs(sample @ normal + d) |
| threshold = max(float(np.percentile(residual, 75)), float(np.median(residual)) * 2.5, 1e-6) |
| retained = sample[residual <= threshold] |
| if retained.shape[0] < 50 or retained.shape[0] >= sample.shape[0] * 0.98: |
| return normal.astype(np.float32), d |
| sample = retained |
| return normal.astype(np.float32), d |
|
|
|
|
| def plane_depth_for_pixels( |
| shape: tuple[int, int], |
| plane: tuple[np.ndarray, float], |
| fx: float, |
| fy: float, |
| cx: float, |
| cy: float, |
| ) -> np.ndarray: |
| h, w = shape |
| normal, d = plane |
| xs, ys = np.meshgrid(np.arange(w, dtype=np.float32), np.arange(h, dtype=np.float32)) |
| rays = np.stack([(xs - cx) / fx, (ys - cy) / fy, np.ones((h, w), dtype=np.float32)], axis=-1) |
| denom = rays @ normal |
| with np.errstate(divide="ignore", invalid="ignore"): |
| depth = -d / denom |
| depth[~np.isfinite(depth)] = 0 |
| depth[depth <= 0] = 0 |
| return depth.astype(np.float32) |
|
|
|
|
| def robust_extent_from_points(points: np.ndarray) -> dict[str, Any]: |
| if points.shape[0] == 0: |
| return { |
| "span_x": 1.0, |
| "span_y": 0.2, |
| "span_z": 1.0, |
| "footprint_diag": 1.0, |
| "median_z": 1.0, |
| "percentile_low": [0.0, 0.0, 0.0], |
| "percentile_high": [1.0, 0.2, 1.0], |
| } |
| sample = points |
| if sample.shape[0] > 80000: |
| sample = sample[np.linspace(0, sample.shape[0] - 1, 80000).astype(np.int64)] |
| low = np.percentile(sample, 3, axis=0) |
| high = np.percentile(sample, 97, axis=0) |
| span = np.maximum(high - low, 1e-4) |
| return { |
| "span_x": float(span[0]), |
| "span_y": float(span[1]), |
| "span_z": float(span[2]), |
| "footprint_diag": float(np.hypot(span[0], span[2])), |
| "median_z": float(np.median(sample[:, 2])), |
| "percentile_low": [float(value) for value in low], |
| "percentile_high": [float(value) for value in high], |
| } |
|
|
|
|
| def depth_scale_for_mask(depth: np.ndarray, target_mask: np.ndarray) -> dict[str, Any]: |
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| valid = target_mask & np.isfinite(depth) & (depth > 0) |
| points = pixels_to_points(depth, fx, fy, cx, cy)[valid] |
| scale = robust_extent_from_points(points) |
| scale["valid_points"] = int(points.shape[0]) |
| return scale |
|
|
|
|
| def continuous_depth_for_category( |
| depth: np.ndarray, |
| target_mask: np.ndarray, |
| visible_mask: np.ndarray | None, |
| category: str, |
| ) -> tuple[np.ndarray, dict[str, Any]]: |
| if category not in CONTINUOUS_SURFACE_CATEGORIES: |
| return depth, {"surface_model": "depth_surface"} |
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| points = pixels_to_points(depth, fx, fy, cx, cy) |
| fit_mask = target_mask if visible_mask is None else (visible_mask & target_mask) |
| fit_mask &= np.isfinite(depth) & (depth > 0) |
| plane = fit_plane(points[fit_mask]) |
| if plane is None: |
| return depth, {"surface_model": "depth_surface_plane_fit_failed", "fit_points": int(fit_mask.sum())} |
| plane_depth = plane_depth_for_pixels((h, w), plane, fx, fy, cx, cy) |
| observed = depth[target_mask & np.isfinite(depth) & (depth > 0)] |
| if observed.size == 0: |
| return depth, {"surface_model": "depth_surface_no_observed_values", "fit_points": int(fit_mask.sum())} |
| low, high = np.percentile(observed, [1, 99]) |
| plausible = target_mask & np.isfinite(plane_depth) & (plane_depth > 0.5 * low) & (plane_depth < 1.5 * high) |
| continuous = depth.copy() |
| continuous[plausible] = plane_depth[plausible] |
| surface_model = ( |
| "raised_curb_planar_cap" if category in RAISED_CURB_CATEGORIES else "robust_single_plane" |
| ) |
| return continuous, { |
| "surface_model": surface_model, |
| "fit_points": int(fit_mask.sum()), |
| "plane_normal": [float(v) for v in plane[0]], |
| "plane_d": float(plane[1]), |
| "plane_applied_pixels": int(plausible.sum()), |
| } |
|
|
|
|
| def rotation_matrix(yaw_degrees: float, pitch_degrees: float, roll_degrees: float = 0.0) -> np.ndarray: |
| yaw = math.radians(yaw_degrees) |
| pitch = math.radians(pitch_degrees) |
| roll = math.radians(roll_degrees) |
| cy, sy = math.cos(yaw), math.sin(yaw) |
| cp, sp = math.cos(pitch), math.sin(pitch) |
| cr, sr = math.cos(roll), math.sin(roll) |
| ry = np.array([[cy, 0.0, sy], [0.0, 1.0, 0.0], [-sy, 0.0, cy]], dtype=np.float32) |
| rx = np.array([[1.0, 0.0, 0.0], [0.0, cp, -sp], [0.0, sp, cp]], dtype=np.float32) |
| rz = np.array([[cr, -sr, 0.0], [sr, cr, 0.0], [0.0, 0.0, 1.0]], dtype=np.float32) |
| return rz @ rx @ ry |
|
|
|
|
| def quality_review_ids(path: Path | None) -> set[str]: |
| if path is None or not path.is_file(): |
| return set() |
| data = read_json(path) |
| return {row["sample_id"] for row in data.get("review_candidates", [])} |
|
|
|
|
| def load_taxonomy(path: Path | None) -> dict[str, Any]: |
| if path is None or not path.is_file(): |
| return {} |
| return read_json(path).get("categories", {}) |
|
|
|
|
| def select_rows( |
| rows: list[dict[str, Any]], |
| review_ids: set[str], |
| limit: int, |
| per_category_limit: int, |
| include_review: bool, |
| sample_ids: list[str] | None = None, |
| ) -> list[dict[str, Any]]: |
| if sample_ids: |
| by_id = {row["sample_id"]: row for row in rows} |
| missing = [sample_id for sample_id in sample_ids if sample_id not in by_id] |
| if missing: |
| raise ValueError(f"sample_ids not found in manifest: {', '.join(missing)}") |
| return [by_id[sample_id] for sample_id in sample_ids] |
|
|
| selected = [] |
| counts: Counter[str] = Counter() |
| for row in rows: |
| sample_id = row["sample_id"] |
| if not include_review and sample_id in review_ids: |
| continue |
| category = str(row.get("category") or "unknown") |
| if per_category_limit > 0 and counts[category] >= per_category_limit: |
| continue |
| selected.append(row) |
| counts[category] += 1 |
| if limit > 0 and len(selected) >= limit: |
| break |
| return selected |
|
|
|
|
| def line_x_bounds_on_mask( |
| y_center: float, |
| slope: float, |
| mask: np.ndarray, |
| ) -> tuple[int, int] | None: |
| h, w = mask.shape |
| x_center = (w - 1) / 2.0 |
| xs = np.arange(w, dtype=np.float32) |
| ys = np.rint(y_center + slope * (xs - x_center)).astype(np.int32) |
| in_frame = (ys >= 0) & (ys < h) |
| if not np.any(in_frame): |
| return None |
| valid_xs = xs[in_frame].astype(np.int32) |
| valid_ys = ys[in_frame] |
| on_mask = mask[valid_ys, valid_xs] |
| if np.any(on_mask): |
| selected_xs = valid_xs[on_mask] |
| return int(selected_xs.min()), int(selected_xs.max()) |
| mask_ys, mask_xs = np.nonzero(mask) |
| if mask_xs.size == 0: |
| return None |
| return int(mask_xs.min()), int(mask_xs.max()) |
|
|
|
|
| def sample_nearest_depth( |
| depth: np.ndarray, |
| valid: np.ndarray, |
| x: float, |
| y: float, |
| fallback: float, |
| ) -> float: |
| h, w = depth.shape |
| cx = int(np.clip(round(x), 0, w - 1)) |
| cy = int(np.clip(round(y), 0, h - 1)) |
| if valid[cy, cx]: |
| return float(depth[cy, cx]) |
| for radius in (2, 4, 8, 14, 22): |
| y0 = max(0, cy - radius) |
| y1 = min(h, cy + radius + 1) |
| x0 = max(0, cx - radius) |
| x1 = min(w, cx + radius + 1) |
| patch_valid = valid[y0:y1, x0:x1] |
| if np.any(patch_valid): |
| return float(np.median(depth[y0:y1, x0:x1][patch_valid])) |
| return fallback |
|
|
|
|
| def sample_nearest_color(colors: np.ndarray, x: float, y: float) -> np.ndarray: |
| h, w = colors.shape[:2] |
| cx = int(np.clip(round(x), 0, w - 1)) |
| cy = int(np.clip(round(y), 0, h - 1)) |
| return colors[cy, cx].astype(np.uint8) |
|
|
|
|
| def regularize_depth_row(depth_values: np.ndarray) -> np.ndarray: |
| values = depth_values.astype(np.float32).copy() |
| finite = np.isfinite(values) & (values > 0) |
| if finite.sum() == 0: |
| return values |
| fallback = float(np.median(values[finite])) |
| values[~finite] = fallback |
| if len(values) < 4: |
| return values |
| xs = np.linspace(-1.0, 1.0, len(values), dtype=np.float32) |
| residual_center = float(np.median(values)) |
| residual_scale = float(np.median(np.abs(values - residual_center))) + 1e-6 |
| inliers = np.abs(values - residual_center) <= 3.0 * residual_scale |
| if inliers.sum() >= 2: |
| degree = 1 |
| coefficients = np.polyfit(xs[inliers], values[inliers], degree) |
| fitted = np.polyval(coefficients, xs).astype(np.float32) |
| values = 0.35 * values + 0.65 * fitted |
| kernel = np.array([0.20, 0.60, 0.20], dtype=np.float32) |
| padded = np.pad(values, (1, 1), mode="edge") |
| return np.convolve(padded, kernel, mode="valid").astype(np.float32) |
|
|
|
|
| def polygon_signed_area(points: np.ndarray) -> float: |
| """Return the signed area of an Nx2 polygon without repeating its first point.""" |
| xs = points[:, 0].astype(np.float64) |
| ys = points[:, 1].astype(np.float64) |
| return 0.5 * float(np.sum(xs * np.roll(ys, -1) - np.roll(xs, -1) * ys)) |
|
|
|
|
| def triangulate_simple_polygon(points: np.ndarray) -> list[tuple[int, int, int]]: |
| """Triangulate a simple image-space polygon with deterministic ear clipping.""" |
| polygon = np.asarray(points, dtype=np.float64).reshape((-1, 2)) |
| if len(polygon) < 3: |
| raise ValueError("A polygon needs at least three vertices") |
| orientation = 1.0 if polygon_signed_area(polygon) > 0 else -1.0 |
| remaining = list(range(len(polygon))) |
| triangles: list[tuple[int, int, int]] = [] |
|
|
| def cross(a: np.ndarray, b: np.ndarray, c: np.ndarray) -> float: |
| ab = b - a |
| ac = c - a |
| return float(ab[0] * ac[1] - ab[1] * ac[0]) |
|
|
| def inside_triangle(point: np.ndarray, a: np.ndarray, b: np.ndarray, c: np.ndarray) -> bool: |
| c1 = orientation * cross(a, b, point) |
| c2 = orientation * cross(b, c, point) |
| c3 = orientation * cross(c, a, point) |
| return c1 >= -1e-8 and c2 >= -1e-8 and c3 >= -1e-8 |
|
|
| guard = 0 |
| while len(remaining) > 3: |
| ear_found = False |
| for offset, current in enumerate(remaining): |
| previous = remaining[offset - 1] |
| following = remaining[(offset + 1) % len(remaining)] |
| a, b, c = polygon[[previous, current, following]] |
| if orientation * cross(a, b, c) <= 1e-8: |
| continue |
| if any( |
| inside_triangle(polygon[candidate], a, b, c) |
| for candidate in remaining |
| if candidate not in {previous, current, following} |
| ): |
| continue |
| triangles.append((previous, current, following)) |
| del remaining[offset] |
| ear_found = True |
| break |
| guard += 1 |
| if not ear_found or guard > len(polygon) * len(polygon): |
| raise ValueError("Could not triangulate simplified curb contour") |
| triangles.append(tuple(remaining)) |
| return triangles |
|
|
|
|
| def simplified_largest_contour( |
| mask: np.ndarray, |
| epsilon_ratio: float, |
| ) -> tuple[np.ndarray, dict[str, Any]]: |
| """Extract a simplified continuous outline, avoiding raster-staircase side walls.""" |
| contours, _ = cv2.findContours( |
| mask.astype(np.uint8), |
| cv2.RETR_EXTERNAL, |
| cv2.CHAIN_APPROX_NONE, |
| ) |
| if not contours: |
| raise ValueError("No target contour for raised curb") |
| contour = max(contours, key=cv2.contourArea) |
| perimeter = float(cv2.arcLength(contour, True)) |
| epsilon = max(1.0, float(epsilon_ratio) * perimeter) |
| simplified = cv2.approxPolyDP(contour, epsilon, True).reshape((-1, 2)).astype(np.float32) |
| if len(simplified) < 3: |
| raise ValueError("Raised-curb contour collapsed during simplification") |
| return simplified, { |
| "raw_contour_vertices": int(len(contour)), |
| "contour_vertices": int(len(simplified)), |
| "contour_perimeter_pixels": perimeter, |
| "contour_epsilon_pixels": epsilon, |
| "contour_epsilon_ratio": float(epsilon_ratio), |
| } |
|
|
|
|
| def build_raised_curb_prism_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| height: float | None, |
| height_ratio: float, |
| contour_epsilon_ratio: float, |
| stride: int, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Build one smooth, vertically extruded curb body from a simplified mask contour.""" |
| h, w = depth.shape |
| valid = target_mask & np.isfinite(depth) & (depth > 0) |
| if not np.any(valid): |
| raise ValueError("No valid target depth values for raised curb") |
| contour, contour_metadata = simplified_largest_contour( |
| target_mask, |
| contour_epsilon_ratio, |
| ) |
| triangles = triangulate_simple_polygon(contour) |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| points = pixels_to_points(depth, fx, fy, cx, cy) |
| fallback_depth = float(np.median(depth[valid])) |
|
|
| top_vertices: list[np.ndarray] = [] |
| for x, y in contour: |
| sampled_depth = sample_nearest_depth(depth, valid, float(x), float(y), fallback_depth) |
| ray = np.array([(float(x) - cx) / fx, (float(y) - cy) / fy, 1.0], dtype=np.float32) |
| top_vertices.append(ray * sampled_depth) |
|
|
| scale = robust_extent_from_points(points[valid]) |
| median_depth = float(np.median(depth[valid])) |
| if height is not None: |
| curb_height = float(height) |
| height_source = "explicit_relative_height" |
| else: |
| |
| |
| |
| automatic_height_cap = 0.35 * median_depth |
| curb_height = min( |
| max( |
| float(height_ratio) * median_depth, |
| 0.065 * median_depth, |
| 1.15 * max(float(scale["span_y"]), 1e-3), |
| ), |
| automatic_height_cap, |
| ) |
| height_source = "median_depth_ratio_with_vertical_evidence_cap" |
| top_array = np.asarray(top_vertices, dtype=np.float32) |
| bottom_array = top_array.copy() |
| bottom_array[:, 1] += curb_height |
| target_colors = colors[valid].astype(np.float32) |
| base_color = np.median(target_colors, axis=0) |
| base_color = 0.84 * base_color + 0.16 * np.array([172.0, 172.0, 166.0], dtype=np.float32) |
| top_color_array = np.repeat( |
| np.clip(base_color, 0, 255).astype(np.uint8)[None, :], |
| len(top_array), |
| axis=0, |
| ) |
| bottom_color_array = np.clip(top_color_array.astype(np.float32) * 0.62, 0, 255).astype(np.uint8) |
| vertices = np.concatenate([top_array, bottom_array], axis=0) |
| vertex_colors = np.concatenate([top_color_array, bottom_color_array], axis=0) |
|
|
| count = len(top_array) |
| top_faces = list(triangles) |
| bottom_faces = [(c + count, b + count, a + count) for a, b, c in triangles] |
| side_faces: list[tuple[int, int, int]] = [] |
| for a in range(count): |
| b = (a + 1) % count |
| |
| |
| side_faces.append((b, a, a + count)) |
| side_faces.append((b, a + count, b + count)) |
| faces = np.asarray(top_faces + bottom_faces + side_faces, dtype=np.int64) |
| metadata = { |
| "front_faces": len(top_faces), |
| "back_faces": len(bottom_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(side_faces), |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": 0, |
| "stair_sidewall_bottom_faces": 0, |
| "solid_thickness": curb_height, |
| "max_depth_jump": None, |
| "stride": stride, |
| "support_mode": "raised_curb_prism", |
| "stair_surface_mode": "not_applicable", |
| "stair_rows": [], |
| "stair_cross_samples": None, |
| "stair_edge_slope": None, |
| "stair_step_count": None, |
| "surface_model": "raised_curb_contour_prism", |
| "curb_height": curb_height, |
| "curb_height_ratio": float(height_ratio), |
| "curb_height_source": height_source, |
| **contour_metadata, |
| "depth_scaled_dimensions": { |
| "height": curb_height, |
| "source_span_x": float(scale["span_x"]), |
| "source_span_y": float(scale["span_y"]), |
| "source_span_z": float(scale["span_z"]), |
| "footprint_diag": float(scale["footprint_diag"]), |
| "median_depth": median_depth, |
| "automatic_height_cap": 0.35 * median_depth, |
| "depth_units": "relative_pseudo_depth", |
| }, |
| } |
| return vertices, vertex_colors, faces, metadata |
|
|
|
|
| def build_continuous_stair_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| stride: int, |
| thickness: float | None, |
| thickness_ratio: float, |
| support_mode: str, |
| wedge_base_drop_ratio: float, |
| stair_edges_y: list[int] | None, |
| stair_edge_slope: float, |
| cross_samples: int, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| h, w = depth.shape |
| valid = target_mask & np.isfinite(depth) & (depth > 0) |
| depth_values = depth[valid] |
| if depth_values.size == 0: |
| raise ValueError("No valid target depth values") |
| median_depth = float(np.median(depth_values)) |
| depth_span = float(np.percentile(depth_values, 95) - np.percentile(depth_values, 5)) |
| solid_thickness = float(thickness) if thickness is not None else max( |
| 0.04 * median_depth, |
| thickness_ratio * max(depth_span, 1e-3), |
| ) |
|
|
| mask_ys, mask_xs = np.nonzero(target_mask) |
| if mask_ys.size == 0: |
| raise ValueError("No target mask pixels") |
| top_y = int(mask_ys.min()) |
| bottom_y = int(mask_ys.max()) |
| min_gap = max(3, int(round(stride * 0.75))) |
| requested_rows = [top_y] |
| requested_rows.extend(int(edge) for edge in (stair_edges_y or []) if top_y < int(edge) < bottom_y) |
| requested_rows.append(bottom_y) |
| requested_rows = sorted(requested_rows) |
| row_centers: list[int] = [] |
| for row in requested_rows: |
| if not row_centers or abs(row - row_centers[-1]) >= min_gap: |
| row_centers.append(int(row)) |
| else: |
| row_centers[-1] = int(round((row_centers[-1] + row) * 0.5)) |
| if len(row_centers) < 2: |
| row_centers = [top_y, bottom_y] |
|
|
| bounds = [line_x_bounds_on_mask(float(row), stair_edge_slope, target_mask) for row in row_centers] |
| if any(item is None for item in bounds): |
| raise ValueError("Could not derive stair line bounds from target mask") |
| typed_bounds = [item for item in bounds if item is not None] |
| widest = max(right - left for left, right in typed_bounds) |
| if cross_samples <= 0: |
| cross_samples = int(np.clip(math.ceil(widest / max(stride, 1)) + 1, 12, 96)) |
| cross_samples = max(4, int(cross_samples)) |
|
|
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| image_x_center = (w - 1) / 2.0 |
| vertices: list[np.ndarray] = [] |
| vertex_colors: list[np.ndarray] = [] |
| front_id = np.full((len(row_centers), cross_samples), -1, dtype=np.int64) |
| back_id = np.full_like(front_id, -1) |
|
|
| row_depths: list[np.ndarray] = [] |
| row_pixels: list[list[tuple[float, float]]] = [] |
| global_fallback = float(np.median(depth_values)) |
| for row, (left, right) in zip(row_centers, typed_bounds): |
| xs = np.linspace(float(left), float(right), cross_samples, dtype=np.float32) |
| pixels = [(float(x), float(row + stair_edge_slope * (float(x) - image_x_center))) for x in xs] |
| sampled = np.array( |
| [sample_nearest_depth(depth, valid, x, y, global_fallback) for x, y in pixels], |
| dtype=np.float32, |
| ) |
| row_depths.append(regularize_depth_row(sampled)) |
| row_pixels.append(pixels) |
|
|
| for row_index, (pixels, depths) in enumerate(zip(row_pixels, row_depths)): |
| for col_index, ((x, y), z) in enumerate(zip(pixels, depths)): |
| z = float(max(z, 1e-4)) |
| point = np.array( |
| [ |
| (float(x) - cx) * z / fx, |
| (float(y) - cy) * z / fy, |
| z, |
| ], |
| dtype=np.float32, |
| ) |
| front_id[row_index, col_index] = len(vertices) |
| vertices.append(point) |
| vertex_colors.append(sample_nearest_color(colors, x, y)) |
|
|
| back_depth = z + solid_thickness |
| back_point = point * (back_depth / max(z, 1e-6)) |
| back_id[row_index, col_index] = len(vertices) |
| vertices.append(back_point.astype(np.float32)) |
| vertex_colors.append(np.clip(vertex_colors[-1].astype(np.float32) * 0.54, 0, 255).astype(np.uint8)) |
|
|
| front_faces: list[tuple[int, int, int]] = [] |
| back_faces: list[tuple[int, int, int]] = [] |
| for row_index in range(len(row_centers) - 1): |
| for col_index in range(cross_samples - 1): |
| a = int(front_id[row_index, col_index]) |
| b = int(front_id[row_index, col_index + 1]) |
| c = int(front_id[row_index + 1, col_index]) |
| d = int(front_id[row_index + 1, col_index + 1]) |
| ba = int(back_id[row_index, col_index]) |
| bb = int(back_id[row_index, col_index + 1]) |
| bc = int(back_id[row_index + 1, col_index]) |
| bd = int(back_id[row_index + 1, col_index + 1]) |
| front_faces.append((a, c, b)) |
| front_faces.append((b, c, d)) |
| back_faces.append((bb, bc, ba)) |
| back_faces.append((bd, bc, bb)) |
|
|
| boundary_side_faces: list[tuple[int, int, int]] = [] |
|
|
| def append_backed_strip(front_a: int, front_b: int, back_a: int, back_b: int) -> None: |
| boundary_side_faces.append((front_a, front_b, back_a)) |
| boundary_side_faces.append((front_b, back_b, back_a)) |
|
|
| last_row = len(row_centers) - 1 |
| last_col = cross_samples - 1 |
| for col_index in range(cross_samples - 1): |
| append_backed_strip( |
| int(front_id[0, col_index]), |
| int(front_id[0, col_index + 1]), |
| int(back_id[0, col_index]), |
| int(back_id[0, col_index + 1]), |
| ) |
| append_backed_strip( |
| int(front_id[last_row, col_index + 1]), |
| int(front_id[last_row, col_index]), |
| int(back_id[last_row, col_index + 1]), |
| int(back_id[last_row, col_index]), |
| ) |
| for row_index in range(len(row_centers) - 1): |
| append_backed_strip( |
| int(front_id[row_index + 1, 0]), |
| int(front_id[row_index, 0]), |
| int(back_id[row_index + 1, 0]), |
| int(back_id[row_index, 0]), |
| ) |
| append_backed_strip( |
| int(front_id[row_index, last_col]), |
| int(front_id[row_index + 1, last_col]), |
| int(back_id[row_index, last_col]), |
| int(back_id[row_index + 1, last_col]), |
| ) |
|
|
| stair_sidewall_faces: list[tuple[int, int, int]] = [] |
| stair_sidewall_bottom_faces: list[tuple[int, int, int]] = [] |
| if support_mode == "stair_sidewall": |
| front_points = np.asarray([vertices[int(front_id[row_index, col])] for row_index in range(len(row_centers)) for col in (0, last_col)], dtype=np.float32) |
| base_y = float(front_points[:, 1].max()) + max( |
| solid_thickness * 1.5, |
| wedge_base_drop_ratio * max(median_depth, 1e-3), |
| 1e-3, |
| ) |
|
|
| def make_bottom(top_id: int) -> int: |
| bottom = vertices[top_id].astype(np.float32).copy() |
| bottom[1] = base_y |
| bottom_id = len(vertices) |
| vertices.append(bottom) |
| vertex_colors.append(np.clip(vertex_colors[top_id].astype(np.float32) * 0.70, 0, 255).astype(np.uint8)) |
| return bottom_id |
|
|
| left_ids = [int(front_id[row_index, 0]) for row_index in range(len(row_centers))] |
| right_ids = [int(front_id[row_index, last_col]) for row_index in range(len(row_centers))] |
| left_bottom = [make_bottom(vertex_id) for vertex_id in left_ids] |
| right_bottom = [make_bottom(vertex_id) for vertex_id in right_ids] |
| for top_a, top_b, bottom_a, bottom_b in zip(left_ids[:-1], left_ids[1:], left_bottom[:-1], left_bottom[1:]): |
| stair_sidewall_faces.append((top_a, top_b, bottom_a)) |
| stair_sidewall_faces.append((top_b, bottom_b, bottom_a)) |
| for top_a, top_b, bottom_a, bottom_b in zip(right_ids[:-1], right_ids[1:], right_bottom[:-1], right_bottom[1:]): |
| stair_sidewall_faces.append((top_a, bottom_a, top_b)) |
| stair_sidewall_faces.append((top_b, bottom_a, bottom_b)) |
| for left_a, left_b, right_a, right_b in zip(left_bottom[:-1], left_bottom[1:], right_bottom[:-1], right_bottom[1:]): |
| stair_sidewall_bottom_faces.append((left_a, right_a, left_b)) |
| stair_sidewall_bottom_faces.append((left_b, right_a, right_b)) |
|
|
| side_faces = boundary_side_faces + stair_sidewall_faces + stair_sidewall_bottom_faces |
| face_rows = front_faces + back_faces + side_faces |
| faces = ( |
| np.asarray(face_rows, dtype=np.int64).reshape((-1, 3)) |
| if face_rows |
| else np.empty((0, 3), dtype=np.int64) |
| ) |
| metadata = { |
| "front_faces": len(front_faces), |
| "back_faces": len(back_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(boundary_side_faces), |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": len(stair_sidewall_faces), |
| "stair_sidewall_bottom_faces": len(stair_sidewall_bottom_faces), |
| "solid_thickness": solid_thickness, |
| "max_depth_jump": None, |
| "stride": stride, |
| "support_mode": support_mode, |
| "stair_surface_mode": "continuous_step_strips", |
| "stair_rows": row_centers, |
| "stair_edge_slope": float(stair_edge_slope), |
| "stair_cross_samples": int(cross_samples), |
| } |
| return np.asarray(vertices, dtype=np.float32), np.asarray(vertex_colors, dtype=np.uint8), faces, metadata |
|
|
|
|
| def infer_stair_edge_rows( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| stair_edges_y: list[int] | None, |
| ) -> tuple[list[int], dict[str, Any]]: |
| """Filter duplicate/noisy tread candidates without inventing extra stairs. |
| |
| The upstream detector intentionally has high recall. A presentation mesh |
| must not interpret every weak or repeated line as a physical riser. This |
| filter combines target support, RGB horizontal-edge evidence, and depth |
| discontinuity evidence, then merges the characteristic clusters produced |
| when one physical edge is detected several times. |
| """ |
|
|
| mask_ys, _ = np.nonzero(target_mask) |
| if mask_ys.size == 0: |
| return [], { |
| "policy": "empty_target", |
| "candidate_rows": [], |
| "retained_rows": [], |
| "confidence": 0.0, |
| } |
| top_y = int(mask_ys.min()) |
| bottom_y = int(mask_ys.max()) |
| candidates = sorted( |
| { |
| int(row) |
| for row in (stair_edges_y or []) |
| if top_y <= int(row) <= bottom_y |
| } |
| ) |
| if not candidates: |
| return [], { |
| "policy": "no_supported_detector_rows", |
| "candidate_rows": [], |
| "retained_rows": [], |
| "confidence": 0.0, |
| } |
|
|
| gray = cv2.cvtColor(colors, cv2.COLOR_RGB2GRAY).astype(np.float32) / 255.0 |
| rgb_gradient = np.abs(cv2.Sobel(gray, cv2.CV_32F, 0, 1, ksize=3)) |
| depth_f = depth.astype(np.float32).copy() |
| finite_depth = np.isfinite(depth_f) & (depth_f > 0) |
| replacement = ( |
| float(np.median(depth_f[finite_depth])) if np.any(finite_depth) else 1.0 |
| ) |
| depth_f[~finite_depth] = replacement |
| depth_gradient = np.abs(cv2.Sobel(depth_f, cv2.CV_32F, 0, 1, ksize=3)) |
| rgb_reference = max(float(np.percentile(rgb_gradient[target_mask], 75)), 1e-4) |
| depth_reference = max( |
| float(np.percentile(depth_gradient[target_mask], 75)), |
| 1e-4, |
| ) |
| band_radius = max(2, int(round(target_mask.shape[0] / 512.0))) |
| evidence: list[dict[str, Any]] = [] |
| for row in candidates: |
| row_slice = slice( |
| max(0, row - band_radius), |
| min(target_mask.shape[0], row + band_radius + 1), |
| ) |
| support = target_mask[row_slice] |
| rgb_values = rgb_gradient[row_slice][support] |
| depth_values = depth_gradient[row_slice][support] |
| evidence.append( |
| { |
| "row": row, |
| "support_pixels": int(support.sum()), |
| "rgb_strength": ( |
| float(np.percentile(rgb_values, 75)) / rgb_reference |
| if rgb_values.size |
| else 0.0 |
| ), |
| "depth_strength": ( |
| float(np.percentile(depth_values, 75)) / depth_reference |
| if depth_values.size |
| else 0.0 |
| ), |
| } |
| ) |
| maximum_support = max(item["support_pixels"] for item in evidence) |
| for item in evidence: |
| support_ratio = item["support_pixels"] / max(maximum_support, 1) |
| item["support_ratio"] = float(support_ratio) |
| item["score"] = float( |
| 0.50 * min(item["rgb_strength"], 3.0) / 3.0 |
| + 0.30 * min(item["depth_strength"], 3.0) / 3.0 |
| + 0.20 * support_ratio |
| ) |
|
|
| supported = [ |
| item |
| for item in evidence |
| if ( |
| item["support_ratio"] >= 0.12 |
| and item["score"] >= 0.16 |
| and max(item["rgb_strength"], item["depth_strength"]) >= 0.25 |
| ) |
| ] |
| if not supported: |
| return [], { |
| "policy": "insufficient_rgb_depth_edge_evidence", |
| "candidate_rows": candidates, |
| "evidence": evidence, |
| "retained_rows": [], |
| "confidence": 0.0, |
| "step_count_policy": ( |
| "no_physical_step_count_inferred_from_unsupported_rows" |
| ), |
| } |
| supported.sort(key=lambda item: item["row"]) |
|
|
| |
| |
| |
| rows = [int(item["row"]) for item in supported] |
| gap_policy = "non_maximum_suppression" |
| if len(rows) >= 5: |
| gaps = np.diff(rows).astype(np.float32) |
| sorted_gaps = np.sort(gaps) |
| ratios = sorted_gaps[1:] / np.maximum(sorted_gaps[:-1], 1.0) |
| split_index = int(np.argmax(ratios)) if ratios.size else -1 |
| split_ratio = float(ratios[split_index]) if split_index >= 0 else 1.0 |
| threshold = ( |
| 0.5 * float(sorted_gaps[split_index] + sorted_gaps[split_index + 1]) |
| if split_index >= 0 |
| else 0.0 |
| ) |
| large_gap_count = int(np.sum(gaps > threshold)) if threshold > 0 else 0 |
| small_gap_count = int(np.sum(gaps <= threshold)) if threshold > 0 else 0 |
| if split_ratio >= 1.8 and large_gap_count >= 2 and small_gap_count >= 2: |
| groups: list[list[dict[str, Any]]] = [[supported[0]]] |
| for gap, item in zip(gaps, supported[1:]): |
| if float(gap) > threshold: |
| groups.append([item]) |
| else: |
| groups[-1].append(item) |
| merged: list[dict[str, Any]] = [] |
| for group in groups: |
| weights = np.asarray( |
| [max(float(item["score"]), 1e-4) for item in group], |
| dtype=np.float64, |
| ) |
| merged_row = int( |
| round( |
| np.average( |
| [float(item["row"]) for item in group], |
| weights=weights, |
| ) |
| ) |
| ) |
| merged.append( |
| { |
| **max(group, key=lambda item: item["score"]), |
| "row": merged_row, |
| "merged_rows": [int(item["row"]) for item in group], |
| } |
| ) |
| supported = merged |
| gap_policy = "bimodal_gap_cluster_merge" |
|
|
| if gap_policy == "non_maximum_suppression": |
| minimum_separation = max(4, int(round(0.055 * max(bottom_y - top_y, 1)))) |
| retained: list[dict[str, Any]] = [] |
| for item in sorted(supported, key=lambda value: value["score"], reverse=True): |
| if all( |
| abs(int(item["row"]) - int(other["row"])) >= minimum_separation |
| for other in retained |
| ): |
| retained.append(item) |
| supported = sorted(retained, key=lambda item: item["row"]) |
|
|
| retained_rows = [int(item["row"]) for item in supported] |
| return retained_rows, { |
| "policy": gap_policy, |
| "candidate_rows": candidates, |
| "evidence": evidence, |
| "retained_rows": retained_rows, |
| "confidence": ( |
| float(np.mean([item["score"] for item in supported])) |
| if supported |
| else 0.0 |
| ), |
| "step_count_policy": "one_retained_physical_riser_row_equals_one_step", |
| } |
|
|
|
|
| def trusted_stair_edges_from_manifest( |
| geometry_manifest: dict[str, Any], |
| *, |
| minimum_confidence: float = 0.35, |
| ) -> tuple[list[int], dict[str, Any]]: |
| """Reject detector rows whose provenance cannot support a physical count.""" |
|
|
| raw_rows = [ |
| int(value) for value in geometry_manifest.get("stair_edges_y", []) |
| ] |
| used_regular_fallback = bool( |
| geometry_manifest.get("used_regular_fallback_stair_edges", False) |
| ) |
| confidence_value = geometry_manifest.get("stair_edge_confidence") |
| confidence = ( |
| float(confidence_value) if confidence_value is not None else None |
| ) |
| source = ( |
| geometry_manifest.get("stair_edge_source") |
| or geometry_manifest.get("edge_source") |
| or ( |
| "regular_fallback" |
| if used_regular_fallback |
| else "detector_rows_with_manifest_confidence" |
| ) |
| ) |
| if used_regular_fallback: |
| trusted_rows: list[int] = [] |
| policy = "reject_regular_fallback_rows_for_physical_step_count" |
| elif confidence is None: |
| trusted_rows = [] |
| policy = "reject_rows_without_detector_confidence" |
| elif confidence < minimum_confidence: |
| trusted_rows = [] |
| policy = "reject_low_confidence_detector_rows" |
| else: |
| trusted_rows = raw_rows |
| policy = "accept_detector_rows_for_rgb_depth_filtering" |
| return trusted_rows, { |
| "raw_rows": raw_rows, |
| "trusted_rows": trusted_rows, |
| "used_regular_fallback": used_regular_fallback, |
| "detector_confidence": confidence, |
| "minimum_confidence": float(minimum_confidence), |
| "edge_source": str(source), |
| "policy": policy, |
| } |
|
|
|
|
| def stair_source_band_bounds( |
| rows: list[int], |
| top_y: int, |
| bottom_y: int, |
| ) -> list[int]: |
| """Turn physical stair-edge centers into one source-texture band per step.""" |
|
|
| if not rows: |
| return [top_y, bottom_y] |
| if len(rows) == 1: |
| half_span = max(2, int(round(0.20 * max(bottom_y - top_y, 1)))) |
| return [max(top_y, rows[0] - half_span), min(bottom_y, rows[0] + half_span)] |
| midpoints = [ |
| int(round(0.5 * (left + right))) |
| for left, right in zip(rows[:-1], rows[1:]) |
| ] |
| first_gap = max(rows[1] - rows[0], 2) |
| last_gap = max(rows[-1] - rows[-2], 2) |
| bounds = [ |
| max(top_y, int(round(rows[0] - 0.5 * first_gap))), |
| *midpoints, |
| min(bottom_y, int(round(rows[-1] + 0.5 * last_gap))), |
| ] |
| for index in range(1, len(bounds)): |
| bounds[index] = max(bounds[index], bounds[index - 1] + 1) |
| if bounds[-1] > bottom_y: |
| return ( |
| np.linspace(top_y, bottom_y, len(rows) + 1) |
| .round() |
| .astype(int) |
| .tolist() |
| ) |
| return bounds |
|
|
|
|
| def estimate_stair_dimensions( |
| depth: np.ndarray, |
| target_mask: np.ndarray, |
| stair_rows: list[int], |
| step_count: int, |
| *, |
| metric_style_depth: bool, |
| ) -> dict[str, Any]: |
| """Estimate staircase dimensions from depth points with conservative bounds.""" |
|
|
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| points = pixels_to_points(depth, fx, fy, cx, cy) |
| scale = depth_scale_for_mask(depth, target_mask) |
| row_points: list[np.ndarray] = [] |
| for row in stair_rows: |
| radius = max(2, int(round(h / 512.0))) |
| row_mask = np.zeros_like(target_mask) |
| row_mask[max(0, row - radius) : min(h, row + radius + 1)] = True |
| valid = row_mask & target_mask & np.isfinite(depth) & (depth > 0) |
| if np.any(valid): |
| row_points.append(np.median(points[valid], axis=0)) |
| raw_treads: list[float] = [] |
| raw_rises: list[float] = [] |
| for left, right in zip(row_points[:-1], row_points[1:]): |
| raw_treads.append(abs(float(right[2] - left[2]))) |
| raw_rises.append(abs(float(right[1] - left[1]))) |
|
|
| count = max(int(step_count), 1) |
| positive_treads = [value for value in raw_treads if value > 1e-6] |
| positive_rises = [value for value in raw_rises if value > 1e-6] |
| raw_tread = ( |
| float(np.median(positive_treads)) if positive_treads else 0.0 |
| ) |
| raw_rise = ( |
| float(np.median(positive_rises)) if positive_rises else 0.0 |
| ) |
| median_depth = max(float(scale["median_z"]), 1e-3) |
| depth_discontinuity_threshold = ( |
| max(0.03, 0.01 * median_depth) |
| if metric_style_depth |
| else 0.01 * median_depth |
| ) |
| depth_edge_evidence = bool( |
| len(row_points) >= 2 |
| and raw_tread >= depth_discontinuity_threshold |
| ) |
| if metric_style_depth: |
| width = float(np.clip(float(scale["span_x"]), 0.75, 6.0)) |
| units = "metric_style_depth_meters_not_calibrated_truth" |
| if depth_edge_evidence: |
| tread_depth = float(np.clip(raw_tread, 0.22, 0.55)) |
| |
| |
| |
| rise = float(np.clip(raw_rise, 0.08, 0.22)) |
| source = ( |
| "depth_edge_points_with_accessibility_plausibility_bounds" |
| ) |
| confidence = "medium" |
| scale_status = ( |
| "metric_style_model_with_unverified_monocular_scale" |
| ) |
| else: |
| tread_depth = 0.30 |
| rise = 0.15 |
| source = ( |
| "category_prior_due_to_insufficient_depth_discontinuity" |
| ) |
| confidence = "low" |
| scale_status = ( |
| "category_prior_fallback_no_metric_depth_edge_evidence" |
| ) |
| else: |
| width = float( |
| np.clip( |
| float(scale["span_x"]), |
| 0.18 * median_depth, |
| 1.5 * median_depth, |
| ) |
| ) |
| units = "relative_pseudo_depth" |
| if depth_edge_evidence: |
| tread_depth = float( |
| np.clip( |
| raw_tread, |
| 0.04 * median_depth, |
| 0.22 * median_depth, |
| ) |
| ) |
| rise = float( |
| np.clip( |
| raw_rise, |
| 0.018 * median_depth, |
| 0.09 * median_depth, |
| ) |
| ) |
| source = ( |
| "relative_depth_edge_points_with_scale_relative_bounds" |
| ) |
| confidence = "low" |
| scale_status = "relative_monocular_scale_only" |
| else: |
| tread_depth = 0.10 * median_depth |
| rise = 0.045 * median_depth |
| source = ( |
| "scale_relative_category_prior_due_to_insufficient_depth_discontinuity" |
| ) |
| confidence = "low" |
| scale_status = ( |
| "relative_category_prior_no_depth_edge_evidence" |
| ) |
| total_length = tread_depth * count |
| total_height = rise * count |
| return { |
| "width": width, |
| "length": total_length, |
| "height": total_height, |
| "tread_depth": tread_depth, |
| "riser_height": rise, |
| "raw_tread_depth": raw_tread, |
| "raw_riser_height": raw_rise, |
| "row_point_count": len(row_points), |
| "row_pair_count": max(len(row_points) - 1, 0), |
| "depth_edge_evidence": depth_edge_evidence, |
| "depth_discontinuity_threshold": depth_discontinuity_threshold, |
| "dimension_confidence": confidence, |
| "source_span_x": float(scale["span_x"]), |
| "source_span_y": float(scale["span_y"]), |
| "source_span_z": float(scale["span_z"]), |
| "footprint_diag": float(scale["footprint_diag"]), |
| "valid_points": int(scale["valid_points"]), |
| "depth_units": units, |
| "dimension_source": source, |
| "physical_scale_status": scale_status, |
| "intrinsics_source": "default_focal_equals_image_extent", |
| "gravity_alignment": "unavailable", |
| "camera_pose_assumption": ( |
| "approximately_level_camera_only_for_bounded_projected_vertical_evidence" |
| ), |
| "calibrated_metric_truth": False, |
| } |
|
|
|
|
| def build_canonical_staircase_prism_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| stride: int, |
| thickness: float | None, |
| thickness_ratio: float, |
| stair_edges_y: list[int] | None, |
| step_count: int, |
| stair_edge_slope: float = 0.0, |
| metric_style_depth: bool = False, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| mask_ys, mask_xs = np.nonzero(target_mask) |
| if mask_ys.size == 0: |
| raise ValueError("No target mask pixels") |
| h, w = target_mask.shape |
| bbox_width = max(int(mask_xs.max() - mask_xs.min()), 1) |
| bbox_height = max(int(mask_ys.max() - mask_ys.min()), 1) |
| filtered_rows, stair_inference = infer_stair_edge_rows( |
| depth, |
| colors, |
| target_mask, |
| stair_edges_y, |
| ) |
| if step_count > 0: |
| inferred_steps = step_count |
| step_count_source = "explicit_review_override" |
| elif filtered_rows: |
| inferred_steps = len(filtered_rows) |
| step_count_source = "filtered_rgb_depth_riser_evidence" |
| else: |
| inferred_steps = 1 |
| step_count_source = "conservative_single_step_due_to_missing_edge_evidence" |
| inferred_steps = int(np.clip(inferred_steps, 1, 18)) |
| dimensions = estimate_stair_dimensions( |
| depth, |
| target_mask, |
| filtered_rows, |
| inferred_steps, |
| metric_style_depth=metric_style_depth, |
| ) |
|
|
| target_colors = colors[target_mask] |
| if target_colors.size == 0: |
| base_color = np.array([176, 176, 168], dtype=np.float32) |
| else: |
| base_color = np.median(target_colors.astype(np.float32), axis=0) |
| neutral = np.array([178, 178, 170], dtype=np.float32) |
| base_color = 0.45 * base_color + 0.55 * neutral |
|
|
| total_length = float(dimensions["length"]) |
| total_width = float(dimensions["width"]) |
| total_height = float(dimensions["height"]) |
| side_drop = float(thickness) if thickness is not None else max( |
| float(dimensions["riser_height"]) * 0.35, |
| total_height * max(thickness_ratio * 0.30, 0.025), |
| ) |
| tread_depth = total_length / inferred_steps |
| rise = total_height / inferred_steps |
| x_left = -0.5 * total_width |
| x_right = 0.5 * total_width |
|
|
| vertices: list[np.ndarray] = [] |
| vertex_colors: list[np.ndarray] = [] |
| faces: list[tuple[int, int, int]] = [] |
| front_faces = 0 |
| side_faces = 0 |
|
|
| def scene_to_camera(point: tuple[float, float, float]) -> np.ndarray: |
| x, y, z = point |
| return np.array([x, -z, y], dtype=np.float32) |
|
|
| def add_vertex(point: tuple[float, float, float], color: np.ndarray) -> int: |
| vertex_id = len(vertices) |
| vertices.append(scene_to_camera(point)) |
| vertex_colors.append(np.clip(color, 0, 255).astype(np.uint8)) |
| return vertex_id |
|
|
| def add_quad( |
| a: tuple[float, float, float], |
| b: tuple[float, float, float], |
| c: tuple[float, float, float], |
| d: tuple[float, float, float], |
| color: np.ndarray, |
| *, |
| front: bool, |
| ) -> None: |
| nonlocal front_faces, side_faces |
| ids = [add_vertex(point, color) for point in (a, b, c, d)] |
| faces.append((ids[0], ids[1], ids[2])) |
| faces.append((ids[0], ids[2], ids[3])) |
| if front: |
| front_faces += 2 |
| else: |
| side_faces += 2 |
|
|
| texture_cross_samples = int( |
| np.clip(math.ceil(bbox_width / max(float(stride), 1.0)), 24, 160) |
| ) |
| texture_depth_samples = int( |
| np.clip( |
| math.ceil( |
| bbox_height |
| / max(float(inferred_steps * max(stride, 1)), 1.0) |
| ), |
| 6, |
| 32, |
| ) |
| ) |
|
|
| def add_textured_quad( |
| a: tuple[float, float, float], |
| b: tuple[float, float, float], |
| c: tuple[float, float, float], |
| d: tuple[float, float, float], |
| *, |
| source_y0: int, |
| source_y1: int, |
| shade: float, |
| ) -> None: |
| nonlocal front_faces |
| corner_a = np.asarray(a, dtype=np.float32) |
| corner_b = np.asarray(b, dtype=np.float32) |
| corner_c = np.asarray(c, dtype=np.float32) |
| corner_d = np.asarray(d, dtype=np.float32) |
| grid = np.empty( |
| (texture_depth_samples + 1, texture_cross_samples + 1), |
| dtype=np.int64, |
| ) |
| for depth_index in range(texture_depth_samples + 1): |
| v = depth_index / texture_depth_samples |
| left = (1.0 - v) * corner_a + v * corner_d |
| right = (1.0 - v) * corner_b + v * corner_c |
| source_y = int( |
| np.clip( |
| round((1.0 - v) * source_y0 + v * source_y1), |
| 0, |
| h - 1, |
| ) |
| ) |
| for cross_index in range(texture_cross_samples + 1): |
| u = cross_index / texture_cross_samples |
| point = (1.0 - u) * left + u * right |
| line_bounds = line_x_bounds_on_mask( |
| float(source_y), |
| float(stair_edge_slope), |
| target_mask, |
| ) |
| source_left, source_right = ( |
| line_bounds |
| if line_bounds is not None |
| else (int(mask_xs.min()), int(mask_xs.max())) |
| ) |
| source_x = int( |
| np.clip( |
| round( |
| source_left |
| + u |
| * ( |
| source_right |
| - source_left |
| ) |
| ), |
| 0, |
| w - 1, |
| ) |
| ) |
| color = colors[source_y, source_x].astype(np.float32) * shade |
| grid[depth_index, cross_index] = add_vertex( |
| tuple(float(value) for value in point), |
| color, |
| ) |
| for depth_index in range(texture_depth_samples): |
| for cross_index in range(texture_cross_samples): |
| top_left = int(grid[depth_index, cross_index]) |
| top_right = int(grid[depth_index, cross_index + 1]) |
| bottom_left = int(grid[depth_index + 1, cross_index]) |
| bottom_right = int( |
| grid[depth_index + 1, cross_index + 1] |
| ) |
| faces.append((top_left, top_right, bottom_right)) |
| faces.append((top_left, bottom_right, bottom_left)) |
| front_faces += 2 |
|
|
| row_edges = filtered_rows |
| if step_count <= 0 and row_edges: |
| row_bounds = stair_source_band_bounds( |
| row_edges, |
| int(mask_ys.min()), |
| int(mask_ys.max()), |
| ) |
| else: |
| row_bounds = np.linspace(int(mask_ys.min()), int(mask_ys.max()), inferred_steps + 1).round().astype(int).tolist() |
| if len(row_bounds) < inferred_steps + 1: |
| row_bounds = np.linspace(int(mask_ys.min()), int(mask_ys.max()), inferred_steps + 1).round().astype(int).tolist() |
|
|
| def step_color(index: int, shade: float) -> np.ndarray: |
| start = row_bounds[min(index, len(row_bounds) - 2)] |
| end = row_bounds[min(index + 1, len(row_bounds) - 1)] |
| y0, y1 = sorted((start, end)) |
| band = np.zeros_like(target_mask, dtype=bool) |
| center_x = (w - 1) * 0.5 |
| ys, xs = np.indices(target_mask.shape, dtype=np.float32) |
| aligned_y = ys - float(stair_edge_slope) * (xs - center_x) |
| band = (aligned_y >= y0) & (aligned_y <= y1) & target_mask |
| band_colors = colors[band] |
| color = base_color if band_colors.size == 0 else 0.55 * np.median(band_colors.astype(np.float32), axis=0) + 0.45 * base_color |
| return np.clip(color * shade, 0, 255) |
|
|
| for step_index in range(inferred_steps): |
| y0 = step_index * tread_depth |
| y1 = (step_index + 1) * tread_depth |
| z0 = step_index * rise |
| z1 = (step_index + 1) * rise |
| local_bottom_z = z0 - side_drop |
| |
| |
| source_band_index = inferred_steps - 1 - step_index |
| side_color = step_color(source_band_index, 0.66) |
| source_y_near = row_bounds[ |
| min(source_band_index + 1, len(row_bounds) - 1) |
| ] |
| source_y_far = row_bounds[ |
| min(source_band_index, len(row_bounds) - 2) |
| ] |
|
|
| add_textured_quad( |
| (x_left, y0, z1), |
| (x_right, y0, z1), |
| (x_right, y1, z1), |
| (x_left, y1, z1), |
| source_y0=source_y_near, |
| source_y1=source_y_far, |
| shade=1.02, |
| ) |
| add_textured_quad( |
| (x_left, y0, z0), |
| (x_right, y0, z0), |
| (x_right, y0, z1), |
| (x_left, y0, z1), |
| source_y0=source_y_near, |
| source_y1=source_y_far, |
| shade=0.82, |
| ) |
|
|
| |
| |
| |
| add_quad( |
| (x_left, y0, local_bottom_z), |
| (x_left, y0, z1), |
| (x_left, y1, z1), |
| (x_left, y1, local_bottom_z), |
| side_color, |
| front=False, |
| ) |
| add_quad( |
| (x_right, y0, z1), |
| (x_right, y0, local_bottom_z), |
| (x_right, y1, local_bottom_z), |
| (x_right, y1, z1), |
| side_color, |
| front=False, |
| ) |
| add_quad( |
| (x_left, y0, local_bottom_z), |
| (x_right, y0, local_bottom_z), |
| (x_right, y0, z0), |
| (x_left, y0, z0), |
| side_color * 0.92, |
| front=False, |
| ) |
| add_quad( |
| (x_left, y1, local_bottom_z), |
| (x_right, y1, local_bottom_z), |
| (x_right, y0, local_bottom_z), |
| (x_left, y0, local_bottom_z), |
| side_color * 0.78, |
| front=False, |
| ) |
|
|
| top_y = total_length |
| top_z = total_height |
| back_color = np.clip(base_color * 0.70, 0, 255) |
| final_bottom_z = total_height - rise - side_drop |
| add_quad( |
| (x_right, top_y, final_bottom_z), |
| (x_left, top_y, final_bottom_z), |
| (x_left, top_y, top_z), |
| (x_right, top_y, top_z), |
| back_color, |
| front=False, |
| ) |
|
|
| |
| |
| if metric_style_depth: |
| front_landing = max(0.75, 2.0 * tread_depth) |
| top_landing = max(0.60, 1.5 * tread_depth) |
| else: |
| front_landing = max(2.0 * tread_depth, 0.50 * total_length) |
| top_landing = max(1.5 * tread_depth, 0.35 * total_length) |
| add_textured_quad( |
| (x_left, -front_landing, 0.0), |
| (x_right, -front_landing, 0.0), |
| (x_right, 0.0, 0.0), |
| (x_left, 0.0, 0.0), |
| source_y0=int(mask_ys.max()), |
| source_y1=row_bounds[-1], |
| shade=1.0, |
| ) |
| add_textured_quad( |
| (x_left, top_y, top_z), |
| (x_right, top_y, top_z), |
| (x_right, top_y + top_landing, top_z), |
| (x_left, top_y + top_landing, top_z), |
| source_y0=row_bounds[0], |
| source_y1=int(mask_ys.min()), |
| shade=0.98, |
| ) |
|
|
| metadata = { |
| "front_faces": front_faces, |
| "back_faces": 0, |
| "side_faces": side_faces, |
| "boundary_side_faces": side_faces, |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": side_faces, |
| "stair_sidewall_bottom_faces": 2 * inferred_steps, |
| "solid_thickness": side_drop, |
| "max_depth_jump": None, |
| "stride": stride, |
| "support_mode": "open_world_local_support", |
| "stair_surface_mode": "depth_scaled_staircase_prism", |
| "stair_step_count": inferred_steps, |
| "stair_step_count_source": step_count_source, |
| "stair_inference": stair_inference, |
| "stair_rows": row_bounds, |
| "stair_physical_edge_rows": filtered_rows, |
| "stair_cross_samples": texture_cross_samples, |
| "stair_texture_depth_samples": texture_depth_samples, |
| "texture_policy": "dense per-vertex frozen completed RGB sampled inside each perspective-aware source stair band", |
| "texture_band_order": "near_low_step_uses_bottom_image_band", |
| "texture_coordinate_policy": ( |
| "image_x_maps_to_scene_width_and_image_y_maps_to_tread_depth_or_riser_height" |
| ), |
| "stair_edge_slope": float(stair_edge_slope), |
| "depth_scaled_dimensions": { |
| **dimensions, |
| "side_drop": side_drop, |
| }, |
| "scene_semantics": "open_world_accessibility_surface_patch", |
| "context_landings": { |
| "front_length": front_landing, |
| "top_length": top_landing, |
| "purpose": "scene_context_not_navigability_proof", |
| }, |
| "automatic_passability_claim": False, |
| "mesh_topology_role": "renderable_surface_patch_not_collision_mesh", |
| "watertight_collision_mesh": False, |
| } |
| return np.asarray(vertices, dtype=np.float32), np.asarray(vertex_colors, dtype=np.uint8), np.asarray(faces, dtype=np.int64), metadata |
|
|
|
|
| def build_solid_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| stride: int, |
| max_depth_jump: float | None, |
| thickness: float | None, |
| thickness_ratio: float, |
| support_mode: str, |
| wedge_base_drop_ratio: float, |
| category: str = "unknown", |
| stair_edges_y: list[int] | None = None, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| points = pixels_to_points(depth, fx, fy, cx, cy) |
| valid = target_mask & np.isfinite(depth) & (depth > 0) |
| depth_values = depth[valid] |
| if depth_values.size == 0: |
| raise ValueError("No valid target depth values") |
| depth_span = float(np.percentile(depth_values, 95) - np.percentile(depth_values, 5)) |
| median_depth = float(np.median(depth_values)) |
| mesh_jump = float(max_depth_jump) if max_depth_jump is not None else max(0.03, 0.08 * max(depth_span, 1e-3)) |
| top_points = points[valid] |
| scale = robust_extent_from_points(top_points) |
| category_drop_ratio = { |
| "tactile_paving": 0.035, |
| "curb_cut": 0.060, |
| "ramp": 0.080, |
| "walkway": 0.055, |
| }.get(category, 0.060) |
| category_median_ratio = { |
| "tactile_paving": 0.010, |
| "curb_cut": 0.018, |
| "ramp": 0.024, |
| "walkway": 0.016, |
| }.get(category, 0.018) |
| footprint_drop = category_drop_ratio * max(float(scale["footprint_diag"]), 1e-3) |
| depth_drop = max(thickness_ratio * 0.55, category_drop_ratio) * max(depth_span, 1e-3) |
| median_drop = category_median_ratio * max(median_depth, 1e-3) |
| solid_thickness = float(thickness) if thickness is not None else max(footprint_drop, depth_drop, median_drop) |
| vertical_drop = solid_thickness if support_mode == "vertical_slab" else max(solid_thickness, wedge_base_drop_ratio * max(median_depth, 1e-3)) |
| if top_points.size: |
| top_y_high = float(np.percentile(top_points[:, 1], 98)) |
| else: |
| top_y_high = 0.0 |
| vertical_base_y = top_y_high + vertical_drop |
|
|
| ys = np.arange(0, h, stride) |
| xs = np.arange(0, w, stride) |
| front_id = -np.ones((len(ys), len(xs)), dtype=np.int64) |
| back_id = -np.ones((len(ys), len(xs)), dtype=np.int64) |
| vertices: list[np.ndarray] = [] |
| vertex_colors: list[np.ndarray] = [] |
|
|
| for iy, y in enumerate(ys): |
| for ix, x in enumerate(xs): |
| if not valid[y, x]: |
| continue |
| point = points[y, x].astype(np.float32) |
| if not np.all(np.isfinite(point)) or point[2] <= 0: |
| continue |
| front_id[iy, ix] = len(vertices) |
| vertices.append(point) |
| vertex_colors.append(colors[y, x]) |
| if support_mode == "vertical_slab": |
| back_point = point.copy() |
| back_point[1] = max(vertical_base_y, float(point[1] + 0.35 * vertical_drop)) |
| else: |
| back_depth = float(point[2] + solid_thickness) |
| back_point = point * (back_depth / max(float(point[2]), 1e-6)) |
| back_id[iy, ix] = len(vertices) |
| vertices.append(back_point.astype(np.float32)) |
| shade = 0.62 if support_mode == "vertical_slab" else 0.50 |
| vertex_colors.append(np.clip(colors[y, x].astype(np.float32) * shade, 0, 255).astype(np.uint8)) |
|
|
| front_faces: list[tuple[int, int, int]] = [] |
| back_faces: list[tuple[int, int, int]] = [] |
| edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
|
|
| def add_front(face: tuple[int, int, int]) -> None: |
| front_faces.append(face) |
| for a, b in ((face[0], face[1]), (face[1], face[2]), (face[2], face[0])): |
| edge_counts[tuple(sorted((a, b)))] += 1 |
|
|
| for iy in range(len(ys) - 1): |
| for ix in range(len(xs) - 1): |
| ids = [ |
| front_id[iy, ix], |
| front_id[iy, ix + 1], |
| front_id[iy + 1, ix], |
| front_id[iy + 1, ix + 1], |
| ] |
| bids = [ |
| back_id[iy, ix], |
| back_id[iy, ix + 1], |
| back_id[iy + 1, ix], |
| back_id[iy + 1, ix + 1], |
| ] |
| if min(ids) < 0: |
| continue |
| z = np.array([vertices[i][2] for i in ids], dtype=np.float32) |
| if float(z.max() - z.min()) > mesh_jump: |
| continue |
| add_front((ids[0], ids[2], ids[1])) |
| add_front((ids[1], ids[2], ids[3])) |
| back_faces.append((bids[1], bids[2], bids[0])) |
| back_faces.append((bids[3], bids[2], bids[1])) |
|
|
| boundary_side_faces: list[tuple[int, int, int]] = [] |
| for (a, b), count in edge_counts.items(): |
| if count != 1: |
| continue |
| a_back = a + 1 if a % 2 == 0 else a - 1 |
| b_back = b + 1 if b % 2 == 0 else b - 1 |
| if a_back < 0 or b_back < 0 or a_back >= len(vertices) or b_back >= len(vertices): |
| continue |
| boundary_side_faces.append((a, b, a_back)) |
| boundary_side_faces.append((b, b_back, a_back)) |
|
|
| wedge_support_faces: list[tuple[int, int, int]] = [] |
| stair_sidewall_faces: list[tuple[int, int, int]] = [] |
| stair_sidewall_bottom_faces: list[tuple[int, int, int]] = [] |
|
|
| def append_triangular_side_panel(side_ids: list[int]) -> tuple[int, int] | None: |
| ordered_ids: list[int] = [] |
| for vertex_id in side_ids: |
| if vertex_id < 0: |
| continue |
| if not ordered_ids or ordered_ids[-1] != int(vertex_id): |
| ordered_ids.append(int(vertex_id)) |
| if len(ordered_ids) < 2: |
| return None |
| ordered_ids = sorted( |
| ordered_ids, |
| key=lambda vertex_id: float(vertices[vertex_id][1]), |
| ) |
| side_points = np.asarray([vertices[vertex_id] for vertex_id in ordered_ids], dtype=np.float32) |
| upper_id = ordered_ids[0] |
| lower_id = ordered_ids[-1] |
| upper_point = vertices[upper_id].astype(np.float32) |
| lower_point = vertices[lower_id].astype(np.float32) |
| upper_base_point = upper_point.copy() |
| minimum_drop = max( |
| solid_thickness * 1.5, |
| wedge_base_drop_ratio * max(median_depth, 1e-3), |
| 1e-3, |
| ) |
| upper_base_point[1] = max(float(lower_point[1]), float(upper_point[1] + minimum_drop)) |
| base_color = np.clip( |
| np.asarray([vertex_colors[vertex_id] for vertex_id in ordered_ids], dtype=np.float32).mean(axis=0) * 0.68, |
| 0, |
| 255, |
| ).astype(np.uint8) |
| bottom_ids: list[int] = [] |
| denominator = max(len(ordered_ids) - 1, 1) |
| for index, top_id in enumerate(ordered_ids): |
| if top_id == lower_id: |
| bottom_ids.append(lower_id) |
| continue |
| t = index / denominator |
| bottom_point = ((1.0 - t) * upper_base_point + t * lower_point).astype(np.float32) |
| bottom_color = np.clip( |
| (1.0 - t) * base_color.astype(np.float32) + t * vertex_colors[lower_id].astype(np.float32) * 0.72, |
| 0, |
| 255, |
| ).astype(np.uint8) |
| bottom_id = len(vertices) |
| vertices.append(bottom_point) |
| vertex_colors.append(bottom_color) |
| bottom_ids.append(bottom_id) |
| for top_a, top_b, bottom_a, bottom_b in zip( |
| ordered_ids[:-1], |
| ordered_ids[1:], |
| bottom_ids[:-1], |
| bottom_ids[1:], |
| ): |
| if len({top_a, top_b, bottom_a}) == 3: |
| wedge_support_faces.append((top_a, top_b, bottom_a)) |
| if len({top_b, bottom_b, bottom_a}) == 3: |
| wedge_support_faces.append((top_b, bottom_b, bottom_a)) |
| return bottom_ids[0], lower_id |
|
|
| def collect_side_rows(side: str) -> list[tuple[int, int]]: |
| side_rows: list[tuple[int, int]] = [] |
| for iy, y in enumerate(ys): |
| valid_ix = np.flatnonzero(front_id[iy] >= 0) |
| if valid_ix.size == 0: |
| continue |
| ix = int(valid_ix[0] if side == "left" else valid_ix[-1]) |
| side_rows.append((int(y), int(front_id[iy, ix]))) |
| return side_rows |
|
|
| def thin_side_profile(side_rows: list[tuple[int, int]]) -> list[tuple[int, int]]: |
| if len(side_rows) <= 2: |
| return side_rows |
| edge_rows = sorted( |
| {int(edge) for edge in stair_edges_y or [] if 0 <= int(edge) < h} |
| ) |
| if not edge_rows: |
| return side_rows |
| selected_indices = {0, len(side_rows) - 1} |
| row_values = np.asarray([row for row, _ in side_rows], dtype=np.int32) |
| for edge in edge_rows: |
| for offset in (-stride, 0, stride): |
| index = int(np.argmin(np.abs(row_values - (edge + offset)))) |
| selected_indices.add(index) |
| return [side_rows[index] for index in sorted(selected_indices)] |
|
|
| def append_stair_sidewall( |
| left_rows: list[tuple[int, int]], |
| right_rows: list[tuple[int, int]], |
| ) -> tuple[int, int]: |
| if len(left_rows) < 2 or len(right_rows) < 2: |
| return 0, 0 |
| by_y_left = {row: vertex_id for row, vertex_id in thin_side_profile(left_rows)} |
| by_y_right = {row: vertex_id for row, vertex_id in thin_side_profile(right_rows)} |
| common_rows = sorted(set(by_y_left) & set(by_y_right)) |
| if len(common_rows) < 2: |
| return 0, 0 |
|
|
| left_ids = [by_y_left[row] for row in common_rows] |
| right_ids = [by_y_right[row] for row in common_rows] |
| top_ids = left_ids + right_ids |
| top_points = np.asarray([vertices[vertex_id] for vertex_id in top_ids], dtype=np.float32) |
| base_y = float(top_points[:, 1].max()) + max( |
| solid_thickness * 1.5, |
| wedge_base_drop_ratio * max(median_depth, 1e-3), |
| 1e-3, |
| ) |
|
|
| def make_bottom(top_id: int) -> int: |
| bottom = vertices[top_id].astype(np.float32).copy() |
| bottom[1] = base_y |
| color = np.clip(vertex_colors[top_id].astype(np.float32) * 0.70, 0, 255).astype(np.uint8) |
| bottom_id = len(vertices) |
| vertices.append(bottom) |
| vertex_colors.append(color) |
| return bottom_id |
|
|
| left_bottom_ids = [make_bottom(vertex_id) for vertex_id in left_ids] |
| right_bottom_ids = [make_bottom(vertex_id) for vertex_id in right_ids] |
|
|
| side_count = 0 |
| bottom_count = 0 |
| for top_a, top_b, bottom_a, bottom_b in zip( |
| left_ids[:-1], |
| left_ids[1:], |
| left_bottom_ids[:-1], |
| left_bottom_ids[1:], |
| ): |
| if len({top_a, top_b, bottom_a}) == 3: |
| stair_sidewall_faces.append((top_a, top_b, bottom_a)) |
| side_count += 1 |
| if len({top_b, bottom_b, bottom_a}) == 3: |
| stair_sidewall_faces.append((top_b, bottom_b, bottom_a)) |
| side_count += 1 |
| for top_a, top_b, bottom_a, bottom_b in zip( |
| right_ids[:-1], |
| right_ids[1:], |
| right_bottom_ids[:-1], |
| right_bottom_ids[1:], |
| ): |
| if len({top_a, bottom_a, top_b}) == 3: |
| stair_sidewall_faces.append((top_a, bottom_a, top_b)) |
| side_count += 1 |
| if len({top_b, bottom_a, bottom_b}) == 3: |
| stair_sidewall_faces.append((top_b, bottom_a, bottom_b)) |
| side_count += 1 |
|
|
| for left_a, left_b, right_a, right_b in zip( |
| left_bottom_ids[:-1], |
| left_bottom_ids[1:], |
| right_bottom_ids[:-1], |
| right_bottom_ids[1:], |
| ): |
| if len({left_a, right_a, left_b}) == 3: |
| stair_sidewall_bottom_faces.append((left_a, right_a, left_b)) |
| bottom_count += 1 |
| if len({left_b, right_a, right_b}) == 3: |
| stair_sidewall_bottom_faces.append((left_b, right_a, right_b)) |
| bottom_count += 1 |
| return side_count, bottom_count |
|
|
| if support_mode == "triangular_wedge": |
| left_ids = [vertex_id for _, vertex_id in collect_side_rows("left")] |
| right_ids = [vertex_id for _, vertex_id in collect_side_rows("right")] |
| left_support = append_triangular_side_panel(left_ids) |
| right_support = append_triangular_side_panel(right_ids) |
| if left_support is not None and right_support is not None: |
| left_base, left_lower = left_support |
| right_base, right_lower = right_support |
| if left_lower != right_lower and left_base != right_base: |
| wedge_support_faces.append((left_lower, right_lower, left_base)) |
| wedge_support_faces.append((right_lower, right_base, left_base)) |
| elif support_mode == "stair_sidewall": |
| append_stair_sidewall(collect_side_rows("left"), collect_side_rows("right")) |
|
|
| side_faces = boundary_side_faces + wedge_support_faces + stair_sidewall_faces + stair_sidewall_bottom_faces |
| face_rows = front_faces + back_faces + side_faces |
| faces = ( |
| np.asarray(face_rows, dtype=np.int64).reshape((-1, 3)) |
| if face_rows |
| else np.empty((0, 3), dtype=np.int64) |
| ) |
| metadata = { |
| "front_faces": len(front_faces), |
| "back_faces": len(back_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(boundary_side_faces), |
| "wedge_support_faces": len(wedge_support_faces), |
| "stair_sidewall_faces": len(stair_sidewall_faces), |
| "stair_sidewall_bottom_faces": len(stair_sidewall_bottom_faces), |
| "solid_thickness": solid_thickness, |
| "max_depth_jump": mesh_jump, |
| "stride": stride, |
| "support_mode": support_mode, |
| "vertical_drop": vertical_drop if support_mode == "vertical_slab" else None, |
| "vertical_base_y": vertical_base_y if support_mode == "vertical_slab" else None, |
| "depth_scaled_dimensions": { |
| "source_span_x": float(scale["span_x"]), |
| "source_span_y": float(scale["span_y"]), |
| "source_span_z": float(scale["span_z"]), |
| "footprint_diag": float(scale["footprint_diag"]), |
| "median_depth": median_depth, |
| "depth_span": depth_span, |
| "solid_thickness": solid_thickness, |
| "vertical_drop": vertical_drop if support_mode == "vertical_slab" else None, |
| "category_drop_ratio": category_drop_ratio, |
| "depth_units": "relative_pseudo_depth", |
| }, |
| } |
| return np.asarray(vertices, dtype=np.float32), np.asarray(vertex_colors, dtype=np.uint8), faces, metadata |
|
|
|
|
| def build_perspective_continuous_surface_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| category: str, |
| stride: int, |
| minimum_cell_coverage: float = 0.15, |
| plane: tuple[np.ndarray, float] | None = None, |
| plane_validation_mask: np.ndarray | None = None, |
| texture_support_mask: np.ndarray | None = None, |
| must_retain_mask: np.ndarray | None = None, |
| bridge_evidence_mask: np.ndarray | None = None, |
| maximum_bridge_distance_pixels: float | None = None, |
| maximum_bridge_added_ratio: float = 0.005, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Build a dense, perspective-preserving road-surface patch. |
| |
| Unlike the legacy solid/slab builders, this representation has no inferred |
| back, vertical wall, or wedge whose copied colors could visually stretch. |
| Every front vertex retains an image-space coordinate and samples the frozen |
| completed RGB at that coordinate (or the nearest supported boundary pixel). |
| """ |
|
|
| if stride < 1: |
| raise ValueError("continuous surface stride must be positive") |
| if not 0.0 < minimum_cell_coverage <= 1.0: |
| raise ValueError("minimum_cell_coverage must be in (0, 1]") |
| if colors.shape[:2] != depth.shape or target_mask.shape != depth.shape: |
| raise ValueError("Depth, RGB, and continuous target mask must share a raster") |
| for name, auxiliary in ( |
| ("plane_validation_mask", plane_validation_mask), |
| ("texture_support_mask", texture_support_mask), |
| ("must_retain_mask", must_retain_mask), |
| ("bridge_evidence_mask", bridge_evidence_mask), |
| ): |
| if auxiliary is not None and auxiliary.shape != depth.shape: |
| raise ValueError(f"{name} must share the continuous surface raster") |
|
|
| h, w = depth.shape |
| bridge_maximum_distance_pixels = ( |
| 3.0 * float(stride) * math.sqrt(2.0) |
| if maximum_bridge_distance_pixels is None |
| else float(maximum_bridge_distance_pixels) |
| ) |
| support = target_mask.astype(bool) |
| texture_support = ( |
| support |
| if texture_support_mask is None |
| else texture_support_mask.astype(bool) |
| ) |
| exact_geometry_support_texture = bool( |
| np.array_equal(texture_support, support) |
| ) |
| if not np.any(texture_support): |
| raise ValueError("No source pixels for continuous surface texture") |
| mask_y, mask_x = np.nonzero(support) |
| if mask_x.size == 0: |
| raise ValueError("No target pixels for perspective continuous surface") |
|
|
| x_min, x_max = int(mask_x.min()), int(mask_x.max()) |
| y_min, y_max = int(mask_y.min()), int(mask_y.max()) |
|
|
| def grid_axis(low: int, high: int) -> list[int]: |
| values = list(range(low, high + 1, int(stride))) |
| if not values: |
| values = [low] |
| if values[-1] != high: |
| values.append(high) |
| return values |
|
|
| xs = grid_axis(x_min, x_max) |
| ys = grid_axis(y_min, y_max) |
| if len(xs) < 2 or len(ys) < 2: |
| raise ValueError("Continuous surface bounding box is too small for meshing") |
|
|
| raw_cell_mask = np.zeros((len(ys) - 1, len(xs) - 1), dtype=np.uint8) |
| retained_evidence_cells = np.zeros_like(raw_cell_mask, dtype=bool) |
| bridge_evidence_cells = np.zeros_like(raw_cell_mask, dtype=bool) |
| expanded_bridge_evidence = None |
| if bridge_evidence_mask is not None and np.any(bridge_evidence_mask): |
| evidence_kernel_size = 4 * int(stride) + 1 |
| evidence_kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (evidence_kernel_size, evidence_kernel_size), |
| ) |
| expanded_bridge_evidence = cv2.dilate( |
| bridge_evidence_mask.astype(np.uint8), |
| evidence_kernel, |
| ).astype(bool) |
| for iy, (y0, y1) in enumerate(zip(ys[:-1], ys[1:])): |
| for ix, (x0, x1) in enumerate(zip(xs[:-1], xs[1:])): |
| patch = support[y0 : y1 + 1, x0 : x1 + 1] |
| center_y = min((y0 + y1) // 2, h - 1) |
| center_x = min((x0 + x1) // 2, w - 1) |
| coverage = float(patch.mean()) if patch.size else 0.0 |
| retain_evidence = bool( |
| must_retain_mask is not None |
| and np.any(must_retain_mask[y0 : y1 + 1, x0 : x1 + 1]) |
| ) |
| retained_evidence_cells[iy, ix] = retain_evidence |
| if expanded_bridge_evidence is not None: |
| bridge_evidence_cells[iy, ix] = bool( |
| np.any(expanded_bridge_evidence[y0 : y1 + 1, x0 : x1 + 1]) |
| ) |
| if ( |
| support[center_y, center_x] |
| or coverage >= minimum_cell_coverage |
| or retain_evidence |
| ): |
| raw_cell_mask[iy, ix] = 1 |
|
|
| if not np.any(raw_cell_mask): |
| raise ValueError("Perspective continuous surface produced no occupied cells") |
| cell_kernel = np.ones((3, 3), dtype=np.uint8) |
| closed_cell_candidate = cv2.morphologyEx( |
| raw_cell_mask, |
| cv2.MORPH_CLOSE, |
| cell_kernel, |
| ).astype(bool) |
| cell_close_added = closed_cell_candidate & ~raw_cell_mask.astype(bool) |
| cell_close_limit = max(1, int(math.floor(0.02 * int(raw_cell_mask.sum())))) |
| cell_close_gate_passed = int(cell_close_added.sum()) <= cell_close_limit |
| cell_mask = ( |
| closed_cell_candidate |
| if cell_close_gate_passed |
| else raw_cell_mask.astype(bool) |
| ) |
| cell_mask_after_close = cell_mask.copy() |
|
|
| |
| |
| |
| component_count, component_labels, component_stats, _ = ( |
| cv2.connectedComponentsWithStats( |
| cell_mask.astype(np.uint8), |
| connectivity=4, |
| ) |
| ) |
| occupied_cells = int(cell_mask.sum()) |
| minimum_component_cells = max(2, int(math.ceil(0.001 * occupied_cells))) |
| main_component_label = ( |
| 1 |
| + int(np.argmax(component_stats[1:, cv2.CC_STAT_AREA])) |
| if component_count > 1 |
| else 0 |
| ) |
| kept_labels: set[int] = set() |
| for label in range(1, component_count): |
| component = component_labels == label |
| area = int(component_stats[label, cv2.CC_STAT_AREA]) |
| if ( |
| label == main_component_label |
| or area >= minimum_component_cells |
| or np.any(component & retained_evidence_cells) |
| ): |
| kept_labels.add(label) |
| filtered_cell_mask = np.isin(component_labels, list(kept_labels)) |
| discarded_tiny_cells = int(cell_mask.sum() - filtered_cell_mask.sum()) |
| cell_mask = filtered_cell_mask |
|
|
| bridge_limit_cells = max( |
| 1, |
| int(math.floor(maximum_bridge_added_ratio * max(int(cell_mask.sum()), 1))), |
| ) |
| bridge_added_cells = 0 |
| bridge_count = 0 |
| bridge_rejected_count = 0 |
| bridge_distances_pixels: list[float] = [] |
| if main_component_label in kept_labels: |
| anchor = component_labels == main_component_label |
| retained_components = [ |
| component_labels == label |
| for label in sorted(kept_labels) |
| if label != main_component_label |
| and np.any((component_labels == label) & retained_evidence_cells) |
| ] |
| for component in retained_components: |
| if np.any(component & anchor): |
| continue |
| distance, nearest_label_map = cv2.distanceTransformWithLabels( |
| (~anchor).astype(np.uint8), |
| cv2.DIST_L2, |
| 5, |
| labelType=cv2.DIST_LABEL_PIXEL, |
| ) |
| component_y, component_x = np.nonzero(component) |
| nearest_index = int(np.argmin(distance[component_y, component_x])) |
| source_y = int(component_y[nearest_index]) |
| source_x = int(component_x[nearest_index]) |
| distance_pixels = float(distance[source_y, source_x]) * float(stride) |
| nearest_label = int(nearest_label_map[source_y, source_x]) |
| anchor_y, anchor_x = np.nonzero(anchor) |
| anchor_labels = nearest_label_map[anchor_y, anchor_x] |
| matching = np.flatnonzero(anchor_labels == nearest_label) |
| if matching.size == 0: |
| bridge_rejected_count += 1 |
| continue |
| target_index = int(matching[0]) |
| target_y = int(anchor_y[target_index]) |
| target_x = int(anchor_x[target_index]) |
| candidate = np.zeros_like(cell_mask, dtype=np.uint8) |
| cv2.line( |
| candidate, |
| (source_x, source_y), |
| (target_x, target_y), |
| color=1, |
| thickness=2, |
| lineType=cv2.LINE_8, |
| ) |
| candidate = candidate.astype(bool) |
| added_candidate = candidate & ~cell_mask |
| evidence_overlap = ( |
| float(np.count_nonzero(candidate & bridge_evidence_cells)) |
| / max(int(candidate.sum()), 1) |
| if expanded_bridge_evidence is not None |
| else 0.0 |
| ) |
| if ( |
| distance_pixels > bridge_maximum_distance_pixels |
| or evidence_overlap < 0.25 |
| or bridge_added_cells + int(added_candidate.sum()) |
| > bridge_limit_cells |
| ): |
| bridge_rejected_count += 1 |
| continue |
| cell_mask |= candidate |
| anchor |= component | candidate |
| added = int(added_candidate.sum()) |
| bridge_added_cells += added |
| bridge_count += 1 |
| bridge_distances_pixels.append(distance_pixels) |
|
|
| |
| |
| |
| |
| detached_rejected_cells = int(np.count_nonzero(cell_mask & ~anchor)) |
| detached_rejected_hidden_cells = int( |
| np.count_nonzero(cell_mask & ~anchor & retained_evidence_cells) |
| ) |
| cell_mask &= anchor |
| else: |
| detached_rejected_cells = 0 |
| detached_rejected_hidden_cells = 0 |
|
|
| |
| |
| |
| |
| |
| cell_mask_before_final_hole_fill = cell_mask.copy() |
| cell_mask = fill_enclosed_mask_holes(cell_mask) |
| final_enclosed_hole_added_cells = int( |
| np.count_nonzero(cell_mask & ~cell_mask_before_final_hole_fill) |
| ) |
|
|
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| camera_points = pixels_to_points(depth, fx, fy, cx, cy) |
| validation_support = ( |
| support |
| if plane_validation_mask is None |
| else plane_validation_mask.astype(bool) |
| ) |
| fit_mask = validation_support & np.isfinite(depth) & (depth > 0) |
| if not np.any(fit_mask): |
| fit_mask = support & np.isfinite(depth) & (depth > 0) |
| surface_plane = plane if plane is not None else fit_plane(camera_points[fit_mask]) |
| if surface_plane is None: |
| raise ValueError("Could not fit the continuous surface plane") |
| surface_depth = plane_depth_for_pixels((h, w), surface_plane, fx, fy, cx, cy) |
| observed = depth[fit_mask] |
| depth_low, depth_high = np.percentile(observed, [1, 99]) |
| allowed_low = max(float(depth_low) * 0.5, 1e-4) |
| allowed_high = max(float(depth_high) * 1.5, allowed_low + 1e-4) |
|
|
| referenced_nodes = np.zeros((len(ys), len(xs)), dtype=bool) |
| for iy, ix in np.argwhere(cell_mask): |
| referenced_nodes[iy, ix] = True |
| referenced_nodes[iy + 1, ix] = True |
| referenced_nodes[iy, ix + 1] = True |
| referenced_nodes[iy + 1, ix + 1] = True |
|
|
| node_grid_y, node_grid_x = np.nonzero(referenced_nodes) |
| node_image_y = np.asarray([ys[index] for index in node_grid_y], dtype=np.int32) |
| node_image_x = np.asarray([xs[index] for index in node_grid_x], dtype=np.int32) |
| node_plane_depth = surface_depth[node_image_y, node_image_x] |
| finite_positive = np.isfinite(node_plane_depth) & (node_plane_depth > 0) |
| if not np.all(finite_positive): |
| raise ValueError("Continuous surface plane is invalid inside the mesh ROI") |
| plausible_depth = ( |
| (node_plane_depth >= allowed_low) |
| & (node_plane_depth <= allowed_high) |
| ) |
| plane_plausible_fraction = float(np.mean(plausible_depth)) |
| if plane_plausible_fraction < 0.95: |
| raise ValueError( |
| "Continuous surface plane failed the whole-ROI plausibility gate: " |
| f"{plane_plausible_fraction:.6f}" |
| ) |
|
|
| nearest_distance, nearest_labels = cv2.distanceTransformWithLabels( |
| (~texture_support).astype(np.uint8), |
| cv2.DIST_L2, |
| 5, |
| labelType=cv2.DIST_LABEL_PIXEL, |
| ) |
| max_label = int(nearest_labels.max()) |
| nearest_y = np.zeros(max_label + 1, dtype=np.int32) |
| nearest_x = np.zeros(max_label + 1, dtype=np.int32) |
| support_y, support_x = np.nonzero(texture_support) |
| support_labels = nearest_labels[support_y, support_x] |
| nearest_y[support_labels] = support_y |
| nearest_x[support_labels] = support_x |
|
|
| vertex_ids = -np.ones((len(ys), len(xs)), dtype=np.int64) |
| vertices: list[np.ndarray] = [] |
| vertex_colors: list[np.ndarray] = [] |
| boundary_clamped_vertices = 0 |
| texture_clamp_distances_pixels: list[float] = [] |
| depth_clipped_vertices = 0 |
| for iy, ix in np.argwhere(referenced_nodes): |
| y, x = int(ys[iy]), int(xs[ix]) |
| z = float(surface_depth[y, x]) |
| point = np.asarray( |
| [ |
| (float(x) - cx) * z / fx, |
| (float(y) - cy) * z / fy, |
| z, |
| ], |
| dtype=np.float32, |
| ) |
| sample_y, sample_x = y, x |
| if not texture_support[y, x]: |
| label = int(nearest_labels[y, x]) |
| if 0 < label <= max_label: |
| sample_y = int(nearest_y[label]) |
| sample_x = int(nearest_x[label]) |
| boundary_clamped_vertices += 1 |
| texture_clamp_distances_pixels.append( |
| float(nearest_distance[y, x]) |
| ) |
| vertex_ids[iy, ix] = len(vertices) |
| vertices.append(point) |
| vertex_colors.append(colors[sample_y, sample_x].astype(np.uint8)) |
|
|
| front_faces: list[tuple[int, int, int]] = [] |
| for iy, ix in np.argwhere(cell_mask): |
| ids = ( |
| int(vertex_ids[iy, ix]), |
| int(vertex_ids[iy, ix + 1]), |
| int(vertex_ids[iy + 1, ix]), |
| int(vertex_ids[iy + 1, ix + 1]), |
| ) |
| if min(ids) < 0: |
| raise RuntimeError("Internal continuous-surface grid indexing error") |
| front_faces.append((ids[0], ids[2], ids[1])) |
| front_faces.append((ids[1], ids[2], ids[3])) |
|
|
| vertex_array = np.asarray(vertices, dtype=np.float32) |
| color_array = np.asarray(vertex_colors, dtype=np.uint8) |
| face_array = np.asarray(front_faces, dtype=np.int64).reshape((-1, 3)) |
| scale = robust_extent_from_points(vertex_array) |
| metadata = { |
| "front_faces": len(front_faces), |
| "back_faces": 0, |
| "side_faces": 0, |
| "boundary_side_faces": 0, |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": 0, |
| "stair_sidewall_bottom_faces": 0, |
| "solid_thickness": 0.0, |
| "max_depth_jump": None, |
| "stride": int(stride), |
| "support_mode": "open_world_surface_only", |
| "vertical_drop": None, |
| "vertical_base_y": None, |
| "surface_model": "perspective_planar_accessibility_surface_patch", |
| "category": category, |
| "stair_surface_mode": "not_stairs", |
| "stair_rows": [], |
| "stair_cross_samples": None, |
| "stair_edge_slope": None, |
| "stair_step_count": None, |
| "texture_policy": ( |
| ( |
| "dense frozen completed RGB sampled exactly at original image " |
| "coordinates for reviewed and evidence-gated repair pixels; " |
| "grid-boundary-only nodes use the nearest geometry texel; no " |
| "copied side or back texture" |
| ) |
| if exact_geometry_support_texture |
| else ( |
| "dense frozen completed RGB sampled at original image coordinates; " |
| "texture-repair nodes use the nearest explicit texture-support " |
| "pixel; no copied side or back texture" |
| ) |
| ), |
| "texture_coordinate_policy": "perspective_image_xy_reprojection_without_bbox_warp", |
| "boundary_clamped_vertices": int(boundary_clamped_vertices), |
| "texture_source_support_pixels": int(texture_support.sum()), |
| "exact_geometry_support_texture": exact_geometry_support_texture, |
| "texture_completion_vertices": int(boundary_clamped_vertices), |
| "texture_completion_max_distance_pixels": ( |
| max(texture_clamp_distances_pixels) |
| if texture_clamp_distances_pixels |
| else 0.0 |
| ), |
| "texture_completion_mean_distance_pixels": ( |
| float(np.mean(texture_clamp_distances_pixels)) |
| if texture_clamp_distances_pixels |
| else 0.0 |
| ), |
| "texture_completion_policy": ( |
| "nearest_reviewed_target_pixel_without_blur_or_bbox_warp" |
| ), |
| "depth_clipped_vertices": int(depth_clipped_vertices), |
| "plane_plausible_fraction": plane_plausible_fraction, |
| "plane_depth_policy": ( |
| "whole_roi_validated_single_plane_without_per_vertex_clipping" |
| ), |
| "surface_cells_before_topology_close": int(raw_cell_mask.sum()), |
| "surface_cells": int(cell_mask.sum()), |
| "topology_close_added_cells": int( |
| (cell_mask_after_close & ~raw_cell_mask.astype(bool)).sum() |
| ), |
| "topology_close_candidate_added_cells": int(cell_close_added.sum()), |
| "topology_close_added_limit_cells": int(cell_close_limit), |
| "topology_close_gate_passed": cell_close_gate_passed, |
| "minimum_component_cells": int(minimum_component_cells), |
| "discarded_tiny_cells_without_hidden_evidence": int( |
| discarded_tiny_cells |
| ), |
| "hidden_evidence_cells": int(retained_evidence_cells.sum()), |
| "hidden_evidence_covered_cells": int( |
| np.count_nonzero(cell_mask & retained_evidence_cells) |
| ), |
| "hidden_cell_coverage_fraction": ( |
| float( |
| np.count_nonzero(cell_mask & retained_evidence_cells) |
| / max(int(retained_evidence_cells.sum()), 1) |
| ) |
| if must_retain_mask is not None |
| else None |
| ), |
| "bridge_policy": ( |
| "hidden_component_shortest_cell_path_with_full_obstacle_evidence" |
| ), |
| "bridge_count": int(bridge_count), |
| "bridge_rejected_count": int(bridge_rejected_count), |
| "bridge_added_cells": int(bridge_added_cells), |
| "bridge_added_limit_cells": int(bridge_limit_cells), |
| "bridge_maximum_distance_pixels": bridge_maximum_distance_pixels, |
| "bridge_distances_pixels": bridge_distances_pixels, |
| "bridge_evidence_mask_used": expanded_bridge_evidence is not None, |
| "detached_rejected_cells": int(detached_rejected_cells), |
| "detached_rejected_hidden_cells": int( |
| detached_rejected_hidden_cells |
| ), |
| "detached_component_policy": ( |
| "exclude_when_not_connected_to_primary_surface_by_validated_obstacle_bridge" |
| ), |
| "final_enclosed_hole_added_cells": int( |
| final_enclosed_hole_added_cells |
| ), |
| "final_cell_holes_filled": True, |
| "cell_components_before": mask_component_count( |
| raw_cell_mask.astype(bool), |
| connectivity=4, |
| ), |
| "cell_components_after": mask_component_count( |
| cell_mask, |
| connectivity=4, |
| ), |
| "minimum_cell_coverage": float(minimum_cell_coverage), |
| "source_pixel_bounds": { |
| "x_min": x_min, |
| "x_max": x_max, |
| "y_min": y_min, |
| "y_max": y_max, |
| }, |
| "plane_normal": [float(value) for value in surface_plane[0]], |
| "plane_d": float(surface_plane[1]), |
| "plane_source": ( |
| "validated_visible_target_plane" |
| if plane is not None |
| else "robust_fit_on_repaired_support" |
| ), |
| "plane_validation_pixels": int(fit_mask.sum()), |
| "depth_scaled_dimensions": { |
| "source_span_x": float(scale["span_x"]), |
| "source_span_y": float(scale["span_y"]), |
| "source_span_z": float(scale["span_z"]), |
| "footprint_diag": float(scale["footprint_diag"]), |
| "median_depth": float(scale["median_z"]), |
| "depth_units": "metric_style_depth_or_input_pseudo_depth_not_calibrated_truth", |
| "calibrated_metric_truth": False, |
| }, |
| "scene_semantics": "open_world_accessibility_surface_patch", |
| "automatic_passability_claim": False, |
| "mesh_topology_role": "renderable_surface_patch_not_collision_mesh", |
| "watertight_collision_mesh": False, |
| } |
| return vertex_array, color_array, face_array, metadata |
|
|
|
|
| def build_hybrid_instance_footprint_mask( |
| target_mask: np.ndarray, |
| *, |
| category: str, |
| stride: int, |
| visible_mask: np.ndarray | None = None, |
| hidden_mask: np.ndarray | None = None, |
| ) -> tuple[np.ndarray, dict[str, Any]]: |
| """Return one smooth, hole-free instance footprint without using a hull. |
| |
| The reviewed amodal target remains the only semantic source. The component |
| with the most visible-target evidence is retained (which rejects detached |
| hidden islands such as a fence/car false positive), then a bounded |
| close/open removes narrow obstacle cuts and render-only hairs. A simplified |
| external contour removes raster stair-steps while preserving broad Y/T path |
| branches. Convex hulls, row spans, bounding boxes, and generation masks are |
| deliberately excluded. |
| """ |
|
|
| if target_mask.ndim != 2 or not np.any(target_mask): |
| raise ValueError("Hybrid instance footprint needs a non-empty 2D mask") |
| if stride < 1: |
| raise ValueError("Hybrid instance footprint stride must be positive") |
| for name, mask in (("visible_mask", visible_mask), ("hidden_mask", hidden_mask)): |
| if mask is not None and mask.shape != target_mask.shape: |
| raise ValueError(f"{name} must share the target raster") |
|
|
| source = fill_enclosed_mask_holes(target_mask.astype(bool)) |
| component_count, labels, stats, _ = cv2.connectedComponentsWithStats( |
| source.astype(np.uint8), |
| connectivity=8, |
| ) |
| if component_count <= 1: |
| raise ValueError("Hybrid instance footprint has no connected target") |
|
|
| visible = ( |
| visible_mask.astype(bool) |
| if visible_mask is not None |
| else np.zeros_like(source) |
| ) |
| component_rows: list[tuple[int, int, int]] = [] |
| for label in range(1, component_count): |
| component = labels == label |
| component_rows.append( |
| ( |
| int(np.count_nonzero(component & visible)), |
| int(stats[label, cv2.CC_STAT_AREA]), |
| label, |
| ) |
| ) |
| |
| |
| |
| _, _, retained_label = max(component_rows) |
| authoritative = fill_enclosed_mask_holes(labels == retained_label) |
| source_pixels = int(authoritative.sum()) |
| source_visible = int(np.count_nonzero(authoritative & visible)) |
| hidden = ( |
| hidden_mask.astype(bool) |
| if hidden_mask is not None |
| else np.zeros_like(authoritative) |
| ) |
| source_hidden = int(np.count_nonzero(authoritative & hidden)) |
|
|
| category_multiplier = { |
| "ramp": 8, |
| "walkway": 10, |
| "curb_cut": 12, |
| "tactile_paving": 12, |
| }.get(category, 8) |
| requested_kernel = category_multiplier * int(stride) + 1 |
| if requested_kernel % 2 == 0: |
| requested_kernel += 1 |
| kernel_candidates = [] |
| for value in ( |
| requested_kernel, |
| max(4 * int(stride) + 1, requested_kernel - 2 * int(stride)), |
| 4 * int(stride) + 1, |
| ): |
| if value % 2 == 0: |
| value += 1 |
| if value not in kernel_candidates: |
| kernel_candidates.append(value) |
|
|
| selected = authoritative.copy() |
| selected_kernel = 1 |
| selected_gate = "authoritative_component_fallback" |
| selected_metrics: dict[str, float] = {} |
| for kernel_size in kernel_candidates: |
| kernel = cv2.getStructuringElement( |
| cv2.MORPH_ELLIPSE, |
| (kernel_size, kernel_size), |
| ) |
| candidate = cv2.morphologyEx( |
| authoritative.astype(np.uint8), |
| cv2.MORPH_CLOSE, |
| kernel, |
| ) |
| candidate = cv2.morphologyEx(candidate, cv2.MORPH_OPEN, kernel) |
| contours, _ = cv2.findContours( |
| candidate, |
| cv2.RETR_EXTERNAL, |
| cv2.CHAIN_APPROX_NONE, |
| ) |
| if not contours: |
| continue |
| largest = max(contours, key=cv2.contourArea) |
| candidate = np.zeros_like(candidate) |
| cv2.drawContours(candidate, [largest], -1, color=1, thickness=-1) |
| candidate_bool = fill_enclosed_mask_holes(candidate.astype(bool)) |
| visible_retention = ( |
| float(np.count_nonzero(candidate_bool & authoritative & visible)) |
| / max(source_visible, 1) |
| if source_visible > 0 |
| else 1.0 |
| ) |
| hidden_retention = ( |
| float(np.count_nonzero(candidate_bool & authoritative & hidden)) |
| / max(source_hidden, 1) |
| if source_hidden > 0 |
| else 1.0 |
| ) |
| authoritative_retention = float( |
| np.count_nonzero(candidate_bool & authoritative) |
| ) / max(source_pixels, 1) |
| growth_ratio = float( |
| np.count_nonzero(candidate_bool & ~authoritative) |
| ) / max(source_pixels, 1) |
| if ( |
| visible_retention >= 0.94 |
| and hidden_retention >= 0.90 |
| and authoritative_retention >= 0.84 |
| and growth_ratio <= 0.20 |
| ): |
| selected = candidate_bool |
| selected_kernel = kernel_size |
| selected_gate = "bounded_close_open_retention_gate_passed" |
| selected_metrics = { |
| "visible_retention_fraction": visible_retention, |
| "hidden_retention_fraction": hidden_retention, |
| "authoritative_retention_fraction": authoritative_retention, |
| "growth_fraction": growth_ratio, |
| } |
| break |
|
|
| |
| |
| |
| contours, _ = cv2.findContours( |
| selected.astype(np.uint8), |
| cv2.RETR_EXTERNAL, |
| cv2.CHAIN_APPROX_NONE, |
| ) |
| contour = max(contours, key=cv2.contourArea) |
| raw_contour_vertices = int(len(contour)) |
| perimeter = float(cv2.arcLength(contour, True)) |
| epsilon = max(1.0, 0.0015 * perimeter) |
| simplified_contour = cv2.approxPolyDP(contour, epsilon, True) |
| simplified = np.zeros_like(selected, dtype=np.uint8) |
| cv2.drawContours( |
| simplified, |
| [simplified_contour], |
| -1, |
| color=1, |
| thickness=-1, |
| ) |
| simplified_bool = fill_enclosed_mask_holes(simplified.astype(bool)) |
| simplify_visible_retention = ( |
| float(np.count_nonzero(simplified_bool & authoritative & visible)) |
| / max(source_visible, 1) |
| if source_visible > 0 |
| else 1.0 |
| ) |
| simplify_hidden_retention = ( |
| float(np.count_nonzero(simplified_bool & authoritative & hidden)) |
| / max(source_hidden, 1) |
| if source_hidden > 0 |
| else 1.0 |
| ) |
| if simplify_visible_retention >= 0.94 and simplify_hidden_retention >= 0.90: |
| selected = simplified_bool |
| contour_simplification_used = True |
| else: |
| contour_simplification_used = False |
|
|
| return selected.astype(bool), { |
| "role": "render_only_instance_footprint_not_annotation", |
| "policy": ( |
| "visible_anchored_component_plus_bounded_close_open_and_" |
| "external_contour_simplification" |
| ), |
| "generation_mask_used": False, |
| "convex_hull_used": False, |
| "row_span_used": False, |
| "bbox_footprint_used": False, |
| "annotation_mask_modified": False, |
| "source_components": int(component_count - 1), |
| "retained_component_label": int(retained_label), |
| "source_pixels": int(target_mask.sum()), |
| "authoritative_component_pixels": source_pixels, |
| "footprint_pixels": int(selected.sum()), |
| "requested_smoothing_kernel": int(requested_kernel), |
| "selected_smoothing_kernel": int(selected_kernel), |
| "smoothing_gate": selected_gate, |
| **selected_metrics, |
| "raw_contour_vertices": raw_contour_vertices, |
| "simplified_contour_vertices": int(len(simplified_contour)), |
| "contour_simplification_used": contour_simplification_used, |
| "simplified_visible_retention_fraction": simplify_visible_retention, |
| "simplified_hidden_retention_fraction": simplify_hidden_retention, |
| "final_components": mask_component_count(selected), |
| "final_holes_filled": True, |
| } |
|
|
|
|
| def _contiguous_mask_runs(row: np.ndarray) -> list[tuple[int, int]]: |
| """Return inclusive true runs for one binary mask row.""" |
|
|
| columns = np.flatnonzero(row) |
| if columns.size == 0: |
| return [] |
| split_indices = np.flatnonzero(np.diff(columns) > 1) + 1 |
| return [ |
| (int(run[0]), int(run[-1])) |
| for run in np.split(columns, split_indices) |
| if run.size > 0 |
| ] |
|
|
|
|
| def _robust_corridor_boundary( |
| rows: np.ndarray, |
| values: np.ndarray, |
| *, |
| side: str, |
| ) -> tuple[np.ndarray, float]: |
| """Fit a smooth boundary while treating obstacle bites as outliers.""" |
|
|
| if side not in {"left", "right"}: |
| raise ValueError("Corridor boundary side must be left or right") |
| if len(rows) != len(values) or len(rows) < 2: |
| raise ValueError("Corridor boundary needs at least two aligned samples") |
| normalized_rows = (rows - float(rows.min())) / max( |
| float(np.ptp(rows)), |
| 1.0, |
| ) |
| degree = min(2, len(rows) - 1) |
| retained = np.ones(len(rows), dtype=bool) |
| coefficients = np.polyfit(normalized_rows, values, degree) |
| for _ in range(4): |
| retained_degree = min(degree, max(int(retained.sum()) - 1, 1)) |
| coefficients = np.polyfit( |
| normalized_rows[retained], |
| values[retained], |
| retained_degree, |
| ) |
| prediction = np.polyval(coefficients, normalized_rows) |
| residual = values - prediction |
| median = float(np.median(residual[retained])) |
| mad = float( |
| np.median(np.abs(residual[retained] - median)) |
| ) + 1.0 |
| if side == "left": |
| |
| |
| candidate = ( |
| (residual < median + 2.2 * mad) |
| & (residual > median - 4.0 * mad) |
| ) |
| else: |
| |
| candidate = ( |
| (residual > median - 2.2 * mad) |
| & (residual < median + 4.0 * mad) |
| ) |
| if int(candidate.sum()) < max(8, degree + 2): |
| break |
| retained = candidate |
| return ( |
| np.polyval(coefficients, normalized_rows).astype(np.float64), |
| float(retained.mean()), |
| ) |
|
|
|
|
| def build_obstacle_centered_corridor_profile( |
| footprint_mask: np.ndarray, |
| *, |
| stride: int, |
| visible_mask: np.ndarray | None = None, |
| hidden_mask: np.ndarray | None = None, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Extract one smooth local channel through the completed obstacle zone. |
| |
| A full perspective road mask often reaches the image horizon and produces |
| extremely long triangular shards after monocular plane reprojection. For |
| amodal review the useful object is instead the local, obstacle-free travel |
| channel. This routine follows the mask run that crosses the hidden-region |
| centre, keeps enough visible context on both sides, and robustly removes |
| boundary notches caused by foreground silhouettes. It never edits the |
| reviewed masks or the frozen completed RGB. |
| """ |
|
|
| if footprint_mask.ndim != 2 or not np.any(footprint_mask): |
| raise ValueError("Obstacle-centred corridor needs a non-empty footprint") |
| if stride < 1: |
| raise ValueError("Obstacle-centred corridor stride must be positive") |
| for name, mask in (("visible_mask", visible_mask), ("hidden_mask", hidden_mask)): |
| if mask is not None and mask.shape != footprint_mask.shape: |
| raise ValueError(f"{name} must share the corridor footprint raster") |
|
|
| footprint = footprint_mask.astype(bool) |
| visible = ( |
| visible_mask.astype(bool) & footprint |
| if visible_mask is not None |
| else footprint.copy() |
| ) |
| if not np.any(visible): |
| visible = footprint.copy() |
| hidden = ( |
| hidden_mask.astype(bool) & footprint |
| if hidden_mask is not None |
| else np.zeros_like(footprint) |
| ) |
| hidden_source = ( |
| hidden |
| if np.any(hidden) |
| else ( |
| hidden_mask.astype(bool) |
| if hidden_mask is not None and np.any(hidden_mask) |
| else np.zeros_like(footprint) |
| ) |
| ) |
|
|
| visible_y, visible_x = np.nonzero(visible) |
| visible_y_min = int(visible_y.min()) |
| visible_y_max = int(visible_y.max()) |
| visible_span = visible_y_max - visible_y_min + 1 |
| hidden_y, hidden_x = np.nonzero(hidden_source) |
| obstacle_centered = bool(hidden_x.size > 0) |
| if obstacle_centered: |
| anchor_x = float(np.median(hidden_x)) |
| anchor_y = float(np.median(hidden_y)) |
| hidden_y_min = int(hidden_y.min()) |
| hidden_y_max = int(hidden_y.max()) |
| hidden_span = hidden_y_max - hidden_y_min + 1 |
| row_min = max( |
| visible_y_min, |
| int(round(hidden_y_min - 0.25 * hidden_span)), |
| ) |
| row_max = min( |
| visible_y_max, |
| int(round(hidden_y_max + 0.80 * hidden_span)), |
| ) |
| minimum_span = max( |
| int(round(0.38 * visible_span)), |
| 8 * int(stride), |
| ) |
| current_span = row_max - row_min + 1 |
| if current_span < minimum_span: |
| missing = minimum_span - current_span |
| row_min = max(visible_y_min, row_min - missing // 3) |
| current_span = row_max - row_min + 1 |
| row_max = min( |
| visible_y_max, |
| row_max + (minimum_span - current_span), |
| ) |
| current_span = row_max - row_min + 1 |
| if current_span < minimum_span: |
| row_min = max( |
| visible_y_min, |
| row_min - (minimum_span - current_span), |
| ) |
| else: |
| anchor_x = float(np.median(visible_x)) |
| anchor_y = float(np.median(visible_y)) |
| hidden_y_min = None |
| hidden_y_max = None |
| row_min = visible_y_min |
| row_max = visible_y_max |
|
|
| sampled_rows = np.arange( |
| row_min, |
| row_max + 1, |
| int(stride), |
| dtype=np.int32, |
| ) |
| if sampled_rows.size == 0: |
| sampled_rows = np.asarray([row_min, row_max], dtype=np.int32) |
| elif int(sampled_rows[-1]) != row_max: |
| sampled_rows = np.concatenate( |
| (sampled_rows, np.asarray([row_max], dtype=np.int32)) |
| ) |
| if sampled_rows.size < 2: |
| raise ValueError("Obstacle-centred corridor is too short") |
|
|
| row_runs = [ |
| _contiguous_mask_runs(footprint[int(row)]) |
| for row in sampled_rows |
| ] |
| available = [index for index, runs in enumerate(row_runs) if runs] |
| if not available: |
| raise ValueError("Obstacle-centred corridor has no occupied row") |
| anchor_index = min( |
| available, |
| key=lambda index: abs(float(sampled_rows[index]) - anchor_y), |
| ) |
|
|
| selected_runs: list[tuple[int, int] | None] = [None] * len(sampled_rows) |
|
|
| def choose_run( |
| candidates: list[tuple[int, int]], |
| reference_x: float, |
| ) -> tuple[int, int]: |
| def score(run: tuple[int, int]) -> tuple[float, int]: |
| left, right = run |
| distance = ( |
| 0.0 |
| if left <= reference_x <= right |
| else min(abs(reference_x - left), abs(reference_x - right)) |
| ) |
| return distance, -(right - left + 1) |
|
|
| return min(candidates, key=score) |
|
|
| selected_runs[anchor_index] = choose_run( |
| row_runs[anchor_index], |
| anchor_x, |
| ) |
| for direction in (1, -1): |
| reference_x = float(np.clip(anchor_x, *selected_runs[anchor_index])) |
| index = anchor_index + direction |
| while 0 <= index < len(sampled_rows): |
| if row_runs[index]: |
| chosen = choose_run(row_runs[index], reference_x) |
| selected_runs[index] = chosen |
| reference_x = float(np.clip(reference_x, *chosen)) |
| index += direction |
|
|
| valid_indices = np.asarray( |
| [index for index, run in enumerate(selected_runs) if run is not None], |
| dtype=np.int32, |
| ) |
| all_indices = np.arange(len(sampled_rows), dtype=np.float64) |
| raw_left = np.interp( |
| all_indices, |
| valid_indices.astype(np.float64), |
| np.asarray( |
| [selected_runs[index][0] for index in valid_indices], |
| dtype=np.float64, |
| ), |
| ) |
| raw_right = np.interp( |
| all_indices, |
| valid_indices.astype(np.float64), |
| np.asarray( |
| [selected_runs[index][1] for index in valid_indices], |
| dtype=np.float64, |
| ), |
| ) |
| left, left_retention = _robust_corridor_boundary( |
| sampled_rows.astype(np.float64), |
| raw_left, |
| side="left", |
| ) |
| right, right_retention = _robust_corridor_boundary( |
| sampled_rows.astype(np.float64), |
| raw_right, |
| side="right", |
| ) |
| raw_width = np.maximum(raw_right - raw_left + 1.0, 1.0) |
| minimum_width = max( |
| 6.0 * float(stride), |
| 0.45 * float(np.percentile(raw_width, 20.0)), |
| ) |
| center = 0.5 * (left + right) |
| width = np.maximum(right - left + 1.0, minimum_width) |
| left = center - 0.5 * (width - 1.0) |
| right = center + 0.5 * (width - 1.0) |
| image_width = footprint.shape[1] |
| left = np.clip(left, 0.0, float(image_width - 2)) |
| right = np.clip(right, left + 1.0, float(image_width - 1)) |
|
|
| |
| |
| |
| |
| |
| profile_width_before_crop = right - left + 1.0 |
| anchor_profile_index = int( |
| np.argmin(np.abs(sampled_rows.astype(np.float64) - anchor_y)) |
| ) |
| if obstacle_centered and hidden_y_min is not None and hidden_y_max is not None: |
| reference_indices = np.flatnonzero( |
| (sampled_rows >= hidden_y_min - int(stride)) |
| & (sampled_rows <= hidden_y_max + int(stride)) |
| ) |
| else: |
| reference_indices = np.asarray([anchor_profile_index], dtype=np.int64) |
| if reference_indices.size == 0: |
| reference_indices = np.asarray([anchor_profile_index], dtype=np.int64) |
| reference_width = float(np.median(profile_width_before_crop[reference_indices])) |
| maximum_local_width = float( |
| max( |
| 8.0 * float(stride), |
| min(2.75 * reference_width, 0.45 * float(image_width)), |
| ) |
| ) |
| minimum_tail_width = float( |
| max(8.0 * float(stride), 0.18 * min(reference_width, maximum_local_width)) |
| ) |
| required_start = int(np.min(reference_indices)) |
| required_end = int(np.max(reference_indices)) |
| eligible_tail = profile_width_before_crop >= minimum_tail_width |
| trim_start = required_start |
| while trim_start > 0 and bool(eligible_tail[trim_start - 1]): |
| trim_start -= 1 |
| trim_end = required_end |
| while trim_end + 1 < len(sampled_rows) and bool(eligible_tail[trim_end + 1]): |
| trim_end += 1 |
| minimum_sampled_rows = min(12, len(sampled_rows)) |
| while trim_end - trim_start + 1 < minimum_sampled_rows: |
| if trim_start > 0: |
| trim_start -= 1 |
| elif trim_end + 1 < len(sampled_rows): |
| trim_end += 1 |
| else: |
| break |
|
|
| original_sampled_row_count = len(sampled_rows) |
| selection = slice(trim_start, trim_end + 1) |
| sampled_rows = sampled_rows[selection] |
| left = left[selection] |
| right = right[selection] |
| center = 0.5 * (left + right) |
| selected_width = right - left + 1.0 |
| cropped_width = np.minimum(selected_width, maximum_local_width) |
| cropped_width_pixels = float(np.sum(selected_width - cropped_width)) |
| left = center - 0.5 * (cropped_width - 1.0) |
| right = center + 0.5 * (cropped_width - 1.0) |
| left = np.clip(left, 0.0, float(image_width - 2)) |
| right = np.clip(right, left + 1.0, float(image_width - 1)) |
| row_min = int(sampled_rows[0]) |
| row_max = int(sampled_rows[-1]) |
|
|
| return ( |
| sampled_rows.astype(np.float64), |
| left.astype(np.float64), |
| right.astype(np.float64), |
| { |
| "policy": ( |
| "obstacle_centered_single_channel_with_robust_quadratic_" |
| "boundary_completion" |
| ), |
| "obstacle_centered": obstacle_centered, |
| "hidden_region_used_for_localization": obstacle_centered, |
| "anchor_x": anchor_x, |
| "anchor_y": anchor_y, |
| "row_min": int(row_min), |
| "row_max": int(row_max), |
| "sampled_rows": int(len(sampled_rows)), |
| "raw_valid_rows": int(len(valid_indices)), |
| "left_fit_retention_fraction": left_retention, |
| "right_fit_retention_fraction": right_retention, |
| "raw_width_pixels_median": float(np.median(raw_width)), |
| "profile_width_pixels_median": float(np.median(right - left + 1.0)), |
| "profile_width_pixels_min": float(np.min(right - left + 1.0)), |
| "profile_width_pixels_max": float(np.max(right - left + 1.0)), |
| "local_roi_policy": ( |
| "preserve_original_image_coordinates_while_cropping_wide_" |
| "foreground_branches_and_narrow_horizon_tails" |
| ), |
| "local_roi_reference_width_pixels": reference_width, |
| "local_roi_maximum_width_pixels": maximum_local_width, |
| "local_roi_minimum_tail_width_pixels": minimum_tail_width, |
| "local_roi_width_pixels_cropped_total": cropped_width_pixels, |
| "local_roi_original_sampled_rows": original_sampled_row_count, |
| "local_roi_tail_rows_trimmed": int( |
| original_sampled_row_count - len(sampled_rows) |
| ), |
| "hidden_y_min": hidden_y_min, |
| "hidden_y_max": hidden_y_max, |
| "branch_policy": ( |
| "follow_the_channel_crossing_the_completion_zone_instead_of_" |
| "meshing_every_horizon_branch" |
| ), |
| "reviewed_masks_modified": False, |
| }, |
| ) |
|
|
|
|
| def build_obstacle_centered_depth_scaled_corridor_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| footprint_mask: np.ndarray, |
| *, |
| category: str, |
| stride: int, |
| visible_mask: np.ndarray | None = None, |
| hidden_mask: np.ndarray | None = None, |
| plane_validation_mask: np.ndarray | None = None, |
| metric_style_depth: bool = False, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Build a watertight, depth-scaled local context patch around an obstacle. |
| |
| A regular metric grid is created in the fitted local surface plane and each |
| vertex is reprojected through the camera model into the frozen completed |
| RGB. Width and length therefore select physical context rather than |
| independently normalizing texture axes. The result keeps nearby path cues, |
| covers the hidden obstacle zone, and avoids the horizon spikes and isolated |
| ribbon appearance of whole-mask inverse projection. |
| """ |
|
|
| if colors.shape[:2] != depth.shape or footprint_mask.shape != depth.shape: |
| raise ValueError("Depth, RGB, and corridor footprint must share a raster") |
| for name, mask in ( |
| ("visible_mask", visible_mask), |
| ("hidden_mask", hidden_mask), |
| ("plane_validation_mask", plane_validation_mask), |
| ): |
| if mask is not None and mask.shape != depth.shape: |
| raise ValueError(f"{name} must share the corridor raster") |
|
|
| profile_rows, profile_left, profile_right, profile_metadata = ( |
| build_obstacle_centered_corridor_profile( |
| footprint_mask, |
| stride=stride, |
| visible_mask=visible_mask, |
| hidden_mask=hidden_mask, |
| ) |
| ) |
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| camera_points = pixels_to_points(depth, fx, fy, cx, cy) |
| validation = ( |
| plane_validation_mask.astype(bool) |
| if plane_validation_mask is not None |
| else footprint_mask.astype(bool) |
| ) |
| fit_mask = validation & np.isfinite(depth) & (depth > 0) |
| if not np.any(fit_mask): |
| fit_mask = footprint_mask.astype(bool) & np.isfinite(depth) & (depth > 0) |
| plane = fit_plane(camera_points[fit_mask]) |
| if plane is None: |
| raise ValueError("Could not fit the obstacle-centred corridor plane") |
| plane_normal = np.asarray(plane[0], dtype=np.float64) |
| plane_normal_norm = float(np.linalg.norm(plane_normal)) |
| if plane_normal_norm <= 1e-12: |
| raise ValueError("Obstacle-centred corridor plane normal collapsed") |
| plane_normal /= plane_normal_norm |
| plane_d = float(plane[1]) / plane_normal_norm |
|
|
| def exact_plane_points_for( |
| image_x: np.ndarray, |
| image_y: np.ndarray, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray]: |
| image_x = np.asarray(image_x, dtype=np.float64) |
| image_y = np.asarray(image_y, dtype=np.float64) |
| rays = np.stack( |
| ( |
| (image_x - cx) / fx, |
| (image_y - cy) / fy, |
| np.ones_like(image_x), |
| ), |
| axis=-1, |
| ) |
| denominator = rays @ plane_normal |
| with np.errstate(divide="ignore", invalid="ignore"): |
| image_depth = -plane_d / denominator |
| points = rays * image_depth[..., None] |
| return points.astype(np.float64), image_depth, denominator |
|
|
| profile_center = 0.5 * (profile_left + profile_right) |
| left_points, left_depth, left_denominator = exact_plane_points_for( |
| profile_left, |
| profile_rows, |
| ) |
| right_points, right_depth, right_denominator = exact_plane_points_for( |
| profile_right, |
| profile_rows, |
| ) |
| center_points, center_depth, center_denominator = exact_plane_points_for( |
| profile_center, |
| profile_rows, |
| ) |
| control_depths = np.concatenate((left_depth, right_depth, center_depth)) |
| if not np.all(np.isfinite(control_depths)) or np.any(control_depths <= 0): |
| raise ValueError("Corridor ray/plane intersections are not positive and finite") |
|
|
| segment_lengths = np.linalg.norm(np.diff(center_points, axis=0), axis=1) |
| finite_segments = segment_lengths[ |
| np.isfinite(segment_lengths) & (segment_lengths > 0) |
| ] |
| if finite_segments.size == 0: |
| raise ValueError("Corridor has no valid depth-scaled length") |
| segment_cap = float(np.percentile(finite_segments, 90.0)) |
| robust_segments = np.minimum(segment_lengths, segment_cap) |
| cumulative = np.concatenate( |
| (np.asarray([0.0], dtype=np.float64), np.cumsum(robust_segments)) |
| ) |
| raw_centerline_length = float(cumulative[-1]) |
| if raw_centerline_length <= 1e-6: |
| raise ValueError("Corridor robust length collapsed") |
|
|
| longitudinal_axis = center_points[-1] - center_points[0] |
| longitudinal_axis -= float(np.dot(longitudinal_axis, plane_normal)) * plane_normal |
| if float(np.linalg.norm(longitudinal_axis)) <= 1e-8: |
| centered_points = center_points - np.mean(center_points, axis=0, keepdims=True) |
| try: |
| _, _, principal_axes = np.linalg.svd(centered_points, full_matrices=False) |
| except np.linalg.LinAlgError as error: |
| raise ValueError("Could not derive the corridor plane basis") from error |
| longitudinal_axis = principal_axes[0] |
| longitudinal_axis -= ( |
| float(np.dot(longitudinal_axis, plane_normal)) * plane_normal |
| ) |
| longitudinal_axis /= max(float(np.linalg.norm(longitudinal_axis)), 1e-12) |
| if float(np.dot(longitudinal_axis, center_points[-1] - center_points[0])) < 0: |
| longitudinal_axis *= -1.0 |
| lateral_hint = np.median(right_points - left_points, axis=0) |
| lateral_axis = np.cross(plane_normal, longitudinal_axis) |
| lateral_axis /= max(float(np.linalg.norm(lateral_axis)), 1e-12) |
| if float(np.dot(lateral_axis, lateral_hint)) < 0: |
| lateral_axis *= -1.0 |
|
|
| anchor_x = float(profile_metadata["anchor_x"]) |
| anchor_y = float(profile_metadata["anchor_y"]) |
| anchor_points, anchor_depth_array, anchor_denominator_array = ( |
| exact_plane_points_for( |
| np.asarray([anchor_x], dtype=np.float64), |
| np.asarray([anchor_y], dtype=np.float64), |
| ) |
| ) |
| anchor_point = anchor_points[0] |
| anchor_depth = float(anchor_depth_array[0]) |
| if not np.isfinite(anchor_depth) or anchor_depth <= 0: |
| raise ValueError("Obstacle anchor has no positive plane intersection") |
|
|
| physical_widths = np.linalg.norm(right_points - left_points, axis=1) |
| hidden_y_min = profile_metadata.get("hidden_y_min") |
| hidden_y_max = profile_metadata.get("hidden_y_max") |
| near_hidden_rows = np.ones(len(profile_rows), dtype=bool) |
| if hidden_y_min is not None and hidden_y_max is not None: |
| near_hidden_rows = ( |
| (profile_rows >= float(hidden_y_min)) |
| & (profile_rows <= float(hidden_y_max)) |
| ) |
| if not np.any(near_hidden_rows): |
| near_hidden_rows = np.ones(len(profile_rows), dtype=bool) |
| control_depth_median = float(np.median(control_depths)) |
| stable_width_rows = ( |
| (np.abs(left_denominator) >= 0.03) |
| & (np.abs(right_denominator) >= 0.03) |
| & (left_depth <= 4.0 * control_depth_median) |
| & (right_depth <= 4.0 * control_depth_median) |
| ) |
| valid_widths = physical_widths[ |
| near_hidden_rows |
| & stable_width_rows |
| & np.isfinite(physical_widths) |
| & (physical_widths > 1e-6) |
| ] |
| if valid_widths.size == 0: |
| valid_widths = physical_widths[ |
| stable_width_rows |
| & np.isfinite(physical_widths) |
| & (physical_widths > 1e-6) |
| ] |
| if valid_widths.size == 0: |
| raise ValueError("Corridor has no valid local physical width") |
| channel_width_estimate = float(np.median(valid_widths)) |
|
|
| hidden_bool = ( |
| hidden_mask.astype(bool) |
| if hidden_mask is not None |
| else np.zeros_like(footprint_mask, dtype=bool) |
| ) |
| hidden_y_pixels, hidden_x_pixels = np.nonzero(hidden_bool) |
| if hidden_x_pixels.size: |
| hidden_points, hidden_depths, hidden_denominators = exact_plane_points_for( |
| hidden_x_pixels.astype(np.float64), |
| hidden_y_pixels.astype(np.float64), |
| ) |
| valid_hidden_points = ( |
| np.isfinite(hidden_depths) |
| & (hidden_depths > 0) |
| & np.all(np.isfinite(hidden_points), axis=1) |
| ) |
| if int(np.count_nonzero(valid_hidden_points)) != len(hidden_x_pixels): |
| raise ValueError( |
| "Every reviewed hidden pixel must have a positive finite " |
| "intersection with the fitted local surface plane" |
| ) |
| hidden_points = hidden_points[valid_hidden_points] |
| hidden_depths = hidden_depths[valid_hidden_points] |
| hidden_denominators = hidden_denominators[valid_hidden_points] |
| hidden_point_image_x = hidden_x_pixels[valid_hidden_points] |
| hidden_point_image_y = hidden_y_pixels[valid_hidden_points] |
| else: |
| hidden_depths = np.empty((0,), dtype=np.float64) |
| hidden_denominators = np.empty((0,), dtype=np.float64) |
| hidden_points = np.empty((0, 3), dtype=np.float64) |
| hidden_point_image_x = np.empty((0,), dtype=np.int64) |
| hidden_point_image_y = np.empty((0,), dtype=np.int64) |
| if len(hidden_points): |
| hidden_relative = hidden_points - anchor_point[None, :] |
| hidden_local_x = hidden_relative @ lateral_axis |
| hidden_local_z = hidden_relative @ longitudinal_axis |
| stable_hidden_points = ( |
| (np.abs(hidden_denominators) >= 0.03) |
| & (hidden_depths <= 4.0 * control_depth_median) |
| ) |
| if int(np.count_nonzero(stable_hidden_points)) < max( |
| 16, |
| int(math.ceil(0.25 * len(hidden_points))), |
| ): |
| stable_hidden_points = np.ones(len(hidden_points), dtype=bool) |
| stable_hidden_x = hidden_local_x[stable_hidden_points] |
| stable_hidden_z = hidden_local_z[stable_hidden_points] |
| |
| |
| |
| hidden_x_low = float(np.min(hidden_local_x)) |
| hidden_x_high = float(np.max(hidden_local_x)) |
| hidden_z_low = float(np.min(hidden_local_z)) |
| hidden_z_high = float(np.max(hidden_local_z)) |
| context_center_x = float(np.median(stable_hidden_x)) |
| context_center_z = float(np.median(stable_hidden_z)) |
| else: |
| stable_hidden_points = np.empty((0,), dtype=bool) |
| hidden_x_low = hidden_x_high = 0.0 |
| hidden_z_low = hidden_z_high = 0.0 |
| context_center_x = 0.0 |
| context_center_z = 0.0 |
| hidden_width_span = float(max(hidden_x_high - hidden_x_low, 0.0)) |
| hidden_length_span = float(max(hidden_z_high - hidden_z_low, 0.0)) |
| |
| |
| |
| |
| context_origin = anchor_point.copy() |
|
|
| center_relative = center_points - context_origin[None, :] |
| center_longitudinal = center_relative @ longitudinal_axis |
| center_lateral = center_relative @ lateral_axis |
| stable_center_rows = ( |
| (np.abs(center_denominator) >= 0.03) |
| & (center_depth <= 4.0 * control_depth_median) |
| ) |
| if np.any(stable_center_rows): |
| stable_center_longitudinal = center_longitudinal[stable_center_rows] |
| stable_center_lateral = center_lateral[stable_center_rows] |
| else: |
| stable_center_longitudinal = center_longitudinal |
| stable_center_lateral = center_lateral |
| available_centerline_span = float( |
| np.percentile(stable_center_longitudinal, 95.0) |
| - np.percentile(stable_center_longitudinal, 5.0) |
| ) |
| raw_width = float( |
| max(1.20 * channel_width_estimate, 1.15 * hidden_width_span) |
| ) |
| raw_length = float( |
| max( |
| 1.80 * channel_width_estimate, |
| 1.35 * hidden_length_span, |
| 0.55 * available_centerline_span, |
| ) |
| ) |
| if raw_width <= 1e-6 or raw_length <= 1e-6: |
| raise ValueError("Depth-scaled context patch dimensions collapsed") |
|
|
| metric_dimension_bounds = { |
| "curb_cut": { |
| "width": (1.4, 4.5), |
| "length": (2.5, 6.0), |
| }, |
| "tactile_paving": { |
| "width": (0.8, 2.5), |
| "length": (2.5, 8.0), |
| }, |
| "walkway": { |
| "width": (1.4, 5.5), |
| "length": (3.0, 8.0), |
| }, |
| "ramp": { |
| "width": (1.2, 4.0), |
| "length": (2.5, 7.0), |
| }, |
| } |
| category_metric_bounds = metric_dimension_bounds.get(category) |
| requested_width = raw_width |
| requested_length = raw_length |
| if metric_style_depth and category_metric_bounds is not None: |
| requested_width = float( |
| np.clip(requested_width, *category_metric_bounds["width"]) |
| ) |
| requested_length = float( |
| np.clip(requested_length, *category_metric_bounds["length"]) |
| ) |
|
|
| def interval_containing_required_span( |
| center: float, |
| span: float, |
| required_low: float, |
| required_high: float, |
| margin: float, |
| ) -> tuple[float, float]: |
| required_span = max(required_high - required_low, 0.0) + 2.0 * margin |
| span = max(span, required_span, 1e-6) |
| low = center - 0.5 * span |
| high = center + 0.5 * span |
| required_low -= margin |
| required_high += margin |
| if required_low < low: |
| shift = required_low - low |
| low += shift |
| high += shift |
| if required_high > high: |
| shift = required_high - high |
| low += shift |
| high += shift |
| return float(low), float(high) |
|
|
| profile_center_x = float(np.median(stable_center_lateral)) |
| if len(hidden_points): |
| lateral_center = 0.65 * context_center_x + 0.35 * profile_center_x |
| longitudinal_center = context_center_z |
| else: |
| lateral_center = profile_center_x |
| longitudinal_center = float(np.median(stable_center_longitudinal)) |
| lateral_margin = 0.08 * channel_width_estimate |
| longitudinal_margin = 0.12 * channel_width_estimate |
| desired_x_low, desired_x_high = interval_containing_required_span( |
| lateral_center, |
| requested_width, |
| float(hidden_x_low), |
| float(hidden_x_high), |
| lateral_margin, |
| ) |
| desired_z_low, desired_z_high = interval_containing_required_span( |
| longitudinal_center, |
| requested_length, |
| float(hidden_z_low), |
| float(hidden_z_high), |
| longitudinal_margin, |
| ) |
|
|
| def clip_convex_polygon_axis( |
| polygon: np.ndarray, |
| *, |
| axis: int, |
| bound: float, |
| keep_greater: bool, |
| ) -> np.ndarray: |
| if len(polygon) == 0: |
| return polygon |
| clipped: list[np.ndarray] = [] |
|
|
| def inside(point: np.ndarray) -> bool: |
| value = float(point[axis]) |
| return value >= bound - 1e-10 if keep_greater else value <= bound + 1e-10 |
|
|
| previous = polygon[-1] |
| previous_inside = inside(previous) |
| for current in polygon: |
| current_inside = inside(current) |
| if current_inside != previous_inside: |
| denominator = float(current[axis] - previous[axis]) |
| if abs(denominator) > 1e-12: |
| fraction = (bound - float(previous[axis])) / denominator |
| clipped.append(previous + fraction * (current - previous)) |
| if current_inside: |
| clipped.append(current) |
| previous = current |
| previous_inside = current_inside |
| return np.asarray(clipped, dtype=np.float64).reshape((-1, 2)) |
|
|
| footprint_bool = footprint_mask.astype(bool) |
| row_counts = np.count_nonzero(footprint_bool, axis=1) |
| core_rows = np.flatnonzero(row_counts >= max(1, int(math.ceil(0.10 * w)))) |
| footprint_y, footprint_x = np.nonzero(footprint_bool) |
| if core_rows.size: |
| roi_y_low = int(core_rows[0]) |
| roi_y_high = int(core_rows[-1]) |
| elif footprint_y.size: |
| roi_y_low = int(np.min(footprint_y)) |
| roi_y_high = int(np.max(footprint_y)) |
| else: |
| raise ValueError("Context ROI requires a non-empty reviewed footprint") |
| if hidden_y_pixels.size: |
| roi_y_low = min(roi_y_low, int(np.min(hidden_y_pixels))) |
| roi_y_high = max(roi_y_high, int(np.max(hidden_y_pixels))) |
| roi_y_margin = max(1, int(math.ceil(0.01 * h))) |
| roi_y_low = max(0, roi_y_low - roi_y_margin) |
| roi_y_high = min(h - 1, roi_y_high + roi_y_margin) |
| roi_height = roi_y_high - roi_y_low + 1 |
| column_counts = np.count_nonzero( |
| footprint_bool[roi_y_low : roi_y_high + 1], |
| axis=0, |
| ) |
| core_columns = np.flatnonzero( |
| column_counts >= max(1, int(math.ceil(0.15 * roi_height))) |
| ) |
| if core_columns.size: |
| roi_x_low = int(core_columns[0]) |
| roi_x_high = int(core_columns[-1]) |
| elif footprint_x.size: |
| roi_x_low = int(np.min(footprint_x)) |
| roi_x_high = int(np.max(footprint_x)) |
| else: |
| raise ValueError("Context ROI requires horizontal footprint support") |
| if hidden_x_pixels.size: |
| roi_x_low = min(roi_x_low, int(np.min(hidden_x_pixels))) |
| roi_x_high = max(roi_x_high, int(np.max(hidden_x_pixels))) |
| roi_x_margin = max(1, int(math.ceil(0.02 * w))) |
| roi_x_low = max(0, roi_x_low - roi_x_margin) |
| roi_x_high = min(w - 1, roi_x_high + roi_x_margin) |
|
|
| image_margin = 0.5 |
| image_context_roi = { |
| "x_min": float(max(image_margin, roi_x_low)), |
| "x_max": float(min(w - 1.0 - image_margin, roi_x_high)), |
| "y_min": float(max(image_margin, roi_y_low)), |
| "y_max": float(min(h - 1.0 - image_margin, roi_y_high)), |
| } |
| image_boundary_polygon = np.asarray( |
| [ |
| [image_margin, image_margin], |
| [w - 1.0 - image_margin, image_margin], |
| [w - 1.0 - image_margin, h - 1.0 - image_margin], |
| [image_margin, h - 1.0 - image_margin], |
| ], |
| dtype=np.float64, |
| ) |
| plane_depth_sign = 1.0 if -plane_d >= 0.0 else -1.0 |
| minimum_required_hidden_denominator = ( |
| float(np.min(np.abs(hidden_denominators))) |
| if len(hidden_denominators) |
| else 0.03 |
| ) |
| plane_stability_floor = max( |
| 1e-4, |
| min(0.03, 0.98 * minimum_required_hidden_denominator), |
| ) |
|
|
| def stable_plane_halfspace_value(point: np.ndarray) -> float: |
| image_x_value = float(point[0]) |
| image_y_value = float(point[1]) |
| denominator = ( |
| plane_normal[0] * (image_x_value - cx) / fx |
| + plane_normal[1] * (image_y_value - cy) / fy |
| + plane_normal[2] |
| ) |
| return plane_depth_sign * float(denominator) - plane_stability_floor |
|
|
| clipped_image_polygon: list[np.ndarray] = [] |
| previous = image_boundary_polygon[-1] |
| previous_value = stable_plane_halfspace_value(previous) |
| for current in image_boundary_polygon: |
| current_value = stable_plane_halfspace_value(current) |
| previous_inside = previous_value >= -1e-10 |
| current_inside = current_value >= -1e-10 |
| if previous_inside != current_inside: |
| value_delta = previous_value - current_value |
| if abs(value_delta) <= 1e-12: |
| raise ValueError("Plane-stability half-space edge collapsed") |
| fraction = previous_value / value_delta |
| clipped_image_polygon.append( |
| previous + np.clip(fraction, 0.0, 1.0) * (current - previous) |
| ) |
| if current_inside: |
| clipped_image_polygon.append(current) |
| previous = current |
| previous_value = current_value |
| image_boundary_polygon = np.asarray( |
| clipped_image_polygon, |
| dtype=np.float64, |
| ).reshape((-1, 2)) |
| if len(image_boundary_polygon) < 3: |
| raise ValueError("The reviewed context ROI has no stable positive plane region") |
| frustum_points, frustum_depths, frustum_denominators = exact_plane_points_for( |
| image_boundary_polygon[:, 0], |
| image_boundary_polygon[:, 1], |
| ) |
| if ( |
| not np.all(np.isfinite(frustum_points)) |
| or not np.all(np.isfinite(frustum_depths)) |
| or np.any(frustum_depths <= 0) |
| ): |
| raise ValueError("The local surface plane does not span the camera image") |
| frustum_relative = frustum_points - context_origin[None, :] |
| context_polygon = np.stack( |
| ( |
| frustum_relative @ lateral_axis, |
| frustum_relative @ longitudinal_axis, |
| ), |
| axis=1, |
| ) |
| for axis, bound, keep_greater in ( |
| (0, desired_x_low, True), |
| (0, desired_x_high, False), |
| (1, desired_z_low, True), |
| (1, desired_z_high, False), |
| ): |
| context_polygon = clip_convex_polygon_axis( |
| context_polygon, |
| axis=axis, |
| bound=bound, |
| keep_greater=keep_greater, |
| ) |
| if len(context_polygon) < 3: |
| raise ValueError("Depth-scaled context patch does not intersect the camera field of view") |
| polygon_area = 0.5 * abs( |
| float( |
| np.sum( |
| context_polygon[:, 0] * np.roll(context_polygon[:, 1], -1) |
| - context_polygon[:, 1] * np.roll(context_polygon[:, 0], -1) |
| ) |
| ) |
| ) |
| if not np.isfinite(polygon_area) or polygon_area <= 1e-8: |
| raise ValueError("Camera-clipped context polygon collapsed") |
|
|
| context_polygon_camera = ( |
| context_origin[None, :] |
| + context_polygon[:, :1] * lateral_axis[None, :] |
| + context_polygon[:, 1:] * longitudinal_axis[None, :] |
| ) |
| context_polygon_depth = context_polygon_camera[:, 2] |
| context_image_polygon = np.stack( |
| ( |
| fx * context_polygon_camera[:, 0] / context_polygon_depth + cx, |
| fy * context_polygon_camera[:, 1] / context_polygon_depth + cy, |
| ), |
| axis=1, |
| ) |
| if not np.all(np.isfinite(context_image_polygon)): |
| raise ValueError("Camera-clipped context polygon did not reproject finitely") |
| projected_width = float(np.ptp(context_image_polygon[:, 0])) |
| projected_height = float(np.ptp(context_image_polygon[:, 1])) |
| polygon_metric_width = float(np.ptp(context_polygon[:, 0])) |
| polygon_metric_length = float(np.ptp(context_polygon[:, 1])) |
| target_metric_cell = max( |
| max(polygon_metric_width, polygon_metric_length) / 480.0, |
| min(polygon_metric_width, polygon_metric_length) / 120.0, |
| 1e-4, |
| ) |
| across_count = int( |
| np.clip(math.ceil(polygon_metric_width / target_metric_cell) + 1, 24, 481) |
| ) |
| row_count = int( |
| np.clip(math.ceil(polygon_metric_length / target_metric_cell) + 1, 24, 481) |
| ) |
|
|
| def polygon_x_bounds_at_z(z_value: float) -> tuple[float, float]: |
| intersections: list[float] = [] |
| for edge_index in range(len(context_polygon)): |
| start = context_polygon[edge_index] |
| end = context_polygon[(edge_index + 1) % len(context_polygon)] |
| z_start = float(start[1]) |
| z_end = float(end[1]) |
| if abs(z_end - z_start) <= 1e-12: |
| if abs(z_value - z_start) <= 1e-9: |
| intersections.extend((float(start[0]), float(end[0]))) |
| continue |
| fraction = (z_value - z_start) / (z_end - z_start) |
| if -1e-9 <= fraction <= 1.0 + 1e-9: |
| intersections.append( |
| float(start[0] + np.clip(fraction, 0.0, 1.0) * (end[0] - start[0])) |
| ) |
| if len(intersections) < 2: |
| raise ValueError("Could not scan-convert the camera-clipped context polygon") |
| return float(min(intersections)), float(max(intersections)) |
|
|
| polygon_z_low = float(np.min(context_polygon[:, 1])) |
| polygon_z_high = float(np.max(context_polygon[:, 1])) |
| length = polygon_z_high - polygon_z_low |
| if length <= 1e-6: |
| raise ValueError("Camera-clipped context patch has no longitudinal extent") |
| endpoint_low = polygon_x_bounds_at_z(polygon_z_low) |
| endpoint_high = polygon_x_bounds_at_z(polygon_z_high) |
| maximum_polygon_width = float(np.ptp(context_polygon[:, 0])) |
| if min( |
| endpoint_low[1] - endpoint_low[0], |
| endpoint_high[1] - endpoint_high[0], |
| ) < max(1e-5, 0.01 * maximum_polygon_width): |
| endpoint_inset = min(0.0025 * length, 0.02 * channel_width_estimate) |
| polygon_z_low += endpoint_inset |
| polygon_z_high -= endpoint_inset |
| length = polygon_z_high - polygon_z_low |
| z_values = np.linspace(polygon_z_low, polygon_z_high, row_count, dtype=np.float64) |
| row_x_bounds = np.asarray( |
| [polygon_x_bounds_at_z(float(value)) for value in z_values], |
| dtype=np.float64, |
| ) |
| if np.any(row_x_bounds[:, 1] - row_x_bounds[:, 0] <= 1e-6): |
| raise ValueError("Camera-clipped context grid contains a collapsed row") |
| across_fraction = np.linspace(0.0, 1.0, across_count, dtype=np.float64) |
| longitudinal_fraction = np.linspace(0.0, 1.0, row_count, dtype=np.float64) |
| local_x = ( |
| row_x_bounds[:, :1] |
| + across_fraction[None, :] |
| * (row_x_bounds[:, 1:] - row_x_bounds[:, :1]) |
| ) |
| local_z = np.broadcast_to(z_values[:, None], (row_count, across_count)) |
| width = float(np.ptp(local_x)) |
| length = float(np.ptp(local_z)) |
| final_x_low = float(np.min(local_x)) |
| final_x_high = float(np.max(local_x)) |
| final_z_low = float(np.min(local_z)) |
| final_z_high = float(np.max(local_z)) |
| mesh_context_polygon = np.concatenate( |
| ( |
| np.stack((row_x_bounds[:, 0], z_values), axis=1), |
| np.stack((row_x_bounds[::-1, 1], z_values[::-1]), axis=1), |
| ), |
| axis=0, |
| ) |
| mesh_context_polygon_camera = ( |
| context_origin[None, :] |
| + mesh_context_polygon[:, :1] * lateral_axis[None, :] |
| + mesh_context_polygon[:, 1:] * longitudinal_axis[None, :] |
| ) |
| mesh_context_polygon_depth = mesh_context_polygon_camera[:, 2] |
| context_image_polygon = np.stack( |
| ( |
| fx * mesh_context_polygon_camera[:, 0] / mesh_context_polygon_depth + cx, |
| fy * mesh_context_polygon_camera[:, 1] / mesh_context_polygon_depth + cy, |
| ), |
| axis=1, |
| ) |
| polygon_area = 0.5 * abs( |
| float( |
| np.sum( |
| mesh_context_polygon[:, 0] |
| * np.roll(mesh_context_polygon[:, 1], -1) |
| - mesh_context_polygon[:, 1] |
| * np.roll(mesh_context_polygon[:, 0], -1) |
| ) |
| ) |
| ) |
| grid_camera_points = ( |
| context_origin[None, None, :] |
| + local_x[..., None] * lateral_axis[None, None, :] |
| + local_z[..., None] * longitudinal_axis[None, None, :] |
| ) |
| if np.any(grid_camera_points[..., 2] <= 1e-6): |
| raise ValueError("Metric context grid crossed the camera plane") |
| grid_image_x = fx * grid_camera_points[..., 0] / grid_camera_points[..., 2] + cx |
| grid_image_y = fy * grid_camera_points[..., 1] / grid_camera_points[..., 2] + cy |
| if ( |
| np.any(grid_image_x < 0) |
| or np.any(grid_image_x > w - 1) |
| or np.any(grid_image_y < 0) |
| or np.any(grid_image_y > h - 1) |
| ): |
| raise ValueError("Metric context grid projected outside the source image") |
| sampled_grid_colors = cv2.remap( |
| colors, |
| grid_image_x.astype(np.float32), |
| grid_image_y.astype(np.float32), |
| interpolation=cv2.INTER_LINEAR, |
| borderMode=cv2.BORDER_REPLICATE, |
| ).astype(np.uint8) |
|
|
| projected_patch_mask = np.zeros((h, w), dtype=np.uint8) |
| cv2.fillConvexPoly( |
| projected_patch_mask, |
| np.rint(context_image_polygon).astype(np.int32), |
| 1, |
| ) |
| projected_patch_bool = projected_patch_mask.astype(bool) |
| hidden_pixel_coverage_fraction = float( |
| np.count_nonzero(projected_patch_bool & hidden_bool) |
| / max(int(np.count_nonzero(hidden_bool)), 1) |
| ) |
| context_target_fraction = float( |
| np.count_nonzero(projected_patch_bool & footprint_mask.astype(bool)) |
| / max(int(np.count_nonzero(projected_patch_bool)), 1) |
| ) |
| continuous_hidden_coverage_fraction = 1.0 |
| continuous_hidden_membership = np.ones(len(hidden_points), dtype=bool) |
| if len(hidden_points): |
| polygon_orientation = np.sign( |
| np.sum( |
| mesh_context_polygon[:, 0] |
| * np.roll(mesh_context_polygon[:, 1], -1) |
| - mesh_context_polygon[:, 1] |
| * np.roll(mesh_context_polygon[:, 0], -1) |
| ) |
| ) |
| if polygon_orientation == 0: |
| raise ValueError("Metric context polygon has no winding") |
| hidden_local_points = np.stack((hidden_local_x, hidden_local_z), axis=1) |
| edge_start = mesh_context_polygon |
| edge_delta = np.roll(mesh_context_polygon, -1, axis=0) - edge_start |
| coordinate_scale = max( |
| float(np.max(np.abs(mesh_context_polygon))), |
| 1.0, |
| ) |
| continuous_hidden_membership = np.ones(len(hidden_local_points), dtype=bool) |
| for start, delta in zip(edge_start, edge_delta): |
| relative = hidden_local_points - start[None, :] |
| signed_cross = ( |
| delta[0] * relative[:, 1] - delta[1] * relative[:, 0] |
| ) |
| continuous_hidden_membership &= ( |
| polygon_orientation * signed_cross >= -1e-7 * coordinate_scale |
| ) |
| if not np.any(continuous_hidden_membership): |
| break |
| continuous_hidden_coverage_fraction = float( |
| np.count_nonzero(continuous_hidden_membership) |
| / len(continuous_hidden_membership) |
| ) |
| hidden_component_coverage: list[dict[str, Any]] = [] |
| minimum_hidden_component_coverage = 1.0 |
| minimum_continuous_hidden_component_coverage = 1.0 |
| if np.any(hidden_bool): |
| component_count, component_labels = cv2.connectedComponents( |
| hidden_bool.astype(np.uint8), |
| connectivity=8, |
| ) |
| minimum_component_pixels = max( |
| 8, |
| int(math.ceil(0.001 * np.count_nonzero(hidden_bool))), |
| ) |
| for component_id in range(1, component_count): |
| component = component_labels == component_id |
| component_pixels = int(np.count_nonzero(component)) |
| if component_pixels < minimum_component_pixels: |
| continue |
| coverage = float( |
| np.count_nonzero(projected_patch_bool & component) |
| / component_pixels |
| ) |
| point_component = ( |
| component_labels[hidden_point_image_y, hidden_point_image_x] |
| == component_id |
| ) |
| continuous_coverage = float( |
| np.count_nonzero( |
| continuous_hidden_membership[point_component] |
| ) |
| / max(int(np.count_nonzero(point_component)), 1) |
| ) |
| hidden_component_coverage.append( |
| { |
| "component_id": component_id, |
| "pixels": component_pixels, |
| "coverage_fraction": coverage, |
| "continuous_local_coordinate_coverage_fraction": ( |
| continuous_coverage |
| ), |
| } |
| ) |
| if hidden_component_coverage: |
| minimum_hidden_component_coverage = min( |
| float(row["coverage_fraction"]) |
| for row in hidden_component_coverage |
| ) |
| minimum_continuous_hidden_component_coverage = min( |
| float(row["continuous_local_coordinate_coverage_fraction"]) |
| for row in hidden_component_coverage |
| ) |
|
|
| observed_vertical = -center_points[:, 1] |
| variance_longitudinal = float(np.var(center_longitudinal)) |
| vertical_slope = ( |
| float( |
| np.mean( |
| (center_longitudinal - float(np.mean(center_longitudinal))) |
| * (observed_vertical - float(np.mean(observed_vertical))) |
| ) |
| / max(variance_longitudinal, 1e-12) |
| ) |
| if variance_longitudinal > 1e-12 |
| else 0.0 |
| ) |
| raw_depth_rise = abs(vertical_slope) * length |
| rise_source = "local_plane_camera_y_gradient_without_gravity_calibration" |
| rise_confidence = "low" |
| if category == "ramp": |
| minimum_rise = 0.015 * length |
| maximum_rise = (1.0 / 12.0) * length |
| rise = float(np.clip(raw_depth_rise, minimum_rise, maximum_rise)) |
| if raw_depth_rise < minimum_rise or raw_depth_rise > maximum_rise: |
| rise_source += "_with_accessibility_bounds" |
| rise_confidence = "low" |
| category_signature = "depth_scaled_uniform_local_channel_slope" |
| maximum_semantic_relief = 0.0 |
| elif category == "curb_cut": |
| minimum_rise = ( |
| max(0.012 * length, 0.04) |
| if metric_style_depth |
| else 0.012 * length |
| ) |
| maximum_rise = ( |
| min((1.0 / 12.0) * length, 0.20) |
| if metric_style_depth |
| else (1.0 / 12.0) * length |
| ) |
| rise = float(np.clip(raw_depth_rise, minimum_rise, maximum_rise)) |
| if raw_depth_rise < minimum_rise or raw_depth_rise > maximum_rise: |
| rise_source += "_with_accessibility_bounds" |
| rise_confidence = "low" |
| category_signature = "complete_central_low_channel_with_lateral_flares" |
| maximum_semantic_relief = float(0.65 * rise) |
| elif category == "tactile_paving": |
| rise = 0.0 |
| rise_source = "near_level_with_scale_bounded_tactile_relief" |
| rise_confidence = "category_verified" |
| category_signature = "complete_channel_with_visible_tactile_ribs" |
| maximum_semantic_relief = float( |
| np.clip(0.006 * length, 0.008, 0.025) |
| ) |
| else: |
| rise = float(min(raw_depth_rise, 0.02 * length)) |
| if raw_depth_rise > 0.02 * length: |
| rise_source += "_near_level_bound" |
| rise_confidence = "low" |
| category_signature = "depth_scaled_near_level_complete_walkway_channel" |
| maximum_semantic_relief = 0.0 |
|
|
| front_vertices: list[np.ndarray] = [] |
| front_colors: list[np.ndarray] = [] |
| sampled_inside_footprint = 0 |
| sampled_inside_hidden = 0 |
| rib_count = int(np.clip(round(length / 0.45), 6, 24)) |
| for row_index, v in enumerate(longitudinal_fraction): |
| image_x = grid_image_x[row_index] |
| image_y = grid_image_y[row_index] |
| sample_x = np.clip(np.rint(image_x), 0, w - 1).astype(np.int32) |
| sample_y = np.clip(np.rint(image_y), 0, h - 1).astype(np.int32) |
| sampled_inside_footprint += int( |
| np.count_nonzero(footprint_mask[sample_y, sample_x]) |
| ) |
| sampled_inside_hidden += int( |
| np.count_nonzero(hidden_bool[sample_y, sample_x]) |
| ) |
| x_coordinate = local_x[row_index] |
| z_coordinate = local_z[row_index] |
| if category == "ramp": |
| height = np.full_like(x_coordinate, (1.0 - v) * rise) |
| elif category == "curb_cut": |
| lateral = np.abs(2.0 * across_fraction - 1.0) |
| height = ( |
| np.full_like(x_coordinate, (1.0 - v) * rise) |
| + 0.65 * rise * v * lateral**3 |
| ) |
| elif category == "tactile_paving": |
| rib_phase = 2.0 * math.pi * float(rib_count) * v |
| rib = max(0.0, math.cos(rib_phase)) ** 6 |
| height = np.full_like( |
| x_coordinate, |
| maximum_semantic_relief * rib, |
| ) |
| else: |
| height = np.full_like(x_coordinate, (1.0 - v) * rise) |
| front_vertices.extend( |
| np.stack((x_coordinate, -height, z_coordinate), axis=1).astype( |
| np.float32 |
| ) |
| ) |
| front_colors.extend(sampled_grid_colors[row_index]) |
|
|
| front_faces: list[tuple[int, int, int]] = [] |
| for row_index in range(row_count - 1): |
| row_start = row_index * across_count |
| next_start = (row_index + 1) * across_count |
| for column_index in range(across_count - 1): |
| a = row_start + column_index |
| b = a + 1 |
| c = next_start + column_index |
| d = c + 1 |
| front_faces.append((a, c, b)) |
| front_faces.append((b, c, d)) |
|
|
| transformed_top = np.asarray(front_vertices, dtype=np.float32) |
| top_colors = np.asarray(front_colors, dtype=np.uint8) |
| footprint_diag = float(math.hypot(width, length)) |
| thickness = float(np.clip(0.018 * footprint_diag, 0.025, 0.18)) |
| back_vertices = transformed_top.copy() |
| back_vertices[:, 1] += thickness |
| front_vertex_count = len(transformed_top) |
| back_faces = [ |
| ( |
| face[2] + front_vertex_count, |
| face[1] + front_vertex_count, |
| face[0] + front_vertex_count, |
| ) |
| for face in front_faces |
| ] |
| edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
| for face in front_faces: |
| for a, b in ( |
| (face[0], face[1]), |
| (face[1], face[2]), |
| (face[2], face[0]), |
| ): |
| edge_counts[tuple(sorted((a, b)))] += 1 |
| boundary_edges = [edge for edge, count in edge_counts.items() if count == 1] |
| side_faces: list[tuple[int, int, int]] = [] |
| for a, b in boundary_edges: |
| side_faces.append((a, b, a + front_vertex_count)) |
| side_faces.append((b, b + front_vertex_count, a + front_vertex_count)) |
| all_faces = front_faces + back_faces + side_faces |
| full_edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
| for face in all_faces: |
| for a, b in ( |
| (face[0], face[1]), |
| (face[1], face[2]), |
| (face[2], face[0]), |
| ): |
| full_edge_counts[tuple(sorted((a, b)))] += 1 |
| watertight_edge_check = bool( |
| full_edge_counts and all(value == 2 for value in full_edge_counts.values()) |
| ) |
| if not watertight_edge_check: |
| raise ValueError("Obstacle-centred corridor failed the watertight edge check") |
|
|
| neutral = np.clip( |
| np.median(top_colors.astype(np.float32), axis=0) * 0.52, |
| 0, |
| 255, |
| ).astype(np.uint8) |
| front_face_array = np.asarray(front_faces, dtype=np.int64) |
| front_triangles = transformed_top[front_face_array].astype(np.float64) |
| triangle_edges = np.stack( |
| ( |
| front_triangles[:, 1] - front_triangles[:, 0], |
| front_triangles[:, 2] - front_triangles[:, 1], |
| front_triangles[:, 0] - front_triangles[:, 2], |
| ), |
| axis=1, |
| ) |
| triangle_edge_lengths = np.linalg.norm(triangle_edges, axis=2) |
| triangle_twice_area = np.linalg.norm( |
| np.cross( |
| front_triangles[:, 1] - front_triangles[:, 0], |
| front_triangles[:, 2] - front_triangles[:, 0], |
| ), |
| axis=1, |
| ) |
| if np.any(~np.isfinite(triangle_twice_area)) or np.any( |
| triangle_twice_area <= 1e-12 |
| ): |
| raise ValueError("Perspective-preserving corridor has degenerate front faces") |
| triangle_area = 0.5 * triangle_twice_area |
| triangle_longest_edge = np.max(triangle_edge_lengths, axis=1) |
| triangle_aspect = triangle_longest_edge**2 / triangle_twice_area |
| edge_length_median = float(np.median(triangle_edge_lengths)) |
| triangle_area_median = float(np.median(triangle_area)) |
| front_xz = transformed_top[:, (0, 2)] |
| front_xz_triangles = front_xz[front_face_array] |
| signed_twice_area_xz = ( |
| (front_xz_triangles[:, 1, 0] - front_xz_triangles[:, 0, 0]) |
| * (front_xz_triangles[:, 2, 1] - front_xz_triangles[:, 0, 1]) |
| - (front_xz_triangles[:, 1, 1] - front_xz_triangles[:, 0, 1]) |
| * (front_xz_triangles[:, 2, 0] - front_xz_triangles[:, 0, 0]) |
| ) |
| winding_sign = -1.0 if float(np.median(signed_twice_area_xz)) < 0 else 1.0 |
| flipped_front_faces = int( |
| np.count_nonzero(winding_sign * signed_twice_area_xz <= 1e-12) |
| ) |
| source_pixel_points = np.stack((grid_image_x, grid_image_y), axis=-1) |
| source_pixel_edges = np.concatenate( |
| ( |
| np.linalg.norm(np.diff(source_pixel_points, axis=1), axis=2).reshape(-1), |
| np.linalg.norm(np.diff(source_pixel_points, axis=0), axis=2).reshape(-1), |
| ) |
| ) |
| source_pixel_edge_median = float(np.median(source_pixel_edges)) |
| control_denominators = np.abs( |
| np.concatenate((left_denominator, right_denominator, center_denominator)) |
| ) |
| grid_rays = np.stack( |
| ( |
| (grid_image_x - cx) / fx, |
| (grid_image_y - cy) / fy, |
| np.ones_like(grid_image_x), |
| ), |
| axis=-1, |
| ) |
| grid_denominators = np.abs(grid_rays @ plane_normal) |
| grid_depths = grid_camera_points[..., 2] |
| fit_residuals = np.abs(camera_points[fit_mask] @ plane_normal + plane_d) |
| desired_rectangle_area = max( |
| (desired_x_high - desired_x_low) * (desired_z_high - desired_z_low), |
| 1e-12, |
| ) |
| field_of_view_retained_area_fraction = float( |
| np.clip(polygon_area / desired_rectangle_area, 0.0, 1.0) |
| ) |
| field_of_view_scale = float(math.sqrt(field_of_view_retained_area_fraction)) |
| geometry_quality = { |
| "front_triangle_area_max_to_median": float( |
| np.max(triangle_area) / max(triangle_area_median, 1e-12) |
| ), |
| "front_edge_length_max_to_median": float( |
| np.max(triangle_edge_lengths) / max(edge_length_median, 1e-12) |
| ), |
| "front_triangle_aspect_p99": float(np.percentile(triangle_aspect, 99.0)), |
| "front_triangle_aspect_max": float(np.max(triangle_aspect)), |
| "front_triangle_min_area": float(np.min(triangle_area)), |
| "front_triangle_median_area": triangle_area_median, |
| "front_edge_median_length": edge_length_median, |
| "actual_width_span": float(np.ptp(transformed_top[:, 0])), |
| "actual_length_span": float(np.ptp(transformed_top[:, 2])), |
| "reported_width_absolute_error": float( |
| abs(float(np.ptp(transformed_top[:, 0])) - width) |
| ), |
| "reported_length_absolute_error": float( |
| abs(float(np.ptp(transformed_top[:, 2])) - length) |
| ), |
| "flipped_or_degenerate_front_faces_xz": flipped_front_faces, |
| "basis_lateral_longitudinal_dot_absolute": float( |
| abs(float(np.dot(lateral_axis, longitudinal_axis))) |
| ), |
| "basis_lateral_normal_dot_absolute": float( |
| abs(float(np.dot(lateral_axis, plane_normal))) |
| ), |
| "basis_longitudinal_normal_dot_absolute": float( |
| abs(float(np.dot(longitudinal_axis, plane_normal))) |
| ), |
| "source_pixel_edge_max_to_median": float( |
| np.max(source_pixel_edges) / max(source_pixel_edge_median, 1e-12) |
| ), |
| "source_pixel_edge_min_to_median": float( |
| np.min(source_pixel_edges) / max(source_pixel_edge_median, 1e-12) |
| ), |
| "source_pixel_edge_median": source_pixel_edge_median, |
| "source_pixel_edge_p99": float( |
| np.percentile(source_pixel_edges, 99.0) |
| ), |
| "source_pixel_edge_max": float(np.max(source_pixel_edges)), |
| "control_ray_plane_denominator_min_abs": float( |
| np.min(control_denominators) |
| ), |
| "control_plane_depth_max_to_median": float( |
| np.max(control_depths) / max(float(np.median(control_depths)), 1e-12) |
| ), |
| "context_grid_ray_plane_denominator_min_abs": float( |
| np.min(grid_denominators) |
| ), |
| "context_grid_plane_depth_max_to_median": float( |
| np.max(grid_depths) / max(float(np.median(grid_depths)), 1e-12) |
| ), |
| "plane_fit_residual_median": float(np.median(fit_residuals)), |
| "plane_fit_residual_p95": float(np.percentile(fit_residuals, 95.0)), |
| "field_of_view_retained_area_fraction": ( |
| field_of_view_retained_area_fraction |
| ), |
| "hidden_pixel_coverage_fraction": hidden_pixel_coverage_fraction, |
| "continuous_local_coordinate_hidden_coverage_fraction": ( |
| continuous_hidden_coverage_fraction |
| ), |
| "minimum_hidden_component_coverage_fraction": ( |
| minimum_hidden_component_coverage |
| ), |
| "minimum_continuous_hidden_component_coverage_fraction": ( |
| minimum_continuous_hidden_component_coverage |
| ), |
| "context_target_fraction": context_target_fraction, |
| "profile_anchor_y_inside_sampled_rows": bool( |
| float(profile_metadata["row_min"]) |
| <= float(profile_metadata["anchor_y"]) |
| <= float(profile_metadata["row_max"]) |
| ), |
| } |
| back_colors = np.repeat(neutral[None, :], front_vertex_count, axis=0) |
| vertices = np.concatenate((transformed_top, back_vertices), axis=0) |
| vertex_colors = np.concatenate((top_colors, back_colors), axis=0) |
| sample_count = row_count * across_count |
| width_bound_applied = bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and not ( |
| category_metric_bounds["width"][0] |
| <= raw_width |
| <= category_metric_bounds["width"][1] |
| ) |
| ) |
| length_bound_applied = bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and not ( |
| category_metric_bounds["length"][0] |
| <= raw_length |
| <= category_metric_bounds["length"][1] |
| ) |
| ) |
| return vertices, vertex_colors, np.asarray(all_faces, dtype=np.int64), { |
| "front_faces": len(front_faces), |
| "back_faces": len(back_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(side_faces), |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": 0, |
| "stair_sidewall_bottom_faces": 0, |
| "solid_thickness": thickness, |
| "max_depth_jump": None, |
| "stride": int(stride), |
| "support_mode": "obstacle_centered_metric_context_patch_slab", |
| "surface_model": "obstacle_centered_depth_scaled_context_surface", |
| "complete_footprint": False, |
| "complete_obstacle_zone": bool( |
| not np.any(hidden_bool) |
| or ( |
| hidden_pixel_coverage_fraction >= 0.98 |
| and minimum_hidden_component_coverage >= 0.90 |
| and continuous_hidden_coverage_fraction >= 0.999 |
| and minimum_continuous_hidden_component_coverage >= 0.999 |
| ) |
| ), |
| "open_world_channel_model": False, |
| "local_context_patch_model": True, |
| "open_world_scene_compatible": True, |
| "open_world_topology_claim": False, |
| "context_pixels_are_target_claim": False, |
| "footprint_policy": ( |
| "asymmetric_depth_scaled_local_plane_context_polygon_clipped_to_" |
| "the_camera_frustum_and_required_hidden_extent" |
| ), |
| "texture_coordinate_policy": ( |
| "regular_metric_plane_grid_reprojected_to_frozen_completed_rgb_" |
| "original_image_xy_with_bilinear_sampling_without_extrapolation" |
| ), |
| "texture_mapping": "metric_plane_grid_reprojected_to_original_image", |
| "texture_geometry_warp": ( |
| "metric_plane_pinhole_reprojection_plus_bounded_semantic_height" |
| ), |
| "independent_axis_normalization_used": False, |
| "uniform_geometry_scale": 1.0, |
| "uniform_geometry_scale_applied": False, |
| "texture_resampling": "bilinear_original_image_pixel", |
| "texture_extrapolation_used": False, |
| "texture_extension_max_distance_pixels": 0.0, |
| "texture_profile_inside_reviewed_footprint_fraction": float( |
| sampled_inside_footprint / max(sample_count, 1) |
| ), |
| "texture_profile_hidden_completion_fraction": float( |
| sampled_inside_hidden / max(sample_count, 1) |
| ), |
| "projected_context_hidden_pixel_coverage_fraction": ( |
| hidden_pixel_coverage_fraction |
| ), |
| "projected_context_target_fraction": context_target_fraction, |
| "projected_context_hidden_component_coverage": ( |
| hidden_component_coverage |
| ), |
| "projected_context_minimum_hidden_component_coverage_fraction": ( |
| minimum_hidden_component_coverage |
| ), |
| "continuous_local_coordinate_hidden_coverage_fraction": ( |
| continuous_hidden_coverage_fraction |
| ), |
| "minimum_continuous_hidden_component_coverage_fraction": ( |
| minimum_continuous_hidden_component_coverage |
| ), |
| "all_hidden_components_covered": bool( |
| continuous_hidden_coverage_fraction >= 0.999 |
| and minimum_continuous_hidden_component_coverage >= 0.999 |
| ), |
| "projected_context_image_corners": [ |
| [float(value) for value in corner] |
| for corner in context_image_polygon |
| ], |
| "projected_context_image_polygon": [ |
| [float(value) for value in point] |
| for point in context_image_polygon |
| ], |
| "context_roi_image_xyxy": image_context_roi, |
| "context_roi_policy": ( |
| "reviewed_target_core_plus_all_hidden_components_for_diagnostics_" |
| "while_camera_frustum_clipping_uses_the_full_source_image" |
| ), |
| "category_geometry_signature": category_signature, |
| "maximum_semantic_relief": maximum_semantic_relief, |
| "front_boundary_loops": 1, |
| "final_cell_holes_filled": True, |
| "regular_grid_has_no_internal_topological_holes": True, |
| "final_enclosed_hole_added_cells": 0, |
| "watertight_edge_check": watertight_edge_check, |
| "calibrated_metric_truth": False, |
| "metric_style_depth_input": bool(metric_style_depth), |
| "automatic_passability_claim": False, |
| "dimension_source": ( |
| "asymmetric_hidden_component_local_ray_plane_extent_with_category_" |
| "roi_bounds_and_camera_frustum_polygon_clip" |
| ), |
| "rise_source": rise_source, |
| "rise_confidence": rise_confidence, |
| "raw_depth_rise": float(raw_depth_rise), |
| "raw_depth_scaled_dimensions": { |
| "width": raw_width, |
| "length": raw_length, |
| "robust_centerline_length": raw_centerline_length, |
| "channel_width_estimate": channel_width_estimate, |
| "hidden_width_span_full_reviewed_support": hidden_width_span, |
| "hidden_length_span_full_reviewed_support": hidden_length_span, |
| "stable_hidden_point_fraction": float( |
| np.count_nonzero(stable_hidden_points) |
| / max(len(stable_hidden_points), 1) |
| ), |
| }, |
| "dimension_bounds": { |
| "independent_texture_axis_normalization_used": False, |
| "physical_roi_axis_bounds_used": bool( |
| width_bound_applied or length_bound_applied |
| ), |
| "metric_style_width_min": ( |
| category_metric_bounds["width"][0] |
| if metric_style_depth and category_metric_bounds is not None |
| else None |
| ), |
| "metric_style_width_max": ( |
| category_metric_bounds["width"][1] |
| if metric_style_depth and category_metric_bounds is not None |
| else None |
| ), |
| "metric_style_length_min": ( |
| category_metric_bounds["length"][0] |
| if metric_style_depth and category_metric_bounds is not None |
| else None |
| ), |
| "metric_style_length_max": ( |
| category_metric_bounds["length"][1] |
| if metric_style_depth and category_metric_bounds is not None |
| else None |
| ), |
| "width_bound_applied": width_bound_applied, |
| "length_bound_applied": length_bound_applied, |
| "hard_required_local_extents": { |
| "x": [float(hidden_x_low), float(hidden_x_high)], |
| "z": [float(hidden_z_low), float(hidden_z_high)], |
| }, |
| "desired_local_extents": { |
| "x": [desired_x_low, desired_x_high], |
| "z": [desired_z_low, desired_z_high], |
| }, |
| "final_local_extents": { |
| "x": [final_x_low, final_x_high], |
| "z": [final_z_low, final_z_high], |
| }, |
| "required_core_clipped": bool( |
| continuous_hidden_coverage_fraction < 0.999 |
| ), |
| "required_width_exceeds_recommended_max": bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and hidden_width_span |
| > category_metric_bounds["width"][1] |
| ), |
| "required_length_exceeds_recommended_max": bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and hidden_length_span |
| > category_metric_bounds["length"][1] |
| ), |
| "camera_field_of_view_scale": field_of_view_scale, |
| "camera_field_of_view_retained_area_fraction": ( |
| field_of_view_retained_area_fraction |
| ), |
| "plane_stability_denominator_floor": plane_stability_floor, |
| "minimum_required_hidden_denominator_abs": ( |
| minimum_required_hidden_denominator |
| ), |
| "camera_field_of_view_crop_applied": not math.isclose( |
| field_of_view_retained_area_fraction, |
| 1.0, |
| rel_tol=1e-8, |
| ), |
| "width_below_recommended_minimum": bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and width < category_metric_bounds["width"][0] |
| ), |
| "length_below_recommended_minimum": bool( |
| metric_style_depth |
| and category_metric_bounds is not None |
| and length < category_metric_bounds["length"][0] |
| ), |
| }, |
| "surface_dimensions": { |
| "width": width, |
| "length": length, |
| "rise": float(rise), |
| "thickness": thickness, |
| "footprint_diag": footprint_diag, |
| "units": ( |
| "metric_style_depth_units_not_calibrated_truth" |
| if metric_style_depth |
| else "relative_pseudo_depth_units" |
| ), |
| }, |
| "depth_scaled_dimensions": { |
| "width": width, |
| "length": length, |
| "height": float(rise), |
| "thickness": thickness, |
| "footprint_diag": footprint_diag, |
| "raw_width": raw_width, |
| "raw_length": raw_length, |
| "dimension_source": ( |
| "asymmetric_hidden_component_local_ray_plane_extent_with_" |
| "category_roi_bounds_and_camera_frustum_polygon_clip" |
| ), |
| "calibrated_metric_truth": False, |
| "depth_units": ( |
| "metric_style_depth_units_not_calibrated_truth" |
| if metric_style_depth |
| else "relative_pseudo_depth_units" |
| ), |
| }, |
| "corridor_profile": profile_metadata, |
| "corridor_grid": { |
| "rows": row_count, |
| "columns": across_count, |
| "front_vertices": front_vertex_count, |
| "rib_count": rib_count if category == "tactile_paving" else 0, |
| "metric_cell_width": float( |
| np.median(row_x_bounds[:, 1] - row_x_bounds[:, 0]) |
| / max(across_count - 1, 1) |
| ), |
| "metric_cell_length": float(length / max(row_count - 1, 1)), |
| "target_metric_cell_size": target_metric_cell, |
| "projected_width_pixels": projected_width, |
| "projected_height_pixels": projected_height, |
| "row_width_min": float( |
| np.min(row_x_bounds[:, 1] - row_x_bounds[:, 0]) |
| ), |
| "row_width_max": float( |
| np.max(row_x_bounds[:, 1] - row_x_bounds[:, 0]) |
| ), |
| "regular_longitudinal_spacing": True, |
| "variable_lateral_width_due_to_camera_frustum_clip": bool( |
| field_of_view_retained_area_fraction < 1.0 - 1e-8 |
| ), |
| }, |
| "geometry_quality": geometry_quality, |
| "local_plane_basis": { |
| "lateral_axis": [float(value) for value in lateral_axis], |
| "longitudinal_axis": [float(value) for value in longitudinal_axis], |
| "normal_axis": [float(value) for value in plane_normal], |
| "context_origin_camera": [float(value) for value in context_origin], |
| "anchor_point_camera": [float(value) for value in anchor_point], |
| "anchor_depth": anchor_depth, |
| "hidden_median_center_offset": [ |
| context_center_x, |
| context_center_z, |
| ], |
| "desired_local_bounds": { |
| "x": [desired_x_low, desired_x_high], |
| "z": [desired_z_low, desired_z_high], |
| }, |
| }, |
| "plane_normal": [float(value) for value in plane_normal], |
| "plane_d": plane_d, |
| "plane_fit_pixels": int(fit_mask.sum()), |
| "plane_fit_residual_median": float(np.median(fit_residuals)), |
| "plane_fit_residual_p95": float(np.percentile(fit_residuals, 95.0)), |
| } |
|
|
|
|
| def build_hybrid_depth_scaled_surface_mesh( |
| depth: np.ndarray, |
| colors: np.ndarray, |
| footprint_mask: np.ndarray, |
| *, |
| category: str, |
| stride: int, |
| minimum_cell_coverage: float = 0.15, |
| plane_validation_mask: np.ndarray | None = None, |
| metric_style_depth: bool = False, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Build a watertight, depth-scaled instance surface with direct RGB. |
| |
| The front topology and every color sample retain the reviewed image-space |
| footprint. Ray/plane reconstruction supplies physical-like lateral and |
| longitudinal dimensions; only accessibility-plausibility bounds are used |
| when the monocular vertical gradient is weak or implausible. A thin neutral |
| underside makes the open-world patch stable in oblique views without |
| stretching photographic texture down legacy walls. |
| """ |
|
|
| top_vertices, top_colors, top_faces, perspective = ( |
| build_perspective_continuous_surface_mesh( |
| depth, |
| colors, |
| footprint_mask, |
| category=category, |
| stride=stride, |
| minimum_cell_coverage=minimum_cell_coverage, |
| plane_validation_mask=plane_validation_mask, |
| texture_support_mask=footprint_mask, |
| ) |
| ) |
| if len(top_vertices) < 4 or len(top_faces) < 2: |
| raise ValueError("Hybrid instance surface is too small") |
|
|
| h, w = depth.shape |
| fx, fy, cx, cy = camera_intrinsics(w, h, None, None, None, None) |
| image_x = np.clip( |
| np.rint(top_vertices[:, 0] * fx / top_vertices[:, 2] + cx), |
| 0, |
| w - 1, |
| ).astype(np.int32) |
| image_y = np.clip( |
| np.rint(top_vertices[:, 1] * fy / top_vertices[:, 2] + cy), |
| 0, |
| h - 1, |
| ).astype(np.int32) |
|
|
| center = top_vertices.mean(axis=0) |
| centered = top_vertices - center |
| pixels = np.stack( |
| ( |
| image_x.astype(np.float64) - float(np.mean(image_x)), |
| image_y.astype(np.float64) - float(np.mean(image_y)), |
| ), |
| axis=1, |
| ) |
| coefficients, _, rank, _ = np.linalg.lstsq( |
| pixels, |
| centered.astype(np.float64), |
| rcond=None, |
| ) |
| axis_x = coefficients[0].astype(np.float64) |
| axis_y = coefficients[1].astype(np.float64) |
| if rank < 2 or np.linalg.norm(axis_x) < 1e-8: |
| _, _, vh = np.linalg.svd(centered, full_matrices=False) |
| axis_x = vh[0].astype(np.float64) |
| axis_y = vh[1].astype(np.float64) |
| axis_x /= max(float(np.linalg.norm(axis_x)), 1e-8) |
| axis_y -= axis_x * float(np.dot(axis_y, axis_x)) |
| if np.linalg.norm(axis_y) < 1e-8: |
| _, _, vh = np.linalg.svd(centered, full_matrices=False) |
| axis_y = vh[1].astype(np.float64) |
| axis_y -= axis_x * float(np.dot(axis_y, axis_x)) |
| axis_y /= max(float(np.linalg.norm(axis_y)), 1e-8) |
|
|
| local_x = centered @ axis_x |
| local_y = centered @ axis_y |
| if np.corrcoef(local_x, image_x.astype(np.float64))[0, 1] < 0: |
| axis_x *= -1.0 |
| local_x *= -1.0 |
| if np.corrcoef(local_y, image_y.astype(np.float64))[0, 1] < 0: |
| axis_y *= -1.0 |
| local_y *= -1.0 |
|
|
| x_low, x_high = np.percentile(local_x, [2.0, 98.0]) |
| y_low, y_high = np.percentile(local_y, [2.0, 98.0]) |
| width = max(float(x_high - x_low), 1e-4) |
| length = max(float(y_high - y_low), 1e-4) |
| local_x_center = 0.5 * float(x_low + x_high) |
| local_y_center = 0.5 * float(y_low + y_high) |
| canonical_x = local_x - local_x_center |
| canonical_y = local_y - local_y_center |
| v = np.clip((local_y - y_low) / length, 0.0, 1.0) |
| u = np.clip((local_x - x_low) / width, 0.0, 1.0) |
|
|
| observed_vertical = -top_vertices[:, 1].astype(np.float64) |
| variance_y = float(np.var(local_y)) |
| vertical_slope = ( |
| float( |
| np.mean( |
| (local_y - float(np.mean(local_y))) |
| * (observed_vertical - float(np.mean(observed_vertical))) |
| ) |
| / max(variance_y, 1e-12) |
| ) |
| if variance_y > 1e-12 |
| else 0.0 |
| ) |
| raw_depth_rise = abs(vertical_slope) * length |
| rise_source = "depth_plane_vertical_gradient" |
| rise_confidence = "medium" |
| if category == "ramp": |
| minimum_rise = 0.015 * length |
| maximum_rise = 0.12 * length |
| rise = float(np.clip(raw_depth_rise, minimum_rise, maximum_rise)) |
| if raw_depth_rise < minimum_rise or raw_depth_rise > maximum_rise: |
| rise_source += "_with_accessibility_bounds" |
| rise_confidence = "low" |
| height = (1.0 - v) * rise |
| category_signature = "depth_scaled_uniform_longitudinal_slope" |
| maximum_semantic_relief = 0.0 |
| elif category == "curb_cut": |
| minimum_rise = 0.012 * length |
| maximum_rise = 0.10 * length |
| rise = float(np.clip(raw_depth_rise, minimum_rise, maximum_rise)) |
| if raw_depth_rise < minimum_rise or raw_depth_rise > maximum_rise: |
| rise_source += "_with_accessibility_bounds" |
| rise_confidence = "low" |
| lateral = np.abs(2.0 * u - 1.0) |
| flare = 0.65 * rise * v * lateral**3 |
| height = (1.0 - v) * rise + flare |
| category_signature = "depth_scaled_central_low_channel_with_lateral_flares" |
| maximum_semantic_relief = float(np.max(flare)) |
| elif category == "tactile_paving": |
| rise = 0.0 |
| rise_source = "near_level_with_scale_bounded_tactile_relief" |
| rise_confidence = "category_verified" |
| rib_height = float(np.clip(0.006 * length, 0.006, 0.018)) |
| rib_phase = 2.0 * math.pi * 10.0 * v |
| height = rib_height * np.maximum(0.0, np.cos(rib_phase)) ** 6 |
| category_signature = "instance_footprint_with_low_transverse_tactile_ribs" |
| maximum_semantic_relief = rib_height |
| else: |
| rise = float(min(raw_depth_rise, 0.02 * length)) |
| height = (1.0 - v) * rise |
| category_signature = "depth_scaled_near_level_instance_walkway" |
| maximum_semantic_relief = 0.0 |
| if raw_depth_rise > 0.02 * length: |
| rise_source += "_near_level_bound" |
| rise_confidence = "low" |
|
|
| transformed_top = np.stack( |
| ( |
| canonical_x, |
| -height, |
| canonical_y, |
| ), |
| axis=1, |
| ).astype(np.float32) |
| footprint_diag = float(math.hypot(width, length)) |
| thickness = float(np.clip(0.018 * footprint_diag, 0.025, 0.22)) |
| back_vertices = transformed_top.copy() |
| back_vertices[:, 1] += thickness |
|
|
| front_vertex_count = len(transformed_top) |
| front_faces = [tuple(int(value) for value in face) for face in top_faces] |
| edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
| for face in front_faces: |
| for a, b in ( |
| (face[0], face[1]), |
| (face[1], face[2]), |
| (face[2], face[0]), |
| ): |
| edge_counts[tuple(sorted((a, b)))] += 1 |
| boundary_edges = [edge for edge, count in edge_counts.items() if count == 1] |
| boundary_adjacency: defaultdict[int, set[int]] = defaultdict(set) |
| for a, b in boundary_edges: |
| boundary_adjacency[a].add(b) |
| boundary_adjacency[b].add(a) |
| if not boundary_adjacency or any( |
| len(neighbors) != 2 for neighbors in boundary_adjacency.values() |
| ): |
| raise ValueError("Hybrid front boundary is not a simple closed loop") |
| remaining_boundary = set(boundary_adjacency) |
| boundary_loop_count = 0 |
| while remaining_boundary: |
| boundary_loop_count += 1 |
| stack = [next(iter(remaining_boundary))] |
| while stack: |
| current = stack.pop() |
| if current not in remaining_boundary: |
| continue |
| remaining_boundary.remove(current) |
| stack.extend(boundary_adjacency[current]) |
| if boundary_loop_count != 1: |
| raise ValueError( |
| "Hybrid instance surface must have exactly one outer boundary; " |
| f"found {boundary_loop_count}" |
| ) |
|
|
| back_faces = [ |
| ( |
| face[2] + front_vertex_count, |
| face[1] + front_vertex_count, |
| face[0] + front_vertex_count, |
| ) |
| for face in front_faces |
| ] |
| side_faces: list[tuple[int, int, int]] = [] |
| for a, b in boundary_edges: |
| side_faces.append((a, b, a + front_vertex_count)) |
| side_faces.append((b, b + front_vertex_count, a + front_vertex_count)) |
| all_faces = front_faces + back_faces + side_faces |
| full_edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
| for face in all_faces: |
| for a, b in ( |
| (face[0], face[1]), |
| (face[1], face[2]), |
| (face[2], face[0]), |
| ): |
| full_edge_counts[tuple(sorted((a, b)))] += 1 |
| watertight_edge_check = bool( |
| full_edge_counts and all(value == 2 for value in full_edge_counts.values()) |
| ) |
| if not watertight_edge_check: |
| raise ValueError("Hybrid instance extrusion failed the watertight edge check") |
|
|
| neutral = np.clip( |
| np.median(top_colors.astype(np.float32), axis=0) * 0.52, |
| 0, |
| 255, |
| ).astype(np.uint8) |
| back_colors = np.repeat(neutral[None, :], front_vertex_count, axis=0) |
| vertices = np.concatenate((transformed_top, back_vertices), axis=0) |
| vertex_colors = np.concatenate((top_colors, back_colors), axis=0) |
| return vertices, vertex_colors, np.asarray(all_faces, dtype=np.int64), { |
| "front_faces": len(front_faces), |
| "back_faces": len(back_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(side_faces), |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": 0, |
| "stair_sidewall_bottom_faces": 0, |
| "solid_thickness": thickness, |
| "max_depth_jump": None, |
| "stride": int(stride), |
| "support_mode": "hybrid_depth_scaled_instance_surface_slab", |
| "surface_model": "hybrid_depth_scaled_instance_accessibility_surface", |
| "complete_footprint": True, |
| "footprint_policy": ( |
| "smoothed_visible_anchored_reviewed_instance_topology_without_hull" |
| ), |
| "texture_coordinate_policy": ( |
| "direct_frozen_completed_rgb_at_original_image_xy_without_donor_" |
| "extrapolation" |
| ), |
| "texture_extrapolation_used": False, |
| "texture_extension_max_distance_pixels": 0.0, |
| "category_geometry_signature": category_signature, |
| "maximum_semantic_relief": maximum_semantic_relief, |
| "front_boundary_loops": int(boundary_loop_count), |
| "front_boundary_edges": int(len(boundary_edges)), |
| "final_cell_holes_filled": bool( |
| perspective.get("final_cell_holes_filled") |
| ), |
| "final_enclosed_hole_added_cells": int( |
| perspective.get("final_enclosed_hole_added_cells") or 0 |
| ), |
| "watertight_edge_check": watertight_edge_check, |
| "calibrated_metric_truth": False, |
| "metric_style_depth_input": bool(metric_style_depth), |
| "automatic_passability_claim": False, |
| "dimension_source": ( |
| "ray_plane_depth_reprojection_robust_extents_with_" |
| "accessibility_plausibility_bounds" |
| ), |
| "rise_source": rise_source, |
| "rise_confidence": rise_confidence, |
| "raw_depth_rise": float(raw_depth_rise), |
| "surface_dimensions": { |
| "width": width, |
| "length": length, |
| "rise": float(rise), |
| "thickness": thickness, |
| "footprint_diag": footprint_diag, |
| "units": ( |
| "metric_style_depth_units_not_calibrated_truth" |
| if metric_style_depth |
| else "relative_pseudo_depth_units" |
| ), |
| }, |
| "depth_scaled_dimensions": { |
| "width": width, |
| "length": length, |
| "height": float(rise), |
| "thickness": thickness, |
| "footprint_diag": footprint_diag, |
| "dimension_source": ( |
| "ray_plane_depth_reprojection_robust_extents_with_" |
| "accessibility_plausibility_bounds" |
| ), |
| "calibrated_metric_truth": False, |
| "depth_units": ( |
| "metric_style_depth_units_not_calibrated_truth" |
| if metric_style_depth |
| else "relative_pseudo_depth_units" |
| ), |
| }, |
| "local_basis": { |
| "image_x_axis": [float(value) for value in axis_x], |
| "image_y_axis": [float(value) for value in axis_y], |
| }, |
| "perspective_front_audit": perspective, |
| } |
|
|
|
|
| def build_canonical_textured_surface_mesh( |
| colors: np.ndarray, |
| target_mask: np.ndarray, |
| *, |
| category: str, |
| stride: int, |
| thickness_ratio: float, |
| complete_footprint: bool = False, |
| contextual_crop: bool = False, |
| context_margin_ratio: float = 0.06, |
| ) -> tuple[np.ndarray, np.ndarray, np.ndarray, dict[str, Any]]: |
| """Build a clean category surface while retaining image-scale RGB texture. |
| |
| ``complete_footprint`` is a qualitative presentation prior for categories |
| whose reviewed mask is a fragmented perspective corridor. It uses the |
| largest rectangular patch fully contained by the hole-filled reviewed |
| surface, so every vertex samples the accepted completed RGB at its original |
| image coordinate. No texture extrapolation or bbox warp is used. |
| |
| ``contextual_crop`` instead keeps a bounded rectangular crop around the |
| reviewed target. This is useful for non-stair surfaces whose semantics are |
| lost when an isolated patch of pavement is rendered: the accepted completed |
| RGB supplies the obstacle-free surroundings, while every mesh vertex still |
| samples its original image coordinate. It is a qualitative scene-context |
| presentation, not a claim that all pixels in the crop belong to the target. |
| """ |
|
|
| if contextual_crop and not complete_footprint: |
| raise ValueError("contextual_crop requires complete_footprint") |
| if context_margin_ratio < 0.0 or context_margin_ratio > 0.5: |
| raise ValueError("context_margin_ratio must be in [0, 0.5]") |
| texture_support = target_mask.astype(bool) |
| if contextual_crop: |
| texture_support = fill_enclosed_mask_holes(texture_support) |
| support_y, support_x = np.nonzero(texture_support) |
| if support_x.size == 0: |
| raise ValueError("No target pixels for contextual textured surface") |
| image_height, image_width = target_mask.shape |
| raw_x_min, raw_x_max = int(support_x.min()), int(support_x.max()) |
| raw_y_min, raw_y_max = int(support_y.min()), int(support_y.max()) |
| margin_x = int(round(image_width * context_margin_ratio)) |
| margin_y = int(round(image_height * context_margin_ratio)) |
| x_min = max(0, raw_x_min - margin_x) |
| x_max = min(image_width - 1, raw_x_max + margin_x) |
| y_min = max(0, raw_y_min - margin_y) |
| y_max = min(image_height - 1, raw_y_max + margin_y) |
| mask_y, mask_x = np.nonzero( |
| np.ones((y_max - y_min + 1, x_max - x_min + 1), dtype=bool) |
| ) |
| elif complete_footprint: |
| texture_support = fill_enclosed_mask_holes(texture_support) |
| x_min, x_max, y_min, y_max = largest_axis_aligned_rectangle( |
| texture_support |
| ) |
| mask_y, mask_x = np.nonzero(texture_support[y_min : y_max + 1, x_min : x_max + 1]) |
| else: |
| mask_y, mask_x = np.nonzero(texture_support) |
| if mask_x.size == 0: |
| raise ValueError("No target pixels for canonical textured surface") |
| if not complete_footprint: |
| x_min, x_max = int(mask_x.min()), int(mask_x.max()) |
| y_min, y_max = int(mask_y.min()), int(mask_y.max()) |
| bbox_width = max(x_max - x_min + 1, 2) |
| bbox_height = max(y_max - y_min + 1, 2) |
| aspect = float(bbox_width / bbox_height) |
| surface_length = 2.4 |
| surface_width = float(surface_length * aspect) |
| category_rise_ratio = { |
| "ramp": 0.24, |
| "curb_cut": 0.10, |
| "walkway": 0.015, |
| "tactile_paving": 0.010, |
| }.get(category, 0.03) |
| total_rise = surface_length * category_rise_ratio |
| thickness = max(surface_length * thickness_ratio * 0.35, 0.035) |
|
|
| xs = list(range(x_min, x_max + 1, max(stride, 1))) |
| ys = list(range(y_min, y_max + 1, max(stride, 1))) |
| if xs[-1] != x_max: |
| xs.append(x_max) |
| if ys[-1] != y_max: |
| ys.append(y_max) |
|
|
| vertex_ids: dict[tuple[int, int], int] = {} |
| vertices: list[np.ndarray] = [] |
| vertex_colors: list[np.ndarray] = [] |
| front_faces: list[tuple[int, int, int]] = [] |
| edge_counts: defaultdict[tuple[int, int], int] = defaultdict(int) |
|
|
| extended_texture_vertices = 0 |
|
|
| def scene_to_camera(point: tuple[float, float, float]) -> np.ndarray: |
| x, y, z = point |
| return np.array([x, -z, y], dtype=np.float32) |
|
|
| def add_vertex(grid_y: int, grid_x: int) -> int: |
| nonlocal extended_texture_vertices |
| key = (grid_y, grid_x) |
| if key in vertex_ids: |
| return vertex_ids[key] |
| u = (grid_x - x_min) / max(float(bbox_width - 1), 1.0) |
| v = (grid_y - y_min) / max(float(bbox_height - 1), 1.0) |
| x = (u - 0.5) * surface_width |
| y = v * surface_length |
| z = (1.0 - v) * total_rise |
| if complete_footprint and category == "curb_cut": |
| |
| |
| |
| lateral = abs(2.0 * u - 1.0) |
| z += total_rise * 0.65 * v * lateral**3 |
| elif complete_footprint and category == "tactile_paving": |
| |
| |
| rib_phase = 2.0 * math.pi * 10.0 * v |
| ridge = max(0.0, math.cos(rib_phase)) ** 6 |
| z += 0.018 * ridge |
| sample_y, sample_x = grid_y, grid_x |
| if complete_footprint and not target_mask[grid_y, grid_x]: |
| |
| |
| |
| extended_texture_vertices += 1 |
| vertex_id = len(vertices) |
| vertex_ids[key] = vertex_id |
| vertices.append(scene_to_camera((x, y, z))) |
| vertex_colors.append(colors[sample_y, sample_x].astype(np.uint8)) |
| return vertex_id |
|
|
| for row_index in range(len(ys) - 1): |
| for column_index in range(len(xs) - 1): |
| y0, y1 = ys[row_index], ys[row_index + 1] |
| x0, x1 = xs[column_index], xs[column_index + 1] |
| center_y = min((y0 + y1) // 2, target_mask.shape[0] - 1) |
| center_x = min((x0 + x1) // 2, target_mask.shape[1] - 1) |
| if ( |
| not complete_footprint |
| and not target_mask[center_y, center_x] |
| ): |
| continue |
| ids = ( |
| add_vertex(y0, x0), |
| add_vertex(y0, x1), |
| add_vertex(y1, x0), |
| add_vertex(y1, x1), |
| ) |
| cell_faces = ( |
| (ids[0], ids[2], ids[1]), |
| (ids[1], ids[2], ids[3]), |
| ) |
| front_faces.extend(cell_faces) |
| for face in cell_faces: |
| for a, b in ( |
| (face[0], face[1]), |
| (face[1], face[2]), |
| (face[2], face[0]), |
| ): |
| edge_counts[tuple(sorted((a, b)))] += 1 |
|
|
| if not front_faces: |
| raise ValueError("Canonical textured surface produced no faces") |
|
|
| front_vertex_count = len(vertices) |
| for vertex_id in range(front_vertex_count): |
| point = vertices[vertex_id].copy() |
| point[1] += thickness |
| vertices.append(point) |
| vertex_colors.append( |
| np.clip( |
| vertex_colors[vertex_id].astype(np.float32) * 0.58, |
| 0, |
| 255, |
| ).astype(np.uint8) |
| ) |
| back_faces = [ |
| ( |
| face[2] + front_vertex_count, |
| face[1] + front_vertex_count, |
| face[0] + front_vertex_count, |
| ) |
| for face in front_faces |
| ] |
| side_faces: list[tuple[int, int, int]] = [] |
| for (a, b), count in edge_counts.items(): |
| if count != 1: |
| continue |
| side_faces.append((a, b, a + front_vertex_count)) |
| side_faces.append( |
| (b, b + front_vertex_count, a + front_vertex_count) |
| ) |
| all_faces = front_faces + back_faces + side_faces |
| metadata = { |
| "front_faces": len(front_faces), |
| "back_faces": len(back_faces), |
| "side_faces": len(side_faces), |
| "boundary_side_faces": len(side_faces), |
| "wedge_support_faces": 0, |
| "stair_sidewall_faces": 0, |
| "stair_sidewall_bottom_faces": 0, |
| "solid_thickness": thickness, |
| "max_depth_jump": None, |
| "stride": stride, |
| "support_mode": "canonical_textured_surface_slab", |
| "stair_surface_mode": "not_stairs", |
| "stair_rows": [], |
| "stair_cross_samples": None, |
| "surface_model": ( |
| "canonical_textured_context_surface" |
| if contextual_crop |
| else "canonical_textured_category_surface" |
| ), |
| "footprint_policy": ( |
| "bounded_context_rectangle_around_hole_filled_reviewed_surface" |
| if contextual_crop |
| else ( |
| "largest_rectangle_inside_hole_filled_reviewed_surface" |
| if complete_footprint |
| else "reviewed_target_mask_cells" |
| ) |
| ), |
| "complete_footprint": bool(complete_footprint), |
| "contextual_crop": bool(contextual_crop), |
| "context_margin_ratio": float(context_margin_ratio), |
| "texture_coordinate_policy": ( |
| "direct_completed_rgb_image_xy_context_crop_without_extrapolation_or_bbox_warp" |
| if contextual_crop |
| else "direct_completed_rgb_image_xy_without_extrapolation_or_bbox_warp" |
| ), |
| "extended_texture_vertices": int(extended_texture_vertices), |
| "texture_extension_max_distance_pixels": 0.0, |
| "texture_extension_mean_distance_pixels": 0.0, |
| "texture_extrapolation_used": False, |
| "context_pixels_are_target_claim": False, |
| "contextual_crop_pixel_count": int(bbox_width * bbox_height), |
| "contextual_crop_target_fraction": float( |
| target_mask[y_min : y_max + 1, x_min : x_max + 1].sum() |
| / max(bbox_width * bbox_height, 1) |
| ), |
| "footprint_source_bounds": { |
| "x_min": int(x_min), |
| "x_max": int(x_max), |
| "y_min": int(y_min), |
| "y_max": int(y_max), |
| }, |
| "calibrated_metric_truth": False, |
| "automatic_passability_claim": False, |
| "category_rise_ratio": category_rise_ratio, |
| "category_geometry_signature": { |
| "ramp": "uniform_longitudinal_slope", |
| "curb_cut": "central_low_channel_with_lateral_flares", |
| "walkway": "near_level_continuous_surface", |
| "tactile_paving": "near_level_surface_with_low_transverse_ribs", |
| }.get(category, "generic_continuous_surface"), |
| "maximum_semantic_relief": ( |
| 0.65 * total_rise |
| if complete_footprint and category == "curb_cut" |
| else ( |
| 0.018 |
| if complete_footprint and category == "tactile_paving" |
| else 0.0 |
| ) |
| ), |
| "surface_dimensions": { |
| "width": surface_width, |
| "length": surface_length, |
| "rise": total_rise, |
| "thickness": thickness, |
| }, |
| } |
| return ( |
| np.asarray(vertices, dtype=np.float32), |
| np.asarray(vertex_colors, dtype=np.uint8), |
| np.asarray(all_faces, dtype=np.int64), |
| metadata, |
| ) |
|
|
|
|
| def write_ply(path: Path, vertices: np.ndarray, colors: np.ndarray, faces: np.ndarray) -> None: |
| path.parent.mkdir(parents=True, exist_ok=True) |
| with path.open("w", encoding="ascii") as handle: |
| handle.write("ply\nformat ascii 1.0\n") |
| handle.write(f"element vertex {len(vertices)}\n") |
| handle.write("property float x\nproperty float y\nproperty float z\n") |
| handle.write("property uchar red\nproperty uchar green\nproperty uchar blue\n") |
| handle.write(f"element face {len(faces)}\n") |
| handle.write("property list uchar int vertex_indices\n") |
| handle.write("end_header\n") |
| for point, color in zip(vertices, colors): |
| handle.write( |
| f"{point[0]:.6f} {point[1]:.6f} {point[2]:.6f} " |
| f"{int(color[0])} {int(color[1])} {int(color[2])}\n" |
| ) |
| for face in faces: |
| handle.write(f"3 {int(face[0])} {int(face[1])} {int(face[2])}\n") |
|
|
|
|
| def write_vertex_color_glb( |
| path: Path, |
| vertices: np.ndarray, |
| colors: np.ndarray, |
| faces: np.ndarray, |
| ) -> dict[str, Any]: |
| import trimesh |
|
|
| rgba = np.concatenate( |
| ( |
| colors.astype(np.uint8), |
| np.full((len(colors), 1), 255, dtype=np.uint8), |
| ), |
| axis=1, |
| ) |
| mesh = trimesh.Trimesh( |
| vertices=vertices, |
| faces=faces, |
| vertex_colors=rgba, |
| process=False, |
| ) |
| path.parent.mkdir(parents=True, exist_ok=True) |
| mesh.export(path) |
| return { |
| "watertight": bool(mesh.is_watertight), |
| "euler_number": int(mesh.euler_number), |
| "vertex_color_texture": True, |
| } |
|
|
|
|
| def render_mesh_frame( |
| vertices: np.ndarray, |
| colors: np.ndarray, |
| faces: np.ndarray, |
| *, |
| size: tuple[int, int], |
| yaw: float, |
| pitch: float, |
| title: str, |
| ) -> Image.Image: |
| width, height = size |
| centered = vertices.astype(np.float32).copy() |
| centered[:, 1] *= -1.0 |
| centered -= np.median(centered, axis=0, keepdims=True) |
| rotated = centered @ rotation_matrix(yaw, pitch).T |
| finite = np.all(np.isfinite(rotated), axis=1) |
| if not np.any(finite): |
| return Image.new("RGB", size, (24, 24, 24)) |
| if faces.size == 0: |
| image = Image.new("RGB", size, (24, 24, 24)) |
| draw = ImageDraw.Draw(image) |
| draw.rectangle((0, 0, width, 18), fill=(245, 245, 245)) |
| draw.text((6, 4), f"{title} | no faces", fill=(0, 0, 0)) |
| return image |
| x_values = rotated[:, 0] |
| y_values = rotated[:, 1] |
| x_low, x_high = np.percentile(x_values[finite], [1, 99]) |
| y_low, y_high = np.percentile(y_values[finite], [1, 99]) |
| x_scale = (width - 36) / max(float(x_high - x_low), 1e-6) |
| y_scale = (height - 46) / max(float(y_high - y_low), 1e-6) |
| scale = min(x_scale, y_scale) |
| x_center = (x_low + x_high) * 0.5 |
| y_center = (y_low + y_high) * 0.5 |
| projected = np.empty((len(vertices), 2), dtype=np.float32) |
| projected[:, 0] = (rotated[:, 0] - x_center) * scale + width * 0.5 |
| projected[:, 1] = height * 0.5 - (rotated[:, 1] - y_center) * scale + 8 |
|
|
| image = Image.new("RGB", size, (24, 24, 24)) |
| draw = ImageDraw.Draw(image) |
| face_depth = rotated[faces].mean(axis=1)[:, 2] |
| order = np.argsort(face_depth) |
| z_low, z_high = np.percentile(face_depth, [5, 95]) if len(face_depth) else (0.0, 1.0) |
| for face_index in order: |
| face = faces[face_index] |
| polygon = [(float(projected[i, 0]), float(projected[i, 1])) for i in face] |
| if all((x < -5 or x > width + 5 or y < 13 or y > height + 5) for x, y in polygon): |
| continue |
| tri = rotated[face] |
| normal = np.cross(tri[1] - tri[0], tri[2] - tri[0]) |
| normal_norm = float(np.linalg.norm(normal)) |
| facing = abs(float(normal[2]) / normal_norm) if normal_norm > 1e-6 else 0.0 |
| depth_weight = np.clip((face_depth[face_index] - z_low) / max(float(z_high - z_low), 1e-6), 0.0, 1.0) |
| shade = 0.52 + 0.28 * float(depth_weight) + 0.20 * facing |
| color = np.clip(colors[face].astype(np.float32).mean(axis=0) * shade, 0, 255).astype(np.uint8) |
| draw.polygon(polygon, fill=tuple(int(v) for v in color)) |
|
|
| draw.rectangle((0, 0, width, 18), fill=(245, 245, 245)) |
| draw.text((6, 4), title, fill=(0, 0, 0)) |
| return image |
|
|
|
|
| def write_turntable_gif( |
| path: Path, |
| vertices: np.ndarray, |
| colors: np.ndarray, |
| faces: np.ndarray, |
| *, |
| sample_id: str, |
| category: str, |
| frames: int, |
| size: tuple[int, int], |
| pitch: float, |
| duration_ms: int, |
| orbit_mode: str = "front_arc", |
| orbit_span: float = 140.0, |
| ) -> None: |
| if orbit_mode not in {"front_arc", "full_360"}: |
| raise ValueError(f"Unsupported preview orbit mode: {orbit_mode}") |
| if orbit_mode == "front_arc" and not 20.0 <= orbit_span <= 240.0: |
| raise ValueError("front_arc preview span must be in [20, 240] degrees") |
| images = [] |
| for index in range(frames): |
| yaw = ( |
| 0.5 |
| * orbit_span |
| * math.sin(2.0 * math.pi * index / frames) |
| if orbit_mode == "front_arc" |
| else 360.0 * index / frames |
| ) |
| images.append( |
| render_mesh_frame( |
| vertices, |
| colors, |
| faces, |
| size=size, |
| yaw=yaw, |
| pitch=pitch, |
| title=f"{sample_id} | {category} | open-world surface", |
| ) |
| ) |
| path.parent.mkdir(parents=True, exist_ok=True) |
| images[0].save( |
| path, |
| save_all=True, |
| append_images=images[1:], |
| optimize=True, |
| duration=duration_ms, |
| loop=0, |
| ) |
|
|
|
|
| def build_html(output_dir: Path, rows: list[dict[str, Any]], summary_rows: list[dict[str, Any]]) -> None: |
| page_title = output_dir.name.replace("_", " ").replace("-", " ").title() |
| cards = [] |
| for row in summary_rows: |
| sample_id = row["sample_id"] |
| gif_markup = ( |
| f'<img src="{html.escape(row["gif"])}" ' |
| f'alt="{html.escape(sample_id)} solid mesh GIF">' |
| if row.get("gif") |
| else '<div class="missing">Turntable render pending</div>' |
| ) |
| cards.append( |
| f""" |
| <article> |
| <h2>{html.escape(sample_id)} <span>{html.escape(row['category'])}</span></h2> |
| {gif_markup} |
| <p> |
| <a href="{html.escape(row['mesh'])}">solid_mesh.ply</a> |
| <a href="{html.escape(row['glb'])}">solid_mesh.glb</a> |
| <a href="{html.escape(row['source_mesh'])}">source completed_mesh.ply</a> |
| <span>{row['vertices']} vertices / {row['faces']} faces</span> |
| <span>{html.escape(row.get('geometry_model') or 'untyped')}</span> |
| <span>{html.escape(row.get('surface_model') or 'depth_surface')}</span> |
| </p> |
| </article>""" |
| ) |
| page = f"""<!doctype html> |
| <html lang="en"> |
| <head> |
| <meta charset="utf-8"> |
| <meta name="viewport" content="width=device-width, initial-scale=1"> |
| <title>{html.escape(page_title)}</title> |
| <style> |
| body {{ margin: 0; font-family: Arial, sans-serif; background: #f3f3f0; color: #151515; }} |
| header {{ padding: 18px 24px; background: #202020; color: white; }} |
| h1 {{ margin: 0 0 6px; font-size: 22px; }} |
| header p {{ margin: 0; color: #d8d8d8; font-size: 13px; line-height: 1.4; }} |
| main {{ padding: 18px; display: grid; grid-template-columns: repeat(auto-fill, minmax(390px, 1fr)); gap: 14px; }} |
| article {{ background: white; border: 1px solid #d0d0cc; border-radius: 6px; overflow: hidden; }} |
| h2 {{ margin: 0; padding: 10px 12px; font-size: 13px; background: #efefea; }} |
| h2 span {{ float: right; color: #606060; font-weight: 400; }} |
| img {{ display: block; width: 100%; height: auto; background: #181818; }} |
| .missing {{ min-height: 220px; display: grid; place-items: center; background: #181818; color: #ddd; }} |
| p {{ margin: 0; padding: 10px 12px; display: flex; gap: 12px; flex-wrap: wrap; font-size: 13px; }} |
| a {{ color: #174a8b; text-decoration: none; }} |
| a:hover {{ text-decoration: underline; }} |
| span {{ color: #505050; }} |
| </style> |
| </head> |
| <body> |
| <header> |
| <h1>{html.escape(page_title)}</h1> |
| <p>Geometry-first open-world accessibility surface patches with trusted stair-count evidence, depth-scaled dimensions, and dense colors sampled from the selected completion. These are review artifacts, not calibrated metric truth, collision meshes, or passability certification.</p> |
| </header> |
| <main> |
| {''.join(cards)} |
| </main> |
| </body> |
| </html> |
| """ |
| (output_dir / "index.html").write_text(page, encoding="utf-8") |
|
|
|
|
| def main() -> int: |
| parser = argparse.ArgumentParser(description=__doc__) |
| parser.add_argument("--results-dir", required=True) |
| parser.add_argument("--manifest", required=True) |
| parser.add_argument("--quality-summary", default=None) |
| parser.add_argument("--taxonomy", default="configs/accessibility_taxonomy.json") |
| parser.add_argument( |
| "--category-overrides", |
| default="configs/accessibility_review_overrides.json", |
| help="Human-reviewed sample category overrides applied before geometry routing.", |
| ) |
| parser.add_argument( |
| "--allow-unreviewed-level-change-categories", |
| action="store_true", |
| help="Legacy escape hatch. By default stairs/ramp/curb categories require explicit review.", |
| ) |
| parser.add_argument("--output-dir", required=True) |
| parser.add_argument( |
| "--texture-package", |
| default=None, |
| help=( |
| "Optional frozen review package whose " |
| "results/<sample_id>/03_2d_completion/completed_rgb_selected.png " |
| "is used only as the mesh texture. The 2D files are read-only." |
| ), |
| ) |
| parser.add_argument( |
| "--sample-ids", |
| default="", |
| help="Optional comma-separated sample ids. When set, preserves this order and ignores limit/per-category selection.", |
| ) |
| parser.add_argument("--limit", type=int, default=12) |
| parser.add_argument("--per-category-limit", type=int, default=6) |
| parser.add_argument("--include-review", action="store_true") |
| parser.add_argument("--stride", type=int, default=6) |
| parser.add_argument("--max-depth-jump", type=float, default=None) |
| parser.add_argument("--thickness", type=float, default=None) |
| parser.add_argument("--thickness-ratio", type=float, default=0.08) |
| parser.add_argument( |
| "--surface-coordinate-mode", |
| choices=[ |
| "relative_depth", |
| "perspective_surface", |
| "hybrid_instance_complete", |
| "canonical_textured", |
| "canonical_textured_complete", |
| "canonical_textured_context", |
| ], |
| default="perspective_surface", |
| help=( |
| "For non-stair continuous categories, use the legacy solid depth " |
| "mesh, a perspective-preserving surface-only patch, a watertight " |
| "depth-scaled instance surface, the legacy canonical presentation " |
| "fallback, a complete category surface, or a bounded contextual " |
| "crop. Stair routing is independent." |
| ), |
| ) |
| parser.add_argument( |
| "--continuous-stride", |
| type=int, |
| default=3, |
| help="Image-grid stride for the perspective continuous front surface.", |
| ) |
| parser.add_argument( |
| "--continuous-minimum-cell-coverage", |
| type=float, |
| default=0.15, |
| help="Minimum target coverage for a perspective surface grid cell.", |
| ) |
| parser.add_argument( |
| "--continuous-support-close-kernel", |
| type=int, |
| default=0, |
| help="Render-only support close kernel; 0 uses 4*--continuous-stride+1.", |
| ) |
| parser.add_argument( |
| "--continuous-obstacle-texture-cleanup", |
| action=argparse.BooleanOptionalAction, |
| default=False, |
| help=( |
| "Optional render-only experiment that replaces obstacle appearance " |
| "with same-surface exemplar texels in memory. Disabled by default: " |
| "the accepted completed RGB is already obstacle-completed, and a " |
| "second texture repair can introduce repeated or blocky pavement. " |
| "The frozen 2D file always remains read-only." |
| ), |
| ) |
| parser.add_argument( |
| "--support-mode", |
| choices=["boundary", "auto", "triangular_wedge", "stair_sidewall", "vertical_slab"], |
| default="boundary", |
| help="Add category-aware structural support faces. auto applies stair_sidewall to stairs and vertical_slab to continuous categories, unless wedge categories are requested.", |
| ) |
| parser.add_argument("--wedge-support-categories", default="ramp,curb_cut") |
| parser.add_argument("--stair-sidewall-categories", default="stairs") |
| parser.add_argument("--wedge-base-drop-ratio", type=float, default=0.10) |
| parser.add_argument( |
| "--stair-surface-mode", |
| choices=["depth_grid", "continuous_step_strips", "canonical_staircase_prism"], |
| default="depth_grid", |
| help="For stair categories, choose the presentation mesh surface model.", |
| ) |
| parser.add_argument( |
| "--stair-cross-samples", |
| type=int, |
| default=0, |
| help="Number of samples across each continuous stair strip. 0 derives it from stride and mask width.", |
| ) |
| parser.add_argument( |
| "--stair-step-count", |
| type=int, |
| default=0, |
| help="Step count for canonical_staircase_prism. 0 derives it from detected edges with a visual minimum.", |
| ) |
| parser.add_argument("--hidden-tint", type=float, default=0.0) |
| parser.add_argument("--mask-close-kernel", type=int, default=5) |
| parser.add_argument("--keep-largest-component", action=argparse.BooleanOptionalAction, default=True) |
| parser.add_argument( |
| "--continuous-surface-categories", |
| default="walkway,ramp,curb_cut,raised_curb,tactile_paving", |
| ) |
| parser.add_argument("--raised-curb-categories", default="raised_curb") |
| parser.add_argument("--raised-curb-height", type=float, default=None) |
| parser.add_argument("--raised-curb-height-ratio", type=float, default=0.18) |
| parser.add_argument("--raised-curb-contour-epsilon-ratio", type=float, default=0.015) |
| parser.add_argument("--frames", type=int, default=48) |
| parser.add_argument( |
| "--render-gif", |
| action=argparse.BooleanOptionalAction, |
| default=False, |
| help=( |
| "Explicitly render the dependency-light CPU preview GIF. The " |
| "production default is GPU-required Open3D/EGL rendering through " |
| "tools/render_accessibility_turntable.py under Slurm." |
| ), |
| ) |
| parser.add_argument("--duration-ms", type=int, default=70) |
| parser.add_argument("--width", type=int, default=420) |
| parser.add_argument("--height", type=int, default=320) |
| parser.add_argument("--pitch", type=float, default=-25.0) |
| args = parser.parse_args() |
|
|
| if ( |
| args.surface_coordinate_mode == "hybrid_instance_complete" |
| and args.render_gif |
| ): |
| parser.error( |
| "hybrid_instance_complete is a production GPU-only path; " |
| "render it with tools/render_accessibility_turntable.py " |
| "inside a Slurm GPU allocation" |
| ) |
|
|
| results_dir = Path(args.results_dir) |
| texture_package = ( |
| Path(args.texture_package).expanduser().resolve() |
| if args.texture_package |
| else None |
| ) |
| if texture_package is not None and not texture_package.is_dir(): |
| raise FileNotFoundError( |
| f"Frozen 2D texture package does not exist: {texture_package}" |
| ) |
| output_dir = Path(args.output_dir) |
| mesh_dir = output_dir / "meshes" |
| gif_dir = output_dir / "gifs" |
| surface_texture_dir = output_dir / "surface_textures" |
| output_dir.mkdir(parents=True, exist_ok=True) |
| mesh_dir.mkdir(parents=True, exist_ok=True) |
| if args.render_gif: |
| gif_dir.mkdir(parents=True, exist_ok=True) |
| surface_texture_dir.mkdir(parents=True, exist_ok=True) |
| taxonomy = load_taxonomy(Path(args.taxonomy) if args.taxonomy else None) |
| category_overrides = load_reviewed_category_overrides( |
| Path(args.category_overrides) if args.category_overrides else None |
| ) |
|
|
| rows = select_rows( |
| read_jsonl(Path(args.manifest)), |
| quality_review_ids(Path(args.quality_summary) if args.quality_summary else None), |
| args.limit, |
| args.per_category_limit, |
| args.include_review, |
| [item.strip() for item in args.sample_ids.split(",") if item.strip()], |
| ) |
|
|
| summary_rows = [] |
| global CONTINUOUS_SURFACE_CATEGORIES |
| CONTINUOUS_SURFACE_CATEGORIES = { |
| item.strip() for item in args.continuous_surface_categories.split(",") if item.strip() |
| } |
| wedge_support_categories = { |
| item.strip() for item in args.wedge_support_categories.split(",") if item.strip() |
| } |
| stair_sidewall_categories = { |
| item.strip() for item in args.stair_sidewall_categories.split(",") if item.strip() |
| } |
| global RAISED_CURB_CATEGORIES |
| RAISED_CURB_CATEGORIES = { |
| item.strip() for item in args.raised_curb_categories.split(",") if item.strip() |
| } |
| for row in rows: |
| sample_id = row["sample_id"] |
| source_category = str( |
| row.get("source_category") or row.get("category") or "unknown" |
| ) |
| category_override = category_overrides.get(sample_id, {}) |
| category = resolve_geometry_category_for_modeling( |
| sample_id, |
| source_category, |
| row, |
| category_overrides, |
| allow_unreviewed_level_change=( |
| args.allow_unreviewed_level_change_categories |
| or category_override.get("geometry_profile_reviewed") is True |
| ), |
| ) |
| geometry_profile = category_override.get("geometry_profile") |
| category_shape_enabled = bool( |
| category_override.get("category_shape_enabled", True) |
| ) |
| modeled_geometry_category = category |
| if not category_shape_enabled and geometry_profile in { |
| "generic_walkway_no_synthetic_tactile_relief", |
| "platform_walkway_without_synthetic_curb_cut", |
| "crossing_walkway_without_synthetic_curb_cut", |
| }: |
| modeled_geometry_category = "walkway" |
| category_info = taxonomy.get(category, {}) |
| sample_dir = results_dir / sample_id |
| geometry_dir = sample_dir / "geometry" |
| depth = np.load(geometry_dir / "completed_depth.npy").astype(np.float32) |
| h, w = depth.shape |
| raw_target_mask = read_mask(geometry_dir / "amodal_target_mask.png", (h, w)) |
| visible_path = geometry_dir / "target_visible_mask.png" |
| visible_mask = read_mask(visible_path, (h, w)) if visible_path.is_file() else None |
| hidden_path = geometry_dir / "hidden_completion_mask.png" |
| hidden_mask = read_mask(hidden_path, (h, w)) if hidden_path.is_file() else None |
| target_mask = clean_target_mask( |
| raw_target_mask, |
| args.mask_close_kernel, |
| args.keep_largest_component, |
| ) |
| mesh_target_mask = target_mask |
| surface_support_metadata: dict[str, Any] = { |
| "role": "not_applied", |
| "support_pixels": int(target_mask.sum()), |
| "support_added_pixels": 0, |
| "generation_mask_used": False, |
| "annotation_mask_modified": False, |
| } |
| full_obstacle_mask: np.ndarray | None = None |
| if ( |
| args.surface_coordinate_mode == "perspective_surface" |
| and category in CONTINUOUS_SURFACE_CATEGORIES |
| and category not in RAISED_CURB_CATEGORIES |
| ): |
| full_obstacle_path = sample_dir / "obstacle_mask.png" |
| full_obstacle_mask = ( |
| read_mask_nearest(full_obstacle_path, (h, w)) |
| if full_obstacle_path.is_file() |
| else None |
| ) |
| mesh_target_mask, surface_support_metadata = ( |
| build_continuous_surface_support_mask( |
| raw_target_mask, |
| stride=args.continuous_stride, |
| close_kernel=args.continuous_support_close_kernel, |
| visible_mask=visible_mask, |
| hidden_mask=hidden_mask, |
| obstacle_mask=full_obstacle_mask, |
| ) |
| ) |
| depth, surface_metadata = continuous_depth_for_category( |
| depth, |
| mesh_target_mask, |
| visible_mask, |
| category, |
| ) |
| surface_metadata = { |
| **surface_metadata, |
| "geometry_support": surface_support_metadata, |
| } |
| frozen_texture_path = ( |
| texture_package |
| / "results" |
| / sample_id |
| / "03_2d_completion" |
| / "completed_rgb_selected.png" |
| if texture_package is not None |
| else None |
| ) |
| colors, texture_source = prepare_texture( |
| sample_dir, |
| geometry_dir, |
| (h, w), |
| hidden_tint=args.hidden_tint, |
| texture_override=frozen_texture_path, |
| ) |
| surface_texture_path: Path | None = None |
| surface_texture_metadata: dict[str, Any] = { |
| "role": "not_applied", |
| "source_file_modified": False, |
| "frozen_2d_consumed_read_only": True, |
| } |
| if ( |
| args.continuous_obstacle_texture_cleanup |
| and args.surface_coordinate_mode == "perspective_surface" |
| and category in CONTINUOUS_SURFACE_CATEGORIES |
| and category not in RAISED_CURB_CATEGORIES |
| ): |
| colors, surface_texture_metadata = ( |
| complete_continuous_surface_texture( |
| colors, |
| mesh_target_mask, |
| full_obstacle_mask, |
| stride=args.continuous_stride, |
| ) |
| ) |
| surface_texture_path = ( |
| surface_texture_dir / f"{sample_id}_surface_texture.png" |
| ) |
| Image.fromarray(colors, mode="RGB").save(surface_texture_path) |
| texture_source += " + render_only_same_surface_obstacle_exemplar" |
| surface_metadata = { |
| **surface_metadata, |
| "render_only_texture_completion": surface_texture_metadata, |
| } |
| if args.support_mode == "auto": |
| if category in stair_sidewall_categories: |
| support_mode = "stair_sidewall" |
| elif category in wedge_support_categories: |
| support_mode = "triangular_wedge" |
| elif category in CONTINUOUS_SURFACE_CATEGORIES: |
| support_mode = "vertical_slab" |
| else: |
| support_mode = "boundary" |
| else: |
| support_mode = args.support_mode |
| geometry_manifest_path = geometry_dir / "geometry_manifest.json" |
| geometry_manifest = read_json(geometry_manifest_path) if geometry_manifest_path.is_file() else {} |
| depth_manifest_path = sample_dir / "depth_manifest.json" |
| depth_manifest = ( |
| read_json(depth_manifest_path) |
| if depth_manifest_path.is_file() |
| else {} |
| ) |
| metric_style_depth = bool(depth_manifest.get("metric_depth", False)) |
| stair_edges_y, stair_edge_provenance = ( |
| trusted_stair_edges_from_manifest(geometry_manifest) |
| ) |
| depth_output = depth_manifest.get("output_depth") |
| geometry_depth_source = geometry_manifest.get("depth_source") |
| depth_source_matches_manifest = bool( |
| depth_output |
| and geometry_depth_source |
| and Path(depth_output).expanduser().resolve() |
| == Path(geometry_depth_source).expanduser().resolve() |
| ) |
| checkpoint_value = depth_manifest.get("checkpoint") |
| checkpoint_path = ( |
| Path(checkpoint_value).expanduser() |
| if checkpoint_value |
| else None |
| ) |
| depth_model_identity = { |
| "engine": depth_manifest.get("engine"), |
| "encoder": depth_manifest.get("encoder"), |
| "checkpoint": ( |
| str(checkpoint_path) if checkpoint_path is not None else None |
| ), |
| "checkpoint_sha256": ( |
| sha256_file(checkpoint_path) |
| if checkpoint_path is not None and checkpoint_path.is_file() |
| else None |
| ), |
| "metric_style_depth": metric_style_depth, |
| "max_depth": depth_manifest.get("max_depth"), |
| "geometry_depth_source": geometry_depth_source, |
| "manifest_output_depth": depth_output, |
| "geometry_depth_source_matches_manifest": ( |
| depth_source_matches_manifest |
| ), |
| "selection_policy": ( |
| "best_available_local_single_image_metric_depth_model" |
| if ( |
| depth_manifest.get("engine") == "depth_anything_v2" |
| and metric_style_depth |
| and checkpoint_path is not None |
| and "metric_hypersim_vitl" in checkpoint_path.name |
| ) |
| else "input_pipeline_depth_model" |
| ), |
| } |
| if category in RAISED_CURB_CATEGORIES: |
| sample_curb_height = category_override.get("raised_curb_height", args.raised_curb_height) |
| sample_curb_height_ratio = float( |
| category_override.get("raised_curb_height_ratio", args.raised_curb_height_ratio) |
| ) |
| sample_contour_epsilon_ratio = float( |
| category_override.get( |
| "raised_curb_contour_epsilon_ratio", |
| args.raised_curb_contour_epsilon_ratio, |
| ) |
| ) |
| vertices, vertex_colors, faces, metadata = build_raised_curb_prism_mesh( |
| depth, |
| colors, |
| target_mask, |
| height=None if sample_curb_height is None else float(sample_curb_height), |
| height_ratio=sample_curb_height_ratio, |
| contour_epsilon_ratio=sample_contour_epsilon_ratio, |
| stride=args.stride, |
| ) |
| surface_metadata = { |
| **surface_metadata, |
| "surface_model": metadata["surface_model"], |
| "curb_height": metadata["curb_height"], |
| "contour_vertices": metadata["contour_vertices"], |
| "raw_contour_vertices": metadata["raw_contour_vertices"], |
| } |
| elif args.stair_surface_mode == "canonical_staircase_prism" and category in stair_sidewall_categories: |
| vertices, vertex_colors, faces, metadata = build_canonical_staircase_prism_mesh( |
| depth, |
| colors, |
| target_mask, |
| stride=args.stride, |
| thickness=args.thickness, |
| thickness_ratio=args.thickness_ratio, |
| stair_edges_y=stair_edges_y, |
| step_count=int( |
| category_override.get( |
| "stair_step_count", |
| args.stair_step_count, |
| ) |
| ), |
| stair_edge_slope=float( |
| geometry_manifest.get("stair_edge_slope", 0.0) |
| ), |
| metric_style_depth=metric_style_depth, |
| ) |
| elif ( |
| args.stair_surface_mode == "continuous_step_strips" |
| and category in stair_sidewall_categories |
| and len(stair_edges_y) > 0 |
| ): |
| vertices, vertex_colors, faces, metadata = build_continuous_stair_mesh( |
| depth, |
| colors, |
| target_mask, |
| stride=args.stride, |
| thickness=args.thickness, |
| thickness_ratio=args.thickness_ratio, |
| support_mode=support_mode, |
| wedge_base_drop_ratio=args.wedge_base_drop_ratio, |
| stair_edges_y=stair_edges_y, |
| stair_edge_slope=float(geometry_manifest.get("stair_edge_slope", 0.0)), |
| cross_samples=args.stair_cross_samples, |
| ) |
| elif ( |
| args.surface_coordinate_mode == "perspective_surface" |
| and category in CONTINUOUS_SURFACE_CATEGORIES |
| ): |
| vertices, vertex_colors, faces, metadata = ( |
| build_perspective_continuous_surface_mesh( |
| depth, |
| colors, |
| mesh_target_mask, |
| category=category, |
| stride=args.continuous_stride, |
| minimum_cell_coverage=args.continuous_minimum_cell_coverage, |
| plane=( |
| ( |
| np.asarray( |
| surface_metadata["plane_normal"], |
| dtype=np.float32, |
| ), |
| float(surface_metadata["plane_d"]), |
| ) |
| if ( |
| surface_metadata.get("plane_normal") is not None |
| and surface_metadata.get("plane_d") is not None |
| ) |
| else None |
| ), |
| plane_validation_mask=( |
| visible_mask |
| if visible_mask is not None and np.any(visible_mask) |
| else raw_target_mask |
| ), |
| |
| |
| |
| |
| texture_support_mask=mesh_target_mask, |
| must_retain_mask=hidden_mask, |
| bridge_evidence_mask=full_obstacle_mask, |
| ) |
| ) |
| surface_metadata = { |
| **surface_metadata, |
| "surface_model": metadata["surface_model"], |
| "texture_policy": metadata["texture_policy"], |
| "texture_coordinate_policy": metadata[ |
| "texture_coordinate_policy" |
| ], |
| "surface_cells": metadata["surface_cells"], |
| "cell_components_after": metadata["cell_components_after"], |
| "hidden_cell_coverage_fraction": metadata[ |
| "hidden_cell_coverage_fraction" |
| ], |
| "bridge_count": metadata["bridge_count"], |
| "bridge_added_cells": metadata["bridge_added_cells"], |
| "discarded_tiny_cells_without_hidden_evidence": metadata[ |
| "discarded_tiny_cells_without_hidden_evidence" |
| ], |
| "detached_rejected_cells": metadata[ |
| "detached_rejected_cells" |
| ], |
| "detached_rejected_hidden_cells": metadata[ |
| "detached_rejected_hidden_cells" |
| ], |
| "boundary_clamped_vertices": metadata[ |
| "boundary_clamped_vertices" |
| ], |
| "texture_completion_vertices": metadata[ |
| "texture_completion_vertices" |
| ], |
| "texture_completion_max_distance_pixels": metadata[ |
| "texture_completion_max_distance_pixels" |
| ], |
| "texture_completion_mean_distance_pixels": metadata[ |
| "texture_completion_mean_distance_pixels" |
| ], |
| "texture_completion_policy": metadata[ |
| "texture_completion_policy" |
| ], |
| "depth_clipped_vertices": metadata[ |
| "depth_clipped_vertices" |
| ], |
| "plane_plausible_fraction": metadata[ |
| "plane_plausible_fraction" |
| ], |
| "plane_depth_policy": metadata["plane_depth_policy"], |
| } |
| elif ( |
| args.surface_coordinate_mode == "hybrid_instance_complete" |
| and category in CONTINUOUS_SURFACE_CATEGORIES |
| and category not in RAISED_CURB_CATEGORIES |
| ): |
| mesh_target_mask, surface_support_metadata = ( |
| build_hybrid_instance_footprint_mask( |
| target_mask, |
| category=modeled_geometry_category, |
| stride=args.continuous_stride, |
| visible_mask=visible_mask, |
| hidden_mask=hidden_mask, |
| ) |
| ) |
| vertices, vertex_colors, faces, metadata = ( |
| build_obstacle_centered_depth_scaled_corridor_mesh( |
| depth, |
| colors, |
| mesh_target_mask, |
| category=modeled_geometry_category, |
| stride=args.continuous_stride, |
| visible_mask=visible_mask, |
| hidden_mask=hidden_mask, |
| plane_validation_mask=( |
| visible_mask & mesh_target_mask |
| if visible_mask is not None and np.any(visible_mask) |
| else mesh_target_mask |
| ), |
| metric_style_depth=metric_style_depth, |
| ) |
| ) |
| surface_metadata = { |
| **surface_metadata, |
| "geometry_support": surface_support_metadata, |
| "source_semantic_category": category, |
| "modeled_geometry_category": modeled_geometry_category, |
| "geometry_profile": geometry_profile, |
| "geometry_profile_source": category_override.get( |
| "geometry_profile_source" |
| ), |
| "category_shape_enabled": category_shape_enabled, |
| "surface_model": metadata["surface_model"], |
| "support_mode": metadata["support_mode"], |
| "footprint_policy": metadata["footprint_policy"], |
| "complete_footprint": metadata["complete_footprint"], |
| "category_geometry_signature": metadata[ |
| "category_geometry_signature" |
| ], |
| "maximum_semantic_relief": metadata[ |
| "maximum_semantic_relief" |
| ], |
| "texture_coordinate_policy": metadata[ |
| "texture_coordinate_policy" |
| ], |
| "texture_mapping": metadata["texture_mapping"], |
| "texture_geometry_warp": metadata[ |
| "texture_geometry_warp" |
| ], |
| "independent_axis_normalization_used": metadata[ |
| "independent_axis_normalization_used" |
| ], |
| "uniform_geometry_scale": metadata[ |
| "uniform_geometry_scale" |
| ], |
| "uniform_geometry_scale_applied": metadata[ |
| "uniform_geometry_scale_applied" |
| ], |
| "texture_extrapolation_used": metadata[ |
| "texture_extrapolation_used" |
| ], |
| "texture_resampling": metadata["texture_resampling"], |
| "texture_extension_max_distance_pixels": metadata[ |
| "texture_extension_max_distance_pixels" |
| ], |
| "surface_dimensions": metadata["surface_dimensions"], |
| "dimension_source": metadata["dimension_source"], |
| "rise_source": metadata["rise_source"], |
| "rise_confidence": metadata["rise_confidence"], |
| "raw_depth_rise": metadata["raw_depth_rise"], |
| "calibrated_metric_truth": metadata[ |
| "calibrated_metric_truth" |
| ], |
| "metric_style_depth_input": metadata[ |
| "metric_style_depth_input" |
| ], |
| "automatic_passability_claim": metadata[ |
| "automatic_passability_claim" |
| ], |
| "front_boundary_loops": metadata["front_boundary_loops"], |
| "final_cell_holes_filled": metadata[ |
| "final_cell_holes_filled" |
| ], |
| "final_enclosed_hole_added_cells": metadata[ |
| "final_enclosed_hole_added_cells" |
| ], |
| "watertight_edge_check": metadata[ |
| "watertight_edge_check" |
| ], |
| "complete_obstacle_zone": metadata[ |
| "complete_obstacle_zone" |
| ], |
| "open_world_channel_model": metadata[ |
| "open_world_channel_model" |
| ], |
| "local_context_patch_model": metadata[ |
| "local_context_patch_model" |
| ], |
| "open_world_scene_compatible": metadata[ |
| "open_world_scene_compatible" |
| ], |
| "open_world_topology_claim": metadata[ |
| "open_world_topology_claim" |
| ], |
| "context_pixels_are_target_claim": metadata[ |
| "context_pixels_are_target_claim" |
| ], |
| "projected_context_hidden_pixel_coverage_fraction": metadata[ |
| "projected_context_hidden_pixel_coverage_fraction" |
| ], |
| "projected_context_target_fraction": metadata[ |
| "projected_context_target_fraction" |
| ], |
| "projected_context_hidden_component_coverage": metadata[ |
| "projected_context_hidden_component_coverage" |
| ], |
| "projected_context_minimum_hidden_component_coverage_fraction": metadata[ |
| "projected_context_minimum_hidden_component_coverage_fraction" |
| ], |
| "continuous_local_coordinate_hidden_coverage_fraction": metadata[ |
| "continuous_local_coordinate_hidden_coverage_fraction" |
| ], |
| "minimum_continuous_hidden_component_coverage_fraction": metadata[ |
| "minimum_continuous_hidden_component_coverage_fraction" |
| ], |
| "all_hidden_components_covered": metadata[ |
| "all_hidden_components_covered" |
| ], |
| "projected_context_image_corners": metadata[ |
| "projected_context_image_corners" |
| ], |
| "projected_context_image_polygon": metadata[ |
| "projected_context_image_polygon" |
| ], |
| "context_roi_image_xyxy": metadata["context_roi_image_xyxy"], |
| "context_roi_policy": metadata["context_roi_policy"], |
| "regular_grid_has_no_internal_topological_holes": metadata[ |
| "regular_grid_has_no_internal_topological_holes" |
| ], |
| "corridor_profile": metadata["corridor_profile"], |
| "corridor_grid": metadata["corridor_grid"], |
| "geometry_quality": metadata["geometry_quality"], |
| "local_plane_basis": metadata["local_plane_basis"], |
| "raw_depth_scaled_dimensions": metadata[ |
| "raw_depth_scaled_dimensions" |
| ], |
| "dimension_bounds": metadata["dimension_bounds"], |
| "plane_fit_residual_median": metadata[ |
| "plane_fit_residual_median" |
| ], |
| "plane_fit_residual_p95": metadata["plane_fit_residual_p95"], |
| "texture_profile_inside_reviewed_footprint_fraction": metadata[ |
| "texture_profile_inside_reviewed_footprint_fraction" |
| ], |
| "texture_profile_hidden_completion_fraction": metadata[ |
| "texture_profile_hidden_completion_fraction" |
| ], |
| } |
| elif ( |
| args.surface_coordinate_mode in { |
| "canonical_textured", |
| "canonical_textured_complete", |
| "canonical_textured_context", |
| } |
| and category in CONTINUOUS_SURFACE_CATEGORIES |
| ): |
| vertices, vertex_colors, faces, metadata = ( |
| build_canonical_textured_surface_mesh( |
| colors, |
| target_mask, |
| category=category, |
| stride=args.stride, |
| thickness_ratio=args.thickness_ratio, |
| complete_footprint=( |
| args.surface_coordinate_mode |
| in { |
| "canonical_textured_complete", |
| "canonical_textured_context", |
| } |
| ), |
| contextual_crop=( |
| args.surface_coordinate_mode |
| == "canonical_textured_context" |
| ), |
| ) |
| ) |
| surface_metadata = { |
| **surface_metadata, |
| "surface_model": metadata["surface_model"], |
| "category_rise_ratio": metadata["category_rise_ratio"], |
| "category_geometry_signature": metadata[ |
| "category_geometry_signature" |
| ], |
| "maximum_semantic_relief": metadata[ |
| "maximum_semantic_relief" |
| ], |
| "surface_dimensions": metadata["surface_dimensions"], |
| "footprint_policy": metadata["footprint_policy"], |
| "complete_footprint": metadata["complete_footprint"], |
| "contextual_crop": metadata["contextual_crop"], |
| "context_margin_ratio": metadata["context_margin_ratio"], |
| "context_pixels_are_target_claim": metadata[ |
| "context_pixels_are_target_claim" |
| ], |
| "contextual_crop_pixel_count": metadata[ |
| "contextual_crop_pixel_count" |
| ], |
| "contextual_crop_target_fraction": metadata[ |
| "contextual_crop_target_fraction" |
| ], |
| "texture_coordinate_policy": metadata[ |
| "texture_coordinate_policy" |
| ], |
| "extended_texture_vertices": metadata[ |
| "extended_texture_vertices" |
| ], |
| "texture_extension_max_distance_pixels": metadata[ |
| "texture_extension_max_distance_pixels" |
| ], |
| "texture_extension_mean_distance_pixels": metadata[ |
| "texture_extension_mean_distance_pixels" |
| ], |
| "texture_extrapolation_used": metadata[ |
| "texture_extrapolation_used" |
| ], |
| "footprint_source_bounds": metadata[ |
| "footprint_source_bounds" |
| ], |
| "calibrated_metric_truth": metadata[ |
| "calibrated_metric_truth" |
| ], |
| "automatic_passability_claim": metadata[ |
| "automatic_passability_claim" |
| ], |
| } |
| else: |
| vertices, vertex_colors, faces, metadata = build_solid_mesh( |
| depth, |
| colors, |
| target_mask, |
| stride=args.stride, |
| max_depth_jump=args.max_depth_jump, |
| thickness=args.thickness, |
| thickness_ratio=args.thickness_ratio, |
| support_mode=support_mode, |
| wedge_base_drop_ratio=args.wedge_base_drop_ratio, |
| category=category, |
| stair_edges_y=stair_edges_y, |
| ) |
| metadata["stair_edge_provenance"] = stair_edge_provenance |
| mesh_path = mesh_dir / f"{sample_id}_solid_mesh.ply" |
| glb_path = mesh_dir / f"{sample_id}_solid_mesh.glb" |
| gif_path = gif_dir / f"{sample_id}_solid_mesh.gif" |
| write_ply(mesh_path, vertices, vertex_colors, faces) |
| glb_info = write_vertex_color_glb( |
| glb_path, |
| vertices, |
| vertex_colors, |
| faces, |
| ) |
| if args.render_gif: |
| write_turntable_gif( |
| gif_path, |
| vertices, |
| vertex_colors, |
| faces, |
| sample_id=sample_id, |
| category=category, |
| frames=args.frames, |
| size=(args.width, args.height), |
| pitch=args.pitch, |
| duration_ms=args.duration_ms, |
| ) |
| summary_rows.append( |
| { |
| "sample_id": sample_id, |
| "category": category, |
| "source_category": source_category, |
| "category_override_applied": category != source_category, |
| "geometry_profile": category_override.get("geometry_profile"), |
| "vertices": int(len(vertices)), |
| "faces": int(len(faces)), |
| "front_faces": metadata["front_faces"], |
| "back_faces": metadata["back_faces"], |
| "side_faces": metadata["side_faces"], |
| "boundary_side_faces": metadata["boundary_side_faces"], |
| "wedge_support_faces": metadata["wedge_support_faces"], |
| "stair_sidewall_faces": metadata["stair_sidewall_faces"], |
| "stair_sidewall_bottom_faces": metadata["stair_sidewall_bottom_faces"], |
| "solid_thickness": metadata["solid_thickness"], |
| "max_depth_jump": metadata["max_depth_jump"], |
| "stride": metadata["stride"], |
| "support_mode": metadata["support_mode"], |
| "vertical_drop": metadata.get("vertical_drop"), |
| "vertical_base_y": metadata.get("vertical_base_y"), |
| "depth_scaled_dimensions": metadata.get("depth_scaled_dimensions", {}), |
| "stair_surface_mode": metadata.get("stair_surface_mode", "depth_grid"), |
| "stair_rows": metadata.get("stair_rows", []), |
| "stair_cross_samples": metadata.get("stair_cross_samples"), |
| "stair_edge_slope": metadata.get("stair_edge_slope"), |
| "stair_step_count": metadata.get("stair_step_count"), |
| "stair_step_count_source": metadata.get("stair_step_count_source"), |
| "stair_inference": metadata.get("stair_inference"), |
| "stair_edge_provenance": stair_edge_provenance, |
| "stair_physical_edge_rows": metadata.get( |
| "stair_physical_edge_rows", |
| [], |
| ), |
| "curb_height": metadata.get("curb_height"), |
| "curb_height_ratio": metadata.get("curb_height_ratio"), |
| "curb_height_source": metadata.get("curb_height_source"), |
| "contour_vertices": metadata.get("contour_vertices"), |
| "raw_contour_vertices": metadata.get("raw_contour_vertices"), |
| "surface_model": surface_metadata.get("surface_model"), |
| "surface_metadata": surface_metadata, |
| "stair_edges_y": stair_edges_y, |
| "raw_stair_edges_y": stair_edge_provenance["raw_rows"], |
| "geometry_model": category_info.get("geometry_model", "untyped_depth_surface"), |
| "primary_users": category_info.get("primary_users", []), |
| "texture_source": texture_source, |
| "texture_sha256": ( |
| sha256_file(frozen_texture_path) |
| if frozen_texture_path is not None |
| else None |
| ), |
| "texture_package": ( |
| str(texture_package) if texture_package is not None else None |
| ), |
| "surface_texture": ( |
| str(Path("surface_textures") / surface_texture_path.name) |
| if surface_texture_path is not None |
| else None |
| ), |
| "surface_texture_sha256": ( |
| sha256_file(surface_texture_path) |
| if surface_texture_path is not None |
| else None |
| ), |
| "surface_texture_metadata": surface_texture_metadata, |
| "raw_target_pixels": int(raw_target_mask.sum()), |
| "clean_target_pixels": int(target_mask.sum()), |
| "render_support_pixels": int(mesh_target_mask.sum()), |
| "render_support_added_pixels": int( |
| mesh_target_mask.sum() - raw_target_mask.sum() |
| ), |
| "render_support_metadata": surface_support_metadata, |
| "hidden_cell_coverage_fraction": metadata.get( |
| "hidden_cell_coverage_fraction" |
| ), |
| "surface_cell_components": metadata.get( |
| "cell_components_after" |
| ), |
| "surface_bridge_count": metadata.get("bridge_count"), |
| "surface_bridge_added_cells": metadata.get( |
| "bridge_added_cells" |
| ), |
| "surface_detached_rejected_cells": metadata.get( |
| "detached_rejected_cells" |
| ), |
| "surface_detached_rejected_hidden_cells": metadata.get( |
| "detached_rejected_hidden_cells" |
| ), |
| "surface_boundary_clamped_vertices": metadata.get( |
| "boundary_clamped_vertices" |
| ), |
| "surface_depth_clipped_vertices": metadata.get( |
| "depth_clipped_vertices" |
| ), |
| "surface_plane_plausible_fraction": metadata.get( |
| "plane_plausible_fraction" |
| ), |
| "mesh": str(Path("meshes") / mesh_path.name), |
| "glb": str(Path("meshes") / glb_path.name), |
| "glb_info": glb_info, |
| "gif": ( |
| str(Path("gifs") / gif_path.name) |
| if args.render_gif |
| else None |
| ), |
| "source_mesh": os.path.relpath(geometry_dir / "completed_mesh.ply", output_dir), |
| "depth_is_metric": False, |
| "depth_model_is_metric_style": metric_style_depth, |
| "depth_model_identity": depth_model_identity, |
| "amodal_3d_completion_is_ground_truth": False, |
| } |
| ) |
|
|
| summary = { |
| "sample_count": len(summary_rows), |
| "rows": summary_rows, |
| "depth_is_metric": False, |
| "amodal_3d_completion_is_ground_truth": False, |
| "continuous_surface_categories": sorted(CONTINUOUS_SURFACE_CATEGORIES), |
| "wedge_support_categories": sorted(wedge_support_categories), |
| "stair_sidewall_categories": sorted(stair_sidewall_categories), |
| "raised_curb_categories": sorted(RAISED_CURB_CATEGORIES), |
| "stair_surface_mode": args.stair_surface_mode, |
| "surface_coordinate_mode": args.surface_coordinate_mode, |
| "taxonomy": str(Path(args.taxonomy)) if args.taxonomy else None, |
| "texture_policy": ( |
| "frozen quality-gated completed RGB used read-only; perspective-aware " |
| "dense vertex sampling; no semantic tint" |
| if texture_package is not None and args.hidden_tint <= 0 |
| else ( |
| "clean completed RGB with hidden-region inpainting; no mask-color tint" |
| if args.hidden_tint <= 0 |
| else f"clean completed RGB with hidden-region inpainting and hidden tint {args.hidden_tint:.3f}" |
| ) |
| ), |
| "texture_package": ( |
| str(texture_package) if texture_package is not None else None |
| ), |
| "solid_mesh_version": ( |
| "v14_obstacle_centered_metric_context_surface_gpu_required" |
| if args.surface_coordinate_mode == "hybrid_instance_complete" |
| else "v10_continuous_surface_gpu_required" |
| ), |
| "preview_orbit": { |
| "mode": "front_arc", |
| "span_degrees": 140.0, |
| "fixed_projection_scale": False, |
| "role": ( |
| "cpu_preview_forbidden_for_hybrid_gpu_only_production_path" |
| if args.surface_coordinate_mode == "hybrid_instance_complete" |
| else "optional_dependency_light_cpu_preview_disabled_by_default" |
| ), |
| }, |
| "production_renderer_policy": ( |
| "physical_nvidia_gpu_required_fail_closed_no_implicit_cpu_fallback" |
| ), |
| "allow_unreviewed_level_change_categories": args.allow_unreviewed_level_change_categories, |
| "note": ( |
| ( |
| "Hybrid non-stair surfaces retain a smoothed, visible-anchored " |
| "reviewed instance, then follow the local travel channel crossing " |
| "the hidden obstacle zone instead of turning every horizon branch " |
| "into a long standalone shard. A regular depth-scaled metric grid is " |
| "built in the fitted local surface plane and reprojected through the " |
| "camera model into the frozen accepted 2D completion. This keeps a " |
| "bounded amount of path context without independently normalizing " |
| "texture axes. The local context slab is watertight and uses no " |
| "second obstacle cleanup, donor extrapolation, fog, or repeated " |
| "texture. It does not claim that all context pixels are target " |
| "surface or that the patch recovers full open-world topology. " |
| "Production rendering requires a verified physical " |
| "NVIDIA GPU and fails closed instead of falling back to CPU. Values " |
| "remain monocular metric-style estimates rather " |
| "than calibrated physical truth or passability certification." |
| ) |
| if args.surface_coordinate_mode == "hybrid_instance_complete" |
| else ( |
| "Geometry-first accessibility review surfaces. Stair count is " |
| "filtered from trusted detector provenance plus RGB/depth riser " |
| "evidence. Frozen 2D output is used read-only for appearance; " |
| "geometry and dimensions remain model estimates rather than " |
| "calibrated physical truth." |
| ) |
| ), |
| } |
| (output_dir / "summary.json").write_text( |
| json.dumps(summary, ensure_ascii=False, indent=2, sort_keys=True) + "\n", |
| encoding="utf-8", |
| ) |
| build_html(output_dir, rows, summary_rows) |
| print(json.dumps({"output": str(output_dir), "sample_count": len(summary_rows)}, ensure_ascii=False)) |
| return 0 |
|
|
|
|
| if __name__ == "__main__": |
| raise SystemExit(main()) |
|
|