"""CRE-ROI v2b: FastSAM segment-everything -> cross-view common object -> fixed-320 ROI crop -> ONE batched CRE-Stereo pass over the 3 camera pairs -> 3-view fusion -> B/C/D. Why each piece is there ROI crop the item covers a small part of a 1280-wide frame; cropping spends the network's resolution on the cargo instead of on belt. same L/R window disparity is invariant to an equal column shift, so the crop must use identical [x0,x1] in both eyes and be left-padded by the max disparity, or the right-hand counterpart falls outside the crop. batching the 3 crops go through CRE as one forward pass; that is the "v2b" part. gate pixel the inspection point is projected into every camera, and the blob covering it is kept - by construction all views then measure the SAME physical item. consistency a view whose 3D centroid disagrees with the median is dropped; that is the cross-view rule enforced again in 3D, where it is unambiguous. The stereo rig must be RECTIFIED (parallel optical axes). Verged pairs break depth = fx*b/disp and the reconstruction lands metres away. """ from __future__ import annotations import json import time from pathlib import Path import cv2 import numpy as np import torch import torch.nn.functional as F from .. import config as C # ------------------------------------------------------------------ geometry helpers def _taubin(xy): """algebraic circle fit; returns (cx, cy, R, mean relative residual)""" x = xy[:, 0].astype(np.float64) y = xy[:, 1].astype(np.float64) xm, ym = x.mean(), y.mean() u, v = x - xm, y - ym Suu, Svv, Suv = (u * u).sum(), (v * v).sum(), (u * v).sum() Suuu, Svvv = (u ** 3).sum(), (v ** 3).sum() Suvv, Svuu = (u * v * v).sum(), (v * u * u).sum() try: uc, vc = np.linalg.solve(np.array([[Suu, Suv], [Suv, Svv]]), 0.5 * np.array([Suuu + Suvv, Svvv + Svuu])) except np.linalg.LinAlgError: return None cx, cy = uc + xm, vc + ym R = np.sqrt(max(uc * uc + vc * vc + (Suu + Svv) / len(x), 1e-12)) if not np.isfinite(R) or R < 1e-6: return None return cx, cy, R, float(np.abs(np.hypot(x - cx, y - cy) - R).mean() / R) def _rin_rout(xy): """K = inscribed / circumscribed radius of the convex hull; 1.0 for a perfect circle""" pts = np.ascontiguousarray(xy.astype(np.float32)) if len(pts) < 3: return 0.0 _, r_out = cv2.minEnclosingCircle(pts) if r_out < 1e-6: return 0.0 xmin, ymin = xy.min(0) w, h = float(np.ptp(xy[:, 0])), float(np.ptp(xy[:, 1])) sc = 180.0 / max(w, h, 1e-6) img = np.zeros((int(h * sc) + 10, int(w * sc) + 10), np.uint8) try: hull = cv2.convexHull(pts).reshape(-1, 2) except cv2.error: return 0.0 cv2.fillConvexPoly(img, ((hull - [xmin, ymin]) * sc + 5).astype(np.int32), 255) return float(cv2.distanceTransform(img, cv2.DIST_L2, 5).max()) / sc / r_out def _section_k(xy, res_tol=0.06, min_span=120.0): if len(xy) < 15: return 0.0 fit = _taubin(xy) if fit is not None: cx, cy, R, res = fit ang = np.arctan2(xy[:, 1] - cy, xy[:, 0] - cx) span = np.unique((((ang + np.pi) / (2 * np.pi)) * 48).astype(int) % 48).size / 48.0 * 360.0 ext = max(np.ptp(xy[:, 0]), np.ptp(xy[:, 1])) + 1e-9 if res < res_tol and span >= min_span and 0.35 * ext < R < 2.0 * ext: return 1.0 return _rin_rout(xy) def voxel(P, v=0.004): key = np.floor(P / v).astype(np.int64) _, idx = np.unique(key, axis=0, return_index=True) return P[idx] def roundness(P): """max circular-section K over the top-down section and one belt-aligned cross section""" if len(P) < 60: return 0.0 P = voxel(P) ks = [_section_k(P[:, :2])] xy = P[:, :2] - P[:, :2].mean(0) try: _, _, V = np.linalg.svd(xy, full_matrices=False) ax = V[0] except np.linalg.LinAlgError: ax = np.array([1.0, 0.0]) along = xy @ ax perp = xy @ np.array([-ax[1], ax[0]]) mid = np.abs(along - np.median(along)) < 0.15 * (np.ptp(along) + 1e-9) if mid.sum() > 15: ks.append(_section_k(np.c_[perp[mid], P[mid, 2]])) return max(ks) def classify(P, belt_z: float): """belt-plane OBB -> dims in real millimetres -> B / C / D""" if len(P) < 60: return "?", [0, 0, 0], 0.0 P = voxel(P) centre = P.mean(0) radial = np.linalg.norm(P - centre, axis=1) P = P[radial < np.percentile(radial, 94)] # trim segmentation fringe if len(P) < 40: return "?", [0, 0, 0], 0.0 height = (np.percentile(P[:, 2], 98) - belt_z) / C.DIM_SCALE * 1000.0 (_, _), (rw, rh), _ = cv2.minAreaRect(np.ascontiguousarray(P[:, :2].astype(np.float32))) foot = sorted([rw / C.DIM_SCALE * 1000.0, rh / C.DIM_SCALE * 1000.0]) dims = sorted([height, foot[0], foot[1]], reverse=True) k = roundness(P) a, b, c = C.OVERSIZE_MAX if dims[0] > a or dims[1] > b or dims[2] > c or dims[2] < C.MIN_DIM: return "C", [round(x) for x in dims], k return ("D" if k > C.ROUND_K else "B"), [round(x) for x in dims], k # ------------------------------------------------------------------ the pipeline class CreRoiV2b: """Holds the models, the render products and the per-frame decision.""" def __init__(self, calib_path: Path | str = None, device="cuda"): self.device = device calib_path = Path(calib_path or C.CONFIG / "calib.json") self.calib = json.loads(calib_path.read_text()) self.belt_z = self.calib.get("belt_top", C.BELT_Z) self.aim = np.array(self.calib["center"], dtype=float) if not self.calib.get("rectified"): raise ValueError( f"{calib_path} is not marked rectified. depth = fx*b/disp assumes parallel " "optical axes; rebuild the rig with sim.scene.build_camera_rig()." ) self._load_models() self.cams = {} # -- models ------------------------------------------------------------- def _load_models(self): import sys cre_dir = C.MODELS / "crestereo" if str(cre_dir) not in sys.path: sys.path.insert(0, str(cre_dir)) from nets import Model # noqa: E402 (vendored CRE-Stereo) self.cre = Model(max_disp=256, mixed_precision=False, test_mode=True) weights = C.MODELS / "crestereo_eth3d.pth" self.cre.load_state_dict(torch.load(weights, map_location="cpu"), strict=True) self.cre.to(self.device).eval() from ultralytics import FastSAM # noqa: E402 self.fsam = FastSAM(str(C.MODELS / "FastSAM-s.pt")) # -- render products ---------------------------------------------------- def attach_cameras(self): """one RGB annotator per eye; call once, after the stage is populated""" import omni.replicator.core as rep for name, cc in self.calib["cameras"].items(): K = cc["intrinsics"] res = (K["width"], K["height"]) ann = {} for side, path in (("L", cc["left_path"]), ("R", cc["right_path"])): rp = rep.create.render_product(path, res) a = rep.AnnotatorRegistry.get_annotator("rgb") a.attach(rp) ann[side] = a self.cams[name] = dict(K=K, b=cc["baseline_m"], LW=np.array(cc["left_world"], dtype=float), **ann) self.gate_px = {n: self._gate_px(n) for n in self.cams} return self.cams async def warmup(self, steps=120, tries=4): """Step the app until the eyes actually return an image. `attach_cameras` creates the render products and returns at once, but RTX yields nothing for a fresh render product for a long while, so the first `get_data()` comes back PURE BLACK - RGB max 0 with alpha ~255, meaning geometry is being hit and is simply unresolved, not that the view is empty. FastSAM then segments nothing and every early item is logged unclassified, which reads as "the pipeline is broken". Measured on this cell: 60 steps still black; 120 steps plus a second settle gives RGB max 92 / mean 21 on the D435 left eye. So this waits and CHECKS rather than trusting a fixed count - it returns the darkest eye it saw, for the run log. """ import isaacsim.core.experimental.utils.app as app_utils worst = 0 for attempt in range(tries): await app_utils.update_app_async(steps=steps) worst = 255 for c in self.cams.values(): for side in ("L", "R"): a = np.asarray(c[side].get_data()) worst = min(worst, int(a[..., :3].max()) if a.size else 0) if worst > 3: return dict(ok=True, attempts=attempt + 1, darkest_eye_max=worst) return dict(ok=False, attempts=tries, darkest_eye_max=worst) def _gate_px(self, name): c = self.cams[name] K = c["K"] p = np.append(self.aim, 1.0) @ np.linalg.inv(c["LW"]) d = max(-p[2], 1e-6) u = K["cx"] + K["fx"] * p[0] / d v = K["cy"] - K["fy"] * p[1] / d return int(np.clip(u, 0, K["width"] - 1)), int(np.clip(v, 0, K["height"] - 1)) # -- stages ------------------------------------------------------------- def segment(self, rgb, gate): """smallest plausible blob covering the gate pixel""" t0 = time.time() res = self.fsam(rgb[..., ::-1], device=self.device, retina_masks=True, imgsz=1024, conf=0.35, iou=0.9, verbose=False) ms = (time.time() - t0) * 1000 if not res or res[0].masks is None: return None, ms gx, gy = gate best, best_area = None, np.inf for m in res[0].masks.data.cpu().numpy(): mb = m > 0.5 area = int(mb.sum()) if area < 150 or area > C.MAX_MASK_FRAC * mb.size: continue if mb[gy, gx] and area < best_area: best, best_area = mb, area return best, ms @staticmethod def crop(left, right, mask, K, baseline): """identical column window in both eyes, left-padded by the max disparity""" H, W = mask.shape ys, xs = np.where(mask) max_disp = int(np.ceil(C.DISP_PAD * K["fx"] * baseline / C.Z_MIN)) x0 = max(0, xs.min() - max_disp - C.ROI_PAD_X) x1 = min(W, xs.max() + C.ROI_PAD_X) y0 = max(0, ys.min() - C.ROI_PAD_V) y1 = min(H, ys.max() + C.ROI_PAD_V) cl, cr = left[y0:y1, x0:x1], right[y0:y1, x0:x1] h, w = cl.shape[:2] s = min(C.ROI_FIXED / max(h, w, 1), 2.5) if abs(s - 1) > 0.02: cl = cv2.resize(cl, (max(int(round(w * s)), 8), max(int(round(h * s)), 8))) cr = cv2.resize(cr, (cl.shape[1], cl.shape[0])) return np.ascontiguousarray(cl), np.ascontiguousarray(cr), (x0, y0, x1, y1), s, (h, w) def infer_batch(self, crops, iters=20): """all camera crops in ONE two-stage CRE pass, zero-padded to a common /8 canvas""" sizes = [c[0].shape[:2] for c in crops] Hp = (max(h for h, _ in sizes) + 7) // 8 * 8 Wp = (max(w for _, w in sizes) + 7) // 8 * 8 lb = np.zeros((len(crops), 3, Hp, Wp), np.float32) rb = np.zeros_like(lb) for i, (l, r) in enumerate(crops): h, w = l.shape[:2] lb[i, :, :h, :w] = l.transpose(2, 0, 1) rb[i, :, :h, :w] = r.transpose(2, 0, 1) il = torch.from_numpy(lb).to(self.device) ir = torch.from_numpy(rb).to(self.device) ild = F.interpolate(il, (Hp // 2, Wp // 2), mode="bilinear", align_corners=True) ird = F.interpolate(ir, (Hp // 2, Wp // 2), mode="bilinear", align_corners=True) with torch.inference_mode(): init = self.cre(ild, ird, iters=iters, flow_init=None) flow = self.cre(il, ir, iters=iters, flow_init=init) disp = np.abs(flow[:, 0].detach().cpu().numpy()) return [disp[i, :h, :w] for i, (h, w) in enumerate(sizes)] def backproject(self, disp_small, mask, box, s, orig, K, baseline, LW): h, w = orig disp = cv2.resize(disp_small, (w, h)) / s depth = np.where(disp > 0.5, K["fx"] * baseline / np.maximum(disp, 1e-6), np.nan) x0, y0, x1, y1 = box # erode: silhouette pixels straddle the depth discontinuity and smear the cloud m = cv2.erode(mask[y0:y1, x0:x1].astype(np.uint8), np.ones((3, 3), np.uint8), iterations=2).astype(bool) vs, us = np.mgrid[y0:y1, x0:x1] sel = m & np.isfinite(depth) & (depth > 1e-3) u, v, d = us[sel], vs[sel], depth[sel] cam = np.stack([(u - K["cx"]) * d / K["fx"], -(v - K["cy"]) * d / K["fy"], -d, np.ones_like(d)], 1) return (cam @ LW)[:, :3] def _in_workspace(self, P): return P[(np.abs(P[:, 0] - self.aim[0]) < 0.45) & (np.abs(P[:, 1]) < 0.35) & (P[:, 2] > self.belt_z + 0.012) & (P[:, 2] < self.belt_z + 0.80)] # -- one decision ------------------------------------------------------- def measure(self): """capture -> segment -> crop -> batched CRE -> fuse -> classify""" t0 = time.time() crops, meta, seg_ms = [], [], [] for name, c in self.cams.items(): left = c["L"].get_data()[..., :3] right = c["R"].get_data()[..., :3] mask, ms = self.segment(left, self.gate_px[name]) seg_ms.append(round(ms)) if mask is None: continue cl, cr, box, s, orig = self.crop(left.astype(np.float32) / 255.0, right.astype(np.float32) / 255.0, mask, c["K"], c["b"]) crops.append((cl, cr)) meta.append((name, mask, box, s, orig)) if not crops: return dict(cls="?", dims=[0, 0, 0], k=0.0, views=0, pts=0, seg_ms=seg_ms, cre_ms=0, total_ms=round((time.time() - t0) * 1000)) t_cre = time.time() disps = self.infer_batch(crops) cre_ms = round((time.time() - t_cre) * 1000) clouds = [] for (name, mask, box, s, orig), disp in zip(meta, disps): c = self.cams[name] P = self._in_workspace( self.backproject(disp, mask, box, s, orig, c["K"], c["b"], c["LW"])) if len(P) >= 40: clouds.append((name, P)) dropped = [] if len(clouds) > 1: cent = np.array([p.mean(0) for _, p in clouds]) med = np.median(cent, 0) keep = [i for i in range(len(clouds)) if np.linalg.norm(cent[i] - med) < C.VIEW_CONSISTENCY] dropped = [clouds[i][0] for i in range(len(clouds)) if i not in keep] if keep: clouds = [clouds[i] for i in keep] if not clouds: return dict(cls="?", dims=[0, 0, 0], k=0.0, views=0, pts=0, seg_ms=seg_ms, cre_ms=cre_ms, total_ms=round((time.time() - t0) * 1000)) P = np.vstack([p for _, p in clouds]) cls, dims, k = classify(P, self.belt_z) return dict(cls=cls, dims=dims, k=round(float(k), 3), views=len(clouds), dropped=dropped, pts=len(P), seg_ms=seg_ms, cre_ms=cre_ms, total_ms=round((time.time() - t0) * 1000))