diff --git a/anomavision/algorithm/common/feature_extraction.py b/anomavision/algorithm/common/feature_extraction.py index 132ca64d..e48ab5c5 100644 --- a/anomavision/algorithm/common/feature_extraction.py +++ b/anomavision/algorithm/common/feature_extraction.py @@ -163,7 +163,16 @@ def forward( layers.append(out4) if layer_indices is not None: - layers = [layers[i] for i in layer_indices] + # Indices refer to the original ResNet stages: + # 0=layer1, 1=layer2, 2=layer3, 3=layer4. + stage_outputs = [out1] + if max_l >= 1: + stage_outputs.append(out2) + if max_l >= 2: + stage_outputs.append(out3) + if max_l >= 3: + stage_outputs.append(out4) + layers = [stage_outputs[i] for i in layer_indices] if layer_hook is not None: layers = [layer_hook(layer) for layer in layers] diff --git a/anomavision/algorithm/patchcore/patchcore.py b/anomavision/algorithm/patchcore/patchcore.py index c2ec64c2..52fa16b1 100644 --- a/anomavision/algorithm/patchcore/patchcore.py +++ b/anomavision/algorithm/patchcore/patchcore.py @@ -43,8 +43,8 @@ class PatchCore(torch.nn.Module): backbone: Feature-extraction backbone. Supported values are ``resnet18`` and ``wide_resnet50``. device: Device used for feature extraction and nearest-neighbor distance. - layer_indices: ResNet feature stages to concatenate. Defaults to ``[0, 1]`` - to keep the lightweight model fast and compact. + layer_indices: ResNet feature stages to concatenate. Defaults to ``[1, 2]`` + to provide multi-scale mid/deep features while remaining lightweight. memory_bank: Optional precomputed bank with shape ``(num_patches, dim)``. Providing it creates a ready-to-infer model. coreset_ratio: Fraction of extracted normal patches to retain. Must be in @@ -90,7 +90,7 @@ def __init__( ) self.device = torch.device(device) self.backbone = backbone - self.layer_indices = list(layer_indices or [0, 1]) + self.layer_indices = list(layer_indices or [1, 2]) self.coreset_ratio = float(coreset_ratio) self.max_memory_patches = max_memory_patches self.patch_grid = patch_grid diff --git a/anomavision/detect.py b/anomavision/detect.py index cec66d2a..ad0e2862 100644 --- a/anomavision/detect.py +++ b/anomavision/detect.py @@ -515,23 +515,49 @@ def _save_live_drift_status() -> None: score_maps, kernel_size=33, sigma=4 ) if config.thresh is not None: - # Localization is the source of truth for anomaly - # classification: an image is anomalous only when at - # least one pixel in its anomaly map reaches the - # configured threshold. This prevents the image-level - # score from reporting ANOMALY without localization. - localization_masks = anomavision.classification( - score_maps, config.thresh - ) - is_anomaly = ( - np.any( - np.asarray(localization_masks).reshape( - len(localization_masks), -1 - ) - > 0, - axis=1, + if str(config.get("algorithm", "")).lower() == "patchcore": + # PatchCore: config.thresh is an IMAGE-level threshold + # (cosine-distance scale). The blurred pixel map lives + # on a different scale, so thresholding it directly + # gives either an empty or a full-image mask. Classify + # by image score, then localize with a per-image + # relative cutoff on the score map. + scores_np = np.asarray( + image_scores.detach().float().cpu().numpy() + if hasattr(image_scores, "detach") + else image_scores + ).reshape(-1) + is_anomaly = (scores_np >= float(config.thresh)).astype( + np.int64 ) - ).astype(np.int64) + # Cheap relative cutoff on the min-max normalised map + maps_np = ( + score_maps.detach().float().cpu().numpy() + if hasattr(score_maps, "detach") + else np.asarray(score_maps) + ) + lo = maps_np.min(axis=(1, 2), keepdims=True) + hi = maps_np.max(axis=(1, 2), keepdims=True) + norm = (maps_np - lo) / (hi - lo + 1e-8) + localization_masks = ( + (norm >= float(config.get("patchcore_loc_rel", 0.60))) + & (is_anomaly[:, None, None] > 0) + ).astype(np.uint8) + else: + # PaDiM etc.: an image is anomalous only when at + # least one pixel reaches the configured threshold. + localization_masks = anomavision.classification( + score_maps, config.thresh + ) + is_anomaly = ( + np.any( + np.asarray(localization_masks).reshape( + len(localization_masks), -1 + ) + > 0, + axis=1, + ) + ).astype(np.int64) else: localization_masks = np.zeros_like(score_maps) is_anomaly = np.zeros(score_maps.shape[0], dtype=np.int64) diff --git a/anomavision/train.py b/anomavision/train.py index dc5c1838..df6c8461 100644 --- a/anomavision/train.py +++ b/anomavision/train.py @@ -93,6 +93,13 @@ def create_parser(add_help: bool = True) -> argparse.ArgumentParser: default=None, help="Backbone feature layers.", ) + parser.add_argument( + "--patchcore_layer_indices", + type=int, + nargs="+", + default=None, + help="ResNet stages used by PatchCore; defaults to [1, 2].", + ) parser.add_argument( "--coreset_ratio", type=float, @@ -226,14 +233,18 @@ def run_training(args): "cfg: algorithm=%s | backbone=%s | layers=%s", config.algorithm, config.backbone, - config.layer_indices, + ( + config.get("patchcore_layer_indices", [1, 2]) + if str(config.algorithm).lower() == "patchcore" + else config.layer_indices + ), ) if str(config.algorithm).lower() == "patchcore": model = anomavision.PatchCore( backbone=config.backbone, device=device, - layer_indices=config.layer_indices, + layer_indices=config.get("patchcore_layer_indices", [1, 2]), coreset_ratio=float(config.coreset_ratio), max_memory_patches=config.max_memory_patches, patch_grid=config.patch_grid, diff --git a/anomavision/visualization/boundary.py b/anomavision/visualization/boundary.py index 6357afd9..21e97be5 100644 --- a/anomavision/visualization/boundary.py +++ b/anomavision/visualization/boundary.py @@ -1,5 +1,6 @@ from typing import Tuple, Union +import cv2 import numpy as np import torch from skimage.segmentation import find_boundaries @@ -92,12 +93,42 @@ def boundary_image( """ image = to_numpy(image).copy() - mask = to_numpy(patch_classification).copy() - - found_boundaries = find_boundaries(mask).astype(np.uint8) - layer_two = np.zeros(image.shape, dtype=np.uint8) - layer_two[:] = boundary_color + mask = np.squeeze(to_numpy(patch_classification).copy()) + + if mask.ndim != 2: + raise ValueError( + f"patch_classification must be a 2D mask after squeezing; got shape {mask.shape}" + ) + + # Resize the localization mask itself, not its one-pixel boundary. + # This preserves the defect region and lets OpenCV trace a visible + # contour at the final image resolution. + binary_mask = (mask > 0.5).astype(np.uint8) + + if binary_mask.shape != image.shape[:2]: + binary_mask = cv2.resize( + binary_mask, + (image.shape[1], image.shape[0]), + interpolation=cv2.INTER_NEAREST, + ) + + # Fill tiny gaps introduced by patch/grid localization while keeping + # separate defects as separate regions. + kernel = np.ones((3, 3), dtype=np.uint8) + binary_mask = cv2.morphologyEx(binary_mask, cv2.MORPH_CLOSE, kernel) + + contours, _ = cv2.findContours( + binary_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE + ) - b_image = composite_image(image, layer_two, found_boundaries) + b_image = image.copy() + if contours: + cv2.drawContours( + b_image, + contours, + contourIdx=-1, + color=tuple(int(v) for v in boundary_color), + thickness=max(2, min(image.shape[:2]) // 150), + ) return b_image diff --git a/apps/api/fastapi_app.py b/apps/api/fastapi_app.py index ae096eca..42f645b3 100644 --- a/apps/api/fastapi_app.py +++ b/apps/api/fastapi_app.py @@ -30,9 +30,26 @@ model_type: Optional[ModelType] = None drift_runtime: Optional[InferenceDriftRuntime] = None -ANOMALY_THRESHOLD = 13.0 +# Pixel thresholds are algorithm-specific. A single PaDiM threshold (13.0) +# makes PatchCore masks empty even when the image is correctly classified as +# anomalous (PatchCore scores are typically much smaller). +_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD") +ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0 RESIZE_SIZE = (224, 224) + +def _threshold_for_model(model_path: str) -> float: + """Return the pixel-localization threshold for the active model.""" + if _THRESHOLD_OVERRIDE: + return float(_THRESHOLD_OVERRIDE) + normalized = os.path.normpath(model_path).lower() + if "patchcore" in normalized: + return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25")) + if "efficientad" in normalized: + return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) + return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) + + # You can override these via environment variables MODEL_DATA_PATH = os.getenv( "ANOMAVISION_MODEL_DATA_PATH", "distributions/padim/bottle/anomav_exp" @@ -56,7 +73,7 @@ async def load_model(): model = ModelWrapper(model_path, device_str) model_type = ModelType.from_extension(model_path) """ - global model, model_type, drift_runtime + global model, model_type, drift_runtime, ANOMALY_THRESHOLD device_str = determine_device(DEVICE) # "cpu" or "cuda" model_path = os.path.realpath(os.path.join(MODEL_DATA_PATH, MODEL_FILE)) @@ -67,6 +84,8 @@ async def load_model(): # ModelType is inferred from extension (.pt/.onnx/.engine/...) model_type = ModelType.from_extension(model_path) model = ModelWrapper(model_path, device_str) + ANOMALY_THRESHOLD = _threshold_for_model(model_path) + print(f"[api] Localization threshold: {ANOMALY_THRESHOLD}") if DRIFT_REFERENCE: reference = load_embeddings(DRIFT_REFERENCE) @@ -201,11 +220,23 @@ def create_visualizations( ): """ Mirror detect.py's visualization path. + + The localization mask is the source of truth for both the defect contour + and image anomaly status. The image-level score is intentionally not used + to decide whether a localization frame is drawn. """ score_map_classifications = anomavision.classification( score_maps, ANOMALY_THRESHOLD ) - image_classifications = anomavision.classification(image_scores, ANOMALY_THRESHOLD) + image_classifications = ( + np.any( + np.asarray(score_map_classifications).reshape( + score_map_classifications.shape[0], -1 + ) + > 0, + axis=1, + ) + ).astype(np.int64) test_images = np.array([image_np]) diff --git a/apps/inference_engine.py b/apps/inference_engine.py index c419cead..0918d8a4 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -17,6 +17,7 @@ from dataclasses import dataclass from typing import Optional +import cv2 import numpy as np import onnxruntime as ort from onnxruntime import GraphOptimizationLevel, SessionOptions @@ -27,7 +28,24 @@ # ----------------------------------------------------------------------------- # Config — all overridable via environment variables # ----------------------------------------------------------------------------- -ANOMALY_THRESHOLD = float(os.getenv("ANOMAVISION_THRESHOLD", "13.0")) +# Keep an explicit environment override authoritative. Otherwise the +# threshold follows the algorithm encoded by the active model path. +_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD") +ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0 + + +def _threshold_for_model(model_path: str) -> float: + """Return the algorithm-appropriate pixel threshold for a model artifact.""" + if _THRESHOLD_OVERRIDE: + return float(_THRESHOLD_OVERRIDE) + normalized = os.path.normpath(model_path).lower() + if "patchcore" in normalized: + return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25")) + if "efficientad" in normalized: + return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) + return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) + + MODEL_DATA_PATH = os.getenv("ANOMAVISION_MODEL_DATA_PATH", "") MODEL_FILE = os.getenv("ANOMAVISION_MODEL_FILE", "model.onnx") STUDIO_ROOT = os.path.expanduser( @@ -214,6 +232,13 @@ def load_model(project_id: Optional[str] = None) -> str: _sess = ort.InferenceSession(model_path, providers=providers, sess_options=opts) _input_name = _sess.get_inputs()[0].name + # The previous API used a fixed PaDiM threshold (13.0) for every model. + # PatchCore maps use a much smaller score scale, so that made the heatmap + # show the defect while the binary localization mask was completely empty. + global ANOMALY_THRESHOLD + ANOMALY_THRESHOLD = _threshold_for_model(model_path) + print(f"[inference] Localization threshold: {ANOMALY_THRESHOLD}") + # Warmup — run twice so JIT compile happens now, not on the first real request dummy_shape = tuple( d if isinstance(d, int) and d > 0 else 1 for d in _sess.get_inputs()[0].shape @@ -283,15 +308,50 @@ def run( # Visualizations are optional. Live/camera inference does not need them; # skipping this CPU-heavy path keeps latency close to the raw ONNX runtime. if include_visualizations: - score_map_cls = classification(score_maps, threshold) - # Use the localized pixel mask as the source of truth. The image-level - # score alone must not produce ANOMALY when no pixel is localized. - image_cls = ( - np.any( - np.asarray(score_map_cls).reshape(score_map_cls.shape[0], -1) > 0, - axis=1, + # The image score decides whether the frame is anomalous. + # Localization is derived independently from the spatial PatchCore map. + # + # Do not use a fixed normalized threshold such as 0.60 here: for a + # localized defect that can select a large portion of the object/image. + # Instead, keep only the strongest spatial response (top 5% of pixels) + # and then retain the connected high-score regions. This makes the + # contour follow the defect rather than the whole image. + raw_map = np.asarray(score_maps, dtype=np.float32) + flat_map = raw_map.reshape(raw_map.shape[0], -1) + map_min = flat_map.min(axis=1)[:, None, None] + map_max = flat_map.max(axis=1)[:, None, None] + normalized_maps = (raw_map - map_min) / np.maximum(map_max - map_min, 1e-8) + + score_map_cls = np.zeros_like(normalized_maps, dtype=np.uint8) + for i, normalized_map in enumerate(normalized_maps): + localization_threshold = float(np.quantile(normalized_map, 0.95)) + mask = (normalized_map >= localization_threshold).astype(np.uint8) + + # Close small gaps without expanding the defect excessively. + kernel = np.ones((3, 3), dtype=np.uint8) + mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel) + + # Keep only connected regions containing one of the strongest + # PatchCore responses. This removes scattered background pixels. + num_labels, labels, stats, _ = cv2.connectedComponentsWithStats( + mask, connectivity=8 ) - ).astype(np.int64) + if num_labels > 1: + peaks = [] + for label in range(1, num_labels): + component = normalized_map[labels == label] + peaks.append((float(component.max()), label)) + keep = {label for _, label in sorted(peaks, reverse=True)[:3]} + mask = np.isin(labels, list(keep)).astype(np.uint8) + + score_map_cls[i] = mask + + # The image-level score is the anomaly decision. The score map is + # localization data and must not determine the outer anomaly frame. + image_cls = np.array( + [int(float(image_score) >= float(threshold))], dtype=np.int64 + ) + score_map_cls[image_cls == 0] = 0 test_images = np.array([image_np]) boundary_np = visualization.framed_boundary_images( test_images, score_map_cls, image_cls, padding=VIZ_PADDING @@ -305,10 +365,7 @@ def run( latency_ms = (time.perf_counter() - t0) * 1000 - pixel_mask = classification(score_maps, threshold) - is_anomaly = bool( - np.any(np.asarray(pixel_mask).reshape(pixel_mask.shape[0], -1) > 0) - ) + is_anomaly = bool(float(image_score) >= float(threshold)) return InferenceResult( anomaly_score=image_score, diff --git a/config.yml b/config.yml index f6ff81a4..fcbe8ac0 100644 --- a/config.yml +++ b/config.yml @@ -1,8 +1,8 @@ # ========================= # Dataset / preprocessing (shared by train, detect, eval, stream) # ========================= -dataset_path: "D:/01-DATA" # Root dataset directory (contains train/test folders) -class_name: "bottle" # Dataset class to train/evaluate +dataset_path: "D:/01-DATA/mvtec" # Root dataset directory (contains train/test folders) +class_name: "cable" # Dataset class to train/evaluate resize: [224, 224] # Input image size [width, height] crop_size: # Optional center crop size [width, height] normalize: true # Apply ImageNet normalization @@ -22,7 +22,8 @@ search_chunk_size: 1024 # PatchCore nearest-neighbor search c coreset_method: "kcenter" # PatchCore coreset method: kcenter | random coreset_seed: 42 # Random seed for PatchCore coreset selection feat_dim: 50 # PaDiM number of feature dimensions to keep -layer_indices: [0] # Backbone feature layers used by PaDiM/PatchCore/EfficientAD +layer_indices: [0] # Backbone feature layers used by PaDiM/EfficientAD +patchcore_layer_indices: [1, 2] # PatchCore uses ResNet layer2 + layer3 for multi-scale anomaly features model_data_path: "./distributions" # Directory for trained models and statistics model: "model.pt" # Model filename used by detect/eval/export output_model: "model.pt" # Filename written by the train command @@ -66,8 +67,8 @@ detailed_timing: false # Enable detailed timing information # ========================= # Visualization (detect/eval) # ========================= -enable_visualization: false # Enable anomaly visualization -save_visualizations: false # Save generated visualizations to disk +enable_visualization: true # Enable anomaly visualization +save_visualizations: true # Save generated visualizations to disk viz_output_dir: "./visualizations/" # Directory for visualization output viz_alpha: 0.5 # Heatmap overlay transparency viz_padding: 40 # Extra visualization border/padding diff --git a/tests/test_patchcore_ultralight.py b/tests/test_patchcore_ultralight.py index e06fe17b..3e1f92be 100644 --- a/tests/test_patchcore_ultralight.py +++ b/tests/test_patchcore_ultralight.py @@ -59,3 +59,36 @@ def test_patchcore_stats_round_trip_preserves_ultralight_settings( assert restored.patch_grid == 3 assert restored.search_chunk_size == 7 assert restored.max_memory_patches == 11 + + +def test_patchcore_fit_and_inference_are_deterministic(monkeypatch): + """Guard against regressions that make PatchCore appear random between runs.""" + monkeypatch.setattr(patchcore_module, "ResnetEmbeddingsExtractor", FakeExtractor) + + images = torch.arange(6 * 3 * 8 * 8, dtype=torch.float32).reshape(6, 3, 8, 8) + loader = DataLoader(TensorDataset(images), batch_size=2, shuffle=False) + query = images[:2] + + def build_and_run(): + model = patchcore_module.PatchCore( + device="cpu", + layer_indices=[0], + coreset_ratio=0.5, + max_memory_patches=5, + patch_grid=2, + search_chunk_size=2, + coreset_method="kcenter", + coreset_seed=42, + ) + model.fit(loader) + scores, maps = model.predict(query) + return model.memory_bank.clone(), scores.clone(), maps.clone() + + bank_a, scores_a, maps_a = build_and_run() + bank_b, scores_b, maps_b = build_and_run() + + assert torch.equal(bank_a, bank_b) + assert torch.equal(scores_a, scores_b) + assert torch.equal(maps_a, maps_b) + assert torch.isfinite(scores_a).all() + assert torch.isfinite(maps_a).all() diff --git a/tests/test_visualization_boundary.py b/tests/test_visualization_boundary.py new file mode 100644 index 00000000..ca6b5885 --- /dev/null +++ b/tests/test_visualization_boundary.py @@ -0,0 +1,15 @@ +import numpy as np + +from anomavision.visualization.boundary import boundary_image + + +def test_boundary_image_draws_thin_localization_boundary(): + image = np.zeros((32, 32, 3), dtype=np.uint8) + mask = np.zeros((8, 8), dtype=np.uint8) + mask[2:6, 2:6] = 1 + + result = boundary_image(image, mask, boundary_color=(255, 0, 0)) + + # The resized localization must produce visible boundary pixels. + red_pixels = np.all(result == np.array([255, 0, 0], dtype=np.uint8), axis=-1) + assert red_pixels.any()