From c6c8eeb5b37d24afaf7123eea1175526aaacece3 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:26:18 +0200 Subject: [PATCH 01/17] test(patchcore): guard deterministic inference regression --- tests/test_patchcore_ultralight.py | 33 ++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/tests/test_patchcore_ultralight.py b/tests/test_patchcore_ultralight.py index e06fe17b..3e1f92be 100644 --- a/tests/test_patchcore_ultralight.py +++ b/tests/test_patchcore_ultralight.py @@ -59,3 +59,36 @@ def test_patchcore_stats_round_trip_preserves_ultralight_settings( assert restored.patch_grid == 3 assert restored.search_chunk_size == 7 assert restored.max_memory_patches == 11 + + +def test_patchcore_fit_and_inference_are_deterministic(monkeypatch): + """Guard against regressions that make PatchCore appear random between runs.""" + monkeypatch.setattr(patchcore_module, "ResnetEmbeddingsExtractor", FakeExtractor) + + images = torch.arange(6 * 3 * 8 * 8, dtype=torch.float32).reshape(6, 3, 8, 8) + loader = DataLoader(TensorDataset(images), batch_size=2, shuffle=False) + query = images[:2] + + def build_and_run(): + model = patchcore_module.PatchCore( + device="cpu", + layer_indices=[0], + coreset_ratio=0.5, + max_memory_patches=5, + patch_grid=2, + search_chunk_size=2, + coreset_method="kcenter", + coreset_seed=42, + ) + model.fit(loader) + scores, maps = model.predict(query) + return model.memory_bank.clone(), scores.clone(), maps.clone() + + bank_a, scores_a, maps_a = build_and_run() + bank_b, scores_b, maps_b = build_and_run() + + assert torch.equal(bank_a, bank_b) + assert torch.equal(scores_a, scores_b) + assert torch.equal(maps_a, maps_b) + assert torch.isfinite(scores_a).all() + assert torch.isfinite(maps_a).all() From 900199a817776e7fbb7fea86f33b26c5a3c8316a Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:10:18 +0200 Subject: [PATCH 02/17] fix(features): select ResNet stages by original indices --- anomavision/algorithm/common/feature_extraction.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/anomavision/algorithm/common/feature_extraction.py b/anomavision/algorithm/common/feature_extraction.py index 132ca64d..e48ab5c5 100644 --- a/anomavision/algorithm/common/feature_extraction.py +++ b/anomavision/algorithm/common/feature_extraction.py @@ -163,7 +163,16 @@ def forward( layers.append(out4) if layer_indices is not None: - layers = [layers[i] for i in layer_indices] + # Indices refer to the original ResNet stages: + # 0=layer1, 1=layer2, 2=layer3, 3=layer4. + stage_outputs = [out1] + if max_l >= 1: + stage_outputs.append(out2) + if max_l >= 2: + stage_outputs.append(out3) + if max_l >= 3: + stage_outputs.append(out4) + layers = [stage_outputs[i] for i in layer_indices] if layer_hook is not None: layers = [layer_hook(layer) for layer in layers] From edc3b4c4d6311cad71e05a1910915bfcfd52fac0 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:10:20 +0200 Subject: [PATCH 03/17] fix(patchcore): use dedicated multi-scale feature stages --- anomavision/train.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/anomavision/train.py b/anomavision/train.py index dc5c1838..df6c8461 100644 --- a/anomavision/train.py +++ b/anomavision/train.py @@ -93,6 +93,13 @@ def create_parser(add_help: bool = True) -> argparse.ArgumentParser: default=None, help="Backbone feature layers.", ) + parser.add_argument( + "--patchcore_layer_indices", + type=int, + nargs="+", + default=None, + help="ResNet stages used by PatchCore; defaults to [1, 2].", + ) parser.add_argument( "--coreset_ratio", type=float, @@ -226,14 +233,18 @@ def run_training(args): "cfg: algorithm=%s | backbone=%s | layers=%s", config.algorithm, config.backbone, - config.layer_indices, + ( + config.get("patchcore_layer_indices", [1, 2]) + if str(config.algorithm).lower() == "patchcore" + else config.layer_indices + ), ) if str(config.algorithm).lower() == "patchcore": model = anomavision.PatchCore( backbone=config.backbone, device=device, - layer_indices=config.layer_indices, + layer_indices=config.get("patchcore_layer_indices", [1, 2]), coreset_ratio=float(config.coreset_ratio), max_memory_patches=config.max_memory_patches, patch_grid=config.patch_grid, From be9b2d4d4de4dd107d51d088b0d5cdf7fb7f19e2 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:10:23 +0200 Subject: [PATCH 04/17] fix(config): use multi-scale PatchCore feature layers --- config.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/config.yml b/config.yml index f6ff81a4..85491ead 100644 --- a/config.yml +++ b/config.yml @@ -22,7 +22,8 @@ search_chunk_size: 1024 # PatchCore nearest-neighbor search c coreset_method: "kcenter" # PatchCore coreset method: kcenter | random coreset_seed: 42 # Random seed for PatchCore coreset selection feat_dim: 50 # PaDiM number of feature dimensions to keep -layer_indices: [0] # Backbone feature layers used by PaDiM/PatchCore/EfficientAD +layer_indices: [0] # Backbone feature layers used by PaDiM/EfficientAD +patchcore_layer_indices: [1, 2] # PatchCore uses ResNet layer2 + layer3 for multi-scale anomaly features model_data_path: "./distributions" # Directory for trained models and statistics model: "model.pt" # Model filename used by detect/eval/export output_model: "model.pt" # Filename written by the train command From 1162480613d98717e0049b317ec7af44da81f5b3 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:10:28 +0200 Subject: [PATCH 05/17] fix(patchcore): default to multi-scale ResNet features --- anomavision/algorithm/patchcore/patchcore.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/anomavision/algorithm/patchcore/patchcore.py b/anomavision/algorithm/patchcore/patchcore.py index c2ec64c2..52fa16b1 100644 --- a/anomavision/algorithm/patchcore/patchcore.py +++ b/anomavision/algorithm/patchcore/patchcore.py @@ -43,8 +43,8 @@ class PatchCore(torch.nn.Module): backbone: Feature-extraction backbone. Supported values are ``resnet18`` and ``wide_resnet50``. device: Device used for feature extraction and nearest-neighbor distance. - layer_indices: ResNet feature stages to concatenate. Defaults to ``[0, 1]`` - to keep the lightweight model fast and compact. + layer_indices: ResNet feature stages to concatenate. Defaults to ``[1, 2]`` + to provide multi-scale mid/deep features while remaining lightweight. memory_bank: Optional precomputed bank with shape ``(num_patches, dim)``. Providing it creates a ready-to-infer model. coreset_ratio: Fraction of extracted normal patches to retain. Must be in @@ -90,7 +90,7 @@ def __init__( ) self.device = torch.device(device) self.backbone = backbone - self.layer_indices = list(layer_indices or [0, 1]) + self.layer_indices = list(layer_indices or [1, 2]) self.coreset_ratio = float(coreset_ratio) self.max_memory_patches = max_memory_patches self.patch_grid = patch_grid From 93d2dad8d31c48a1d3e768d7f85a633862a947d4 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:20:10 +0200 Subject: [PATCH 06/17] fix(visualization): preserve thin defect boundaries --- anomavision/visualization/boundary.py | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/anomavision/visualization/boundary.py b/anomavision/visualization/boundary.py index 6357afd9..c7072679 100644 --- a/anomavision/visualization/boundary.py +++ b/anomavision/visualization/boundary.py @@ -1,5 +1,6 @@ from typing import Tuple, Union +import cv2 import numpy as np import torch from skimage.segmentation import find_boundaries @@ -94,10 +95,25 @@ def boundary_image( image = to_numpy(image).copy() mask = to_numpy(patch_classification).copy() - found_boundaries = find_boundaries(mask).astype(np.uint8) - layer_two = np.zeros(image.shape, dtype=np.uint8) - layer_two[:] = boundary_color - - b_image = composite_image(image, layer_two, found_boundaries) + # Keep the localization boundary binary until it is applied to the image. + # Resizing a one-pixel boundary with area interpolation can make values + # fractional; exact binary-mask checks then erase the boundary. + mask = np.squeeze(mask) + if mask.ndim != 2: + raise ValueError( + f"patch_classification must be a 2D mask after squeezing; got shape {mask.shape}" + ) + + found_boundaries = find_boundaries(mask > 0.5, mode="outer") + + if found_boundaries.shape != image.shape[:2]: + found_boundaries = cv2.resize( + found_boundaries.astype(np.uint8), + (image.shape[1], image.shape[0]), + interpolation=cv2.INTER_NEAREST, + ).astype(bool) + + b_image = image.copy() + b_image[found_boundaries] = np.asarray(boundary_color, dtype=b_image.dtype) return b_image From e032082c25bb43598464284d471cdf0cd3d1972b Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:20:17 +0200 Subject: [PATCH 07/17] test(visualization): guard defect boundary rendering --- tests/test_visualization_boundary.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 tests/test_visualization_boundary.py diff --git a/tests/test_visualization_boundary.py b/tests/test_visualization_boundary.py new file mode 100644 index 00000000..ca6b5885 --- /dev/null +++ b/tests/test_visualization_boundary.py @@ -0,0 +1,15 @@ +import numpy as np + +from anomavision.visualization.boundary import boundary_image + + +def test_boundary_image_draws_thin_localization_boundary(): + image = np.zeros((32, 32, 3), dtype=np.uint8) + mask = np.zeros((8, 8), dtype=np.uint8) + mask[2:6, 2:6] = 1 + + result = boundary_image(image, mask, boundary_color=(255, 0, 0)) + + # The resized localization must produce visible boundary pixels. + red_pixels = np.all(result == np.array([255, 0, 0], dtype=np.uint8), axis=-1) + assert red_pixels.any() From b77e04d19609514478b93ed08cf5c197a4d5ba4c Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:23:23 +0200 Subject: [PATCH 08/17] fix(localization): draw contours from defect masks --- anomavision/visualization/boundary.py | 37 +++++++++++++++++++-------- 1 file changed, 26 insertions(+), 11 deletions(-) diff --git a/anomavision/visualization/boundary.py b/anomavision/visualization/boundary.py index c7072679..21e97be5 100644 --- a/anomavision/visualization/boundary.py +++ b/anomavision/visualization/boundary.py @@ -93,27 +93,42 @@ def boundary_image( """ image = to_numpy(image).copy() - mask = to_numpy(patch_classification).copy() + mask = np.squeeze(to_numpy(patch_classification).copy()) - # Keep the localization boundary binary until it is applied to the image. - # Resizing a one-pixel boundary with area interpolation can make values - # fractional; exact binary-mask checks then erase the boundary. - mask = np.squeeze(mask) if mask.ndim != 2: raise ValueError( f"patch_classification must be a 2D mask after squeezing; got shape {mask.shape}" ) - found_boundaries = find_boundaries(mask > 0.5, mode="outer") + # Resize the localization mask itself, not its one-pixel boundary. + # This preserves the defect region and lets OpenCV trace a visible + # contour at the final image resolution. + binary_mask = (mask > 0.5).astype(np.uint8) - if found_boundaries.shape != image.shape[:2]: - found_boundaries = cv2.resize( - found_boundaries.astype(np.uint8), + if binary_mask.shape != image.shape[:2]: + binary_mask = cv2.resize( + binary_mask, (image.shape[1], image.shape[0]), interpolation=cv2.INTER_NEAREST, - ).astype(bool) + ) + + # Fill tiny gaps introduced by patch/grid localization while keeping + # separate defects as separate regions. + kernel = np.ones((3, 3), dtype=np.uint8) + binary_mask = cv2.morphologyEx(binary_mask, cv2.MORPH_CLOSE, kernel) + + contours, _ = cv2.findContours( + binary_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE + ) b_image = image.copy() - b_image[found_boundaries] = np.asarray(boundary_color, dtype=b_image.dtype) + if contours: + cv2.drawContours( + b_image, + contours, + contourIdx=-1, + color=tuple(int(v) for v in boundary_color), + thickness=max(2, min(image.shape[:2]) // 150), + ) return b_image From fb40fc11a7883fb7d43c74481be517fc77e6efc7 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:23:30 +0200 Subject: [PATCH 09/17] fix(api): use localization mask for boundary status --- apps/api/fastapi_app.py | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/apps/api/fastapi_app.py b/apps/api/fastapi_app.py index ae096eca..636733a6 100644 --- a/apps/api/fastapi_app.py +++ b/apps/api/fastapi_app.py @@ -201,11 +201,23 @@ def create_visualizations( ): """ Mirror detect.py's visualization path. + + The localization mask is the source of truth for both the defect contour + and image anomaly status. The image-level score is intentionally not used + to decide whether a localization frame is drawn. """ score_map_classifications = anomavision.classification( score_maps, ANOMALY_THRESHOLD ) - image_classifications = anomavision.classification(image_scores, ANOMALY_THRESHOLD) + image_classifications = ( + np.any( + np.asarray(score_map_classifications).reshape( + score_map_classifications.shape[0], -1 + ) + > 0, + axis=1, + ) + ).astype(np.int64) test_images = np.array([image_np]) From 996dd562e274604ca60ae3444bad3112affc4ac7 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:29:25 +0200 Subject: [PATCH 10/17] fix(api): resolve localization threshold per algorithm --- apps/inference_engine.py | 24 +++++++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/apps/inference_engine.py b/apps/inference_engine.py index c419cead..a5775fe1 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -27,7 +27,22 @@ # ----------------------------------------------------------------------------- # Config — all overridable via environment variables # ----------------------------------------------------------------------------- -ANOMALY_THRESHOLD = float(os.getenv("ANOMAVISION_THRESHOLD", "13.0")) +# Keep an explicit environment override authoritative. Otherwise the +# threshold follows the algorithm encoded by the active model path. +_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD") +ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0 + + +def _threshold_for_model(model_path: str) -> float: + """Return the algorithm-appropriate pixel threshold for a model artifact.""" + if _THRESHOLD_OVERRIDE: + return float(_THRESHOLD_OVERRIDE) + normalized = os.path.normpath(model_path).lower() + if "patchcore" in normalized: + return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25")) + if "efficientad" in normalized: + return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) + return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) MODEL_DATA_PATH = os.getenv("ANOMAVISION_MODEL_DATA_PATH", "") MODEL_FILE = os.getenv("ANOMAVISION_MODEL_FILE", "model.onnx") STUDIO_ROOT = os.path.expanduser( @@ -214,6 +229,13 @@ def load_model(project_id: Optional[str] = None) -> str: _sess = ort.InferenceSession(model_path, providers=providers, sess_options=opts) _input_name = _sess.get_inputs()[0].name + # The previous API used a fixed PaDiM threshold (13.0) for every model. + # PatchCore maps use a much smaller score scale, so that made the heatmap + # show the defect while the binary localization mask was completely empty. + global ANOMALY_THRESHOLD + ANOMALY_THRESHOLD = _threshold_for_model(model_path) + print(f"[inference] Localization threshold: {ANOMALY_THRESHOLD}") + # Warmup — run twice so JIT compile happens now, not on the first real request dummy_shape = tuple( d if isinstance(d, int) and d > 0 else 1 for d in _sess.get_inputs()[0].shape From de93be10cf0fba7818518c6e6c22adebf852e73b Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:37:30 +0200 Subject: [PATCH 11/17] fix(localization): derive visible defect mask from anomaly heatmap --- apps/inference_engine.py | 25 +++++++++++++++++++++---- 1 file changed, 21 insertions(+), 4 deletions(-) diff --git a/apps/inference_engine.py b/apps/inference_engine.py index a5775fe1..792e3492 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -18,6 +18,7 @@ from typing import Optional import numpy as np +import cv2 import onnxruntime as ort from onnxruntime import GraphOptimizationLevel, SessionOptions from PIL import Image @@ -305,15 +306,31 @@ def run( # Visualizations are optional. Live/camera inference does not need them; # skipping this CPU-heavy path keeps latency close to the raw ONNX runtime. if include_visualizations: - score_map_cls = classification(score_maps, threshold) - # Use the localized pixel mask as the source of truth. The image-level - # score alone must not produce ANOMALY when no pixel is localized. + # The absolute anomaly threshold decides whether the image is anomalous. + # The display mask is derived from the same score map, but normalized + # per image so the spatial peak visible in the heatmap remains drawable. + raw_map = np.asarray(score_maps, dtype=np.float32) + flat_map = raw_map.reshape(raw_map.shape[0], -1) + map_min = flat_map.min(axis=1)[:, None, None] + map_max = flat_map.max(axis=1)[:, None, None] + normalized_maps = (raw_map - map_min) / np.maximum(map_max - map_min, 1e-8) + score_map_cls = (normalized_maps >= 0.60).astype(np.uint8) + + kernel = np.ones((5, 5), dtype=np.uint8) + for i in range(score_map_cls.shape[0]): + score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_CLOSE, kernel) + score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_OPEN, kernel) + + model_mask = classification(score_maps, threshold) image_cls = ( np.any( - np.asarray(score_map_cls).reshape(score_map_cls.shape[0], -1) > 0, + np.asarray(model_mask).reshape(model_mask.shape[0], -1) > 0, axis=1, ) ).astype(np.int64) + score_map_cls[image_cls == 0] = 0 + # Use the localized pixel mask as the source of truth. The image-level + # score alone must not produce ANOMALY when no pixel is localized. test_images = np.array([image_np]) boundary_np = visualization.framed_boundary_images( test_images, score_map_cls, image_cls, padding=VIZ_PADDING From 5f188bdf1eaef7d618ca7d4045e1b3e05eb24eeb Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:52:23 +0200 Subject: [PATCH 12/17] fix(api): use algorithm-specific localization threshold --- apps/api/fastapi_app.py | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/apps/api/fastapi_app.py b/apps/api/fastapi_app.py index 636733a6..d1aeaa52 100644 --- a/apps/api/fastapi_app.py +++ b/apps/api/fastapi_app.py @@ -30,9 +30,25 @@ model_type: Optional[ModelType] = None drift_runtime: Optional[InferenceDriftRuntime] = None -ANOMALY_THRESHOLD = 13.0 +# Pixel thresholds are algorithm-specific. A single PaDiM threshold (13.0) +# makes PatchCore masks empty even when the image is correctly classified as +# anomalous (PatchCore scores are typically much smaller). +_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD") +ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0 RESIZE_SIZE = (224, 224) + +def _threshold_for_model(model_path: str) -> float: + """Return the pixel-localization threshold for the active model.""" + if _THRESHOLD_OVERRIDE: + return float(_THRESHOLD_OVERRIDE) + normalized = os.path.normpath(model_path).lower() + if "patchcore" in normalized: + return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25")) + if "efficientad" in normalized: + return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) + return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) + # You can override these via environment variables MODEL_DATA_PATH = os.getenv( "ANOMAVISION_MODEL_DATA_PATH", "distributions/padim/bottle/anomav_exp" @@ -56,7 +72,7 @@ async def load_model(): model = ModelWrapper(model_path, device_str) model_type = ModelType.from_extension(model_path) """ - global model, model_type, drift_runtime + global model, model_type, drift_runtime, ANOMALY_THRESHOLD device_str = determine_device(DEVICE) # "cpu" or "cuda" model_path = os.path.realpath(os.path.join(MODEL_DATA_PATH, MODEL_FILE)) @@ -67,6 +83,8 @@ async def load_model(): # ModelType is inferred from extension (.pt/.onnx/.engine/...) model_type = ModelType.from_extension(model_path) model = ModelWrapper(model_path, device_str) + ANOMALY_THRESHOLD = _threshold_for_model(model_path) + print(f"[api] Localization threshold: {ANOMALY_THRESHOLD}") if DRIFT_REFERENCE: reference = load_embeddings(DRIFT_REFERENCE) From dffebadd491d25cfcb76c75ec829c0eeb8cb8451 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:52:40 +0200 Subject: [PATCH 13/17] fix(localization): keep anomaly decision separate from pixel map --- apps/inference_engine.py | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/apps/inference_engine.py b/apps/inference_engine.py index 792e3492..65933f0d 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -321,12 +321,11 @@ def run( score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_CLOSE, kernel) score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_OPEN, kernel) - model_mask = classification(score_maps, threshold) + # The image-level score is the anomaly decision. The score map is + # localization data and may have a different numerical scale from the + # image score, so do not use a pixel threshold to decide the frame color. image_cls = ( - np.any( - np.asarray(model_mask).reshape(model_mask.shape[0], -1) > 0, - axis=1, - ) + np.asarray(image_scores).reshape(-1) >= float(threshold) ).astype(np.int64) score_map_cls[image_cls == 0] = 0 # Use the localized pixel mask as the source of truth. The image-level @@ -344,10 +343,7 @@ def run( latency_ms = (time.perf_counter() - t0) * 1000 - pixel_mask = classification(score_maps, threshold) - is_anomaly = bool( - np.any(np.asarray(pixel_mask).reshape(pixel_mask.shape[0], -1) > 0) - ) + is_anomaly = bool(float(image_score) >= float(threshold)) return InferenceResult( anomaly_score=image_score, From 036377eaed78f793aa7ed8fe38b7111dab8e401b Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:05:15 +0200 Subject: [PATCH 14/17] fix(localization): isolate strongest PatchCore defect regions --- apps/inference_engine.py | 54 ++++++++++++++++++++++++++++------------ 1 file changed, 38 insertions(+), 16 deletions(-) diff --git a/apps/inference_engine.py b/apps/inference_engine.py index 65933f0d..1ffe5d64 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -306,30 +306,52 @@ def run( # Visualizations are optional. Live/camera inference does not need them; # skipping this CPU-heavy path keeps latency close to the raw ONNX runtime. if include_visualizations: - # The absolute anomaly threshold decides whether the image is anomalous. - # The display mask is derived from the same score map, but normalized - # per image so the spatial peak visible in the heatmap remains drawable. + # The image score decides whether the frame is anomalous. + # Localization is derived independently from the spatial PatchCore map. + # + # Do not use a fixed normalized threshold such as 0.60 here: for a + # localized defect that can select a large portion of the object/image. + # Instead, keep only the strongest spatial response (top 5% of pixels) + # and then retain the connected high-score regions. This makes the + # contour follow the defect rather than the whole image. raw_map = np.asarray(score_maps, dtype=np.float32) flat_map = raw_map.reshape(raw_map.shape[0], -1) map_min = flat_map.min(axis=1)[:, None, None] map_max = flat_map.max(axis=1)[:, None, None] - normalized_maps = (raw_map - map_min) / np.maximum(map_max - map_min, 1e-8) - score_map_cls = (normalized_maps >= 0.60).astype(np.uint8) + normalized_maps = (raw_map - map_min) / np.maximum( + map_max - map_min, 1e-8 + ) + + score_map_cls = np.zeros_like(normalized_maps, dtype=np.uint8) + for i, normalized_map in enumerate(normalized_maps): + localization_threshold = float(np.quantile(normalized_map, 0.95)) + mask = (normalized_map >= localization_threshold).astype(np.uint8) + + # Close small gaps without expanding the defect excessively. + kernel = np.ones((3, 3), dtype=np.uint8) + mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel) - kernel = np.ones((5, 5), dtype=np.uint8) - for i in range(score_map_cls.shape[0]): - score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_CLOSE, kernel) - score_map_cls[i] = cv2.morphologyEx(score_map_cls[i], cv2.MORPH_OPEN, kernel) + # Keep only connected regions containing one of the strongest + # PatchCore responses. This removes scattered background pixels. + num_labels, labels, stats, _ = cv2.connectedComponentsWithStats( + mask, connectivity=8 + ) + if num_labels > 1: + peaks = [] + for label in range(1, num_labels): + component = normalized_map[labels == label] + peaks.append((float(component.max()), label)) + keep = {label for _, label in sorted(peaks, reverse=True)[:3]} + mask = np.isin(labels, list(keep)).astype(np.uint8) + + score_map_cls[i] = mask # The image-level score is the anomaly decision. The score map is - # localization data and may have a different numerical scale from the - # image score, so do not use a pixel threshold to decide the frame color. - image_cls = ( - np.asarray(image_scores).reshape(-1) >= float(threshold) - ).astype(np.int64) + # localization data and must not determine the outer anomaly frame. + image_cls = np.array( + [int(float(image_score) >= float(threshold))], dtype=np.int64 + ) score_map_cls[image_cls == 0] = 0 - # Use the localized pixel mask as the source of truth. The image-level - # score alone must not produce ANOMALY when no pixel is localized. test_images = np.array([image_np]) boundary_np = visualization.framed_boundary_images( test_images, score_map_cls, image_cls, padding=VIZ_PADDING From 9c76ab6c6540c67f788e25f677000a994d3c42b9 Mon Sep 17 00:00:00 2001 From: Deep Knowledge <66887716+DeepKnowledge1@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:07:34 +0200 Subject: [PATCH 15/17] fix(detect): separate anomaly decision from defect localization --- anomavision/detect.py | 40 +++++++++++++++++++++++----------------- 1 file changed, 23 insertions(+), 17 deletions(-) diff --git a/anomavision/detect.py b/anomavision/detect.py index cec66d2a..b7553f16 100644 --- a/anomavision/detect.py +++ b/anomavision/detect.py @@ -515,26 +515,32 @@ def _save_live_drift_status() -> None: score_maps, kernel_size=33, sigma=4 ) if config.thresh is not None: - # Localization is the source of truth for anomaly - # classification: an image is anomalous only when at - # least one pixel in its anomaly map reaches the - # configured threshold. This prevents the image-level - # score from reporting ANOMALY without localization. - localization_masks = anomavision.classification( - score_maps, config.thresh - ) + # Keep the image-level anomaly decision based on the + # model's image score. The pixel score map is used + # independently for spatial localization so a valid + # image-level anomaly is not lost just because the + # fixed image threshold is not a good pixel threshold. is_anomaly = ( - np.any( - np.asarray(localization_masks).reshape( - len(localization_masks), -1 - ) - > 0, - axis=1, - ) + np.asarray(image_scores).reshape(-1) >= float(config.thresh) ).astype(np.int64) + + # Build a spatial defect mask from the strongest + # responses in each anomalous image. Do not use the + # image threshold directly as a pixel threshold: + # PatchCore's image score and pixel-map values have + # different distributions. + localization_masks = make_localization_mask( + score_maps, + is_anomaly, + quantile=0.90, + ) else: - localization_masks = np.zeros_like(score_maps) - is_anomaly = np.zeros(score_maps.shape[0], dtype=np.int64) + localization_masks = np.zeros_like( + np.asarray(score_maps), dtype=np.uint8 + ) + is_anomaly = np.zeros( + np.asarray(score_maps).shape[0], dtype=np.int64 + ) if not stream_mode: results_accumulator["scores"].extend(image_scores.tolist()) From 67a026677ba3f00a94f9ae756a6d2f22f484299e Mon Sep 17 00:00:00 2001 From: DeepKnowledge1 Date: Thu, 1 Oct 2026 15:31:17 +0200 Subject: [PATCH 16/17] fix(detect): restore PatchCore localization masks PatchCore pixel maps are cosine distances, so thresholding them with the image-level threshold gave empty or full-image masks. For PatchCore, classify by image score and build the mask from a per-image min-max normalized map with a relative cutoff (patchcore_loc_rel, default 0.60). PaDiM behavior is unchanged. --- anomavision/detect.py | 70 +++++++++++++++++++++++++++---------------- config.yml | 8 ++--- 2 files changed, 49 insertions(+), 29 deletions(-) diff --git a/anomavision/detect.py b/anomavision/detect.py index b7553f16..ad0e2862 100644 --- a/anomavision/detect.py +++ b/anomavision/detect.py @@ -515,32 +515,52 @@ def _save_live_drift_status() -> None: score_maps, kernel_size=33, sigma=4 ) if config.thresh is not None: - # Keep the image-level anomaly decision based on the - # model's image score. The pixel score map is used - # independently for spatial localization so a valid - # image-level anomaly is not lost just because the - # fixed image threshold is not a good pixel threshold. - is_anomaly = ( - np.asarray(image_scores).reshape(-1) >= float(config.thresh) - ).astype(np.int64) - - # Build a spatial defect mask from the strongest - # responses in each anomalous image. Do not use the - # image threshold directly as a pixel threshold: - # PatchCore's image score and pixel-map values have - # different distributions. - localization_masks = make_localization_mask( - score_maps, - is_anomaly, - quantile=0.90, - ) + if str(config.get("algorithm", "")).lower() == "patchcore": + # PatchCore: config.thresh is an IMAGE-level threshold + # (cosine-distance scale). The blurred pixel map lives + # on a different scale, so thresholding it directly + # gives either an empty or a full-image mask. Classify + # by image score, then localize with a per-image + # relative cutoff on the score map. + scores_np = np.asarray( + image_scores.detach().float().cpu().numpy() + if hasattr(image_scores, "detach") + else image_scores + ).reshape(-1) + is_anomaly = (scores_np >= float(config.thresh)).astype( + np.int64 + ) + # Cheap relative cutoff on the min-max normalised map + maps_np = ( + score_maps.detach().float().cpu().numpy() + if hasattr(score_maps, "detach") + else np.asarray(score_maps) + ) + lo = maps_np.min(axis=(1, 2), keepdims=True) + hi = maps_np.max(axis=(1, 2), keepdims=True) + norm = (maps_np - lo) / (hi - lo + 1e-8) + localization_masks = ( + (norm >= float(config.get("patchcore_loc_rel", 0.60))) + & (is_anomaly[:, None, None] > 0) + ).astype(np.uint8) + else: + # PaDiM etc.: an image is anomalous only when at + # least one pixel reaches the configured threshold. + localization_masks = anomavision.classification( + score_maps, config.thresh + ) + is_anomaly = ( + np.any( + np.asarray(localization_masks).reshape( + len(localization_masks), -1 + ) + > 0, + axis=1, + ) + ).astype(np.int64) else: - localization_masks = np.zeros_like( - np.asarray(score_maps), dtype=np.uint8 - ) - is_anomaly = np.zeros( - np.asarray(score_maps).shape[0], dtype=np.int64 - ) + localization_masks = np.zeros_like(score_maps) + is_anomaly = np.zeros(score_maps.shape[0], dtype=np.int64) if not stream_mode: results_accumulator["scores"].extend(image_scores.tolist()) diff --git a/config.yml b/config.yml index 85491ead..fcbe8ac0 100644 --- a/config.yml +++ b/config.yml @@ -1,8 +1,8 @@ # ========================= # Dataset / preprocessing (shared by train, detect, eval, stream) # ========================= -dataset_path: "D:/01-DATA" # Root dataset directory (contains train/test folders) -class_name: "bottle" # Dataset class to train/evaluate +dataset_path: "D:/01-DATA/mvtec" # Root dataset directory (contains train/test folders) +class_name: "cable" # Dataset class to train/evaluate resize: [224, 224] # Input image size [width, height] crop_size: # Optional center crop size [width, height] normalize: true # Apply ImageNet normalization @@ -67,8 +67,8 @@ detailed_timing: false # Enable detailed timing information # ========================= # Visualization (detect/eval) # ========================= -enable_visualization: false # Enable anomaly visualization -save_visualizations: false # Save generated visualizations to disk +enable_visualization: true # Enable anomaly visualization +save_visualizations: true # Save generated visualizations to disk viz_output_dir: "./visualizations/" # Directory for visualization output viz_alpha: 0.5 # Heatmap overlay transparency viz_padding: 40 # Extra visualization border/padding From d220e771ddc466b9af8e3c3f07a95491a3f71f4b Mon Sep 17 00:00:00 2001 From: DeepKnowledge1 Date: Thu, 1 Oct 2026 15:34:59 +0200 Subject: [PATCH 17/17] fix(detect): restore PatchCore localization masks PatchCore pixel maps are cosine distances, so thresholding them with the image-level threshold gave empty or full-image masks. For PatchCore, classify by image score and build the mask from a per-image min-max normalized map with a relative cutoff (patchcore_loc_rel, default 0.60). PaDiM behavior is unchanged. --- apps/api/fastapi_app.py | 1 + apps/inference_engine.py | 8 ++++---- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/apps/api/fastapi_app.py b/apps/api/fastapi_app.py index d1aeaa52..42f645b3 100644 --- a/apps/api/fastapi_app.py +++ b/apps/api/fastapi_app.py @@ -49,6 +49,7 @@ def _threshold_for_model(model_path: str) -> float: return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) + # You can override these via environment variables MODEL_DATA_PATH = os.getenv( "ANOMAVISION_MODEL_DATA_PATH", "distributions/padim/bottle/anomav_exp" diff --git a/apps/inference_engine.py b/apps/inference_engine.py index 1ffe5d64..0918d8a4 100644 --- a/apps/inference_engine.py +++ b/apps/inference_engine.py @@ -17,8 +17,8 @@ from dataclasses import dataclass from typing import Optional -import numpy as np import cv2 +import numpy as np import onnxruntime as ort from onnxruntime import GraphOptimizationLevel, SessionOptions from PIL import Image @@ -44,6 +44,8 @@ def _threshold_for_model(model_path: str) -> float: if "efficientad" in normalized: return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0")) return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0")) + + MODEL_DATA_PATH = os.getenv("ANOMAVISION_MODEL_DATA_PATH", "") MODEL_FILE = os.getenv("ANOMAVISION_MODEL_FILE", "model.onnx") STUDIO_ROOT = os.path.expanduser( @@ -318,9 +320,7 @@ def run( flat_map = raw_map.reshape(raw_map.shape[0], -1) map_min = flat_map.min(axis=1)[:, None, None] map_max = flat_map.max(axis=1)[:, None, None] - normalized_maps = (raw_map - map_min) / np.maximum( - map_max - map_min, 1e-8 - ) + normalized_maps = (raw_map - map_min) / np.maximum(map_max - map_min, 1e-8) score_map_cls = np.zeros_like(normalized_maps, dtype=np.uint8) for i, normalized_map in enumerate(normalized_maps):