Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
17 commits
Select commit Hold shift + click to select a range
c6c8eeb
test(patchcore): guard deterministic inference regression
DeepKnowledge1 Oct 1, 2026
900199a
fix(features): select ResNet stages by original indices
DeepKnowledge1 Oct 1, 2026
edc3b4c
fix(patchcore): use dedicated multi-scale feature stages
DeepKnowledge1 Oct 1, 2026
be9b2d4
fix(config): use multi-scale PatchCore feature layers
DeepKnowledge1 Oct 1, 2026
1162480
fix(patchcore): default to multi-scale ResNet features
DeepKnowledge1 Oct 1, 2026
93d2dad
fix(visualization): preserve thin defect boundaries
DeepKnowledge1 Oct 1, 2026
e032082
test(visualization): guard defect boundary rendering
DeepKnowledge1 Oct 1, 2026
b77e04d
fix(localization): draw contours from defect masks
DeepKnowledge1 Oct 1, 2026
fb40fc1
fix(api): use localization mask for boundary status
DeepKnowledge1 Oct 1, 2026
996dd56
fix(api): resolve localization threshold per algorithm
DeepKnowledge1 Oct 1, 2026
de93be1
fix(localization): derive visible defect mask from anomaly heatmap
DeepKnowledge1 Oct 1, 2026
5f188bd
fix(api): use algorithm-specific localization threshold
DeepKnowledge1 Oct 1, 2026
dffebad
fix(localization): keep anomaly decision separate from pixel map
DeepKnowledge1 Oct 1, 2026
036377e
fix(localization): isolate strongest PatchCore defect regions
DeepKnowledge1 Oct 1, 2026
9c76ab6
fix(detect): separate anomaly decision from defect localization
DeepKnowledge1 Oct 1, 2026
67a0266
fix(detect): restore PatchCore localization masks
DeepKnowledge1 Oct 1, 2026
d220e77
fix(detect): restore PatchCore localization masks
DeepKnowledge1 Oct 1, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 10 additions & 1 deletion anomavision/algorithm/common/feature_extraction.py
Original file line number Diff line number Diff line change
Expand Up @@ -163,7 +163,16 @@ def forward(
layers.append(out4)

if layer_indices is not None:
layers = [layers[i] for i in layer_indices]
# Indices refer to the original ResNet stages:
# 0=layer1, 1=layer2, 2=layer3, 3=layer4.
stage_outputs = [out1]
if max_l >= 1:
stage_outputs.append(out2)
if max_l >= 2:
stage_outputs.append(out3)
if max_l >= 3:
stage_outputs.append(out4)
layers = [stage_outputs[i] for i in layer_indices]

if layer_hook is not None:
layers = [layer_hook(layer) for layer in layers]
Expand Down
6 changes: 3 additions & 3 deletions anomavision/algorithm/patchcore/patchcore.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,8 +43,8 @@ class PatchCore(torch.nn.Module):
backbone: Feature-extraction backbone. Supported values are ``resnet18`` and
``wide_resnet50``.
device: Device used for feature extraction and nearest-neighbor distance.
layer_indices: ResNet feature stages to concatenate. Defaults to ``[0, 1]``
to keep the lightweight model fast and compact.
layer_indices: ResNet feature stages to concatenate. Defaults to ``[1, 2]``
to provide multi-scale mid/deep features while remaining lightweight.
memory_bank: Optional precomputed bank with shape ``(num_patches, dim)``.
Providing it creates a ready-to-infer model.
coreset_ratio: Fraction of extracted normal patches to retain. Must be in
Expand Down Expand Up @@ -90,7 +90,7 @@ def __init__(
)
self.device = torch.device(device)
self.backbone = backbone
self.layer_indices = list(layer_indices or [0, 1])
self.layer_indices = list(layer_indices or [1, 2])
self.coreset_ratio = float(coreset_ratio)
self.max_memory_patches = max_memory_patches
self.patch_grid = patch_grid
Expand Down
58 changes: 42 additions & 16 deletions anomavision/detect.py
Original file line number Diff line number Diff line change
Expand Up @@ -515,23 +515,49 @@ def _save_live_drift_status() -> None:
score_maps, kernel_size=33, sigma=4
)
if config.thresh is not None:
# Localization is the source of truth for anomaly
# classification: an image is anomalous only when at
# least one pixel in its anomaly map reaches the
# configured threshold. This prevents the image-level
# score from reporting ANOMALY without localization.
localization_masks = anomavision.classification(
score_maps, config.thresh
)
is_anomaly = (
np.any(
np.asarray(localization_masks).reshape(
len(localization_masks), -1
)
> 0,
axis=1,
if str(config.get("algorithm", "")).lower() == "patchcore":
# PatchCore: config.thresh is an IMAGE-level threshold
# (cosine-distance scale). The blurred pixel map lives
# on a different scale, so thresholding it directly
# gives either an empty or a full-image mask. Classify
# by image score, then localize with a per-image
# relative cutoff on the score map.
scores_np = np.asarray(
image_scores.detach().float().cpu().numpy()
if hasattr(image_scores, "detach")
else image_scores
).reshape(-1)
is_anomaly = (scores_np >= float(config.thresh)).astype(
np.int64
)
).astype(np.int64)
# Cheap relative cutoff on the min-max normalised map
maps_np = (
score_maps.detach().float().cpu().numpy()
if hasattr(score_maps, "detach")
else np.asarray(score_maps)
)
lo = maps_np.min(axis=(1, 2), keepdims=True)
hi = maps_np.max(axis=(1, 2), keepdims=True)
norm = (maps_np - lo) / (hi - lo + 1e-8)
localization_masks = (
(norm >= float(config.get("patchcore_loc_rel", 0.60)))
& (is_anomaly[:, None, None] > 0)
).astype(np.uint8)
else:
# PaDiM etc.: an image is anomalous only when at
# least one pixel reaches the configured threshold.
localization_masks = anomavision.classification(
score_maps, config.thresh
)
is_anomaly = (
np.any(
np.asarray(localization_masks).reshape(
len(localization_masks), -1
)
> 0,
axis=1,
)
).astype(np.int64)
else:
localization_masks = np.zeros_like(score_maps)
is_anomaly = np.zeros(score_maps.shape[0], dtype=np.int64)
Expand Down
15 changes: 13 additions & 2 deletions anomavision/train.py
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,13 @@ def create_parser(add_help: bool = True) -> argparse.ArgumentParser:
default=None,
help="Backbone feature layers.",
)
parser.add_argument(
"--patchcore_layer_indices",
type=int,
nargs="+",
default=None,
help="ResNet stages used by PatchCore; defaults to [1, 2].",
)
parser.add_argument(
"--coreset_ratio",
type=float,
Expand Down Expand Up @@ -226,14 +233,18 @@ def run_training(args):
"cfg: algorithm=%s | backbone=%s | layers=%s",
config.algorithm,
config.backbone,
config.layer_indices,
(
config.get("patchcore_layer_indices", [1, 2])
if str(config.algorithm).lower() == "patchcore"
else config.layer_indices
),
)

if str(config.algorithm).lower() == "patchcore":
model = anomavision.PatchCore(
backbone=config.backbone,
device=device,
layer_indices=config.layer_indices,
layer_indices=config.get("patchcore_layer_indices", [1, 2]),
coreset_ratio=float(config.coreset_ratio),
max_memory_patches=config.max_memory_patches,
patch_grid=config.patch_grid,
Expand Down
43 changes: 37 additions & 6 deletions anomavision/visualization/boundary.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
from typing import Tuple, Union

import cv2
import numpy as np
import torch
from skimage.segmentation import find_boundaries
Expand Down Expand Up @@ -92,12 +93,42 @@ def boundary_image(
"""

image = to_numpy(image).copy()
mask = to_numpy(patch_classification).copy()

found_boundaries = find_boundaries(mask).astype(np.uint8)
layer_two = np.zeros(image.shape, dtype=np.uint8)
layer_two[:] = boundary_color
mask = np.squeeze(to_numpy(patch_classification).copy())

if mask.ndim != 2:
raise ValueError(
f"patch_classification must be a 2D mask after squeezing; got shape {mask.shape}"
)

# Resize the localization mask itself, not its one-pixel boundary.
# This preserves the defect region and lets OpenCV trace a visible
# contour at the final image resolution.
binary_mask = (mask > 0.5).astype(np.uint8)

if binary_mask.shape != image.shape[:2]:
binary_mask = cv2.resize(
binary_mask,
(image.shape[1], image.shape[0]),
interpolation=cv2.INTER_NEAREST,
)

# Fill tiny gaps introduced by patch/grid localization while keeping
# separate defects as separate regions.
kernel = np.ones((3, 3), dtype=np.uint8)
binary_mask = cv2.morphologyEx(binary_mask, cv2.MORPH_CLOSE, kernel)

contours, _ = cv2.findContours(
binary_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE
)

b_image = composite_image(image, layer_two, found_boundaries)
b_image = image.copy()
if contours:
cv2.drawContours(
b_image,
contours,
contourIdx=-1,
color=tuple(int(v) for v in boundary_color),
thickness=max(2, min(image.shape[:2]) // 150),
)

return b_image
37 changes: 34 additions & 3 deletions apps/api/fastapi_app.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,9 +30,26 @@
model_type: Optional[ModelType] = None
drift_runtime: Optional[InferenceDriftRuntime] = None

ANOMALY_THRESHOLD = 13.0
# Pixel thresholds are algorithm-specific. A single PaDiM threshold (13.0)
# makes PatchCore masks empty even when the image is correctly classified as
# anomalous (PatchCore scores are typically much smaller).
_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD")
ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0
RESIZE_SIZE = (224, 224)


def _threshold_for_model(model_path: str) -> float:
"""Return the pixel-localization threshold for the active model."""
if _THRESHOLD_OVERRIDE:
return float(_THRESHOLD_OVERRIDE)
normalized = os.path.normpath(model_path).lower()
if "patchcore" in normalized:
return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25"))
if "efficientad" in normalized:
return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0"))
return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0"))


# You can override these via environment variables
MODEL_DATA_PATH = os.getenv(
"ANOMAVISION_MODEL_DATA_PATH", "distributions/padim/bottle/anomav_exp"
Expand All @@ -56,7 +73,7 @@ async def load_model():
model = ModelWrapper(model_path, device_str)
model_type = ModelType.from_extension(model_path)
"""
global model, model_type, drift_runtime
global model, model_type, drift_runtime, ANOMALY_THRESHOLD

device_str = determine_device(DEVICE) # "cpu" or "cuda"
model_path = os.path.realpath(os.path.join(MODEL_DATA_PATH, MODEL_FILE))
Expand All @@ -67,6 +84,8 @@ async def load_model():
# ModelType is inferred from extension (.pt/.onnx/.engine/...)
model_type = ModelType.from_extension(model_path)
model = ModelWrapper(model_path, device_str)
ANOMALY_THRESHOLD = _threshold_for_model(model_path)
print(f"[api] Localization threshold: {ANOMALY_THRESHOLD}")

if DRIFT_REFERENCE:
reference = load_embeddings(DRIFT_REFERENCE)
Expand Down Expand Up @@ -201,11 +220,23 @@ def create_visualizations(
):
"""
Mirror detect.py's visualization path.

The localization mask is the source of truth for both the defect contour
and image anomaly status. The image-level score is intentionally not used
to decide whether a localization frame is drawn.
"""
score_map_classifications = anomavision.classification(
score_maps, ANOMALY_THRESHOLD
)
image_classifications = anomavision.classification(image_scores, ANOMALY_THRESHOLD)
image_classifications = (
np.any(
np.asarray(score_map_classifications).reshape(
score_map_classifications.shape[0], -1
)
> 0,
axis=1,
)
).astype(np.int64)

test_images = np.array([image_np])

Expand Down
83 changes: 70 additions & 13 deletions apps/inference_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
from dataclasses import dataclass
from typing import Optional

import cv2
import numpy as np
import onnxruntime as ort
from onnxruntime import GraphOptimizationLevel, SessionOptions
Expand All @@ -27,7 +28,24 @@
# -----------------------------------------------------------------------------
# Config β€” all overridable via environment variables
# -----------------------------------------------------------------------------
ANOMALY_THRESHOLD = float(os.getenv("ANOMAVISION_THRESHOLD", "13.0"))
# Keep an explicit environment override authoritative. Otherwise the
# threshold follows the algorithm encoded by the active model path.
_THRESHOLD_OVERRIDE = os.getenv("ANOMAVISION_THRESHOLD")
ANOMALY_THRESHOLD = float(_THRESHOLD_OVERRIDE) if _THRESHOLD_OVERRIDE else 13.0


def _threshold_for_model(model_path: str) -> float:
"""Return the algorithm-appropriate pixel threshold for a model artifact."""
if _THRESHOLD_OVERRIDE:
return float(_THRESHOLD_OVERRIDE)
normalized = os.path.normpath(model_path).lower()
if "patchcore" in normalized:
return float(os.getenv("ANOMAVISION_PATCHCORE_THRESHOLD", "0.25"))
if "efficientad" in normalized:
return float(os.getenv("ANOMAVISION_EFFICIENTAD_THRESHOLD", "1.0"))
return float(os.getenv("ANOMAVISION_PADIM_THRESHOLD", "13.0"))


MODEL_DATA_PATH = os.getenv("ANOMAVISION_MODEL_DATA_PATH", "")
MODEL_FILE = os.getenv("ANOMAVISION_MODEL_FILE", "model.onnx")
STUDIO_ROOT = os.path.expanduser(
Expand Down Expand Up @@ -214,6 +232,13 @@ def load_model(project_id: Optional[str] = None) -> str:
_sess = ort.InferenceSession(model_path, providers=providers, sess_options=opts)
_input_name = _sess.get_inputs()[0].name

# The previous API used a fixed PaDiM threshold (13.0) for every model.
# PatchCore maps use a much smaller score scale, so that made the heatmap
# show the defect while the binary localization mask was completely empty.
global ANOMALY_THRESHOLD
ANOMALY_THRESHOLD = _threshold_for_model(model_path)
print(f"[inference] Localization threshold: {ANOMALY_THRESHOLD}")

# Warmup β€” run twice so JIT compile happens now, not on the first real request
dummy_shape = tuple(
d if isinstance(d, int) and d > 0 else 1 for d in _sess.get_inputs()[0].shape
Expand Down Expand Up @@ -283,15 +308,50 @@ def run(
# Visualizations are optional. Live/camera inference does not need them;
# skipping this CPU-heavy path keeps latency close to the raw ONNX runtime.
if include_visualizations:
score_map_cls = classification(score_maps, threshold)
# Use the localized pixel mask as the source of truth. The image-level
# score alone must not produce ANOMALY when no pixel is localized.
image_cls = (
np.any(
np.asarray(score_map_cls).reshape(score_map_cls.shape[0], -1) > 0,
axis=1,
# The image score decides whether the frame is anomalous.
# Localization is derived independently from the spatial PatchCore map.
#
# Do not use a fixed normalized threshold such as 0.60 here: for a
# localized defect that can select a large portion of the object/image.
# Instead, keep only the strongest spatial response (top 5% of pixels)
# and then retain the connected high-score regions. This makes the
# contour follow the defect rather than the whole image.
raw_map = np.asarray(score_maps, dtype=np.float32)
flat_map = raw_map.reshape(raw_map.shape[0], -1)
map_min = flat_map.min(axis=1)[:, None, None]
map_max = flat_map.max(axis=1)[:, None, None]
normalized_maps = (raw_map - map_min) / np.maximum(map_max - map_min, 1e-8)

score_map_cls = np.zeros_like(normalized_maps, dtype=np.uint8)
for i, normalized_map in enumerate(normalized_maps):
localization_threshold = float(np.quantile(normalized_map, 0.95))
mask = (normalized_map >= localization_threshold).astype(np.uint8)

# Close small gaps without expanding the defect excessively.
kernel = np.ones((3, 3), dtype=np.uint8)
mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)

# Keep only connected regions containing one of the strongest
# PatchCore responses. This removes scattered background pixels.
num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(
mask, connectivity=8
)
).astype(np.int64)
if num_labels > 1:
peaks = []
for label in range(1, num_labels):
component = normalized_map[labels == label]
peaks.append((float(component.max()), label))
keep = {label for _, label in sorted(peaks, reverse=True)[:3]}
mask = np.isin(labels, list(keep)).astype(np.uint8)

score_map_cls[i] = mask

# The image-level score is the anomaly decision. The score map is
# localization data and must not determine the outer anomaly frame.
image_cls = np.array(
[int(float(image_score) >= float(threshold))], dtype=np.int64
)
score_map_cls[image_cls == 0] = 0
test_images = np.array([image_np])
boundary_np = visualization.framed_boundary_images(
test_images, score_map_cls, image_cls, padding=VIZ_PADDING
Expand All @@ -305,10 +365,7 @@ def run(

latency_ms = (time.perf_counter() - t0) * 1000

pixel_mask = classification(score_maps, threshold)
is_anomaly = bool(
np.any(np.asarray(pixel_mask).reshape(pixel_mask.shape[0], -1) > 0)
)
is_anomaly = bool(float(image_score) >= float(threshold))

return InferenceResult(
anomaly_score=image_score,
Expand Down
Loading
Loading