Anomaly Detection(AD)은 여러 분야에서 다양하게 사용됩니다. 보안에서는 외부 침입 감지에 사용될 수 있고 의료에서는 정상과 환자를 구분하는데 사용될 수 있습니다. 오늘 소개할 논문은 제조 분야에서 사용되는 이미지 기반의 AD기법에서 가장 대표 모델인 PatchCore를 소개하겠습니다. 해당 논문은 2022년에 출판되었으며 오랜시간 이미지 기반의 AD 연구에서 비교모델로 사용되고 있으며 대표적인 모델이라고 볼 수 있습니다.
산업 분야에서 2D 이미지 기반의 AD는 제품 외관에있는 결함을 탐지하는데 주로 사용됩니다. 일반적으로 computer vision에서는 classification과 segmentation 모델로 제품의 양/불량을 판단하고 결함 부위를 감지할 수 있지만 두 모델은 지도학습(supervised learning) 방식이기 때문에 라벨링이 반드시 필요합니다. Deep Learning 기반의 지도학습 모델은 학습 데이터가 적어도 1000개 이상이 있어야지 믿을 만한 성능의 모델을 구축할 수 있습니다. 하지만 실제 산업에서는 불량 제품이 많이 생산되지 않기 때문에 데이터를 다량으로 수집하는데 오랜시간이 소요됩니다. 반면에 양품 데이터는 쉽게 얻을 수 있으며 이처럼 불량을 감지하고 싶으나 불량 데이터가 희소한 경우에 AD 기법이 사용 될 수 있습니다.
PatchCore는 대표적인 AD 모델 중에 하나로, 메모리 뱅크 기반으로 이상탐지를 수행합니다. 메모리 뱅크 방식은 embedding-based methods에 속하며, 이외에도 reconstruction-based methods와 synthesis-based methods가 있습니다. PatchCore는 PaDiM을 보완한 모델이며 PaDiM보다 계산 속도가 빠르고 개별 패치를 사용하는 PaDiM과 달리 주변 패치 정보까지 활용하여 이상 탐지 능력을 더 향상시킨 모델입니다. PaDiM에 대한 설명은 지금은 생략하겠습니다.
PatchCore의 전체 프로세스는 아래 그림과 같습니다.

정상 샘플 이미지 데이터를(Normal samples) Wide-ResNet과 같은 CNN기반 모델에 대입하여 특징을 추출합니다.
"""wide_resnet101_2 encoder를 이용해 이미지 feature를 추출하는 스크립트.""" import argparse from pathlib import Path import cv2 import numpy as np import pandas as pd import torch import torch.nn as nn from tqdm import tqdm from torchvision.models import Wide_ResNet101_2_Weights, wide_resnet101_2 class WideResNetFeatureExtractor(nn.Module): """wide_resnet101_2를 stem/layer1~4 5구간으로 나눠 각 구간의 출력을 반환.""" def __init__(self, backbone: nn.Module): super().__init__() self.stem = nn.Sequential(backbone.conv1, backbone.bn1, backbone.relu, backbone.maxpool) self.layer1 = backbone.layer1 self.layer2 = backbone.layer2 self.layer3 = backbone.layer3 self.layer4 = backbone.layer4 def forward(self, x: torch.Tensor) -> list[torch.Tensor]: """5구간(stem, layer1, layer2, layer3, layer4)의 출력 feature map을 순서대로 반환.""" outputs = [] x = self.stem(x) outputs.append(x) x = self.layer1(x) outputs.append(x) x = self.layer2(x) outputs.append(x) x = self.layer3(x) outputs.append(x) x = self.layer4(x) outputs.append(x) return outputs def build_wide_resnet_encoder( pretrained_weights_path: str | None = None, imagenet_pretrained: bool = True, ) -> WideResNetFeatureExtractor: """ wide_resnet101_2 image encoder를 생성. Args: pretrained_weights_path: 로컬 checkpoint(.pt) 경로. 지정하면 torchvision의 ImageNet weight 대신 이 checkpoint를 로드한다. imagenet_pretrained: pretrained_weights_path가 None일 때 torchvision의 ImageNet 사전학습 weight를 사용할지 여부. False면 random 초기화 weight 사용 (구조 테스트용). """ weights = Wide_ResNet101_2_Weights.IMAGENET1K_V2 if imagenet_pretrained else None backbone = wide_resnet101_2(weights=None if pretrained_weights_path is not None else weights) if pretrained_weights_path is not None: with open(pretrained_weights_path, "rb") as f: ckpt = torch.load(f, map_location="cpu") if "model" in ckpt and isinstance(ckpt["model"], dict): ckpt = ckpt["model"] if "state_dict" in ckpt and isinstance(ckpt["state_dict"], dict): ckpt = ckpt["state_dict"] missing, unexpected = backbone.load_state_dict(ckpt, strict=False) print(f"로드 완료. missing keys: {len(missing)}개, unexpected keys: {len(unexpected)}개") encoder = WideResNetFeatureExtractor(backbone) encoder.eval() return encoder def preprocess_image_array(image_bgr: np.ndarray, img_size: int = 640) -> torch.Tensor: """BGR 이미지 배열을 wide_resnet101_2 입력 형식으로 전처리.""" img_rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB) img_rgb = cv2.resize(img_rgb, (img_size, img_size), interpolation=cv2.INTER_LINEAR) img_np = img_rgb.astype(np.float32) / 255.0 # 0~255 -> 0~1 정규화 # ImageNet 표준 mean/std로 정규화 (wide_resnet101_2가 ImageNet으로 사전학습됨) mean = np.array([0.485, 0.456, 0.406]) std = np.array([0.229, 0.224, 0.225]) img_np = (img_np - mean) / std # (H, W, C) -> (C, H, W) -> (1, C, H, W) img_tensor = torch.from_numpy(img_np).permute(2, 0, 1).unsqueeze(0).float() return img_tensor if __name__ == "__main__": img_size=520 encoder = build_wide_resnet_encoder( pretrained_weights_path=args.pretrained_weights_path, imagenet_pretrained=not args.no_imagenet_pretrained, ) total_params = sum(p.numel() for p in encoder.parameters()) print(f"wide_resnet101_2 encoder 전체 파라미터 수: {total_params:,}") output_dir = Path(output_dir) output_dir.mkdir(parents=True, exist_ok=True) print(f"[대상] {len(img_paths)}개 이미지") print(f"[설정] img_size={img_size}, layer={layer}") for img_path in tqdm(img_paths, desc="wide_resnet101_2 feature 추출", unit="img"): image_bgr = cv2.imread(image_path) input_tensor = preprocess_image_array(image_bgr, img_size=img_size) with torch.no_grad(): features = encoder(input_tensor) if not (1 <= layer <= len(features)): raise ValueError(f"layer={layer}는 유효 범위(1~{len(features)})를 벗어났습니다.") feat = features[layer - 1].numpy() save_path = output_dir / f"{img_path.stem}.npy" np.save(save_path, feat) print(f"[완료] {len(img_paths)}개 feature -> {output_dir}")
추출된 특징 패치를 이웃 패치와 결합(aggregation)하여 하나의 patch로 사용합니다.
"""PatchCore의 locally aware patch feature 및 multi-hierarchy aggregation 모듈.""" import argparse import json from pathlib import Path import numpy as np import torch import torch.nn.functional as F class LocalNeighborhoodAggregation(torch.nn.Module): """단일 feature map에 대한 locally aware patch feature 생성. Args: patchsize: 이웃 크기 p (논문 기본값 3) stride: 위치 샘플링 stride (논문 기본값 1) output_dim: 목표 feature 차원 d. None이면 입력 채널 수 C를 그대로 사용. """ def __init__(self, patchsize: int = 3, stride: int = 1, output_dim: int | None = None): super().__init__() self.patchsize = patchsize self.stride = stride self.output_dim = output_dim def forward(self, features: torch.Tensor) -> torch.Tensor: """ Args: features: [B, C, H, W] backbone feature map Returns: [B, d, H', W'] (stride=1이면 H'=H, W'=W) """ B, C, H, W = features.shape p = self.patchsize d = self.output_dim or C # 1) p x p 이웃 추출: [B, C*p*p, L], L = H' * W' padding = (p - 1) // 2 unfolded = F.unfold(features, kernel_size=p, stride=self.stride, padding=padding) H_out = (H + 2 * padding - p) // self.stride + 1 W_out = (W + 2 * padding - p) // self.stride + 1 L = unfolded.shape[-1] #(batch, 한 patch의 feature 길이, 총 patch 개수), [B, C*p*p, L] # 2) [B, C*p*p, L] -> permute:[B, L, C*p*p] -> reshape:[B*L, 1, C*p*p] # 메소드 체이닝 : 함수를 . 으로 연달아서 사용하는 방식 unfolded = unfolded.permute(0, 2, 1).reshape(B * L, 1, C * p * p) # 3) adaptive avg pooling으로 d차원으로 압축: [B*L, d], # input (N,C,L) : 샘플 개수, 채널, 길이 aggregated = F.adaptive_avg_pool1d(unfolded, d).squeeze(1) # 4) feature map 형태로 복원: [B, d, H', W'] return aggregated.reshape(B, H_out, W_out, d).permute(0, 3, 1, 2) class MultiHierarchyAggregation(torch.nn.Module): """두 hierarchy의 feature를 PatchCore 방식으로 결합. 각 레벨에 이웃 aggregation 적용 -> 깊은 레벨을 얕은 레벨 해상도로 bilinear upsample -> 채널 concat -> 다시 adaptive pooling으로 최종 차원 d로 통합. """ def __init__(self, patchsize: int = 3, per_level_dim: int = 1024, target_dim: int = 1024): super().__init__() self.agg = LocalNeighborhoodAggregation(patchsize=patchsize, stride=1, output_dim=per_level_dim) self.target_dim = target_dim def forward(self, feat_low: torch.Tensor, feat_high: torch.Tensor) -> torch.Tensor: """ Args: feat_low: 얕은 레벨 [B, C1, H1, W1] (고해상도) feat_high: 깊은 레벨 [B, C2, H2, W2] (저해상도) Returns: [B, target_dim, H1, W1] """ f1 = self.agg(feat_low) # [B, d1, H1, W1] f2 = self.agg(feat_high) # [B, d1, H2, W2] f2 = F.interpolate(f2, size=f1.shape[-2:], mode="bilinear", align_corners=False) combined = torch.cat([f1, f2], dim=1) # [B, 2*d1, H1, W1] # 최종 통합 pooling (공식 구현의 Aggregator에 해당) B, C, H, W = combined.shape combined = combined.permute(0, 2, 3, 1).reshape(B * H * W, 1, C) combined = F.adaptive_avg_pool1d(combined, self.target_dim).squeeze(1) return combined.reshape(B, H, W, self.target_dim).permute(0, 3, 1, 2) def _load_feature(path: Path) -> torch.Tensor: array = np.load(path) return torch.from_numpy(array).float() def run_aggregation(input_dir: str, output_dir: str, input_dir2: str | None = None, use_hierarchy: bool = False, patchsize: int = 3, output_dim: int | None = None, per_level_dim: int | None = None, target_dim: int | None = None) -> Path: """input_dir의 .npy feature들에 aggregation을 적용해 output_dir에 저장. - use_hierarchy=False (default): 각 파일에 LocalNeighborhoodAggregation만 적용 (예: SAM3 layer15). - use_hierarchy=True: input_dir2(다른 hierarchy 레벨, 동일 파일명 필요)가 반드시 있어야 하며, 위 결과에 더해 두 레벨을 결합하는 MultiHierarchyAggregation을 추가로 수행한다. 두 경우 모두 처리한 모든 샘플의 input/output shape 메타 정보를 output_dir 아래 단일 JSON 파일(aggregation_metadata.json)로 저장한다. """ if use_hierarchy and not input_dir2: raise ValueError("--hierarchy를 사용하려면 --input2가 필요합니다.") input_path = Path(input_dir) input_path2 = Path(input_dir2) if input_dir2 else None output_path = Path(output_dir) output_path.mkdir(parents=True, exist_ok=True) local_agg = LocalNeighborhoodAggregation(patchsize=patchsize, output_dim=output_dim) multi_agg = None if use_hierarchy: sample_files = sorted(input_path.glob("*.npy")) if not sample_files: raise FileNotFoundError(f"{input_path}에 .npy 파일이 없습니다.") primary_channels = _load_feature(sample_files[0]).shape[1] resolved_per_level_dim = per_level_dim or primary_channels resolved_target_dim = target_dim or resolved_per_level_dim multi_agg = MultiHierarchyAggregation(patchsize=patchsize, per_level_dim=resolved_per_level_dim, target_dim=resolved_target_dim) metadata = { "config": { "mode": "multi_hierarchy" if use_hierarchy else "local_neighborhood", "input_dir": str(input_path), "input_dir2": str(input_path2) if input_path2 else None, "output_dir": str(output_path), "patchsize": patchsize, "output_dim": output_dim, "per_level_dim": multi_agg.agg.output_dim if multi_agg else None, "target_dim": multi_agg.target_dim if multi_agg else None, }, "samples": {}, } files = sorted(input_path.glob("*.npy")) for f in files: name = f.stem feat1 = _load_feature(f) sample_meta = {"input_shape": list(feat1.shape)} if not use_hierarchy: with torch.no_grad(): local_out = local_agg(feat1) local_file = f"{name}_local.npy" np.save(output_path / local_file, local_out.numpy()) sample_meta["local_output_shape"] = list(local_out.shape) sample_meta["local_output_file"] = local_file else: assert input_path2 is not None f2 = input_path2 / f.name if not f2.exists(): sample_meta["multi_error"] = f"{f2}에 대응 파일이 없어 multi-hierarchy aggregation을 건너뜀" else: feat2 = _load_feature(f2) sample_meta["input2_shape"] = list(feat2.shape) with torch.no_grad(): multi_out = multi_agg(feat1, feat2) multi_file = f"{name}_multi.npy" np.save(output_path / multi_file, multi_out.numpy()) sample_meta["multi_output_shape"] = list(multi_out.shape) sample_meta["multi_output_file"] = multi_file metadata["samples"][name] = sample_meta meta_path = output_path / "aggregation_metadata.json" with open(meta_path, "w", encoding="utf-8") as fp: json.dump(metadata, fp, indent=2, ensure_ascii=False) return meta_path if __name__ == "__main__": meta_path = run_aggregation( input_dir=args.input1, input_dir2=args.input2, use_hierarchy=args.hierarchy, output_dir=args.output, patchsize=args.patchsize, output_dim=args.output_dim, per_level_dim=args.per_level_dim, target_dim=args.target_dim, ) print(f"메타 정보 저장 완료: {meta_path}")
메모리 뱅크 생성

1. 첫번째 iteration

거리값이 가장 큰 16은 2번째 patch와 4번째 patch가 많이 다른 형태의 데이터라는 것을 의미
2. 두번째 iteration


3. 최종 메모리 뱅크
Selected index에 해당하는 patch들만 메모리 뱅크에 저장함
selected_index = Coreset Subsample index = [2,4,1]
"""PatchCore의 greedy coreset selection 모듈.""" import argparse import shutil import tempfile from pathlib import Path import numpy as np import torch @torch.no_grad() def greedy_coreset_indices( work: torch.Tensor, target_size: int, seed: int = 0, verbose: bool = True, ) -> torch.Tensor: """Greedy minimax facility location coreset 선택. Args: work: [N, d*] 선택 기준으로 쓸 feature 행렬 (JL 사영된 것 또는 원본) target_size: 선택할 coreset 크기 l seed: 시작점 재현용 시드 Returns: [target_size] 선택된 인덱스 (LongTensor) """ N = work.shape[0] device = work.device gen = torch.Generator(device="cpu").manual_seed(seed) # --- greedy 선택 --- # min_dists[i] = i번째 점에서 현재까지 선택된 coreset까지의 최소 거리 제곱. # 매 스텝 argmax(min_dists)를 추가하고, 새 점과의 거리로 min_dists를 갱신. # ||a-b||^2 = ||a||^2 - 2*a·b + ||b||^2 전개로 계산한다: torch.cdist는 매 스텝 # [N, d*] 크기의 임시 버퍼를 새로 할당해 GPU에서 OOM을 유발하므로, 대신 # 행렬-벡터 곱(GEMV)만 써서 O(N) 메모리로 거리를 구한다. norms_sq = (work * work).sum(dim=1) # [N] def sq_dists_to(idx: int) -> torch.Tensor: # idx번째 점과 나머지 모든 점들이 얼마나 멀리 떨어져 있는지를 구하는 함수, 유클리안 거리, (a-b)^2 = a^2-2(a*b)+b^2 식에서 각 항을 미리 계산한 후에 수식에 대입 cross = work @ work[idx] # [N] return (norms_sq - 2.0 * cross + norms_sq[idx]).clamp_(min=0) start = int(torch.randint(N, (1,), generator=gen).item()) selected = torch.empty(target_size, dtype=torch.long, device=device) selected[0] = start # patch index 값 random으로 선택 min_dists = sq_dists_to(start) #start번째 patch와 자기자신을 포함한 다른 patch와의 유클리드 거리 계산 for i in range(1, target_size): idx = int(torch.argmax(min_dists).item()) #전체 patch와 선택된 한개의 patch 중에서 가장 거리가 먼 patch index selected[i] = idx #다음 거리 계산 patch index new_dists = sq_dists_to(idx) torch.minimum(min_dists, new_dists, out=min_dists) min_dists[idx] = -1.0 # 이미 선택된 점은 다시 뽑히지 않도록 if verbose and (i % max(1, target_size // 10) == 0): # coverage radius: 아직 커버되지 않은 점 중 가장 먼 거리 (Eq. 5의 max-min 항) print(f" [{i}/{target_size}] coverage radius^2 = {min_dists.max().item():.4f}") return selected # 선택된 이전 patch들과 거리가 먼 순서대로 patch index 나열 [3,77,5,2,8] 첫번째는 random이므로 무시, def scan_patch_files(input_dir: Path) -> tuple[list[Path], int, int]: """.npy 파일들의 헤더만 읽어 (전체 로드 없이) 총 patch 수 N과 차원 d를 파악.""" npy_files = sorted(p for p in input_dir.iterdir() if p.suffix == ".npy") if not npy_files: raise FileNotFoundError(f"{input_dir}에 .npy 확장자 파일이 없습니다.") total = 0 dim: int | None = None for f in npy_files: arr = np.load(f, mmap_mode="r") if arr.ndim == 4: B, C, H, W = arr.shape n = B * H * W #n : 1채널에서의 벡터 크기,batch 포함, 기본 batch가 1이므로 batch는 없는거나 다름없음 elif arr.ndim == 2: # 필요없는 코드 인듯 n, C = arr.shape else: raise ValueError(f"{f.name}: 지원하지 않는 shape {arr.shape} (2D 또는 4D[B,C,H,W]만 지원)") if dim is None: dim = C elif dim != C: raise ValueError(f"{f.name}: 차원 불일치 (기대 {dim}, 실제 {C})") total += n assert dim is not None return npy_files, total, dim #디렉토리에 있는 모든 npy파일 리스트, 디렉토리에 있는 모든 npy의 patch 개수(모든 npy파일에서의 row 개수), npy파일에서의 채널수 def stream_build_bank( npy_files: list[Path], total: int #npy_files의 총 patch 개수, dim: int #npy_files의 channel, bank_path: Path, proj_dim: int | None #해당 값으로 차원 축소, seed: int, ) -> tuple[np.memmap, torch.Tensor]: """파일을 하나씩 읽어 원본 차원 patch bank는 디스크 memmap에 채우고, JL 사영된 저차원 표현(work)은 RAM에 누적한다. 한 시점에 RAM에 올라오는 원본 데이터는 파일 하나 분량뿐이므로, 전체 patch 수가 많아도 메모리 사용량은 O(파일 하나 크기 + N x proj_dim)로 유지된다. """ bank = np.lib.format.open_memmap(bank_path, mode="w+", dtype=np.float32, shape=(total, dim)) use_proj = proj_dim is not None and proj_dim < dim psi: torch.Tensor work: torch.Tensor if use_proj and proj_dim is not None: gen = torch.Generator(device="cpu").manual_seed(seed) psi = torch.randn(dim, proj_dim, generator=gen) / np.sqrt(proj_dim) work = torch.empty((total, proj_dim), dtype=torch.float32) offset = 0 for f in npy_files: arr = np.load(f) if arr.ndim == 4: B, C, H, W = arr.shape patches = arr.transpose(0, 2, 3, 1).reshape(-1, C) else: patches = arr n = patches.shape[0] #B*H*W bank[offset:offset + n] = patches if use_proj: work[offset:offset + n] = torch.from_numpy(patches).float() @ psi print(f" {f.name}: {arr.shape} -> {n} patches ({offset + n}/{total})") offset += n bank.flush() if not use_proj: # 원본 차원을 그대로 선택 공간으로 사용 (memmap 기반, page cache로 관리되어 # concatenate 방식보다 안전하지만 여전히 RAM 부담이 클 수 있음) work = torch.from_numpy(bank) return bank, work #work : patch 개수는 그대로, 벡터 크기만 줄어듬, (16,32) -> (16,8) if __name__ == "__main__": input_dir = Path(args.input) npy_files, N, d = scan_patch_files(input_dir) target = max(1, int(N * args.percentage)) # 최종 patch 개수 proj = args.proj_dim if args.proj_dim and args.proj_dim > 0 else None print(f"입력: 총 {N} patches x {d} dims -> coreset {target}개 선택 " f"({args.percentage:.1%}), proj_dim={proj or '없음'}") if proj is None: print(" [경고] proj-dim 생략: 선택 단계에서 원본 차원 전체를 사용하므로 메모리 부담이 큽니다.") output_dir = Path(args.output) if args.output else input_dir.parent / "coreset" output_dir.mkdir(parents=True, exist_ok=True) tmp_dir = Path(tempfile.mkdtemp(dir=output_dir)) bank_path = tmp_dir / "patch_bank.npy" try: print(f"patch bank를 디스크에 스트리밍 저장 중: {bank_path}") bank, work = stream_build_bank(npy_files, N, d, bank_path, proj, args.seed) if args.device == "auto": device = "cuda" if torch.cuda.is_available() else "cpu" else: device = args.device print(f"선택 연산 장치: {device}") work = work.to(device) indices = greedy_coreset_indices(work, target, seed=args.seed) idx_np = indices.cpu().numpy() coreset = np.array(bank[idx_np]) # 원본 차원 feature를 memmap에서 필요한 행만 읽어 복사 del bank, work finally: shutil.rmtree(tmp_dir, ignore_errors=True) coreset_path = output_dir / "MB.npy" indices_path = output_dir / "MB_indices.npy" np.save(coreset_path, coreset) np.save(indices_path, idx_np) print(f"저장 완료: {coreset_path} shape={coreset.shape}") print(f"인덱스 저장 완료: {indices_path}")


<계산 로직>




"""PatchCore anomaly scoring (feature extraction 이후 단계만). 입력: test_feat : (P, C) 또는 (B, P, C) -- 추출 완료된 test patch feature P = Hf * Wf (예: 72*72 = 5184), C = feature 차원 (예: 1024) memory_bank : (M, C) -- 구축 완료된 coreset memory bank 출력: patch_scores : (B, Hf, Wf) patch별 anomaly score (= 1-NN 거리) image_scores : (B,) softmax re-weighting 적용된 이미지 score (Eq. 7) anomaly_map : (B, H, W) 원본 해상도로 upsample + Gaussian blur """ import argparse import json from pathlib import Path import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt import numpy as np import torch import torch.nn.functional as F from PIL import Image DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu") SCRIPT_DIR = Path(__file__).resolve().parent PROJECT_DIR = SCRIPT_DIR.parent DEFAULT_MEMORY_BANK = PROJECT_DIR / "Data" / "memory_bank" / "masked_img_view1" / "MB.npy" DEFAULT_TEST_IMAGE = ( PROJECT_DIR / "Data" / "features" / "test" / "bad" / "masked_img" / "masked_rubbish_bin_01_bad_01_0.png" ) DEFAULT_OUTPUT_DIR = PROJECT_DIR / "Data" / "features" / "test" / "bad" / "output" DEFAULT_HEATMAP_RANGE = Path("/mnt/data/github/Manufacutring_Inspection/data/memory_bank/heatmap_range.json") # memory bank와 grid가 항상 일치하는 최신 feature 저장소 (pages/2_Analysis.py가 채움). # test_image의 feature는 항상 여기서 찾는다 (다른 img_size로 추출된 stale feature 혼입 방지). APP_AGG_DIR = Path("/mnt/data/github/Manufacutring_Inspection/data/features/app/agg") class PatchCoreScorer: def __init__(self, memory_bank: torch.Tensor, k_nn: int = 3, blur_sigma: float = 4.0, chunk: int = 2048): """ Args: memory_bank: (M, C) coreset memory bank k_nn: re-weighting에 사용할 bank 내 neighbor 수 (논문의 b) blur_sigma: anomaly map Gaussian smoothing sigma chunk: cdist 메모리 절약용 chunk 크기 """ self.bank = memory_bank.to(DEVICE) # (M, C) self.k_nn = min(k_nn, self.bank.shape[0]) self.blur_sigma = blur_sigma self.chunk = chunk # ------------------------------------------------------------------ # Step 1. patch별 1-NN 거리 (= patch anomaly score) # ------------------------------------------------------------------ @torch.no_grad() def _nn_distance(self, flat: torch.Tensor): """ flat: (P_total, C) Returns: dmin (P_total,) 각 patch에서 bank까지의 최소 거리 imin (P_total,) 그 최근접 bank 점의 index (re-weighting에 필요) """ dists, idxs = [], [] for i in range(0, flat.shape[0], self.chunk): d = torch.cdist(flat[i:i + self.chunk], self.bank) # (chunk, M) dmin, imin = d.min(dim=1) dists.append(dmin) idxs.append(imin) #dists = [tensor([0.0000, 3.6056, 0.0000]), # tensor([1.2345, 2.3456, 0.5000])] return torch.cat(dists), torch.cat(idxs) # ------------------------------------------------------------------ # Step 2. 이미지 score: softmax re-weighting (논문 Eq. 7) # ------------------------------------------------------------------ @torch.no_grad() def _image_score(self, feat_b: torch.Tensor, patch_dist_b: torch.Tensor, nn_idx_b: torch.Tensor) -> torch.Tensor: """ feat_b : (P, C) 한 이미지의 patch feature patch_dist_b : (P,) patch별 1-NN 거리 nn_idx_b : (P,) patch별 최근접 bank index s = (1 - exp(s*) / sum_{m in N_b(m*)} exp(d(x*, m))) * s* 직관: 가장 anomalous한 patch x*의 최근접 bank 점 m*가 bank 내에서 '외딴 점'이면(이웃이 멀면) 확신을 낮추고, '밀집 영역의 점'이면 s*를 거의 그대로 사용. => 하나의 이미지에 대한 anomaly score 스칼라 값 """ # (1) 가장 anomalous한 test patch x* 선택 p_star = torch.argmax(patch_dist_b) s_star = patch_dist_b[p_star] # 최대 patch 거리 x_star = feat_b[p_star:p_star + 1] # (1, C) m_star = nn_idx_b[p_star] # x*의 최근접 bank 점 # (2) m*의 bank 내 b-nearest neighbors N_b(m*) d_bank = torch.cdist(self.bank[m_star:m_star + 1], self.bank).squeeze(0) nb_idx = torch.topk(d_bank, k=self.k_nn, largest=False).indices # (3) x*에서 그 neighbor들까지의 거리 d_nb = torch.cdist(x_star, self.bank[nb_idx]).squeeze(0) #(k_nn,) # 수치 안정화: 최대값 빼고 exp (오버플로 방지, 비율은 동일) m = torch.maximum(s_star, d_nb.max()) w = 1 - torch.exp(s_star - m) / torch.exp(d_nb - m).sum() return w * s_star # ------------------------------------------------------------------ # Step 3. anomaly map: 72x72 -> 원본 해상도 + Gaussian blur # ------------------------------------------------------------------ @torch.no_grad() def _make_anomaly_map(self, patch_scores: torch.Tensor, out_size: tuple[int, int]) -> torch.Tensor: amap = patch_scores.unsqueeze(1) # (B,1,Hf,Wf) amap = F.interpolate(amap, size=out_size, mode="bilinear", align_corners=False).squeeze(1) # (B,H,W) sigma = self.blur_sigma ks = int(4 * sigma + 1) | 1 coords = torch.arange(ks, device=amap.device) - ks // 2 g = torch.exp(-(coords ** 2) / (2 * sigma ** 2)) g = (g / g.sum()).view(1, 1, -1) x = amap.unsqueeze(1) x = F.conv2d(x, g.unsqueeze(2), padding=(0, ks // 2)) x = F.conv2d(x, g.unsqueeze(3), padding=(ks // 2, 0)) return x.squeeze(1) # ------------------------------------------------------------------ # 전체 scoring 파이프라인 # ------------------------------------------------------------------ @torch.no_grad() def score(self, test_feat: torch.Tensor, grid: tuple[int, int], out_size: tuple[int, int] | None = None, fg_mask: torch.Tensor | None = None) -> dict: """ Args: test_feat: (P, C) 단일 이미지 또는 (B, P, C) 배치 grid: (Hf, Wf), 예: (72, 72) out_size: anomaly map 출력 해상도 (None이면 map 생략) fg_mask: (B, Hf, Wf) bool, True=foreground. 지정 시 background patch score를 0으로 마스킹 (image score의 argmax에서도 제외됨) Returns: dict(patch_scores, image_scores[, anomaly_map]) """ if test_feat.dim() == 2: test_feat = test_feat.unsqueeze(0) # (1, P, C) test_feat = test_feat.to(DEVICE) B, P, C = test_feat.shape Hf, Wf = grid assert P == Hf * Wf, f"P={P} != Hf*Wf={Hf * Wf}" # ---- Step 1: patch scores ---- flat = test_feat.reshape(-1, C) # (1(B)*P, C) dmin, imin = self._nn_distance(flat) patch_dist = dmin.reshape(B, P) # (1(B), P) nn_idx = imin.reshape(B, P) # ---- foreground masking (선택) ---- if fg_mask is not None: mask_flat = fg_mask.reshape(B, P).to(DEVICE) patch_dist = patch_dist * mask_flat # bg patch -> 0 # ---- Step 2: image scores (re-weighting) ---- image_scores = torch.empty(B, device=DEVICE) for b in range(B): image_scores[b] = self._image_score(test_feat[b], patch_dist[b], nn_idx[b]) out = { "patch_scores": patch_dist.reshape(B, Hf, Wf).cpu(), "image_scores": image_scores.cpu(), } # ---- Step 3: anomaly map (선택) ---- if out_size is not None: out["anomaly_map"] = self._make_anomaly_map( patch_dist.reshape(B, Hf, Wf), out_size).cpu() return out # ---------------------------------------------------------------------------- # 입출력 helper # ---------------------------------------------------------------------------- def load_memory_bank(path: Path) -> torch.Tensor: """(M, C) memory bank npy -> tensor""" bank = np.load(path) return torch.from_numpy(bank).float() def load_test_feature(path: Path) -> tuple[torch.Tensor, tuple[int, int]]: """(1, C, Hf, Wf) test feature npy -> (1, Hf*Wf, C) tensor, (Hf, Wf)""" feat = np.load(path) if feat.ndim != 4 or feat.shape[0] != 1: raise ValueError(f"예상하지 못한 shape: {feat.shape} (기대: (1, C, Hf, Wf))") _, c, hf, wf = feat.shape flat = feat[0].transpose(1, 2, 0).reshape(1, hf * wf, c) # (1, P, C) return torch.from_numpy(np.ascontiguousarray(flat)).float(), (hf, wf) def foreground_mask_from_image(image: Image.Image, threshold: int = 10) -> np.ndarray: """masked_img(배경=검정)에서 foreground mask 추출. Returns: (H, W) bool array, True=foreground (밝기 > threshold) """ gray = np.asarray(image.convert("L"), dtype=np.uint8) return gray > threshold #든 픽셀마다 하나하나 조건 비교(Element-wise comparison)를 수행하여 True, False로 모든 배열 표시 def calibrate_score_range( scorer: "PatchCoreScorer", normal_feature_paths: list[Path], sample_n: int = 30, vmin_percentile: float = 0.5, vmax_percentile: float = 99.5, vmax_margin: float = 1.5, seed: int = 0, ) -> tuple[float, float]: """정상 이미지들의 patch score 분포로 히트맵에 쓸 고정 (vmin, vmax)를 계산. """ if not normal_feature_paths: raise ValueError("calibrate_score_range: normal_feature_paths가 비어 있습니다.") rng = np.random.default_rng(seed) paths = list(normal_feature_paths) if len(paths) > sample_n: idx = rng.choice(len(paths), size=sample_n, replace=False) paths = [paths[i] for i in idx] all_scores = [] for p in paths: test_feat, _ = load_test_feature(p) flat = test_feat.reshape(-1, test_feat.shape[-1]).to(DEVICE) dmin, _ = scorer._nn_distance(flat) all_scores.append(dmin.cpu().numpy()) scores = np.concatenate(all_scores) vmin = float(np.percentile(scores, vmin_percentile)) vmax = float(np.percentile(scores, vmax_percentile)) * vmax_margin print(f"[캘리브레이션] 정상 이미지 {len(paths)}장, patch {scores.size}개 " f"-> vmin={vmin:.4f} (p{vmin_percentile}), " f"vmax={vmax:.4f} (p{vmax_percentile} x {vmax_margin})") return vmin, vmax def calibrate_image_score_threshold( scorer: "PatchCoreScorer", normal_pairs: list[tuple[Path, Path]], sample_n: int = 30, percentile: float = 99.5, margin: float = 1.0, seed: int = 0, ) -> float: """정상 이미지들의 image_score 분포로 Normal/Abnormal 판정 threshold를 계산. """ if not normal_pairs: raise ValueError("calibrate_image_score_threshold: normal_pairs가 비어 있습니다.") rng = np.random.default_rng(seed) pairs = list(normal_pairs) if len(pairs) > sample_n: idx = rng.choice(len(pairs), size=sample_n, replace=False) pairs = [pairs[i] for i in idx] image_scores = [] for feature_path, image_path in pairs: test_feat, grid = load_test_feature(feature_path) image = Image.open(image_path) fg_mask_full = foreground_mask_from_image(image) fg_mask_grid = F.adaptive_max_pool2d( torch.from_numpy(fg_mask_full).float()[None, None], output_size=grid ).squeeze(1).bool() result = scorer.score(test_feat, grid=grid, fg_mask=fg_mask_grid) image_scores.append(float(result["image_scores"][0])) scores = np.array(image_scores) threshold = float(np.percentile(scores, percentile)) * margin print(f"[캘리브레이션] 정상 이미지 {len(pairs)}장 image_score 분포 " f"-> threshold={threshold:.4f} (p{percentile} x {margin})") return threshold def save_result(image_path: Path, anomaly_map: np.ndarray, image_score: float, out_dir: Path, stem: str, fg_mask: np.ndarray | None = None, vmin: float | None = None, vmax: float | None = None) -> None: """anomaly map 히트맵(원본 사진 없이 raw score 그대로)과 raw anomaly map npy를 out_dir에 저장. """ out_dir.mkdir(parents=True, exist_ok=True) image = Image.open(image_path).convert("RGB") npy_path = out_dir / f"{stem}_anomaly_map.npy" np.save(npy_path, anomaly_map) display_map = np.ma.masked_array(anomaly_map, mask=~fg_mask) if fg_mask is not None else anomaly_map fig, ax = plt.subplots(figsize=(image.width / 100, image.height / 100), dpi=100) ax.imshow(display_map, cmap="jet", vmin=vmin, vmax=vmax) ax.set_title(f"image_score={image_score:.4f}") ax.axis("off") fig.tight_layout() png_path = out_dir / f"{stem}_anomaly_map.png" fig.savefig(png_path, dpi=100) plt.close(fig) print(f"[저장] {npy_path}") print(f"[저장] {png_path}") def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="memory bank + test feature -> PatchCore anomaly score/map 계산 및 저장", formatter_class=argparse.RawTextHelpFormatter, ) parser.add_argument("--memory_bank", type=str, default=str(DEFAULT_MEMORY_BANK), help="memory bank (M, C) npy 경로") parser.add_argument("--test_image", type=str, default=str(DEFAULT_TEST_IMAGE), help="test sample 원본 이미지 경로. feature는 " f"{APP_AGG_DIR}/{{test_image의 stem}}_local.npy 에서 자동으로 찾는다.") parser.add_argument("--output_dir", type=str, default=str(DEFAULT_OUTPUT_DIR), help="결과 저장 상위 디렉토리 (test feature 파일명으로 하위 폴더 생성)") parser.add_argument("--k_nn", type=int, default=3, help="image score re-weighting neighbor 수") parser.add_argument("--blur_sigma", type=float, default=4.0, help="anomaly map Gaussian blur sigma") return parser.parse_args() def main() -> None: args = parse_args() memory_bank_path = Path(args.memory_bank) test_image_path = Path(args.test_image) output_dir = Path(args.output_dir) # ---- test feature 경로: data/features/app/agg에 미리 추출된 feature만 사용 ---- # (다른 img_size로 추출된 stale feature를 잘못 넘겨서 memory bank와 grid가 # 어긋나는 문제를 원천 차단하기 위해, 경로를 test_image로부터 항상 여기서 계산한다) test_feature_path = APP_AGG_DIR / f"{test_image_path.stem}_local.npy" if not test_feature_path.exists(): raise FileNotFoundError( f"app/agg에 feature가 없습니다: {test_feature_path}\n" f"먼저 feature extraction 파이프라인으로 {APP_AGG_DIR}에 해당 이미지의 " f"feature를 생성하세요." ) print(f"[feature] {test_feature_path}") # ---- 1. bank + test feature 로드 ---- bank = load_memory_bank(memory_bank_path) print(f"memory bank : {tuple(bank.shape)}") test_feat, grid = load_test_feature(test_feature_path) print(f"test feature: {tuple(test_feat.shape)} grid={grid}") image = Image.open(test_image_path) out_size = (image.height, image.width) # ---- fg_mask: masked_img(배경=검정)에서 foreground 영역 추출 ---- fg_mask_full = foreground_mask_from_image(image) # (H, W) bool fg_mask_grid = F.adaptive_max_pool2d( torch.from_numpy(fg_mask_full).float()[None, None], output_size=grid ).squeeze(1).bool() # (1, Hf, Wf) # fg_mask_grid # tensor([[[False, False, False], # [False, True, False], # [False, True, False]]]) # ---- 2. scoring ---- scorer = PatchCoreScorer(bank, k_nn=args.k_nn, blur_sigma=args.blur_sigma) result = scorer.score(test_feat, grid=grid, out_size=out_size, fg_mask=fg_mask_grid) image_score = float(result["image_scores"][0]) anomaly_map = result["anomaly_map"][0].numpy() print(f"image_score : {image_score:.4f}") print(f"anomaly_map : {anomaly_map.shape}") # ---- 2b. 히트맵 색 스케일(vmin/vmax): find_heapmap_range.py가 미리 계산해둔 json에서 읽음 ---- with open(DEFAULT_HEATMAP_RANGE) as f: heatmap_range = json.load(f) vmin, vmax = heatmap_range["vmin"], heatmap_range["vmax"] print(f"heatmap range: vmin={vmin:.4f}, vmax={vmax:.4f} (출처: {DEFAULT_HEATMAP_RANGE})") # ---- 3. 저장: output_dir/{test feature 파일명}/ 에 img + feature map npy ---- stem = test_feature_path.stem save_result(test_image_path, anomaly_map, image_score, output_dir / stem, stem, fg_mask=fg_mask_full, vmin=vmin, vmax=vmax) if __name__ == "__main__": main()
"""Anomaly 판단을 위한 threshold 계산 스크립트 memory bank + 정상 이미지 feature들로 1) anomaly heatmap 색 스케일(vmin/vmax) 2) Normal/Abnormal 판정 threshold (image_score 기준) 를 모두 percentile 방식으로 계산해 json으로 저장하는 스크립트. anomaly_score.py의 calibrate_score_range()(patch score용)와 calibrate_image_score_threshold()(image_score용)를 그대로 재사용한다. 여기서 계산한 값들을 pages/2_Analysis.py가 그대로 읽어서 쓰므로, 하드코딩된 threshold 없이 memory bank가 바뀔 때마다 다시 계산해서 갱신하면 된다. """ import argparse import glob import json from pathlib import Path import pandas as pd from anomaly_score import ( PatchCoreScorer, calibrate_image_score_threshold, calibrate_score_range, load_memory_bank, ) def _resolve_normal_pairs(calibrate_glob: str, csv_path: str, label: str) -> list[tuple[Path, Path]]: """calibrate_glob의 feature npy들과 csv의 원본 이미지 경로를 파일명으로 매칭.""" df = pd.read_csv(csv_path) normal_rows = df[df["labels"] == label] stem_to_image = {Path(p).stem: Path(p) for p in normal_rows["data_path"]} pairs: list[tuple[Path, Path]] = [] for feature_path in sorted(Path(p) for p in glob.glob(calibrate_glob)): image_stem = feature_path.stem.removesuffix("_local") image_path = stem_to_image.get(image_stem) if image_path is not None and image_path.exists(): pairs.append((feature_path, image_path)) return pairs def main() -> None: args = parse_args() bank = load_memory_bank(Path(args.memory_bank)) print(f"memory bank : {tuple(bank.shape)}") normal_paths = sorted(Path(p) for p in glob.glob(args.calibrate_glob)) if not normal_paths: raise FileNotFoundError(f"--calibrate_glob에 해당하는 파일이 없습니다: {args.calibrate_glob}") scorer = PatchCoreScorer(bank, k_nn=args.k_nn) vmin, vmax = calibrate_score_range( scorer, normal_paths, sample_n=args.calibrate_sample_n, vmin_percentile=args.vmin_percentile, vmax_percentile=args.vmax_percentile, vmax_margin=args.vmax_margin, seed=args.seed, ) result = {"vmin": vmin, "vmax": vmax} if args.calibrate_csv: normal_pairs = _resolve_normal_pairs(args.calibrate_glob, args.calibrate_csv, args.calibrate_label) if not normal_pairs: raise FileNotFoundError( f"--calibrate_csv({args.calibrate_csv})에서 --calibrate_glob과 매칭되는 " f"'{args.calibrate_label}' 이미지를 찾지 못했습니다." ) anomaly_threshold = calibrate_image_score_threshold( scorer, normal_pairs, sample_n=args.calibrate_sample_n, percentile=args.threshold_percentile, margin=args.threshold_margin, seed=args.seed, ) result["anomaly_threshold"] = anomaly_threshold else: print("[건너뜀] --calibrate_csv 미지정 -> anomaly_threshold 계산 생략") output_path = Path(args.output) output_path.parent.mkdir(parents=True, exist_ok=True) with open(output_path, "w") as f: json.dump(result, f, indent=2) print(f"[저장] {output_path} {result}") if __name__ == "__main__": main()

[코드] : https://github.com/amazon-science/patchcore-inspection
[0] Towards total recall in industrial anomaly detection (https://arxiv.org/pdf/2106.08265)
[1] Roth, Karsten, et al. "Towards total recall in industrial anomaly detection." 2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 2022.
[2] Chen, Qiyu, et al. "A unified anomaly synthesis strategy with gradient ascent for industrial anomaly detection and localization." European Conference on Computer Vision. Cham: Springer Nature Switzerland, 2024.