Files
bachelor-thesis/notebooks/caml.ipynb
T
lukas 882c6f54bb
Build Typst document / build_typst_documents (push) Successful in 22s
ploting notebooks
2025-01-01 20:53:05 +01:00

10 KiB

In [1]:
import sys
import torch
from pyprojroot import here as project_root
import numpy as np

sys.path.insert(0, str(project_root()))

from src.evaluation.utils import get_test_path, get_model
from src.evaluation.eval import meta_test

from src.train_utils.trainer import train_parser
from src.models.feature_extractors.pretrained_fe import get_fe_metadata
import torchvision.transforms as transforms
from PIL import Image

device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
/home/q315433/micromamba/envs/pmf/lib/python3.12/site-packages/torch/__init__.py:749: UserWarning: torch.set_default_tensor_type() is deprecated as of PyTorch 2.1, please use torch.set_default_dtype() and torch.set_default_device() as alternatives. (Triggered internally at ../torch/csrc/tensor/python_tensor.cpp:431.)
  _C._set_default_tensor_type(t)
In [2]:
def test_transform():
    def _convert_image_to_rgb(im):
        return im.convert('RGB')

    return transforms.Compose([
        #transforms.Resize(224),
        transforms.Resize(224),
        #transforms.CenterCrop(224),
        _convert_image_to_rgb,
        transforms.ToTensor(),
        transforms.Normalize(mean=torch.tensor([0.4815, 0.4578, 0.4082]), std=torch.tensor([0.2686, 0.2613, 0.2758])),
        ])

preprocess = test_transform()
In [3]:
import enum

class T:
    fe_type = "timm:vit_base_patch16_clip_224.openai:768"
    #fe_type = "timm:vit_huge_patch14_clip_224.laion2b:1280"
    fe_dim = 768
    fe_dtype = "float32"
    model = "CAML"
    dropout = 0.0
    encoder_size = "large"

fe_metadata = get_fe_metadata(T())
#test_path = get_test_path(args, data_path)
#device = torch.device(f'cuda:{args.gpu}')

# Get the model and load its weights.
model, model_path = get_model(T(), fe_metadata, device)
print(model_path)
#print(model)
if model_path:
    model.load_state_dict(torch.load(model_path, map_location=f'cuda:0'), strict=False)
model.to(device)
_= model.eval()
Defaulting to float32 dtype
Loaded pretrained timm model vit_base_patch16_clip_224.openai
../caml_pretrained_models/CAML_CLIP/model.pth
In [9]:
import os

img_path = "../pmf_cvpr22/data_custom"

def filecnt_in_dir(dirr, typ):
    _, _, files = next(os.walk(f"{img_path}/{dirr}/test/{typ}/"))
    return len(files)

def evaluate(shot, way, folder):
    ts = ["good", "broken_small", "broken_large", "contamination"]
    tss = ["good", "cable_swap", "combined", "cut_inner_insulation", "cut_outer_insulation", "missing_cable", "missing_wire", "poke_insulation"]
    tss = ts
    cat = ["bottle", "cable"]

    #goodnr = (len(tss)-1) * shot
    
    with torch.no_grad():
        #img_supp = [preprocess(Image.open(f"{img_path}/{folder}/train/good/{i:03d}.png")).unsqueeze(0).to(device) for i in range(shot)]
        img_supp = [preprocess(Image.open(f"{img_path}/{folder}/test/{n}/{i:03d}.png")).unsqueeze(0).to(device) for n in tss[1:4] for i in range(shot)]
        
        tmp = [(preprocess(Image.open(f"{img_path}/{folder}/test/{n}/{i:03d}.png")).unsqueeze(0).to(device), tss.index(n)-1) for n in tss[1:4] for i in range(shot, filecnt_in_dir(folder, n))]
        img_query, query_labels = zip(*tmp)
        #print(tmp)
        print(query_labels)
    
        img_concat = img_supp + list(img_query)
        img_concat = torch.cat(img_concat, 0)
        print(img_concat.shape)
        print(len(img_supp))
        print(len(img_query))
        #shot = (len(tss)-1) * shot
    
        #logits = model.meta_test(img_concat, way=4, shot=shot, query_shot=1)
        #print(logits)
        #
        feature_vector = model.get_feature_vector(img_concat)
        support_features = feature_vector[:way * shot]
        query_features = feature_vector[way * shot:]
        b, d = query_features.shape
    
        # Reshape query and support to a sequence.
        support = support_features.reshape(1, way * shot, d).repeat(b, 1, 1)
        query = query_features.reshape(-1, 1, d)
        feature_sequences = torch.cat([query, support], dim=1)
        print(feature_sequences.shape)
    
        #labels = torch.LongTensor([i // shot for i in range(shot * way)]).to(device)
        labels = torch.arange(way).repeat(shot, 1).T.flatten().to(model.device)
        print(labels)
        #labels = torch.from_numpy(np.ones(shape=shot, dtype=int)).to(device)
        #labels = torch.cat([torch.from_numpy(np.zeros(shape=shot, dtype=int)).to(device), labels])
        print(labels)
        
        #labels = torch.LongTensor([0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ]).to(device)
        print(labels.shape)
        print(feature_sequences.shape)
        logits = model.transformer_encoder.forward_imagenet_v2(feature_sequences, labels, way, shot)
        #print(logits)
        _, max_index = torch.max(logits[:, :way], 1)
        #print(max_index.cpu().numpy())
        #bbb = np.ones(shape=(14*4))
        #bbb[:14] = 0
        #print(np.mean(max_index.cpu().numpy() == bbb))
        print("herre")
        print(max_index.shape)
        print(np.array(query_labels).shape)

        return np.mean(max_index.cpu().numpy() == np.array(query_labels))

scores = [evaluate(shot, 3, "bottle") for shot in [1,3,5]]
print(np.array(scores))
(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2)
torch.Size([65, 3, 224, 224])
3
62
torch.Size([62, 4, 768])
tensor([0, 1, 2], device='cuda:0')
tensor([0, 1, 2], device='cuda:0')
torch.Size([3])
torch.Size([62, 4, 768])
herre
torch.Size([62])
(62,)
(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2)
torch.Size([65, 3, 224, 224])
9
56
torch.Size([56, 10, 768])
tensor([0, 0, 0, 1, 1, 1, 2, 2, 2], device='cuda:0')
tensor([0, 0, 0, 1, 1, 1, 2, 2, 2], device='cuda:0')
torch.Size([9])
torch.Size([56, 10, 768])
herre
torch.Size([56])
(56,)
(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2)
torch.Size([65, 3, 224, 224])
15
50
torch.Size([50, 16, 768])
tensor([0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2], device='cuda:0')
tensor([0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2], device='cuda:0')
torch.Size([15])
torch.Size([50, 16, 768])
herre
torch.Size([50])
(50,)
[0.58064516 0.51785714 0.52      ]

CAML: Resulsts:

bottle: jeweils 1,3,5 shots normal [0.40740741 0.39726027 0.30769231]

inbalanced - mehr good shots 5,10,15,30 -> alle anderen nur 5

  • not possible 1q 2 ways nur detektieren ob fehlerhaft oder nicht 3,6,9 shots -> wegen model restrictions [0.79012346 0.84415584 0.87671233]

inbalance 2 way 5,10,15,30 -> rest 5

  • not possible

nur fehlerklasse erkennen 1,3,5 [0.58064516 0.51785714 0.52 ]

cable: jeweils 1,3,5 shots normal [0.24031008 0.19834711 0.15929204]

inbalanced - mehr good shots 5,10,15,30 -> alle anderen nur 5

  • not possible

2 ways nur detektieren ob fehlerhaft oder nicht 1,3,5 shots [0.57364341 0.54545455 0.59292035]

inbalance 2 way 5,10,15,30 -> rest 5

  • not possible

nur fehlerklasse erkennen 1,3,5 [0.12962963 0.36363636 0.58823529]