INNER CODE UNIT · Python
load_clip_pretrain
FoundationVision/UniTok · main.py:60
def load_clip_pretrain(model, ckpt_path):
ckpt = torch.load(ckpt_path, map_location='cpu')
converted_state_dict = dict()
for k, v in ckpt.items():
if k.startswith('visual.'):
if 'head' in k or 'pos_embed' in k:
continue
new_k = k.replace('visual.trunk.', 'encoder.')
converted_state_dict[new_k] = v
elif k.startswith('text.'):
new_k = k.replace('text.', 'text_encoder.')
converted_state_dict[new_k] = v
# elif k == 'logit_scale':
# converted_state_dict[k] = v
model.load_state_dict(converted_state_dict, strict=False)
def train_one_ep(