| { | |
| "model_cfg": { | |
| "embed_dim": 384, | |
| "vision_cfg": { | |
| "image_size": 160, | |
| "layers": 12, | |
| "width": 384, | |
| "patch_size": 16, | |
| "no_ln_pre": true, | |
| "pool_type": "avg", | |
| "final_ln_after_pool": true, | |
| "norm_kwargs": { | |
| "eps": 1e-06 | |
| }, | |
| "head_width": 64 | |
| }, | |
| "text_cfg": { | |
| "context_length": 80, | |
| "vocab_size": 32000, | |
| "hf_tokenizer_name": "bert-base-uncased", | |
| "tokenizer_kwargs": { | |
| "strip_sep_token": true | |
| }, | |
| "width": 384, | |
| "heads": 6, | |
| "layers": 12, | |
| "pool_type": "last", | |
| "no_causal_mask": true, | |
| "act_kwargs": { | |
| "approximate": "tanh" | |
| }, | |
| "norm_kwargs": { | |
| "eps": 1e-06 | |
| } | |
| } | |
| }, | |
| "preprocess_cfg": { | |
| "mean": [ | |
| 0.485, | |
| 0.456, | |
| 0.406 | |
| ], | |
| "std": [ | |
| 0.229, | |
| 0.224, | |
| 0.225 | |
| ], | |
| "interpolation": "bilinear", | |
| "resize_mode": "squash" | |
| } | |
| } |