| { | |
| "model_cfg": { | |
| "embed_dim": 768, | |
| "vision_cfg": { | |
| "image_size": 224, | |
| "layers": 24, | |
| "width": 1024, | |
| "patch_size": 14, | |
| "output_tokens": true, | |
| "pool_type": "avg_all", | |
| "final_ln_after_pool": true | |
| }, | |
| "text_cfg": { | |
| "context_length": 77, | |
| "vocab_size": 49408, | |
| "width": 768, | |
| "heads": 12, | |
| "layers": 12, | |
| "output_tokens": true, | |
| "cross_attn_ratio": 2, | |
| "does_full_decoding": true | |
| }, | |
| "custom_text": true | |
| }, | |
| "preprocess_cfg": { | |
| "mean": [ | |
| 0.48145466, | |
| 0.4578275, | |
| 0.40821073 | |
| ], | |
| "std": [ | |
| 0.26862954, | |
| 0.26130258, | |
| 0.27577711 | |
| ], | |
| "interpolation": "bicubic", | |
| "resize_mode": "shortest" | |
| } | |
| } |