mixvideo-v2/cargos/tvai-v2/视觉-语言模型配置/RN50x16-quickgelu.json

22 lines
389 B
JSON

{
"embed_dim": 768,
"quick_gelu": true,
"vision_cfg": {
"image_size": 384,
"layers": [
6,
8,
18,
8
],
"width": 96,
"patch_size": null
},
"text_cfg": {
"context_length": 77,
"vocab_size": 49408,
"width": 768,
"heads": 12,
"layers": 12
}
}