{ "metadata": { "total_size": "11399028736" }, "weight_map": { "decoder.block.0.cross_attention_layer.cross_attention.Wk.weight": "model-00001-of-00003.safetensors", "decoder.block.0.cross_attention_layer.cross_attention.Wq.weight": "model-00001-of-00003.safetensors", "decoder.block.0.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.0.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.0.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.0.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.0.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.0.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.0.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.0.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "decoder.block.0.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "decoder.block.0.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "decoder.block.0.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "decoder.block.0.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "decoder.block.0.self_attention_layer.self_attention.pe_encoding.relative_attention_bias.weight": "model-00001-of-00003.safetensors", "decoder.block.1.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.1.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.1.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.1.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.1.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.1.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.1.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.1.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.1.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.1.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.1.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.1.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.1.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.1.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.10.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.10.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.10.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.10.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.10.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.10.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.10.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.10.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.10.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.10.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.10.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.10.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.10.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.10.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.11.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.11.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.11.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.11.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.11.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.11.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.11.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.11.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.11.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.11.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.11.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.11.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.11.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.11.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.12.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.12.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.12.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.12.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.12.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.12.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.12.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.12.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.12.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.12.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.12.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.12.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.12.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.12.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.13.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.13.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.13.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.13.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.13.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.13.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.13.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.13.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.13.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.13.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.13.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.13.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.13.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.13.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.14.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.14.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.14.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.14.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.14.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.14.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.14.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.14.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.14.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.14.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.14.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.14.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.14.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.14.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.15.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.15.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.15.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.15.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.15.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.15.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.15.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.15.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.15.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.15.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.15.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.15.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.15.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.15.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.16.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.16.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.16.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.16.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.16.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.16.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.16.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.16.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.16.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.16.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.16.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.16.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.16.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.16.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.17.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.17.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.17.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.17.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.17.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.17.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.17.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.17.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.17.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.17.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.17.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.17.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.17.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.17.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.18.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.18.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.18.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.18.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.18.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.18.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.18.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.18.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.18.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.18.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.18.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.18.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.18.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.18.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.19.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.19.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.19.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.19.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.19.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.19.ff_layer.act.wi_0.weight": "model-00003-of-00003.safetensors", "decoder.block.19.ff_layer.act.wi_1.weight": "model-00003-of-00003.safetensors", "decoder.block.19.ff_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.19.ff_layer.wo.weight": "model-00003-of-00003.safetensors", "decoder.block.19.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.19.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.19.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.19.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.19.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.2.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.2.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.2.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.2.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.2.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.2.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.2.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.2.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.2.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.2.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.2.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.2.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.2.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.2.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.20.cross_attention_layer.cross_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.20.cross_attention_layer.cross_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.20.cross_attention_layer.cross_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.20.cross_attention_layer.cross_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.20.cross_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.20.ff_layer.act.wi_0.weight": "model-00003-of-00003.safetensors", "decoder.block.20.ff_layer.act.wi_1.weight": "model-00003-of-00003.safetensors", "decoder.block.20.ff_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.20.ff_layer.wo.weight": "model-00003-of-00003.safetensors", "decoder.block.20.self_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.20.self_attention_layer.self_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.20.self_attention_layer.self_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.20.self_attention_layer.self_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.20.self_attention_layer.self_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.21.cross_attention_layer.cross_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.21.cross_attention_layer.cross_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.21.cross_attention_layer.cross_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.21.cross_attention_layer.cross_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.21.cross_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.21.ff_layer.act.wi_0.weight": "model-00003-of-00003.safetensors", "decoder.block.21.ff_layer.act.wi_1.weight": "model-00003-of-00003.safetensors", "decoder.block.21.ff_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.21.ff_layer.wo.weight": "model-00003-of-00003.safetensors", "decoder.block.21.self_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.21.self_attention_layer.self_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.21.self_attention_layer.self_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.21.self_attention_layer.self_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.21.self_attention_layer.self_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.22.cross_attention_layer.cross_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.22.cross_attention_layer.cross_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.22.cross_attention_layer.cross_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.22.cross_attention_layer.cross_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.22.cross_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.22.ff_layer.act.wi_0.weight": "model-00003-of-00003.safetensors", "decoder.block.22.ff_layer.act.wi_1.weight": "model-00003-of-00003.safetensors", "decoder.block.22.ff_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.22.ff_layer.wo.weight": "model-00003-of-00003.safetensors", "decoder.block.22.self_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.22.self_attention_layer.self_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.22.self_attention_layer.self_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.22.self_attention_layer.self_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.22.self_attention_layer.self_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.23.cross_attention_layer.cross_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.23.cross_attention_layer.cross_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.23.cross_attention_layer.cross_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.23.cross_attention_layer.cross_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.23.cross_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.23.ff_layer.act.wi_0.weight": "model-00003-of-00003.safetensors", "decoder.block.23.ff_layer.act.wi_1.weight": "model-00003-of-00003.safetensors", "decoder.block.23.ff_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.23.ff_layer.wo.weight": "model-00003-of-00003.safetensors", "decoder.block.23.self_attention_layer.layer_norm.weight": "model-00003-of-00003.safetensors", "decoder.block.23.self_attention_layer.self_attention.Wk.weight": "model-00003-of-00003.safetensors", "decoder.block.23.self_attention_layer.self_attention.Wq.weight": "model-00003-of-00003.safetensors", "decoder.block.23.self_attention_layer.self_attention.Wv.weight": "model-00003-of-00003.safetensors", "decoder.block.23.self_attention_layer.self_attention.o.weight": "model-00003-of-00003.safetensors", "decoder.block.3.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.3.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.3.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.3.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.3.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.3.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.3.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.3.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.3.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.3.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.3.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.3.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.3.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.3.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.4.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.4.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.4.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.4.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.4.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.4.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.4.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.4.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.4.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.4.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.4.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.4.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.4.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.4.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.5.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.5.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.5.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.5.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.5.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.5.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.5.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.5.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.5.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.5.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.5.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.5.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.5.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.5.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.6.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.6.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.6.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.6.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.6.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.6.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.6.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.6.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.6.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.6.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.6.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.6.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.6.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.6.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.7.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.7.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.7.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.7.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.7.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.7.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.7.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.7.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.7.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.7.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.7.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.7.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.7.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.7.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.8.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.8.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.8.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.8.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.8.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.8.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.8.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.8.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.8.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.8.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.8.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.8.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.8.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.8.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.9.cross_attention_layer.cross_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.9.cross_attention_layer.cross_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.9.cross_attention_layer.cross_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.9.cross_attention_layer.cross_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.block.9.cross_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.9.ff_layer.act.wi_0.weight": "model-00002-of-00003.safetensors", "decoder.block.9.ff_layer.act.wi_1.weight": "model-00002-of-00003.safetensors", "decoder.block.9.ff_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.9.ff_layer.wo.weight": "model-00002-of-00003.safetensors", "decoder.block.9.self_attention_layer.layer_norm.weight": "model-00002-of-00003.safetensors", "decoder.block.9.self_attention_layer.self_attention.Wk.weight": "model-00002-of-00003.safetensors", "decoder.block.9.self_attention_layer.self_attention.Wq.weight": "model-00002-of-00003.safetensors", "decoder.block.9.self_attention_layer.self_attention.Wv.weight": "model-00002-of-00003.safetensors", "decoder.block.9.self_attention_layer.self_attention.o.weight": "model-00002-of-00003.safetensors", "decoder.final_layer_norm.weight": "model-00003-of-00003.safetensors", "encoder.block.0.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.0.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.0.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.0.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.0.self_attention_layer.self_attention.pe_encoding.relative_attention_bias.weight": "model-00001-of-00003.safetensors", "encoder.block.1.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.1.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.1.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.1.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.1.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.1.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.1.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.1.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.1.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.10.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.10.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.10.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.10.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.10.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.10.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.10.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.10.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.10.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.11.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.11.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.11.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.11.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.11.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.11.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.11.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.11.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.11.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.12.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.12.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.12.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.12.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.12.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.12.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.12.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.12.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.12.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.13.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.13.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.13.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.13.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.13.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.13.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.13.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.13.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.13.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.14.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.14.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.14.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.14.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.14.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.14.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.14.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.14.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.14.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.15.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.15.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.15.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.15.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.15.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.15.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.15.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.15.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.15.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.16.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.16.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.16.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.16.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.16.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.16.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.16.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.16.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.16.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.17.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.17.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.17.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.17.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.17.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.17.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.17.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.17.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.17.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.18.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.18.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.18.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.18.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.18.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.18.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.18.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.18.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.18.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.19.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.19.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.19.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.19.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.19.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.19.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.19.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.19.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.19.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.2.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.2.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.2.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.2.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.2.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.2.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.2.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.2.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.2.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.20.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.20.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.20.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.20.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.20.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.20.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.20.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.20.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.20.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.21.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.21.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.21.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.21.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.21.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.21.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.21.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.21.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.21.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.22.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.22.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.22.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.22.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.22.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.22.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.22.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.22.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.22.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.23.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.23.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.23.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.23.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.23.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.23.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.23.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.23.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.23.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.3.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.3.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.3.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.3.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.3.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.3.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.3.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.3.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.3.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.4.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.4.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.4.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.4.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.4.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.4.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.4.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.4.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.4.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.5.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.5.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.5.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.5.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.5.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.5.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.5.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.5.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.5.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.6.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.6.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.6.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.6.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.6.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.6.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.6.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.6.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.6.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.7.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.7.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.7.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.7.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.7.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.7.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.7.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.7.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.7.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.8.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.8.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.8.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.8.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.8.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.8.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.8.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.8.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.8.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.block.9.ff_layer.act.wi_0.weight": "model-00001-of-00003.safetensors", "encoder.block.9.ff_layer.act.wi_1.weight": "model-00001-of-00003.safetensors", "encoder.block.9.ff_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.9.ff_layer.wo.weight": "model-00001-of-00003.safetensors", "encoder.block.9.self_attention_layer.layer_norm.weight": "model-00001-of-00003.safetensors", "encoder.block.9.self_attention_layer.self_attention.Wk.weight": "model-00001-of-00003.safetensors", "encoder.block.9.self_attention_layer.self_attention.Wq.weight": "model-00001-of-00003.safetensors", "encoder.block.9.self_attention_layer.self_attention.Wv.weight": "model-00001-of-00003.safetensors", "encoder.block.9.self_attention_layer.self_attention.o.weight": "model-00001-of-00003.safetensors", "encoder.final_layer_norm.weight": "model-00001-of-00003.safetensors", "lm_head.weight": "model-00003-of-00003.safetensors", "shared.weight": "model-00001-of-00003.safetensors" } }