{ "_name_or_path": "/basemodels/models/pythia-410m", "abl_setting": null, "add_GT_ratio": 0, "architectures": [ "HiarachicalPythia" ], "attention_bias": true, "attention_dropout": 0.0, "attn_implementation": "flash_attention_2", "block_size": 2048, "bos_token_id": 0, "chunk_merge_method": "meanpooling", "chunk_size": 4, "classifier_dropout": 0.1, "decoder_layers": 23, "detach_or_not": "pre_post", "dist_topk": 64, "distribution_merge": true, "encoder_layers": 1, "eos_token_id": 0, "ex_hlm_loss_ratio": 1.0, "ex_loss_ratio": 1.0, "ex_vq_loss_display": 1.0, "base_model_name_or_path": "/basemodels/models/pythia-410m", "head_factor": 1.0, "hidden_act": "gelu", "hidden_dropout": 0.0, "hidden_size": 1024, "hlm_hs_lambda": 0.5, "hlm_type": "vq_gpt2_ce", "initializer_range": 0.02, "intermediate_size": 4096, "layer_norm_eps": 1e-05, "layer_norm_option": "rawadd", "loss_abl_method": "loss2-module13-loss3-module1234", "loss_type": "hlm_MSE_loss", "max_position_embeddings": 2048, "model_name_or_path": "/basemodels/models/pythia-410m", "model_type": "gpt_neox", "num_attention_heads": 16, "num_hidden_layers": 24, "padding_factor": 1, "projector_pos": "no_projector", "rope_scaling": null, "rotary_emb_base": 10000, "rotary_pct": 0.25, "shift_feature": true, "special_layers": 2, "tie_word_embeddings": false, "torch_dtype": "float32", "training_type": "full_from_scratch_addvqfeature", "transformers_version": "4.44.0", "use_cache": true, "use_parallel_residual": true, "vae_config_path": "/conf/SimVQ_cb64.yaml", "vocab_size": 50304, "vq_config": { "codebook_size": 64, "kmeans_init": false, "learnable_codebook": false, "vq_type": "SimVQ" }, "vq_patch_ratio": 1 }