{ "model_type": "fock-parflm-v2.1-depthcond-multicontext-gaussian-vtheta", "framework": "semsimula (Semantic Simulation SPLM family)", "architecture": "FockMultiXiPARFLM", "vtheta": "DepthConditionedMultiContextGaussianVTheta (5 heads x 8 wells = 40 attractors, per-layer depth codes)", "corpus": "openwebtext", "tokenizer": "gpt2 (BPE, vocab 50257)", "total_parameters": 53378075, "total_state_dict_tensors": 53428333, "final_val_ppl": 27.23, "final_val_loss": 3.3043, "training_steps_total": 250000, "training_steps_phase1": 100000, "training_steps_phase2": 150000, "training_tokens": "graduated 1B -> 2B (2.05B consumed)", "reverse_channel": true, "tie_embeddings": false, "model_cfg": { "vocab_size": 50257, "d": 384, "max_len": 1024, "L": 16, "v_hidden": 1024, "v_depth": 3, "dt": 1.0, "init_m": 1.0, "init_gamma": 1.0, "learn_mgamma": true, "fixed_gamma": 0.3, "v_phi_kind": "structural_competitive", "v_phi_d_type": 32, "v_phi_d_angle": 16, "v_phi_phi_hidden": 128, "v_phi_theta_hidden": 128, "v_phi_mlp_hidden": 128, "v_phi_C": 1.0, "v_phi_eps": 0.1, "v_phi_init_scale": 0.02, "v_phi_n_heads": 4, "v_phi_add_value_transport": false, "v_phi_vt_n_heads": 4, "v_phi_vt_d_head": 32, "v_phi_vt_sigma": 1.0, "v_phi_competitive_temp": 1.0, "v_phi_competitive_scale": "row", "ln_before_distance": true, "per_layer_v_phi_scale": true, "per_layer_scale_init": -3.0, "theta_activation": "tanh", "theta_form": "mlp", "mass_mode": "logfreq", "logfreq_init_alpha": 0.1, "logfreq_path": "logfreq_surprisal_openwebtext.npy", "ln_after_step": true, "ln_eps": 1e-05, "causal_force": true, "tie_embeddings": false, "use_output_bias": true, "use_grad_checkpoint": false, "use_layer_checkpoint": true, "top_k": 16, "score_head_hidden": 32, "score_head_init_scale": 0.02, "gumbel_tau_init": 1.0, "gumbel_tau_min": 0.3, "gumbel_noise": true, "score_head_use_detached_h_src": true, "use_gathered_v_phi": true, "xi_channels": 5, "xi_alpha_inits": [ 0.5, 0.75, 0.95, 0.99, 0.995 ], "xi_learnable": true, "xi_alpha_init_mode": "explicit", "xi_tau_max": 100.0, "force_clamp_max": null, "ln_before_vtheta": false, "fock_version": "v2", "n_registers": 32, "register_salience_decay": 0.5, "register_salience_threshold": 0.005, "creation_gate_hidden": 64, "stack_discipline": true, "register_init_scale": 0.02, "d_k": 64, "tau_create_init": 8.0, "destruction_gate_hidden": 64, "reverse_channel": true, "per_register_tau": true, "per_register_keys": true, "ortho_register_init": true, "register_repulsion": true, "register_repulsion_coeff": 0.05, "register_repulsion_kind": "gram", "reverse_channel_stable": true, "reverse_channel_pre_ln": true, "reverse_channel_soft_norm": true, "reverse_channel_warmup_steps": 4000, "reverse_channel_per_layer": true }, "train_cfg_phase2": { "batch_size": 8, "block_size": 512, "grad_accum": 2, "effective_batch": 16, "steps": 150000, "lr": 0.00015, "weight_decay": 0.01, "warmup_steps": 2000, "grad_clip": 1.0, "grad_clip_vphi": 0.3, "optimizer": "adamw", "grad_centralization": false, "lambda_v": 0.01, "v_theta_variant": "gaussian", "lr_schedule": "wsd", "v_theta_n_heads": 5, "v_theta_wells_per_head": 8, "v_theta_depth_condition": true, "v_theta_depth_code_init_std": 0.02 }, "final_learned_xi_alphas": [ 0.1722, 0.3218, 0.4975, 0.6386, 0.9649 ], "seed": 0 }