ifmanzhang's picture
Add files using upload-large-folder tool
038aa4b verified
Raw
History Blame Contribute Delete
1.74 kB
model:
id: "Qwen/Qwen3.5-9B"
# Local path override (optional, leave empty to download from HF)
local_path: ""
quantization: "4bit" # "4bit" | "8bit" | "none"
torch_dtype: "bfloat16"
steering:
# Which transformer layer indices to extract vectors from.
# "all" to sweep every layer; or a list like [16, 20, 24, 28]
extract_layers: "all"
# Scaling coefficient at inference time
alpha: 20.0
# Traits to apply: each entry specifies the trait name, best layer, and optional alpha override
active_traits:
- name: taciturn
layer: 20
alpha: 20
- name: ruthless
layer: 14
alpha: 12
data:
pairs_dir: "data/pairs"
# Number of samples used for vector extraction (per trait, per polarity)
max_samples: 256
finetune:
output_dir: "finetune/checkpoints"
lora_r: 16
lora_alpha: 32
lora_dropout: 0.05
target_modules: "all-linear"
learning_rate: 2.0e-4
num_epochs: 3
per_device_train_batch_size: 1
gradient_accumulation_steps: 8
max_seq_length: 512
warmup_ratio: 0.03
fp16: false
bf16: true
gemini:
model: "gemini-3-pro"
# Set GEMINI_API_KEY env var; do not hard-code keys here
samples_per_trait: 300
traits:
- name: taciturn
description: "极度寡言少语、言简意赅,不废话不解释"
opposite: "热情健谈、喜欢展开解释"
- name: ruthless
description: "冷酷无情、不在意对方感受,只讲结果"
opposite: "温暖体贴、充满同理心"
style:
# Bilingual style fine-tune: train on top of merged_model
base_model: "merged_model"
output_dir: "finetune/style_checkpoint"
merged_output: "merged_model_v2"
zh_samples: 200 # Chinese archaic pairs
en_samples: 100 # English archaic pairs