File size: 1,798 Bytes
d90f91c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
# ─────────────────────────────────────────────
# Chinese BabyLM Evaluation Pipeline Config
# ─────────────────────────────────────────────

# Models to evaluate.
# Each entry needs:
#   path    β€” HuggingFace repo ID or local directory path
#   backend β€” one of: causal, mlm, mntp, enc_dec_mask, enc_dec_prefix
models:
  - path: timorobrecht/full_chinese_gpu3.2-dpo
    backend: causal
  # - path: /path/to/local/model
  #   backend: mlm

# Tasks to run. Comment out any group or individual task to skip it.
tasks:
  # NLU Track β€” zero-shot minimal pairs
  zero_shot:
    - zhoblimp
    - hanzi_structure
    - hanzi_pinyin

  # Cog Track β€” fMRI brain encoding
  cogbench:
    - word_fmri
    - fmri

  # Fine-tuning Track β€” CLUE tasks
  finetune:
    - afqmc
    - ocnli
    - tnews
    - cluewsc2020

# Directories
eval_dir: evaluation_data   # where prepare_chinese_data.py puts data
results_dir: results        # where eval results are written

# Save items containing UNK tokens for hanzi track tasks
save_item_with_unk: true

# Fine-tuning hyperparameters
# Global defaults are applied first; per-task overrides are merged on top.
finetune_hparams:
  lr: 3.0e-5
  batch_size: 32
  max_epochs: 10
  sequence_length: 128
  seed: 42

  task_overrides:
    afqmc:
      lr: 3.0e-5
      batch_size: 32
      max_epochs: 10
    ocnli:
      lr: 3.0e-5
      batch_size: 32
      max_epochs: 10
    tnews:
      lr: 3.0e-5
      batch_size: 32
      max_epochs: 10
    cluewsc2020:
      lr: 3.0e-5
      batch_size: 32
      max_epochs: 30