TaoNet-mini-A2 / configs /tokenizer.yaml
Lobakkang's picture
Upload folder using huggingface_hub
d092869 verified
Raw
History Blame Contribute Delete
1.36 kB
# Example configuration for training a SentencePiece tokenizer from JSONL data
# Dataset source - JSONL file
jsonl_path: /home/student/Data/TaoData/pretrain.jsonl
text_field: text # Field name in JSON for text data
# Tokenizer training parameters
vocab_size: 8192 # Keep aligned with pretrain/sft model vocab_size
model_type: unigram # SentencePiece model type: unigram, bpe, char, word
character_coverage: 0.9995
# Output configuration
output_dir: tokenizer
tokenizer_prefix: tokenizer
# Custom special tokens
# Built-in tokens are managed by SentencePiece and resolved at runtime.
# Entries here are registered as user-defined symbols and should encode as
# single tokens, but SentencePiece does not guarantee their exact IDs.
# Note: Use \n for newline token, \t for tab, etc.
special_tokens:
- "\n" # Newline token - quoted to preserve literal \n in YAML
- <think> # Special token for chain-of-thought reasoning
- <user> # User message token
- <assistant> # Assistant message token
- <image> # Image token for multimodal models
# Data sampling (optional)
# Set to a number to train on only the first N samples from the JSONL file
# Useful for quick testing or sub-sampling large datasets
# Omit or set to null to use entire file
max_samples: 1000000
# Optional metadata
tokenizer_name: tokenizer