# Example configuration for training a SentencePiece tokenizer from JSONL data # Dataset source - JSONL file jsonl_path: /home/student/Data/TaoData/pretrain.jsonl text_field: text # Field name in JSON for text data # Tokenizer training parameters vocab_size: 8192 # Keep aligned with pretrain/sft model vocab_size model_type: unigram # SentencePiece model type: unigram, bpe, char, word character_coverage: 0.9995 # Output configuration output_dir: tokenizer tokenizer_prefix: tokenizer # Custom special tokens # Built-in tokens are managed by SentencePiece and resolved at runtime. # Entries here are registered as user-defined symbols and should encode as # single tokens, but SentencePiece does not guarantee their exact IDs. # Note: Use \n for newline token, \t for tab, etc. special_tokens: - "\n" # Newline token - quoted to preserve literal \n in YAML - # Special token for chain-of-thought reasoning - # User message token - # Assistant message token - # Image token for multimodal models # Data sampling (optional) # Set to a number to train on only the first N samples from the JSONL file # Useful for quick testing or sub-sampling large datasets # Omit or set to null to use entire file max_samples: 1000000 # Optional metadata tokenizer_name: tokenizer