Skip to content

Commit

Permalink
fix(config): Set eos/bos to tokenizer if different (axolotl-ai-cloud#801
Browse files Browse the repository at this point in the history
)

* fix(config): Set eos/bos to tokenizer if different

* chore: fix lint
  • Loading branch information
NanoCode012 committed Oct 29, 2023
1 parent ae306fb commit 874d485
Showing 1 changed file with 14 additions and 0 deletions.
14 changes: 14 additions & 0 deletions src/axolotl/utils/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -386,6 +386,20 @@ def load_model(
)
model.config.max_position_embeddings = cfg.sequence_len

if (
hasattr(model.config, "bos_token_id")
and model.config.bos_token_id
and model.config.bos_token_id != tokenizer.bos_token_id
):
model.config.bos_token_id = tokenizer.bos_token_id

if (
hasattr(model.config, "eos_token_id")
and model.config.eos_token_id
and model.config.eos_token_id != tokenizer.eos_token_id
):
model.config.eos_token_id = tokenizer.eos_token_id

if model.device.type == "cuda":
log_gpu_memory_usage(LOG, "after model load", model.device)

Expand Down

0 comments on commit 874d485

Please sign in to comment.