Nightly (#784)
* Update __init__.py * dynamic RoPE * Update mistral.py * Update llama.py * Update tokenizer_utils.py * Update mistral.py * Update llama.py * Update __init__.py * Update flex_attention.py * Update llama.py * Update llama.py * Mistral Nemo * Update tokenizer_utils.py * Update tokenizer_utils.py * Update tokenizer_utils.py
This commit is contained in:
parent
6786fa425f
commit
ded20b2462
1 changed files with 5 additions and 0 deletions
|
|
@ -682,6 +682,11 @@ def fix_untrained_tokens(model, tokenizer, train_dataset, eps = 1e-16):
|
|||
embedding_matrix = model.get_input_embeddings ().weight
|
||||
lm_head_matrix = model.get_output_embeddings().weight
|
||||
|
||||
# Ignore some model checks for now
|
||||
if model.config._name_or_path in IGNORED_TOKENIZER_NAMES:
|
||||
return
|
||||
pass
|
||||
|
||||
# Get untrained tokens
|
||||
indicator_untrained = torch.amax(embedding_matrix, axis = 1) <= eps
|
||||
where_untrained = torch.where(indicator_untrained)[0]
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue