hf_text-generation-inference/server/text_generation_server/models/santacoder.py

65 lines
2.0 KiB
Python
Raw Normal View History

2023-01-20 04:24:39 -07:00
import torch
import torch.distributed
2023-02-14 05:02:16 -07:00
from typing import Optional, List
2023-01-20 04:24:39 -07:00
from transformers import AutoTokenizer, AutoModelForCausalLM
2023-03-07 10:52:22 -07:00
from text_generation_server.models import CausalLM
2023-01-20 04:24:39 -07:00
FIM_PREFIX = "<fim-prefix>"
FIM_MIDDLE = "<fim-middle>"
FIM_SUFFIX = "<fim-suffix>"
FIM_PAD = "<fim-pad>"
EOD = "<|endoftext|>"
class SantaCoder(CausalLM):
def __init__(self, model_id: str, revision: Optional[str] = None, quantize=False):
2023-01-20 04:24:39 -07:00
if torch.cuda.is_available():
device = torch.device("cuda")
dtype = torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float32
else:
if quantize:
raise ValueError("quantization is not available on CPU")
device = torch.device("cpu")
dtype = torch.float32
2023-01-31 10:53:56 -07:00
tokenizer = AutoTokenizer.from_pretrained(
model_id, revision=revision, padding_side="left", truncation_side="left"
2023-01-31 10:53:56 -07:00
)
2023-01-20 04:24:39 -07:00
tokenizer.add_special_tokens(
{
"additional_special_tokens": [
EOD,
FIM_PREFIX,
FIM_MIDDLE,
FIM_SUFFIX,
FIM_PAD,
],
"pad_token": EOD,
}
)
self.model = (
AutoModelForCausalLM.from_pretrained(
model_id,
2023-01-31 10:53:56 -07:00
revision=revision,
torch_dtype=dtype,
load_in_8bit=quantize,
trust_remote_code=True, # required
)
.to(device)
.eval()
)
2023-01-20 04:24:39 -07:00
super(CausalLM, self).__init__(
tokenizer=tokenizer, device=device, decode_buffer=1
2023-01-20 04:24:39 -07:00
)
def decode(self, generated_ids: List[int]) -> str:
# Do not skip special tokens as they are used for custom parsing rules of the generated text
return self.tokenizer.decode(
generated_ids, skip_special_tokens=False, cleanup_tokenization_spaces=False
)