HuggingFace Transformers
在 Clore.ai 上将 HuggingFace Transformers 用于 NLP、视觉和音频
最后更新于
这有帮助吗?
这有帮助吗?
pytorch/pytorch:2.5.1-cuda12.4-cudnn9-devel22/tcppip install transformers accelerate datasets huggingface_hubpip install transformers[torch]
pip install accelerate # 用于大型模型
pip install datasets # 用于训练数据from transformers import AutoModelForCausalLM, AutoTokenizer
import torch
model_name = "mistralai/Mistral-7B-Instruct-v0.2"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForCausalLM.from_pretrained(
model_name,
torch_dtype=torch.float16,
device_map="auto"
)
prompt = "用简单的语言解释量子计算:"
inputs = tokenizer(prompt, return_tensors="pt").to("cuda")
outputs = model.generate(
**inputs,
max_new_tokens=200,
temperature=0.7,
do_sample=True
)
response = tokenizer.decode(outputs[0], skip_special_tokens=True)
print(response)from transformers import pipeline
pipe = pipeline(
"text-generation",
model="meta-llama/Llama-2-7b-chat-hf",
torch_dtype=torch.float16,
device_map="auto"
)
messages = [
{"role": "user", "content": "什么是机器学习?"}
]
outputs = pipe(
messages,
max_new_tokens=256,
do_sample=True,
temperature=0.7
)
print(outputs[0]["generated_text"][-1]["content"])from transformers import TextStreamer
streamer = TextStreamer(tokenizer)
model.generate(
**inputs,
max_new_tokens=200,
streamer=streamer
)from transformers import AutoModelForCausalLM, BitsAndBytesConfig
import torch
bnb_config = BitsAndBytesConfig(
load_in_4bit=True,
bnb_4bit_quant_type="nf4",
bnb_4bit_compute_dtype=torch.float16,
bnb_4bit_use_double_quant=True
)
model = AutoModelForCausalLM.from_pretrained(
"meta-llama/Llama-2-13b-hf",
quantization_config=bnb_config,
device_map="auto"
)model = AutoModelForCausalLM.from_pretrained(
"meta-llama/Llama-2-7b-hf",
load_in_8bit=True,
device_map="auto"
)from transformers import AutoModel, AutoTokenizer
import torch
model_name = "sentence-transformers/all-MiniLM-L6-v2"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModel.from_pretrained(model_name).cuda()
def get_embedding(text):
inputs = tokenizer(text, return_tensors="pt", padding=True, truncation=True).to("cuda")
with torch.no_grad():
outputs = model(**inputs)
# 平均池化
embedding = outputs.last_hidden_state.mean(dim=1)
return embedding
emb = get_embedding("Hello, world!")
print(f"Embedding shape: {emb.shape}")from transformers import pipeline
from PIL import Image
classifier = pipeline("image-classification", model="google/vit-base-patch16-224", device=0)
image = Image.open("cat.jpg")
results = classifier(image)
for result in results:
print(f"{result['label']}: {result['score']:.4f}")from transformers import pipeline
from PIL import Image
detector = pipeline("object-detection", model="facebook/detr-resnet-50", device=0)
image = Image.open("street.jpg")
results = detector(image)
for result in results:
print(f"{result['label']}: {result['score']:.4f} at {result['box']}")from transformers import pipeline
from PIL import Image
segmenter = pipeline("image-segmentation", model="facebook/maskformer-swin-base-ade", device=0)
image = Image.open("scene.jpg")
results = segmenter(image)
for segment in results:
print(f"{segment['label']}: score {segment['score']:.4f}")from transformers import pipeline
transcriber = pipeline(
"automatic-speech-recognition",
model="openai/whisper-large-v3",
device=0
)
result = transcriber("audio.mp3")
print(result["text"])from transformers import pipeline
import scipy
synthesizer = pipeline("text-to-speech", model="microsoft/speecht5_tts", device=0)
speech = synthesizer("Hello, this is a test of text to speech.")
scipy.io.wavfile.write("output.wav", rate=speech["sampling_rate"], data=speech["audio"])from datasets import load_dataset
dataset = load_dataset("imdb")
train_dataset = dataset["train"].select(range(1000))
eval_dataset = dataset["test"].select(range(200))from transformers import (
AutoModelForSequenceClassification,
AutoTokenizer,
TrainingArguments,
Trainer
)
model_name = "bert-base-uncased"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=2)
def tokenize_function(examples):
return tokenizer(examples["text"], padding="max_length", truncation=True)
tokenized_train = train_dataset.map(tokenize_function, batched=True)
tokenized_eval = eval_dataset.map(tokenize_function, batched=True)
training_args = TrainingArguments(
output_dir="./results",
evaluation_strategy="epoch",
learning_rate=2e-5,
per_device_train_batch_size=16,
per_device_eval_batch_size=16,
num_train_epochs=3,
weight_decay=0.01,
fp16=True,
)
trainer = Trainer(
model=model,
args=training_args,
train_dataset=tokenized_train,
eval_dataset=tokenized_eval,
)
trainer.train()from transformers import AutoModelForCausalLM
# 自动设备放置
model = AutoModelForCausalLM.from_pretrained(
"meta-llama/Llama-2-70b-hf",
device_map="auto",
torch_dtype=torch.float16
)
# 手动设备映射
device_map = {
"model.embed_tokens": 0,
"model.layers.0": 0,
"model.layers.1": 0,
# ...
"model.layers.39": 1,
"model.norm": 1,
"lm_head": 1
}
model = AutoModelForCausalLM.from_pretrained(
"model_name",
device_map=device_map
)
# Flash Attention 2
model = AutoModelForCausalLM.from_pretrained(
"meta-llama/Llama-2-7b-hf",
torch_dtype=torch.float16,
attn_implementation="flash_attention_2",
device_map="auto"
)
# 梯度检查点
model.gradient_checkpointing_enable()from huggingface_hub import snapshot_download
snapshot_download(
repo_id="meta-llama/Llama-2-7b-hf",
local_dir="./llama-2-7b"
)from huggingface_hub import HfApi
api = HfApi()
api.upload_folder(
folder_path="./my_model",
repo_id="username/my-model",
repo_type="model"
)