初始化项目,由ModelHub XC社区提供模型
Model: ericoh929/qwen3-1.7b-lamini-qlora-instruction-tuned Source: Original Platform
This commit is contained in:
39
infer_simple.py
Normal file
39
infer_simple.py
Normal file
@@ -0,0 +1,39 @@
|
||||
# infer_simple.py
|
||||
import torch
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
from prompt_format import format_instruction_prompt
|
||||
|
||||
MODEL_ID = "ericoh929/qwen3-1.7b-lamini-qlora-instruction-tuned"
|
||||
|
||||
def main():
|
||||
tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True)
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
MODEL_ID,
|
||||
trust_remote_code=True,
|
||||
device_map="auto",
|
||||
torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32,
|
||||
)
|
||||
model.eval()
|
||||
|
||||
instruction = "Answer the question. Be concise."
|
||||
inp = "If Tom has 3 apples and buys 4 more, how many apples does he have?"
|
||||
|
||||
prompt = format_instruction_prompt(instruction, inp)
|
||||
inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
|
||||
|
||||
with torch.inference_mode():
|
||||
out = model.generate(
|
||||
**inputs,
|
||||
max_new_tokens=256,
|
||||
do_sample=False,
|
||||
pad_token_id=tokenizer.pad_token_id or tokenizer.eos_token_id,
|
||||
eos_token_id=tokenizer.eos_token_id,
|
||||
)
|
||||
|
||||
gen_ids = out[0][inputs["input_ids"].shape[1]:]
|
||||
answer = tokenizer.decode(gen_ids, skip_special_tokens=True).strip()
|
||||
|
||||
print(answer)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user