Downloads · 30 days
2
14% of all-time downloads
automateyournetwork/Cisco_DevNet_Sandbox_Running_Config
Cisco_DevNet_Sandbox_Running_Config is a machine learning model from automateyournetwork. Use it for the machine learning task on the model card, and read the license before you ship it in a product. It is set up for transformers.
Downloads · 30 days
2
14% of all-time downloads
All-time downloads
14
Public
Repo size
904 KB
Likes
2
Public
Click a slice to open those files.
.bin100 KB · 96%
From the Hugging Face model README
To use this model:
import torch
import transformers
import pyreft
device = "cuda"
# Load the base model
model_name_or_path = "meta-llama/Meta-Llama-3-8B"
model = transformers.AutoModelForCausalLM.from_pretrained(
model_name_or_path, torch_dtype=torch.bfloat16, device_map={"": device}
)
# Load the ReFT model
reft_model = pyreft.ReftModel.load(
"./CiscoDevNetSandboxRunningConfig", model
)
# Ensure the ReFT model components are also on the GPU
reft_model.set_device(device)
# Define the prompt template
prompt_no_input_template = """<s>[INST] <<SYS>>
You are a computer networking expert specialized in Cisco IOS XE running configurations.
<</SYS>>
%s [/INST]
"""
# Load the tokenizer
tokenizer = transformers.AutoTokenizer.from_pretrained(
model_name_or_path, model_max_length=2048,
padding_side="right", use_fast=False
)
# Set pad_token as eos_token
tokenizer.pad_token = tokenizer.eos_token
def generate_response(instruction):
# Tokenize and prepare the input
prompt = prompt_no_input_template % instruction
prompt = tokenizer(prompt, return_tensors="pt").to(device)
base_unit_location = prompt["input_ids"].shape[-1] - 1 # Last position
# Move all relevant tensors and operations to the GPU
prompt = {key: value.to(device) for key, value in prompt.items()} # Ensure the prompt is on the GPU
# Generate the response using the reft_model
_, reft_response = reft_model.generate(
prompt, unit_locations={"sources->base": (None, [[[base_unit_location]]])},
intervene_on_prompt=True, max_new_tokens=512, do_sample=True,
eos_token_id=tokenizer.eos_token_id, early_stopping=True
)
fine_tuned_answer = tokenizer.decode(reft_response[0], skip_special_tokens=True)
return fine_tuned_answer
while True:
instruction = input("You: ")
if instruction.lower() == "exit":
print("Goodbye!")
break
# Generate and print the response
fine_tuned_answer = generate_response(instruction)
print(f"Question: {instruction}")
print(f"Fine-tuned Answer: {fine_tuned_answer}")
# Log intervention details
print(f"Intervention applied: {reft_model.interventions}")