Instructions to use Nexusflow/NexusRaven-13B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Nexusflow/NexusRaven-13B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Nexusflow/NexusRaven-13B")# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("Nexusflow/NexusRaven-13B") model = AutoModelForCausalLM.from_pretrained("Nexusflow/NexusRaven-13B", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Nexusflow/NexusRaven-13B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Nexusflow/NexusRaven-13B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Nexusflow/NexusRaven-13B", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/Nexusflow/NexusRaven-13B
- SGLang
How to use Nexusflow/NexusRaven-13B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Nexusflow/NexusRaven-13B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Nexusflow/NexusRaven-13B", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Nexusflow/NexusRaven-13B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Nexusflow/NexusRaven-13B", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use Nexusflow/NexusRaven-13B with Docker Model Runner:
docker model run hf.co/Nexusflow/NexusRaven-13B
File size: 3,965 Bytes
ef90054 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 | from typing import Literal
import math
import inspect
from transformers import pipeline
##########################################################
# Step 1: Define the functions you want to articulate. ###
##########################################################
def calculator(
input_a: float,
input_b: float,
operation: Literal["add", "subtract", "multiply", "divide"],
):
"""
Computes a calculation.
Args:
input_a (float) : Required. The first input.
input_b (float) : Required. The second input.
operation (string): The operation. Choices include: add to add two numbers, subtract to subtract two numbers, multiply to multiply two numbers, and divide to divide them.
"""
match operation:
case "add":
return input_a + input_b
case "subtract":
return input_a - input_b
case "multiply":
return input_a * input_b
case "divide":
return input_a / input_b
def cylinder_volume(radius, height):
"""
Calculate the volume of a cylinder.
Parameters:
- radius (float): The radius of the base of the cylinder.
- height (float): The height of the cylinder.
Returns:
- float: The volume of the cylinder.
"""
if radius < 0 or height < 0:
raise ValueError("Radius and height must be non-negative.")
volume = math.pi * (radius**2) * height
return volume
#############################################################
# Step 2: Let's define some utils for building the prompt ###
#############################################################
def format_functions_for_prompt(*functions):
formatted_functions = []
for func in functions:
source_code = inspect.getsource(func)
docstring = inspect.getdoc(func)
formatted_functions.append(
f"OPTION:\n<func_start>{source_code}<func_end>\n<docstring_start>\n{docstring}\n<docstring_end>"
)
return "\n".join(formatted_functions)
##############################
# Step 3: Construct Prompt ###
##############################
def construct_prompt(user_query: str):
formatted_prompt = format_functions_for_prompt(calculator, cylinder_volume)
formatted_prompt += f"\n\nUser Query: Question: {user_query}\n"
prompt = (
"<human>:\n"
+ formatted_prompt
+ "Please pick a function from the above options that best answers the user query and fill in the appropriate arguments.<human_end>"
)
return prompt
#######################################
# Step 4: Execute the function call ###
#######################################
def execute_function_call(model_output):
# Ignore everything after "Reflection" since that is not essential.
function_call = (
model_output[0]["generated_text"]
.strip()
.split("\n")[1]
.replace("Initial Answer:", "")
.strip()
)
try:
return eval(function_call)
except Exception as e:
return str(e)
if __name__ == "__main__":
# Build the model
text_gen = pipeline(
"text-generation",
model="Nexusflow/NexusRaven-13B",
device="cuda",
)
# Comp[ute a Simple equation
prompt = construct_prompt("What is 1+10?")
model_output = text_gen(
prompt, do_sample=False, max_new_tokens=400, return_full_text=False
)
result = execute_function_call(model_output)
print("Model Output:", model_output)
print("Execution Result:", result)
prompt = construct_prompt(
"I have a cake that is about 3 centimenters high and 200 centimeters in diameter. How much cake do I have?"
)
model_output = text_gen(
prompt,
do_sample=False,
max_new_tokens=400,
return_full_text=False,
stop=["\nReflection:"],
)
result = execute_function_call(model_output)
print("Model Output:", model_output)
print("Execution Result:", result)
|