This commit is contained in:
2025-03-25 19:48:34 +07:00
parent 1bef081ca2
commit 05d3dfcf25
8 changed files with 778 additions and 0 deletions

131
app/ctypes_bindings.py Normal file
View File

@@ -0,0 +1,131 @@
import ctypes
# Define global variables to store the callback function output for displaying in the Gradio interface
global_text = []
global_state = -1
split_byte_data = bytes(b"") # Used to store the segmented byte data
# Set the dynamic library path
# Default is v1.1.2
rkllm_lib = ctypes.CDLL('/usr/lib/librkllmrt.so')
# Define the structures from the library
RKLLM_Handle_t = ctypes.c_void_p
userdata = ctypes.c_void_p(None)
LLMCallState = ctypes.c_int
LLMCallState.RKLLM_RUN_NORMAL = 0
LLMCallState.RKLLM_RUN_WAITING = 1
LLMCallState.RKLLM_RUN_FINISH = 2
LLMCallState.RKLLM_RUN_ERROR = 3
LLMCallState.RKLLM_RUN_GET_LAST_HIDDEN_LAYER = 4
RKLLMInputMode = ctypes.c_int
RKLLMInputMode.RKLLM_INPUT_PROMPT = 0
RKLLMInputMode.RKLLM_INPUT_TOKEN = 1
RKLLMInputMode.RKLLM_INPUT_EMBED = 2
RKLLMInputMode.RKLLM_INPUT_MULTIMODAL = 3
RKLLMInferMode = ctypes.c_int
RKLLMInferMode.RKLLM_INFER_GENERATE = 0
RKLLMInferMode.RKLLM_INFER_GET_LAST_HIDDEN_LAYER = 1
class RKLLMExtendParam(ctypes.Structure):
_fields_ = [
("base_domain_id", ctypes.c_int32),
("reserved", ctypes.c_uint8 * 112)
]
class RKLLMParam(ctypes.Structure):
_fields_ = [
("model_path", ctypes.c_char_p),
("max_context_len", ctypes.c_int32),
("max_new_tokens", ctypes.c_int32),
("top_k", ctypes.c_int32),
("top_p", ctypes.c_float),
("temperature", ctypes.c_float),
("repeat_penalty", ctypes.c_float),
("frequency_penalty", ctypes.c_float),
("presence_penalty", ctypes.c_float),
("mirostat", ctypes.c_int32),
("mirostat_tau", ctypes.c_float),
("mirostat_eta", ctypes.c_float),
("skip_special_token", ctypes.c_bool),
("is_async", ctypes.c_bool),
("img_start", ctypes.c_char_p),
("img_end", ctypes.c_char_p),
("img_content", ctypes.c_char_p),
("extend_param", RKLLMExtendParam),
]
class RKLLMLoraAdapter(ctypes.Structure):
_fields_ = [
("lora_adapter_path", ctypes.c_char_p),
("lora_adapter_name", ctypes.c_char_p),
("scale", ctypes.c_float)
]
class RKLLMEmbedInput(ctypes.Structure):
_fields_ = [
("embed", ctypes.POINTER(ctypes.c_float)),
("n_tokens", ctypes.c_size_t)
]
class RKLLMTokenInput(ctypes.Structure):
_fields_ = [
("input_ids", ctypes.POINTER(ctypes.c_int32)),
("n_tokens", ctypes.c_size_t)
]
class RKLLMMultiModelInput(ctypes.Structure):
_fields_ = [
("prompt", ctypes.c_char_p),
("image_embed", ctypes.POINTER(ctypes.c_float)),
("n_image_tokens", ctypes.c_size_t)
]
class RKLLMInputUnion(ctypes.Union):
_fields_ = [
("prompt_input", ctypes.c_char_p),
("embed_input", RKLLMEmbedInput),
("token_input", RKLLMTokenInput),
("multimodal_input", RKLLMMultiModelInput)
]
class RKLLMInput(ctypes.Structure):
_fields_ = [
("input_mode", ctypes.c_int),
("input_data", RKLLMInputUnion)
]
class RKLLMLoraParam(ctypes.Structure):
_fields_ = [
("lora_adapter_name", ctypes.c_char_p)
]
class RKLLMPromptCacheParam(ctypes.Structure):
_fields_ = [
("save_prompt_cache", ctypes.c_int),
("prompt_cache_path", ctypes.c_char_p)
]
class RKLLMInferParam(ctypes.Structure):
_fields_ = [
("mode", RKLLMInferMode),
("lora_params", ctypes.POINTER(RKLLMLoraParam)),
("prompt_cache_params", ctypes.POINTER(RKLLMPromptCacheParam))
]
class RKLLMResultLastHiddenLayer(ctypes.Structure):
_fields_ = [
("hidden_states", ctypes.POINTER(ctypes.c_float)),
("embd_size", ctypes.c_int),
("num_tokens", ctypes.c_int)
]
class RKLLMResult(ctypes.Structure):
_fields_ = [
("text", ctypes.c_char_p),
("size", ctypes.c_int),
("last_hidden_layer", RKLLMResultLastHiddenLayer)
]

39
app/entrypoint.sh Executable file
View File

@@ -0,0 +1,39 @@
#!/bin/bash
message_print() {
echo
echo "#########################################"
echo $1
echo "#########################################"
echo
}
message_print "install apt packages"
apt update
apt install -y pip git curl wget nano sudo apt-utils cmake
cd /
message_print "Changing to repository..."
git clone https://github.com/DrHo1y/ezrknn-llm
cd ezrknn-llm/
cp ./rkllm-runtime/runtime/Linux/librkllm_api/aarch64/* /usr/lib
cp ./rkllm-runtime/runtime/Linux/librkllm_api/include/* /usr/local/include
message_print "Compiling LLM runtime for Linux..."
cd ./rkllm-runtime/examples/rkllm_api_demo/
bash build-linux.sh
message_print "Moving rkllm to /usr/bin..."
cp ./build/build_linux_aarch64_Release/llm_demo /usr/bin/rkllm
echo "* soft nofile 16384" >> /etc/security/limits.conf
echo "* hard nofile 1048576" >> /etc/security/limits.conf
message_print "Increasing file limit for all users (needed for LLMs to run)..."
echo "root soft nofile 16384" >> /etc/security/limits.conf
echo "root hard nofile 1048576" >> /etc/security/limits.conf
message_print "Done installing ezrknn-llm!"
message_print "Install python packages"
cd /offline_packages
python -m pip install --no-index --find-links . -r requirements.txt
message_print "Start Gradio server"
cd /app
python rkllm_server_gradio.py

54
app/mesh_utils.py Normal file
View File

@@ -0,0 +1,54 @@
# From https://github.com/nv-tlabs/LLaMA-Mesh/
# For use with https://huggingface.co/c01zaut/LLaMA-Mesh-rk3588-1.1.1
from trimesh.exchange.gltf import export_glb
import trimesh
import numpy as np
import tempfile
def apply_gradient_color(mesh_text):
"""
Apply a gradient color to the mesh vertices based on the Y-axis and save as GLB.
Args:
mesh_text (str): The input mesh in OBJ format as a string.
Returns:
str: Path to the GLB file with gradient colors applied.
"""
# Load the mesh
temp_file = tempfile.NamedTemporaryFile(suffix=f"", delete=False).name
with open(temp_file+".obj", "w") as f:
f.write(mesh_text)
# return temp_file
mesh = trimesh.load_mesh(temp_file+".obj", file_type='obj')
# Get vertex coordinates
vertices = mesh.vertices
y_values = vertices[:, 1] # Y-axis values
# Normalize Y values to range [0, 1] for color mapping
y_normalized = (y_values - y_values.min()) / (y_values.max() - y_values.min())
# Generate colors: Map normalized Y values to RGB gradient (e.g., blue to red)
colors = np.zeros((len(vertices), 4)) # RGBA
colors[:, 0] = y_normalized # Red channel
colors[:, 2] = 1 - y_normalized # Blue channel
colors[:, 3] = 1.0 # Alpha channel (fully opaque)
# Attach colors to mesh vertices
mesh.visual.vertex_colors = colors
# Export to GLB format
glb_path = temp_file+".glb"
with open(glb_path, "wb") as f:
f.write(export_glb(mesh))
return glb_path
def visualize_mesh(mesh_text):
"""
Convert the provided 3D mesh text into a visualizable format.
This function assumes the input is in OBJ format.
"""
temp_file = "temp_mesh.obj"
with open(temp_file, "w") as f:
f.write(mesh_text)
return temp_file

203
app/model_class.py Normal file
View File

@@ -0,0 +1,203 @@
from transformers import AutoTokenizer
from ctypes_bindings import *
from model_configs import model_configs
import threading
import time
import sys
import os
MODEL_PATH = "./models"
# Create a dict of various model configs, and then check the ./models directory if any exist
# This will become the content of the model selector's drop down menu
def available_models():
if not os.path.exists(MODEL_PATH):
os.mkdir(MODEL_PATH)
# Initialize the dict of available models as empty
rkllm_model_files = {}
# Populate the dictionary with found models, and their base configurations
for family, config in model_configs.items():
for model, details in config["models"].items():
filename = details["filename"]
if os.path.exists(os.path.join(MODEL_PATH, filename)):
rkllm_model_files[model] = {}
rkllm_model_files[model].update({"name": model,"family": family, "filename": filename, "config": config["base_config"]})
return rkllm_model_files
# Define the callback function
def callback_impl(result, userdata, state):
global global_text, global_state, split_byte_data
if state == LLMCallState.RKLLM_RUN_FINISH:
global_state = state
print("\n")
sys.stdout.flush()
elif state == LLMCallState.RKLLM_RUN_ERROR:
global_state = state
print("run error")
sys.stdout.flush()
elif state == LLMCallState.RKLLM_RUN_GET_LAST_HIDDEN_LAYER:
'''
If using the GET_LAST_HIDDEN_LAYER function, the callback interface will return the memory pointer: last_hidden_layer, the number of tokens: num_tokens, and the size of the hidden layer: embd_size.
With these three parameters, you can retrieve the data from last_hidden_layer.
Note: The data needs to be retrieved during the current callback; if not obtained in time, the pointer will be released by the next callback.
'''
if result.last_hidden_layer.embd_size != 0 and result.last_hidden_layer.num_tokens != 0:
data_size = result.last_hidden_layer.embd_size * result.last_hidden_layer.num_tokens * ctypes.sizeof(ctypes.c_float)
print(f"data_size: {data_size}")
global_text.append(f"data_size: {data_size}\n")
output_path = os.getcwd() + "/last_hidden_layer.bin"
with open(output_path, "wb") as outFile:
data = ctypes.cast(result.last_hidden_layer.hidden_states, ctypes.POINTER(ctypes.c_float))
float_array_type = ctypes.c_float * (data_size // ctypes.sizeof(ctypes.c_float))
float_array = float_array_type.from_address(ctypes.addressof(data.contents))
outFile.write(bytearray(float_array))
print(f"Data saved to {output_path} successfully!")
global_text.append(f"Data saved to {output_path} successfully!")
else:
print("Invalid hidden layer data.")
global_text.append("Invalid hidden layer data.")
global_state = state
time.sleep(0.05)
sys.stdout.flush()
else:
# Save the output token text and the RKLLM running state
global_state = state
# Monitor if the current byte data is complete; if incomplete, record it for later parsing
try:
if split_byte_data == None or split_byte_data == "" or split_byte_data == '':
global_text.append((b"" + result.contents.text).decode('utf-8'))
print((split_byte_data + result.contents.text).decode('utf-8'), end='')
split_byte_data = bytes(b"")
else:
global_text.append((split_byte_data + result.contents.text).decode('utf-8'))
print((split_byte_data + result.contents.text).decode('utf-8'), end='')
split_byte_data = bytes(b"")
except:
if result.contents.text is not None:
split_byte_data += result.contents.text
sys.stdout.flush()
# Connect the callback function between the Python side and the C++ side
callback_type = ctypes.CFUNCTYPE(None, ctypes.POINTER(RKLLMResult), ctypes.c_void_p, ctypes.c_int)
callback = callback_type(callback_impl)
class RKLLMLoaderClass:
def __init__(self, model="", qtype="w8a8", opt="1", hybrid_quant="1.0"):
self.qtype = qtype
self.opt = opt
self.model = model
self.hybrid_quant = hybrid_quant
if self.model == "":
print("No models loaded yet!")
else:
self.available_models = available_models()
self.model = self.available_models[model]
self.family = self.model["family"]
self.model_path = "models/" + self.model["filename"]
self.base_config = self.model["config"]
self.model_name = self.model["name"]
self.st_model_id = self.base_config["st_model_id"]
self.system_prompt = self.base_config["system_prompt"]
self.rkllm_param = RKLLMParam()
self.rkllm_param.model_path = bytes(self.model_path, 'utf-8')
self.rkllm_param.max_context_len = self.base_config["max_context_len"]
self.rkllm_param.max_new_tokens = self.base_config["max_new_tokens"]
self.rkllm_param.skip_special_token = True
self.rkllm_param.top_k = self.base_config["top_k"]
self.rkllm_param.top_p = self.base_config["top_p"]
# self.rkllm_param.min_p = 0.1
self.rkllm_param.temperature = self.base_config["temperature"]
self.rkllm_param.repeat_penalty = self.base_config["repeat_penalty"]
self.rkllm_param.frequency_penalty = self.base_config["frequency_penalty"]
self.rkllm_param.presence_penalty = 0.0
self.rkllm_param.mirostat = 0
self.rkllm_param.mirostat_tau = 5.0
self.rkllm_param.mirostat_eta = 0.1
self.rkllm_param.is_async = False
self.rkllm_param.img_start = "<image>".encode('utf-8')
self.rkllm_param.img_end = "</image>".encode('utf-8')
self.rkllm_param.img_content = "<unk>".encode('utf-8')
self.rkllm_param.extend_param.base_domain_id = 0
self.handle = RKLLM_Handle_t()
self.rkllm_init = rkllm_lib.rkllm_init
self.rkllm_init.argtypes = [ctypes.POINTER(RKLLM_Handle_t), ctypes.POINTER(RKLLMParam), callback_type]
self.rkllm_init.restype = ctypes.c_int
self.rkllm_init(self.handle, self.rkllm_param, callback)
self.rkllm_run = rkllm_lib.rkllm_run
self.rkllm_run.argtypes = [RKLLM_Handle_t, ctypes.POINTER(RKLLMInput), ctypes.POINTER(RKLLMInferParam), ctypes.c_void_p]
self.rkllm_run.restype = ctypes.c_int
self.rkllm_abort = rkllm_lib.rkllm_abort
self.rkllm_abort.argtypes = [RKLLM_Handle_t]
self.rkllm_abort.restype = ctypes.c_int
self.rkllm_destroy = rkllm_lib.rkllm_destroy
self.rkllm_destroy.argtypes = [RKLLM_Handle_t]
self.rkllm_destroy.restype = ctypes.c_int
# Record the user's input prompt
def get_user_input(self, user_message, history):
history = history + [[user_message, None]]
return "", history
def tokens_to_ctypes_array(self, tokens, ctype):
# Converts a Python list to a ctypes array.
# The tokenizer outputs as a Python list.
return (ctype * len(tokens))(*tokens)
# Run inference
def run(self, prompt):
self.rkllm_infer_params = RKLLMInferParam()
ctypes.memset(ctypes.byref(self.rkllm_infer_params), 0, ctypes.sizeof(RKLLMInferParam))
self.rkllm_infer_params.mode = RKLLMInferMode.RKLLM_INFER_GENERATE
self.rkllm_input = RKLLMInput()
self.rkllm_input.input_mode = RKLLMInputMode.RKLLM_INPUT_TOKEN
self.rkllm_input.input_data.token_input.input_ids = self.tokens_to_ctypes_array(prompt, ctypes.c_int)
self.rkllm_input.input_data.token_input.n_tokens = ctypes.c_ulong(len(prompt))
self.rkllm_run(self.handle, ctypes.byref(self.rkllm_input), ctypes.byref(self.rkllm_infer_params), None)
return
# Release RKLLM object from memory
def release(self):
self.rkllm_abort(self.handle)
self.rkllm_destroy(self.handle)
# Retrieve the output from the RKLLM model and print it in a streaming manner
def get_RKLLM_output(self, message, history):
# Link global variables to retrieve the output information from the callback function
global global_text, global_state
global_text = []
global_state = -1
user_prompt = {"role": "user", "content": message}
history.append(user_prompt)
# Gemma 2 does not support system prompt.
if self.system_prompt == "":
prompt = [user_prompt]
else:
prompt = [
{"role": "system", "content": self.system_prompt},
user_prompt
]
# print(prompt)
TOKENIZER_PATH="%s/%s"%(MODEL_PATH,self.st_model_id.replace("/","-"))
if not os.path.exists(TOKENIZER_PATH):
print("Tokenizer not cached locally, downloading to %s"%TOKENIZER_PATH)
os.mkdir(TOKENIZER_PATH)
tokenizer = AutoTokenizer.from_pretrained(self.st_model_id, trust_remote_code=True)
tokenizer.save_pretrained(TOKENIZER_PATH)
else:
tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_PATH, trust_remote_code=True)
prompt = tokenizer.apply_chat_template(prompt, tokenize=True, add_generation_prompt=True)
# response = {"role": "assistant", "content": "Loading..."}
response = {"role": "assistant", "content": ""}
history.append(response)
model_thread = threading.Thread(target=self.run, args=(prompt,))
model_thread.start()
model_thread_finished = False
while not model_thread_finished:
while len(global_text) > 0:
response["content"] += global_text.pop(0)
# Marco-o1
response["content"] = str(response["content"]).replace("<Thought>", "\\<Thought\\>")
response["content"] = str(response["content"]).replace("</Thought>", "\\<\\/Thought\\>")
response["content"] = str(response["content"]).replace("<Output>", "\\<Output\\>")
response["content"] = str(response["content"]).replace("</Output>", "\\<\\/Output\\>")
time.sleep(0.005)
# Gradio automatically pushes the result returned by the yield statement when calling the then method
yield response
model_thread.join(timeout=0.005)
model_thread_finished = not model_thread.is_alive()

221
app/model_configs.py Normal file
View File

@@ -0,0 +1,221 @@
model_configs = {
"Qwen2.5-3B-Instruct-w8w8": {
"base_config": {
"st_model_id": "Qwen/Qwen2.5-14B-Instruct",
"max_context_len": 128000,
"max_new_tokens": 8192,
"top_k": 5,
"top_p": 0.8,
"temperature": 0.2,
"repeat_penalty": 1.00,
"frequency_penalty": 0.2,
"system_prompt": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."
},
"models": {
"Qwen2.5-3B-Instruct-w8w8": {"filename": "Qwen2.5-3B-Instruct-w8w8.rkllm"}
}
},
"Qwen2.5-Coder-3B-Instruct-w8w8": {
"base_config": {
"st_model_id": "Qwen/Qwen2.5-14B-Instruct",
"max_context_len": 128000,
"max_new_tokens": 8192,
"top_k": 5,
"top_p": 0.8,
"temperature": 0.2,
"repeat_penalty": 1.00,
"frequency_penalty": 0.2,
"system_prompt": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."
},
"models": {
"Qwen2.5-Coder-3B-Instruct-w8w8": {"filename": "Qwen2.5-Coder-3B-Instruct-w8w8.rkllm"}
}
}
# "LLaMA-Mesh": {
# "base_config": {
# "st_model_id": "c01zaut/LLaMA-Mesh-rk3588-1.1.1",
# "max_context_len": 4096,
# "max_new_tokens": 131072,
# "top_k": 1,
# "top_p": 0.9,
# "temperature": 0.6,
# "repeat_penalty": 1.0,
# "frequency_penalty": 0.2,
# "system_prompt": ""
# # "system_prompt": "You are a helpful assistant who renders 3d meshes in obj format. When prompted with an object to generate, respond back with a full description of the object's physical features that you will render. Think carefully about its shape, texture, and size, and then respond in obj format. Always generate a valid mesh that is rendered counter-clockwise. Always respond back with a description and mesh in obj format. If you are given an obj as input, assume that it was not rendered properly, and respond back with the item that it should depict. Then, generate the correct mesh so that it contains a proper, coherent object."
# # "system_prompt": "You are an expert graphic designer who renders 3d meshes in obj format. When prompted with an object to generate, describe the object's physical features that you will render, in the order that you render them in. Think carefully about the object's shape, and then respond in obj format. Always generate a valid and coherent mesh."
# # "system_prompt": "**System Prompt:**\n\n**Objective:**\nYou are a 3D mesh generation model that outputs a valid `.obj` format mesh based on a text description of an object. Your goal is to ensure the generated mesh is geometrically accurate, structurally valid, and aligns with the description provided.\n\n**Important Guidelines:**\n\n1. **Mesh Validity:**\n - **Proper format:** The output must be a valid `.obj` file that includes vertices (`v`), normals (`vn`), and faces (`f`) in the correct format.\n - **No errors in geometry:** Ensure there are no degenerate faces, duplicate vertices, or non-manifold geometry. Every face should reference existing vertices in a consistent and closed manner.\n\n2. **Accuracy in Representation:**\n - **Geometric fidelity:** The generated 3D object must closely match the description, including basic shape, proportions, and key features. If a description mentions a specific detail (e.g., \"a smooth curved surface\"), the model should represent this feature with appropriate vertex density and structure.\n - **Simple objects:** For simpler objects (e.g., cubes, spheres), ensure the geometry is clean and low-poly but still represents the object accurately.\n - **Complex objects:** For detailed descriptions (e.g., animals, vehicles, furniture), ensure the mesh has enough detail to represent the key features, but avoid overcomplicating the mesh with unnecessary vertices or faces.\n\n3. **Mesh Quality and Cleanliness:**\n - **No overlapping faces or vertices:** Make sure there are no duplicate vertices or redundant faces in the mesh.\n - **Closed mesh:** Ensure the mesh is closed (no gaps or missing faces) unless the description explicitly specifies an open structure (e.g., a hollow object).\n - **Normals:** Ensure that the normals are correctly defined to avoid lighting issues or rendering artifacts.\n\n4. **Smoothness and Curvature:**\n - For objects with curved surfaces (e.g., spheres, organic shapes), represent smooth curves by using an appropriate level of vertex density. Avoid sharp angles unless explicitly stated in the description.\n\n5. **Material and Texture:**\n - The model should focus on generating clean meshes; detailed textures can be omitted unless the description specifies them (e.g., \"a red apple with a smooth, shiny texture\"). If no specific textures are mentioned, assume default material properties (e.g., a basic color or a placeholder).\n\n6. **Handling Multiple Parts:**\n - For objects with multiple components (e.g., a car with wheels, a character with accessories), generate separate meshes for each component and ensure they are properly aligned or grouped together.\n\n---\n\n### Mesh Characteristics\n The mesh should be simple, with clear definitions for the body and wheels, ensuring no overlapping or incorrect face definitions.\n\n---\n\n**Reminder for the Model:**\nEnsure that the generated mesh is **structurally valid**, with no geometry errors. The description should be followed as closely as possible, focusing on the object’s shape and features. Aim for **clean** geometry, **accurate representation**, and **valid format** in `.obj`. Avoid creating overly complex meshes if the object does not require it."
# },
# "models": {
# "LLaMA-Mesh-w8a8-opt-hybrid": {"filename": "LLaMA-Mesh-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "LLaMA-Mesh-w8a8_g256-opt": {"filename": "models/LLaMA-Mesh-rk3588-w8a8_g256-opt-1-hybrid-ratio-0.0.rkllm"},
# "LLaMA-Mesh-w8a8-opt": {"filename": "LLaMA-Mesh-rk3588-w8a8-opt-1-hybrid-ratio-0.0.rkllm"},
# "LLaMA-Mesh-w8a8_g512-hybrid": {"filename": "LLaMA-Mesh-rk3588-w8a8_g512-opt-0-hybrid-ratio-1.0.rkllm"},
# "LLaMA-Mesh-w8a8_g128-opt-hybrid-0.5": {"filename": "LLaMA-Mesh-rk3588-w8a8_g128-opt-1-hybrid-ratio-0.5.rkllm"},
# "LLaMA-Mesh-w8a8_g512": {"filename": "LLaMA-Mesh-rk3588-w8a8_g512-opt-0-hybrid-ratio-0.0.rkllm"},
# "LLaMA-Mesh-w8a8_g512-hybrid-0.5": {"filename": "LLaMA-Mesh-rk3588-w8a8_g512-opt-0-hybrid-ratio-0.5.rkllm"},
# "LLaMA-Mesh-w8a8_g512-opt-1.1.3": {"filename": "LLaMA-Mesh-rk3588-w8a8_g512-opt-1-hybrid-ratio-0.0.rkllm"}
# }
# },
# "Phi-3.5-Mini-Instruct": {
# "base_config": {
# "st_model_id": "microsoft/Phi-3.5-mini-instruct",
# "max_context_len": 4096,
# "max_new_tokens": 4096,
# "top_k": 1,
# "top_p": 0.9,
# "temperature": 0.7,
# "repeat_penalty": 1.1,
# "frequency_penalty": 1.0,
# "system_prompt": "You are Phi 3.5 Mini, an artificial intelligence model trained by Microsoft. You are a helpful AI assistant."
# },
# "models": {
# "Phi-3.5-Mini-Instruct": {"filename": "Phi-3.5-mini-instruct-rk3588-w8a8-opt-0-hybrid-ratio-0.0.rkllm"},
# "Phi-3.5-Mini-Instruct-w8a8_g256-opt": {"filename": "Phi-3.5-mini-instruct-rk3588-w8a8_g256-opt-1-hybrid-ratio-0.5.rkllm"},
# "Phi-3.5-Mini-Instruct-w8a8_g512-opt": {"filename": "Phi-3.5-mini-instruct-rk3588-w8a8_g512-opt-1-hybrid-ratio-0.5.rkllm"}
# }
# },
# "Phi-3-Mini-Instruct": {
# "base_config": {
# "st_model_id": "c01zaut/Phi-3-mini-128k-instruct-rk3588-1.1.2",
# "max_context_len": 4096,
# "max_new_tokens": 4096,
# "top_k": 1,
# "top_p": 0.8,
# "temperature": 0.7,
# "repeat_penalty": 1.1,
# "frequency_penalty": 0.9,
# "system_prompt": "You are Phi 3.5 Mini, an artificial intelligence model trained by Microsoft. You are a helpful AI assistant."
# },
# "models": {
# "Phi-3-mini-128k-instruct-w8a8": {"filename": "Phi-3-mini-128k-instruct-rk3588-w8a8-opt-0-hybrid-ratio-0.0.rkllm"},
# "Phi-3-mini-128k-instruct-w8a8-opt": {"filename": "Phi-3-mini-128k-instruct-rk3588-w8a8-opt-1-hybrid-ratio-0.0.rkllm"},
# "Phi-3-mini-128k-instruct-w8a8-opt-hybrid-0.5": {"filename": "Phi-3-mini-128k-instruct-rk3588-w8a8-opt-1-hybrid-ratio-0.5.rkllm"}
# }
# },
# "Qwen-2.5-Instruct": {
# "base_config": {
# "st_model_id": "Qwen/Qwen2.5-14B-Instruct",
# "max_context_len": 4096,
# "max_new_tokens": 8192,
# "top_k": 5,
# "top_p": 0.8,
# "temperature": 0.2,
# "repeat_penalty": 1.00,
# "frequency_penalty": 0.2,
# "system_prompt": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."
# },
# "models": {
# "Qwen2.5-1.5B-Instruct": {"filename": "Qwen2.5-1.5B-Instruct-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "Qwen2.5-3B-Instruct": {"filename": "Qwen2.5-3B-Instruct-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "Qwen2.5-7B-Instruct": {"filename": "Qwen2.5-7B-Instruct-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "Qwen2.5-Coder-7B-Instruct": {"filename": "Qwen2.5-Coder-7B-Instruct-rk3588-w8a8_g128-opt-1-hybrid-ratio-0.5.rkllm"},
# "Qwen2.5-Coder-1.5B-Instruct-w8a8-hybrid": {"filename": "Qwen2.5-Coder-1.5B-Instruct-rk3588-w8a8-opt-0-hybrid-ratio-1.0.rkllm"}
# },
# },
# "Marco-O1": {
# "base_config": {
# "st_model_id": "AIDC-AI/Marco-o1",
# "max_context_len": 4096,
# "max_new_tokens": 8192,
# "top_k": 1,
# "top_p": 0.8,
# "temperature": 0.9,
# "repeat_penalty": 1.00,
# "frequency_penalty": 0.8,
# "system_prompt": "You are a well-trained AI assistant, your name is Marco-o1. Created by AI Business of Alibaba International Digital Business Group.\n\n## IMPORTANT!!!!!!\nWhen you answer questions, your thinking should be done in <Thought>, and your results should be output in <Output>.\n<Thought> should be in English as much as possible, but there are 2 exceptions, one is the reference to the original text, and the other is that mathematics should use markdown format, and the output in <Output> needs to follow the language of the user input."
# },
# "models": {
# "Marco-o1": {"filename": "Marco-o1-rk3588-w8a8_g512-opt-0-hybrid-ratio-0.5.rkllm"}
# }
# },
# "Gemma-2-LongCoT": {
# "base_config": {
# "st_model_id": "c01zaut/OpenLongCoT-Base-Gemma2-2B-rk3588-1.1.1",
# "max_context_len": 4096,
# "max_new_tokens": 4096,
# "top_k": 1,
# "top_p": 0.8,
# "temperature": 0.1,
# "repeat_penalty": 1.05,
# "frequency_penalty": 0.5,
# "system_prompt": ""
# },
# "models": {
# "OpenLongCoT-Base-Gemma2-2B-rk3588-1.1.1": {"filename": "OpenLongCoT-Base-Gemma2-2B-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"}
# }
# },
# "Gemma-2-IT-Google": {
# "base_config": {
# "st_model_id": "c01zaut/gemma-2-9b-it-rk3588-1.1.2",
# "max_context_len": 4096,
# "max_new_tokens": 8192,
# "top_k": 15,
# "top_p": 0.95,
# "temperature": 0.7,
# "repeat_penalty": 1.05,
# "frequency_penalty": 0.5,
# "system_prompt": ""
# },
# "models": {
# "gemma-2-2b-it-opt": {"filename": "gemma-2-2b-it-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "gemma-2-9b-it-opt": {"filename": "gemma-2-9b-it-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "codegemma-7b-it-opt": {"filename": "codegemma-7b-it-rk3588-w8a8-opt-1-hybrid-ratio-0.0.rkllm"},
# "codegemma-7b-it-opt-hybrid": {"filename": "codegemma-7b-it-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"}
# }
# },
# "InternLM2_5": {
# "base_config": {
# "st_model_id": "internlm/internlm2_5-7b-chat",
# "max_context_len": 4096,
# "max_new_tokens": 8192,
# "top_k": 30,
# "top_p": 0.8,
# "temperature": 0.5,
# "repeat_penalty": 1.0005,
# "frequency_penalty": 0.2,
# "system_prompt": "You are InternLM (书生·浦语), a helpful, honest, and harmless AI assistant developed by Shanghai AI Laboratory (上海人工智能实验室)."
# },
# "models": {
# "internlm2_5-1_8b-chat-w8a8_g512-opt": {"filename": "internlm2_5-1_8b-chat-rk3588-w8a8_g512-opt-1-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-1_8b-chat-w8a8": {"filename": "internlm2_5-1_8b-chat-rk3588-w8a8-opt-0-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-1_8b-chat-w8a8-opt": {"filename": "internlm2_5-1_8b-chat-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-20b-chat-w8a8-opt": {"filename": "internlm2_5-20b-chat-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-20b-chat-w8a8_g512-opt": {"filename": "internlm2_5-20b-chat-rk3588-w8a8_g512-opt-1-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-20b-chat-w8a8_g512": {"filename": "internlm2_5-20b-chat-rk3588-w8a8_g512-opt-0-hybrid-ratio-1.0.rkllm"},
# "internlm2_5-7b-chat-w8a8-opt": {"filename": "internlm2_5-7b-chat-rk3588-w8a8-opt-1-hybrid-ratio-1.0.rkllm"}
# }
# },
# "Deepseek-LLM": {
# "base_config": {
# "st_model_id": "c01zaut/deepseek-llm-7b-chat-rk3588-1.1.1",
# "max_context_len": 4096,
# "max_new_tokens": 8192,
# "top_k": 1,
# "top_p": 0.9,
# "temperature": 0.7,
# "repeat_penalty": 1.2,
# "frequency_penalty": 0.8,
# "system_prompt": "You are DeepSeek Chat, a helpful, respectful and honest AI assistant developed by DeepSeek. The knowledge cut-off date for your training data is up to May 2023. Always answer as helpfully as possible, while being safe. Your answers should not include any harmful, unethical, racist, sexist, toxic, dangerous, or illegal content. Please ensure that your responses are socially unbiased and positive in nature. If a question does not make any sense, or is not factually coherent, explain why instead of answering something not correct. If you don’t know the answer to a question, please don’t share false information."
# },
# "models": {
# "deepseek-llm-7b": {"filename": "deepseek-llm-7b-chat-rk3588-w8a8_g256-opt-1-hybrid-ratio-0.5.rkllm"}
# }
# },
# "ChatGLM3-6B": {
# "base_config": {
# "st_model_id": "c01zaut/chatglm3-6b-rk3588-1.1.2",
# "max_context_len": 4096,
# "max_new_tokens": 4096,
# "top_k": 5,
# "top_p": 0.7,
# "temperature": 0.8,
# "repeat_penalty": 1.2,
# "frequency_penalty": 0.8,
# "system_prompt": "You are ChatGLM3, a large language model trained by Zhipu.AI. Follow the user's instructions carefully. Respond using markdown."
# },
# "models": {
# "ChatGLM3-6B": {"filename": "chatglm3-6b-rk3588-w8a8-opt-0-hybrid-ratio-0.0.rkllm"}
# }
# }
}

View File

@@ -0,0 +1,95 @@
import sys
import resource
import gradio as gr
from ctypes_bindings import *
from model_class import *
from mesh_utils import *
# Set environment variables
os.environ["GRADIO_SERVER_NAME"] = "0.0.0.0"
os.environ["GRADIO_SERVER_PORT"] = "8080"
os.environ["RKLLM_LOG_LEVEL"] = "1"
# Set resource limit
resource.setrlimit(resource.RLIMIT_NOFILE, (102400, 102400))
history = []
if __name__ == "__main__":
# Helper function to define initializing model before class is declared
# Without this, you would need to initialize the class before you select the model
def initialize_model(model):
global rkllm_model
# Have to unload previous model in single-threaded mode
try:
rkllm_model.release()
except:
print("No model loaded! Continuing with initialization...")
# Initialize RKLLM model
init_msg = "=========INITIALIZING==========="
print(init_msg)
sys.stdout.flush()
rkllm_model = RKLLMLoaderClass(model=model)
model_init = f"RKLLM Model, {rkllm_model.model_name} has been initialized successfully!"
print(model_init)
complete_init = "=============================="
print(complete_init)
output = [[f"<h4 style=\"text-align:center;\">{model_init}\n</h4>", None]]
sys.stdout.flush()
return output
# Helper function to stream LLM output into the chat box
def get_RKLLM_output(message, history):
try:
yield from rkllm_model.get_RKLLM_output(message, history)
except RuntimeError as e:
print(f"ERROR: {e}")
return history
# Create a Gradio interface
with gr.Blocks(title="Chat with RKLLM") as chatRKLLM:
available_models = available_models()
gr.Markdown("<div align='center'><font size='10'> Definitely Not Skynet </font></div>")
with gr.Tabs():
with gr.TabItem("Select Model"):
model_dropdown = gr.Dropdown(choices=available_models, label="Select Model", value="None", allow_custom_value=True)
statusBox = gr.Chatbot(height=100)
model_dropdown.input(initialize_model, [model_dropdown], [statusBox])
with gr.TabItem("Txt2Txt"):
txt2txt = gr.ChatInterface(fn=get_RKLLM_output, type="messages")
txt2txt.chatbot.height = "70vh"
txt2txt.chatbot.resizeable = True
with gr.TabItem("Txt2Mesh"):
with gr.Row():
with gr.Column(scale=2):
txt2txt = gr.ChatInterface(fn=get_RKLLM_output, type="messages")
txt2txt.chatbot.height = "70vh"
txt2txt.chatbot.resizeable = True
with gr.Column(scale=2):
# Add the text box for 3D mesh input and button
mesh_input = gr.Textbox(
label="3D Mesh Input",
placeholder="Paste your 3D mesh in OBJ format here...",
lines=5,
)
visualize_button = gr.Button("Visualize 3D Mesh")
output_model = gr.Model3D(
label="3D Mesh Visualization",
interactive=False,
)
# Link the button to the visualization function
visualize_button.click(
fn=apply_gradient_color,
inputs=[mesh_input],
outputs=[output_model]
)
print("\nNo model loaded yet!\n")
# Enable the event queue system.
chatRKLLM.queue()
# Start the Gradio application.
chatRKLLM.launch()
print("====================")
print("RKLLM model inference completed, releasing RKLLM model resources...")
rkllm_model.release()
print("====================")

27
docker-compose.yaml Executable file
View File

@@ -0,0 +1,27 @@
services:
rkllm:
platform: linux/arm64/v8
container_name: rkllm
image: python:3.11.11
restart: always
privileged: true
volumes:
- ./app:/app
- ./cache/cache:/root/.cache
- ./cache/site-packages:/usr/local/lib/python3.11/site-packages
- ./offline_packages:/offline_packages
- ./ezrknn-llm:/ezrknn-llm
entrypoint: ./app/entrypoint.sh
ports:
- "8080:8080"
networks:
- apps
networks:
apps:
name: apps
driver: bridge
ipam:
driver: default
config:
- subnet: "172.100.0.0/24"
gateway: "172.100.0.1"

View File

@@ -0,0 +1,8 @@
gradio==5.14.0
gradio_client==1.7.0
huggingface-hub==0.26.2
Jinja2==3.1.4
numpy==2.1.3
transformers==4.46.2
trimesh==4.5.2
sentencepiece==0.2.0