202 lines
7.2 KiB
Python
202 lines
7.2 KiB
Python
"""
|
|
|
|
AUTHOR:
|
|
|
|
Khushal P Soonderji
|
|
|
|
DATE:
|
|
|
|
Thursday, 5th Dec., 2024
|
|
|
|
OBJECTIVE:
|
|
|
|
To use LLMs to perform activities like chat completion, text summarization, etc.
|
|
|
|
REFERENCES:
|
|
|
|
N/A
|
|
|
|
DOWNLOADS:
|
|
|
|
N/A
|
|
|
|
NOTES:
|
|
|
|
N/A
|
|
|
|
"""
|
|
|
|
|
|
# *****************************************************************************************************************
|
|
# ***** ****
|
|
# *** IMPORT ***
|
|
# ***** ****
|
|
# *****************************************************************************************************************
|
|
|
|
|
|
# To make sibling directories accessible for imports:
|
|
import sys
|
|
sys.path.append(".")
|
|
sys.path.append("..")
|
|
|
|
# For using Quart:
|
|
from quart import Blueprint, current_app, request
|
|
|
|
# My utils:
|
|
from utils_v2.string import json
|
|
from utils_v2.api.codes import StatusCodes, HttpCodes
|
|
from utils_v2.api.response import ResponseModel
|
|
from utils_v2.api.async_quart import (
|
|
set_api_version,
|
|
read_input,
|
|
get_session_info,
|
|
log_request_to_mongo,
|
|
log_chain_to_mongo,
|
|
should_not_be_under_maintenance,
|
|
only_whitelisted_ips,
|
|
limit_rate,
|
|
validate_input,
|
|
handle_cancelled_request
|
|
)
|
|
|
|
# Common:
|
|
from shared import constants
|
|
|
|
# Data Models:
|
|
from models.data.api.ai.llm import LLMRequestHeaders, LLMInput
|
|
from models.data.core.user_info import CoreUserInfoModel
|
|
|
|
# For asynchronous activities:
|
|
import asyncio
|
|
|
|
|
|
# *****************************************************************************************************************
|
|
# ***** ****
|
|
# *** MACROS / ONE-TIME INIT ***
|
|
# ***** ****
|
|
# *****************************************************************************************************************
|
|
|
|
|
|
# Related to Quart:
|
|
llm_invoke_bp = Blueprint("llm_invoke", __name__)
|
|
|
|
|
|
# *****************************************************************************************************************
|
|
# ***** ****
|
|
# *** VARIABLES ***
|
|
# ***** ****
|
|
# *****************************************************************************************************************
|
|
|
|
|
|
# --- Nothing Yet
|
|
|
|
|
|
# *****************************************************************************************************************
|
|
# ***** ****
|
|
# *** FUNCTIONS ***
|
|
# ***** ****
|
|
# *****************************************************************************************************************
|
|
|
|
|
|
@llm_invoke_bp.record_once
|
|
def init(blueprint_setup_state):
|
|
|
|
# This gets called when the blueprint is registered.
|
|
# Consider this to be a one-time setup for the whole blueprint:
|
|
pass
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------------------------
|
|
|
|
|
|
@llm_invoke_bp.route("/llm/invoke", methods = ["POST"])
|
|
@set_api_version(api_version = "1.0.0")
|
|
@read_input(sanitize_headers = False, sanitize_data = False)
|
|
@get_session_info(key = "X-Session-Token", session_coro = "get_session")
|
|
@log_request_to_mongo(
|
|
attr_name = "logs_mongo",
|
|
project = constants.PROJECT_NAME,
|
|
log_type = constants.MODULE_NAME,
|
|
operation = "llmInvokeApi",
|
|
log_input = True,
|
|
log_output = True,
|
|
sensitive_keys = ["sessionToken", "X-Session-Token"]
|
|
)
|
|
@log_chain_to_mongo(attr_name = "logs_mongo")
|
|
@should_not_be_under_maintenance(attr_name = "is_under_maintenance")
|
|
@validate_input(
|
|
header_validator = lambda x: LLMRequestHeaders(**x).model_dump(),
|
|
data_validator = lambda x: LLMInput(**x)
|
|
)
|
|
@handle_cancelled_request()
|
|
async def invoke_llm(
|
|
inbound_headers: dict | LLMRequestHeaders = None,
|
|
inbound_data: dict | LLMInput = None,
|
|
inbound_files: dict = None,
|
|
**kwargs
|
|
):
|
|
|
|
"""
|
|
Use this to invoke an LLM for text completion kind of activities.
|
|
:param inbound_headers: auto-extracted by the decorators.
|
|
:param inbound_data: auto-extracted by the decorators.
|
|
:param inbound_files: auto-extracted by the decorators.
|
|
:param kwargs: Any number of extra inputs supplied by the decorators.
|
|
:return: A standard response structure.
|
|
"""
|
|
|
|
# ┏┓
|
|
# ┃┃┏┓┏┓┏┓┏┓┏┓┏┏┓┏┏
|
|
# ┣┛┛ ┗ ┣┛┛ ┗┛┗┗ ┛┛
|
|
# ┛
|
|
|
|
# If the session token is invalid/expired:
|
|
if kwargs.get("session_info") is None:
|
|
return ResponseModel(
|
|
status_code = StatusCodes.FAILED,
|
|
http_code = HttpCodes.UNAUTHORIZED
|
|
)
|
|
|
|
# ┳ ┓
|
|
# ┃┏┓┓┏┏┓┃┏┏┓
|
|
# ┻┛┗┗┛┗┛┛┗┗
|
|
|
|
# Call the LLM and see if its service worked or not:
|
|
llm_response = await current_app.llm.invoke(
|
|
mongo_conn = current_app.data_mongo,
|
|
user_info = CoreUserInfoModel(**kwargs["session_info"]),
|
|
llm_input = inbound_data
|
|
)
|
|
success = False if llm_response.output is None else True
|
|
|
|
# ┳┓
|
|
# ┣┫┏┓┏┏┓┏┓┏┓┏┏┓
|
|
# ┛┗┗ ┛┣┛┗┛┛┗┛┗
|
|
# ┛
|
|
|
|
# Done here:
|
|
return ResponseModel(
|
|
status_code = StatusCodes.OK if success else StatusCodes.FAILED,
|
|
http_code = HttpCodes.SUCCESS if success else HttpCodes.INTERNAL_SERVER_ERROR,
|
|
data = {
|
|
"ts": llm_response.ts.isoformat(),
|
|
"client": llm_response.client,
|
|
"model": llm_response.model,
|
|
"output": llm_response.output,
|
|
"tokens": llm_response.tokens.model_dump(),
|
|
"invocationId": llm_response.invocationId
|
|
} if success else None
|
|
)
|
|
|
|
|
|
# *****************************************************************************************************************
|
|
# ***** ****
|
|
# *** MAIN PROGRAM ***
|
|
# ***** ****
|
|
# *****************************************************************************************************************
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
pass
|