Merge pull request 'v0.7.4-add_vector_search' (#32) from v0.7.4-add_vector_search into main
Reviewed-on: #32
This commit was merged in pull request #32.
This commit is contained in:
+5
-5
@@ -2,7 +2,7 @@
|
||||
|
||||
julia_version = "1.12.6"
|
||||
manifest_format = "2.0"
|
||||
project_hash = "dc7878808bbc4637a12e709dd495979a784824a5"
|
||||
project_hash = "1c1379a2cec320abc347f3acb5ee815ba9855aa6"
|
||||
|
||||
[[deps.Accessors]]
|
||||
deps = ["CompositionsBase", "ConstructionBase", "Dates", "InverseFunctions", "MacroTools"]
|
||||
@@ -300,11 +300,11 @@ version = "1.1.0"
|
||||
|
||||
[[deps.GeneralUtils]]
|
||||
deps = ["CSV", "DataFrames", "DataStructures", "Dates", "Distributions", "Graphs", "HTTP", "JSON", "LibPQ", "NATS", "PrettyPrinting", "Random", "Revise", "SHA", "StringDistances", "UUIDs"]
|
||||
git-tree-sha1 = "a75a088ee8e5faf10f554ca00748e0e6ca58d1ca"
|
||||
git-tree-sha1 = "93293126d24d3929ef6a5067f347bc28c6582c71"
|
||||
repo-rev = "main"
|
||||
repo-url = "https://git.yiem.cc/ton/GeneralUtils"
|
||||
uuid = "c6c72f09-b708-4ac8-ac7c-2084d70108fe"
|
||||
version = "0.5.1"
|
||||
version = "0.5.10"
|
||||
|
||||
[[deps.Graphs]]
|
||||
deps = ["ArnoldiMethod", "DataStructures", "Inflate", "LinearAlgebra", "Random", "SimpleTraits", "SparseArrays", "Statistics"]
|
||||
@@ -1089,10 +1089,10 @@ uuid = "ddb6d928-2868-570f-bddf-ab3f9cf99eb6"
|
||||
version = "0.4.16"
|
||||
|
||||
[[deps.YiemAgent]]
|
||||
deps = ["Base64", "CSV", "DataFrames", "DataStructures", "Dates", "GeneralUtils", "HTTP", "JSON", "LLMMCTS", "LibPQ", "NATS", "PrettyPrinting", "Random", "Revise", "SQLLLM", "Serialization", "URIs", "UUIDs"]
|
||||
deps = ["Base64", "CSV", "DataFrames", "DataStructures", "Dates", "HTTP", "JSON", "LLMMCTS", "LibPQ", "NATS", "PrettyPrinting", "Random", "Revise", "SQLLLM", "Serde", "Serialization", "URIs", "UUIDs"]
|
||||
path = "."
|
||||
uuid = "e012c34b-7f78-48e0-971c-7abb83b6f0a2"
|
||||
version = "0.7.2"
|
||||
version = "0.7.4"
|
||||
|
||||
[[deps.Zlib_jll]]
|
||||
deps = ["Libdl"]
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
name = "YiemAgent"
|
||||
uuid = "e012c34b-7f78-48e0-971c-7abb83b6f0a2"
|
||||
version = "0.7.2"
|
||||
version = "0.7.4"
|
||||
authors = ["narawat lamaiin <narawat@outlook.com>"]
|
||||
|
||||
[deps]
|
||||
@@ -28,7 +28,7 @@ UUIDs = "cf7118a7-6976-5b1a-9a39-7adc72f591a4"
|
||||
Base64 = "1.11.0"
|
||||
CSV = "0.10.15"
|
||||
DataFrames = "1.7.0"
|
||||
GeneralUtils = "0.5.1"
|
||||
GeneralUtils = "0.5.10"
|
||||
HTTP = "2.4.0"
|
||||
JSON = "1.6.1"
|
||||
LLMMCTS = "0.1.5"
|
||||
|
||||
@@ -1,111 +1,10 @@
|
||||
using HTTP, JSON, URIs, Random, PrettyPrinting, UUIDs, Dates, DataFrames, DataStructures
|
||||
using GeneralUtils, SQLLLM, YiemAgent
|
||||
|
||||
config = JSON.parsefile("./appconfig.json")
|
||||
host_url, _port = split(config["externalservice"]["sommpanion_db"]["url"], ':')
|
||||
port = parse(Int, _port)
|
||||
dbname = "winedb"
|
||||
user = config["externalservice"]["sommpanion_db"]["user"]
|
||||
password = config["externalservice"]["sommpanion_db"]["password"]
|
||||
pg_conn_str = "host=$host_url port=$port dbname=$dbname user=$user password=$password"
|
||||
|
||||
function execute_sql_winedb(sql::T) where {T<:AbstractString}
|
||||
host_url, _port = split(config["externalservice"]["sommpanion_db"]["url"], ':')
|
||||
port = parse(Int, _port)
|
||||
dbname = "winedb"
|
||||
user = config["externalservice"]["sommpanion_db"]["user"]
|
||||
password = config["externalservice"]["sommpanion_db"]["password"]
|
||||
db_connection = LibPQ.Connection("host=$host_url port=$port dbname=$dbname user=$user password=$password")
|
||||
result = nothing
|
||||
try
|
||||
result = LibPQ.execute(db_connection, sql)
|
||||
catch e
|
||||
LibPQ.close(db_connection)
|
||||
end
|
||||
|
||||
LibPQ.close(db_connection)
|
||||
return result
|
||||
end
|
||||
|
||||
|
||||
|
||||
sql =
|
||||
"""
|
||||
SELECT T1.winery, T1.wine_name, T1.wine_id, T1.vintage, T1.region, T1.country, T1.wine_type, T1.grape, T1.serving_temperature, T1.sweetness, T1.intensity, T1.tannin, T1.acidity, T1.tasting_notes, T2.price, T2.currency, T1.image_url, T3.retailer_name, T3.retailer_id FROM "wine" AS T1 JOIN "retailer_wine" AS T2 ON T1.wine_id = T2.wine_id JOIN "retailer" AS T3 ON T2.retailer_id = T3.retailer_id WHERE T1.wine_name = 'Montrachet Grand Cru' AND T1.winery = 'Domaine Jacques Prieur' AND T3.retailer_name = 'Yiem Wines Ltd' AND T3.retailer_id = 'f54eab6b-7650-4448-b009-c53f3efbcc3b';
|
||||
"""
|
||||
|
||||
textresult, sql_result_raw, _, _ = YiemAgent.SQLexecution(execute_sql_winedb, sql)
|
||||
result_vec = GeneralUtils.dfToVectorDict(sql_result_raw)
|
||||
|
||||
for d in result_vec
|
||||
wine_name = d["wine_name"]
|
||||
image_url_json_str = d["image_url"]
|
||||
image_url_json_obj = JSON.parse(image_url_json)
|
||||
base_url = "http://192.168.88.106:8080/"
|
||||
image_base64 =
|
||||
if haskey(image_url_json_obj, "bottle")
|
||||
url = base_url * image_url_json_obj["bottle"]
|
||||
image_data = HTTP.get(url) # vector{int} data
|
||||
image_base64_string = base64encode(image_data)
|
||||
else
|
||||
nothing
|
||||
end
|
||||
d["image"] = image_base64
|
||||
end
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
using LibPQ
|
||||
using Tables
|
||||
|
||||
"""
|
||||
update_car_regions_one_by_one(conn::LibPQ.Connection, target_word::String)
|
||||
|
||||
Iterates through all rows in the 'car' table where the region is "German",
|
||||
and updates them one-by-one to the `target_word`.
|
||||
"""
|
||||
function update_car_regions_one_by_one(pg_conn_str::String, replace_word::String , target_word::String)
|
||||
conn = LibPQ.Connection(pg_conn_str)
|
||||
# 1. Fetch the target rows. Assumes 'id' is the primary key.
|
||||
# We select the ID to target rows individually during the update step.
|
||||
select_query = "SELECT id FROM car WHERE region = '$replace_word';"
|
||||
|
||||
result = execute(conn, select_query)
|
||||
rows = Tables.rows(result)
|
||||
|
||||
# 2. Prepare the update statement for execution reuse
|
||||
# Using explicit types for parameter placeholders ($1, $2)
|
||||
update_query = "UPDATE car SET region = \$1 WHERE id = \$2;"
|
||||
|
||||
println("Starting one-by-one update...")
|
||||
updated_count = 0
|
||||
|
||||
# 3. Iterate through rows one-by-one
|
||||
for row in rows
|
||||
# LibPQ row values are accessed via properties or column names
|
||||
row_id = row.id
|
||||
|
||||
# Execute the parameterized statement safely
|
||||
execute(conn, update_query, [target_word, row_id])
|
||||
updated_count += 1
|
||||
end
|
||||
|
||||
println("Successfully updated \$updated_count rows.")
|
||||
return updated_count
|
||||
end
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
# check if this column has vector embedding. if there is one, seach vector version instead
|
||||
column_name_embedding = column_name * "_embedding"
|
||||
if occursin(column_name_embedding, tables_schema[column_name_embedding])
|
||||
vector_column = Dict(
|
||||
"table_name"=> table_name,
|
||||
"column_name"=> column_name_embedding,
|
||||
"operator"=> "vector_similarity",
|
||||
"value"=> column_obj["value"]
|
||||
)
|
||||
end
|
||||
+86
-36
@@ -4,7 +4,7 @@ export addNewMessage, conversation, decisionMaker, reflector, generatechat,
|
||||
generalconversation, detectWineryName, generateSituationReport
|
||||
|
||||
using JSON, DataStructures, Dates, UUIDs, HTTP, Random, PrettyPrinting, Serialization,
|
||||
DataFrames, CSV
|
||||
DataFrames, Serde
|
||||
using GeneralUtils
|
||||
using ..type, ..util, ..llmfunction
|
||||
|
||||
@@ -122,53 +122,103 @@ function decisionMaker(a::T; recentevents::Integer=20, maxattempt=3
|
||||
errornote = "N/A"
|
||||
response = nothing # placeholder for show when error msg show up
|
||||
|
||||
for attempt in 1:maxattempt
|
||||
"""
|
||||
{
|
||||
"model": "your-model.gguf",
|
||||
"messages": [ ... ],
|
||||
"response_format": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "agent_action",
|
||||
"strict": true,
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"think": {
|
||||
"type": "string",
|
||||
"description": "Your step-by-step reasoning process. Explain why you are choosing this action."
|
||||
},
|
||||
"action_name": {
|
||||
"type": "string",
|
||||
"enum": ["search_web", "get_weather", "calculate_math"],
|
||||
"description": "The exact name of the tool to execute."
|
||||
},
|
||||
"action_input": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": { "type": ["string", "null"], "description": "For search_web" },
|
||||
"location": { "type": ["string", "null"], "description": "For get_weather" },
|
||||
"equation": { "type": ["string", "null"], "description": "For calculate_math" }
|
||||
},
|
||||
"required": ["query", "location", "equation"],
|
||||
"additionalProperties": false
|
||||
}
|
||||
},
|
||||
"required": ["think", "action_name", "action_input"],
|
||||
"additionalProperties": false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
msg = Dict(
|
||||
"model" => "gemma-4-E4B-it-UD-Q4_K_XL",
|
||||
"messages" => a.chathistory,
|
||||
"temperature" => 0.7
|
||||
# strict output format
|
||||
response_format = Dict(
|
||||
"type"=> "json_schema",
|
||||
"json_schema"=> Dict(
|
||||
"name"=> "user_profile",
|
||||
"strict"=> true,
|
||||
"schema"=> Dict(
|
||||
"type"=> "object",
|
||||
"properties"=> Dict(
|
||||
"plan"=> Dict("type"=> "string"),
|
||||
"action_name"=> Dict("type"=> "string"),
|
||||
"action_input"=> Dict("type"=> "string"),
|
||||
),
|
||||
"required"=> ["plan", "action_name", "action_input"],
|
||||
"additionalProperties"=> false
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
response = a.context.text2textInstructLLM(a.id, msg)
|
||||
msg = Dict(
|
||||
"model"=> "gemma-4-E4B-it-UD-Q4_K_XL",
|
||||
"messages"=> a.chathistory,
|
||||
"temperature"=> 0.7,
|
||||
"response_format"=> response_format,
|
||||
)
|
||||
|
||||
response = GeneralUtils.clean_json_response(response)
|
||||
for attempt in 1:maxattempt
|
||||
response = a.context.text2textInstructLLM(a.id, msg)
|
||||
response = GeneralUtils.remove_french_accents(response)
|
||||
think, response = GeneralUtils.extractthink(response)
|
||||
response = String(split(response, ", observation")[1]) # in case LLM generate observation key which it isn't supposed to
|
||||
response = strip(response)
|
||||
# think, response = GeneralUtils.extractthink(response)
|
||||
|
||||
# dollar sign in Julia means string interpolation
|
||||
while occursin('$', response)
|
||||
response = replace(response, '$' => "USD")
|
||||
end
|
||||
|
||||
responsedict = nothing
|
||||
if occursin(requiredKeys[2], response)
|
||||
try
|
||||
_responsedict = JSON.parse(response)
|
||||
responsedict = GeneralUtils.dictify(_responsedict; keytype=String, sort_order=requiredKeys)
|
||||
catch
|
||||
println("\nERROR YiemAgent decisionMaker() failed to parse response: $response ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
continue
|
||||
end
|
||||
else
|
||||
println("\nERROR YiemAgent decisionMaker() $errornote --(not qualify response)-> $responsedict ", @__FILE__, ":", @__LINE__, " $(Dates.now())\n")
|
||||
continue
|
||||
end
|
||||
# responsedict = nothing
|
||||
# try
|
||||
# responsedict = Serde.parse_yaml(response)
|
||||
# catch e
|
||||
# println("\nERROR YiemAgent decisionMaker() Error: $e --(not qualify response)-> $response ", @__FILE__, ":", @__LINE__, " $(Dates.now())\n")
|
||||
# continue
|
||||
# end
|
||||
|
||||
# check whether all answer's key points are in responsedict
|
||||
println("\n---")
|
||||
println(responsedict)
|
||||
println("---\n")
|
||||
ispass, errormsg = GeneralUtils.checkAgentResponse_JSON(responsedict, requiredKeys)
|
||||
# # check whether all answer's key points are in responsedict
|
||||
# println("\n---")
|
||||
# println(responsedict)
|
||||
# println("---\n")
|
||||
# ispass, errormsg = GeneralUtils.checkAgentResponse_JSON(responsedict, requiredKeys)
|
||||
|
||||
if !ispass
|
||||
errornote = errormsg
|
||||
println("\nERROR YiemAgent decisionMaker() $errornote --(not qualify response)-> $responsedict ", @__FILE__, ":", @__LINE__, " $(Dates.now())\n")
|
||||
continue
|
||||
end
|
||||
# if !ispass
|
||||
# errornote = errormsg
|
||||
# println("\nERROR YiemAgent decisionMaker() $errornote --(not qualify response)-> $responsedict ", @__FILE__, ":", @__LINE__, " $(Dates.now())\n")
|
||||
# continue
|
||||
# end
|
||||
|
||||
responsedict = JSON.parse(response)
|
||||
|
||||
if responsedict["action_input"] == "CHAT_BOX" &&
|
||||
occursin("similar", responsedict["action_input"])
|
||||
@@ -576,7 +626,7 @@ function think(a::T)::NamedTuple{(:thoughtdict, :result_raw), Tuple{OrderedDict,
|
||||
end
|
||||
|
||||
@info "YiemAgent think() end " @__LINE__
|
||||
@show thoughtdict
|
||||
# @show thoughtdict
|
||||
println("---\n")
|
||||
return (thoughtdict=thoughtdict, result_raw=result_raw)
|
||||
end
|
||||
|
||||
+472
-339
@@ -5,7 +5,7 @@ export virtualWineUserChatbox, jsoncorrection, search_wine_database!, # recomme
|
||||
extractWineAttributes_2, paraphrase, SQLexecution
|
||||
|
||||
using HTTP, JSON, URIs, Random, PrettyPrinting, UUIDs, Dates, DataFrames, DataStructures,
|
||||
Base64, Serde
|
||||
Base64, Serde, LibPQ, NATS
|
||||
using GeneralUtils, SQLLLM
|
||||
using ..type, ..util
|
||||
|
||||
@@ -288,18 +288,7 @@ julia> thoughtdict =
|
||||
function search_wine_database!(a::T, thoughtdict::AbstractDict; useSQLLLM::Bool=false
|
||||
)::NamedTuple{(:thoughtdict, :result_raw), Tuple{OrderedDict, Any}} where {T<:agent}
|
||||
|
||||
# WORKING
|
||||
# look_for_wine_in_wine_database(a, thoughtdict["action_input"])
|
||||
|
||||
println("\ncheckinventory order: $(thoughtdict["action_input"]) ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
wineattributes_1 = extractWineAttributes_1(a, thoughtdict["action_input"])
|
||||
wineattributes_2 = extractWineAttributes_2(a, thoughtdict["action_input"])
|
||||
|
||||
retrieve_attributes = ["winery", "wine_name", "wine_id", "vintage", "region", "country", "wine_type", "grape", "serving_temperature", "sweetness", "intensity", "tannin", "acidity", "tasting_notes", "price", "currency", "image_url", "retailer_name", "retailer_id"]
|
||||
_inventoryquery = "$(thoughtdict["action_input"]), $wineattributes_1, $wineattributes_2, retailer_name: $(a.retailername), retailerid: $(a.retailerid)"
|
||||
inventoryquery = "Retrieves $retrieve_attributes of wines that match the following criteria - {$_inventoryquery}"
|
||||
println("\ncheckinventory input: $inventoryquery ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
|
||||
if useSQLLLM
|
||||
# add suppport for similarSQLVectorDB
|
||||
textresult, result_raw = SQLLLM.query(
|
||||
@@ -313,10 +302,47 @@ function search_wine_database!(a::T, thoughtdict::AbstractDict; useSQLLLM::Bool=
|
||||
else
|
||||
|
||||
# direct query with possible sql instead of SQLLLM.
|
||||
sql = generatesql(a, inventoryquery)
|
||||
println("\nSQL: $sql ", @__FILE__, ":", @__LINE__, " $(Dates.now()) \n")
|
||||
hard_conditions, vector_search = wine_search_term_classification(a, thoughtdict["action_input"])
|
||||
|
||||
# do hard filter
|
||||
# sql = generatesql(a, inventoryquery)
|
||||
sql = predefined_wine_search_sql(hard_conditions)
|
||||
@info "\nsql: $sql, \nvector_search: $vector_search"
|
||||
textresult, sql_result_df, success, _ = SQLexecution(a.context.executeSQL, sql)
|
||||
|
||||
# do vector search
|
||||
vector_search_str = ""
|
||||
for i in vector_search
|
||||
vector_search_str = vector_search_str * " " * i["value"]
|
||||
end
|
||||
vector_search_str = String(strip(vector_search_str))
|
||||
vector_search_str = GeneralUtils.removestring(vector_search_str, ["%"])
|
||||
@show vector_search
|
||||
@show vector_search_str
|
||||
|
||||
|
||||
config = a.context.agentconfig
|
||||
host_url, _port = split(config["externalservice"]["sommpanion_db"]["url"], ':')
|
||||
port = parse(Int, _port)
|
||||
dbname = "winedb"
|
||||
user = config["externalservice"]["sommpanion_db"]["user"]
|
||||
password = config["externalservice"]["sommpanion_db"]["password"]
|
||||
pg_conn_str = "host=$host_url port=$port dbname=$dbname user=$user password=$password"
|
||||
|
||||
#WORKING
|
||||
# df = GeneralUtils.find_text_vector_similarity(
|
||||
# vector_search_str,
|
||||
# "wine",
|
||||
# "tasting_notes_embedding",
|
||||
# GeneralUtils.execute_postgres_sql(pg_conn_str, sql), #BUG input pair (F, arg)
|
||||
# a.context.getTextEmbedding([vector_search_str]) #BUG input pair (F, arg)
|
||||
# )
|
||||
|
||||
|
||||
# @show df
|
||||
# error(888888)
|
||||
|
||||
|
||||
items = nothing
|
||||
if sql_result_df !== nothing
|
||||
result_vec = GeneralUtils.dfToVectorDict(sql_result_df)
|
||||
@@ -627,11 +653,11 @@ julia> thoughtdict =
|
||||
"action_name" => "SEARCH_WINE_DATABASE",
|
||||
"action_input" => "Brunello di Montalcino from Tenuta CastelGiocondo")
|
||||
```
|
||||
julia> look_for_wine_in_wine_database(agent, thoughtdict["action_input"])
|
||||
julia> predefined_wine_search_sql(agent, thoughtdict["action_input"])
|
||||
"""
|
||||
function look_for_wine_in_wine_database(a::T, searchterm::String,
|
||||
function wine_search_term_classification(a::T, searchterm::String,
|
||||
; maxattempt=10
|
||||
)::String where {T<:agent}
|
||||
) where {T<:agent}
|
||||
|
||||
systemmsg =
|
||||
"""
|
||||
@@ -647,43 +673,47 @@ function look_for_wine_in_wine_database(a::T, searchterm::String,
|
||||
# your responsibility includes
|
||||
Fulfill the objective.
|
||||
|
||||
# you should only respond in YAML format as described below
|
||||
table_name_1:
|
||||
column_name_1:
|
||||
operator: "="
|
||||
value: "..."
|
||||
column_name_2:
|
||||
operator: "="
|
||||
value: "..."
|
||||
...
|
||||
table_name_2:
|
||||
column_name_1:
|
||||
operator: "="
|
||||
value: "..."
|
||||
column_name_2:
|
||||
operator: "="
|
||||
value: "..."
|
||||
...
|
||||
# You must output your response as a JSON object containing a single key: "extracted_info".
|
||||
The "extracted_info" key must contain an array of objects. Each object must contain:
|
||||
1) "table_name": The name of the table.
|
||||
2) "column_name": The specific column being filtered.
|
||||
3) "operator": The comparison operator (e.g., "=", ">").
|
||||
4) "value": The value to compare against.
|
||||
|
||||
If the user does not specify any filters, return an empty array for "extracted_info": {"extracted_info": []}.
|
||||
|
||||
# here are some example
|
||||
<user>
|
||||
4-wheel drive car with red color that will give me fast and furious emotion. No more than 7000 USD
|
||||
</user>
|
||||
<assistant>
|
||||
car_info: # table_name
|
||||
drive_type: # column_name
|
||||
operator: "=" # operator is not "N/A" because drive_type column store quantitative value
|
||||
value: "4-wheel" # column_value
|
||||
color:
|
||||
operator: "=" # operator is not "N/A" because color column store quantitative value
|
||||
value: "red"
|
||||
drive_feeling:
|
||||
operator: "N/A" # operator is "N/A" because drive_feeling column store qualitative value
|
||||
value: "fast and furious"
|
||||
price_list:
|
||||
price:
|
||||
operator: "<" # operator is not "N/A" because drive_type column store quantitative value
|
||||
value: "7000"
|
||||
{
|
||||
"extracted_info": [
|
||||
{
|
||||
"table_name": "car_info",
|
||||
"column_name": "drive_type",
|
||||
"operator": "=",
|
||||
"value": "4-wheel"
|
||||
},
|
||||
{
|
||||
"table_name": "car_info",
|
||||
"column_name": "color",
|
||||
"operator": "=",
|
||||
"value": "red"
|
||||
},
|
||||
{
|
||||
"table_name": "car_info",
|
||||
"column_name": "drive_feeling",
|
||||
"operator": "=",
|
||||
"value": "fast and furious"
|
||||
},
|
||||
{
|
||||
"table_name": "price_list",
|
||||
"column_name": "price",
|
||||
"operator": "<",
|
||||
"value": "7000"
|
||||
}
|
||||
}
|
||||
</assistant>
|
||||
"""
|
||||
|
||||
@@ -693,7 +723,9 @@ function look_for_wine_in_wine_database(a::T, searchterm::String,
|
||||
related_tables = a.context.find_related_tables_for_user_question(searchterm)
|
||||
table_schema = ""
|
||||
for table in related_tables
|
||||
_table_schema_str = GeneralUtils.get_db_table_schema_simple(a.context.pg_conn_str, table)
|
||||
_table_schema_str = get_db_table_schema_simple_with_samples(a.context.pg_conn_str, table)
|
||||
|
||||
# _table_schema_str = GeneralUtils.get_db_table_schema_simple(a.context.pg_conn_str, table)
|
||||
table_schema_str = sprint(show, _table_schema_str) * "\n"
|
||||
table_schema = table_schema * table_schema_str
|
||||
end
|
||||
@@ -708,6 +740,48 @@ function look_for_wine_in_wine_database(a::T, searchterm::String,
|
||||
"""
|
||||
input = context * searchterm
|
||||
|
||||
response_format = Dict(
|
||||
"type" => "json_schema",
|
||||
"json_schema" => Dict(
|
||||
"name" => "extracted_conditions",
|
||||
"strict" => true,
|
||||
"schema" => Dict(
|
||||
"type" => "object",
|
||||
"properties" => Dict(
|
||||
"extracted_info" => Dict(
|
||||
"type" => "array",
|
||||
"items" => Dict(
|
||||
"type" => "object",
|
||||
"properties" => Dict(
|
||||
"table_name" => Dict(
|
||||
"type" => "string",
|
||||
"description" => "The name of the database table."
|
||||
),
|
||||
"column_name" => Dict(
|
||||
"type" => "string",
|
||||
"description" => "The name of the column to filter on."
|
||||
),
|
||||
"operator" => Dict(
|
||||
"type" => "string",
|
||||
"enum" => ["=", "!=", ">", "<", ">=", "<=", "LIKE", "IN", "IS NULL", "IS NOT NULL"],
|
||||
"description" => "The SQL comparison operator."
|
||||
),
|
||||
"value" => Dict(
|
||||
"type" => ["string", "null"],
|
||||
"description" => "The value to compare against. Use null for IS NULL/IS NOT NULL."
|
||||
)
|
||||
),
|
||||
"required" => ["table_name", "column_name", "operator", "value"],
|
||||
"additionalProperties" => false
|
||||
)
|
||||
)
|
||||
),
|
||||
"required" => ["extracted_info"],
|
||||
"additionalProperties" => false
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
msg = Dict(
|
||||
"model" => "gemma-4-E4B-it-UD-Q4_K_XL",
|
||||
"messages" => [
|
||||
@@ -724,121 +798,142 @@ function look_for_wine_in_wine_database(a::T, searchterm::String,
|
||||
]
|
||||
),
|
||||
],
|
||||
"temperature" => 0.7
|
||||
"temperature" => 0.7,
|
||||
"response_format"=> response_format,
|
||||
)
|
||||
|
||||
for attempt in 1:maxattempt
|
||||
response = a.context.text2textInstructLLM("random_id", msg)
|
||||
responsedict = Serde.parse_yaml(response)
|
||||
responsedict = JSON.parse(response)
|
||||
# responsedict = nothing
|
||||
# try
|
||||
# responsedict = Serde.parse_yaml(response)
|
||||
# catch e
|
||||
# println("\nERROR YiemAgent predefined_wine_search_sql() Error: $e --(not qualify response)-> $response ", @__FILE__, ":", @__LINE__, " $(Dates.now())\n")
|
||||
# continue
|
||||
# end
|
||||
|
||||
"""
|
||||
responsedict = Dict(
|
||||
"wine" => Dict(
|
||||
"tasting_notes" => Dict(
|
||||
"operator" => "N/A",
|
||||
"value" => "casual dinner"
|
||||
),
|
||||
"wine_type" => Dict(
|
||||
"operator" => "=",
|
||||
"value" => "red"
|
||||
)
|
||||
),
|
||||
"retailer_wine" => Dict(
|
||||
"currency" => Dict(
|
||||
"operator" => "=", "value" => "USD"
|
||||
),
|
||||
"price" => Dict(
|
||||
"operator" => "<", "value" => "1000"
|
||||
)
|
||||
)
|
||||
)
|
||||
"""
|
||||
# println("\n ", table_schema)
|
||||
println("\n ", responsedict)
|
||||
@info "before BM25 " @__LINE__
|
||||
|
||||
for (table_name, table_info_dict) in responsedict
|
||||
for (column_name, v) in table_info_dict
|
||||
# to ensure user input is correct
|
||||
for entry in responsedict["extracted_info"]
|
||||
table_name = entry["table_name"]::String
|
||||
column_name = entry["column_name"]::String
|
||||
|
||||
#
|
||||
do_not_resolve_BM25_list = ["tasting_notes", "seo_name", "vintage", "grape"]
|
||||
if column_name ∉ do_not_resolve_list
|
||||
words_catalog = GeneralUtils.harvest_entity_catalog(a.context.pg_conn_str, table_name, column_name)
|
||||
resolved_word = GeneralUtils.resolve_entity(v["value"], words_catalog; threshold=0.9)
|
||||
table_info_dict[column_name] = resolved_word
|
||||
end
|
||||
bucket = classify_column(a.context.pg_conn_str, table_name, column_name)
|
||||
|
||||
if bucket == "fuzzy_correction"
|
||||
words_catalog = GeneralUtils.harvest_entity_catalog(a.context.pg_conn_str, table_name, column_name)
|
||||
resolved_word = GeneralUtils.resolve_entity(entry["value"], words_catalog; threshold=0.9)
|
||||
entry["value"] = resolved_word
|
||||
end
|
||||
end
|
||||
|
||||
# filter for column that will be used for hard condition (SQL where clause)
|
||||
# column with non-standard operator will be used in vector search
|
||||
vector_search_words = ""
|
||||
hard_operators = ["=","<>","!=",">","<",">=","<=","!<","!>","<=>"]
|
||||
|
||||
# check each attributes against each column in a database table with BM25 and get the closest
|
||||
# word match because there is a typo sometimes.
|
||||
for (k, v) in responsedict
|
||||
if k ∉ ["tasting_notes"]
|
||||
words_catalog = GeneralUtils.harvest_entity_catalog(a.context.pg_conn_str, "wine", k)
|
||||
resolved_word = GeneralUtils.resolve_entity(v, words_catalog; threshold=0.9)
|
||||
responsedict[k] = resolved_word
|
||||
# Build new list of hard condition entries
|
||||
hard_conditions = JSON.Object{String, Any}[]
|
||||
vector_search = JSON.Object{String, Any}[]
|
||||
for entry in responsedict["extracted_info"]
|
||||
if entry["operator"] ∈ hard_operators
|
||||
push!(hard_conditions, entry)
|
||||
else
|
||||
push!(vector_search, entry)
|
||||
end
|
||||
end
|
||||
responsedict = hard_conditions
|
||||
|
||||
println("")
|
||||
@show responsedict
|
||||
@info "predefined_wine_search_sql() " @__LINE__
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
#WORKING
|
||||
println("\n", responsedict)
|
||||
@info "test done " @__LINE__
|
||||
error(9999)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
think, response = GeneralUtils.extractthink(response)
|
||||
responsedict = nothing
|
||||
try
|
||||
_responsedict = JSON.parse(response)
|
||||
responsedict = GeneralUtils.dictify(_responsedict, keytype=String)
|
||||
catch
|
||||
println("\nERROR decisionMaker() failed to parse response: $response ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
continue
|
||||
end
|
||||
|
||||
# check each attributes against each column in a database table with BM25 and get the closest
|
||||
# word match because there is a typo sometimes.
|
||||
for (k, v) in responsedict
|
||||
if k ∉ ["tasting_notes"]
|
||||
words_catalog = GeneralUtils.harvest_entity_catalog(a.context.pg_conn_str, "wine", k)
|
||||
resolved_word = GeneralUtils.resolve_entity(v, words_catalog; threshold=0.9)
|
||||
responsedict[k] = resolved_word
|
||||
end
|
||||
end
|
||||
|
||||
# LLM already extract user search term against tables schema
|
||||
# Ex. responsedict = Dict(
|
||||
# "wine_type"=> "red", # hard constraint
|
||||
# "region"=> "bordeaux", # hard constraint
|
||||
# "price_max"=> "100", # hard constraint
|
||||
# "tasting_notes"=> "fruity, oak" # semantic search)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
return items
|
||||
return (hard_conditions=hard_conditions, vector_search=vector_search)
|
||||
end
|
||||
error("SQLLLM DecisionMaker() failed to generate a thought \n", response)
|
||||
end
|
||||
|
||||
function predefined_wine_search_sql(conditions::Vector{JSON.Object{String, Any}})::String
|
||||
# 1. Base SQL structure
|
||||
base_query =
|
||||
"""
|
||||
SELECT
|
||||
w.winery,
|
||||
w.wine_name,
|
||||
w.wine_id,
|
||||
w.vintage,
|
||||
w.region,
|
||||
w.country,
|
||||
w.wine_type,
|
||||
w.grape,
|
||||
w.serving_temperature,
|
||||
w.sweetness,
|
||||
w.intensity,
|
||||
w.tannin,
|
||||
w.acidity,
|
||||
w.tasting_notes,
|
||||
rw.price,
|
||||
rw.currency,
|
||||
w.image_url,
|
||||
r.retailer_name,
|
||||
rw.retailer_id
|
||||
FROM wine AS w
|
||||
JOIN retailer_wine AS rw ON w.wine_id = rw.wine_id
|
||||
JOIN retailer AS r ON rw.retailer_id = r.retailer_id
|
||||
"""
|
||||
|
||||
# 2. Dynamic WHERE Clause Builder
|
||||
where_clauses = String[]
|
||||
|
||||
# Iterate over each condition object in the array
|
||||
for cond in conditions
|
||||
table_name = String(cond["table_name"])
|
||||
column_name = String(cond["column_name"])
|
||||
op = String(cond["operator"])
|
||||
raw_val = cond["value"]
|
||||
|
||||
# Determine table alias
|
||||
alias = if table_name == "wine"
|
||||
"w"
|
||||
elseif table_name == "retailer_wine"
|
||||
"rw"
|
||||
else
|
||||
continue
|
||||
end
|
||||
|
||||
# --- Value Type Handling ---
|
||||
final_val = raw_val
|
||||
|
||||
if op in ("=", "<", ">", "<=", ">=")
|
||||
str_val = string(raw_val)
|
||||
num_val = tryparse(Float64, str_val)
|
||||
|
||||
if !isnothing(num_val)
|
||||
final_val = isinteger(num_val) ? round(Int, num_val) : num_val
|
||||
end
|
||||
end
|
||||
|
||||
# --- SQL Formatting ---
|
||||
if isa(final_val, Number)
|
||||
clause = "$(alias).$(column_name) $(op) $(final_val)"
|
||||
else
|
||||
escaped_val = replace(string(final_val), "'" => "''")
|
||||
clause = "$(alias).$(column_name) $(op) '$(escaped_val)'"
|
||||
end
|
||||
|
||||
push!(where_clauses, clause)
|
||||
end
|
||||
|
||||
# 3. Assemble Final Query
|
||||
where_sql = isempty(where_clauses) ? "" : "WHERE " * join(where_clauses, " AND ")
|
||||
|
||||
return string(base_query, where_sql, ";")
|
||||
end
|
||||
|
||||
function SQLexecution(executeSQL::Function, sql::T
|
||||
)::NamedTuple where {T<:AbstractString}
|
||||
|
||||
@@ -889,6 +984,64 @@ function SQLexecution(executeSQL::Function, sql::T
|
||||
end
|
||||
end
|
||||
|
||||
function DEPRECIATED_search_wine_database!(a::T, thoughtdict::AbstractDict; useSQLLLM::Bool=false
|
||||
)::NamedTuple{(:thoughtdict, :result_raw), Tuple{OrderedDict, Any}} where {T<:agent}
|
||||
|
||||
# XXX
|
||||
predefined_wine_search_sql(a, thoughtdict["action_input"])
|
||||
|
||||
println("\ncheckinventory order: $(thoughtdict["action_input"]) ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
wineattributes_1 = extractWineAttributes_1(a, thoughtdict["action_input"])
|
||||
wineattributes_2 = extractWineAttributes_2(a, thoughtdict["action_input"])
|
||||
|
||||
retrieve_attributes = ["winery", "wine_name", "wine_id", "vintage", "region", "country", "wine_type", "grape", "serving_temperature", "sweetness", "intensity", "tannin", "acidity", "tasting_notes", "price", "currency", "image_url", "retailer_name", "retailer_id"]
|
||||
_inventoryquery = "$(thoughtdict["action_input"]), $wineattributes_1, $wineattributes_2, retailer_name: $(a.retailername), retailerid: $(a.retailerid)"
|
||||
inventoryquery = "Retrieves $retrieve_attributes of wines that match the following criteria - {$_inventoryquery}"
|
||||
println("\ncheckinventory input: $inventoryquery ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
|
||||
if useSQLLLM
|
||||
# add suppport for similarSQLVectorDB
|
||||
textresult, result_raw = SQLLLM.query(
|
||||
inventoryquery,
|
||||
a.context.executeSQL,
|
||||
a.context.text2textInstructLLM;
|
||||
insertSQLVectorDB=a.context.insertSQLVectorDB,
|
||||
similarSQLVectorDB=a.context.similarSQLVectorDB,
|
||||
llmFormatName="qwen3")
|
||||
thoughtdict["action_result"] = textresult
|
||||
else
|
||||
|
||||
# direct query with possible sql instead of SQLLLM.
|
||||
sql = generatesql(a, inventoryquery)
|
||||
println("\nSQL: $sql ", @__FILE__, ":", @__LINE__, " $(Dates.now()) \n")
|
||||
textresult, sql_result_df, success, _ = SQLexecution(a.context.executeSQL, sql)
|
||||
|
||||
items = nothing
|
||||
if sql_result_df !== nothing
|
||||
result_vec = GeneralUtils.dfToVectorDict(sql_result_df)
|
||||
|
||||
# get image
|
||||
for d in result_vec
|
||||
image_url_json_str = d["image_url"]
|
||||
image_url_json_obj = JSON.parse(image_url_json_str)
|
||||
base_url = "http://192.168.88.106:8080/"
|
||||
if haskey(image_url_json_obj, "bottle")
|
||||
url = base_url * image_url_json_obj["bottle"]
|
||||
image_data = HTTP.get(url) # vector{int} data
|
||||
image_base64_string = base64encode(image_data.body)
|
||||
d["image"] = image_base64_string
|
||||
else
|
||||
d["image"] = nothing
|
||||
end
|
||||
end
|
||||
items = result_vec # image is added to each item
|
||||
end
|
||||
|
||||
thoughtdict["action_result"] = textresult
|
||||
end
|
||||
|
||||
return (thoughtdict=thoughtdict, result_raw=items)
|
||||
end
|
||||
|
||||
"""
|
||||
|
||||
@@ -1266,219 +1419,199 @@ function extractWineAttributes_2(a::T1, input::T2)::String where {T1<:agent, T2<
|
||||
end
|
||||
|
||||
|
||||
function paraphrase(text2textInstructLLM::Function, text::String)
|
||||
systemmsg =
|
||||
"""
|
||||
Your name: N/A
|
||||
Your vision:
|
||||
- You are a helpful assistant who help the user to paraphrase their text.
|
||||
Your mission:
|
||||
- To help paraphrase the user's text
|
||||
Mission's objective includes:
|
||||
- To help paraphrase the user's text
|
||||
Your responsibility includes:
|
||||
1) To help paraphrase the user's text
|
||||
Your responsibility does NOT includes:
|
||||
1) N/A
|
||||
Your profile:
|
||||
- N/A
|
||||
Additional information:
|
||||
- N/A
|
||||
|
||||
At each round of conversation, you will be given the following information:
|
||||
Text: The user's given text
|
||||
|
||||
You MUST follow the following guidelines:
|
||||
- N/A
|
||||
|
||||
You should follow the following guidelines:
|
||||
- N/A
|
||||
|
||||
You should then respond to the user with:
|
||||
Paraphrase: Paraphrased text
|
||||
|
||||
You should only respond in format as described below:
|
||||
Paraphrase: ...
|
||||
|
||||
Let's begin!
|
||||
"""
|
||||
#[PENDING] use JSON the same as extractWineAttributes_1 is better. change this function to use the same format use decisionMaker
|
||||
header = ["Paraphrase:"]
|
||||
dictkey = ["paraphrase"]
|
||||
|
||||
errornote = "N/A"
|
||||
response = nothing # placeholder for show when error msg show up
|
||||
|
||||
|
||||
for attempt in 1:10
|
||||
usermsg = """
|
||||
Text: $text
|
||||
P.S. $errornote
|
||||
"""
|
||||
|
||||
_prompt =
|
||||
[
|
||||
Dict("name" => "system", "text" => systemmsg),
|
||||
Dict("name" => "user", "text" => usermsg)
|
||||
]
|
||||
|
||||
# put in model format
|
||||
prompt = GeneralUtils.formatLLMtext(_prompt, a.llmFormatName)
|
||||
function get_db_table_schema_simple_with_samples(pg_conn_str::String, table_name::String;
|
||||
schema_name::String="public")::String
|
||||
conn = LibPQ.Connection(pg_conn_str)
|
||||
return get_db_table_schema_simple_with_samples(conn, table_name; schema_name=schema_name)
|
||||
end
|
||||
|
||||
try
|
||||
response = text2textInstructLLM(prompt)
|
||||
response = GeneralUtils.deFormatLLMtext(response, a.llmFormatName)
|
||||
think, response = GeneralUtils.extractthink(response)
|
||||
# sometime the model response like this "here's how I would respond: ..."
|
||||
if occursin("respond:", response)
|
||||
errornote = "You don't need to intro your response"
|
||||
error("\nparaphrase() response contain : ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
end
|
||||
response = GeneralUtils.remove_french_accents(response)
|
||||
response = replace(response, '*'=>"")
|
||||
response = replace(response, '$' => "USD")
|
||||
response = replace(response, '`' => "")
|
||||
response = GeneralUtils.remove_french_accents(response)
|
||||
function get_db_table_schema_simple_with_samples(conn, table_name::String; schema_name::String="public", sample_count::Int=3)::String
|
||||
# 1. SQL query for catalog metadata
|
||||
meta_sql = """
|
||||
SELECT
|
||||
a.attname AS column_name,
|
||||
format_type(a.atttypid, a.atttypmod) AS data_type,
|
||||
pg_get_expr(def.adbin, def.adrelid) AS default_value,
|
||||
COALESCE(
|
||||
(SELECT pg_get_constraintdef(p.oid)
|
||||
FROM pg_catalog.pg_constraint p
|
||||
WHERE p.conrelid = c.oid AND a.attnum = ANY(p.conkey)
|
||||
LIMIT 1), ''
|
||||
) AS constraint_definition
|
||||
FROM pg_catalog.pg_attribute a
|
||||
JOIN pg_catalog.pg_class c ON a.attrelid = c.oid
|
||||
JOIN pg_catalog.pg_namespace n ON c.relnamespace = n.oid
|
||||
LEFT JOIN pg_catalog.pg_attrdef def ON def.adrelid = c.oid AND def.adnum = a.attnum
|
||||
WHERE c.relname = \$1
|
||||
AND n.nspname = \$2
|
||||
AND a.attnum > 0
|
||||
AND NOT a.attisdropped
|
||||
ORDER BY a.attnum;
|
||||
"""
|
||||
|
||||
# check whether response has all answer's key points
|
||||
detected_kw = GeneralUtils.detect_keyword(header, response)
|
||||
if 0 ∈ values(detected_kw)
|
||||
errornote = "\nYiemAgent paraphrase() response does not have all answer's key points"
|
||||
continue
|
||||
elseif sum(values(detected_kw)) > length(header)
|
||||
errornote = "\nnYiemAgent paraphrase() response has duplicated answer's key points"
|
||||
continue
|
||||
end
|
||||
meta_res = DataFrame(execute(conn, meta_sql, [table_name, schema_name]))
|
||||
|
||||
responsedict = GeneralUtils.textToDict(response, header;
|
||||
dictKey=dictkey, symbolkey=true)
|
||||
|
||||
for i ∈ [:paraphrase]
|
||||
if length(JSON.json(responsedict[i])) == 0
|
||||
error("$i is empty ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
end
|
||||
end
|
||||
|
||||
# check if there are more than 1 key per categories
|
||||
for i ∈ [:paraphrase]
|
||||
matchkeys = GeneralUtils.findMatchingDictKey(responsedict, i)
|
||||
if length(matchkeys) > 1
|
||||
error("paraphrase() has more than one key per categories")
|
||||
end
|
||||
end
|
||||
|
||||
println("\nparaphrase() ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
pprintln(Dict(responsedict))
|
||||
|
||||
result = responsedict["paraphrase"]
|
||||
|
||||
return result
|
||||
catch e
|
||||
io = IOBuffer()
|
||||
showerror(io, e)
|
||||
errorMsg = String(take!(io))
|
||||
st = sprint((io, v) -> show(io, "text/plain", v), stacktrace(catch_backtrace()))
|
||||
println("\nAttempt $attempt. Error occurred: $errorMsg\n$st ", @__FILE__, ":", @__LINE__, " $(Dates.now())")
|
||||
if nrow(meta_res) == 0
|
||||
error("Table '$schema_name.$table_name' not found.")
|
||||
end
|
||||
end
|
||||
error("paraphrase() failed to generate a response")
|
||||
|
||||
# 2. Build single dynamic query to fetch non-null samples for all columns
|
||||
sample_selects = String[]
|
||||
for row in eachrow(meta_res)
|
||||
c_name = row.column_name
|
||||
push!(sample_selects, """
|
||||
(SELECT json_agg(s."$c_name")
|
||||
FROM (
|
||||
SELECT "$c_name"
|
||||
FROM "$schema_name"."$table_name"
|
||||
WHERE "$c_name" IS NOT NULL
|
||||
LIMIT $sample_count
|
||||
) s
|
||||
) AS "$c_name"
|
||||
""")
|
||||
end
|
||||
|
||||
sample_sql = "SELECT " * join(sample_selects, ",\n ") * ";"
|
||||
sample_df = DataFrame(execute(conn, sample_sql))
|
||||
|
||||
# 3. Build DDL definitions with inline sample comments
|
||||
ddl_lines = String[]
|
||||
constraints = String[]
|
||||
|
||||
for row in eachrow(meta_res)
|
||||
col_name = row.column_name
|
||||
data_type = row.data_type
|
||||
default_val = ismissing(row.default_value) ? "" : " DEFAULT " * row.default_value
|
||||
|
||||
col_def = " \"$col_name\" $data_type$default_val"
|
||||
|
||||
# Fetch sample data for this column from the single-row sample DataFrame
|
||||
samples_comment = ""
|
||||
if nrow(sample_df) > 0
|
||||
raw_samples = sample_df[1, Symbol(col_name)]
|
||||
samples_str = ismissing(raw_samples) || isnothing(raw_samples) ? "[]" : string(raw_samples)
|
||||
samples_comment = " -- Samples: $samples_str"
|
||||
end
|
||||
|
||||
push!(ddl_lines, col_def * samples_comment)
|
||||
|
||||
# Handle table-level constraints
|
||||
con_def = ismissing(row.constraint_definition) ? "" : row.constraint_definition
|
||||
if !isempty(con_def) && !(con_def in constraints)
|
||||
push!(constraints, " " * con_def)
|
||||
end
|
||||
end
|
||||
|
||||
all_definitions = vcat(ddl_lines, constraints)
|
||||
body = join(all_definitions, ",\n")
|
||||
|
||||
return "CREATE TABLE \"$schema_name\".\"$table_name\" (\n$body\n);"
|
||||
end
|
||||
|
||||
|
||||
|
||||
""" Attemp to correct LLM response's incorrect JSON response.
|
||||
|
||||
# Arguments
|
||||
- `a::T1`
|
||||
one of Yiem's agent
|
||||
- `input::T2`
|
||||
text to be send to virtual wine customer
|
||||
|
||||
# Return
|
||||
- `correctjson::String`
|
||||
corrected json string
|
||||
|
||||
# Example
|
||||
```jldoctest
|
||||
julia>
|
||||
```
|
||||
|
||||
# Signature
|
||||
"""
|
||||
function jsoncorrection(config::T1, input::T2, correctJsonExample::T3;
|
||||
maxattempt::Integer=3
|
||||
) where {T1<:AbstractDict, T2<:AbstractString, T3<:AbstractString}
|
||||
|
||||
incorrectjson = deepcopy(input)
|
||||
correctjson = nothing
|
||||
|
||||
for attempt in 1:maxattempt
|
||||
try
|
||||
d = copy(JSON.parsefile(incorrectjson))
|
||||
correctjson = incorrectjson
|
||||
return correctjson
|
||||
catch e
|
||||
@warn "Attempting to correct JSON string. Attempt $attempt"
|
||||
e = """$e"""
|
||||
if occursin("EOF", e)
|
||||
e = split(e, "EOF")[1] * "EOF"
|
||||
end
|
||||
incorrectjson = deepcopy(input)
|
||||
_prompt =
|
||||
"""
|
||||
Your goal are:
|
||||
1) Use the expected JSON format as a guideline to check why the given JSON string failed to load and provide a corrected version that can be loaded by Python's json.load function.
|
||||
2) Provide Corrected JSON string only. Do not provide any other info.
|
||||
|
||||
$correctJsonExample
|
||||
|
||||
Let's begin!
|
||||
Given JSON string: $incorrectjson
|
||||
The given JSON string failed to load previously because: $e
|
||||
Corrected JSON string:
|
||||
"""
|
||||
|
||||
# apply LLM specific instruct format
|
||||
externalService = config["externalservice"]["text2textinstruct"]
|
||||
llminfo = externalService["llminfo"]
|
||||
prompt =
|
||||
if llminfo["name"] == "llama3instruct"
|
||||
formatLLMtext_llama3instruct("system", _prompt)
|
||||
else
|
||||
error("llm model name is not defied yet $(@__LINE__)")
|
||||
end
|
||||
|
||||
# send formatted input to user using GeneralUtils.sendReceiveMqttMsg
|
||||
msgMeta = GeneralUtils.generate_msgMeta(
|
||||
externalService["mqtttopic"],
|
||||
senderName= "jsoncorrection",
|
||||
senderId= string(uuid4()),
|
||||
receiverName= "text2textinstruct",
|
||||
mqttBroker= config["mqttServerInfo"]["broker"],
|
||||
mqttBrokerPort= config["mqttServerInfo"]["port"],
|
||||
)
|
||||
|
||||
outgoingMsg = Dict(
|
||||
"msgMeta"=> msgMeta,
|
||||
"payload"=> Dict(
|
||||
"text"=> prompt,
|
||||
"kwargs"=> Dict(
|
||||
"max_tokens"=> 512,
|
||||
"stop"=> ["<|eot_id|>"],
|
||||
)
|
||||
)
|
||||
)
|
||||
result = GeneralUtils.sendReceiveMqttMsg(outgoingMsg; timeout=120)
|
||||
incorrectjson = result[:response][:text]
|
||||
end
|
||||
end
|
||||
function classify_column(pg_conn_str::String, table_name::String, column_name::String;
|
||||
sample_size::Integer=1000)
|
||||
conn = LibPQ.Connection(pg_conn_str)
|
||||
return classify_column(conn, table_name, column_name; sample_size=sample_size)
|
||||
end
|
||||
|
||||
|
||||
function classify_column(conn::LibPQ.Connection, table_name::String, column_name::String; sample_size::Int=1000)
|
||||
# 1. Fetch BOTH data_type and udt_name (User Defined Type name)
|
||||
meta_query = """
|
||||
SELECT data_type, udt_name
|
||||
FROM information_schema.columns
|
||||
WHERE table_name = lower('$(table_name)')
|
||||
AND column_name = lower('$(column_name)');
|
||||
"""
|
||||
|
||||
pg_type = "unknown"
|
||||
udt_name = "unknown"
|
||||
|
||||
try
|
||||
df = DataFrame(LibPQ.execute(conn, meta_query))
|
||||
if !isempty(df)
|
||||
pg_type = df[1, :data_type]
|
||||
udt_name = df[1, :udt_name]
|
||||
end
|
||||
catch e
|
||||
@error "Failed to fetch metadata for $table_name.$column_name" exception=e
|
||||
return "error"
|
||||
end
|
||||
|
||||
# 2. FAST-TRACK: Check for pgvector FIRST
|
||||
# pgvector registers as "USER-DEFINED" in data_type, but "vector" in udt_name
|
||||
if udt_name == "vector"
|
||||
return "semantic_search"
|
||||
end
|
||||
|
||||
# 3. FAST-TRACK: Hard rules for standard non-text Postgres types
|
||||
if pg_type in ["integer", "bigint", "smallint", "numeric", "real",
|
||||
"double precision", "boolean", "date",
|
||||
"timestamp without time zone", "timestamp with time zone", "uuid"]
|
||||
return "exact_or_range"
|
||||
end
|
||||
|
||||
# 4. SAMPLE: Get text statistics for remaining text columns
|
||||
stats_query = """
|
||||
SELECT
|
||||
COUNT(*)::int AS total_count,
|
||||
COUNT(DISTINCT $(column_name)::text)::int AS unique_count,
|
||||
COALESCE(AVG(LENGTH($(column_name)::text)), 0)::float AS avg_len,
|
||||
COALESCE(STDDEV(LENGTH($(column_name)::text)), 0)::float AS std_len
|
||||
FROM (
|
||||
SELECT $(column_name)
|
||||
FROM $(table_name)
|
||||
WHERE $(column_name) IS NOT NULL
|
||||
LIMIT $sample_size
|
||||
) AS sampled_data;
|
||||
"""
|
||||
|
||||
try
|
||||
df = DataFrame(LibPQ.execute(conn, stats_query))
|
||||
if isempty(df) || df[1, :total_count] == 0
|
||||
return "unknown"
|
||||
end
|
||||
|
||||
total = df[1, :total_count]
|
||||
unique = df[1, :unique_count]
|
||||
avg_len = df[1, :avg_len]
|
||||
std_len = df[1, :std_len]
|
||||
ratio = unique / total
|
||||
|
||||
# 5. HEURISTICS: Route the column_name to the correct text bucket
|
||||
return classify_text_column(unique, ratio, avg_len, std_len)
|
||||
|
||||
catch e
|
||||
@warn "Failed to sample column_name $table_name.$column_name" exception=e
|
||||
return "unknown"
|
||||
end
|
||||
end
|
||||
|
||||
# The Decision Tree for Text Columns (Unchanged, but kept for completeness)
|
||||
function classify_text_column(unique_count::Integer, ratio::Float64, avg_len::Float64, std_len::Float64)
|
||||
if avg_len > 60 && std_len > 25
|
||||
return "full_text_search"
|
||||
end
|
||||
if ratio > 0.90 && avg_len < 40
|
||||
return "exact_or_regex"
|
||||
end
|
||||
if unique_count <= 100
|
||||
return "fuzzy_correction"
|
||||
end
|
||||
if ratio > 0.10 && avg_len < 35
|
||||
return "fuzzy_correction"
|
||||
end
|
||||
if avg_len < 60
|
||||
return "fuzzy_correction"
|
||||
end
|
||||
return "full_text_search"
|
||||
end
|
||||
|
||||
|
||||
|
||||
|
||||
+1
-80
@@ -23,75 +23,6 @@ end
|
||||
|
||||
abstract type agent end
|
||||
|
||||
mutable struct companion <: agent
|
||||
name::String # agent name
|
||||
id::String # agent id
|
||||
systemmsg::String # system message
|
||||
tools::Dict # tools
|
||||
maxHistoryMsg::Integer # e.g. 21th and earlier messages will get summarized
|
||||
chathistory::Vector{Dict{String, Any}}
|
||||
memory::Dict{String, Any}
|
||||
context::NamedTuple # NamedTuple of functions
|
||||
llmFormatName::String
|
||||
end
|
||||
|
||||
function companion(
|
||||
context::agentcontext # NamedTuple of functions
|
||||
;
|
||||
name::String= "Assistant",
|
||||
id::String= GeneralUtils.uuid4snakecase(),
|
||||
maxHistoryMsg::Integer= 20,
|
||||
chathistory::Vector{Dict{String, String}} = Vector{Dict{String, String}}(),
|
||||
llmFormatName::String= "granite3",
|
||||
systemmsg::String=
|
||||
"""
|
||||
Your name: $name
|
||||
Your sex: Female
|
||||
Your role: You are a helpful assistant.
|
||||
You should follow the following guidelines:
|
||||
- Focus on the latest conversation.
|
||||
- Your like to be short and concise.
|
||||
|
||||
Let's begin!
|
||||
""",
|
||||
)
|
||||
|
||||
tools = Dict( # update input format
|
||||
"CHAT_BOX"=> Dict(
|
||||
"description" => "- CHAT_BOX which you can use to talk with the user. The input is your intentions for the dialogue. Be specific.",
|
||||
),
|
||||
)
|
||||
|
||||
""" Memory
|
||||
Ref: Chat prompt format https://huggingface.co/TheBloke/Llama-2-7B-Chat-GGML/discussions/3
|
||||
NO "system" message in chathistory because I want to add it at the inference time
|
||||
chathistory= [
|
||||
Dict("name"=>"user", "text"=> "Wassup!", "timestamp"=> Dates.now()),
|
||||
Dict("name"=>"assistant", "text"=> "Hi I'm your assistant.", "timestamp"=> Dates.now()),
|
||||
]
|
||||
"""
|
||||
memory = Dict{String, Any}(
|
||||
"events"=> Vector{Dict{String, Any}}(),
|
||||
"state"=> Dict{String, Any}(), # state of the agent
|
||||
"recap"=> OrderedDict{String, Any}(), # recap summary of the conversation
|
||||
)
|
||||
|
||||
newAgent = companion(
|
||||
name,
|
||||
id,
|
||||
systemmsg,
|
||||
tools,
|
||||
maxHistoryMsg,
|
||||
chathistory,
|
||||
memory,
|
||||
context,
|
||||
llmFormatName
|
||||
)
|
||||
|
||||
return newAgent
|
||||
end
|
||||
|
||||
|
||||
mutable struct sommelier <: agent
|
||||
name::String # agent name
|
||||
id::String # agent id
|
||||
@@ -210,11 +141,7 @@ function sommelier(
|
||||
memory = Dict{String, Any}(
|
||||
"shortmem"=> OrderedDict{String, Any}(),
|
||||
"scratchpad"=> "",
|
||||
"events"=> Vector{Dict{String, Any}}(),
|
||||
"state"=> Dict{String, Any}(
|
||||
),
|
||||
"recap"=> OrderedDict{String, Any}(),
|
||||
|
||||
)
|
||||
|
||||
newAgent = sommelier(
|
||||
@@ -237,7 +164,6 @@ function sommelier(
|
||||
- You can only recommend wines that are currently in our inventory
|
||||
- Before searching the database for wine, ensure you have at least the following information: 1) budget, 2) wine type, and 3) occasion. Additional details are always helpful. If the user is unsure, provide relevant information and gather insights to make reasonable inferences.
|
||||
- Ask the user one question at a time.
|
||||
- Do not ask the user about wine's flavor e.g. floral, citrusy, nutty or some thing similar as these terms cannot be used to search the database.
|
||||
- Once the user has selected their wine, if you haven't already, ask the user whether they need any further assistance. Do not offer any additional services.
|
||||
- Only end the conversation when the user explicitly intends to do so. When ending, ensure a polite farewell and an invitation to return in the future.
|
||||
- Spicy foods should be paired only with light red wines.
|
||||
@@ -273,17 +199,12 @@ function sommelier(
|
||||
- Processing sales orders or engaging in any other sales-related activities. These are the job of our sales team at the store.
|
||||
- Answering questions or offering additional services beyond those related to your store's wine recommendations such as discounts, quantity, rewards programs, promotions, delivery options, shipping, boxes, gift wrapping, packaging, personalized messages or something similar. These are the job of our sales team at the store.
|
||||
|
||||
# you should then respond to the user with interleaving plan, action_name, action_input
|
||||
# you should then respond to the user with interleaving plan, action_name, action_input in JSON format
|
||||
1) "plan", Based on the current situation, state a complete action plan to complete the task and rationale. Be specific.
|
||||
2) "action_name", (Typically corresponds to the execution of the first step in your plan) Can be one of the available_actions name
|
||||
3) "action_input", The input to the action you are about to perform according to your plan.
|
||||
After the action is executed you gets "action_result". It is the output from the action you selected.
|
||||
|
||||
# you should only respond in JSON format as described below (not Markdown format)
|
||||
"plan": "...",
|
||||
"action_name": "...",
|
||||
"action_input": "..."
|
||||
|
||||
# available actions
|
||||
"CHAT_BOX", which you can use to talk with the user. The input is dialogue you want to chat with the user according to your plan.
|
||||
"SEARCH_WINE_DATABASE", allows you to search information about wines you want in your inventory's database. The input is strictly supported search term including: retailer_name, wine price, winery, name, vintage, region, country, type of wine, grape varietal, tasting notes, occasion, food pairing, intensity, tannin, sweetness, and acidity.
|
||||
|
||||
Reference in New Issue
Block a user