13 Commits

Author SHA1 Message Date
ton 407447831a update 2026-07-24 20:23:54 +07:00
ton c475eb169c update 2026-07-24 20:12:54 +07:00
ton d1a279cca2 update 2026-07-24 20:09:21 +07:00
ton bf3b65ee7b update 2026-07-24 17:35:31 +07:00
ton 360d64c474 update 2026-07-24 16:42:23 +07:00
ton 41a354fa73 update 2026-07-24 15:34:02 +07:00
ton 801596fa7f update 2026-07-24 14:32:44 +07:00
ton 2fbe9d6e1a update 2026-07-24 13:21:04 +07:00
ton 2c2690e5dd update 2026-07-24 13:18:35 +07:00
ton 95db5f877d up version 2026-07-24 13:04:04 +07:00
ton 7e2ddd846e update 2026-07-24 13:03:40 +07:00
ton ab113acde5 up version 2026-07-24 12:45:55 +07:00
ton 7391f0f2ce add execute_postgres_sql 2026-07-24 12:45:14 +07:00
2 changed files with 97 additions and 3 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
name = "GeneralUtils" name = "GeneralUtils"
uuid = "c6c72f09-b708-4ac8-ac7c-2084d70108fe" uuid = "c6c72f09-b708-4ac8-ac7c-2084d70108fe"
version = "0.5.1" version = "0.5.11"
authors = ["tonaerospace <tonaerospace.etc@gmail.com>"] authors = ["tonaerospace <tonaerospace.etc@gmail.com>"]
[deps] [deps]
+96 -2
View File
@@ -1,13 +1,107 @@
module dbUtil module dbUtil
export dictToPostgresKeyValueString, generateInsertSQL, generateUpdateSQL export dictToPostgresKeyValueString, generateInsertSQL, generateUpdateSQL, find_text_vector_similarity,
execute_postgres_sql
using JSON, DataStructures, Distributions, Random, Dates, UUIDs, DataFrames, using JSON, DataStructures, Distributions, Random, Dates, UUIDs, DataFrames,
SHA SHA, NATS, LibPQ
using ..util using ..util
#[PENDING] update code to use JSON #[PENDING] update code to use JSON
# ---------------------------------------------- 100 --------------------------------------------- # # ---------------------------------------------- 100 --------------------------------------------- #
""" Execute SQL against a PostgreSQL database using LibPQ connection string.
# Arguments
- `pg_conn_str::AbstractString`: PostgreSQL connection string in format "host=... port=... dbname=... user=... password=..."
- `sql::AbstractString`: SQL query to execute
# Returns
- `LibPQ.Result` on success, `nothing` on failure
# Example
```julia
pg_conn_str = "host=localhost port=5432 dbname=mydb user=myuser password=mypass"
sql = "SELECT * FROM wine;"
result = execute_postgres_sql(pg_conn_str, sql)
```
"""
function execute_postgres_sql(pg_conn_str::T, sql::T) where {T<:AbstractString}
db_connection = LibPQ.Connection(pg_conn_str)
result = nothing
try
result = LibPQ.execute(db_connection, sql)
catch e
@error e
LibPQ.close(db_connection)
end
LibPQ.close(db_connection)
return result
end
""" find_text_vector_similarity
Find the most similar text records in a PostgreSQL database using vector embeddings and cosine similarity.
This function computes an embedding for the input text using the provided embedding function,
then queries the database to find records with the most similar vector representations using
PostgreSQL's cosine similarity operator (`<->`).
# Arguments
- `text::AbstractString`: The input text to find similar records for
- `tablename::AbstractString`: Name of the database table containing the embedding column
- `embeddingColumnName::AbstractString`: Name of the column storing vector embeddings
- `executesql::Function`: Function that executes SQL queries and returns results
- `get_embedding::Function`: Function that generates embeddings for text inputs
# Keyword Arguments
- `limit::Integer=1`: Maximum number of similar records to return
# Returns
- `DataFrame`: Database records ordered by similarity (most similar first), including a `distance` column
where smaller values indicate higher similarity
# Example
```julia
# Assume you have embedding and SQL execution functions
text = "a rich structured red wine"
tablename = "wine"
embeddingColumnName = "description_embedding"
df = find_text_vector_similarity(
text, tablename, embeddingColumnName,
executesql, get_embedding;
limit = 5
)
# Result contains columns from the table plus a 'distance' column
# where distance = 1 - cosine_similarity (smaller = more similar)
```
"""
function find_text_vector_similarity(text::T1, tablename::T2, embeddingColumnName::T3,
executesql::Function, get_embedding::Function;
limit::Integer=1
)::DataFrame where {T1<:AbstractString, T2<:AbstractString, T3<:AbstractString}
# get embedding from LLM service
_embedding = get_embedding([text])
_embedding = _embedding["data"][1]["embedding"]
_embedding = "$_embedding"
embedding = _embedding[4:end] # remove 'Any' from Any[...]
# check whether there is close enough vector already store in executesql. if no, add, else skip
sql = """
SELECT *, $embeddingColumnName <-> '$embedding' as distance
FROM $tablename
ORDER BY distance LIMIT $limit;
"""
response = executesql(sql)
df = DataFrame(response)
return df
end
""" """
dictToPostgresKeyValueString - Convert dictionary to PostgreSQL key-value string format dictToPostgresKeyValueString - Convert dictionary to PostgreSQL key-value string format