Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d31b0eaddc | ||
|
|
f53ad4de5c | ||
|
|
053d7d668d | ||
|
|
21e3ed2b39 | ||
|
|
16ca4e9293 | ||
|
|
75f82be88d | ||
|
|
22f6a02b08 | ||
|
|
11aa1dbbb1 | ||
|
|
d961d85a28 | ||
|
|
d0c1ad5144 | ||
|
|
0d30b2fe83 | ||
|
|
f33bae6683 | ||
|
|
eaa0df125b | ||
|
|
b0e01b7127 | ||
|
|
2a421e7406 | ||
|
|
23150b68db | ||
|
|
e5c46010a4 | ||
|
|
b2bf68d7a4 | ||
|
|
31cec4f5c9 | ||
|
|
84a0ad8dfb | ||
|
|
69466fd43b | ||
|
|
324e078c85 | ||
|
|
7385fc4529 | ||
|
|
66d1fa427f | ||
|
|
bee396a433 | ||
|
|
8727a52931 | ||
|
|
ed5c076a5a | ||
|
|
db5c575ae0 | ||
|
|
2a0f482578 | ||
|
|
c041cf371a | ||
|
|
c6daf09cd2 | ||
|
|
7296e2e012 | ||
|
|
43bf176ef7 | ||
|
|
6b48d7771c | ||
|
|
cd0d39b3c2 | ||
|
|
8fcbcdd29d | ||
|
|
e24f65c51d | ||
|
|
38e6464219 | ||
|
|
86f5c885a1 | ||
|
|
c0ebd06c0c | ||
|
|
0ce5ddcd00 | ||
|
|
7539af3bd9 | ||
|
|
60f9147873 | ||
|
|
a594c142e0 | ||
|
|
18905a9872 | ||
|
|
27f5dd837d | ||
|
|
a921f9db94 | ||
|
|
b3cd940c73 | ||
|
|
f91cf1ca65 | ||
|
|
c438e89bd0 | ||
|
|
9b1746b7ed | ||
|
|
0611a72c40 | ||
|
|
0ba66526e9 | ||
|
|
425202dd1e | ||
|
|
882ad83aad | ||
|
|
9b12783d08 | ||
|
|
c6817ae32f | ||
|
|
42711a8cc9 | ||
|
|
0344ab2cf9 | ||
|
|
df0d12af43 | ||
|
|
d058ba8861 | ||
|
|
d1309905d0 | ||
|
|
ca6ac3912f | ||
|
|
9eabc32681 | ||
|
|
8798e4a499 | ||
|
|
ea8b2da37c | ||
|
|
b7f7d10fea | ||
|
|
0279c0f48e | ||
|
|
c29cf3855f | ||
|
|
c73901369c | ||
|
|
0e70f03fc3 | ||
|
|
74d1f7f717 | ||
|
|
dfcf67184d | ||
|
|
ce641d1e2b | ||
|
|
166b84a81a | ||
|
|
f4097a1ae5 | ||
|
|
8374e889cf | ||
|
|
62deaf7e00 | ||
|
|
a4cd68b1ad | ||
|
|
dde28f21fa | ||
|
|
a4c040b314 | ||
|
|
87995625ca | ||
|
|
e717854909 | ||
|
|
9e37f2cb17 | ||
|
|
fc08db08f3 | ||
|
|
270bc793ba | ||
|
|
ced2d339ee | ||
|
|
43acee5a29 | ||
|
|
723098994b | ||
|
|
684159abd1 | ||
|
|
dbb482553f | ||
|
|
c99a43559d | ||
|
|
3cbe56861f | ||
|
|
2ef9aad41c | ||
|
|
087f4a526b | ||
|
|
f876cf6e34 | ||
|
|
75fbd56c67 | ||
|
|
0c326e56ba | ||
|
|
2068ba7314 | ||
|
|
9841973b16 | ||
|
|
1c44000933 | ||
|
|
f71e1aebed | ||
|
|
252fe184b6 | ||
|
|
8034148313 | ||
|
|
f6746e8f84 | ||
|
|
a48817300b | ||
|
|
a74a820e32 | ||
|
|
e0f9333d2f | ||
|
|
34d4feae27 | ||
|
|
f09ff1f070 | ||
|
|
9a89c3f002 | ||
|
|
f1e55b8ac7 | ||
|
|
d908fb13b3 | ||
|
|
3c02e525bc | ||
|
|
2cf04e4ab0 | ||
|
|
24a549dc54 | ||
|
|
63749970c6 | ||
|
|
0d1b4f384a | ||
|
|
8517670ac2 | ||
|
|
6157287549 | ||
|
|
041345ad3d | ||
|
|
7ed865cb4a | ||
|
|
b4d59960c0 | ||
|
|
611b32e965 | ||
|
|
fc94c02058 | ||
|
|
853ca3c252 | ||
|
|
74e378fe4d | ||
|
|
a4d1288b00 | ||
|
|
91d27dd231 | ||
|
|
6d90635476 | ||
|
|
e9be67efb4 | ||
|
|
71d1f2d8e6 | ||
|
|
56233d52eb | ||
|
|
5b5c3ba439 | ||
|
|
a0c6acf4e4 | ||
|
|
bf54c123c1 | ||
|
|
6a30fed1c6 | ||
|
|
8c2f8e6247 | ||
|
|
593d1a087d | ||
|
|
633f76f64a | ||
|
|
ec89fabc5d | ||
|
|
26c02ad49c | ||
|
|
e1841e016f | ||
|
|
8ebfb3d8a2 | ||
|
|
4c0320c565 | ||
|
|
5c1f2202cd | ||
|
|
e1abbecabd | ||
|
|
52466da955 | ||
|
|
ede7ad49bf | ||
|
|
c61acee8e8 | ||
|
|
7ea0a368af | ||
|
|
bb07b3508d | ||
|
|
4df24cfde5 | ||
|
|
47c06a77bd | ||
|
|
7a95b1e1da | ||
|
|
b09f9cbf5a | ||
|
|
780880cf07 | ||
|
|
eb6f93bd9c | ||
|
|
a0d4a2e599 | ||
|
|
8458443a2e | ||
|
|
feafc1b7d3 | ||
|
|
fcd54180bc | ||
|
|
d73c3ec8d3 | ||
|
|
ca7210b26f | ||
|
|
d9e22324d4 | ||
|
|
0a18e59274 | ||
|
|
ffc5f81fb5 | ||
|
|
b4668ed851 | ||
|
|
999ff06c58 | ||
|
|
e494b2c215 | ||
|
|
2b2022ae85 | ||
|
|
3afa137130 | ||
|
|
0b88ddf438 | ||
|
|
f878bd1317 | ||
|
|
72075b29de | ||
|
|
edf1a0b879 | ||
|
|
df763ff09c | ||
|
|
83e09bcf47 | ||
|
|
10e9aef044 | ||
|
|
45a0c2eb3d | ||
|
|
99c2a75964 | ||
|
|
9cc61eb851 | ||
|
|
62e51b8cf6 | ||
|
|
b147560a80 | ||
|
|
97a0b1f44c | ||
|
|
8c1e6647e0 | ||
|
|
8432327127 | ||
|
|
231f8e8f0c | ||
|
|
28e43eb74c | ||
|
|
e2d3647d76 | ||
|
|
876509d9e5 | ||
|
|
1647a79198 | ||
|
|
622d8a89a7 | ||
|
|
d5e66e10d6 | ||
|
|
5b1776710f | ||
|
|
7e7e3b77d4 | ||
|
|
28ecd9600a | ||
|
|
1ef6834a31 | ||
|
|
b88f3bed40 | ||
|
|
eaa885f5f5 | ||
|
|
f7f48c2509 | ||
|
|
07344e89b1 | ||
|
|
e6bc717612 | ||
|
|
11521fc24d | ||
|
|
213ec6027d | ||
|
|
259d832d37 | ||
|
|
62deb83520 | ||
|
|
edc4cd7873 | ||
|
|
05b75fbee0 | ||
|
|
d6065e8b89 | ||
|
|
617d129542 | ||
|
|
057424b269 | ||
|
|
63d1b85a6c | ||
|
|
9481c3b424 | ||
|
|
15e174fc2f | ||
|
|
5efd4ec0a4 | ||
|
|
895b61fd62 | ||
|
|
5d4125208c | ||
|
|
ff3bfe75ec | ||
|
|
89c31fdbc2 | ||
|
|
9445c7ca78 | ||
|
|
b3d5833520 | ||
|
|
20ed40510a | ||
|
|
9c5c57f134 | ||
|
|
09b916aac9 | ||
|
|
92843fef28 | ||
|
|
433951afda | ||
|
|
aa1f861e4e | ||
|
|
66b6a09400 | ||
|
|
edf0dc8399 | ||
|
|
8d0c1219ca | ||
|
|
33b3f69076 | ||
|
|
f582941ad3 | ||
|
|
9fd57b4fbf | ||
|
|
f61f69c238 | ||
|
|
55c626d627 | ||
|
|
c6d3257916 | ||
|
|
34f91e9872 | ||
|
|
5acd509fc0 | ||
|
|
765756b002 | ||
|
|
64525744de | ||
|
|
2b89f62add | ||
|
|
0a67adf3e7 | ||
|
|
4ef7296e37 | ||
|
|
0cef46cd78 | ||
|
|
243d72772c | ||
|
|
91214cc6a6 | ||
|
|
da85e80689 | ||
|
|
e22827d380 | ||
|
|
7161643574 | ||
|
|
77f1c6cd0c | ||
|
|
0a4b56eb28 | ||
|
|
b5f94d78f3 | ||
|
|
a9fd1e1832 | ||
|
|
762a7e3489 | ||
|
|
0f3717db0d | ||
|
|
8590ab84ba | ||
|
|
9d17903392 | ||
|
|
7d94831c01 | ||
|
|
b97900f5b8 | ||
|
|
887947a967 | ||
|
|
85661c0e58 | ||
|
|
fc2ba97792 | ||
|
|
4b6cd3fd2b | ||
|
|
d758821817 | ||
|
|
4685d0b1b6 | ||
|
|
e88946f0b5 | ||
|
|
f14c611aac | ||
|
|
84ccdc24ed | ||
|
|
04a0366af5 | ||
|
|
1a956c909c | ||
|
|
8e150701cb | ||
|
|
b9d8b5026e | ||
|
|
a6dfc58c72 | ||
|
|
6266cd8f0b | ||
|
|
e6e6328594 | ||
|
|
8b8826437a | ||
|
|
b77c51512e | ||
|
|
10d2299fae | ||
|
|
83e1eb50bd | ||
|
|
c7546bcafd | ||
|
|
b26a537181 | ||
|
|
e5cbca7f95 | ||
|
|
73c7a554c2 | ||
|
|
c460029704 | ||
|
|
2808157200 | ||
|
|
79913e3816 | ||
|
|
c9b79f4277 | ||
|
|
5897db3880 | ||
|
|
4a4d08d9f1 | ||
|
|
f383a3f50f | ||
|
|
79954cf188 | ||
|
|
f87adc209a | ||
|
|
3a71bfd763 | ||
|
|
c8ff61057e | ||
|
|
10d5975e26 | ||
|
|
7b34c6fc32 | ||
|
|
1ac6e51271 | ||
|
|
8435080ec7 | ||
|
|
3b474e596d | ||
|
|
16b40917c6 | ||
|
|
d0e5a6bfe3 | ||
|
|
0048dbfae2 | ||
|
|
bc612a8384 | ||
|
|
541765f8a0 | ||
|
|
7b4663e373 | ||
|
|
5eb792e2f5 | ||
|
|
5499da8cc3 | ||
|
|
35b212954c | ||
|
|
11c791ae38 | ||
|
|
4ab1cbd03c | ||
|
|
d33fb015d5 | ||
|
|
5b1704c95e | ||
|
|
03659f3e82 | ||
|
|
cf3b7f4603 | ||
|
|
07c6c95eff | ||
|
|
74cbe04fe9 | ||
|
|
05aa62aa9e | ||
|
|
4acb01086b | ||
|
|
d12941576b | ||
|
|
d99f12fe64 | ||
|
|
187e64c8d2 | ||
|
|
de152e4728 | ||
|
|
d3669eeef2 | ||
|
|
38a453eebb | ||
|
|
53dc3c3473 | ||
|
|
b573d9035f | ||
|
|
bc0485c91a | ||
|
|
63fa223e54 | ||
|
|
06b3839fd2 | ||
|
|
254c399ec6 | ||
|
|
6567e93a96 | ||
|
|
7f6de12256 | ||
|
|
5090e521a1 | ||
|
|
81847e7bc5 | ||
|
|
b4fff64169 | ||
|
|
3cba71d5be | ||
|
|
ce6a96f3d1 | ||
|
|
f7a00ff0eb | ||
|
|
56306bb456 | ||
|
|
eab3268386 | ||
|
|
f68dfac8b4 | ||
|
|
bb76ae4aea | ||
|
|
7babcbd170 | ||
|
|
ef12dd1b5f | ||
|
|
6200a9844f | ||
|
|
1e5717d13d | ||
|
|
c97ed69874 | ||
|
|
c258776a68 | ||
|
|
892ff766e8 | ||
|
|
80470cb812 | ||
|
|
5949a659a9 | ||
|
|
151d305b02 | ||
|
|
74819a1d1f | ||
|
|
45edcde550 | ||
|
|
2b04b9d9b8 | ||
|
|
3150c655fd | ||
|
|
26375885a9 | ||
|
|
b584684593 | ||
|
|
6f00b67055 |
@@ -0,0 +1,92 @@
|
||||
# When adding additional environment variables, the schema in "/src/env.mjs"
|
||||
# should be updated accordingly.
|
||||
|
||||
# Prisma
|
||||
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
|
||||
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
|
||||
# Clickhouse
|
||||
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
|
||||
CLICKHOUSE_URL="http://localhost:8123"
|
||||
CLICKHOUSE_USER="clickhouse"
|
||||
CLICKHOUSE_PASSWORD="clickhouse"
|
||||
CLICKHOUSE_MIGRATION_CLUSTER_DISABLED="true"
|
||||
|
||||
# Next Auth
|
||||
# You can generate a new secret on the command line with:
|
||||
# openssl rand -base64 32
|
||||
# https://next-auth.js.org/configuration/options#secret
|
||||
# NEXTAUTH_SECRET=""
|
||||
NEXTAUTH_URL="http://localhost:3000"
|
||||
NEXTAUTH_SECRET="secret"
|
||||
|
||||
# Langfuse Cloud Environment
|
||||
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
|
||||
|
||||
# Langfuse experimental features
|
||||
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
|
||||
|
||||
# Salt for API key hashing
|
||||
SALT="salt"
|
||||
|
||||
# Email
|
||||
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
|
||||
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
|
||||
|
||||
# DON'T PANIC: The Azurite Secrets are well-known and meant to be hard-coded
|
||||
# S3 storage
|
||||
S3_ENDPOINT=http://localhost:10000/devstoreaccount1
|
||||
S3_ACCESS_KEY_ID=devstoreaccount1
|
||||
S3_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
|
||||
S3_BUCKET_NAME=langfuse
|
||||
S3_REGION=auto
|
||||
## Necessary for minio compatibility
|
||||
S3_FORCE_PATH_STYLE=true
|
||||
|
||||
# S3 Media Upload LOCAL
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_REGION=auto
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
|
||||
|
||||
# S3 Event Bucket Upload
|
||||
## Set to true to test uploading all events to S3
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION=auto
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
|
||||
|
||||
LANGFUSE_USE_AZURE_BLOB=true
|
||||
|
||||
# Set during docker build of application
|
||||
# Used to disable environment verification at build time
|
||||
# DOCKER_BUILD=1
|
||||
|
||||
REDIS_HOST="127.0.0.1"
|
||||
REDIS_PORT=6379
|
||||
REDIS_AUTH="myredissecret"
|
||||
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
|
||||
# speeds up local development by not executing init scripts on server startup
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
|
||||
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
|
||||
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
|
||||
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
|
||||
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
|
||||
|
||||
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
|
||||
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
|
||||
+24
-5
@@ -11,6 +11,7 @@ CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
|
||||
CLICKHOUSE_URL="http://localhost:8123"
|
||||
CLICKHOUSE_USER="clickhouse"
|
||||
CLICKHOUSE_PASSWORD="clickhouse"
|
||||
CLICKHOUSE_CLUSTER_DISABLED="true"
|
||||
|
||||
# Next Auth
|
||||
# You can generate a new secret on the command line with:
|
||||
@@ -37,15 +38,26 @@ SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
|
||||
S3_ENDPOINT=http://localhost:9090
|
||||
S3_ACCESS_KEY_ID=minio
|
||||
S3_SECRET_ACCESS_KEY=miniosecret
|
||||
S3_BUCKET_NAME=mybucket
|
||||
S3_BUCKET_NAME=langfuse
|
||||
S3_REGION=us-east-1
|
||||
## Necessary for minio compatibility
|
||||
S3_FORCE_PATH_STYLE=true
|
||||
|
||||
# S3 Media Upload LOCAL
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
|
||||
|
||||
# S3 Event Bucket Upload
|
||||
## Set to true to test uploading all events to S3
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=false
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=mybucket
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
|
||||
@@ -62,9 +74,16 @@ REDIS_HOST="127.0.0.1"
|
||||
REDIS_PORT=6379
|
||||
REDIS_AUTH="myredissecret"
|
||||
|
||||
LANGFUSE_WORKER_PASSWORD=mybasicauthsecret
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
|
||||
# speeds up local development by not executing init scripts on server startup
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
|
||||
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
|
||||
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
|
||||
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
|
||||
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
|
||||
|
||||
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
|
||||
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# When adding additional environment variables, the schema in "/src/env.mjs"
|
||||
# should be updated accordingly.
|
||||
|
||||
# Prisma
|
||||
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
|
||||
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
|
||||
# Clickhouse
|
||||
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
|
||||
CLICKHOUSE_URL="http://localhost:8123"
|
||||
CLICKHOUSE_USER="clickhouse"
|
||||
CLICKHOUSE_PASSWORD="clickhouse"
|
||||
CLICKHOUSE_CLUSTER_ENABLED="false"
|
||||
|
||||
# Next Auth
|
||||
# You can generate a new secret on the command line with:
|
||||
# openssl rand -base64 32
|
||||
# https://next-auth.js.org/configuration/options#secret
|
||||
# NEXTAUTH_SECRET=""
|
||||
NEXTAUTH_URL="http://localhost:3000"
|
||||
NEXTAUTH_SECRET="secret"
|
||||
|
||||
# Langfuse Cloud Environment
|
||||
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
|
||||
|
||||
# Langfuse experimental features
|
||||
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
|
||||
|
||||
# Salt for API key hashing
|
||||
SALT="salt"
|
||||
|
||||
# Email
|
||||
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
|
||||
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
|
||||
|
||||
# S3 storage
|
||||
S3_ENDPOINT=http://localhost:9090
|
||||
S3_ACCESS_KEY_ID=minio
|
||||
S3_SECRET_ACCESS_KEY=miniosecret
|
||||
S3_BUCKET_NAME=langfuse
|
||||
S3_REGION=us-east-1
|
||||
## Necessary for minio compatibility
|
||||
S3_FORCE_PATH_STYLE=true
|
||||
|
||||
# # S3 Media Upload LOCAL
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
|
||||
|
||||
# S3 Event Bucket Upload
|
||||
## Set to true to test uploading all events to S3
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
|
||||
|
||||
# Set during docker build of application
|
||||
# Used to disable environment verification at build time
|
||||
# DOCKER_BUILD=1
|
||||
|
||||
REDIS_HOST="127.0.0.1"
|
||||
REDIS_PORT=6379
|
||||
REDIS_AUTH="myredissecret"
|
||||
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
|
||||
# speeds up local development by not executing init scripts on server startup
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
@@ -18,7 +18,5 @@ REDIS_HOST="127.0.0.1"
|
||||
REDIS_PORT=6379
|
||||
REDIS_AUTH="myredissecret"
|
||||
|
||||
LANGFUSE_WORKER_PASSWORD=myworkerpassword
|
||||
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
+12
-2
@@ -55,6 +55,7 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
|
||||
# Auth, optional configuration
|
||||
# AUTH_DOMAINS_WITH_SSO_ENFORCEMENT=domain1.com,domain2.com
|
||||
# AUTH_IGNORE_ACCOUNT_FIELDS=foo,bar
|
||||
# AUTH_DISABLE_USERNAME_PASSWORD=true
|
||||
# AUTH_DISABLE_SIGNUP=true
|
||||
# AUTH_SESSION_MAX_AGE=43200 # 30 days in minutes (default)
|
||||
@@ -67,6 +68,10 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
# AUTH_GITHUB_CLIENT_ID=
|
||||
# AUTH_GITHUB_CLIENT_SECRET=
|
||||
# AUTH_GITHUB_ALLOW_ACCOUNT_LINKING=false
|
||||
# AUTH_GITHUB_ENTERPRISE_CLIENT_ID=
|
||||
# AUTH_GITHUB_ENTERPRISE_CLIENT_SECRET=
|
||||
# AUTH_GITHUB_ENTERPRISE_BASE_URL=
|
||||
# AUTH_GITHUB_ENTERPRISE_ALLOW_ACCOUNT_LINKING=false
|
||||
# AUTH_GITLAB_CLIENT_ID=
|
||||
# AUTH_GITLAB_CLIENT_SECRET=
|
||||
# AUTH_GITLAB_ALLOW_ACCOUNT_LINKING=false
|
||||
@@ -87,12 +92,17 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
# AUTH_COGNITO_CLIENT_SECRET=
|
||||
# AUTH_COGNITO_ISSUER=
|
||||
# AUTH_COGNITO_ALLOW_ACCOUNT_LINKING=false
|
||||
# AUTH_KEYCLOAK_CLIENT_ID=
|
||||
# AUTH_KEYCLOAK_CLIENT_SECRET=
|
||||
# AUTH_KEYCLOAK_ISSUER=
|
||||
# AUTH_KEYCLOAK_ALLOW_ACCOUNT_LINKING=false
|
||||
# AUTH_CUSTOM_CLIENT_ID=
|
||||
# AUTH_CUSTOM_CLIENT_SECRET=
|
||||
# AUTH_CUSTOM_ISSUER=
|
||||
# AUTH_CUSTOM_NAME=
|
||||
# AUTH_CUSTOM_SCOPE="openid email profile" # optional
|
||||
# AUTH_CUSTOM_ALLOW_ACCOUNT_LINKING=false
|
||||
# AUTH_CUSTOM_ID_TOKEN=false # optional, default is true
|
||||
|
||||
# Transactional email, optional
|
||||
# Defines the email address to use as the from address.
|
||||
@@ -237,7 +247,7 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
# CLICKHOUSE_PASSWORD=
|
||||
|
||||
# Ingestion
|
||||
# LANGFUSE_INGESTION_BUFFER_TTL_SECONDS=
|
||||
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=
|
||||
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_BATCH_SIZE=
|
||||
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=
|
||||
# LANGFUSE_INGESTION_CLICKHOUSE_MAX_ATTEMPTS=
|
||||
@@ -245,4 +255,4 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
# LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
|
||||
# QUEUE_CONSUMER_LEGACY_INGESTION_QUEUE_IS_ENABLED="true"
|
||||
|
||||
## END Langfuse V3 Ingestion
|
||||
## END Langfuse V3 Ingestion
|
||||
|
||||
@@ -3,9 +3,14 @@ name: Codespell
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
branches:
|
||||
- "main"
|
||||
tags:
|
||||
- "v*"
|
||||
pull_request:
|
||||
branches: [main]
|
||||
branches:
|
||||
- "**"
|
||||
merge_group:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -19,7 +19,7 @@ on:
|
||||
type: choice
|
||||
options:
|
||||
- staging
|
||||
- prod-eu-temp
|
||||
- prod-eu
|
||||
- prod-us
|
||||
required: true
|
||||
|
||||
@@ -78,7 +78,7 @@ jobs:
|
||||
return `["staging"]`
|
||||
}
|
||||
if (context.ref === "refs/heads/production") {
|
||||
return `["prod-eu-temp", "prod-us"]`
|
||||
return `["prod-eu", "prod-us"]`
|
||||
}
|
||||
}
|
||||
return "[]"
|
||||
|
||||
+111
-10
@@ -66,10 +66,10 @@ jobs:
|
||||
run: |
|
||||
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
|
||||
|
||||
tests-web:
|
||||
tests-web-sync:
|
||||
timeout-minutes: 20
|
||||
runs-on: ubuntu-latest
|
||||
name: tests-web (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
|
||||
name: tests-web-sync (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
|
||||
strategy:
|
||||
matrix:
|
||||
node-version: [20]
|
||||
@@ -80,6 +80,11 @@ jobs:
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- uses: pnpm/action-setup@v3
|
||||
with:
|
||||
version: 9.5.0
|
||||
@@ -100,7 +105,7 @@ jobs:
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev.example .env
|
||||
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.example > .env
|
||||
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.legacy.example > .env
|
||||
- name: Run + migrate
|
||||
run: |
|
||||
docker compose -f docker-compose.dev.yml up -d
|
||||
@@ -111,6 +116,76 @@ jobs:
|
||||
- name: Seed DB
|
||||
run: |
|
||||
pnpm run db:migrate
|
||||
pnpm --filter=shared ch:up
|
||||
- name: Build
|
||||
run: pnpm run build
|
||||
- name: Start Langfuse
|
||||
run: (pnpm run start&)
|
||||
env:
|
||||
LANGFUSE_INIT_ORG_ID: "seed-org-id"
|
||||
LANGFUSE_INIT_ORG_NAME: "Seed Org"
|
||||
LANGFUSE_INIT_PROJECT_ID: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
|
||||
LANGFUSE_INIT_PROJECT_NAME: "Seed Project"
|
||||
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: "pk-lf-1234567890"
|
||||
LANGFUSE_INIT_PROJECT_SECRET_KEY: "sk-lf-1234567890"
|
||||
LANGFUSE_INIT_USER_EMAIL: "demo@langfuse.com"
|
||||
LANGFUSE_INIT_USER_NAME: "Demo User"
|
||||
LANGFUSE_INIT_USER_PASSWORD: "password"
|
||||
- name: run test-sync
|
||||
run: pnpm --filter=web run test-sync
|
||||
|
||||
tests-web-async:
|
||||
timeout-minutes: 20
|
||||
runs-on: ubuntu-latest
|
||||
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
|
||||
strategy:
|
||||
matrix:
|
||||
node-version: [20]
|
||||
postgres-version: [12, 15]
|
||||
blob-provider: ["", "-azure"]
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- uses: pnpm/action-setup@v3
|
||||
with:
|
||||
version: 9.5.0
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN_READ }}
|
||||
- name: Use Node.js ${{ matrix.node-version }}
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: ${{ matrix.node-version }}
|
||||
cache: "pnpm"
|
||||
cache-dependency-path: "pnpm-lock.yaml"
|
||||
- name: install dependencies
|
||||
run: |
|
||||
pnpm install
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev${{ matrix.blob-provider }}.example .env
|
||||
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
|
||||
- name: Run + migrate
|
||||
run: |
|
||||
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
docker compose ps
|
||||
env:
|
||||
POSTGRES_VERSION: ${{ matrix.postgres-version }}
|
||||
- name: Seed DB
|
||||
run: |
|
||||
pnpm run db:migrate
|
||||
pnpm --filter=shared ch:up
|
||||
- name: Build
|
||||
run: pnpm run build
|
||||
- name: Start Langfuse
|
||||
@@ -131,11 +206,12 @@ jobs:
|
||||
tests-worker:
|
||||
timeout-minutes: 20
|
||||
runs-on: ubuntu-latest
|
||||
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
|
||||
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
|
||||
strategy:
|
||||
matrix:
|
||||
node-version: [20]
|
||||
postgres-version: [12, 15]
|
||||
blob-provider: ["", "-azure"]
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
@@ -161,17 +237,17 @@ jobs:
|
||||
pnpm install
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev.example .env
|
||||
cp .env.dev.example web/.env
|
||||
cp .env.dev.example worker/.env
|
||||
cp .env.dev${{ matrix.blob-provider }}.example .env
|
||||
cp .env.dev${{ matrix.blob-provider }}.example web/.env
|
||||
cp .env.dev${{ matrix.blob-provider }}.example worker/.env
|
||||
- name: Run + migrate
|
||||
run: |
|
||||
docker compose -f docker-compose.dev.yml up -d
|
||||
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
docker compose ps
|
||||
- name: Ensure no unhealthy status
|
||||
@@ -252,9 +328,15 @@ jobs:
|
||||
- name: install dependencies
|
||||
run: |
|
||||
pnpm install
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev.example .env
|
||||
echo "LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING=true" >> .env
|
||||
echo "LANGFUSE_ASYNC_INGESTION_PROCESSING=true" >> .env
|
||||
echo "LANGFUSE_CACHE_API_KEY_ENABLED=true" >> .env
|
||||
echo "LANGFUSE_CACHE_PROMPT_ENABLED=true" >> .env
|
||||
@@ -263,14 +345,29 @@ jobs:
|
||||
docker compose -f docker-compose.dev.yml up -d
|
||||
docker compose ps
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
- name: Ensure Docker dependencies are healthy
|
||||
run: |
|
||||
if docker-compose ps | grep "(unhealthy)"; then
|
||||
echo "One or more services are unhealthy"
|
||||
exit 1
|
||||
else
|
||||
echo "All services are healthy"
|
||||
fi
|
||||
- name: Seed DB
|
||||
run: |
|
||||
pnpm run db:migrate
|
||||
pnpm --filter=shared run ch:up
|
||||
pnpm run db:seed:examples
|
||||
- name: Build
|
||||
run: pnpm run build
|
||||
- name: Run server
|
||||
run: (pnpm run start&)
|
||||
- name: Check worker health
|
||||
run: |
|
||||
timeout 10 bash -c 'until curl -f http://localhost:3030/api/health; do sleep 2; done'
|
||||
- name: Check server health
|
||||
run: |
|
||||
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
|
||||
- name: Run e2e tests
|
||||
run: pnpm --filter=web run test:e2e:server
|
||||
|
||||
@@ -280,11 +377,12 @@ jobs:
|
||||
needs:
|
||||
[
|
||||
lint,
|
||||
tests-web,
|
||||
tests-web-sync,
|
||||
tests-worker,
|
||||
e2e-tests,
|
||||
test-docker-build,
|
||||
e2e-server-tests,
|
||||
tests-web-async,
|
||||
]
|
||||
if: always()
|
||||
steps:
|
||||
@@ -343,6 +441,8 @@ jobs:
|
||||
images: |
|
||||
ghcr.io/langfuse/langfuse # GitHub
|
||||
langfuse/langfuse # Docker Hub
|
||||
flavor: |
|
||||
latest=false
|
||||
tags: |
|
||||
type=ref,event=branch
|
||||
type=ref,event=pr
|
||||
@@ -350,6 +450,7 @@ jobs:
|
||||
type=semver,pattern={{version}}
|
||||
type=semver,pattern={{major}}.{{minor}}
|
||||
type=semver,pattern={{major}}
|
||||
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') }}
|
||||
- name: Build and push Docker image (web)
|
||||
uses: docker/build-push-action@v4
|
||||
with:
|
||||
|
||||
@@ -3,7 +3,7 @@ on:
|
||||
push:
|
||||
# Pattern matched against refs/tags
|
||||
tags:
|
||||
- "v[0-9]+.[0-9]+.[0-9]+" # Semantic version tags
|
||||
- "v3.[0-9]+.[0-9]+" # Semantic version tags
|
||||
|
||||
jobs:
|
||||
release:
|
||||
|
||||
+6
-3
@@ -35,9 +35,12 @@ yarn-error.log*
|
||||
|
||||
# local env files
|
||||
# do not commit any .env files to git, except for the .env.example file. https://create.t3.gg/en/usage/env-variables#using-environment-variables
|
||||
.env
|
||||
.env.local
|
||||
.env*.local
|
||||
.env*
|
||||
!.env.dev.example
|
||||
!.env.dev-azure.example
|
||||
!.env.local.example
|
||||
!.env.prod.example
|
||||
!.env.dev.legacy.example
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
Vendored
+1
@@ -20,6 +20,7 @@
|
||||
"prettier.documentSelectors": [
|
||||
"**/*.{cjs,mjs,ts,tsx,astro,md,mdx,json,yaml,yml}"
|
||||
],
|
||||
"prettier.trailingComma": "all",
|
||||
"mdx.experimentalLanguageServer": true,
|
||||
"typescript.preferences.importModuleSpecifier": "non-relative",
|
||||
"docwriter.style": "JSDoc",
|
||||
|
||||
@@ -432,6 +432,20 @@ Example:
|
||||
op run --env-file="./.env" -- pnpm --filter=shared run db:deploy
|
||||
```
|
||||
|
||||
### Editing default models and prices
|
||||
|
||||
You can update the default AI models and prices by adding or updating an entry in `worker/src/constants/default-model-prices.json`.
|
||||
|
||||
Please note that
|
||||
|
||||
- prices are in USD
|
||||
- the list is ordered by ID, so make sure to keep this order
|
||||
- the `updated_at` field must be updated with the current date in ISO 8601 format. Otherwise, the change will be ignored.
|
||||
|
||||
### Transition period until V3 release
|
||||
|
||||
Until the V3 release, both the JSON record must be updated **and** a migration must be created to continue supporting self-hosted users. Note that the migration must updated both the `models` as well as the `prices` table accordingly.
|
||||
|
||||
## License
|
||||
|
||||
Langfuse is MIT licensed, except for `ee/` folder. See [LICENSE](LICENSE) and [docs](https://langfuse.com/docs/open-source) for more details.
|
||||
|
||||
@@ -181,3 +181,13 @@ This helps us to:
|
||||
None of the data is shared with third parties and does not include any sensitive information. We want to be super transparent about this and you can find the exact data we collect [here](/web/src/features/telemetry/index.ts).
|
||||
|
||||
You can opt-out by setting `TELEMETRY_ENABLED=false`.
|
||||
|
||||
### Star History
|
||||
|
||||
<a href="https://star-history.com/#langfuse/langfuse&Date">
|
||||
<picture>
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date&theme=dark" />
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date" />
|
||||
<img alt="Star History Chart" src="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date" />
|
||||
</picture>
|
||||
</a>
|
||||
|
||||
@@ -20,8 +20,6 @@ services:
|
||||
- NEXTAUTH_URL=http://localhost:3000
|
||||
- TELEMETRY_ENABLED=${TELEMETRY_ENABLED:-true}
|
||||
- LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES=${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-false}
|
||||
- LANGFUSE_WORKER_HOST=${LANGFUSE_WORKER_HOST:-worker}
|
||||
- LANGFUSE_WORKER_PASSWORD=${LANGFUSE_WORKER_PASSWORD:-mybasicauthsecret}
|
||||
- LANGFUSE_INIT_ORG_ID=${LANGFUSE_INIT_ORG_ID:-}
|
||||
- LANGFUSE_INIT_ORG_NAME=${LANGFUSE_INIT_ORG_NAME:-}
|
||||
- LANGFUSE_INIT_PROJECT_ID=${LANGFUSE_INIT_PROJECT_ID:-}
|
||||
@@ -58,7 +56,6 @@ services:
|
||||
- REDIS_HOST=${REDIS_HOST:-redis}
|
||||
- REDIS_PORT=${REDIS_PORT:-6379}
|
||||
- REDIS_AUTH=${REDIS_AUTH:-myredissecret}
|
||||
- LANGFUSE_WORKER_PASSWORD=${LANGFUSE_WORKER_PASSWORD:-mybasicauthsecret}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:3030/api/health"]
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
services:
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server
|
||||
user: "101:101"
|
||||
container_name: clickhouse
|
||||
hostname: clickhouse
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
CLICKHOUSE_USER: clickhouse
|
||||
CLICKHOUSE_PASSWORD: clickhouse
|
||||
volumes:
|
||||
- langfuse_clickhouse_data:/var/lib/clickhouse
|
||||
- langfuse_clickhouse_logs:/var/log/clickhouse-server
|
||||
ports:
|
||||
- "8123:8123"
|
||||
- "9000:9000"
|
||||
depends_on:
|
||||
- postgres
|
||||
|
||||
azurite:
|
||||
image: mcr.microsoft.com/azure-storage/azurite
|
||||
container_name: azurite
|
||||
command: azurite-blob --blobHost 0.0.0.0
|
||||
ports:
|
||||
- "10000:10000"
|
||||
volumes:
|
||||
- langfuse_azurite_data:/data
|
||||
|
||||
redis:
|
||||
image: redis:7.2.4
|
||||
restart: always
|
||||
command: >
|
||||
--requirepass ${REDIS_AUTH:-myredissecret}
|
||||
ports:
|
||||
- 6379:6379
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
interval: 3s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
command: ["postgres", "-c", "log_statement=all"]
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
- POSTGRES_PASSWORD=postgres
|
||||
- POSTGRES_DB=postgres
|
||||
ports:
|
||||
- 5432:5432
|
||||
volumes:
|
||||
- langfuse_postgres_data:/var/lib/postgresql/data
|
||||
|
||||
volumes:
|
||||
langfuse_postgres_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_logs:
|
||||
driver: local
|
||||
langfuse_azurite_data:
|
||||
driver: local
|
||||
+10
-16
@@ -20,7 +20,9 @@ services:
|
||||
minio:
|
||||
image: minio/minio
|
||||
container_name: minio
|
||||
command: server /data --console-address ":9001"
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
|
||||
environment:
|
||||
MINIO_ACCESS_KEY: minio
|
||||
MINIO_SECRET_KEY: miniosecret
|
||||
@@ -31,23 +33,10 @@ services:
|
||||
- langfuse_minio_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "mc", "ready", "local"]
|
||||
interval: 10s
|
||||
interval: 1s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 5s
|
||||
|
||||
miniocreatebucket:
|
||||
image: minio/mc
|
||||
container_name: miniocreatebucket
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command: >
|
||||
"mc alias set minio http://minio:9000 minio miniosecret &&
|
||||
mc rm -r --force minio/mybucket || true &&
|
||||
mc mb minio/mybucket &&
|
||||
mc policy set download minio/mybucket"
|
||||
depends_on:
|
||||
minio:
|
||||
condition: service_healthy
|
||||
start_period: 1s
|
||||
|
||||
redis:
|
||||
image: redis:7.2.4
|
||||
@@ -60,6 +49,11 @@ services:
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
interval: 3s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
command: ["postgres", "-c", "log_statement=all"]
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
services:
|
||||
langfuse-worker:
|
||||
image: langfuse/langfuse-worker:latest
|
||||
depends_on: &langfuse-depends-on
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
minio:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
ports:
|
||||
- "3030:3030"
|
||||
environment: &langfuse-worker-env
|
||||
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres
|
||||
SALT: "mysalt"
|
||||
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # generate via `openssl rand -hex 32`
|
||||
TELEMETRY_ENABLED: ${TELEMETRY_ENABLED:-true}
|
||||
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-true}
|
||||
LANGFUSE_ASYNC_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_INGESTION_PROCESSING:-true}
|
||||
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING:-true}
|
||||
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE: ${LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE:-true}
|
||||
LANGFUSE_READ_FROM_POSTGRES_ONLY: ${LANGFUSE_READ_FROM_POSTGRES_ONLY:-false}
|
||||
LANGFUSE_RETURN_FROM_CLICKHOUSE: ${LANGFUSE_RETURN_FROM_CLICKHOUSE:-true}
|
||||
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
|
||||
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
|
||||
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: ${LANGFUSE_S3_EVENT_UPLOAD_ENABLED:-true}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-us-east-1}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED: ${LANGFUSE_S3_MEDIA_UPLOAD_ENABLED:-true}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET: ${LANGFUSE_S3_MEDIA_UPLOAD_BUCKET:-langfuse}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_REGION: ${LANGFUSE_S3_MEDIA_UPLOAD_REGION:-us-east-1}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID:-minio}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT: ${LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT:-http://minio:9000}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE:-true}
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX: ${LANGFUSE_S3_MEDIA_UPLOAD_PREFIX:-media/}
|
||||
REDIS_HOST: ${REDIS_HOST:-redis}
|
||||
REDIS_PORT: ${REDIS_PORT:-6379}
|
||||
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
|
||||
|
||||
langfuse-web:
|
||||
image: langfuse/langfuse:latest
|
||||
depends_on: *langfuse-depends-on
|
||||
ports:
|
||||
- "3000:3000"
|
||||
environment:
|
||||
<<: *langfuse-worker-env
|
||||
NEXTAUTH_URL: http://localhost:3000
|
||||
NEXTAUTH_SECRET: mysecret
|
||||
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
|
||||
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
|
||||
LANGFUSE_INIT_PROJECT_ID: ${LANGFUSE_INIT_PROJECT_ID:-}
|
||||
LANGFUSE_INIT_PROJECT_NAME: ${LANGFUSE_INIT_PROJECT_NAME:-}
|
||||
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: ${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
|
||||
LANGFUSE_INIT_PROJECT_SECRET_KEY: ${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
|
||||
LANGFUSE_INIT_USER_EMAIL: ${LANGFUSE_INIT_USER_EMAIL:-}
|
||||
LANGFUSE_INIT_USER_NAME: ${LANGFUSE_INIT_USER_NAME:-}
|
||||
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
|
||||
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server
|
||||
user: "101:101"
|
||||
container_name: clickhouse
|
||||
hostname: clickhouse
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
CLICKHOUSE_USER: clickhouse
|
||||
CLICKHOUSE_PASSWORD: clickhouse
|
||||
volumes:
|
||||
- langfuse_clickhouse_data:/var/lib/clickhouse
|
||||
- langfuse_clickhouse_logs:/var/log/clickhouse-server
|
||||
ports:
|
||||
- "8123:8123"
|
||||
- "9000:9000"
|
||||
depends_on:
|
||||
- postgres
|
||||
|
||||
minio:
|
||||
image: minio/minio
|
||||
container_name: minio
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
|
||||
environment:
|
||||
MINIO_ROOT_USER: minio
|
||||
MINIO_ROOT_PASSWORD: miniosecret
|
||||
ports:
|
||||
- "9090:9000"
|
||||
- "9091:9001"
|
||||
volumes:
|
||||
- langfuse_minio_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "mc", "ready", "local"]
|
||||
interval: 1s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 1s
|
||||
|
||||
redis:
|
||||
image: redis:7
|
||||
restart: always
|
||||
command: >
|
||||
--requirepass ${REDIS_AUTH:-myredissecret}
|
||||
ports:
|
||||
- 6379:6379
|
||||
healthcheck:
|
||||
test: ["CMD", "redis-cli", "ping"]
|
||||
interval: 3s
|
||||
timeout: 10s
|
||||
retries: 10
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
interval: 3s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
environment:
|
||||
POSTGRES_USER: postgres
|
||||
POSTGRES_PASSWORD: postgres
|
||||
POSTGRES_DB: postgres
|
||||
ports:
|
||||
- 5432:5432
|
||||
volumes:
|
||||
- langfuse_postgres_data:/var/lib/postgresql/data
|
||||
|
||||
volumes:
|
||||
langfuse_postgres_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_logs:
|
||||
driver: local
|
||||
langfuse_minio_data:
|
||||
driver: local
|
||||
+4
-3
@@ -27,8 +27,9 @@
|
||||
"@langfuse/shared": "workspace:*",
|
||||
"@opentelemetry/api": ">=1.0.0 <1.10.0",
|
||||
"axios": "^1.7.7",
|
||||
"next": "^14.2.15",
|
||||
"next-auth": "^4.24.7",
|
||||
"https-proxy-agent": "^7.0.6",
|
||||
"next": "^14.2.21",
|
||||
"next-auth": "^4.24.11",
|
||||
"zod": "^3.23.8"
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -47,7 +48,7 @@
|
||||
},
|
||||
"pnpm": {
|
||||
"overrides": {
|
||||
"jsonpath-plus": "10.0.0"
|
||||
"jsonpath-plus": "10.2.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +0,0 @@
|
||||
{
|
||||
"trailingComma": "es5",
|
||||
"printWidth": 120
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
|
||||
imports:
|
||||
pagination: ./utils/pagination.yml
|
||||
commons: ./commons.yml
|
||||
service:
|
||||
auth: true
|
||||
base-path: /api/public
|
||||
endpoints:
|
||||
create:
|
||||
docs: Create a comment. Comments may be attached to different object types (trace, observation, session, prompt).
|
||||
method: POST
|
||||
path: /comments
|
||||
request: CreateCommentRequest
|
||||
response: CreateCommentResponse
|
||||
get:
|
||||
docs: Get all comments
|
||||
method: GET
|
||||
path: /comments
|
||||
request:
|
||||
name: GetCommentsRequest
|
||||
query-parameters:
|
||||
page:
|
||||
type: optional<integer>
|
||||
docs: Page number, starts at 1.
|
||||
limit:
|
||||
type: optional<integer>
|
||||
docs: Limit of items per page. If you encounter api issues due to too large page sizes, try to reduce the limit
|
||||
objectType:
|
||||
type: optional<string>
|
||||
docs: Filter comments by object type (trace, observation, session, prompt).
|
||||
objectId:
|
||||
type: optional<string>
|
||||
docs: Filter comments by object id. If objectType is not provided, an error will be thrown.
|
||||
authorUserId:
|
||||
type: optional<string>
|
||||
docs: Filter comments by author user id.
|
||||
response: GetCommentsResponse
|
||||
get-by-id:
|
||||
docs: Get a comment by id
|
||||
method: GET
|
||||
path: /comments/{commentId}
|
||||
path-parameters:
|
||||
commentId:
|
||||
type: string
|
||||
docs: The unique langfuse identifier of a comment
|
||||
response: commons.Comment
|
||||
types:
|
||||
CreateCommentRequest:
|
||||
properties:
|
||||
projectId:
|
||||
type: string
|
||||
docs: The id of the project to attach the comment to.
|
||||
objectType:
|
||||
type: string
|
||||
docs: The type of the object to attach the comment to (trace, observation, session, prompt).
|
||||
objectId:
|
||||
type: string
|
||||
docs: The id of the object to attach the comment to. If this does not reference a valid existing object, an error will be thrown.
|
||||
content:
|
||||
type: string
|
||||
docs: The content of the comment. May include markdown. Currently limited to 500 characters.
|
||||
authorUserId:
|
||||
type: optional<string>
|
||||
docs: The id of the user who created the comment.
|
||||
CreateCommentResponse:
|
||||
properties:
|
||||
id:
|
||||
type: string
|
||||
docs: The id of the created object in Langfuse
|
||||
GetCommentsResponse:
|
||||
properties:
|
||||
data: list<commons.Comment>
|
||||
meta: pagination.MetaResponse
|
||||
@@ -240,6 +240,9 @@ types:
|
||||
configId:
|
||||
type: optional<string>
|
||||
docs: Reference a score config on a score. When set, config and score name must be equal and value must comply to optionally defined numerical range
|
||||
queueId:
|
||||
type: optional<string>
|
||||
docs: Reference an annotation queue on a score. Populated if the score was initially created in an annotation queue.
|
||||
NumericScore:
|
||||
extends: BaseScore
|
||||
properties:
|
||||
@@ -283,6 +286,18 @@ types:
|
||||
- double
|
||||
- string
|
||||
docs: The value of the score. Must be passed as string for categorical scores, and numeric for boolean and numeric scores
|
||||
|
||||
Comment:
|
||||
properties:
|
||||
id: string
|
||||
projectId: string
|
||||
createdAt: datetime
|
||||
updatedAt: datetime
|
||||
objectType: CommentObjectType
|
||||
objectId: string
|
||||
content: string
|
||||
authorUserId: optional<string>
|
||||
|
||||
Dataset:
|
||||
properties:
|
||||
id: string
|
||||
@@ -402,6 +417,12 @@ types:
|
||||
- optional<integer>
|
||||
- optional<boolean>
|
||||
- optional<list<string>>
|
||||
CommentObjectType:
|
||||
enum:
|
||||
- TRACE
|
||||
- OBSERVATION
|
||||
- SESSION
|
||||
- PROMPT
|
||||
DatasetStatus:
|
||||
enum:
|
||||
- ACTIVE
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
|
||||
imports:
|
||||
commons: ./commons.yml
|
||||
|
||||
service:
|
||||
auth: true
|
||||
base-path: /api/public
|
||||
endpoints:
|
||||
get:
|
||||
docs: Get a media record
|
||||
method: GET
|
||||
path: /media/{mediaId}
|
||||
path-parameters:
|
||||
mediaId:
|
||||
type: string
|
||||
docs: The unique langfuse identifier of a media record
|
||||
response: GetMediaResponse
|
||||
|
||||
patch:
|
||||
docs: Patch a media record
|
||||
method: PATCH
|
||||
path: /media/{mediaId}
|
||||
path-parameters:
|
||||
mediaId:
|
||||
type: string
|
||||
docs: The unique langfuse identifier of a media record
|
||||
request: PatchMediaBody
|
||||
|
||||
getUploadUrl:
|
||||
docs: Get a presigned upload URL for a media record
|
||||
method: POST
|
||||
path: /media
|
||||
request: GetMediaUploadUrlRequest
|
||||
response: GetMediaUploadUrlResponse
|
||||
|
||||
types:
|
||||
GetMediaResponse:
|
||||
properties:
|
||||
mediaId:
|
||||
type: string
|
||||
docs: The unique langfuse identifier of a media record
|
||||
contentType:
|
||||
type: string
|
||||
docs: The MIME type of the media record
|
||||
contentLength:
|
||||
type: integer
|
||||
docs: The size of the media record in bytes
|
||||
uploadedAt:
|
||||
type: datetime
|
||||
docs: The date and time when the media record was uploaded
|
||||
url:
|
||||
type: string
|
||||
docs: The download URL of the media record
|
||||
urlExpiry:
|
||||
type: string
|
||||
docs: The expiry date and time of the media record download URL
|
||||
|
||||
PatchMediaBody:
|
||||
properties:
|
||||
uploadedAt:
|
||||
type: datetime
|
||||
docs: The date and time when the media record was uploaded
|
||||
uploadHttpStatus:
|
||||
type: integer
|
||||
docs: The HTTP status code of the upload
|
||||
uploadHttpError:
|
||||
type: optional<string>
|
||||
docs: The HTTP error message of the upload
|
||||
uploadTimeMs:
|
||||
type: optional<integer>
|
||||
docs: The time in milliseconds it took to upload the media record
|
||||
|
||||
GetMediaUploadUrlRequest:
|
||||
properties:
|
||||
traceId:
|
||||
type: string
|
||||
docs: The trace ID associated with the media record
|
||||
observationId:
|
||||
type: optional<string>
|
||||
docs: The observation ID associated with the media record. If the media record is associated directly with a trace, this will be null.
|
||||
contentType: MediaContentType
|
||||
contentLength:
|
||||
type: integer
|
||||
docs: The size of the media record in bytes
|
||||
sha256Hash:
|
||||
type: string
|
||||
docs: The SHA-256 hash of the media record
|
||||
field:
|
||||
type: string
|
||||
docs: The trace / observation field the media record is associated with. This can be one of `input`, `output`, `metadata`
|
||||
|
||||
GetMediaUploadUrlResponse:
|
||||
properties:
|
||||
uploadUrl:
|
||||
type: optional<string>
|
||||
docs: The presigned upload URL. If the asset is already uploaded, this will be null
|
||||
mediaId:
|
||||
type: string
|
||||
docs: The unique langfuse identifier of a media record
|
||||
|
||||
MediaContentType:
|
||||
type: literal<"image/png","image/jpeg","image/jpg","image/webp","audio/mpeg","audio/mp3","audio/wav","text/plain","application/pdf">
|
||||
docs: The MIME type of the media record
|
||||
@@ -52,10 +52,17 @@ service:
|
||||
configId:
|
||||
type: optional<string>
|
||||
docs: Retrieve only scores with a specific configId.
|
||||
queueId:
|
||||
type: optional<string>
|
||||
docs: Retrieve only scores with a specific annotation queueId.
|
||||
dataType:
|
||||
type: optional<commons.ScoreDataType>
|
||||
docs: Retrieve only scores with a specific dataType.
|
||||
response: Scores
|
||||
traceTags:
|
||||
type: optional<list<string>>
|
||||
allow-multiple: true
|
||||
docs: Only scores linked to traces that include all of these tags will be returned.
|
||||
response: GetScoresResponse
|
||||
get-by-id:
|
||||
docs: Get a score
|
||||
method: GET
|
||||
@@ -132,7 +139,38 @@ types:
|
||||
id:
|
||||
type: string
|
||||
docs: The id of the created object in Langfuse
|
||||
Scores:
|
||||
GetScoresResponseTraceData:
|
||||
properties:
|
||||
data: list<commons.Score>
|
||||
userId:
|
||||
type: optional<string>
|
||||
docs: The user ID associated with the trace referenced by score
|
||||
tags:
|
||||
type: optional<list<string>>
|
||||
docs: A list of tags associated with the trace referenced by score
|
||||
|
||||
GetScoresResponseDataNumeric:
|
||||
extends: commons.NumericScore
|
||||
properties:
|
||||
trace: GetScoresResponseTraceData
|
||||
|
||||
GetScoresResponseDataCategorical:
|
||||
extends: commons.CategoricalScore
|
||||
properties:
|
||||
trace: GetScoresResponseTraceData
|
||||
|
||||
GetScoresResponseDataBoolean:
|
||||
extends: commons.BooleanScore
|
||||
properties:
|
||||
trace: GetScoresResponseTraceData
|
||||
|
||||
GetScoresResponseData:
|
||||
discriminant: dataType
|
||||
union:
|
||||
NUMERIC: GetScoresResponseDataNumeric
|
||||
CATEGORICAL: GetScoresResponseDataCategorical
|
||||
BOOLEAN: GetScoresResponseDataBoolean
|
||||
|
||||
GetScoresResponse:
|
||||
properties:
|
||||
data: list<GetScoresResponseData>
|
||||
meta: pagination.MetaResponse
|
||||
|
||||
+11
-5
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "langfuse",
|
||||
"version": "2.85.1",
|
||||
"version": "2.95.1",
|
||||
"author": "engineering@langfuse.com",
|
||||
"license": "MIT",
|
||||
"private": true,
|
||||
@@ -9,15 +9,16 @@
|
||||
},
|
||||
"scripts": {
|
||||
"preinstall": "npx only-allow pnpm",
|
||||
"infra:dev:up": "docker compose -f ./docker-compose.dev.yml up -d",
|
||||
"infra:dev:up": "docker compose -f ./docker-compose.dev.yml up -d --wait",
|
||||
"infra:dev:down": "docker compose -f ./docker-compose.dev.yml down",
|
||||
"infra:dev:prune": "docker compose -f ./docker-compose.dev.yml down -v",
|
||||
"db:generate": "turbo run db:generate",
|
||||
"db:migrate": "turbo run db:migrate",
|
||||
"db:seed": "turbo run db:seed",
|
||||
"db:seed:examples": "turbo run db:seed:examples",
|
||||
"nuke": "bash ./scripts/nuke.sh",
|
||||
"dx": "pnpm i && pnpm run infra:dev:up && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx-f": "pnpm i && pnpm run infra:dev:up && pnpm --filter=shared run db:reset -f && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx-f": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset -f && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx:skip-infra": "pnpm i && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"build": "turbo run build",
|
||||
"start": "turbo run start",
|
||||
@@ -82,7 +83,12 @@
|
||||
},
|
||||
"pnpm": {
|
||||
"overrides": {
|
||||
"jsonpath-plus": "10.0.0"
|
||||
"jsonpath-plus": "10.2.0",
|
||||
"nanoid": "^3.3.8",
|
||||
"katex": "^0.16.21"
|
||||
},
|
||||
"patchedDependencies": {
|
||||
"next-auth@4.24.11": "patches/next-auth@4.24.11.patch"
|
||||
}
|
||||
},
|
||||
"packageManager": "pnpm@9.5.0"
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
DROP TABLE traces ON CLUSTER default;
|
||||
@@ -0,0 +1,32 @@
|
||||
CREATE TABLE traces ON CLUSTER default (
|
||||
`id` String,
|
||||
`timestamp` DateTime64(3),
|
||||
`name` String,
|
||||
`user_id` Nullable(String),
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`release` Nullable(String),
|
||||
`version` Nullable(String),
|
||||
`project_id` String,
|
||||
`public` Bool,
|
||||
`bookmarked` Bool,
|
||||
`tags` Array(String),
|
||||
`input` Nullable(String) CODEC(ZSTD(3)),
|
||||
`output` Nullable(String) CODEC(ZSTD(3)),
|
||||
`session_id` Nullable(String),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
updated_at DateTime64(3) DEFAULT now(),
|
||||
`event_ts` DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter(0.01) GRANULARITY 1,
|
||||
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
|
||||
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
toDate(timestamp)
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
id
|
||||
);
|
||||
@@ -0,0 +1 @@
|
||||
DROP TABLE observations ON CLUSTER default;
|
||||
@@ -0,0 +1,47 @@
|
||||
CREATE TABLE observations ON CLUSTER default (
|
||||
`id` String,
|
||||
`trace_id` String,
|
||||
`project_id` String,
|
||||
`type` LowCardinality(String),
|
||||
`parent_observation_id` Nullable(String),
|
||||
`start_time` DateTime64(3),
|
||||
`end_time` Nullable(DateTime64(3)),
|
||||
`name` String,
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`level` LowCardinality(String),
|
||||
`status_message` Nullable(String),
|
||||
`version` Nullable(String),
|
||||
`input` Nullable(String) CODEC(ZSTD(3)),
|
||||
`output` Nullable(String) CODEC(ZSTD(3)),
|
||||
`provided_model_name` Nullable(String),
|
||||
`internal_model_id` Nullable(String),
|
||||
`model_parameters` Nullable(String),
|
||||
`provided_usage_details` Map(LowCardinality(String), UInt64),
|
||||
`usage_details` Map(LowCardinality(String), UInt64),
|
||||
`provided_cost_details` Map(LowCardinality(String), Decimal64(12)),
|
||||
`cost_details` Map(LowCardinality(String), Decimal64(12)),
|
||||
`total_cost` Nullable(Decimal64(12)),
|
||||
`completion_start_time` Nullable(DateTime64(3)),
|
||||
`prompt_id` Nullable(String),
|
||||
`prompt_name` Nullable(String),
|
||||
`prompt_version` Nullable(UInt16),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
event_ts DateTime64(3),
|
||||
is_deleted UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_trace_id trace_id TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
|
||||
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
`type`,
|
||||
toDate(start_time)
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
`type`,
|
||||
toDate(start_time),
|
||||
id
|
||||
);
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
DROP TABLE scores ON CLUSTER default;
|
||||
@@ -0,0 +1,33 @@
|
||||
CREATE TABLE scores ON CLUSTER default (
|
||||
`id` String,
|
||||
`timestamp` DateTime64(3),
|
||||
`project_id` String,
|
||||
`trace_id` String,
|
||||
`observation_id` Nullable(String),
|
||||
`name` String,
|
||||
`value` Float64,
|
||||
`source` String,
|
||||
`comment` Nullable(String) CODEC(ZSTD(1)),
|
||||
`author_user_id` Nullable(String),
|
||||
`config_id` Nullable(String),
|
||||
`data_type` String,
|
||||
`string_value` Nullable(String),
|
||||
`queue_id` Nullable(String),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
event_ts DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
name
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
name,
|
||||
id
|
||||
)
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE observations ON CLUSTER default ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE observations ON CLUSTER default DROP INDEX IF EXISTS idx_project_id;
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_session_id;
|
||||
@@ -0,0 +1,2 @@
|
||||
ALTER TABLE traces ON CLUSTER default ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
|
||||
ALTER TABLE traces ON CLUSTER default MATERIALIZE INDEX IF EXISTS idx_session_id;
|
||||
+15
-9
@@ -3,24 +3,30 @@ CREATE TABLE traces (
|
||||
`timestamp` DateTime64(3),
|
||||
`name` String,
|
||||
`user_id` Nullable(String),
|
||||
`metadata` Map(String, String) CODEC(ZSTD(1)),
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`release` Nullable(String),
|
||||
`version` Nullable(String),
|
||||
`project_id` String,
|
||||
`public` Bool,
|
||||
`bookmarked` Bool,
|
||||
`tags` Array(String),
|
||||
`input` Nullable(String) CODEC(ZSTD(1)),
|
||||
`output` Nullable(String) CODEC(ZSTD(1)),
|
||||
`input` Nullable(String) CODEC(ZSTD(3)),
|
||||
`output` Nullable(String) CODEC(ZSTD(3)),
|
||||
`session_id` Nullable(String),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
updated_at DateTime64(3) DEFAULT now(),
|
||||
`event_ts` DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter(0.01) GRANULARITY 1,
|
||||
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
|
||||
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(timestamp)
|
||||
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
toDate(timestamp)
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
toUnixTimestamp(timestamp),
|
||||
id
|
||||
);
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
id
|
||||
);
|
||||
+20
-23
@@ -7,7 +7,7 @@ CREATE TABLE observations (
|
||||
`start_time` DateTime64(3),
|
||||
`end_time` Nullable(DateTime64(3)),
|
||||
`name` String,
|
||||
`metadata` Map(LowCardinality(String), String) CODEC(ZSTD(1)),
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`level` LowCardinality(String),
|
||||
`status_message` Nullable(String),
|
||||
`version` Nullable(String),
|
||||
@@ -16,18 +16,10 @@ CREATE TABLE observations (
|
||||
`provided_model_name` Nullable(String),
|
||||
`internal_model_id` Nullable(String),
|
||||
`model_parameters` Nullable(String),
|
||||
`provided_input_usage_units` Nullable(Decimal64(12)),
|
||||
`provided_output_usage_units` Nullable(Decimal64(12)),
|
||||
`provided_total_usage_units` Nullable(Decimal64(12)),
|
||||
`input_usage_units` Nullable(Decimal64(12)),
|
||||
`output_usage_units` Nullable(Decimal64(12)),
|
||||
`total_usage_units` Nullable(Decimal64(12)),
|
||||
`unit` Nullable(String),
|
||||
`provided_input_cost` Nullable(Decimal64(12)),
|
||||
`provided_output_cost` Nullable(Decimal64(12)),
|
||||
`provided_total_cost` Nullable(Decimal64(12)),
|
||||
`input_cost` Nullable(Decimal64(12)),
|
||||
`output_cost` Nullable(Decimal64(12)),
|
||||
`provided_usage_details` Map(LowCardinality(String), UInt64),
|
||||
`usage_details` Map(LowCardinality(String), UInt64),
|
||||
`provided_cost_details` Map(LowCardinality(String), Decimal64(12)),
|
||||
`cost_details` Map(LowCardinality(String), Decimal64(12)),
|
||||
`total_cost` Nullable(Decimal64(12)),
|
||||
`completion_start_time` Nullable(DateTime64(3)),
|
||||
`prompt_id` Nullable(String),
|
||||
@@ -35,16 +27,21 @@ CREATE TABLE observations (
|
||||
`prompt_version` Nullable(UInt16),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
event_ts DateTime64(3),
|
||||
is_deleted UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_trace_id trace_id TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter() GRANULARITY 1,
|
||||
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter() GRANULARITY 1
|
||||
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(start_time)
|
||||
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
|
||||
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
`type`,
|
||||
toDate(start_time)
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
`type`,
|
||||
trace_id,
|
||||
toUnixTimestamp(start_time),
|
||||
id
|
||||
);
|
||||
project_id,
|
||||
`type`,
|
||||
toDate(start_time),
|
||||
id
|
||||
);
|
||||
|
||||
+15
-7
@@ -12,14 +12,22 @@ CREATE TABLE scores (
|
||||
`config_id` Nullable(String),
|
||||
`data_type` String,
|
||||
`string_value` Nullable(String),
|
||||
`queue_id` Nullable(String),
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
event_ts DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_project_id trace_id TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(timestamp)
|
||||
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
|
||||
PRIMARY KEY (
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
name
|
||||
)
|
||||
ORDER BY (
|
||||
project_id,
|
||||
trace_id,
|
||||
toUnixTimestamp(timestamp),
|
||||
id
|
||||
);
|
||||
project_id,
|
||||
toDate(timestamp),
|
||||
name,
|
||||
id
|
||||
)
|
||||
+1
@@ -0,0 +1 @@
|
||||
ALTER TABLE observations ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE observations DROP INDEX IF EXISTS idx_project_id;
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE traces DROP INDEX IF EXISTS idx_session_id;
|
||||
@@ -0,0 +1,2 @@
|
||||
ALTER TABLE traces ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
|
||||
ALTER TABLE traces MATERIALIZE INDEX IF EXISTS idx_session_id;
|
||||
@@ -1,7 +1,13 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Load environment variables
|
||||
source ../../.env
|
||||
[ -f ../../.env ] && source ../../.env
|
||||
|
||||
# Check if CLICKHOUSE_URL is configured
|
||||
if [ -z "${CLICKHOUSE_URL}" ]; then
|
||||
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Check if golang-migrate is installed
|
||||
if ! command -v migrate &> /dev/null
|
||||
@@ -13,10 +19,22 @@ then
|
||||
fi
|
||||
|
||||
# Construct the database URL
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
|
||||
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
|
||||
fi
|
||||
# Execute the down command
|
||||
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" down
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
|
||||
fi
|
||||
@@ -0,0 +1,75 @@
|
||||
import {
|
||||
clickhouseClient,
|
||||
ObservationRecordReadType,
|
||||
} from "@langfuse/shared/src/server";
|
||||
import { Prisma, prisma } from "../../src/db";
|
||||
import { redis } from "@langfuse/shared/src/server";
|
||||
import { prepareClickhouse } from "../../scripts/prepareClickhouse";
|
||||
import { createDatasets } from "../../prisma/seed";
|
||||
import { queryClickhouse } from "../../src/server/repositories/clickhouse";
|
||||
import { convertObservation } from "../../src/server/repositories/observations_converters";
|
||||
|
||||
async function main() {
|
||||
try {
|
||||
const projectIds = ["7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"]; // Example project IDs
|
||||
if (
|
||||
await prisma.project.findFirst({
|
||||
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
|
||||
})
|
||||
) {
|
||||
projectIds.push("239ad00f-562f-411d-af14-831c75ddd875");
|
||||
}
|
||||
await prepareClickhouse(projectIds, {
|
||||
numberOfDays: 3,
|
||||
totalObservations: 10000,
|
||||
});
|
||||
|
||||
const project1 = await prisma.project.findFirst({
|
||||
where: { id: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a" },
|
||||
});
|
||||
|
||||
const project2 =
|
||||
projectIds.length > 1
|
||||
? await prisma.project.findFirst({
|
||||
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
|
||||
})
|
||||
: await prisma.project.findFirst();
|
||||
|
||||
const query = `
|
||||
SELECT *
|
||||
FROM observations o
|
||||
WHERE o.project_id IN ({projectIds: Array(String)})
|
||||
LIMIT 2000;
|
||||
`;
|
||||
|
||||
const res = await queryClickhouse<ObservationRecordReadType>({
|
||||
query,
|
||||
params: {
|
||||
projectIds,
|
||||
},
|
||||
});
|
||||
|
||||
await createDatasets(
|
||||
project1!,
|
||||
project2!,
|
||||
(await Promise.all(res.map(convertObservation))).map((o) => ({
|
||||
...o,
|
||||
metadata: {},
|
||||
modelParameters: {},
|
||||
input: {},
|
||||
output: {},
|
||||
})),
|
||||
);
|
||||
|
||||
console.log("Clickhouse preparation completed successfully.");
|
||||
} catch (error) {
|
||||
console.error("Error during Clickhouse preparation:", error);
|
||||
} finally {
|
||||
await clickhouseClient().close();
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
console.log("Disconnected from Clickhouse.");
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,7 +1,13 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Load environment variables
|
||||
source ../../.env
|
||||
[ -f ../../.env ] && source ../../.env
|
||||
|
||||
# Check if CLICKHOUSE_URL is configured
|
||||
if [ -z "${CLICKHOUSE_URL}" ]; then
|
||||
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Check if golang-migrate is installed
|
||||
if ! command -v migrate &> /dev/null
|
||||
@@ -13,11 +19,22 @@ then
|
||||
fi
|
||||
|
||||
# Construct the database URL
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
|
||||
fi
|
||||
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" up
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" up
|
||||
else
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
|
||||
else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" up
|
||||
fi
|
||||
|
||||
@@ -43,27 +43,30 @@
|
||||
"db:generate": "dotenv -e ../../.env -- npx prisma generate",
|
||||
"db:seed:examples": "dotenv -e ../../.env -- npx prisma db seed -- --environment examples",
|
||||
"db:seed:load": "dotenv -e ../../.env -- npx prisma db seed -- --environment load",
|
||||
"ch:status": "dotenv -e ../../.env -- goose -dir './clickhouse/migrations/' status",
|
||||
"ch:up": "bash clickhouse/scripts/up.sh",
|
||||
"ch:down": "bash clickhouse/scripts/down.sh",
|
||||
"ch:drop": "bash clickhouse/scripts/drop.sh",
|
||||
"ch:reset": "pnpm run ch:down && pnpm run ch:up"
|
||||
"ch:reset": "pnpm run ch:down && pnpm run ch:up && pnpm run ch:seed",
|
||||
"ch:seed": "dotenv -e ../../.env -- tsx clickhouse/scripts/seed.ts",
|
||||
"load:setup": "dotenv -e ../../.env -- tsx scripts/load-seed.ts"
|
||||
},
|
||||
"prisma": {
|
||||
"seed": "ts-node -r tsconfig-paths/register -r dotenv/config --compiler-options {\"module\":\"CommonJS\"} prisma/seed.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@anthropic-ai/tokenizer": "^0.0.4",
|
||||
"@aws-sdk/client-s3": "^3.627.0",
|
||||
"@aws-sdk/lib-storage": "^3.667.0",
|
||||
"@aws-sdk/s3-request-presigner": "^3.554.0",
|
||||
"@aws-sdk/client-cloudwatch": "^3.675.0",
|
||||
"@aws-sdk/client-s3": "^3.675.0",
|
||||
"@aws-sdk/lib-storage": "^3.675.0",
|
||||
"@aws-sdk/s3-request-presigner": "^3.679.0",
|
||||
"@azure/storage-blob": "^12.26.0",
|
||||
"@clickhouse/client": "^1.4.0",
|
||||
"@langchain/anthropic": "^0.3.1",
|
||||
"@langchain/aws": "^0.1.0",
|
||||
"@langchain/core": "^0.3.9",
|
||||
"@langchain/openai": "^0.3.0",
|
||||
"@langchain/anthropic": "^0.3.8",
|
||||
"@langchain/aws": "^0.1.2",
|
||||
"@langchain/core": "^0.3.18",
|
||||
"@langchain/openai": "^0.3.14",
|
||||
"@opentelemetry/api": ">=1.0.0 <1.10.0",
|
||||
"@prisma/client": "^5.20.0",
|
||||
"@prisma/client": "^5.22.0",
|
||||
"@react-email/components": "^0.0.19",
|
||||
"@react-email/render": "^0.0.15",
|
||||
"@types/bcryptjs": "^2.4.6",
|
||||
@@ -73,17 +76,19 @@
|
||||
"dd-trace": "^5.23.1",
|
||||
"decimal.js": "^10.4.3",
|
||||
"exponential-backoff": "^3.1.1",
|
||||
"https-proxy-agent": "^7.0.6",
|
||||
"ioredis": "^5.4.1",
|
||||
"kysely": "^0.27.4",
|
||||
"langchain": "^0.3.2",
|
||||
"langchain": "^0.3.6",
|
||||
"langfuse-langchain": "3.30.3",
|
||||
"lodash": "^4.17.21",
|
||||
"next-auth": "^4.24.7",
|
||||
"next-auth": "^4.24.11",
|
||||
"nodemailer": "^6.9.15",
|
||||
"prisma-extension-kysely": "^2.1.0",
|
||||
"uuid": "^9.0.1",
|
||||
"winston": "^3.14.2",
|
||||
"winston": "^3.15.0",
|
||||
"zod": "^3.23.8",
|
||||
"zod-to-json-schema": "^3.23.2"
|
||||
"zod-to-json-schema": "^3.23.5"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@repo/eslint-config": "workspace:*",
|
||||
@@ -101,11 +106,12 @@
|
||||
"kysely-codegen": "^0.16.8",
|
||||
"nodemon": "^3.1.7",
|
||||
"prettier": "^3.3.3",
|
||||
"prisma": "^5.20.0",
|
||||
"prisma": "^5.22.0",
|
||||
"prisma-erd-generator": "^1.11.2",
|
||||
"prisma-kysely": "^1.8.0",
|
||||
"ts-node": "^10.9.2",
|
||||
"tsc-watch": "^6.2.0",
|
||||
"tsx": "^4.19.1",
|
||||
"typescript": "^5.4.5",
|
||||
"vitest": "^2.1.2"
|
||||
},
|
||||
@@ -115,7 +121,7 @@
|
||||
},
|
||||
"pnpm": {
|
||||
"overrides": {
|
||||
"jsonpath-plus": "10.0.0"
|
||||
"jsonpath-plus": "10.0.7"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +0,0 @@
|
||||
{
|
||||
"trailingComma": "es5",
|
||||
"printWidth": 120
|
||||
}
|
||||
@@ -143,6 +143,18 @@ export type AuditLog = {
|
||||
before: string | null;
|
||||
after: string | null;
|
||||
};
|
||||
export type BackgroundMigration = {
|
||||
id: string;
|
||||
name: string;
|
||||
script: string;
|
||||
args: unknown;
|
||||
state: Generated<unknown>;
|
||||
finished_at: Timestamp | null;
|
||||
failed_at: Timestamp | null;
|
||||
failed_reason: string | null;
|
||||
worker_id: string | null;
|
||||
locked_at: Timestamp | null;
|
||||
};
|
||||
export type BatchExport = {
|
||||
id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
@@ -266,6 +278,8 @@ export type JobExecution = {
|
||||
end_time: Timestamp | null;
|
||||
error: string | null;
|
||||
job_input_trace_id: string | null;
|
||||
job_input_observation_id: string | null;
|
||||
job_input_dataset_item_id: string | null;
|
||||
job_output_score_id: string | null;
|
||||
};
|
||||
export type LlmApiKeys = {
|
||||
@@ -282,6 +296,20 @@ export type LlmApiKeys = {
|
||||
config: unknown | null;
|
||||
project_id: string;
|
||||
};
|
||||
export type Media = {
|
||||
id: string;
|
||||
sha_256_hash: string;
|
||||
project_id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
updated_at: Generated<Timestamp>;
|
||||
uploaded_at: Timestamp | null;
|
||||
upload_http_status: number | null;
|
||||
upload_http_error: string | null;
|
||||
bucket_path: string;
|
||||
bucket_name: string;
|
||||
content_type: string;
|
||||
content_length: string;
|
||||
};
|
||||
export type MembershipInvitation = {
|
||||
id: string;
|
||||
email: string;
|
||||
@@ -304,7 +332,7 @@ export type Model = {
|
||||
input_price: string | null;
|
||||
output_price: string | null;
|
||||
total_price: string | null;
|
||||
unit: string;
|
||||
unit: string | null;
|
||||
tokenizer_id: string | null;
|
||||
tokenizer_config: unknown | null;
|
||||
};
|
||||
@@ -342,6 +370,16 @@ export type Observation = {
|
||||
completion_start_time: Timestamp | null;
|
||||
prompt_id: string | null;
|
||||
};
|
||||
export type ObservationMedia = {
|
||||
id: string;
|
||||
project_id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
updated_at: Generated<Timestamp>;
|
||||
media_id: string;
|
||||
trace_id: string;
|
||||
observation_id: string;
|
||||
field: string;
|
||||
};
|
||||
export type ObservationView = {
|
||||
id: string;
|
||||
trace_id: string | null;
|
||||
@@ -402,6 +440,14 @@ export type PosthogIntegration = {
|
||||
enabled: boolean;
|
||||
created_at: Generated<Timestamp>;
|
||||
};
|
||||
export type Price = {
|
||||
id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
updated_at: Generated<Timestamp>;
|
||||
model_id: string;
|
||||
usage_type: string;
|
||||
price: string;
|
||||
};
|
||||
export type Project = {
|
||||
id: string;
|
||||
org_id: string;
|
||||
@@ -495,6 +541,15 @@ export type Trace = {
|
||||
created_at: Generated<Timestamp>;
|
||||
updated_at: Generated<Timestamp>;
|
||||
};
|
||||
export type TraceMedia = {
|
||||
id: string;
|
||||
project_id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
updated_at: Generated<Timestamp>;
|
||||
media_id: string;
|
||||
trace_id: string;
|
||||
field: string;
|
||||
};
|
||||
export type TraceSession = {
|
||||
id: string;
|
||||
created_at: Generated<Timestamp>;
|
||||
@@ -546,6 +601,7 @@ export type DB = {
|
||||
annotation_queues: AnnotationQueue;
|
||||
api_keys: ApiKey;
|
||||
audit_logs: AuditLog;
|
||||
background_migrations: BackgroundMigration;
|
||||
batch_exports: BatchExport;
|
||||
comments: Comment;
|
||||
cron_jobs: CronJobs;
|
||||
@@ -558,13 +614,16 @@ export type DB = {
|
||||
job_configurations: JobConfiguration;
|
||||
job_executions: JobExecution;
|
||||
llm_api_keys: LlmApiKeys;
|
||||
media: Media;
|
||||
membership_invitations: MembershipInvitation;
|
||||
models: Model;
|
||||
observation_media: ObservationMedia;
|
||||
observations: Observation;
|
||||
observations_view: ObservationView;
|
||||
organization_memberships: OrganizationMembership;
|
||||
organizations: Organization;
|
||||
posthog_integrations: PosthogIntegration;
|
||||
prices: Price;
|
||||
project_memberships: ProjectMembership;
|
||||
projects: Project;
|
||||
prompts: Prompt;
|
||||
@@ -572,6 +631,7 @@ export type DB = {
|
||||
scores: Score;
|
||||
Session: Session;
|
||||
sso_configs: SsoConfig;
|
||||
trace_media: TraceMedia;
|
||||
trace_sessions: TraceSession;
|
||||
traces: Trace;
|
||||
traces_view: TraceView;
|
||||
|
||||
+107
@@ -0,0 +1,107 @@
|
||||
-- Temporarily drop observations_view as prompts.config is a calculated column in the view and we need to update it to a JSON type
|
||||
DROP VIEW IF EXISTS "observations_view"; -- Drop view as column was added in 20240705154048_observation_view_add_created_at_updated_at and update view must have same columns
|
||||
|
||||
-- Update prompts.config to JSON type. Use JSON as JSONB is reordering keys for optimization, which is undesired
|
||||
ALTER TABLE prompts ALTER COLUMN config TYPE JSON USING config::JSON;
|
||||
ALTER TABLE prompts ALTER COLUMN config SET DEFAULT '{}'::JSON;
|
||||
|
||||
-- Recreate the view with the updated prompts.config column
|
||||
CREATE VIEW "observations_view" AS -- Specify the columns that should be returned in the view, as calculated columns are added but exist in the observations table already
|
||||
SELECT
|
||||
o.id,
|
||||
o.name,
|
||||
o.start_time,
|
||||
o.end_time,
|
||||
o.parent_observation_id,
|
||||
o.type,
|
||||
o.trace_id,
|
||||
o.metadata,
|
||||
o.model,
|
||||
o."modelParameters",
|
||||
o.input,
|
||||
o.output,
|
||||
o.level,
|
||||
o.status_message,
|
||||
o.completion_start_time,
|
||||
o.completion_tokens,
|
||||
o.prompt_tokens,
|
||||
o.total_tokens,
|
||||
o.version,
|
||||
o.project_id,
|
||||
o.created_at,
|
||||
o.updated_at,
|
||||
o.unit,
|
||||
o.prompt_id,
|
||||
p.name as prompt_name, -- added in this change
|
||||
p.version as prompt_version, -- added in this change
|
||||
o.input_cost,
|
||||
o.output_cost,
|
||||
o.total_cost,
|
||||
o.internal_model,
|
||||
m.id AS "model_id",
|
||||
m.start_date AS "model_start_date",
|
||||
m.input_price,
|
||||
m.output_price,
|
||||
m.total_price,
|
||||
m.tokenizer_config AS "tokenizer_config",
|
||||
CASE
|
||||
WHEN o.calculated_input_cost IS NULL AND o.input_cost IS NULL AND o.output_cost IS NULL AND o.total_cost IS NULL THEN
|
||||
o.prompt_tokens::decimal * m.input_price
|
||||
ELSE
|
||||
COALESCE(o.calculated_input_cost, o.input_cost)
|
||||
END AS "calculated_input_cost",
|
||||
CASE
|
||||
WHEN o.calculated_output_cost IS NULL AND o.input_cost IS NULL AND o.output_cost IS NULL AND o.total_cost IS NULL THEN
|
||||
o.completion_tokens::decimal * m.output_price
|
||||
ELSE
|
||||
COALESCE(o.calculated_output_cost, o.output_cost)
|
||||
END AS "calculated_output_cost",
|
||||
CASE
|
||||
WHEN o.calculated_total_cost IS NULL AND o.input_cost IS NULL AND o.output_cost IS NULL AND o.total_cost IS NULL THEN
|
||||
CASE
|
||||
WHEN m.total_price IS NOT NULL AND o.total_tokens IS NOT NULL THEN
|
||||
m.total_price * o.total_tokens
|
||||
ELSE
|
||||
o.prompt_tokens::decimal * m.input_price +
|
||||
o.completion_tokens::decimal * m.output_price
|
||||
END
|
||||
ELSE
|
||||
COALESCE(o.calculated_total_cost, o.total_cost)
|
||||
END AS "calculated_total_cost",
|
||||
CASE WHEN o.end_time IS NULL THEN NULL ELSE (EXTRACT(EPOCH FROM o."end_time") - EXTRACT(EPOCH FROM o."start_time"))::double precision END AS "latency",
|
||||
CASE WHEN o.completion_start_time IS NOT NULL AND o.start_time IS NOT NULL THEN EXTRACT(EPOCH FROM (completion_start_time - start_time))::double precision ELSE NULL END as "time_to_first_token"
|
||||
|
||||
FROM
|
||||
observations o
|
||||
LEFT JOIN LATERAL (
|
||||
SELECT
|
||||
models.*
|
||||
FROM
|
||||
models
|
||||
WHERE (models.project_id = o.project_id OR models.project_id IS NULL)
|
||||
AND models.model_name = o.internal_model
|
||||
AND (models.start_date < o.start_time OR models.start_date IS NULL)
|
||||
AND o.unit::TEXT = models.unit
|
||||
ORDER BY
|
||||
models.project_id ASC, -- in postgres, NULLs are sorted last when ordering ASC
|
||||
models.start_date DESC NULLS LAST -- now, NULLs are sorted last when ordering DESC as well
|
||||
LIMIT 1
|
||||
) m ON TRUE
|
||||
LEFT JOIN LATERAL (
|
||||
SELECT
|
||||
prompts.*
|
||||
FROM
|
||||
prompts
|
||||
WHERE prompts.id = o.prompt_id
|
||||
AND prompts.project_id = o.project_id
|
||||
LIMIT 1
|
||||
) p ON TRUE
|
||||
|
||||
|
||||
-- requirements:
|
||||
-- 1. The view should return all columns from the observations table
|
||||
-- 2. The view should match with only one model for each observation if:
|
||||
-- a. The model has the same project_id as the observation, otherwise the model without project_id.
|
||||
-- b. The model has the same model_name as the observation
|
||||
-- c. The model has a start_date that is less than the observation start_time, otherwise the model without start_date
|
||||
-- d. The model has the same unit as the observation
|
||||
@@ -0,0 +1,18 @@
|
||||
INSERT INTO models (
|
||||
id,
|
||||
project_id,
|
||||
model_name,
|
||||
match_pattern,
|
||||
start_date,
|
||||
input_price,
|
||||
output_price,
|
||||
total_price,
|
||||
unit,
|
||||
tokenizer_id,
|
||||
tokenizer_config
|
||||
)
|
||||
VALUES
|
||||
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
|
||||
('cm2krz1uf000208jjg5653iud', NULL, 'claude-3.5-haiku-20241022', '(?i)^(claude-3-5-sonnet-20241022|anthropic\.claude-3-5-sonnet-20241022-v2:0|claude-3-5-sonnet-V2@20241022)$', NULL, 0.000003, 0.000015, NULL, 'TOKENS', 'claude', NULL),
|
||||
('cm2ks2vzn000308jjh4ze1w7q', NULL, 'claude-3.5-haiku-latest', '(?i)^(claude-3-5-sonnet-latest)$', NULL, 0.000003, 0.000015, NULL, 'TOKENS', 'claude', NULL)
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
|
||||
UPDATE models SET model_name = 'claude-3.5-sonnet-20241022' WHERE id = 'cm2krz1uf000208jjg5653iud';
|
||||
UPDATE models SET model_name = 'claude-3.5-sonnet-latest' WHERE id = 'cm2ks2vzn000308jjh4ze1w7q';
|
||||
@@ -0,0 +1,62 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE "prices" (
|
||||
"id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"model_id" TEXT NOT NULL,
|
||||
"usage_type" TEXT NOT NULL,
|
||||
"price" DECIMAL(65,30) NOT NULL,
|
||||
|
||||
CONSTRAINT "prices_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "prices_model_id_idx" ON "prices"("model_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "prices_model_id_usage_type_key" ON "prices"("model_id", "usage_type");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "prices" ADD CONSTRAINT "prices_model_id_fkey" FOREIGN KEY ("model_id") REFERENCES "models"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- Alter Table to make unit column nullable
|
||||
ALTER TABLE "models" ALTER COLUMN "unit" DROP NOT NULL;
|
||||
|
||||
|
||||
-- Create a temporary table to store the new prices for user-defined models
|
||||
CREATE TEMPORARY TABLE temp_prices AS
|
||||
SELECT
|
||||
id AS model_id,
|
||||
'input' AS usage_type,
|
||||
input_price AS price
|
||||
FROM models
|
||||
WHERE project_id IS NOT NULL AND input_price IS NOT NULL
|
||||
UNION ALL
|
||||
SELECT
|
||||
id AS model_id,
|
||||
'output' AS usage_type,
|
||||
output_price AS price
|
||||
FROM models
|
||||
WHERE project_id IS NOT NULL AND output_price IS NOT NULL
|
||||
UNION ALL
|
||||
SELECT
|
||||
id AS model_id,
|
||||
'total' AS usage_type,
|
||||
total_price AS price
|
||||
FROM models
|
||||
WHERE project_id IS NOT NULL AND total_price IS NOT NULL;
|
||||
|
||||
-- Insert into prices table
|
||||
INSERT INTO prices (id, created_at, updated_at, model_id, usage_type, price)
|
||||
SELECT
|
||||
md5(random()::text || clock_timestamp()::text || model_id::text || usage_type::text)::uuid AS id,
|
||||
NOW() AS created_at,
|
||||
NOW() AS updated_at,
|
||||
model_id,
|
||||
usage_type,
|
||||
price
|
||||
FROM temp_prices
|
||||
ON CONFLICT (model_id, usage_type) DO NOTHING;
|
||||
|
||||
-- Drop the temporary table
|
||||
DROP TABLE temp_prices;
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE "background_migrations" (
|
||||
"id" TEXT NOT NULL,
|
||||
"name" TEXT NOT NULL,
|
||||
"script" TEXT NOT NULL,
|
||||
"args" JSONB NOT NULL,
|
||||
|
||||
"finished_at" TIMESTAMP(3),
|
||||
"failed_at" TIMESTAMP(3),
|
||||
"failed_reason" TEXT,
|
||||
"worker_id" TEXT,
|
||||
"locked_at" TIMESTAMP(3),
|
||||
|
||||
CONSTRAINT "background_migrations_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "background_migrations_name_key" ON "background_migrations"("name");
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('32859a35-98f5-4a4a-b438-ebc579349e00', '20241024_1216_add_generations_cost_backfill', 'addGenerationsCostBackfill', '{}');
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('5960f22a-748f-480c-b2f3-bc4f9d5d84bc', '20241024_1730_migrate_traces_from_pg_to_ch', 'migrateTracesFromPostgresToClickhouse', '{}');
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('7526e7c9-0026-4595-af2c-369dfd9176ec', '20241024_1737_migrate_observations_from_pg_to_ch', 'migrateObservationsFromPostgresToClickhouse', '{}');
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('94e50334-50d3-4e49-ad2e-9f6d92c85ef7', '20241024_1738_migrate_scores_from_pg_to_ch', 'migrateScoresFromPostgresToClickhouse', '{}');
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- DropIndex
|
||||
DROP INDEX CONCURRENTLY "prices_model_id_idx";
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- AlterTable
|
||||
ALTER TABLE "background_migrations" ADD COLUMN "state" jsonb NOT NULL DEFAULT '{}';
|
||||
@@ -0,0 +1,29 @@
|
||||
INSERT INTO models (
|
||||
id,
|
||||
project_id,
|
||||
model_name,
|
||||
match_pattern,
|
||||
start_date,
|
||||
input_price,
|
||||
output_price,
|
||||
total_price,
|
||||
unit,
|
||||
tokenizer_id,
|
||||
tokenizer_config
|
||||
)
|
||||
VALUES
|
||||
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
|
||||
('cm34aq60d000207ml0j1h31ar', NULL, 'claude-3-5-haiku-20241022', '(?i)^(claude-3-5-haiku-20241022|anthropic\.claude-3-5-haiku-20241022-v1:0|claude-3-5-haiku-V1@20241022)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL),
|
||||
('cm34aqb9h000307ml6nypd618', NULL, 'claude-3.5-haiku-latest', '(?i)^(claude-3-5-haiku-latest)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL);
|
||||
|
||||
INSERT INTO prices (
|
||||
id,
|
||||
model_id,
|
||||
usage_type,
|
||||
price
|
||||
)
|
||||
VALUES
|
||||
('cm34ax6mc000008jkfqed92mb', 'cm34aq60d000207ml0j1h31ar', 'input', 0.000001),
|
||||
('cm34axb2o000108jk09wn9b47', 'cm34aqb9h000307ml6nypd618', 'input', 0.000001),
|
||||
('cm34axeie000208jk8b2ke2t8', 'cm34aq60d000207ml0j1h31ar', 'output', 0.000005),
|
||||
('cm34axi67000308jk7x1a7qko', 'cm34aqb9h000307ml6nypd618', 'output', 0.000005);
|
||||
@@ -0,0 +1,71 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE "media" (
|
||||
"id" TEXT NOT NULL,
|
||||
"sha_256_hash" CHAR(44) NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"uploaded_at" TIMESTAMP(3),
|
||||
"upload_http_status" INTEGER,
|
||||
"upload_http_error" TEXT,
|
||||
"bucket_path" TEXT NOT NULL,
|
||||
"bucket_name" TEXT NOT NULL,
|
||||
"content_type" TEXT NOT NULL,
|
||||
"content_length" BIGINT NOT NULL,
|
||||
|
||||
CONSTRAINT "media_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "trace_media" (
|
||||
"id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"media_id" TEXT NOT NULL,
|
||||
"trace_id" TEXT NOT NULL,
|
||||
"field" TEXT NOT NULL,
|
||||
|
||||
CONSTRAINT "trace_media_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "observation_media" (
|
||||
"id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"media_id" TEXT NOT NULL,
|
||||
"trace_id" TEXT NOT NULL,
|
||||
"observation_id" TEXT NOT NULL,
|
||||
"field" TEXT NOT NULL,
|
||||
|
||||
CONSTRAINT "observation_media_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "media_project_id_sha_256_hash_key" ON "media"("project_id", "sha_256_hash");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "trace_media_project_id_trace_id_media_id_field_key" ON "trace_media"("project_id", "trace_id", "media_id", "field");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "observation_media_project_id_observation_id_idx" ON "observation_media"("project_id", "observation_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "observation_media_project_id_trace_id_observation_id_media__key" ON "observation_media"("project_id", "trace_id", "observation_id", "media_id", "field");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "media" ADD CONSTRAINT "media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
-- DropForeignKey
|
||||
ALTER TABLE "job_executions" DROP CONSTRAINT "job_executions_job_input_trace_id_fkey";
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "job_executions" ADD COLUMN "job_input_dataset_item_id" TEXT,
|
||||
ADD COLUMN "job_input_observation_id" TEXT;
|
||||
@@ -0,0 +1,25 @@
|
||||
INSERT INTO models (
|
||||
id,
|
||||
project_id,
|
||||
model_name,
|
||||
match_pattern,
|
||||
start_date,
|
||||
input_price,
|
||||
output_price,
|
||||
total_price,
|
||||
unit,
|
||||
tokenizer_id,
|
||||
tokenizer_config
|
||||
)
|
||||
VALUES
|
||||
('cm3x0p8ev000008kyd96800c8', NULL, 'chatgpt-4o-latest', '(?i)^(chatgpt-4o-latest)$', NULL, 0.000005, 0.000015, NULL, 'TOKENS', 'openai', '{ "tokensPerMessage": 3, "tokensPerName": 1, "tokenizerModel": "gpt-4o" }');
|
||||
|
||||
INSERT INTO prices (
|
||||
id,
|
||||
model_id,
|
||||
usage_type,
|
||||
price
|
||||
)
|
||||
VALUES
|
||||
('cm3x0psrz000108kydpxg9o2k', 'cm3x0p8ev000008kyd96800c8', 'input', 0.000005),
|
||||
('cm3x0pyt7000208ky8737gdla', 'cm3x0p8ev000008kyd96800c8', 'output', 0.000015);
|
||||
File diff suppressed because it is too large
Load Diff
+157
-118
@@ -256,7 +256,7 @@ async function main() {
|
||||
|
||||
const queueIds = await generateQueuesForProject(
|
||||
[project1, project2],
|
||||
configIdsAndNames
|
||||
configIdsAndNames,
|
||||
);
|
||||
|
||||
const promptIds = await generatePromptsForProject([project1, project2]);
|
||||
@@ -282,11 +282,11 @@ async function main() {
|
||||
project2,
|
||||
promptIds,
|
||||
queueIds,
|
||||
configIdsAndNames
|
||||
configIdsAndNames,
|
||||
);
|
||||
|
||||
logger.info(
|
||||
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`
|
||||
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`,
|
||||
);
|
||||
|
||||
await uploadObjects(
|
||||
@@ -296,7 +296,7 @@ async function main() {
|
||||
sessions,
|
||||
events,
|
||||
comments,
|
||||
queueItems
|
||||
queueItems,
|
||||
);
|
||||
|
||||
// If openai key is in environment, add it to the projects LLM API keys
|
||||
@@ -314,7 +314,7 @@ async function main() {
|
||||
});
|
||||
} else {
|
||||
logger.warn(
|
||||
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key."
|
||||
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key.",
|
||||
);
|
||||
}
|
||||
|
||||
@@ -387,16 +387,62 @@ async function main() {
|
||||
update: {},
|
||||
});
|
||||
|
||||
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
|
||||
const dataset = await prisma.dataset.create({
|
||||
data: {
|
||||
name: `demo-dataset-${datasetNumber}`,
|
||||
description:
|
||||
datasetNumber === 0 ? "Dataset test description" : undefined,
|
||||
projectId: project2.id,
|
||||
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
|
||||
},
|
||||
});
|
||||
await createDatasets(project1, project2, observations);
|
||||
}
|
||||
}
|
||||
|
||||
main()
|
||||
.then(async () => {
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
})
|
||||
.catch(async (e) => {
|
||||
logger.error(e);
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
process.exit(1);
|
||||
});
|
||||
|
||||
export async function createDatasets(
|
||||
project1: {
|
||||
id: string;
|
||||
orgId: string;
|
||||
createdAt: Date;
|
||||
updatedAt: Date;
|
||||
name: string;
|
||||
},
|
||||
project2: {
|
||||
id: string;
|
||||
orgId: string;
|
||||
createdAt: Date;
|
||||
updatedAt: Date;
|
||||
name: string;
|
||||
},
|
||||
observations: Prisma.ObservationCreateManyInput[],
|
||||
) {
|
||||
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
|
||||
for (const projectId of [project1.id, project2.id]) {
|
||||
const datasetName = `demo-dataset-${datasetNumber}`;
|
||||
|
||||
// check if ds already exists
|
||||
const dataset =
|
||||
(await prisma.dataset.findFirst({
|
||||
where: {
|
||||
projectId,
|
||||
name: datasetName,
|
||||
},
|
||||
})) ??
|
||||
(await prisma.dataset.create({
|
||||
data: {
|
||||
name: datasetName,
|
||||
description:
|
||||
datasetNumber === 0 ? "Dataset test description" : undefined,
|
||||
projectId,
|
||||
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
|
||||
},
|
||||
}));
|
||||
|
||||
const datasetItemIds = [];
|
||||
for (let i = 0; i < 18; i++) {
|
||||
@@ -406,7 +452,7 @@ async function main() {
|
||||
: undefined;
|
||||
const datasetItem = await prisma.datasetItem.create({
|
||||
data: {
|
||||
projectId: project2.id,
|
||||
projectId,
|
||||
datasetId: dataset.id,
|
||||
sourceTraceId: sourceObservation?.traceId,
|
||||
sourceObservationId:
|
||||
@@ -431,9 +477,16 @@ async function main() {
|
||||
}
|
||||
|
||||
for (let datasetRunNumber = 0; datasetRunNumber < 5; datasetRunNumber++) {
|
||||
const datasetRun = await prisma.datasetRuns.create({
|
||||
data: {
|
||||
projectId: project2.id,
|
||||
const datasetRun = await prisma.datasetRuns.upsert({
|
||||
where: {
|
||||
datasetId_projectId_name: {
|
||||
datasetId: dataset.id,
|
||||
projectId,
|
||||
name: `demo-dataset-run-${datasetRunNumber}`,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
projectId,
|
||||
name: `demo-dataset-run-${datasetRunNumber}`,
|
||||
description: Math.random() > 0.5 ? "Dataset run description" : "",
|
||||
datasetId: dataset.id,
|
||||
@@ -445,11 +498,12 @@ async function main() {
|
||||
["tag1", "tag2"],
|
||||
][datasetRunNumber % 5],
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
for (const datasetItemId of datasetItemIds) {
|
||||
const relevantObservations = observations.filter(
|
||||
(o) => o.projectId === project2.id
|
||||
(o) => o.projectId === projectId,
|
||||
);
|
||||
const observation =
|
||||
relevantObservations[
|
||||
@@ -458,7 +512,7 @@ async function main() {
|
||||
|
||||
await prisma.datasetRunItems.create({
|
||||
data: {
|
||||
projectId: project2.id,
|
||||
projectId,
|
||||
datasetItemId,
|
||||
traceId: observation.traceId as string,
|
||||
observationId: Math.random() > 0.5 ? observation.id : undefined,
|
||||
@@ -471,20 +525,6 @@ async function main() {
|
||||
}
|
||||
}
|
||||
|
||||
main()
|
||||
.then(async () => {
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
})
|
||||
.catch(async (e) => {
|
||||
logger.error(e);
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
process.exit(1);
|
||||
});
|
||||
|
||||
async function uploadObjects(
|
||||
traces: Prisma.TraceCreateManyInput[],
|
||||
observations: Prisma.ObservationCreateManyInput[],
|
||||
@@ -492,7 +532,7 @@ async function uploadObjects(
|
||||
sessions: Prisma.TraceSessionCreateManyInput[],
|
||||
events: Prisma.ObservationCreateManyInput[],
|
||||
comments: Prisma.CommentCreateManyInput[],
|
||||
queueItems: Prisma.AnnotationQueueItemCreateManyInput[]
|
||||
queueItems: Prisma.AnnotationQueueItemCreateManyInput[],
|
||||
) {
|
||||
let promises: Prisma.PrismaPromise<unknown>[] = [];
|
||||
|
||||
@@ -506,14 +546,14 @@ async function uploadObjects(
|
||||
},
|
||||
create: chunk[0]!,
|
||||
update: {},
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -524,13 +564,13 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.trace.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -540,14 +580,14 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.observation.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -557,14 +597,14 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.observation.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -574,13 +614,13 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.score.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -590,13 +630,13 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.comment.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -606,13 +646,13 @@ async function uploadObjects(
|
||||
promises.push(
|
||||
prisma.annotationQueueItem.createMany({
|
||||
data: chunk,
|
||||
})
|
||||
}),
|
||||
);
|
||||
});
|
||||
for (let i = 0; i < promises.length; i++) {
|
||||
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
|
||||
logger.info(
|
||||
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`
|
||||
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`,
|
||||
);
|
||||
await promises[i];
|
||||
}
|
||||
@@ -634,7 +674,7 @@ function createObjects(
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[]
|
||||
>
|
||||
>,
|
||||
) {
|
||||
const traces: Prisma.TraceCreateManyInput[] = [];
|
||||
const observations: Prisma.ObservationCreateManyInput[] = [];
|
||||
@@ -649,7 +689,7 @@ function createObjects(
|
||||
// print progress to console with a progress bar that refreshes every 10 iterations
|
||||
// random date within last 90 days, with a linear bias towards more recent dates
|
||||
const traceTs = new Date(
|
||||
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000)
|
||||
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000),
|
||||
);
|
||||
|
||||
const envTag = envTags[Math.floor(Math.random() * envTags.length)];
|
||||
@@ -805,11 +845,11 @@ function createObjects(
|
||||
for (let j = 0; j < Math.floor(Math.random() * 10) + 1; j++) {
|
||||
// add between 1 and 30 ms to trace timestamp
|
||||
const spanTsStart = new Date(
|
||||
traceTs.getTime() + Math.floor(Math.random() * 30)
|
||||
traceTs.getTime() + Math.floor(Math.random() * 30),
|
||||
);
|
||||
// random duration of upto 5000ms
|
||||
const spanTsEnd = new Date(
|
||||
spanTsStart.getTime() + Math.floor(Math.random() * 5000)
|
||||
spanTsStart.getTime() + Math.floor(Math.random() * 5000),
|
||||
);
|
||||
|
||||
const span = {
|
||||
@@ -844,22 +884,22 @@ function createObjects(
|
||||
const generationTsStart = new Date(
|
||||
spanTsStart.getTime() +
|
||||
Math.floor(
|
||||
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
|
||||
)
|
||||
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
|
||||
),
|
||||
);
|
||||
const generationTsEnd = new Date(
|
||||
generationTsStart.getTime() +
|
||||
Math.floor(
|
||||
Math.random() *
|
||||
(spanTsEnd.getTime() - generationTsStart.getTime())
|
||||
)
|
||||
(spanTsEnd.getTime() - generationTsStart.getTime()),
|
||||
),
|
||||
);
|
||||
// somewhere in the middle
|
||||
const generationTsCompletionStart = new Date(
|
||||
generationTsStart.getTime() +
|
||||
Math.floor(
|
||||
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3
|
||||
)
|
||||
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3,
|
||||
),
|
||||
);
|
||||
|
||||
const promptTokens = Math.floor(Math.random() * 1000) + 300;
|
||||
@@ -880,7 +920,7 @@ function createObjects(
|
||||
const promptId =
|
||||
promptIds.get(projectId)![
|
||||
Math.floor(
|
||||
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2)
|
||||
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2),
|
||||
)
|
||||
];
|
||||
|
||||
@@ -962,8 +1002,8 @@ function createObjects(
|
||||
const eventTs = new Date(
|
||||
spanTsStart.getTime() +
|
||||
Math.floor(
|
||||
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
|
||||
)
|
||||
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
|
||||
),
|
||||
);
|
||||
|
||||
events.push({
|
||||
@@ -985,7 +1025,7 @@ function createObjects(
|
||||
}
|
||||
// find unique sessions by id and projectid
|
||||
const uniqueSessions: Prisma.TraceSessionCreateManyInput[] = Array.from(
|
||||
new Set(sessions.map((session) => JSON.stringify(session)))
|
||||
new Set(sessions.map((session) => JSON.stringify(session))),
|
||||
).map((session) => JSON.parse(session) as Prisma.TraceSessionCreateManyInput);
|
||||
|
||||
return {
|
||||
@@ -1007,65 +1047,64 @@ async function generatePromptsForProject(projects: Project[]) {
|
||||
projects.map(async (project) => {
|
||||
const promptIdsForProject = await generatePrompts(project);
|
||||
promptIds.set(project.id, promptIdsForProject);
|
||||
})
|
||||
}),
|
||||
);
|
||||
return promptIds;
|
||||
}
|
||||
|
||||
async function generatePrompts(project: Project) {
|
||||
const promptIds: string[] = [];
|
||||
const prompts = [
|
||||
{
|
||||
id: `prompt-${v4()}`,
|
||||
projectId: project.id,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 1 content",
|
||||
name: "Prompt 1",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-${v4()}`,
|
||||
projectId: project.id,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 2 content",
|
||||
name: "Prompt 2",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-${v4()}`,
|
||||
projectId: project.id,
|
||||
createdBy: "API",
|
||||
prompt: "Prompt 3 content",
|
||||
name: "Prompt 3 by API",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-${v4()}`,
|
||||
projectId: project.id,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 content",
|
||||
name: "Prompt 4",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
];
|
||||
export const SEED_PROMPTS = [
|
||||
{
|
||||
id: `prompt-123`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 1 content",
|
||||
name: "Prompt 1",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-456`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 2 content",
|
||||
name: "Prompt 2",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-789`,
|
||||
createdBy: "API",
|
||||
prompt: "Prompt 3 content",
|
||||
name: "Prompt 3 by API",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-abc`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 content",
|
||||
name: "Prompt 4",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
];
|
||||
|
||||
for (const prompt of prompts) {
|
||||
export const PROMPT_IDS: string[] = [];
|
||||
|
||||
async function generatePrompts(project: Project) {
|
||||
const promptIds = [];
|
||||
for (const prompt of SEED_PROMPTS) {
|
||||
await prisma.prompt.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: prompt.projectId,
|
||||
projectId: prompt.id + project.id,
|
||||
name: prompt.name,
|
||||
version: prompt.version,
|
||||
},
|
||||
id: prompt.id + project.id,
|
||||
},
|
||||
create: {
|
||||
id: prompt.id,
|
||||
projectId: prompt.projectId,
|
||||
id: prompt.id + project.id,
|
||||
projectId: project.id,
|
||||
createdBy: prompt.createdBy,
|
||||
prompt: prompt.prompt,
|
||||
name: prompt.name,
|
||||
@@ -1073,9 +1112,7 @@ async function generatePrompts(project: Project) {
|
||||
labels: prompt.labels,
|
||||
tags: prompt.tags,
|
||||
},
|
||||
update: {
|
||||
id: prompt.id,
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
promptIds.push(prompt.id);
|
||||
}
|
||||
@@ -1129,6 +1166,7 @@ async function generatePrompts(project: Project) {
|
||||
name: version.name,
|
||||
version: version.version,
|
||||
},
|
||||
id: version.id,
|
||||
},
|
||||
create: {
|
||||
id: version.id,
|
||||
@@ -1159,6 +1197,7 @@ async function generatePrompts(project: Project) {
|
||||
name: promptName,
|
||||
version: i,
|
||||
},
|
||||
id: promptId,
|
||||
},
|
||||
create: {
|
||||
id: promptId,
|
||||
@@ -1193,7 +1232,7 @@ async function generateConfigsForProject(projects: Project[]) {
|
||||
projects.map(async (project) => {
|
||||
const configNameAndId = await generateConfigs(project);
|
||||
projectIdsToConfigs.set(project.id, configNameAndId);
|
||||
})
|
||||
}),
|
||||
);
|
||||
return projectIdsToConfigs;
|
||||
}
|
||||
@@ -1343,7 +1382,7 @@ async function generateQueuesForProject(
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[]
|
||||
>
|
||||
>,
|
||||
) {
|
||||
const projectIdsToQueues: Map<string, string[]> = new Map();
|
||||
|
||||
@@ -1351,10 +1390,10 @@ async function generateQueuesForProject(
|
||||
projects.map(async (project) => {
|
||||
const queueIds = await generateQueues(
|
||||
project,
|
||||
configIdsAndNames.get(project.id) ?? []
|
||||
configIdsAndNames.get(project.id) ?? [],
|
||||
);
|
||||
projectIdsToQueues.set(project.id, queueIds);
|
||||
})
|
||||
}),
|
||||
);
|
||||
return projectIdsToQueues;
|
||||
}
|
||||
@@ -1366,7 +1405,7 @@ async function generateQueues(
|
||||
id: string;
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[]
|
||||
}[],
|
||||
) {
|
||||
const queue = {
|
||||
id: `queue-${v4()}`,
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
import { randomUUID } from "crypto";
|
||||
import { prisma } from "../src/db";
|
||||
import {
|
||||
clickhouseClient,
|
||||
getDisplaySecretKey,
|
||||
hashSecretKey,
|
||||
logger,
|
||||
} from "../src/server";
|
||||
import { prepareClickhouse } from "./prepareClickhouse";
|
||||
import { redis } from "../src/server";
|
||||
|
||||
const createRandomProjectId = () => randomUUID().toString();
|
||||
|
||||
const prepareProjectsAndApiKeys = async (
|
||||
numOfProjects: number,
|
||||
opts: { requiredProjectIds: string[] },
|
||||
) => {
|
||||
const { requiredProjectIds } = opts;
|
||||
const projectsToCreate = numOfProjects - requiredProjectIds.length;
|
||||
const projectIds = [...requiredProjectIds];
|
||||
|
||||
for (let i = 0; i < projectsToCreate; i++) {
|
||||
projectIds.push(createRandomProjectId());
|
||||
}
|
||||
|
||||
const operations = projectIds.map(async (projectId) => {
|
||||
const orgId = `org-${projectId}`;
|
||||
|
||||
await prisma.organization.upsert({
|
||||
where: { id: orgId },
|
||||
update: {},
|
||||
create: {
|
||||
id: orgId,
|
||||
name: `Organization for ${projectId}`,
|
||||
},
|
||||
});
|
||||
|
||||
await prisma.project.upsert({
|
||||
where: { id: projectId },
|
||||
update: {},
|
||||
create: {
|
||||
id: projectId,
|
||||
name: `Project ${projectId}`,
|
||||
orgId: orgId,
|
||||
},
|
||||
});
|
||||
|
||||
const apiKeyId = `api-key-${projectId}`;
|
||||
const apiKeyExists = await prisma.apiKey.findUnique({
|
||||
where: { id: apiKeyId },
|
||||
});
|
||||
if (!apiKeyExists) {
|
||||
const sk = await hashSecretKey(
|
||||
`sk-${Math.random().toString(36).substr(2, 9)}`,
|
||||
);
|
||||
await prisma.apiKey.create({
|
||||
data: {
|
||||
id: apiKeyId,
|
||||
note: `API Key for ${projectId}`,
|
||||
publicKey: `pk-${Math.random().toString(36).substr(2, 9)}`,
|
||||
hashedSecretKey: sk,
|
||||
displaySecretKey: getDisplaySecretKey(sk),
|
||||
project: {
|
||||
connect: {
|
||||
id: projectId,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
await Promise.all(operations);
|
||||
return projectIds;
|
||||
};
|
||||
|
||||
async function main() {
|
||||
let numOfProjects = parseInt(process.argv[2], 10);
|
||||
let numberOfDays = parseInt(process.argv[3], 10);
|
||||
let totalObservations = parseInt(process.argv[4], 10);
|
||||
|
||||
logger.info(process.argv);
|
||||
logger.info(
|
||||
`Preparing Clickhouse for ${numOfProjects} projects and ${numberOfDays} days with max Observations ${totalObservations}.`,
|
||||
);
|
||||
|
||||
if (isNaN(totalObservations)) {
|
||||
logger.warn(
|
||||
"Total observations not provided or invalid. Defaulting to 1000 observations.",
|
||||
);
|
||||
totalObservations = 1000;
|
||||
}
|
||||
|
||||
if (isNaN(numOfProjects)) {
|
||||
logger.warn(
|
||||
"Number of projects not provided or invalid. Defaulting to 10 projects.",
|
||||
);
|
||||
numOfProjects = 10;
|
||||
}
|
||||
|
||||
if (isNaN(numberOfDays)) {
|
||||
logger.warn(
|
||||
"Number of days not provided or invalid. Defaulting to 3 days.",
|
||||
);
|
||||
numberOfDays = 3;
|
||||
}
|
||||
|
||||
try {
|
||||
const projectIds = [
|
||||
"7a88fb47-b4e2-43b8-a06c-a5ce950dc53a",
|
||||
"239ad00f-562f-411d-af14-831c75ddd875",
|
||||
];
|
||||
|
||||
const createdProjectIds = await prepareProjectsAndApiKeys(numOfProjects, {
|
||||
requiredProjectIds: projectIds,
|
||||
});
|
||||
|
||||
await prepareClickhouse(createdProjectIds, {
|
||||
numberOfDays,
|
||||
totalObservations: totalObservations ?? 1000,
|
||||
});
|
||||
|
||||
logger.info("Clickhouse preparation completed successfully.");
|
||||
} catch (error) {
|
||||
logger.error("Error during Clickhouse preparation:", error);
|
||||
} finally {
|
||||
await clickhouseClient().close();
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from Clickhouse.");
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -0,0 +1,256 @@
|
||||
import { SEED_PROMPTS } from "../prisma/seed";
|
||||
import { prisma } from "../src/db";
|
||||
import { clickhouseClient, logger } from "../src/server";
|
||||
|
||||
function randn_bm(min: number, max: number, skew: number) {
|
||||
let u = 0,
|
||||
v = 0;
|
||||
while (u === 0) u = Math.random(); //Converting [0,1) to (0,1)
|
||||
while (v === 0) v = Math.random();
|
||||
let num = Math.sqrt(-2.0 * Math.log(u)) * Math.cos(2.0 * Math.PI * v);
|
||||
|
||||
num = num / 10.0 + 0.5; // Translate to 0 -> 1
|
||||
if (num > 1 || num < 0)
|
||||
num = randn_bm(min, max, skew); // resample between 0 and 1 if out of range
|
||||
else {
|
||||
num = Math.pow(num, skew); // Skew
|
||||
num *= max - min; // Stretch to fill range
|
||||
num += min; // offset to min
|
||||
}
|
||||
return num;
|
||||
}
|
||||
|
||||
export const prepareClickhouse = async (
|
||||
projectIds: string[],
|
||||
opts: {
|
||||
numberOfDays: number;
|
||||
totalObservations: number;
|
||||
},
|
||||
) => {
|
||||
logger.info(
|
||||
`Preparing Clickhouse for ${projectIds.length} projects and ${opts.numberOfDays} days.`,
|
||||
);
|
||||
|
||||
const projectData = projectIds.map((projectId) => {
|
||||
const observationsPerProject = Math.ceil(
|
||||
randn_bm(0, opts.totalObservations, 2),
|
||||
); // Skew the number of observations
|
||||
|
||||
const tracesPerProject = Math.floor(observationsPerProject / 6); // On average, one trace should have 6 observations
|
||||
const scoresPerProject = tracesPerProject * 10; // On average, one trace should have 10 scores
|
||||
return {
|
||||
projectId,
|
||||
observationsPerProject,
|
||||
tracesPerProject,
|
||||
scoresPerProject,
|
||||
};
|
||||
});
|
||||
|
||||
for (const data of projectData) {
|
||||
const {
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
observationsPerProject,
|
||||
scoresPerProject,
|
||||
} = data;
|
||||
logger.info(
|
||||
`Preparing Clickhouse for ${projectId}: Traces: ${tracesPerProject}, Scores: ${scoresPerProject}, Observations: ${observationsPerProject}`,
|
||||
);
|
||||
|
||||
const tracesQuery = `
|
||||
INSERT INTO traces
|
||||
SELECT toString(number) AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
concat('name_', toString(rand() % 100)) AS name,
|
||||
concat('user_id_', toInt64(randExponential(1 / 100))) AS user_id,
|
||||
map('key', 'value') AS metadata,
|
||||
concat('release_', toString(randUniform(0, 100))) AS release,
|
||||
concat('version_', toString(randUniform(0, 100))) AS version,
|
||||
'${projectId}' AS project_id,
|
||||
if(rand() < 0.8, true, false) as public,
|
||||
if(rand() < 0.8, true, false) as bookmarked,
|
||||
array('tag1', 'tag2') as tags,
|
||||
repeat('input', toInt64(randExponential(1 / 100))) AS input,
|
||||
repeat('output', toInt64(randExponential(1 / 100))) AS output,
|
||||
if(randUniform(0, 1) < 0.2, NULL, concat('session_', toString(rand() % 1000))) AS session_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${tracesPerProject});
|
||||
`;
|
||||
|
||||
const observationsQuery = `
|
||||
INSERT INTO observations
|
||||
SELECT toString(number) AS id,
|
||||
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
|
||||
'${projectId}' AS project_id,
|
||||
if(randUniform(0, 1) < 0.47, 'GENERATION', if(randUniform(0, 1) < 0.94, 'SPAN', 'EVENT')) AS type,
|
||||
toString(rand()) AS parent_observation_id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS start_time,
|
||||
addSeconds(start_time, if(rand() < 0.6, floor(randUniform(0, 20)), floor(randUniform(0, 3600)))) AS end_time,
|
||||
concat('name', toString(rand() % 100)) AS name,
|
||||
map('key', 'value') AS metadata,
|
||||
if(randUniform(0, 1) < 0.9, 'DEFAULT', if(randUniform(0, 1) < 0.5, 'ERROR', if(randUniform(0, 1) < 0.5, 'DEBUG', 'WARNING'))) AS level,
|
||||
'status_message' AS status_message,
|
||||
'version' AS version,
|
||||
repeat('input', toInt64(randExponential(1 / 100))) AS input,
|
||||
repeat('output', toInt64(randExponential(1 / 100))) AS output,
|
||||
case
|
||||
when number % 2 = 0 then 'claude-3-haiku-20240307'
|
||||
else 'gpt-4'
|
||||
end as provided_model_name,
|
||||
case
|
||||
when number % 2 = 0 then 'cltr0w45b000008k1407o9qv1'
|
||||
else 'clrntkjgy000f08jx79v9g1xj'
|
||||
end as internal_model_id,
|
||||
'{"temperature": 0.7, "max_tokens": 150}' AS model_parameters,
|
||||
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS provided_usage_details,
|
||||
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS usage_details,
|
||||
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS provided_cost_details,
|
||||
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS cost_details,
|
||||
toDecimal64(randUniform(0, 2000), 12) AS total_cost,
|
||||
start_time AS completion_start_time,
|
||||
array(${SEED_PROMPTS.map((p) => `concat('${p.id}',project_id)`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_id,
|
||||
array(${SEED_PROMPTS.map((p) => `'${p.name}'`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_name,
|
||||
array(${SEED_PROMPTS.map((p) => `'${p.version}'`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_version,
|
||||
start_time AS created_at,
|
||||
start_time AS updated_at,
|
||||
start_time AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${observationsPerProject});
|
||||
`;
|
||||
|
||||
console.log(observationsQuery);
|
||||
|
||||
const scoresQuery = `
|
||||
INSERT INTO scores
|
||||
SELECT toString(floor(randUniform(0, 100))) AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
'${projectId}' AS project_id,
|
||||
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
|
||||
if(
|
||||
rand() > 0.9,
|
||||
toString(floor(randUniform(0, ${observationsPerProject}))),
|
||||
NULL
|
||||
) AS observation_id,
|
||||
concat('name_', toString(rand() % 10)) AS name,
|
||||
randUniform(0, 100) as value,
|
||||
'API' as source,
|
||||
'comment' as comment,
|
||||
toString(rand() % 100) as author_user_id,
|
||||
toString(rand() % 100) as config_id,
|
||||
if (rand() < 0.33, 'NUMERIC', if (rand() < 0.5, 'CATEGORICAL', 'BOOLEAN')) as data_type,
|
||||
toString(rand() % 100) as string_value,
|
||||
NULL as queue_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${scoresPerProject});
|
||||
`;
|
||||
|
||||
const queries = [tracesQuery, scoresQuery, observationsQuery];
|
||||
|
||||
for (const query of queries) {
|
||||
logger.info(`Executing query: ${query}`);
|
||||
await clickhouseClient().command({
|
||||
query,
|
||||
clickhouse_settings: {
|
||||
wait_end_of_query: 1,
|
||||
},
|
||||
});
|
||||
}
|
||||
// we also need to upsert trace sessions in postgres
|
||||
|
||||
const sessionQuery = `
|
||||
SELECT session_id, project_id
|
||||
FROM traces
|
||||
WHERE session_id IS NOT NULL;
|
||||
`;
|
||||
const sessionResult = await clickhouseClient().query({
|
||||
query: sessionQuery,
|
||||
format: "JSONEachRow",
|
||||
});
|
||||
|
||||
const sessionData = await sessionResult.json<{
|
||||
session_id: string;
|
||||
project_id: string;
|
||||
}>();
|
||||
|
||||
const idProjectIdCombinations = sessionData.map((session) => ({
|
||||
id: session.session_id,
|
||||
projectId: session.project_id,
|
||||
public: Math.random() < 0.1,
|
||||
bookmarked: Math.random() < 0.1,
|
||||
}));
|
||||
|
||||
await prisma.traceSession.createMany({
|
||||
data: idProjectIdCombinations,
|
||||
skipDuplicates: true,
|
||||
});
|
||||
}
|
||||
|
||||
const tables = ["traces", "scores", "observations"];
|
||||
for (const table of tables) {
|
||||
const query = `
|
||||
SELECT
|
||||
project_id,
|
||||
count() AS per_project_count,
|
||||
bar(per_project_count, 0, (
|
||||
SELECT count(*)
|
||||
FROM ${table}
|
||||
), 50) AS bar_representation
|
||||
FROM ${table}
|
||||
GROUP BY project_id
|
||||
ORDER BY count() desc
|
||||
`;
|
||||
|
||||
const result = await clickhouseClient().query({
|
||||
query,
|
||||
format: "TabSeparated",
|
||||
});
|
||||
|
||||
logger.info(
|
||||
`${table.charAt(0).toUpperCase() + table.slice(1)} per Project: \n` +
|
||||
(await result.text()),
|
||||
);
|
||||
}
|
||||
|
||||
const tablesWithDateColumns = [
|
||||
{ name: "traces", dateColumn: "timestamp" },
|
||||
{ name: "scores", dateColumn: "timestamp" },
|
||||
{ name: "observations", dateColumn: "start_time" },
|
||||
];
|
||||
|
||||
for (const { name: table, dateColumn } of tablesWithDateColumns) {
|
||||
const query = `
|
||||
SELECT
|
||||
toDate(${dateColumn}) AS event_date,
|
||||
count() AS per_date_count,
|
||||
bar(per_date_count, 0, (
|
||||
SELECT count(*)
|
||||
FROM ${table}
|
||||
), 50) AS bar_representation
|
||||
FROM ${table}
|
||||
GROUP BY event_date
|
||||
ORDER BY event_date desc
|
||||
`;
|
||||
|
||||
const result = await clickhouseClient().query({
|
||||
query,
|
||||
format: "TabSeparated",
|
||||
});
|
||||
|
||||
logger.info(
|
||||
`${table.charAt(0).toUpperCase() + table.slice(1)} per Date: \n` +
|
||||
(await result.text()),
|
||||
);
|
||||
}
|
||||
};
|
||||
+33
-30
@@ -21,37 +21,40 @@ export class PrismaClientSingleton {
|
||||
return PrismaClientSingleton.instance;
|
||||
}
|
||||
|
||||
const client = new PrismaClient<
|
||||
Prisma.PrismaClientOptions,
|
||||
"warn" | "error" | "query"
|
||||
>({
|
||||
log: [
|
||||
{ emit: "event", level: "query" },
|
||||
{ emit: "event", level: "error" },
|
||||
{ emit: "event", level: "warn" },
|
||||
],
|
||||
});
|
||||
|
||||
if (env.NODE_ENV === "development") {
|
||||
client.$on("query", (event) => {
|
||||
logger.info(`prisma:query ${event.query}, ${event.duration}ms`);
|
||||
});
|
||||
}
|
||||
|
||||
client.$on("warn", (event) => {
|
||||
logger.warn(`prisma:warn ${event.message}`);
|
||||
});
|
||||
|
||||
client.$on("error", (event) => {
|
||||
logger.error(`prisma:error ${event.message}`);
|
||||
});
|
||||
|
||||
PrismaClientSingleton.instance = client;
|
||||
PrismaClientSingleton.instance = createPrismaInstance();
|
||||
|
||||
return PrismaClientSingleton.instance;
|
||||
}
|
||||
}
|
||||
|
||||
const createPrismaInstance = () => {
|
||||
const client = new PrismaClient<
|
||||
Prisma.PrismaClientOptions,
|
||||
"warn" | "error" | "query"
|
||||
>({
|
||||
log: [
|
||||
{ emit: "event", level: "query" },
|
||||
{ emit: "event", level: "error" },
|
||||
{ emit: "event", level: "warn" },
|
||||
],
|
||||
});
|
||||
|
||||
if (env.NODE_ENV === "development") {
|
||||
client.$on("query", (event) => {
|
||||
logger.info(`prisma:query ${event.query}, ${event.duration}ms`);
|
||||
});
|
||||
}
|
||||
|
||||
client.$on("warn", (event) => {
|
||||
logger.warn(`prisma:warn ${event.message}`);
|
||||
});
|
||||
|
||||
client.$on("error", (event) => {
|
||||
logger.error(`prisma:error ${event.message}`);
|
||||
});
|
||||
return client;
|
||||
};
|
||||
|
||||
export class KyselySingleton {
|
||||
private static instance: { $kysely: Kysely<DB> };
|
||||
|
||||
@@ -73,7 +76,7 @@ export class KyselySingleton {
|
||||
createQueryCompiler: () => new PostgresQueryCompiler(),
|
||||
},
|
||||
}),
|
||||
})
|
||||
}),
|
||||
);
|
||||
|
||||
return KyselySingleton.instance;
|
||||
@@ -86,7 +89,7 @@ declare const globalThis: {
|
||||
} & typeof global;
|
||||
|
||||
if (process.env.NODE_ENV === "development") {
|
||||
globalThis.prismaGlobal ??= new PrismaClient(); // regular instantiation
|
||||
globalThis.prismaGlobal ??= createPrismaInstance(); // regular instantiation
|
||||
globalThis.kyselyPrismaGlobal ??= globalThis.prismaGlobal.$extends(
|
||||
kyselyExtension({
|
||||
kysely: (driver) =>
|
||||
@@ -100,9 +103,9 @@ if (process.env.NODE_ENV === "development") {
|
||||
createQueryCompiler: () => new PostgresQueryCompiler(),
|
||||
},
|
||||
}),
|
||||
})
|
||||
}),
|
||||
);
|
||||
};
|
||||
}
|
||||
|
||||
export const prisma =
|
||||
globalThis.prismaGlobal ?? PrismaClientSingleton.getInstance();
|
||||
|
||||
@@ -30,15 +30,36 @@ const EnvSchema = z.object({
|
||||
CLICKHOUSE_URL: z.string().url().optional(),
|
||||
CLICKHOUSE_USER: z.string().optional(),
|
||||
CLICKHOUSE_PASSWORD: z.string().optional(),
|
||||
LANGFUSE_INGESTION_BUFFER_TTL_SECONDS: z.coerce
|
||||
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: z
|
||||
.enum(["true", "false"])
|
||||
.default("false"),
|
||||
LANGFUSE_ASYNC_INGESTION_PROCESSING: z
|
||||
.enum(["true", "false"])
|
||||
.default("false"),
|
||||
LANGFUSE_INGESTION_QUEUE_DELAY_MS: z.coerce
|
||||
.number()
|
||||
.positive()
|
||||
.default(60 * 10),
|
||||
.nonnegative()
|
||||
.default(15_000),
|
||||
SALT: z.string().optional(), // used by components imported by web package
|
||||
LANGFUSE_LOG_LEVEL: z
|
||||
.enum(["trace", "debug", "info", "warn", "error", "fatal"])
|
||||
.optional(),
|
||||
LANGFUSE_LOG_FORMAT: z.enum(["text", "json"]).default("text"),
|
||||
ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING: z
|
||||
.enum(["true", "false"])
|
||||
.default("false"),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: z.enum(["true", "false"]).default("false"),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: z.string().optional(),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: z.string().default(""),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION: z.string().optional(),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: z.string().optional(),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: z.string().optional(),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: z.string().optional(),
|
||||
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: z
|
||||
.enum(["true", "false"])
|
||||
.default("false"),
|
||||
LANGFUSE_USE_AZURE_BLOB: z.enum(["true", "false"]).default("false"),
|
||||
STRIPE_SECRET_KEY: z.string().optional(),
|
||||
});
|
||||
|
||||
export const env = EnvSchema.parse(removeEmptyEnvVariables(process.env));
|
||||
|
||||
@@ -5,6 +5,7 @@ export const langfuseObjects = [
|
||||
"span",
|
||||
"generation",
|
||||
"event",
|
||||
"dataset_item",
|
||||
] as const;
|
||||
|
||||
// variable mapping stored in the db for eval templates
|
||||
@@ -21,7 +22,7 @@ export const variableMapping = z
|
||||
(value) => value.langfuseObject === "trace" || value.objectName !== null,
|
||||
{
|
||||
message: "objectName is required for langfuseObjects other than trace",
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
export const variableMappingList = z.array(variableMapping);
|
||||
@@ -44,7 +45,7 @@ const observationCols = [
|
||||
{ name: "Output", id: "output", internal: 'o."output"' },
|
||||
];
|
||||
|
||||
export const availableEvalVariables = [
|
||||
export const availableTraceEvalVariables = [
|
||||
{
|
||||
id: "trace",
|
||||
display: "Trace",
|
||||
@@ -76,6 +77,28 @@ export const availableEvalVariables = [
|
||||
},
|
||||
];
|
||||
|
||||
export const availableDatasetEvalVariables = [
|
||||
{
|
||||
id: "dataset_item",
|
||||
display: "Dataset item",
|
||||
availableColumns: [
|
||||
{
|
||||
name: "Metadata",
|
||||
id: "metadata",
|
||||
type: "stringObject",
|
||||
internal: 'd."metadata"',
|
||||
},
|
||||
{ name: "Input", id: "input", internal: 'd."input"' },
|
||||
{
|
||||
name: "Expected output",
|
||||
id: "expected_output",
|
||||
internal: 'd."expected_output"',
|
||||
},
|
||||
],
|
||||
},
|
||||
...availableTraceEvalVariables,
|
||||
];
|
||||
|
||||
export const OutputSchema = z.object({
|
||||
reasoning: z.string(),
|
||||
score: z.string(),
|
||||
@@ -83,6 +106,7 @@ export const OutputSchema = z.object({
|
||||
|
||||
export enum EvalTargetObject {
|
||||
Trace = "trace",
|
||||
Dataset = "dataset",
|
||||
}
|
||||
|
||||
export const DEFAULT_TRACE_JOB_DELAY = 10_000;
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
export * from "./scoreTypes";
|
||||
export * from "./types";
|
||||
export * from "./scoreConfigTypes";
|
||||
@@ -55,6 +55,7 @@ const ScoreBase = z.object({
|
||||
configId: z.string().nullish(),
|
||||
createdAt: z.coerce.date(),
|
||||
updatedAt: z.coerce.date(),
|
||||
queueId: z.string().nullish(),
|
||||
});
|
||||
|
||||
const BaseScoreBody = z.object({
|
||||
@@ -83,13 +84,13 @@ export const ScoreBodyWithoutConfig = z.discriminatedUnion("dataType", [
|
||||
z.object({
|
||||
value: z.number(),
|
||||
dataType: z.literal("NUMERIC"),
|
||||
}),
|
||||
})
|
||||
),
|
||||
BaseScoreBody.merge(
|
||||
z.object({
|
||||
value: z.string(),
|
||||
dataType: z.literal("CATEGORICAL"),
|
||||
}),
|
||||
})
|
||||
),
|
||||
BaseScoreBody.merge(
|
||||
z.object({
|
||||
@@ -97,7 +98,7 @@ export const ScoreBodyWithoutConfig = z.discriminatedUnion("dataType", [
|
||||
message: "Value must be either 0 or 1",
|
||||
}),
|
||||
dataType: z.literal("BOOLEAN"),
|
||||
}),
|
||||
})
|
||||
),
|
||||
]);
|
||||
|
||||
@@ -161,7 +162,7 @@ export const ScorePropsAgainstConfig = z.union([
|
||||
*/
|
||||
export const filterAndValidateDbScoreList = (
|
||||
scores: Score[],
|
||||
onParseError?: (error: z.ZodError) => void,
|
||||
onParseError?: (error: z.ZodError) => void
|
||||
): APIScore[] =>
|
||||
scores.reduce((acc, ts) => {
|
||||
const result = APIScoreSchema.safeParse(ts);
|
||||
@@ -198,14 +199,14 @@ export const PostScoresBody = z.discriminatedUnion("dataType", [
|
||||
value: z.number(),
|
||||
dataType: z.literal("NUMERIC"),
|
||||
configId: z.string().nullish(),
|
||||
}),
|
||||
})
|
||||
),
|
||||
BaseScoreBody.merge(
|
||||
z.object({
|
||||
value: z.string(),
|
||||
dataType: z.literal("CATEGORICAL"),
|
||||
configId: z.string().nullish(),
|
||||
}),
|
||||
})
|
||||
),
|
||||
BaseScoreBody.merge(
|
||||
z.object({
|
||||
@@ -215,14 +216,14 @@ export const PostScoresBody = z.discriminatedUnion("dataType", [
|
||||
}),
|
||||
dataType: z.literal("BOOLEAN"),
|
||||
configId: z.string().nullish(),
|
||||
}),
|
||||
})
|
||||
),
|
||||
BaseScoreBody.merge(
|
||||
z.object({
|
||||
value: z.union([z.string(), z.number()]),
|
||||
dataType: z.undefined(),
|
||||
configId: z.string().nullish(),
|
||||
}),
|
||||
})
|
||||
),
|
||||
]);
|
||||
|
||||
@@ -234,6 +235,8 @@ export const GetScoresQuery = z.object({
|
||||
userId: z.string().nullish(),
|
||||
dataType: z.enum(ScoreDataType).nullish(),
|
||||
configId: z.string().nullish(),
|
||||
queueId: z.string().nullish(),
|
||||
traceTags: z.union([z.array(z.string()), z.string()]).nullish(),
|
||||
name: z.string().nullish(),
|
||||
fromTimestamp: stringDateTime,
|
||||
toTimestamp: stringDateTime,
|
||||
@@ -255,8 +258,9 @@ const LegacyGetScoreResponseDataV1 = z.intersection(
|
||||
z.object({
|
||||
trace: z.object({
|
||||
userId: z.string().nullish(),
|
||||
tags: z.array(z.string()).nullish(),
|
||||
}),
|
||||
}),
|
||||
})
|
||||
);
|
||||
export const GetScoresResponse = z.object({
|
||||
data: z.array(LegacyGetScoreResponseDataV1),
|
||||
@@ -265,7 +269,7 @@ export const GetScoresResponse = z.object({
|
||||
|
||||
export const legacyFilterAndValidateV1GetScoreList = (
|
||||
scores: unknown[],
|
||||
onParseError?: (error: z.ZodError) => void,
|
||||
onParseError?: (error: z.ZodError) => void
|
||||
): z.infer<typeof LegacyGetScoreResponseDataV1>[] =>
|
||||
scores.reduce(
|
||||
(acc: z.infer<typeof LegacyGetScoreResponseDataV1>[], ts) => {
|
||||
@@ -278,7 +282,7 @@ export const legacyFilterAndValidateV1GetScoreList = (
|
||||
}
|
||||
return acc;
|
||||
},
|
||||
[] as z.infer<typeof LegacyGetScoreResponseDataV1>[],
|
||||
[] as z.infer<typeof LegacyGetScoreResponseDataV1>[]
|
||||
);
|
||||
|
||||
// GET /scores/{scoreId}
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
import { ScoreDataType, ScoreSource } from "@prisma/client";
|
||||
|
||||
export type CategoricalAggregate = {
|
||||
type: "CATEGORICAL";
|
||||
values: string[];
|
||||
valueCounts: { value: string; count: number }[];
|
||||
comment?: string | null;
|
||||
};
|
||||
|
||||
export type NumericAggregate = {
|
||||
type: "NUMERIC";
|
||||
values: number[];
|
||||
average: number;
|
||||
comment?: string | null;
|
||||
};
|
||||
|
||||
export type ScoreAggregate = Record<
|
||||
string,
|
||||
CategoricalAggregate | NumericAggregate
|
||||
>;
|
||||
|
||||
export type ScoreSimplified = {
|
||||
name: string;
|
||||
dataType: ScoreDataType;
|
||||
source: ScoreSource;
|
||||
value?: number | null;
|
||||
comment?: string | null;
|
||||
stringValue?: string | null;
|
||||
};
|
||||
|
||||
export type LastUserScore = ScoreSimplified & {
|
||||
timestamp: string;
|
||||
traceId: string;
|
||||
observationId?: string | null;
|
||||
|
||||
userId: string;
|
||||
};
|
||||
@@ -6,11 +6,12 @@ export * from "./interfaces/parseDbOrg";
|
||||
export * from "./interfaces/customLLMProviderConfigSchemas";
|
||||
export * from "./tableDefinitions";
|
||||
export * from "./types";
|
||||
export * from "./tracesTable";
|
||||
export * from "./tableDefinitions/tracesTable";
|
||||
export * from "./server/auth/apiKeys";
|
||||
export * from "./observationsTable";
|
||||
export * from "./utils/zod";
|
||||
export * from "./utils/json";
|
||||
export * from "./utils/stringChecks";
|
||||
export * from "./utils/objects";
|
||||
export * from "./utils/typeChecks";
|
||||
export * from "./features/entitlements/plans";
|
||||
@@ -28,8 +29,7 @@ export * from "./features/batchExport/types";
|
||||
export * from "./features/annotation/types";
|
||||
|
||||
// scores
|
||||
export * from "./features/scores/scoreConfigTypes";
|
||||
export * from "./features/scores/scoreTypes";
|
||||
export * from "./features/scores";
|
||||
|
||||
// comments
|
||||
export * from "./features/comments/types";
|
||||
|
||||
@@ -15,6 +15,7 @@ export const filterOperators = {
|
||||
],
|
||||
numberObject: ["=", ">", "<", ">=", "<="],
|
||||
boolean: ["=", "<>"],
|
||||
null: ["is null", "is not null"],
|
||||
} as const;
|
||||
|
||||
export const timeFilter = z.object({
|
||||
@@ -68,6 +69,12 @@ export const booleanFilter = z.object({
|
||||
operator: z.enum(filterOperators.boolean),
|
||||
value: z.boolean(),
|
||||
});
|
||||
export const nullFilter = z.object({
|
||||
type: z.literal("null"),
|
||||
column: z.string(),
|
||||
operator: z.enum(filterOperators.null),
|
||||
value: z.literal(""),
|
||||
});
|
||||
export const singleFilter = z.discriminatedUnion("type", [
|
||||
timeFilter,
|
||||
stringFilter,
|
||||
@@ -77,4 +84,5 @@ export const singleFilter = z.discriminatedUnion("type", [
|
||||
stringObjectFilter,
|
||||
numberObjectFilter,
|
||||
booleanFilter,
|
||||
nullFilter,
|
||||
]);
|
||||
|
||||
@@ -17,7 +17,6 @@ export function CustomSSOProvider<P extends CustomSSOUser>(
|
||||
wellKnown: `${options.issuer}/.well-known/openid-configuration`,
|
||||
authorization: { params: { scope: "openid email profile" } }, // overridden by options.authorization to be able to set custom scopes, deep merged with this default
|
||||
checks: ["pkce", "state"],
|
||||
idToken: true,
|
||||
profile(profile) {
|
||||
return {
|
||||
id: profile.sub,
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
import type { OAuthConfig, OAuthUserConfig } from "next-auth/providers/oauth";
|
||||
import type { GithubProfile, GithubEmail } from "next-auth/providers/github";
|
||||
|
||||
export function GitHubEnterpriseProvider<P extends GithubProfile>(
|
||||
options: OAuthUserConfig<P> & {
|
||||
enterprise?: {
|
||||
baseUrl?: string;
|
||||
};
|
||||
}
|
||||
): OAuthConfig<P> {
|
||||
const baseUrl = options?.enterprise?.baseUrl ?? "https://github.com"
|
||||
const apiBaseUrl = options?.enterprise?.baseUrl
|
||||
? `${options?.enterprise?.baseUrl}/api/v3`
|
||||
: "https://api.github.com"
|
||||
|
||||
return {
|
||||
id: "github-enterprise",
|
||||
name: "GitHub Enterprise",
|
||||
type: "oauth",
|
||||
authorization: {
|
||||
url: `${baseUrl}/login/oauth/authorize`,
|
||||
params: { scope: "read:user user:email" },
|
||||
},
|
||||
token: `${baseUrl}/login/oauth/access_token`,
|
||||
userinfo: {
|
||||
url: `${apiBaseUrl}/user`,
|
||||
async request({ client, tokens }) {
|
||||
const profile = await client.userinfo(tokens.access_token!)
|
||||
|
||||
if (!profile.email) {
|
||||
// If the user does not have a public email, get another via the GitHub API
|
||||
// See https://docs.github.com/en/rest/users/emails#list-email-addresses-for-the-authenticated-user
|
||||
const res = await fetch(`${apiBaseUrl}/user/emails`, {
|
||||
headers: { Authorization: `token ${tokens.access_token}` },
|
||||
})
|
||||
|
||||
if (res.ok) {
|
||||
const emails: GithubEmail[] = await res.json()
|
||||
profile.email = (emails.find((e) => e.primary) ?? emails[0]).email
|
||||
}
|
||||
}
|
||||
|
||||
return profile
|
||||
},
|
||||
},
|
||||
profile(profile) {
|
||||
return {
|
||||
id: profile.id.toString(),
|
||||
name: profile.name ?? profile.login,
|
||||
email: profile.email,
|
||||
image: profile.avatar_url,
|
||||
}
|
||||
},
|
||||
style: {
|
||||
logo: "https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github.svg",
|
||||
logoDark:
|
||||
"https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github-dark.svg",
|
||||
bg: "#fff",
|
||||
bgDark: "#000",
|
||||
text: "#000",
|
||||
textDark: "#fff",
|
||||
},
|
||||
options,
|
||||
}
|
||||
}
|
||||
@@ -1,48 +0,0 @@
|
||||
import { createClient } from "@clickhouse/client";
|
||||
|
||||
import { env } from "../env";
|
||||
import { eventTypes } from "./ingestion/types";
|
||||
import { LangfuseNotFoundError } from "../errors";
|
||||
|
||||
export enum ClickhouseEntityType {
|
||||
Trace = "trace",
|
||||
Score = "score",
|
||||
Observation = "observation",
|
||||
SdkLog = "sdk-log",
|
||||
}
|
||||
|
||||
export function getClickhouseEntityType(
|
||||
eventType: string,
|
||||
): ClickhouseEntityType {
|
||||
switch (eventType) {
|
||||
case eventTypes.TRACE_CREATE:
|
||||
return ClickhouseEntityType.Trace;
|
||||
case eventTypes.OBSERVATION_CREATE:
|
||||
case eventTypes.OBSERVATION_UPDATE:
|
||||
case eventTypes.EVENT_CREATE:
|
||||
case eventTypes.SPAN_CREATE:
|
||||
case eventTypes.SPAN_UPDATE:
|
||||
case eventTypes.GENERATION_CREATE:
|
||||
case eventTypes.GENERATION_UPDATE:
|
||||
return ClickhouseEntityType.Observation;
|
||||
case eventTypes.SCORE_CREATE:
|
||||
return ClickhouseEntityType.Score;
|
||||
case eventTypes.SDK_LOG:
|
||||
return ClickhouseEntityType.SdkLog;
|
||||
default:
|
||||
throw new LangfuseNotFoundError(`Unknown event type: ${eventType}`);
|
||||
}
|
||||
}
|
||||
|
||||
export type ClickhouseClientType = ReturnType<typeof createClient>;
|
||||
|
||||
export const clickhouseClient = createClient({
|
||||
url: env.CLICKHOUSE_URL,
|
||||
username: env.CLICKHOUSE_USER,
|
||||
password: env.CLICKHOUSE_PASSWORD,
|
||||
database: "default",
|
||||
clickhouse_settings: {
|
||||
async_insert: 1,
|
||||
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
|
||||
},
|
||||
});
|
||||
@@ -0,0 +1,28 @@
|
||||
import { createClient } from "@clickhouse/client";
|
||||
import { env } from "../../env";
|
||||
import { NodeClickHouseClientConfigOptions } from "@clickhouse/client/dist/config";
|
||||
|
||||
export type ClickhouseClientType = ReturnType<typeof createClient>;
|
||||
|
||||
export const clickhouseClient = (opts?: NodeClickHouseClientConfigOptions) =>
|
||||
createClient({
|
||||
...opts,
|
||||
url: env.CLICKHOUSE_URL,
|
||||
username: env.CLICKHOUSE_USER,
|
||||
password: env.CLICKHOUSE_PASSWORD,
|
||||
database: "default",
|
||||
clickhouse_settings: {
|
||||
async_insert: 1,
|
||||
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
|
||||
},
|
||||
});
|
||||
|
||||
export const defaultClickhouseClient = clickhouseClient();
|
||||
/**
|
||||
* Accepts a JavaScript date and returns the DateTime in format YYYY-MM-DD HH:MM:SS
|
||||
*/
|
||||
|
||||
export const convertDateToClickhouseDateTime = (date: Date): string => {
|
||||
// 2024-11-06T20:37:00.123Z -> 2024-11-06 21:37:00.123
|
||||
return date.toISOString().replace("T", " ").replace("Z", "");
|
||||
};
|
||||
@@ -0,0 +1,7 @@
|
||||
export const ClickhouseTableNames = {
|
||||
traces: "traces",
|
||||
observations: "observations",
|
||||
scores: "scores",
|
||||
} as const;
|
||||
|
||||
export type ClickhouseTableName = keyof typeof ClickhouseTableNames;
|
||||
@@ -0,0 +1,37 @@
|
||||
import { LangfuseNotFoundError } from "../../errors";
|
||||
import { eventTypes } from "../ingestion/types";
|
||||
import { ClickhouseTableName, ClickhouseTableNames } from "./schema";
|
||||
|
||||
export const isValidTableName = (
|
||||
tableName: string,
|
||||
): tableName is ClickhouseTableName =>
|
||||
Object.keys(ClickhouseTableNames).includes(tableName);
|
||||
|
||||
export type IngestionEntityTypes =
|
||||
| "trace"
|
||||
| "observation"
|
||||
| "score"
|
||||
| "sdk_log";
|
||||
|
||||
export const getClickhouseEntityType = (
|
||||
eventType: string,
|
||||
): IngestionEntityTypes => {
|
||||
switch (eventType) {
|
||||
case eventTypes.TRACE_CREATE:
|
||||
return "trace";
|
||||
case eventTypes.OBSERVATION_CREATE:
|
||||
case eventTypes.OBSERVATION_UPDATE:
|
||||
case eventTypes.EVENT_CREATE:
|
||||
case eventTypes.SPAN_CREATE:
|
||||
case eventTypes.SPAN_UPDATE:
|
||||
case eventTypes.GENERATION_CREATE:
|
||||
case eventTypes.GENERATION_UPDATE:
|
||||
return "observation";
|
||||
case eventTypes.SCORE_CREATE:
|
||||
return "score";
|
||||
case eventTypes.SDK_LOG:
|
||||
return "sdk_log";
|
||||
default:
|
||||
throw new LangfuseNotFoundError(`Unknown event type: ${eventType}`);
|
||||
}
|
||||
};
|
||||
@@ -1,167 +0,0 @@
|
||||
import z from "zod";
|
||||
|
||||
export const clickhouseStringDateSchema = z
|
||||
.string()
|
||||
// clickhouse stores UTC like '2024-05-23 18:33:41.602000'
|
||||
// we need to convert it to '2024-05-23T18:33:41.602000Z'
|
||||
.transform((str) => str.replace(" ", "T") + "Z")
|
||||
.pipe(z.string().datetime());
|
||||
|
||||
export const observationRecordBaseSchema = z.object({
|
||||
id: z.string(),
|
||||
trace_id: z.string().nullish(),
|
||||
project_id: z.string(),
|
||||
type: z.string(),
|
||||
parent_observation_id: z.string().nullish(),
|
||||
name: z.string().nullish(),
|
||||
metadata: z.record(z.string()),
|
||||
level: z.string().nullish(),
|
||||
status_message: z.string().nullish(),
|
||||
version: z.string().nullish(),
|
||||
input: z.string().nullish(),
|
||||
output: z.string().nullish(),
|
||||
provided_model_name: z.string().nullish(),
|
||||
internal_model_id: z.string().nullish(),
|
||||
model_parameters: z.string().nullish(),
|
||||
unit: z.string().nullish(),
|
||||
input_usage_units: z.number().nullish(),
|
||||
output_usage_units: z.number().nullish(),
|
||||
total_usage_units: z.number().nullish(),
|
||||
input_cost: z.number().nullish(),
|
||||
output_cost: z.number().nullish(),
|
||||
total_cost: z.number().nullish(),
|
||||
provided_input_usage_units: z.number().nullish(),
|
||||
provided_output_usage_units: z.number().nullish(),
|
||||
provided_total_usage_units: z.number().nullish(),
|
||||
provided_input_cost: z.number().nullish(),
|
||||
provided_output_cost: z.number().nullish(),
|
||||
provided_total_cost: z.number().nullish(),
|
||||
prompt_id: z.string().nullish(),
|
||||
prompt_name: z.string().nullish(),
|
||||
prompt_version: z.number().nullish(),
|
||||
});
|
||||
export type ObservationRecordBaseType = z.infer<
|
||||
typeof observationRecordBaseSchema
|
||||
>;
|
||||
|
||||
export const observationRecordReadSchema = observationRecordBaseSchema.extend({
|
||||
created_at: clickhouseStringDateSchema,
|
||||
updated_at: clickhouseStringDateSchema,
|
||||
start_time: clickhouseStringDateSchema,
|
||||
end_time: clickhouseStringDateSchema.nullish(),
|
||||
completion_start_time: clickhouseStringDateSchema.nullish(),
|
||||
});
|
||||
export type ObservationRecordReadType = z.infer<
|
||||
typeof observationRecordReadSchema
|
||||
>;
|
||||
|
||||
export const observationRecordInsertSchema = observationRecordBaseSchema.extend(
|
||||
{
|
||||
created_at: z.number(),
|
||||
updated_at: z.number(),
|
||||
start_time: z.number(),
|
||||
end_time: z.number().nullish(),
|
||||
completion_start_time: z.number().nullish(),
|
||||
}
|
||||
);
|
||||
export type ObservationRecordInsertType = z.infer<
|
||||
typeof observationRecordInsertSchema
|
||||
>;
|
||||
|
||||
export const traceRecordBaseSchema = z.object({
|
||||
id: z.string(),
|
||||
name: z.string().nullish(),
|
||||
user_id: z.string().nullish(),
|
||||
metadata: z.record(z.string()),
|
||||
release: z.string().nullish(),
|
||||
version: z.string().nullish(),
|
||||
project_id: z.string(),
|
||||
public: z.boolean(),
|
||||
bookmarked: z.boolean(),
|
||||
tags: z.array(z.string()),
|
||||
input: z.string().nullish(),
|
||||
output: z.string().nullish(),
|
||||
session_id: z.string().nullish(),
|
||||
});
|
||||
export type TraceRecordBaseType = z.infer<typeof traceRecordBaseSchema>;
|
||||
|
||||
export const traceRecordReadSchema = traceRecordBaseSchema.extend({
|
||||
timestamp: clickhouseStringDateSchema,
|
||||
created_at: clickhouseStringDateSchema,
|
||||
updated_at: clickhouseStringDateSchema,
|
||||
});
|
||||
export type TraceRecordReadType = z.infer<typeof traceRecordReadSchema>;
|
||||
|
||||
export const traceRecordInsertSchema = traceRecordBaseSchema.extend({
|
||||
timestamp: z.number(),
|
||||
created_at: z.number(),
|
||||
updated_at: z.number(),
|
||||
});
|
||||
export type TraceRecordInsertType = z.infer<typeof traceRecordInsertSchema>;
|
||||
|
||||
export const scoreRecordBaseSchema = z.object({
|
||||
id: z.string(),
|
||||
project_id: z.string(),
|
||||
trace_id: z.string(),
|
||||
observation_id: z.string().nullish(),
|
||||
name: z.string().nullish(),
|
||||
value: z.union([z.number(), z.string()]).nullish(),
|
||||
source: z.string(),
|
||||
comment: z.string().nullish(),
|
||||
author_user_id: z.string().nullish(),
|
||||
config_id: z.string().nullish(),
|
||||
data_type: z.enum(["NUMERIC", "CATEGORICAL", "BOOLEAN"]).nullish(),
|
||||
string_value: z.string().nullish(),
|
||||
});
|
||||
export type ScoreRecordBaseType = z.infer<typeof scoreRecordBaseSchema>;
|
||||
|
||||
export const scoreRecordReadSchema = scoreRecordBaseSchema.extend({
|
||||
created_at: clickhouseStringDateSchema,
|
||||
updated_at: clickhouseStringDateSchema,
|
||||
timestamp: clickhouseStringDateSchema,
|
||||
});
|
||||
export type ScoreRecordReadType = z.infer<typeof scoreRecordReadSchema>;
|
||||
|
||||
export const scoreRecordInsertSchema = scoreRecordBaseSchema.extend({
|
||||
created_at: z.number(),
|
||||
updated_at: z.number(),
|
||||
timestamp: z.number(),
|
||||
});
|
||||
export type ScoreRecordInsertType = z.infer<typeof scoreRecordInsertSchema>;
|
||||
|
||||
export const convertTraceReadToInsert = (
|
||||
record: TraceRecordReadType
|
||||
): TraceRecordInsertType => {
|
||||
return {
|
||||
...record,
|
||||
created_at: new Date(record.created_at).getTime(),
|
||||
updated_at: new Date(record.created_at).getTime(),
|
||||
timestamp: new Date(record.timestamp).getTime(),
|
||||
};
|
||||
};
|
||||
|
||||
export const convertObservationReadToInsert = (
|
||||
record: ObservationRecordReadType
|
||||
): ObservationRecordInsertType => {
|
||||
return {
|
||||
...record,
|
||||
created_at: new Date(record.created_at).getTime(),
|
||||
updated_at: new Date(record.created_at).getTime(),
|
||||
start_time: new Date(record.start_time).getTime(),
|
||||
end_time: record.end_time ? new Date(record.end_time).getTime() : undefined,
|
||||
completion_start_time: record.completion_start_time
|
||||
? new Date(record.completion_start_time).getTime()
|
||||
: undefined,
|
||||
};
|
||||
};
|
||||
|
||||
export const convertScoreReadToInsert = (
|
||||
record: ScoreRecordReadType
|
||||
): ScoreRecordInsertType => {
|
||||
return {
|
||||
...record,
|
||||
created_at: new Date(record.created_at).getTime(),
|
||||
updated_at: new Date(record.updated_at).getTime(),
|
||||
timestamp: new Date(record.timestamp).getTime(),
|
||||
};
|
||||
};
|
||||
@@ -26,7 +26,7 @@ const arrayOperatorReplacements = {
|
||||
export function tableColumnsToSqlFilterAndPrefix(
|
||||
filters: FilterState,
|
||||
tableColumns: ColumnDefinition[],
|
||||
table: TableNames
|
||||
table: TableNames,
|
||||
): Prisma.Sql {
|
||||
const sql = tableColumnsToSqlFilter(filters, tableColumns, table);
|
||||
if (sql === Prisma.empty) {
|
||||
@@ -42,14 +42,14 @@ export function tableColumnsToSqlFilterAndPrefix(
|
||||
export function tableColumnsToSqlFilter(
|
||||
filters: FilterState,
|
||||
tableColumns: ColumnDefinition[],
|
||||
table: TableNames
|
||||
table: TableNames,
|
||||
): Prisma.Sql {
|
||||
const internalFilters = filters.map((filter) => {
|
||||
// Get column definition to map column to internal name, e.g. "t.id"
|
||||
const col = tableColumns.find(
|
||||
(c) =>
|
||||
// TODO: Only use id instead of name
|
||||
c.name === filter.column || c.id === filter.column
|
||||
c.name === filter.column || c.id === filter.column,
|
||||
);
|
||||
if (!col) {
|
||||
logger.error("Invalid filter column", filter.column);
|
||||
@@ -71,13 +71,13 @@ export function tableColumnsToSqlFilter(
|
||||
? Prisma.raw(
|
||||
arrayOperatorReplacements[
|
||||
filter.operator as keyof typeof arrayOperatorReplacements
|
||||
]
|
||||
],
|
||||
)
|
||||
: filter.operator in operatorReplacements
|
||||
? Prisma.raw(
|
||||
operatorReplacements[
|
||||
filter.operator as keyof typeof operatorReplacements
|
||||
]
|
||||
],
|
||||
)
|
||||
: Prisma.raw(filter.operator); //checked by zod
|
||||
|
||||
@@ -97,19 +97,21 @@ export function tableColumnsToSqlFilter(
|
||||
break;
|
||||
case "stringOptions":
|
||||
valuePrisma = Prisma.sql`(${Prisma.join(
|
||||
filter.value.map((v) => Prisma.sql`${v}`)
|
||||
filter.value.map((v) => Prisma.sql`${v}`),
|
||||
)})`;
|
||||
break;
|
||||
case "arrayOptions":
|
||||
valuePrisma = Prisma.sql`ARRAY[${Prisma.join(
|
||||
filter.value.map((v) => Prisma.sql`${v}`),
|
||||
", "
|
||||
", ",
|
||||
)}] `;
|
||||
break;
|
||||
|
||||
case "boolean":
|
||||
valuePrisma = Prisma.sql`${filter.value}`;
|
||||
break;
|
||||
case "null":
|
||||
valuePrisma = Prisma.sql``;
|
||||
break;
|
||||
}
|
||||
const jsonKeyPrisma =
|
||||
filter.type === "stringObject" || filter.type === "numberObject"
|
||||
@@ -123,12 +125,12 @@ export function tableColumnsToSqlFilter(
|
||||
filter.type === "string" || filter.type === "stringObject"
|
||||
? [
|
||||
["contains", "does not contain", "ends with"].includes(
|
||||
filter.operator
|
||||
filter.operator,
|
||||
)
|
||||
? Prisma.raw("'%' || ")
|
||||
: Prisma.empty,
|
||||
["contains", "does not contain", "starts with"].includes(
|
||||
filter.operator
|
||||
filter.operator,
|
||||
)
|
||||
? Prisma.raw(" || '%'")
|
||||
: Prisma.empty,
|
||||
@@ -152,7 +154,7 @@ export function tableColumnsToSqlFilter(
|
||||
|
||||
const castValueToPostgresTypes = (
|
||||
column: ColumnDefinition,
|
||||
table: TableNames
|
||||
table: TableNames,
|
||||
) => {
|
||||
return column.name === "type" &&
|
||||
(table === "observations" ||
|
||||
@@ -168,7 +170,7 @@ const dateOperators = filterOperators["datetime"];
|
||||
export const datetimeFilterToPrismaSql = (
|
||||
safeColumn: string,
|
||||
operator: (typeof dateOperators)[number],
|
||||
value: Date
|
||||
value: Date,
|
||||
) => {
|
||||
if (!dateOperators.includes(operator)) {
|
||||
throw new Error("Invalid operator: " + operator);
|
||||
@@ -178,12 +180,12 @@ export const datetimeFilterToPrismaSql = (
|
||||
}
|
||||
|
||||
return Prisma.sql`AND ${Prisma.raw(safeColumn)} ${Prisma.raw(
|
||||
operator
|
||||
operator,
|
||||
)} ${value}::timestamp with time zone at time zone 'UTC'`;
|
||||
};
|
||||
|
||||
export const datetimeFilterToPrisma = (
|
||||
timestampFilter: z.infer<typeof timeFilter>
|
||||
timestampFilter: z.infer<typeof timeFilter>,
|
||||
) => {
|
||||
const prismaTimestampFilter =
|
||||
timestampFilter.operator === ">="
|
||||
|
||||
@@ -1,25 +1,34 @@
|
||||
export * from "./services/S3StorageService";
|
||||
export * from "./services/StorageService";
|
||||
export * from "./services/email/organizationInvitation/sendMembershipInvitationEmail";
|
||||
export * from "./services/email/batchExportSuccess/sendBatchExportSuccessEmail";
|
||||
export * from "./services/email/passwordReset/sendResetPasswordVerificationRequest";
|
||||
export * from "./services/PromptService";
|
||||
export * from "./services/traces-ui-table-service";
|
||||
export * from "./auth/apiKeys";
|
||||
export * from "./auth/customSsoProvider";
|
||||
export * from "./auth/gitHubEnterpriseProvider";
|
||||
export * from "./llm/fetchLLMCompletion";
|
||||
export * from "./llm/types";
|
||||
export * from "./utils/DatabaseReadStream";
|
||||
export * from "./utils/transforms";
|
||||
export * from "./clickhouse";
|
||||
export * from "../server/definitions";
|
||||
export * from "./clickhouse/client";
|
||||
export * from "./clickhouse/schemaUtils";
|
||||
export * from "./clickhouse/schema";
|
||||
export * from "./repositories/definitions";
|
||||
export * from "../server/ingestion/types";
|
||||
export * from "./ingestion/modelMatch";
|
||||
export * from "./ingestion/processEventBatch";
|
||||
export * from "../server/ingestion/types";
|
||||
export * from "../server/ingestion/validateAndInflateScore";
|
||||
export * from "./redis/redis";
|
||||
export * from "./redis/traceUpsert";
|
||||
export * from "./redis/CloudUsageMeteringQueue";
|
||||
export * from "./redis/getQueue";
|
||||
export * from "./redis/datasetRunItemUpsert";
|
||||
export * from "./redis/batchExport";
|
||||
export * from "./redis/legacyIngestion";
|
||||
export * from "./redis/ingestionQueue";
|
||||
export * from "./redis/experimentCreateQueue";
|
||||
export * from "./auth/types";
|
||||
export * from "./ingestion/legacy/index";
|
||||
export * from "./queues";
|
||||
@@ -29,3 +38,5 @@ export * from "./filterToPrisma";
|
||||
export * from "./instrumentation";
|
||||
export * from "./logger";
|
||||
export * from "./queries";
|
||||
export * from "./repositories";
|
||||
export * from "./redis/evalExecutionQueue";
|
||||
|
||||
@@ -20,6 +20,9 @@ import { jsonSchema } from "../../../utils/zod";
|
||||
import { prisma } from "../../../db";
|
||||
import { LegacyIngestionAccessScope } from ".";
|
||||
import { logger } from "../../logger";
|
||||
import { env } from "../../../env";
|
||||
import { upsertTrace } from "../../repositories";
|
||||
import { convertDateToClickhouseDateTime } from "../../clickhouse/client";
|
||||
|
||||
export interface EventProcessor {
|
||||
auth(apiScope: LegacyIngestionAccessScope): void;
|
||||
@@ -131,19 +134,6 @@ export class ObservationProcessor implements EventProcessor {
|
||||
})
|
||||
: undefined;
|
||||
|
||||
const traceId =
|
||||
!this.event.body.traceId && !existingObservation
|
||||
? // Create trace if no traceid
|
||||
(
|
||||
await prisma.trace.create({
|
||||
data: {
|
||||
projectId: apiScope.projectId,
|
||||
name: this.event.body.name,
|
||||
},
|
||||
})
|
||||
).id
|
||||
: this.event.body.traceId;
|
||||
|
||||
// Token counts
|
||||
const [newInputCount, newOutputCount] =
|
||||
"usage" in this.event.body
|
||||
@@ -220,13 +210,56 @@ export class ObservationProcessor implements EventProcessor {
|
||||
logger.warn("Prompt not found for observation", this.event.body);
|
||||
}
|
||||
|
||||
const observationId = this.event.body.id ?? v4();
|
||||
const observationId =
|
||||
this.event.body.id ??
|
||||
(() => {
|
||||
const newId = v4();
|
||||
logger.info(
|
||||
`observation.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
|
||||
);
|
||||
return newId;
|
||||
})();
|
||||
|
||||
let traceId = this.event.body?.traceId;
|
||||
if (!this.event.body.traceId && !existingObservation) {
|
||||
// Create trace if no traceId
|
||||
traceId = observationId;
|
||||
|
||||
// Insert trace into postgres
|
||||
await prisma.trace.upsert({
|
||||
where: {
|
||||
id: observationId,
|
||||
},
|
||||
create: {
|
||||
projectId: apiScope.projectId,
|
||||
name: this.event.body.name,
|
||||
id: observationId,
|
||||
timestamp: this.event.body.startTime || new Date(),
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
if (env.CLICKHOUSE_URL) {
|
||||
// Insert trace into clickhouse if enabled
|
||||
await upsertTrace({
|
||||
id: observationId,
|
||||
project_id: apiScope.projectId,
|
||||
timestamp: convertDateToClickhouseDateTime(
|
||||
this.event.body.startTime
|
||||
? new Date(this.event.body.startTime)
|
||||
: new Date(),
|
||||
),
|
||||
created_at: convertDateToClickhouseDateTime(new Date()),
|
||||
updated_at: convertDateToClickhouseDateTime(new Date()),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
id: observationId,
|
||||
create: {
|
||||
id: observationId,
|
||||
traceId: traceId,
|
||||
traceId,
|
||||
type: type,
|
||||
name: this.event.body.name,
|
||||
startTime: this.event.body.startTime
|
||||
@@ -359,7 +392,7 @@ export class ObservationProcessor implements EventProcessor {
|
||||
text: body.input,
|
||||
});
|
||||
} else {
|
||||
logger.info(
|
||||
logger.debug(
|
||||
`No input provided, trying to calculate for id: ${existingObservation?.id}`,
|
||||
);
|
||||
const observationInput = await prisma.observation.findFirst({
|
||||
@@ -385,7 +418,7 @@ export class ObservationProcessor implements EventProcessor {
|
||||
text: body.output,
|
||||
});
|
||||
} else {
|
||||
logger.info(
|
||||
logger.debug(
|
||||
`No output provided, trying to calculate for id: ${existingObservation?.id}`,
|
||||
);
|
||||
const observationOutput = await prisma.observation.findFirst({
|
||||
@@ -551,7 +584,15 @@ export class TraceProcessor implements EventProcessor {
|
||||
|
||||
this.auth(apiScope);
|
||||
|
||||
const internalId = body.id ?? v4();
|
||||
const internalId =
|
||||
body.id ??
|
||||
(() => {
|
||||
const newId = v4();
|
||||
logger.info(
|
||||
`trace.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
|
||||
);
|
||||
return newId;
|
||||
})();
|
||||
|
||||
logger.debug(
|
||||
`Trying to create trace, project ${apiScope.projectId}, id: ${internalId}`,
|
||||
@@ -584,19 +625,32 @@ export class TraceProcessor implements EventProcessor {
|
||||
: undefined;
|
||||
|
||||
if (body.sessionId) {
|
||||
await prisma.traceSession.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
try {
|
||||
await prisma.traceSession.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
id: body.sessionId,
|
||||
projectId: apiScope.projectId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: body.sessionId,
|
||||
projectId: apiScope.projectId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: body.sessionId,
|
||||
projectId: apiScope.projectId,
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
update: {},
|
||||
});
|
||||
} catch (e) {
|
||||
if (
|
||||
e instanceof Prisma.PrismaClientKnownRequestError &&
|
||||
e.code === "P2002"
|
||||
) {
|
||||
logger.warn(
|
||||
`Failed to upsert session. Session ${body.sessionId} in project ${apiScope.projectId} already exists`,
|
||||
);
|
||||
} else {
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Do not use nested upserts or multiple where conditions as this should be a single native database upsert
|
||||
@@ -663,7 +717,15 @@ export class ScoreProcessor implements EventProcessor {
|
||||
|
||||
this.auth(apiScope);
|
||||
|
||||
const id = body.id ?? v4();
|
||||
const id =
|
||||
body.id ??
|
||||
(() => {
|
||||
const newId = v4();
|
||||
logger.info(
|
||||
`score.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
|
||||
);
|
||||
return newId;
|
||||
})();
|
||||
|
||||
const existingScore = await prisma.score.findFirst({
|
||||
where: {
|
||||
|
||||
@@ -1,15 +1,9 @@
|
||||
import { env } from "node:process";
|
||||
import z from "zod";
|
||||
import { ForbiddenError, UnauthorizedError } from "../../../errors";
|
||||
import { eventTypes, ingestionApiSchema, ingestionEvent } from "../types";
|
||||
import { eventTypes, ingestionApiSchema, IngestionEventType } from "../types";
|
||||
import { getProcessorForEvent } from "./EventProcessor";
|
||||
import { TraceUpsertEventType } from "../../queues";
|
||||
import {
|
||||
convertTraceUpsertEventsToRedisEvents,
|
||||
getTraceUpsertQueue,
|
||||
} from "../../redis/traceUpsert";
|
||||
import { ApiAccessScope } from "../../auth/types";
|
||||
import { redis } from "../../redis/redis";
|
||||
import { backOff } from "exponential-backoff";
|
||||
import { Model } from "../../..";
|
||||
import { logger } from "../../logger";
|
||||
@@ -77,7 +71,7 @@ export const handleBatch = async (
|
||||
// Decide how to handle the error: rethrow, continue, or push an error object to results
|
||||
// For example, push an error object:
|
||||
errors.push({
|
||||
error: error,
|
||||
error,
|
||||
id: singleEvent.id,
|
||||
type: singleEvent.type,
|
||||
});
|
||||
@@ -102,7 +96,7 @@ async function retry<T>(request: () => Promise<T>): Promise<T> {
|
||||
}
|
||||
|
||||
const handleSingleEvent = async (
|
||||
event: z.infer<typeof ingestionEvent>,
|
||||
event: IngestionEventType,
|
||||
apiScope: LegacyIngestionAccessScope,
|
||||
calculateTokenDelegate: (p: {
|
||||
model: Model;
|
||||
@@ -122,11 +116,11 @@ const handleSingleEvent = async (
|
||||
restEvent = rest;
|
||||
}
|
||||
|
||||
logger.info(
|
||||
logger.debug(
|
||||
`handling single event ${event.id} of type ${event.type}: ${JSON.stringify({ body: restEvent })}`,
|
||||
);
|
||||
|
||||
const cleanedEvent = ingestionEvent.parse(cleanEvent(event));
|
||||
const cleanedEvent = cleanEvent(event) as IngestionEventType;
|
||||
|
||||
// Deny access to non-score events if the access level is not "all"
|
||||
// This is an additional safeguard to auth checks in EventProcessor
|
||||
@@ -163,42 +157,5 @@ export function cleanEvent(obj: unknown): unknown {
|
||||
}
|
||||
}
|
||||
|
||||
export const isNotNullOrUndefined = <T>(
|
||||
val?: T | null,
|
||||
): val is Exclude<T, null | undefined> => !isUndefinedOrNull(val);
|
||||
|
||||
export const isUndefinedOrNull = <T>(val?: T | null): val is undefined | null =>
|
||||
val === undefined || val === null;
|
||||
|
||||
export const addTracesToTraceUpsertQueue = async (
|
||||
batchResults: BatchResult[],
|
||||
projectId: string,
|
||||
): Promise<void> => {
|
||||
const traceEvents: TraceUpsertEventType[] = batchResults
|
||||
.filter((result) => result.type === eventTypes.TRACE_CREATE) // we only have create, no update.
|
||||
.map((result) =>
|
||||
result.result &&
|
||||
typeof result.result === "object" &&
|
||||
"id" in result.result
|
||||
? // ingestion API only gets traces for one projectId
|
||||
{ traceId: result.result.id as string, projectId }
|
||||
: null,
|
||||
)
|
||||
.filter(isNotNullOrUndefined);
|
||||
|
||||
try {
|
||||
if (env.NEXT_PUBLIC_LANGFUSE_CLOUD_REGION && redis) {
|
||||
logger.info(`Sending ${traceEvents.length} events to worker via Redis`);
|
||||
|
||||
const queue = getTraceUpsertQueue();
|
||||
if (!queue) {
|
||||
logger.error("TraceUpsertQueue not initialized");
|
||||
return;
|
||||
}
|
||||
|
||||
await queue.addBulk(convertTraceUpsertEventsToRedisEvents(traceEvents));
|
||||
}
|
||||
} catch (error) {
|
||||
logger.error("Error sending events to worker", error);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -0,0 +1,396 @@
|
||||
import { randomUUID } from "crypto";
|
||||
import { z } from "zod";
|
||||
|
||||
import { type Model } from "../../db";
|
||||
import { env } from "../../env";
|
||||
import {
|
||||
InvalidRequestError,
|
||||
LangfuseNotFoundError,
|
||||
UnauthorizedError,
|
||||
} from "../../errors";
|
||||
import { AuthHeaderValidVerificationResult } from "../auth/types";
|
||||
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
|
||||
import {
|
||||
getCurrentSpan,
|
||||
instrumentAsync,
|
||||
instrumentSync,
|
||||
recordIncrement,
|
||||
traceException,
|
||||
} from "../instrumentation";
|
||||
import { logger } from "../logger";
|
||||
import { LegacyIngestionEventType, QueueJobs } from "../queues";
|
||||
import { IngestionQueue } from "../redis/ingestionQueue";
|
||||
import { LegacyIngestionQueue } from "../redis/legacyIngestion";
|
||||
import { redis } from "../redis/redis";
|
||||
import { handleBatch } from "./legacy";
|
||||
import {
|
||||
StorageService,
|
||||
StorageServiceFactory,
|
||||
} from "../services/StorageService";
|
||||
import { getProcessorForEvent } from "./legacy/EventProcessor";
|
||||
import { eventTypes, ingestionEvent, IngestionEventType } from "./types";
|
||||
|
||||
export type TokenCountDelegate = (p: {
|
||||
model: Model;
|
||||
text: unknown;
|
||||
}) => number | undefined;
|
||||
|
||||
let s3StorageServiceClient: StorageService;
|
||||
|
||||
const getS3StorageServiceClient = (bucketName: string): StorageService => {
|
||||
if (!s3StorageServiceClient) {
|
||||
s3StorageServiceClient = StorageServiceFactory.getInstance({
|
||||
bucketName,
|
||||
accessKeyId: env.LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID,
|
||||
secretAccessKey: env.LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY,
|
||||
endpoint: env.LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT,
|
||||
region: env.LANGFUSE_S3_EVENT_UPLOAD_REGION,
|
||||
forcePathStyle: env.LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE === "true",
|
||||
});
|
||||
}
|
||||
return s3StorageServiceClient;
|
||||
};
|
||||
|
||||
export const processEventBatch = async (
|
||||
input: unknown[],
|
||||
authCheck: AuthHeaderValidVerificationResult,
|
||||
tokenCountDelegate: TokenCountDelegate,
|
||||
): Promise<{
|
||||
successes: { id: string; status: number }[];
|
||||
errors: {
|
||||
id: string;
|
||||
status: number;
|
||||
message?: string;
|
||||
error?: string;
|
||||
}[];
|
||||
}> => {
|
||||
// add context of api call to the span
|
||||
const currentSpan = getCurrentSpan();
|
||||
recordIncrement("langfuse.ingestion.event", input.length);
|
||||
currentSpan?.setAttribute("event_count", input.length);
|
||||
|
||||
/**************
|
||||
* VALIDATION *
|
||||
**************/
|
||||
const validationErrors: { id: string; error: unknown }[] = [];
|
||||
const authenticationErrors: { id: string; error: unknown }[] = [];
|
||||
|
||||
const batch: z.infer<typeof ingestionEvent>[] = input
|
||||
.flatMap((event) => {
|
||||
const parsed = instrumentSync(
|
||||
{ name: "ingestion-zod-parse-individual-event" },
|
||||
(span) => {
|
||||
const parsedBody = ingestionEvent.safeParse(event);
|
||||
if (parsedBody.data?.id !== undefined) {
|
||||
span.setAttribute("object.id", parsedBody.data.id);
|
||||
}
|
||||
return parsedBody;
|
||||
},
|
||||
);
|
||||
if (!parsed.success) {
|
||||
validationErrors.push({
|
||||
id:
|
||||
typeof event === "object" && event && "id" in event
|
||||
? typeof event.id === "string"
|
||||
? event.id
|
||||
: "unknown"
|
||||
: "unknown",
|
||||
error: new InvalidRequestError(parsed.error.message),
|
||||
});
|
||||
return [];
|
||||
}
|
||||
if (!isAuthorized(parsed.data, authCheck, tokenCountDelegate)) {
|
||||
authenticationErrors.push({
|
||||
id: parsed.data.id,
|
||||
error: new UnauthorizedError("Access Scope Denied"),
|
||||
});
|
||||
return [];
|
||||
}
|
||||
return [parsed.data];
|
||||
})
|
||||
.flatMap((event) => {
|
||||
if (event.type === eventTypes.SDK_LOG) {
|
||||
// Log SDK_LOG events, but remove them from further processing
|
||||
logger.info("SDK Log Event", { event });
|
||||
return [];
|
||||
}
|
||||
return [event];
|
||||
});
|
||||
|
||||
const sortedBatch = sortBatch(batch);
|
||||
|
||||
// We group events by eventBodyId which allows us to store and process them
|
||||
// as one which reduces infra interactions per event. Only used in the S3 case.
|
||||
const sortedBatchByEventBodyId = sortedBatch.reduce(
|
||||
(
|
||||
acc: Record<
|
||||
string,
|
||||
{
|
||||
data: IngestionEventType[];
|
||||
key: string;
|
||||
eventBodyId: string;
|
||||
type: (typeof eventTypes)[keyof typeof eventTypes];
|
||||
}
|
||||
>,
|
||||
event,
|
||||
) => {
|
||||
if (!event.body?.id) {
|
||||
return acc;
|
||||
}
|
||||
const key = `${getClickhouseEntityType(event.type)}-${event.body.id}`;
|
||||
if (!acc[key]) {
|
||||
acc[key] = {
|
||||
data: [],
|
||||
key: event.id,
|
||||
type: event.type,
|
||||
eventBodyId: event.body.id,
|
||||
};
|
||||
}
|
||||
acc[key].data.push(event);
|
||||
return acc;
|
||||
},
|
||||
{},
|
||||
);
|
||||
|
||||
/********************
|
||||
* ASYNC PROCESSING *
|
||||
********************/
|
||||
let s3UploadErrored = false;
|
||||
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
|
||||
await instrumentAsync({ name: "s3-upload-events" }, async () => {
|
||||
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
|
||||
throw new Error("S3 event store is enabled but no bucket is set");
|
||||
}
|
||||
const s3Client = getS3StorageServiceClient(
|
||||
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
|
||||
);
|
||||
// S3 Event Upload is currently blocking, but non-failing.
|
||||
// If a promise rejects, we log it below, but do not throw an error.
|
||||
// In this case, we upload the full batch into the Redis queue.
|
||||
const results = await Promise.allSettled(
|
||||
Object.keys(sortedBatchByEventBodyId).map(async (id) => {
|
||||
// We upload the event in an array to the S3 bucket grouped by the eventBodyId.
|
||||
// That way we batch updates from the same invocation into a single file and reduce
|
||||
// write operations on S3.
|
||||
const { data, key, type, eventBodyId } = sortedBatchByEventBodyId[id];
|
||||
return s3Client.uploadJson(
|
||||
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${authCheck.scope.projectId}/${getClickhouseEntityType(type)}/${eventBodyId}/${key}.json`,
|
||||
data,
|
||||
);
|
||||
}),
|
||||
);
|
||||
results.forEach((result) => {
|
||||
if (result.status === "rejected") {
|
||||
s3UploadErrored = true;
|
||||
logger.error("Failed to upload event to S3", {
|
||||
error: result.reason,
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// Send each event individually to IngestionQueue for new processing
|
||||
if (
|
||||
env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" &&
|
||||
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" &&
|
||||
env.LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING === "true" &&
|
||||
redis &&
|
||||
!s3UploadErrored
|
||||
) {
|
||||
const queue = IngestionQueue.getInstance();
|
||||
const results = await Promise.allSettled(
|
||||
Object.keys(sortedBatchByEventBodyId).map(async (id) =>
|
||||
queue
|
||||
? queue.add(
|
||||
QueueJobs.IngestionJob,
|
||||
{
|
||||
id: randomUUID(),
|
||||
timestamp: new Date(),
|
||||
name: QueueJobs.IngestionJob as const,
|
||||
payload: {
|
||||
data: {
|
||||
type: sortedBatchByEventBodyId[id].type,
|
||||
eventBodyId: sortedBatchByEventBodyId[id].eventBodyId,
|
||||
},
|
||||
authCheck,
|
||||
},
|
||||
},
|
||||
{
|
||||
delay: env.LANGFUSE_INGESTION_QUEUE_DELAY_MS,
|
||||
},
|
||||
)
|
||||
: Promise.reject("Failed to instantiate queue"),
|
||||
),
|
||||
);
|
||||
results.forEach((result) => {
|
||||
if (result.status === "rejected") {
|
||||
logger.error("Failed to add event to IngestionQueue", {
|
||||
error: result.reason,
|
||||
});
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// As part of the legacy processing we sent the entire batch to the worker.
|
||||
if (env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" && redis) {
|
||||
const queue = LegacyIngestionQueue.getInstance();
|
||||
|
||||
if (queue) {
|
||||
let addToQueueFailed = false;
|
||||
|
||||
const queuePayload: LegacyIngestionEventType =
|
||||
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" && !s3UploadErrored
|
||||
? {
|
||||
data: Object.keys(sortedBatchByEventBodyId).map((id) => {
|
||||
const { key, type, eventBodyId } = sortedBatchByEventBodyId[id];
|
||||
return {
|
||||
type,
|
||||
eventBodyId,
|
||||
eventId: key,
|
||||
};
|
||||
}),
|
||||
authCheck,
|
||||
useS3EventStore: true,
|
||||
}
|
||||
: { data: sortedBatch, authCheck, useS3EventStore: false };
|
||||
|
||||
try {
|
||||
await queue.add(QueueJobs.LegacyIngestionJob, {
|
||||
payload: queuePayload,
|
||||
id: randomUUID(),
|
||||
timestamp: new Date(),
|
||||
name: QueueJobs.LegacyIngestionJob as const,
|
||||
});
|
||||
} catch (e: unknown) {
|
||||
logger.warn(
|
||||
"Failed to add batch to queue, falling back to sync processing",
|
||||
e,
|
||||
);
|
||||
addToQueueFailed = true;
|
||||
}
|
||||
|
||||
if (!addToQueueFailed) {
|
||||
return aggregateBatchResult(
|
||||
// we are not sending additional server errors to the client in case of early return
|
||||
[...validationErrors, ...authenticationErrors],
|
||||
sortedBatch.map((event) => ({ id: event.id, result: event })),
|
||||
);
|
||||
}
|
||||
} else {
|
||||
logger.error(
|
||||
"Ingestion queue not initialized, falling back to sync processing",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/*******************
|
||||
* SYNC PROCESSING *
|
||||
*******************/
|
||||
const result = await handleBatch(sortedBatch, authCheck, tokenCountDelegate);
|
||||
|
||||
// in case we did not return early, we return the result here
|
||||
return aggregateBatchResult(
|
||||
[...validationErrors, ...authenticationErrors, ...result.errors],
|
||||
result.results,
|
||||
);
|
||||
};
|
||||
|
||||
const isAuthorized = (
|
||||
event: IngestionEventType,
|
||||
authScope: AuthHeaderValidVerificationResult,
|
||||
tokenCountDelegate: TokenCountDelegate,
|
||||
): boolean => {
|
||||
try {
|
||||
getProcessorForEvent(event, tokenCountDelegate).auth(authScope.scope);
|
||||
return true;
|
||||
} catch (error) {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Sorts a batch of ingestion events. Orders by: updating events last, sorted by timestamp asc.
|
||||
*/
|
||||
const sortBatch = (batch: Array<z.infer<typeof ingestionEvent>>) => {
|
||||
const updateEvents: (typeof eventTypes)[keyof typeof eventTypes][] = [
|
||||
eventTypes.GENERATION_UPDATE,
|
||||
eventTypes.SPAN_UPDATE,
|
||||
eventTypes.OBSERVATION_UPDATE, // legacy event type
|
||||
];
|
||||
const updates = batch
|
||||
.filter((event) => updateEvents.includes(event.type))
|
||||
.sort((a, b) => {
|
||||
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
|
||||
});
|
||||
const others = batch
|
||||
.filter((event) => !updateEvents.includes(event.type))
|
||||
.sort((a, b) => {
|
||||
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
|
||||
});
|
||||
|
||||
// Return the array with non-update events first, followed by update events
|
||||
return [...others, ...updates];
|
||||
};
|
||||
|
||||
export const aggregateBatchResult = (
|
||||
errors: Array<{ id: string; error: unknown }>,
|
||||
results: Array<{ id: string; result: unknown }>,
|
||||
) => {
|
||||
const returnedErrors: {
|
||||
id: string;
|
||||
status: number;
|
||||
message?: string;
|
||||
error?: string;
|
||||
}[] = [];
|
||||
|
||||
const successes: {
|
||||
id: string;
|
||||
status: number;
|
||||
}[] = [];
|
||||
|
||||
errors.forEach((error) => {
|
||||
if (error.error instanceof InvalidRequestError) {
|
||||
returnedErrors.push({
|
||||
id: error.id,
|
||||
status: 400,
|
||||
message: "Invalid request data",
|
||||
error: error.error.message,
|
||||
});
|
||||
} else if (error.error instanceof UnauthorizedError) {
|
||||
returnedErrors.push({
|
||||
id: error.id,
|
||||
status: 401,
|
||||
message: "Authentication error",
|
||||
error: error.error.message,
|
||||
});
|
||||
} else if (error.error instanceof LangfuseNotFoundError) {
|
||||
returnedErrors.push({
|
||||
id: error.id,
|
||||
status: 404,
|
||||
message: "Resource not found",
|
||||
error: error.error.message,
|
||||
});
|
||||
} else {
|
||||
returnedErrors.push({
|
||||
id: error.id,
|
||||
status: 500,
|
||||
error: "Internal Server Error",
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
if (returnedErrors.length > 0) {
|
||||
traceException(errors);
|
||||
logger.error("Error processing events", returnedErrors);
|
||||
}
|
||||
|
||||
results.forEach((result) => {
|
||||
successes.push({
|
||||
id: result.id,
|
||||
status: 201,
|
||||
});
|
||||
});
|
||||
|
||||
return { successes, errors: returnedErrors };
|
||||
};
|
||||
@@ -3,7 +3,7 @@ import { z } from "zod";
|
||||
|
||||
import { NonEmptyString, jsonSchema } from "../../utils/zod";
|
||||
import { ModelUsageUnit } from "../../constants";
|
||||
import { ObservationLevel } from "@prisma/client";
|
||||
import { ObservationLevel, ScoreSource } from "@prisma/client";
|
||||
|
||||
export const Usage = z.object({
|
||||
input: z.number().int().nullish(),
|
||||
@@ -163,6 +163,7 @@ const BaseScoreBody = z.object({
|
||||
traceId: z.string(),
|
||||
observationId: z.string().nullish(),
|
||||
comment: z.string().nullish(),
|
||||
source: z.nativeEnum(ScoreSource).default(ScoreSource.API),
|
||||
});
|
||||
|
||||
/**
|
||||
|
||||
@@ -76,7 +76,7 @@ function inflateScoreBody(
|
||||
const { body, projectId, scoreId, config } = params;
|
||||
|
||||
const relevantDataType = config?.dataType ?? body.dataType;
|
||||
const scoreProps = { ...body, id: scoreId, projectId, source: "API" };
|
||||
const scoreProps = { source: "API", ...body, id: scoreId, projectId };
|
||||
|
||||
if (typeof body.value === "number") {
|
||||
if (relevantDataType && relevantDataType === ScoreDataType.BOOLEAN) {
|
||||
|
||||
@@ -1,5 +1,11 @@
|
||||
import {
|
||||
CloudWatchClient,
|
||||
PutMetricDataCommand,
|
||||
} from "@aws-sdk/client-cloudwatch";
|
||||
import * as opentelemetry from "@opentelemetry/api";
|
||||
import * as dd from "dd-trace";
|
||||
import { env } from "../../env";
|
||||
import { logger } from "../logger";
|
||||
|
||||
// type CallbackFn<T> = () => T;
|
||||
|
||||
@@ -32,7 +38,7 @@ export async function instrumentAsync<T>(
|
||||
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
|
||||
ctx.name,
|
||||
{
|
||||
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
|
||||
root: !ctx.traceContext && ctx.rootSpan,
|
||||
kind: ctx.spanKind,
|
||||
},
|
||||
activeContext,
|
||||
@@ -66,7 +72,7 @@ export function instrumentSync<T>(
|
||||
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
|
||||
ctx.name,
|
||||
{
|
||||
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
|
||||
root: !ctx.traceContext && ctx.rootSpan,
|
||||
kind: ctx.spanKind,
|
||||
},
|
||||
activeContext,
|
||||
@@ -153,6 +159,36 @@ export const addUserToSpan = (
|
||||
|
||||
export const getTracer = (name: string) => opentelemetry.trace.getTracer(name);
|
||||
|
||||
const cloudWatchClient = new CloudWatchClient();
|
||||
const cloudWatchLastSubmitted: Record<string, number> = {};
|
||||
const sendCloudWatchMetric = (key: string, value: number | undefined) => {
|
||||
const currentTime = Date.now();
|
||||
const interval = 30 * 1000;
|
||||
|
||||
// Check if the function has been executed in the last 30s for this key
|
||||
if (
|
||||
!cloudWatchLastSubmitted[key] ||
|
||||
currentTime - cloudWatchLastSubmitted[key] >= interval
|
||||
) {
|
||||
cloudWatchLastSubmitted[key] = currentTime;
|
||||
cloudWatchClient
|
||||
.send(
|
||||
new PutMetricDataCommand({
|
||||
Namespace: "Langfuse",
|
||||
MetricData: [
|
||||
{
|
||||
MetricName: key,
|
||||
Value: value ?? 0,
|
||||
},
|
||||
],
|
||||
}),
|
||||
)
|
||||
.catch((error) => {
|
||||
logger.warn("Failed to send metric to CloudWatch", error);
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
export const recordGauge = (
|
||||
stat: string,
|
||||
value?: number | undefined,
|
||||
@@ -162,6 +198,9 @@ export const recordGauge = (
|
||||
}
|
||||
| undefined,
|
||||
) => {
|
||||
if (env.ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING === "true") {
|
||||
sendCloudWatchMetric(stat, value);
|
||||
}
|
||||
dd.dogstatsd.gauge(stat, value, tags);
|
||||
};
|
||||
|
||||
@@ -170,6 +209,9 @@ export const recordIncrement = (
|
||||
value?: number | undefined,
|
||||
tags?: { [tag: string]: string | number } | undefined,
|
||||
) => {
|
||||
if (env.ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING === "true") {
|
||||
sendCloudWatchMetric(stat, value);
|
||||
}
|
||||
dd.dogstatsd.increment(stat, value, tags);
|
||||
};
|
||||
|
||||
@@ -178,6 +220,9 @@ export const recordHistogram = (
|
||||
value?: number | undefined,
|
||||
tags?: { [tag: string]: string | number } | undefined,
|
||||
) => {
|
||||
if (env.ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING === "true") {
|
||||
sendCloudWatchMetric(stat, value);
|
||||
}
|
||||
dd.dogstatsd.histogram(stat, value, tags);
|
||||
};
|
||||
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import type { ZodSchema } from "zod";
|
||||
|
||||
import { CallbackHandler } from "langfuse-langchain";
|
||||
|
||||
import { ChatAnthropic } from "@langchain/anthropic";
|
||||
import { ChatBedrockConverse } from "@langchain/aws";
|
||||
import {
|
||||
@@ -17,11 +19,27 @@ import {
|
||||
BedrockConfigSchema,
|
||||
BedrockCredentialSchema,
|
||||
} from "../../interfaces/customLLMProviderConfigSchemas";
|
||||
|
||||
import { AuthHeaderValidVerificationResult } from "../auth/types";
|
||||
import {
|
||||
processEventBatch,
|
||||
type TokenCountDelegate,
|
||||
} from "../ingestion/processEventBatch";
|
||||
import { logger } from "../logger";
|
||||
import { ChatMessage, ChatMessageRole, LLMAdapter, ModelParams } from "./types";
|
||||
|
||||
import type { BaseCallbackHandler } from "@langchain/core/callbacks/base";
|
||||
|
||||
type ProcessTracedEvents = () => Promise<void>;
|
||||
|
||||
export type TraceParams = {
|
||||
traceName: string;
|
||||
traceId: string;
|
||||
projectId: string;
|
||||
tags: string[];
|
||||
tokenCountDelegate: TokenCountDelegate;
|
||||
authCheck: AuthHeaderValidVerificationResult;
|
||||
};
|
||||
|
||||
type LLMCompletionParams = {
|
||||
messages: ChatMessage[];
|
||||
modelParams: ModelParams;
|
||||
@@ -31,6 +49,7 @@ type LLMCompletionParams = {
|
||||
apiKey: string;
|
||||
maxRetries?: number;
|
||||
config?: Record<string, string> | null;
|
||||
traceParams?: TraceParams;
|
||||
};
|
||||
|
||||
type FetchLLMCompletionParams = LLMCompletionParams & {
|
||||
@@ -40,25 +59,34 @@ type FetchLLMCompletionParams = LLMCompletionParams & {
|
||||
export async function fetchLLMCompletion(
|
||||
params: LLMCompletionParams & {
|
||||
streaming: true;
|
||||
}
|
||||
): Promise<IterableReadableStream<Uint8Array>>;
|
||||
},
|
||||
): Promise<{
|
||||
completion: IterableReadableStream<Uint8Array>;
|
||||
processTracedEvents: ProcessTracedEvents;
|
||||
}>;
|
||||
|
||||
export async function fetchLLMCompletion(
|
||||
params: LLMCompletionParams & {
|
||||
streaming: false;
|
||||
}
|
||||
): Promise<string>;
|
||||
},
|
||||
): Promise<{ completion: string; processTracedEvents: ProcessTracedEvents }>;
|
||||
|
||||
export async function fetchLLMCompletion(
|
||||
params: LLMCompletionParams & {
|
||||
streaming: false;
|
||||
structuredOutputSchema: ZodSchema;
|
||||
}
|
||||
): Promise<unknown>;
|
||||
},
|
||||
): Promise<{
|
||||
completion: unknown;
|
||||
processTracedEvents: ProcessTracedEvents;
|
||||
}>;
|
||||
|
||||
export async function fetchLLMCompletion(
|
||||
params: FetchLLMCompletionParams
|
||||
): Promise<string | IterableReadableStream<Uint8Array> | unknown> {
|
||||
params: FetchLLMCompletionParams,
|
||||
): Promise<{
|
||||
completion: string | IterableReadableStream<Uint8Array> | unknown;
|
||||
processTracedEvents: ProcessTracedEvents;
|
||||
}> {
|
||||
// the apiKey must never be printed to the console
|
||||
const {
|
||||
messages,
|
||||
@@ -69,8 +97,39 @@ export async function fetchLLMCompletion(
|
||||
baseURL,
|
||||
maxRetries,
|
||||
config,
|
||||
traceParams,
|
||||
} = params;
|
||||
|
||||
let finalCallbacks: BaseCallbackHandler[] | undefined = callbacks ?? [];
|
||||
let processTracedEvents: ProcessTracedEvents = () => Promise.resolve();
|
||||
|
||||
if (traceParams) {
|
||||
const handler = new CallbackHandler({
|
||||
_projectId: traceParams.projectId,
|
||||
_isLocalEventExportEnabled: true,
|
||||
tags: traceParams.tags,
|
||||
});
|
||||
|
||||
finalCallbacks.push(handler);
|
||||
|
||||
processTracedEvents = async () => {
|
||||
try {
|
||||
const events = await handler.langfuse._exportLocalEvents(
|
||||
traceParams.projectId,
|
||||
);
|
||||
await processEventBatch(
|
||||
JSON.parse(JSON.stringify(events)), // stringify to emulate network event batch from network call
|
||||
traceParams.authCheck,
|
||||
traceParams.tokenCountDelegate,
|
||||
);
|
||||
} catch (e) {
|
||||
logger.error("Failed to process traced events", { error: e });
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
finalCallbacks = finalCallbacks.length > 0 ? finalCallbacks : undefined;
|
||||
|
||||
const finalMessages = messages.map((message) => {
|
||||
if (message.role === ChatMessageRole.User)
|
||||
return new HumanMessage(message.content);
|
||||
@@ -89,7 +148,7 @@ export async function fetchLLMCompletion(
|
||||
temperature: modelParams.temperature,
|
||||
maxTokens: modelParams.max_tokens,
|
||||
topP: modelParams.top_p,
|
||||
callbacks,
|
||||
callbacks: finalCallbacks,
|
||||
clientOptions: { maxRetries },
|
||||
});
|
||||
} else if (modelParams.adapter === LLMAdapter.OpenAI) {
|
||||
@@ -99,7 +158,8 @@ export async function fetchLLMCompletion(
|
||||
temperature: modelParams.temperature,
|
||||
maxTokens: modelParams.max_tokens,
|
||||
topP: modelParams.top_p,
|
||||
callbacks,
|
||||
streamUsage: false, // https://github.com/langchain-ai/langchainjs/issues/6533
|
||||
callbacks: finalCallbacks,
|
||||
maxRetries,
|
||||
configuration: {
|
||||
baseURL,
|
||||
@@ -114,7 +174,7 @@ export async function fetchLLMCompletion(
|
||||
temperature: modelParams.temperature,
|
||||
maxTokens: modelParams.max_tokens,
|
||||
topP: modelParams.top_p,
|
||||
callbacks,
|
||||
callbacks: finalCallbacks,
|
||||
maxRetries,
|
||||
});
|
||||
} else if (modelParams.adapter === LLMAdapter.Bedrock) {
|
||||
@@ -128,7 +188,7 @@ export async function fetchLLMCompletion(
|
||||
temperature: modelParams.temperature,
|
||||
maxTokens: modelParams.max_tokens,
|
||||
topP: modelParams.top_p,
|
||||
callbacks,
|
||||
callbacks: finalCallbacks,
|
||||
maxRetries,
|
||||
});
|
||||
} else {
|
||||
@@ -137,10 +197,19 @@ export async function fetchLLMCompletion(
|
||||
throw new Error("This model provider is not supported.");
|
||||
}
|
||||
|
||||
const runConfig = {
|
||||
callbacks: finalCallbacks,
|
||||
runId: traceParams?.traceId,
|
||||
runName: traceParams?.traceName,
|
||||
};
|
||||
|
||||
if (params.structuredOutputSchema) {
|
||||
return await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
|
||||
.withStructuredOutput(params.structuredOutputSchema)
|
||||
.invoke(finalMessages);
|
||||
return {
|
||||
completion: await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
|
||||
.withStructuredOutput(params.structuredOutputSchema)
|
||||
.invoke(finalMessages, runConfig),
|
||||
processTracedEvents,
|
||||
};
|
||||
}
|
||||
|
||||
/*
|
||||
@@ -156,27 +225,41 @@ export async function fetchLLMCompletion(
|
||||
Reference: https://platform.openai.com/docs/guides/reasoning/beta-limitations
|
||||
*/
|
||||
if (modelParams.model.startsWith("o1-")) {
|
||||
return await new ChatOpenAI({
|
||||
openAIApiKey: apiKey,
|
||||
modelName: modelParams.model,
|
||||
temperature: 1,
|
||||
maxTokens: undefined,
|
||||
topP: undefined,
|
||||
callbacks,
|
||||
maxRetries,
|
||||
configuration: {
|
||||
baseURL,
|
||||
},
|
||||
})
|
||||
.pipe(new StringOutputParser())
|
||||
.invoke(
|
||||
finalMessages.filter((message) => message._getType() !== "system")
|
||||
);
|
||||
return {
|
||||
completion: await new ChatOpenAI({
|
||||
openAIApiKey: apiKey,
|
||||
modelName: modelParams.model,
|
||||
temperature: 1,
|
||||
maxTokens: undefined,
|
||||
topP: undefined,
|
||||
callbacks,
|
||||
maxRetries,
|
||||
configuration: {
|
||||
baseURL,
|
||||
},
|
||||
})
|
||||
.pipe(new StringOutputParser())
|
||||
.invoke(
|
||||
finalMessages.filter((message) => message._getType() !== "system"),
|
||||
runConfig,
|
||||
),
|
||||
processTracedEvents,
|
||||
};
|
||||
}
|
||||
|
||||
if (streaming) {
|
||||
return chatModel.pipe(new BytesOutputParser()).stream(finalMessages);
|
||||
return {
|
||||
completion: await chatModel
|
||||
.pipe(new BytesOutputParser())
|
||||
.stream(finalMessages, runConfig),
|
||||
processTracedEvents,
|
||||
};
|
||||
}
|
||||
|
||||
return await chatModel.pipe(new StringOutputParser()).invoke(finalMessages);
|
||||
return {
|
||||
completion: await chatModel
|
||||
.pipe(new StringOutputParser())
|
||||
.invoke(finalMessages, runConfig),
|
||||
processTracedEvents,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -26,6 +26,20 @@ export enum ChatMessageRole {
|
||||
|
||||
export const ChatMessageDefaultRoleSchema = z.nativeEnum(ChatMessageRole);
|
||||
|
||||
const ChatMessageSchema = z.object({
|
||||
role: z.union([ChatMessageDefaultRoleSchema, z.string()]), // Users may ingest any string as role via API/SDK
|
||||
content: z.string(),
|
||||
});
|
||||
|
||||
export const ChatMessageListSchema = z.array(ChatMessageSchema);
|
||||
export const TextPromptSchema = z.string().min(1, "Enter a prompt");
|
||||
|
||||
export const PromptContentSchema = z.union([
|
||||
ChatMessageListSchema,
|
||||
TextPromptSchema,
|
||||
]);
|
||||
export type PromptContent = z.infer<typeof PromptContentSchema>;
|
||||
|
||||
export type ModelParams = {
|
||||
provider: string;
|
||||
adapter: LLMAdapter;
|
||||
@@ -49,7 +63,18 @@ export const ZodModelConfig = z.object({
|
||||
top_p: z.coerce.number().optional(),
|
||||
});
|
||||
|
||||
// NOTE: Update docs page when changing this!
|
||||
// Experiment config
|
||||
export const ExperimentMetadataSchema = z
|
||||
.object({
|
||||
prompt_id: z.string(),
|
||||
provider: z.string(),
|
||||
model: z.string(),
|
||||
model_params: ZodModelConfig,
|
||||
})
|
||||
.strict();
|
||||
export type ExperimentMetadata = z.infer<typeof ExperimentMetadataSchema>;
|
||||
|
||||
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
|
||||
export const openAIModels = [
|
||||
"gpt-4o",
|
||||
"gpt-4o-2024-08-06",
|
||||
@@ -76,11 +101,13 @@ export const openAIModels = [
|
||||
|
||||
export type OpenAIModel = (typeof openAIModels)[number];
|
||||
|
||||
// NOTE: Update docs page when changing this!
|
||||
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
|
||||
export const anthropicModels = [
|
||||
"claude-3-5-sonnet-20241022",
|
||||
"claude-3-5-sonnet-20240620",
|
||||
"claude-3-opus-20240229",
|
||||
"claude-3-sonnet-20240229",
|
||||
"claude-3-5-haiku-20241022",
|
||||
"claude-3-haiku-20240307",
|
||||
"claude-2.1",
|
||||
"claude-2.0",
|
||||
|
||||
@@ -0,0 +1,411 @@
|
||||
import { filterOperators } from "../../../interfaces/filters";
|
||||
import { clickhouseCompliantRandomCharacters } from "../../repositories";
|
||||
|
||||
export type ClickhouseOperator =
|
||||
| (typeof filterOperators)[keyof typeof filterOperators][number]
|
||||
| "!=";
|
||||
export interface Filter {
|
||||
apply(): ClickhouseFilter;
|
||||
clickhouseTable: string;
|
||||
operator: ClickhouseOperator;
|
||||
field: string;
|
||||
}
|
||||
type ClickhouseFilter = {
|
||||
query: string;
|
||||
params: { [x: string]: any } | {};
|
||||
};
|
||||
|
||||
export class StringFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public value: string;
|
||||
public operator: (typeof filterOperators)["string"][number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["string"][number];
|
||||
value: string;
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const varName = `stringFilter${clickhouseCompliantRandomCharacters()}`;
|
||||
|
||||
const fieldWithPrefix = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
|
||||
let query: string;
|
||||
switch (this.operator) {
|
||||
case "=":
|
||||
query = `${fieldWithPrefix} = {${varName}: String}`;
|
||||
break;
|
||||
case "contains":
|
||||
query = `position(${fieldWithPrefix}, {${varName}: String}) > 0`;
|
||||
break;
|
||||
case "does not contain":
|
||||
query = `position(${fieldWithPrefix}, {${varName}: String}) = 0`;
|
||||
break;
|
||||
case "starts with":
|
||||
query = `startsWith(${fieldWithPrefix}, {${varName}: String})`;
|
||||
break;
|
||||
case "ends with":
|
||||
query = `endsWith(${fieldWithPrefix}, {${varName}: String})`;
|
||||
break;
|
||||
default:
|
||||
throw new Error(`Unsupported operator: ${this.operator}`);
|
||||
}
|
||||
|
||||
return {
|
||||
query: query,
|
||||
params: { [varName]: this.value },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class NumberFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public value: number;
|
||||
public operator: (typeof filterOperators)["number"][number] | "!=";
|
||||
public clickhouseTypeOverwrite?: string;
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["number"][number] | "!=";
|
||||
value: number;
|
||||
tablePrefix?: string;
|
||||
clickhouseTypeOverwrite?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
this.clickhouseTypeOverwrite = opts.clickhouseTypeOverwrite;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const uid = clickhouseCompliantRandomCharacters();
|
||||
const varName = `numberFilter${uid}`;
|
||||
const type = this.clickhouseTypeOverwrite ?? "Decimal64(12)";
|
||||
return {
|
||||
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: ${type}}`,
|
||||
params: { [varName]: this.value.toString() },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class DateTimeFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public value: Date;
|
||||
public operator: (typeof filterOperators)["datetime"][number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["datetime"][number];
|
||||
value: Date;
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const uid = clickhouseCompliantRandomCharacters();
|
||||
const varName = `dateTimeFilter${uid}`;
|
||||
return {
|
||||
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: DateTime64(3)}`,
|
||||
params: { [varName]: new Date(this.value).getTime() },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class StringOptionsFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public values: string[];
|
||||
public operator: (typeof filterOperators.stringOptions)[number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators.stringOptions)[number];
|
||||
values: string[];
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.values = opts.values;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const uid = clickhouseCompliantRandomCharacters();
|
||||
const varName = `stringOptionsFilter${uid}`;
|
||||
return {
|
||||
query:
|
||||
this.operator === "any of"
|
||||
? `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`
|
||||
: `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`,
|
||||
params: { [varName]: this.values },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// stringObject filter is used when we want to filter on a key value pair in a clickhouse map.
|
||||
// As we use the MAP form clickhouse, we can only filter efficiently on the first level of a json obj.
|
||||
export class StringObjectFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public key: string;
|
||||
public value: string;
|
||||
public operator: (typeof filterOperators)["stringObject"][number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["stringObject"][number];
|
||||
key: string;
|
||||
value: string;
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
this.key = opts.key;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const varKeyName = `stringObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
|
||||
const varValueName = `stringObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
|
||||
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
|
||||
|
||||
// const query: `${column}['{varKeyName: String}'] ${this.operator} {${varValueName}: String}`,
|
||||
let query: string;
|
||||
switch (this.operator) {
|
||||
case "=":
|
||||
query = `${column}[{${varKeyName}: String}] = {${varValueName}: String}`;
|
||||
break;
|
||||
case "contains":
|
||||
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) > 0`;
|
||||
break;
|
||||
case "does not contain":
|
||||
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) = 0`;
|
||||
break;
|
||||
case "starts with":
|
||||
query = `startsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
|
||||
break;
|
||||
case "ends with":
|
||||
query = `endsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
|
||||
break;
|
||||
default:
|
||||
throw new Error(`Unsupported operator: ${this.operator}`);
|
||||
}
|
||||
|
||||
return {
|
||||
query,
|
||||
params: { [varKeyName]: this.key, [varValueName]: this.value },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// this is used when we want to filter multiple values on a clickhouse column which is also an array
|
||||
export class ArrayOptionsFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public values: string[];
|
||||
public operator: (typeof filterOperators.arrayOptions)[number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators.arrayOptions)[number];
|
||||
values: string[];
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.values = opts.values;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const uid = clickhouseCompliantRandomCharacters();
|
||||
const varName = `arrayOptionsFilter${uid}`;
|
||||
let query: string;
|
||||
|
||||
switch (this.operator) {
|
||||
case "any of":
|
||||
query = `hasAny({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`;
|
||||
break;
|
||||
case "none of":
|
||||
query = `hasAny({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`;
|
||||
break;
|
||||
case "all of":
|
||||
query = `hasAll(${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}, {${varName}: Array(String)}) = True`;
|
||||
break;
|
||||
default:
|
||||
throw new Error(`Unsupported operator: ${this.operator}`);
|
||||
}
|
||||
|
||||
return {
|
||||
query,
|
||||
params: { [varName]: this.values },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class NullFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public operator: (typeof filterOperators)["null"][number];
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["null"][number];
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
return {
|
||||
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator}`,
|
||||
params: {},
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class NumberObjectFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public key: string;
|
||||
public value: number;
|
||||
public operator: (typeof filterOperators)["numberObject"][number] | "!=";
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["numberObject"][number] | "!=";
|
||||
key: string;
|
||||
value: number;
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.operator = opts.operator;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
this.key = opts.key;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const varKeyName = `numberObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
|
||||
const varValueName = `numberObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
|
||||
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
|
||||
return {
|
||||
query: `empty(arrayFilter(x -> (((x.1) = {${varKeyName}: String}) AND ((x.2) ${this.operator} {${varValueName}: Decimal64(12)})), ${column})) = 0`,
|
||||
params: { [varKeyName]: this.key, [varValueName]: this.value },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class BooleanFilter implements Filter {
|
||||
public clickhouseTable: string;
|
||||
public field: string;
|
||||
public operator: (typeof filterOperators)["boolean"][number];
|
||||
public value: boolean;
|
||||
protected tablePrefix?: string;
|
||||
|
||||
constructor(opts: {
|
||||
clickhouseTable: string;
|
||||
field: string;
|
||||
operator: (typeof filterOperators)["boolean"][number];
|
||||
value: boolean;
|
||||
tablePrefix?: string;
|
||||
}) {
|
||||
this.clickhouseTable = opts.clickhouseTable;
|
||||
this.field = opts.field;
|
||||
this.value = opts.value;
|
||||
this.tablePrefix = opts.tablePrefix;
|
||||
this.operator = opts.operator;
|
||||
}
|
||||
|
||||
apply(): ClickhouseFilter {
|
||||
const uid = clickhouseCompliantRandomCharacters();
|
||||
const varName = `booleanFilter${uid}`;
|
||||
return {
|
||||
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: Boolean}`,
|
||||
params: { [varName]: this.value },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export class FilterList {
|
||||
private filters: Filter[];
|
||||
|
||||
constructor(filters: Filter[] = []) {
|
||||
this.filters = filters;
|
||||
}
|
||||
|
||||
push(...filter: Filter[]) {
|
||||
this.filters.push(...filter);
|
||||
}
|
||||
|
||||
find(predicate: (filter: Filter) => boolean) {
|
||||
return this.filters.find(predicate);
|
||||
}
|
||||
|
||||
length() {
|
||||
return this.filters.length;
|
||||
}
|
||||
|
||||
public apply(): ClickhouseFilter {
|
||||
if (this.filters.length === 0) {
|
||||
return {
|
||||
query: "",
|
||||
params: {},
|
||||
};
|
||||
}
|
||||
const compiledQueries = this.filters.map((filter) => filter.apply());
|
||||
const { params, queries } = compiledQueries.reduce(
|
||||
(acc, { params, query }) => {
|
||||
acc.params = { ...acc.params, ...params };
|
||||
acc.queries.push(query);
|
||||
return acc;
|
||||
},
|
||||
{ params: {}, queries: [] as string[] },
|
||||
);
|
||||
return {
|
||||
query: queries.join(" AND "),
|
||||
params,
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
import z from "zod";
|
||||
import { singleFilter } from "../../../interfaces/filters";
|
||||
import { FilterCondition } from "../../../types";
|
||||
import { isValidTableName } from "../../clickhouse/schemaUtils";
|
||||
import { logger } from "../../logger";
|
||||
import { UiColumnMapping } from "../../../tableDefinitions";
|
||||
import {
|
||||
StringFilter,
|
||||
DateTimeFilter,
|
||||
StringOptionsFilter,
|
||||
FilterList,
|
||||
NumberFilter,
|
||||
ArrayOptionsFilter,
|
||||
BooleanFilter,
|
||||
NumberObjectFilter,
|
||||
StringObjectFilter,
|
||||
NullFilter,
|
||||
} from "./clickhouse-filter";
|
||||
|
||||
export class QueryBuilderError extends Error {
|
||||
constructor(message: string) {
|
||||
super(message);
|
||||
this.name = "QueryBuilderError";
|
||||
}
|
||||
}
|
||||
|
||||
// This function ensures that the user only selects valid columns from the clickhouse schema.
|
||||
// The filter property in this column needs to be zod verified.
|
||||
// User input for values (e.g. project_id = <value>) are sent to Clickhouse as parameters to prevent SQL injection
|
||||
export const createFilterFromFilterState = (
|
||||
filter: FilterCondition[],
|
||||
columnMapping: UiColumnMapping[],
|
||||
) => {
|
||||
return filter.map((frontEndFilter) => {
|
||||
// checks if the column exists in the clickhouse schema
|
||||
const column = matchAndVerifyTracesUiColumn(frontEndFilter, columnMapping);
|
||||
|
||||
switch (frontEndFilter.type) {
|
||||
case "string":
|
||||
return new StringFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
value: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "datetime":
|
||||
return new DateTimeFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
value: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "stringOptions":
|
||||
return new StringOptionsFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
values: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "number":
|
||||
return new NumberFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
value: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
clickhouseTypeOverwrite: column.clickhouseTypeOverwrite,
|
||||
});
|
||||
case "arrayOptions":
|
||||
return new ArrayOptionsFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
values: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "boolean":
|
||||
return new BooleanFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
value: frontEndFilter.value,
|
||||
operator: frontEndFilter.operator,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "numberObject":
|
||||
return new NumberObjectFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
key: frontEndFilter.key,
|
||||
operator: frontEndFilter.operator,
|
||||
value: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "stringObject":
|
||||
return new StringObjectFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
key: frontEndFilter.key,
|
||||
value: frontEndFilter.value,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
case "null":
|
||||
return new NullFilter({
|
||||
clickhouseTable: column.clickhouseTableName,
|
||||
field: column.clickhouseSelect,
|
||||
operator: frontEndFilter.operator,
|
||||
tablePrefix: column.queryPrefix,
|
||||
});
|
||||
default:
|
||||
const exhaustiveCheck: never = frontEndFilter;
|
||||
logger.error(`Invalid filter type: ${JSON.stringify(exhaustiveCheck)}`);
|
||||
throw new QueryBuilderError(`Invalid filter type`);
|
||||
}
|
||||
});
|
||||
};
|
||||
|
||||
const matchAndVerifyTracesUiColumn = (
|
||||
filter: z.infer<typeof singleFilter>,
|
||||
uiTableDefinitions: UiColumnMapping[],
|
||||
) => {
|
||||
// tries to match the column name to the clickhouse table name
|
||||
logger.debug(`Filter to match: ${JSON.stringify(filter)}`);
|
||||
const uiTable = uiTableDefinitions.find(
|
||||
(col) =>
|
||||
col.uiTableName === filter.column || col.uiTableId === filter.column, // matches on the NAME of the column in the UI.
|
||||
);
|
||||
|
||||
if (!uiTable) {
|
||||
throw new QueryBuilderError(
|
||||
`Column ${filter.column} does not match a UI / CH table mapping.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (!isValidTableName(uiTable.clickhouseTableName)) {
|
||||
throw new QueryBuilderError(
|
||||
`Invalid clickhouse table name: ${uiTable.clickhouseTableName}`,
|
||||
);
|
||||
}
|
||||
|
||||
return uiTable;
|
||||
};
|
||||
|
||||
export function getProjectIdDefaultFilter(
|
||||
projectId: string,
|
||||
opts: { tracesPrefix: string },
|
||||
): {
|
||||
tracesFilter: FilterList;
|
||||
scoresFilter: FilterList;
|
||||
observationsFilter: FilterList;
|
||||
} {
|
||||
return {
|
||||
tracesFilter: new FilterList([
|
||||
new StringFilter({
|
||||
clickhouseTable: "traces",
|
||||
field: "project_id",
|
||||
operator: "=",
|
||||
value: projectId,
|
||||
tablePrefix: opts.tracesPrefix,
|
||||
}),
|
||||
]),
|
||||
scoresFilter: new FilterList([
|
||||
new StringFilter({
|
||||
clickhouseTable: "scores",
|
||||
field: "project_id",
|
||||
operator: "=",
|
||||
value: projectId,
|
||||
}),
|
||||
]),
|
||||
observationsFilter: new FilterList([
|
||||
new StringFilter({
|
||||
clickhouseTable: "observations",
|
||||
field: "project_id",
|
||||
operator: "=",
|
||||
value: projectId,
|
||||
}),
|
||||
]),
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
import z from "zod";
|
||||
import { OrderByState } from "../../../interfaces/orderBy";
|
||||
import { UiColumnMapping } from "../../../tableDefinitions";
|
||||
import { logger } from "../../logger";
|
||||
|
||||
export function orderByToClickhouseSql(
|
||||
orderBy: OrderByState,
|
||||
tableColumns: UiColumnMapping[],
|
||||
): string {
|
||||
if (!orderBy) {
|
||||
return "";
|
||||
}
|
||||
// Get column definition to map column to internal name, e.g. "t.id"
|
||||
const col = tableColumns.find(
|
||||
(c) => c.uiTableName === orderBy.column || c.uiTableId === orderBy.column,
|
||||
);
|
||||
|
||||
if (!col) {
|
||||
logger.warn("Invalid order by column", orderBy.column);
|
||||
throw new Error("Invalid order by column: " + orderBy.column);
|
||||
}
|
||||
|
||||
// Assert that orderBy.order is either "asc" or "desc"
|
||||
const orderByOrder = z.enum(["ASC", "DESC"]);
|
||||
const order = orderByOrder.safeParse(orderBy.order);
|
||||
if (!order.success) {
|
||||
logger.warn("Invalid order", orderBy.order);
|
||||
throw new Error("Invalid order: " + orderBy.order);
|
||||
}
|
||||
|
||||
// Both column and order are safe, can use raw SQL
|
||||
return `ORDER BY ${col.queryPrefix ? col.queryPrefix + "." : ""}${col.clickhouseSelect} ${order.data}`;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
const regexIndefiniteCharacters = "%";
|
||||
|
||||
export const clickhouseSearchCondition = (query?: string) => {
|
||||
return {
|
||||
query: query
|
||||
? `
|
||||
AND (
|
||||
id ILIKE {searchString: String} OR
|
||||
user_id ILIKE {searchString: String} OR
|
||||
name ILIKE {searchString: String}
|
||||
)
|
||||
`
|
||||
: "",
|
||||
params: query
|
||||
? {
|
||||
searchString: `${regexIndefiniteCharacters}${query}${regexIndefiniteCharacters}`,
|
||||
}
|
||||
: {},
|
||||
};
|
||||
};
|
||||
@@ -9,12 +9,10 @@ import { TableFilters } from "./types";
|
||||
|
||||
type AdditionalObservationFields = {
|
||||
traceName: string | null;
|
||||
promptName: string | null;
|
||||
promptVersion: string | null;
|
||||
traceTags: Array<string>;
|
||||
};
|
||||
|
||||
type FullObservation = AdditionalObservationFields & ObservationView;
|
||||
export type FullObservation = AdditionalObservationFields & ObservationView;
|
||||
|
||||
export type FullObservations = Array<FullObservation>;
|
||||
|
||||
@@ -40,17 +38,17 @@ export function parseGetAllGenerationsInput(filters: TableFilters) {
|
||||
const filterCondition = tableColumnsToSqlFilterAndPrefix(
|
||||
filters.filter ?? [],
|
||||
observationsTableCols,
|
||||
"observations"
|
||||
"observations",
|
||||
);
|
||||
|
||||
const orderByCondition = orderByToPrismaSql(
|
||||
filters.orderBy,
|
||||
observationsTableCols
|
||||
observationsTableCols,
|
||||
);
|
||||
|
||||
// to improve query performance, add timeseries filter to observation queries as well
|
||||
const startTimeFilter = filters.filter?.find(
|
||||
(f) => f.column === "Start Time" && f.type === "datetime"
|
||||
(f) => f.column === "Start Time" && f.type === "datetime",
|
||||
);
|
||||
|
||||
const datetimeFilter =
|
||||
@@ -58,7 +56,7 @@ export function parseGetAllGenerationsInput(filters: TableFilters) {
|
||||
? datetimeFilterToPrismaSql(
|
||||
"start_time",
|
||||
startTimeFilter.operator,
|
||||
startTimeFilter.value
|
||||
startTimeFilter.value,
|
||||
)
|
||||
: Prisma.empty;
|
||||
|
||||
|
||||
@@ -4,20 +4,20 @@ import {
|
||||
datetimeFilterToPrismaSql,
|
||||
tableColumnsToSqlFilterAndPrefix,
|
||||
} from "../filterToPrisma";
|
||||
import { tracesTableCols } from "../../tracesTable";
|
||||
import { tracesTableCols } from "../../tableDefinitions/tracesTable";
|
||||
import { orderByToPrismaSql } from "../orderByToPrisma";
|
||||
|
||||
export function parseTraceAllFilters(input: TableFilters) {
|
||||
const filterCondition = tableColumnsToSqlFilterAndPrefix(
|
||||
input.filter ?? [],
|
||||
tracesTableCols,
|
||||
"traces"
|
||||
"traces",
|
||||
);
|
||||
const orderByCondition = orderByToPrismaSql(input.orderBy, tracesTableCols);
|
||||
|
||||
// to improve query performance, add timeseries filter to observation queries as well
|
||||
const timeseriesFilter = input.filter?.find(
|
||||
(f) => f.column === "Timestamp" && f.type === "datetime"
|
||||
(f) => f.column === "Timestamp" && f.type === "datetime",
|
||||
);
|
||||
|
||||
const observationTimeseriesFilter =
|
||||
@@ -25,7 +25,7 @@ export function parseTraceAllFilters(input: TableFilters) {
|
||||
? datetimeFilterToPrismaSql(
|
||||
"start_time",
|
||||
timeseriesFilter.operator,
|
||||
timeseriesFilter.value
|
||||
timeseriesFilter.value,
|
||||
)
|
||||
: Prisma.empty;
|
||||
|
||||
@@ -130,6 +130,6 @@ export function createTracesQuery({
|
||||
${filterCondition}
|
||||
${orderByCondition}
|
||||
${limit ? Prisma.sql`LIMIT ${limit}` : Prisma.empty}
|
||||
${page && limit ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
|
||||
${page !== undefined && limit !== undefined ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
|
||||
`;
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user