Compare commits

..
34 Commits
Author SHA1 Message Date
Max Deichmann e21102664e chore: release v2.95.2 2025-02-15 13:29:57 +01:00
Max DeichmannandGitHub d31b0eaddc security: upgrade dompurify v2 (#5570)
security: upgrade dompurify
2025-02-15 12:26:51 +00:00
Max Deichmann f53ad4de5c chore: release v2.95.1
Codespell / Check for spelling errors (push) Waiting to run
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-02-11 14:19:25 +01:00
Max DeichmannandGitHub 053d7d668d security: upgrade sentry 8.52.0 (#5477)
fix
2025-02-11 14:18:38 +01:00
Max DeichmannandGitHub 21e3ed2b39 security: upgrade clickhouse migration package (#5478)
push
2025-02-11 14:18:26 +01:00
Max DeichmannandGitHub 16ca4e9293 security: upgrade json path (#5475) 2025-02-11 13:44:51 +01:00
Marc KlingenandGitHub 75f82be88d ci(v2): run codespell also on v2 branch and prs (#5224) (#5225) 2025-01-27 13:25:31 +01:00
Marc Klingen 22f6a02b08 chore: release v2.95.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-27 13:16:16 +01:00
Marc KlingenandGitHub 11aa1dbbb1 feat(v2-auth): make checks and auth method configurable across SSO providers (#5203) (#5219) 2025-01-27 13:15:33 +01:00
steffen911 d961d85a28 chore: release v2.94.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-24 16:35:20 +01:00
Steffen SchmitzandGitHub d0c1ad5144 feat: add proxy support for oauth flows (#5198) (#5201)
(cherry picked from commit c442c4290e)
2025-01-24 16:34:53 +01:00
Marc Klingen 0d30b2fe83 chore: release v2.93.9
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-24 14:12:59 +01:00
Marc KlingenandGitHub f33bae6683 feat(auth-v2): Add AUTH_CUSTOM_ID_TOKEN environment variable (#5193) (#5196) 2025-01-24 14:12:00 +01:00
Baptiste Mille-MathiasandGitHub eaa0df125b feat: add support for DATABASE_ARGS config (cherry-pick) (#5152) 2025-01-24 13:50:17 +01:00
Max Deichmann b0e01b7127 chore: release v2.93.8
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-06 10:54:29 +01:00
Max DeichmannandGitHub 2a421e7406 security: upgrade next (#4891)
push
2025-01-06 10:45:23 +01:00
Marc Klingen 23150b68db chore: release v2.93.7
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-19 00:57:59 +01:00
Ildar IapparovandMarc Klingen e5c46010a4 feat(auth): add AUTH_IGNORE_ACCOUNT_FIELDS to sanitize IDP fields before creating an account (#4728)
* feat: Field sanitization before creating an Account

* add comments

---------

Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-19 00:57:07 +01:00
Marc Klingen b2bf68d7a4 chore: release v2.93.6
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-13 02:57:59 +01:00
Marc Klingen 31cec4f5c9 feat: in HF Spaces, prompt opening in new tab when running in iframe (#4713) 2024-12-13 02:45:32 +01:00
Max Deichmann 84a0ad8dfb chore: release v2.93.5
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-12 14:06:33 +01:00
Max DeichmannandGitHub 69466fd43b security: upgrade next-auth (#4702) 2024-12-12 14:06:12 +01:00
Marc Klingen 324e078c85 chore: release v2.93.4
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-04 00:07:38 +01:00
Marc Klingen 7385fc4529 fix: disable x frame options header on Hugging Face (#4558) 2024-12-04 00:07:02 +01:00
Marc Klingen 66d1fa427f chore: release v2.93.3
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 23:37:22 +01:00
bee396a433 feat(auth): add KeyCloak authentication option (#2866)
---------

Co-authored-by: RTae <natthanan.bhu@doctorasa.co>
Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-03 23:35:01 +01:00
jay0129andMarc Klingen 8727a52931 feat(auth): add GitHub Enterprise Authentication Provider (#4463)
feat: Add GitHub Enterprise Authentication Provider

Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-03 23:34:31 +01:00
Marc Klingen ed5c076a5a fix(auth): add nonce check for Cognito NextAuth provider (#4401) 2024-12-03 23:34:15 +01:00
Marc Klingen db5c575ae0 chore: release v2.93.2
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 17:53:55 +01:00
Marc Klingen 2a0f482578 fix: remove LANGFUSE_CSP_DISABLE (did not work) and disable csp headers on HF Spaces (#4545) 2024-12-03 17:51:50 +01:00
Marc Klingen c041cf371a chore: release v2.93.1
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 01:25:53 +01:00
Marc Klingen c6daf09cd2 [cherry-pick][2.93.x] chore: add LANGFUSE_CSP_DISABLE to .env.prod.example (#4533) 2024-12-03 01:25:04 +01:00
Marc KlingenandGitHub 7296e2e012 [cherry-pick][2.93.x] feat: optionally disable csp headers via LANGFUSE_CSP_DISABLE=true (#4532) 2024-12-03 01:22:11 +01:00
steffen911 43bf176ef7 chore: release v2.93.0 2024-11-26 10:04:45 +01:00
534 changed files with 23180 additions and 27222 deletions
+18 -10
View File
@@ -11,7 +11,7 @@ CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_ENABLED="false"
CLICKHOUSE_MIGRATION_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
@@ -35,18 +35,17 @@ EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# DON'T PANIC: The Azurite Secrets are well-known and meant to be hard-coded
# S3 Batch Exports
LANGFUSE_S3_BATCH_EXPORT_ENABLED=true
LANGFUSE_S3_BATCH_EXPORT_BUCKET=langfuse
LANGFUSE_S3_BATCH_EXPORT_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_BATCH_EXPORT_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
LANGFUSE_S3_BATCH_EXPORT_REGION=auto
LANGFUSE_S3_BATCH_EXPORT_ENDPOINT=http://localhost:10000/devstoreaccount1
# S3 storage
S3_ENDPOINT=http://localhost:10000/devstoreaccount1
S3_ACCESS_KEY_ID=devstoreaccount1
S3_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
S3_BUCKET_NAME=langfuse
S3_REGION=auto
## Necessary for minio compatibility
LANGFUSE_S3_BATCH_EXPORT_FORCE_PATH_STYLE=true
LANGFUSE_S3_BATCH_EXPORT_PREFIX=exports/
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
@@ -58,6 +57,7 @@ LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
@@ -82,3 +82,11 @@ ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+18 -10
View File
@@ -11,7 +11,7 @@ CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_ENABLED="false"
CLICKHOUSE_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
@@ -34,18 +34,17 @@ SALT="salt"
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# S3 Batch Exports
LANGFUSE_S3_BATCH_EXPORT_ENABLED=true
LANGFUSE_S3_BATCH_EXPORT_BUCKET=langfuse
LANGFUSE_S3_BATCH_EXPORT_ACCESS_KEY_ID=minio
LANGFUSE_S3_BATCH_EXPORT_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_BATCH_EXPORT_REGION=us-east-1
LANGFUSE_S3_BATCH_EXPORT_ENDPOINT=http://localhost:9090
# S3 storage
S3_ENDPOINT=http://localhost:9090
S3_ACCESS_KEY_ID=minio
S3_SECRET_ACCESS_KEY=miniosecret
S3_BUCKET_NAME=langfuse
S3_REGION=us-east-1
## Necessary for minio compatibility
LANGFUSE_S3_BATCH_EXPORT_FORCE_PATH_STYLE=true
LANGFUSE_S3_BATCH_EXPORT_PREFIX=exports/
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
@@ -57,6 +56,7 @@ LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
@@ -79,3 +79,11 @@ ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+81
View File
@@ -0,0 +1,81 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
# Clickhouse
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_ENABLED="false"
# Next Auth
# You can generate a new secret on the command line with:
# openssl rand -base64 32
# https://next-auth.js.org/configuration/options#secret
# NEXTAUTH_SECRET=""
NEXTAUTH_URL="http://localhost:3000"
NEXTAUTH_SECRET="secret"
# Langfuse Cloud Environment
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
# Langfuse experimental features
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
# Salt for API key hashing
SALT="salt"
# Email
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# S3 storage
S3_ENDPOINT=http://localhost:9090
S3_ACCESS_KEY_ID=minio
S3_SECRET_ACCESS_KEY=miniosecret
S3_BUCKET_NAME=langfuse
S3_REGION=us-east-1
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# # S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
# Set during docker build of application
# Used to disable environment verification at build time
# DOCKER_BUILD=1
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
+22
View File
@@ -0,0 +1,22 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@db:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@db:5432/postgres"
# Next Auth
NEXTAUTH_SECRET="secret"
NEXTAUTH_URL="http://localhost:3000"
# feature flag to enable experimental features locally
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="false"
SALT="salt"
# Redis
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
+11 -10
View File
@@ -101,7 +101,6 @@ OTEL_SERVICE_NAME="langfuse"
# AUTH_CUSTOM_ISSUER=
# AUTH_CUSTOM_NAME=
# AUTH_CUSTOM_SCOPE="openid email profile" # optional
# AUTH_CUSTOM_CLIENT_AUTH_METHOD="client_secret_basic" # optional
# AUTH_CUSTOM_ALLOW_ACCOUNT_LINKING=false
# AUTH_CUSTOM_ID_TOKEN=false # optional, default is true
@@ -111,16 +110,16 @@ OTEL_SERVICE_NAME="langfuse"
# Defines the connection url for smtp server.
# SMTP_CONNECTION_URL=
# S3 Batch Exports
# LANGFUSE_S3_BATCH_EXPORT_ENABLED=
# LANGFUSE_S3_BATCH_EXPORT_BUCKET=
# LANGFUSE_S3_BATCH_EXPORT_ACCESS_KEY_ID=
# LANGFUSE_S3_BATCH_EXPORT_SECRET_ACCESS_KEY=
# LANGFUSE_S3_BATCH_EXPORT_REGION=
# LANGFUSE_S3_BATCH_EXPORT_ENDPOINT=
# LANGFUSE_S3_BATCH_EXPORT_PREFIX=
# S3 storage, optional, used for exports from the UI
# S3_ENDPOINT=
# S3_ACCESS_KEY_ID=
# S3_SECRET_ACCESS_KEY=
# S3_BUCKET_NAME=
# S3_REGION=
# BATCH_EXPORT_DOWNLOAD_LINK_EXPIRATION_HOURS=
# S3 storage for events, optional, used to persist all incoming events
# LANGFUSE_S3_EVENT_UPLOAD_ENABLED="true"
# LANGFUSE_S3_EVENT_UPLOAD_BUCKET=
# Optional prefix to be used within the bucket. Must end with `/` if set
# LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
@@ -246,12 +245,14 @@ OTEL_SERVICE_NAME="langfuse"
# CLICKHOUSE_URL=
# CLICKHOUSE_USER=
# CLICKHOUSE_PASSWORD=
# CLICKHOUSE_DB=
# Ingestion
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_BATCH_SIZE=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_MAX_ATTEMPTS=
# LANGFUSE_LEGACY_INGESTION_WORKER_CONCURRENCY=
# LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
# QUEUE_CONSUMER_LEGACY_INGESTION_QUEUE_IS_ENABLED="true"
## END Langfuse V3 Ingestion
@@ -58,6 +58,7 @@ jobs:
--build-arg SENTRY_PROJECT=${{ vars.SENTRY_PROJECT }} \
.
docker push $REGISTRY/$REPOSITORY:$IMAGE_TAG
- name: Render AWS ECS Task Definition
id: render-task-definition
uses: aws-actions/amazon-ecs-render-task-definition@v1
+10 -59
View File
@@ -10,29 +10,11 @@ on:
merge_group:
pull_request:
branches:
- "**"
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
- "*"
jobs:
pre-job:
runs-on: ubuntu-latest
outputs:
should_skip: ${{ steps.skip_check.outputs.should_skip }}
timeout-minutes: 15
steps:
- id: skip_check
uses: fkirc/skip-duplicate-actions@v5
with:
do_not_skip: '["workflow_dispatch"]'
lint:
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
steps:
- uses: actions/checkout@v4
- uses: pnpm/action-setup@v3
@@ -55,14 +37,10 @@ jobs:
test-docker-build:
timeout-minutes: 20
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -91,9 +69,6 @@ jobs:
tests-web-sync:
timeout-minutes: 20
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
name: tests-web-sync (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
strategy:
matrix:
@@ -114,7 +89,6 @@ jobs:
with:
version: 9.5.0
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -131,9 +105,7 @@ jobs:
- name: Load default env
run: |
cp .env.dev.example .env
grep -v -e '^LANGFUSE_S3_BATCH_EXPORT_ENABLED=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.example > .env
echo "LANGFUSE_INGESTION_QUEUE_DELAY_MS=1" >> .env
echo "LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=1" >> .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.legacy.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev.yml up -d
@@ -161,15 +133,10 @@ jobs:
LANGFUSE_INIT_USER_PASSWORD: "password"
- name: run test-sync
run: pnpm --filter=web run test-sync
- name: run test-client
run: pnpm --filter=web run test-client
tests-web-async:
timeout-minutes: 20
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
@@ -191,7 +158,6 @@ jobs:
with:
version: 9.5.0
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -208,9 +174,7 @@ jobs:
- name: Load default env
run: |
cp .env.dev${{ matrix.blob-provider }}.example .env
grep -v -e '^LANGFUSE_S3_BATCH_EXPORT_ENABLED=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
echo "LANGFUSE_INGESTION_QUEUE_DELAY_MS=1" >> .env
echo "LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=1" >> .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
@@ -242,9 +206,6 @@ jobs:
tests-worker:
timeout-minutes: 20
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
@@ -261,7 +222,6 @@ jobs:
with:
version: 9.5.0
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -310,9 +270,6 @@ jobs:
e2e-tests:
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
steps:
- uses: actions/checkout@v4
- uses: pnpm/action-setup@v3
@@ -324,7 +281,6 @@ jobs:
cache: "pnpm"
cache-dependency-path: "pnpm-lock.yaml"
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -354,13 +310,9 @@ jobs:
e2e-server-tests:
runs-on: ubuntu-latest
needs:
- pre-job
if: needs.pre-job.outputs.should_skip != 'true'
steps:
- uses: actions/checkout@v4
- name: Login to Docker Hub
if: github.repository == 'langfuse/langfuse' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
@@ -384,6 +336,8 @@ jobs:
- name: Load default env
run: |
cp .env.dev.example .env
echo "LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_ASYNC_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_CACHE_API_KEY_ENABLED=true" >> .env
echo "LANGFUSE_CACHE_PROMPT_ENABLED=true" >> .env
- name: Run + migrate
@@ -494,9 +448,9 @@ jobs:
type=ref,event=pr
type=sha
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}},enable=${{ !contains(github.ref, '-rc') }}
type=semver,pattern={{major}},enable=${{ !contains(github.ref, '-rc') }}
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') && !contains(github.ref, '-rc') }}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') }}
- name: Build and push Docker image (web)
uses: docker/build-push-action@v4
with:
@@ -515,16 +469,13 @@ jobs:
images: |
ghcr.io/langfuse/langfuse-worker # GitHub
langfuse/langfuse-worker # Docker Hub
flavor: |
latest=false
tags: |
type=ref,event=branch
type=ref,event=pr
type=sha
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}},enable=${{ !contains(github.ref, '-rc') }}
type=semver,pattern={{major}},enable=${{ !contains(github.ref, '-rc') }}
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') && !contains(github.ref, '-rc') }}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
- name: Build and push Docker image (worker)
uses: docker/build-push-action@v4
with:
+3 -8
View File
@@ -6,11 +6,6 @@ on:
push:
branches:
- main
merge_group:
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
snyk:
@@ -20,13 +15,13 @@ jobs:
- name: Build a Docker image
run: docker compose -f docker-compose.build.yml up -d
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-web)
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-server)
continue-on-error: true
uses: snyk/actions/docker@master
env:
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
with:
image: langfuse-langfuse-web
image: langfuse-server
args: --file=web/Dockerfile
- name: Upload result to GitHub Code Scanning
@@ -41,7 +36,7 @@ jobs:
env:
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
with:
image: langfuse-langfuse-worker
image: langfuse-worker
args: --file=worker/Dockerfile
- name: Upload result to GitHub Code Scanning
+2
View File
@@ -38,7 +38,9 @@ yarn-error.log*
.env*
!.env.dev.example
!.env.dev-azure.example
!.env.local.example
!.env.prod.example
!.env.dev.legacy.example
# vercel
.vercel
+29 -33
View File
@@ -121,40 +121,36 @@ flowchart TB
end
end
subgraph s9 ["VPC (US and EU separated)"]
DB[Postgres Database]
Redis[Redis Cache/Queue]
Clickhouse[Clickhouse Database]
subgraph s1["Application (langfuse/langfuse/web)"]
API[Public HTTP API]
G[TRPC API]
I[NextAuth]
H[React Frontend]
ORM
H --> G
H --> I
G --> I
G --- ORM
API --- ORM
I --- ORM
end
subgraph s5["Application (langfuse/langfuse/worker)"]
Worker
end
Worker --- DB
Worker --- Redis
Worker --- Clickhouse
ORM --- DB
ORM --- Redis
ORM --- Clickhouse
DB[Postgres Database]
Redis[Redis Cache/Queue]
Clickhouse[Clickhouse Database]
subgraph s1["Application (langfuse/langfuse/web)"]
API[Public HTTP API]
G[TRPC API]
I[NextAuth]
H[React Frontend]
ORM
H --> G
H --> I
G --> I
G --- ORM
API --- ORM
I --- ORM
end
subgraph s5["Application (langfuse/langfuse/worker)"]
Worker
end
Worker --- DB
Worker --- Redis
Worker --- Clickhouse
ORM --- DB
ORM --- Redis
ORM --- Clickhouse
JS --- API
Python --- API
```
@@ -401,7 +397,7 @@ The background color of the following component will be `hsl(var(--primary))` an
| --primary-accent | Primary accent color used for branding | Layout |
| --hover-primary-accent | Primary accent color used for hover effects for links | SignIn and AuthCloudRegionSwitch |
| --muted-green | Muted green for Event label | ObservationTree |
| --muted-magenta | Muted magenta for Generation label | ObservationTree |
| --muted-orange | Muted orange for Generation label | ObservationTree |
| --muted-blue | Muted blue for Span label | ObservationTree |
| --muted-gray | Muted gray for disabled status badges | StatusBadge |
| --accent-light-green | Light green accent for background of output and assistant messages | IOPreview, Generations, Traces |
@@ -443,7 +439,7 @@ You can update the default AI models and prices by adding or updating an entry i
Please note that
- prices are in USD
- the list is ordered by ID, so make sure to keep this order and insert new models at the end of the list
- the list is ordered by ID, so make sure to keep this order
- the `updated_at` field must be updated with the current date in ISO 8601 format. Otherwise, the change will be ignored.
### Transition period until V3 release
+2 -2
View File
@@ -2,8 +2,8 @@ Copyright (c) 2023--2024 Langfuse GmbH
Portions of this software are licensed as follows:
- All content that resides under the "ee/", "web/src/ee/", and/or "worker/src/ee/" directories of this repository, if these directories exist, is licensed under the license defined in "ee/LICENSE".
- All third party components incorporated into the Langfuse Software are licensed under the original license provided by the owner of the applicable component.
- All content that resides under the "ee/" and/or "web/src/ee" directories of this repository, if these directories exist, is licensed under the license defined in "ee/LICENSE".
- All third party components incorporated into the Finto Technologies Software are licensed under the original license provided by the owner of the applicable component.
- Content outside of the above mentioned directories or restrictions above is available under the "MIT Expat" license as defined below.
Permission is hereby granted, free of charge, to any person obtaining a copy
+3 -52
View File
@@ -42,7 +42,9 @@
## Langfuse Overview
[![Langfuse Overview Video](https://github.com/user-attachments/assets/3926b288-ff61-4b95-8aa1-45d041c70866)](https://langfuse.com/watch-demo)
_Unmute video for voice-over_
https://github.com/langfuse/langfuse/assets/2834609/a94062e9-c782-4ee9-af59-dee6370149a8
### Develop
@@ -189,54 +191,3 @@ You can opt-out by setting `TELEMETRY_ENABLED=false`.
<img alt="Star History Chart" src="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date" />
</picture>
</a>
### Open Source Projects Using Langfuse
Top open-source Python projects that use Langfuse, ranked by stars ([Source](https://github.com/langfuse/langfuse-docs/blob/main/components-mdx/dependents)):
| Repository | Stars |
| :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----: |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/127165244?s=40&v=4" width="20" height="20" alt=""> &nbsp; [langgenius](https://github.com/langgenius) / [dify](https://github.com/langgenius/dify) | 54865 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/158137808?s=40&v=4" width="20" height="20" alt=""> &nbsp; [open-webui](https://github.com/open-webui) / [open-webui](https://github.com/open-webui/open-webui) | 51531 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/131470832?s=40&v=4" width="20" height="20" alt=""> &nbsp; [lobehub](https://github.com/lobehub) / [lobe-chat](https://github.com/lobehub/lobe-chat) | 49003 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/85702467?s=40&v=4" width="20" height="20" alt=""> &nbsp; [langflow-ai](https://github.com/langflow-ai) / [langflow](https://github.com/langflow-ai/langflow) | 39093 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/130722866?s=40&v=4" width="20" height="20" alt=""> &nbsp; [run-llama](https://github.com/run-llama) / [llama_index](https://github.com/run-llama/llama_index) | 37368 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/139558948?s=40&v=4" width="20" height="20" alt=""> &nbsp; [chatchat-space](https://github.com/chatchat-space) / [Langchain-Chatchat](https://github.com/chatchat-space/Langchain-Chatchat) | 32486 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/128289781?s=40&v=4" width="20" height="20" alt=""> &nbsp; [FlowiseAI](https://github.com/FlowiseAI) / [Flowise](https://github.com/FlowiseAI/Flowise) | 32448 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/31035808?s=40&v=4" width="20" height="20" alt=""> &nbsp; [mindsdb](https://github.com/mindsdb) / [mindsdb](https://github.com/mindsdb/mindsdb) | 26931 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/119600397?s=40&v=4" width="20" height="20" alt=""> &nbsp; [twentyhq](https://github.com/twentyhq) / [twenty](https://github.com/twentyhq/twenty) | 24195 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/60330232?s=40&v=4" width="20" height="20" alt=""> &nbsp; [PostHog](https://github.com/PostHog) / [posthog](https://github.com/PostHog/posthog) | 22618 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/121462774?s=40&v=4" width="20" height="20" alt=""> &nbsp; [BerriAI](https://github.com/BerriAI) / [litellm](https://github.com/BerriAI/litellm) | 15151 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/179202840?s=40&v=4" width="20" height="20" alt=""> &nbsp; [mediar-ai](https://github.com/mediar-ai) / [screenpipe](https://github.com/mediar-ai/screenpipe) | 11037 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/105877416?s=40&v=4" width="20" height="20" alt=""> &nbsp; [formbricks](https://github.com/formbricks) / [formbricks](https://github.com/formbricks/formbricks) | 9386 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/76263028?s=40&v=4" width="20" height="20" alt=""> &nbsp; [anthropics](https://github.com/anthropics) / [courses](https://github.com/anthropics/courses) | 8385 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/78410652?s=40&v=4" width="20" height="20" alt=""> &nbsp; [GreyDGL](https://github.com/GreyDGL) / [PentestGPT](https://github.com/GreyDGL/PentestGPT) | 7374 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/152537519?s=40&v=4" width="20" height="20" alt=""> &nbsp; [superagent-ai](https://github.com/superagent-ai) / [superagent](https://github.com/superagent-ai/superagent) | 5391 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/137907881?s=40&v=4" width="20" height="20" alt=""> &nbsp; [promptfoo](https://github.com/promptfoo) / [promptfoo](https://github.com/promptfoo/promptfoo) | 4976 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/157326433?s=40&v=4" width="20" height="20" alt=""> &nbsp; [onlook-dev](https://github.com/onlook-dev) / [onlook](https://github.com/onlook-dev/onlook) | 4141 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/7250217?s=40&v=4" width="20" height="20" alt=""> &nbsp; [Canner](https://github.com/Canner) / [WrenAI](https://github.com/Canner/WrenAI) | 2526 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/11855343?s=40&v=4" width="20" height="20" alt=""> &nbsp; [pingcap](https://github.com/pingcap) / [autoflow](https://github.com/pingcap/autoflow) | 2061 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/85268109?s=40&v=4" width="20" height="20" alt=""> &nbsp; [MLSysOps](https://github.com/MLSysOps) / [MLE-agent](https://github.com/MLSysOps/MLE-agent) | 1161 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/158137808?s=40&v=4" width="20" height="20" alt=""> &nbsp; [open-webui](https://github.com/open-webui) / [pipelines](https://github.com/open-webui/pipelines) | 1100 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/18422723?s=40&v=4" width="20" height="20" alt=""> &nbsp; [alishobeiri](https://github.com/alishobeiri) / [thread](https://github.com/alishobeiri/thread) | 1074 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/125468716?s=40&v=4" width="20" height="20" alt=""> &nbsp; [topoteretes](https://github.com/topoteretes) / [cognee](https://github.com/topoteretes/cognee) | 971 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/188657705?s=40&v=4" width="20" height="20" alt=""> &nbsp; [bRAGAI](https://github.com/bRAGAI) / [bRAG-langchain](https://github.com/bRAGAI/bRAG-langchain) | 823 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/169500408?s=40&v=4" width="20" height="20" alt=""> &nbsp; [opslane](https://github.com/opslane) / [opslane](https://github.com/opslane/opslane) | 677 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/151867818?s=40&v=4" width="20" height="20" alt=""> &nbsp; [dynamiq-ai](https://github.com/dynamiq-ai) / [dynamiq](https://github.com/dynamiq-ai/dynamiq) | 639 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/48585267?s=40&v=4" width="20" height="20" alt=""> &nbsp; [theopenconversationkit](https://github.com/theopenconversationkit) / [tock](https://github.com/theopenconversationkit/tock) | 514 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/20493493?s=40&v=4" width="20" height="20" alt=""> &nbsp; [andysingal](https://github.com/andysingal) / [llm-course](https://github.com/andysingal/llm-course) | 394 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/132396805?s=40&v=4" width="20" height="20" alt=""> &nbsp; [phospho-app](https://github.com/phospho-app) / [phospho](https://github.com/phospho-app/phospho) | 384 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/178644984?s=40&v=4" width="20" height="20" alt=""> &nbsp; [sentient-engineering](https://github.com/sentient-engineering) / [agent-q](https://github.com/sentient-engineering/agent-q) | 370 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/168552753?s=40&v=4" width="20" height="20" alt=""> &nbsp; [sql-agi](https://github.com/sql-agi) / [DB-GPT](https://github.com/sql-agi/DB-GPT) | 324 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/60330232?s=40&v=4" width="20" height="20" alt=""> &nbsp; [PostHog](https://github.com/PostHog) / [posthog-foss](https://github.com/PostHog/posthog-foss) | 305 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/154247157?s=40&v=4" width="20" height="20" alt=""> &nbsp; [vespperhq](https://github.com/vespperhq) / [vespper](https://github.com/vespperhq/vespper) | 304 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/185116535?s=40&v=4" width="20" height="20" alt=""> &nbsp; [block](https://github.com/block) / [goose](https://github.com/block/goose) | 295 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/609489?s=40&v=4" width="20" height="20" alt=""> &nbsp; [aorwall](https://github.com/aorwall) / [moatless-tools](https://github.com/aorwall/moatless-tools) | 291 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/2357342?s=40&v=4" width="20" height="20" alt=""> &nbsp; [dmayboroda](https://github.com/dmayboroda) / [minima](https://github.com/dmayboroda/minima) | 221 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/66303003?s=40&v=4" width="20" height="20" alt=""> &nbsp; [RobotecAI](https://github.com/RobotecAI) / [rai](https://github.com/RobotecAI/rai) | 172 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/148684274?s=40&v=4" width="20" height="20" alt=""> &nbsp; [i-am-alice](https://github.com/i-am-alice) / [3rd-devs](https://github.com/i-am-alice/3rd-devs) | 148 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/171735272?s=40&v=4" width="20" height="20" alt=""> &nbsp; [8090-inc](https://github.com/8090-inc) / [xrx-sample-apps](https://github.com/8090-inc/xrx-sample-apps) | 138 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/104478511?s=40&v=4" width="20" height="20" alt=""> &nbsp; [babelcloud](https://github.com/babelcloud) / [LLM-RGB](https://github.com/babelcloud/LLM-RGB) | 135 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/15125613?s=40&v=4" width="20" height="20" alt=""> &nbsp; [souzatharsis](https://github.com/souzatharsis) / [tamingLLMs](https://github.com/souzatharsis/tamingLLMs) | 129 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/169401942?s=40&v=4" width="20" height="20" alt=""> &nbsp; [LibreChat-AI](https://github.com/LibreChat-AI) / [librechat.ai](https://github.com/LibreChat-AI/librechat.ai) | 128 |
| <img class="avatar mr-2" src="https://avatars.githubusercontent.com/u/51827949?s=40&v=4" width="20" height="20" alt=""> &nbsp; [deepset-ai](https://github.com/deepset-ai) / [haystack-core-integrations](https://github.com/deepset-ai/haystack-core-integrations) | 126 |
+3 -9
View File
@@ -1,19 +1,13 @@
_We are Hiring_
Join us in building out Langfuse in Berlin, Germany. Langfuse is the open source LLM engineering platform: we build tooling to help developers [build & improve LLM applications](https://langfuse.com/docs).
We are an open source company, we hire in person (4+ days a week), we only hire excellent technical talent. Find more information on our [careers page](https://langfuse.com/careers)
Join us in scaling Langfuse in Berlin, Germany. We are an open source company, we hire in person, we are only hiring technical talent.
_Open Roles_
- Product Engineer, 70-130k EUR, 0.1-0.35% Equity, https://www.ycombinator.com/companies/langfuse/jobs/aAvmoFB-product-engineer
- Backend Engineer, 70-130k EUR, 0.1-0.35% Equity, https://www.ycombinator.com/companies/langfuse/jobs/1bO16H6-backend-engineer
- Design Engineer, 70-130k EUR, 0.1-0.35% Equity, https://www.ycombinator.com/companies/langfuse/jobs/mDquP95-design-engineer
- Developer Advocate, 70-130k EUR, 0.1-0.35% Equity, https://www.ycombinator.com/companies/langfuse/jobs/uHysbKH-developer-advocate-devrel
- Product Engineer, 70-130k EUR, 0.25-0.75% Equity, https://www.ycombinator.com/companies/langfuse/jobs/aAvmoFB-product-engineer
- Developer Advocate, 60-110k EUR, 0.25-0.5% Equity, https://www.ycombinator.com/companies/langfuse/jobs/uHysbKH-developer-advocate-devrel
_More Info_
- https://langfuse.com/careers
- https://langfuse.com/docs
- https://langfuse.com/why
- https://langfuse.com/changelog
+44 -113
View File
@@ -1,53 +1,34 @@
version: "3.5"
services:
langfuse-web:
server:
build:
dockerfile: ./web/Dockerfile
context: .
args:
- NEXT_PUBLIC_LANGFUSE_CLOUD_REGION=${NEXT_PUBLIC_LANGFUSE_CLOUD_REGION}
depends_on: &langfuse-depends-on
postgres:
condition: service_healthy
minio:
condition: service_healthy
redis:
condition: service_healthy
clickhouse:
condition: service_healthy
depends_on:
- db
- redis
ports:
- "3000:3000"
environment: &langfuse-web-env
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres
NEXTAUTH_SECRET: mysecret
SALT: mysalt
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # generate via `openssl rand -hex 32`
NEXTAUTH_URL: http://localhost:3000
TELEMETRY_ENABLED: ${TELEMETRY_ENABLED:-true}
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-false}
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
LANGFUSE_INIT_PROJECT_ID: ${LANGFUSE_INIT_PROJECT_ID:-}
LANGFUSE_INIT_PROJECT_NAME: ${LANGFUSE_INIT_PROJECT_NAME:-}
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: ${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
LANGFUSE_INIT_PROJECT_SECRET_KEY: ${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
LANGFUSE_INIT_USER_EMAIL: ${LANGFUSE_INIT_USER_EMAIL:-}
LANGFUSE_INIT_USER_NAME: ${LANGFUSE_INIT_USER_NAME:-}
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
REDIS_HOST: ${REDIS_HOST:-redis}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
environment:
- DATABASE_URL=postgresql://postgres:postgres@db:5432/postgres
- NEXTAUTH_SECRET=mysecret
- SALT=mysalt
- ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000 # generate via `openssl rand -hex 32`
- NEXTAUTH_URL=http://localhost:3000
- TELEMETRY_ENABLED=${TELEMETRY_ENABLED:-true}
- LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES=${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-false}
- LANGFUSE_INIT_ORG_ID=${LANGFUSE_INIT_ORG_ID:-}
- LANGFUSE_INIT_ORG_NAME=${LANGFUSE_INIT_ORG_NAME:-}
- LANGFUSE_INIT_PROJECT_ID=${LANGFUSE_INIT_PROJECT_ID:-}
- LANGFUSE_INIT_PROJECT_NAME=${LANGFUSE_INIT_PROJECT_NAME:-}
- LANGFUSE_INIT_PROJECT_PUBLIC_KEY=${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
- LANGFUSE_INIT_PROJECT_SECRET_KEY=${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
- LANGFUSE_INIT_USER_EMAIL=${LANGFUSE_INIT_USER_EMAIL:-}
- LANGFUSE_INIT_USER_NAME=${LANGFUSE_INIT_USER_NAME:-}
- LANGFUSE_INIT_USER_PASSWORD=${LANGFUSE_INIT_USER_PASSWORD:-}
restart: always
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:3000/api/public/health"]
@@ -55,17 +36,26 @@ services:
timeout: 10s
retries: 3
langfuse-worker:
worker:
build:
dockerfile: ./worker/Dockerfile
context: .
args:
- NEXT_PUBLIC_LANGFUSE_CLOUD_REGION=${NEXT_PUBLIC_LANGFUSE_CLOUD_REGION}
depends_on: *langfuse-depends-on
depends_on:
- db
- redis
ports:
- "3030:3030"
environment:
<<: *langfuse-web-env
- DATABASE_URL=postgresql://postgres:postgres@db:5432/postgres
- NEXTAUTH_SECRET=mysecret
- TELEMETRY_ENABLED=${TELEMETRY_ENABLED:-true}
- LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES=${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-false}
- PORT=${PORT:-3030}
- REDIS_HOST=${REDIS_HOST:-redis}
- REDIS_PORT=${REDIS_PORT:-6379}
- REDIS_AUTH=${REDIS_AUTH:-myredissecret}
restart: always
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:3030/api/health"]
@@ -73,85 +63,26 @@ services:
timeout: 10s
retries: 3
clickhouse:
image: clickhouse/clickhouse-server
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
healthcheck:
test: wget --no-verbose --tries=1 --spider http://localhost:8123/ping || exit 1
interval: 5s
timeout: 5s
retries: 10
start_period: 1s
minio:
image: minio/minio
container_name: minio
entrypoint: sh
# create the 'langfuse' bucket before starting the service
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
environment:
MINIO_ROOT_USER: minio
MINIO_ROOT_PASSWORD: miniosecret
ports:
- "9090:9000"
- "9091:9001"
volumes:
- langfuse_minio_data:/data
healthcheck:
test: ["CMD", "mc", "ready", "local"]
interval: 1s
timeout: 5s
retries: 5
start_period: 1s
redis:
image: redis:7
restart: always
image: redis:7.2.4
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
restart: always
ports:
- 6379:6379
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 10s
retries: 10
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
db:
image: postgres
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
environment:
POSTGRES_USER: postgres
POSTGRES_PASSWORD: postgres
POSTGRES_DB: postgres
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=postgres
- POSTGRES_DB=postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
- database_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_minio_data:
database_data:
driver: local
+145
View File
@@ -0,0 +1,145 @@
services:
langfuse-worker:
image: langfuse/langfuse-worker:latest
depends_on: &langfuse-depends-on
postgres:
condition: service_healthy
minio:
condition: service_healthy
redis:
condition: service_healthy
ports:
- "3030:3030"
environment: &langfuse-worker-env
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres
SALT: "mysalt"
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # generate via `openssl rand -hex 32`
TELEMETRY_ENABLED: ${TELEMETRY_ENABLED:-true}
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-true}
LANGFUSE_ASYNC_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_INGESTION_PROCESSING:-true}
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING:-true}
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE: ${LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE:-true}
LANGFUSE_READ_FROM_POSTGRES_ONLY: ${LANGFUSE_READ_FROM_POSTGRES_ONLY:-false}
LANGFUSE_RETURN_FROM_CLICKHOUSE: ${LANGFUSE_RETURN_FROM_CLICKHOUSE:-true}
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: ${LANGFUSE_S3_EVENT_UPLOAD_ENABLED:-true}
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED: ${LANGFUSE_S3_MEDIA_UPLOAD_ENABLED:-true}
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET: ${LANGFUSE_S3_MEDIA_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_MEDIA_UPLOAD_REGION: ${LANGFUSE_S3_MEDIA_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT: ${LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX: ${LANGFUSE_S3_MEDIA_UPLOAD_PREFIX:-media/}
REDIS_HOST: ${REDIS_HOST:-redis}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
langfuse-web:
image: langfuse/langfuse:latest
depends_on: *langfuse-depends-on
ports:
- "3000:3000"
environment:
<<: *langfuse-worker-env
NEXTAUTH_URL: http://localhost:3000
NEXTAUTH_SECRET: mysecret
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
LANGFUSE_INIT_PROJECT_ID: ${LANGFUSE_INIT_PROJECT_ID:-}
LANGFUSE_INIT_PROJECT_NAME: ${LANGFUSE_INIT_PROJECT_NAME:-}
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: ${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
LANGFUSE_INIT_PROJECT_SECRET_KEY: ${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
LANGFUSE_INIT_USER_EMAIL: ${LANGFUSE_INIT_USER_EMAIL:-}
LANGFUSE_INIT_USER_NAME: ${LANGFUSE_INIT_USER_NAME:-}
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
clickhouse:
image: clickhouse/clickhouse-server
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
depends_on:
- postgres
minio:
image: minio/minio
container_name: minio
entrypoint: sh
# create the 'langfuse' bucket before starting the service
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
environment:
MINIO_ROOT_USER: minio
MINIO_ROOT_PASSWORD: miniosecret
ports:
- "9090:9000"
- "9091:9001"
volumes:
- langfuse_minio_data:/data
healthcheck:
test: ["CMD", "mc", "ready", "local"]
interval: 1s
timeout: 5s
retries: 5
start_period: 1s
redis:
image: redis:7
restart: always
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
ports:
- 6379:6379
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 10s
retries: 10
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
environment:
POSTGRES_USER: postgres
POSTGRES_PASSWORD: postgres
POSTGRES_DB: postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_minio_data:
driver: local
+27 -131
View File
@@ -1,129 +1,31 @@
services:
langfuse-worker:
image: langfuse/langfuse-worker:3
restart: always
depends_on: &langfuse-depends-on
postgres:
langfuse-server:
image: langfuse/langfuse:2
depends_on:
db:
condition: service_healthy
minio:
condition: service_healthy
redis:
condition: service_healthy
clickhouse:
condition: service_healthy
ports:
- "3030:3030"
environment: &langfuse-worker-env
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres
SALT: "mysalt"
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # generate via `openssl rand -hex 32`
TELEMETRY_ENABLED: ${TELEMETRY_ENABLED:-true}
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-true}
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-auto}
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET: ${LANGFUSE_S3_MEDIA_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_MEDIA_UPLOAD_REGION: ${LANGFUSE_S3_MEDIA_UPLOAD_REGION:-auto}
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT: ${LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX: ${LANGFUSE_S3_MEDIA_UPLOAD_PREFIX:-media/}
LANGFUSE_INGESTION_QUEUE_DELAY_MS: ${LANGFUSE_INGESTION_QUEUE_DELAY_MS:-}
LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS: ${LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS:-}
REDIS_HOST: ${REDIS_HOST:-redis}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
langfuse-web:
image: langfuse/langfuse:3
restart: always
depends_on: *langfuse-depends-on
ports:
- "3000:3000"
environment:
<<: *langfuse-worker-env
NEXTAUTH_URL: http://localhost:3000
NEXTAUTH_SECRET: mysecret
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
LANGFUSE_INIT_PROJECT_ID: ${LANGFUSE_INIT_PROJECT_ID:-}
LANGFUSE_INIT_PROJECT_NAME: ${LANGFUSE_INIT_PROJECT_NAME:-}
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: ${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
LANGFUSE_INIT_PROJECT_SECRET_KEY: ${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
LANGFUSE_INIT_USER_EMAIL: ${LANGFUSE_INIT_USER_EMAIL:-}
LANGFUSE_INIT_USER_NAME: ${LANGFUSE_INIT_USER_NAME:-}
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
- DATABASE_URL=postgresql://postgres:postgres@db:5432/postgres
- NEXTAUTH_SECRET=mysecret
- SALT=mysalt
- ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000 # generate via `openssl rand -hex 32`
- NEXTAUTH_URL=http://localhost:3000
- TELEMETRY_ENABLED=${TELEMETRY_ENABLED:-true}
- LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES=${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-false}
- LANGFUSE_INIT_ORG_ID=${LANGFUSE_INIT_ORG_ID:-}
- LANGFUSE_INIT_ORG_NAME=${LANGFUSE_INIT_ORG_NAME:-}
- LANGFUSE_INIT_PROJECT_ID=${LANGFUSE_INIT_PROJECT_ID:-}
- LANGFUSE_INIT_PROJECT_NAME=${LANGFUSE_INIT_PROJECT_NAME:-}
- LANGFUSE_INIT_PROJECT_PUBLIC_KEY=${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
- LANGFUSE_INIT_PROJECT_SECRET_KEY=${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
- LANGFUSE_INIT_USER_EMAIL=${LANGFUSE_INIT_USER_EMAIL:-}
- LANGFUSE_INIT_USER_NAME=${LANGFUSE_INIT_USER_NAME:-}
- LANGFUSE_INIT_USER_PASSWORD=${LANGFUSE_INIT_USER_PASSWORD:-}
clickhouse:
image: clickhouse/clickhouse-server
restart: always
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
healthcheck:
test: wget --no-verbose --tries=1 --spider http://localhost:8123/ping || exit 1
interval: 5s
timeout: 5s
retries: 10
start_period: 1s
minio:
image: minio/minio
restart: always
container_name: minio
entrypoint: sh
# create the 'langfuse' bucket before starting the service
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
environment:
MINIO_ROOT_USER: minio
MINIO_ROOT_PASSWORD: miniosecret
ports:
- "9090:9000"
- "9091:9001"
volumes:
- langfuse_minio_data:/data
healthcheck:
test: ["CMD", "mc", "ready", "local"]
interval: 1s
timeout: 5s
retries: 5
start_period: 1s
redis:
image: redis:7
restart: always
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
ports:
- 6379:6379
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 10s
retries: 10
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
db:
image: postgres
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
@@ -131,20 +33,14 @@ services:
timeout: 3s
retries: 10
environment:
POSTGRES_USER: postgres
POSTGRES_PASSWORD: postgres
POSTGRES_DB: postgres
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=postgres
- POSTGRES_DB=postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
- database_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_minio_data:
database_data:
driver: local
+1 -2
View File
@@ -48,8 +48,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.7",
"nanoid": "^3.3.8"
"jsonpath-plus": "10.2.0"
}
}
}
+5 -2
View File
@@ -1,9 +1,12 @@
import { z } from "zod";
import { removeEmptyEnvVariables } from "@langfuse/shared";
import { env as sharedEnv, removeEmptyEnvVariables } from "@langfuse/shared";
const EnvSchema = z.object({
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION: z.string().optional(),
LANGFUSE_EE_LICENSE_KEY: z.string().optional(),
});
export const env = EnvSchema.parse(removeEmptyEnvVariables(process.env));
export const env = {
...sharedEnv,
...EnvSchema.parse(removeEmptyEnvVariables(process.env)),
};
+1 -1
View File
@@ -12,7 +12,7 @@ groups:
# namespaceExport: Langfuse
# allowCustomFetcher: true
- name: fernapi/fern-openapi
version: 0.1.7
version: 0.0.26
output:
location: local-file-system
path: ../../../web/public/generated/api-client
+1 -1
View File
@@ -58,7 +58,7 @@ types:
docs: The id of the object to attach the comment to. If this does not reference a valid existing object, an error will be thrown.
content:
type: string
docs: The content of the comment. May include markdown. Currently limited to 3000 characters.
docs: The content of the comment. May include markdown. Currently limited to 500 characters.
authorUserId:
type: optional<string>
docs: The id of the user who created the comment.
+7 -13
View File
@@ -126,7 +126,7 @@ types:
docs: The output data of the observation
usage:
type: optional<Usage>
docs: (Deprecated. Use usageDetails and costDetails instead.) The usage data of the observation
docs: The usage data of the observation
level:
type: ObservationLevel
docs: The level of the observation
@@ -139,12 +139,6 @@ types:
promptId:
type: optional<string>
docs: The prompt ID associated with the observation
usageDetails:
type: optional<map<string, integer>>
docs: The usage details of the observation. Key is the name of the usage metric, value is the number of units consumed. The total key is the sum of all (non-total) usage metrics or the total value ingested.
costDetails:
type: optional<map<string, double>>
docs: The cost details of the observation. Key is the name of the cost metric, value is the cost in USD. The total key is the sum of all (non-total) cost metrics or the total value ingested.
ObservationsView:
extends: Observation
@@ -169,13 +163,13 @@ types:
docs: The total price in USD.
calculatedInputCost:
type: optional<double>
docs: (Deprecated. Use usageDetails and costDetails instead.) The calculated cost of the input in USD
docs: The calculated cost of the input in USD
calculatedOutputCost:
type: optional<double>
docs: (Deprecated. Use usageDetails and costDetails instead.) The calculated cost of the output in USD
docs: The calculated cost of the output in USD
calculatedTotalCost:
type: optional<double>
docs: (Deprecated. Use usageDetails and costDetails instead.) The calculated total cost in USD
docs: The calculated total cost in USD
latency:
type: optional<double>
docs: The latency in seconds.
@@ -184,7 +178,7 @@ types:
docs: The time to the first token in seconds
Usage:
docs: (Deprecated. Use usageDetails and costDetails instead.) Standard interface for usage and cost
docs: Standard interface for usage and cost
properties:
input:
docs: Number of input units (e.g. tokens)
@@ -378,10 +372,10 @@ types:
type: string
startDate:
docs: Apply only to generations which are newer than this ISO date.
type: optional<datetime>
type: optional<date>
unit:
docs: Unit used by this model.
type: optional<ModelUsageUnit>
type: ModelUsageUnit
inputPrice:
docs: Price (USD) per input unit
type: optional<double>
-19
View File
@@ -11,7 +11,6 @@ service:
Batched ingestion for Langfuse Tracing. If you want to use tracing via the API, such as to build your own Langfuse client implementation, this is the only API route you need to implement.
Notes:
- Introduction to data model: https://langfuse.com/docs/tracing-data-model
- Batch sizes are limited to 3.5 MB in total. You need to adjust the number of events per batch accordingly.
- The API does not return a 4xx status code for input errors. Instead, it responds with a 207 status code, which includes a list of the encountered errors.
method: POST
@@ -127,8 +126,6 @@ types:
model: optional<string>
modelParameters: optional<map<string, commons.MapValue>>
usage: optional<IngestionUsage>
usageDetails: optional<UsageDetails>
costDetails: optional<map<string, double>>
promptName: optional<string>
promptVersion: optional<integer>
@@ -140,8 +137,6 @@ types:
modelParameters: optional<map<string, commons.MapValue>>
usage: optional<IngestionUsage>
promptName: optional<string>
usageDetails: optional<UsageDetails>
costDetails: optional<map<string, double>>
promptVersion: optional<integer>
ObservationBody:
@@ -317,17 +312,3 @@ types:
properties:
successes: list<IngestionSuccess>
errors: list<IngestionError>
OpenAIUsageSchema:
properties:
prompt_tokens: integer
completion_tokens: integer
total_tokens: integer
prompt_tokens_details: optional<map<string, integer>>
completion_tokens_details: optional<map<string, integer>>
UsageDetails:
discriminated: false
union:
- map<string, integer>
- OpenAIUsageSchema
+1 -59
View File
@@ -99,63 +99,5 @@ types:
docs: The unique langfuse identifier of a media record
MediaContentType:
enum:
- value: image/png
name: IMAGE_PNG
- value: image/jpeg
name: IMAGE_JPEG
- value: image/jpg
name: IMAGE_JPG
- value: image/webp
name: IMAGE_WEBP
- value: image/gif
name: IMAGE_GIF
- value: image/svg+xml
name: IMAGE_SVG_XML
- value: image/tiff
name: IMAGE_TIFF
- value: image/bmp
name: IMAGE_BMP
- value: audio/mpeg
name: AUDIO_MPEG
- value: audio/mp3
name: AUDIO_MP3
- value: audio/wav
name: AUDIO_WAV
- value: audio/ogg
name: AUDIO_OGG
- value: audio/oga
name: AUDIO_OGA
- value: audio/aac
name: AUDIO_AAC
- value: audio/mp4
name: AUDIO_MP4
- value: audio/flac
name: AUDIO_FLAC
- value: video/mp4
name: VIDEO_MP4
- value: video/webm
name: VIDEO_WEBM
- value: text/plain
name: TEXT_PLAIN
- value: text/html
name: TEXT_HTML
- value: text/css
name: TEXT_CSS
- value: text/csv
name: TEXT_CSV
- value: application/pdf
name: APPLICATION_PDF
- value: application/msword
name: APPLICATION_MSWORD
- value: application/vnd.ms-excel
name: APPLICATION_MS_EXCEL
- value: application/zip
name: APPLICATION_ZIP
- value: application/json
name: APPLICATION_JSON
- value: application/xml
name: APPLICATION_XML
- value: application/octet-stream
name: APPLICATION_OCTET_STREAM
type: literal<"image/png","image/jpeg","image/jpg","image/webp","audio/mpeg","audio/mp3","audio/wav","text/plain","application/pdf">
docs: The MIME type of the media record
+1 -1
View File
@@ -58,7 +58,7 @@ types:
type: optional<datetime>
unit:
docs: Unit used by this model.
type: optional<commons.ModelUsageUnit>
type: commons.ModelUsageUnit
inputPrice:
docs: Price (USD) per input unit
type: optional<double>
@@ -1,27 +0,0 @@
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
imports:
prompts: ./prompts.yml
pagination: ./utils/pagination.yml
service:
auth: true
base-path: /api/public/v2
endpoints:
update:
docs: Update labels for a specific prompt version
method: PATCH
path: /prompts/{name}/versions/{version}
path-parameters:
name:
type: string
docs: The name of the prompt
version:
type: integer
docs: Version of the prompt to update
request:
name: UpdatePromptRequest
body:
properties:
newLabels:
type: list<string>
docs: New labels for the prompt version. Labels are unique across versions. The "latest" label is reserved and managed by Langfuse.
response: prompts.Prompt
-9
View File
@@ -90,9 +90,6 @@ types:
tags:
type: optional<list<string>>
docs: List of tags to apply to all versions of this prompt.
commitMessage:
type: optional<string>
docs: Commit message for this prompt version.
CreateTextPromptRequest:
properties:
@@ -105,9 +102,6 @@ types:
tags:
type: optional<list<string>>
docs: List of tags to apply to all versions of this prompt.
commitMessage:
type: optional<string>
docs: Commit message for this prompt version.
Prompt:
union:
@@ -125,9 +119,6 @@ types:
tags:
type: list<string>
docs: List of tags. Used to filter via UI and API. The same across versions of a prompt.
commitMessage:
type: optional<string>
docs: Commit message for this prompt version.
ChatMessage:
properties:
+1 -1
View File
@@ -59,7 +59,7 @@ service:
type: optional<commons.ScoreDataType>
docs: Retrieve only scores with a specific dataType.
traceTags:
type: optional<string>
type: optional<list<string>>
allow-multiple: true
docs: Only scores linked to traces that include all of these tags will be returned.
response: GetScoresResponse
+1 -1
View File
@@ -3,7 +3,7 @@ groups:
local:
generators:
- name: fernapi/fern-openapi
version: 0.1.7
version: 0.0.31
output:
location: local-file-system
path: ../../../web/public/generated/api
+2 -3
View File
@@ -1,5 +1,4 @@
{
"organization": "langfuse",
"organization": "finto",
"version": "0.43.7"
}
}
+5 -4
View File
@@ -1,6 +1,6 @@
{
"name": "langfuse",
"version": "3.20.0",
"version": "2.95.2",
"author": "engineering@langfuse.com",
"license": "MIT",
"private": true,
@@ -25,6 +25,7 @@
"dev": "turbo run dev",
"lint": "turbo run lint",
"test": "turbo run test",
"models:migrate": "turbo run models:migrate",
"release": "dotenv -e ../.env -- release-it",
"prepare": "husky"
},
@@ -80,15 +81,15 @@
}
}
},
"packageManager": "pnpm@9.5.0",
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.7",
"jsonpath-plus": "10.2.0",
"nanoid": "^3.3.8",
"katex": "^0.16.21"
},
"patchedDependencies": {
"next-auth@4.24.11": "patches/next-auth@4.24.11.patch"
}
}
},
"packageManager": "pnpm@9.5.0"
}
@@ -1 +0,0 @@
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_user_id;
@@ -1,2 +0,0 @@
ALTER TABLE traces ON CLUSTER default ADD INDEX IF NOT EXISTS idx_user_id user_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces ON CLUSTER default MATERIALIZE INDEX IF EXISTS idx_user_id;
@@ -1 +0,0 @@
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_user_id;
@@ -1,2 +0,0 @@
ALTER TABLE traces ADD INDEX IF NOT EXISTS idx_user_id user_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces MATERIALIZE INDEX IF EXISTS idx_user_id;
+12 -22
View File
@@ -18,33 +18,23 @@ then
exit 1
fi
# Ensure CLICKHOUSE_DB is set
if [ -z "${CLICKHOUSE_DB}" ]; then
export CLICKHOUSE_DB="default"
fi
# Ensure CLICKHOUSE_CLUSTER_NAME is set
if [ -z "${CLICKHOUSE_CLUSTER_NAME}" ]; then
export CLICKHOUSE_CLUSTER_NAME="default"
fi
# Construct the database URL
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "false" ] ; then
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
fi
+2 -7
View File
@@ -12,16 +12,11 @@ then
exit 1
fi
# Ensure CLICKHOUSE_DB is set
if [ -z "${CLICKHOUSE_DB}" ]; then
export CLICKHOUSE_DB="default"
fi
# Construct the database URL
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-migrations-table-engine=MergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the drop command
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" drop
+1 -1
View File
@@ -2,7 +2,7 @@ import {
clickhouseClient,
ObservationRecordReadType,
} from "@langfuse/shared/src/server";
import { prisma } from "../../src/db";
import { Prisma, prisma } from "../../src/db";
import { redis } from "@langfuse/shared/src/server";
import { prepareClickhouse } from "../../scripts/prepareClickhouse";
import { createDatasets } from "../../prisma/seed";
+12 -22
View File
@@ -18,33 +18,23 @@ then
exit 1
fi
# Ensure CLICKHOUSE_DB is set
if [ -z "${CLICKHOUSE_DB}" ]; then
export CLICKHOUSE_DB="default"
fi
# Ensure CLICKHOUSE_CLUSTER_NAME is set
if [ -z "${CLICKHOUSE_CLUSTER_NAME}" ]; then
export CLICKHOUSE_CLUSTER_NAME="default"
fi
# Construct the database URL
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "false" ] ; then
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" up
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" up
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" up
fi
+3 -5
View File
@@ -33,7 +33,7 @@
"scripts": {
"build": "tsc",
"dev": "tsc --watch",
"lint": "eslint . --ext .js,.jsx,.ts,.tsx --max-warnings 110",
"lint": "eslint . --ext .js,.jsx,.ts,.tsx",
"lint:fix": "eslint . --ext .js,.jsx,.ts,.tsx --fix",
"db:migrate": "DISABLE_ERD=false dotenv -e ../../.env -- npx prisma migrate dev",
"db:push": "DISABLE_ERD=false dotenv -e ../../.env -- npx prisma db push",
@@ -64,7 +64,6 @@
"@langchain/anthropic": "^0.3.8",
"@langchain/aws": "^0.1.2",
"@langchain/core": "^0.3.18",
"@langchain/google-vertexai": "^0.1.3",
"@langchain/openai": "^0.3.14",
"@opentelemetry/api": ">=1.0.0 <1.10.0",
"@prisma/client": "^5.22.0",
@@ -73,7 +72,7 @@
"@types/bcryptjs": "^2.4.6",
"axios": "^1.7.7",
"bcryptjs": "^2.4.3",
"bullmq": "^5.34.10",
"bullmq": "^5.12.10",
"dd-trace": "^5.23.1",
"decimal.js": "^10.4.3",
"exponential-backoff": "^3.1.1",
@@ -122,8 +121,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.7",
"nanoid": "^3.3.8"
"jsonpath-plus": "10.0.7"
}
}
}
-25
View File
@@ -170,17 +170,6 @@ export type BatchExport = {
url: string | null;
log: string | null;
};
export type BillingMeterBackup = {
stripe_customer_id: string;
meter_id: string;
start_time: Timestamp;
end_time: Timestamp;
aggregated_value: number;
event_name: string;
org_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type Comment = {
id: string;
project_id: string;
@@ -304,8 +293,6 @@ export type LlmApiKeys = {
base_url: string | null;
custom_models: Generated<string[]>;
with_default_models: Generated<boolean>;
extra_headers: string | null;
extra_header_keys: Generated<string[]>;
config: unknown | null;
project_id: string;
};
@@ -466,9 +453,7 @@ export type Project = {
org_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
deleted_at: Timestamp | null;
name: string;
retention_days: number | null;
};
export type ProjectMembership = {
org_membership_id: string;
@@ -492,14 +477,6 @@ export type Prompt = {
config: Generated<unknown>;
tags: Generated<string[]>;
labels: Generated<string[]>;
commit_message: string | null;
};
export type QueueBackUp = {
id: string;
project_id: string | null;
queue_name: string;
content: unknown;
created_at: Generated<Timestamp>;
};
export type Score = {
id: string;
@@ -626,7 +603,6 @@ export type DB = {
audit_logs: AuditLog;
background_migrations: BackgroundMigration;
batch_exports: BatchExport;
billing_meter_backups: BillingMeterBackup;
comments: Comment;
cron_jobs: CronJobs;
dataset_items: DatasetItem;
@@ -651,7 +627,6 @@ export type DB = {
project_memberships: ProjectMembership;
projects: Project;
prompts: Prompt;
queue_backups: QueueBackUp;
score_configs: ScoreConfig;
scores: Score;
Session: Session;
@@ -1,2 +0,0 @@
-- AlterTable
ALTER TABLE "projects" ADD COLUMN "deleted_at" TIMESTAMP(3);
@@ -1,5 +0,0 @@
-- DropForeignKey
ALTER TABLE "job_executions" DROP CONSTRAINT "job_executions_job_output_score_id_fkey";
-- DropForeignKey
ALTER TABLE "traces" DROP CONSTRAINT "traces_session_id_project_id_fkey";
@@ -1,13 +0,0 @@
-- CreateTable
CREATE TABLE "queue_backups" (
"id" TEXT NOT NULL,
"project_id" TEXT,
"queue_name" TEXT NOT NULL,
"content" JSONB NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
CONSTRAINT "queue_backups_pkey" PRIMARY KEY ("id")
);
-- AddForeignKey
ALTER TABLE "traces" ADD CONSTRAINT "traces_session_id_project_id_fkey" FOREIGN KEY ("session_id", "project_id") REFERENCES "trace_sessions"("id", "project_id") ON DELETE RESTRICT ON UPDATE CASCADE;
@@ -1,2 +0,0 @@
-- DropForeignKey
ALTER TABLE "traces" DROP CONSTRAINT "traces_session_id_project_id_fkey";
@@ -1,21 +0,0 @@
-- CreateTable
CREATE TABLE "billing_meter_backups" (
"stripe_customer_id" TEXT NOT NULL,
"meter_id" TEXT NOT NULL,
"start_time" TIMESTAMP(3) NOT NULL,
"end_time" TIMESTAMP(3) NOT NULL,
"aggregated_value" INTEGER NOT NULL,
"event_name" TEXT NOT NULL,
"org_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP
);
-- CreateIndex
CREATE INDEX "billing_meter_backups_stripe_customer_id_meter_id_start_tim_idx" ON "billing_meter_backups"("stripe_customer_id", "meter_id", "start_time", "end_time");
-- CreateIndex
CREATE UNIQUE INDEX "billing_meter_backups_stripe_customer_id_meter_id_start_tim_key" ON "billing_meter_backups"("stripe_customer_id", "meter_id", "start_time", "end_time");
@@ -1,3 +0,0 @@
ALTER TABLE "llm_api_keys"
ADD COLUMN "extra_headers" TEXT,
ADD COLUMN "extra_header_keys" TEXT[] NOT NULL DEFAULT '{}'::TEXT[];
@@ -1,3 +0,0 @@
-- AlterTable
ALTER TABLE "projects"
ADD COLUMN "retention_days" INTEGER;
@@ -1 +0,0 @@
UPDATE "llm_api_keys" SET adapter = 'google-vertex-ai' WHERE adapter = 'vertex-ai';
@@ -1,2 +0,0 @@
-- AlterTable
ALTER TABLE "prompts" ADD COLUMN "commit_message" TEXT;
+39 -74
View File
@@ -113,9 +113,7 @@ model Project {
orgId String @map("org_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
deletedAt DateTime? @map("deleted_at")
name String
retentionDays Int? @map("retention_days")
projectMembers ProjectMembership[]
organization Organization @relation(fields: [orgId], references: [id], onUpdate: Cascade, onDelete: Cascade)
traces Trace[]
@@ -193,8 +191,6 @@ model LlmApiKeys {
baseURL String? @map("base_url")
customModels String[] @default([]) @map("custom_models")
withDefaultModels Boolean @default(true) @map("with_default_models")
extraHeaders String? @map("extra_headers")
extraHeaderKeys String[] @default([]) @map("extra_header_keys")
config Json?
projectId String @map("project_id")
@@ -275,6 +271,7 @@ model TraceSession {
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
bookmarked Boolean @default(false)
public Boolean @default(false)
traces Trace[]
@@id([id, projectId])
@@index([projectId])
@@ -286,24 +283,25 @@ model TraceSession {
// Update TraceView below when making changes to this model!
model Trace {
id String @id @default(cuid())
externalId String? @map("external_id")
timestamp DateTime @default(now())
id String @id @default(cuid())
externalId String? @map("external_id")
timestamp DateTime @default(now())
name String?
userId String? @map("user_id")
userId String? @map("user_id")
metadata Json?
release String?
version String?
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
public Boolean @default(false)
bookmarked Boolean @default(false)
tags String[] @default([])
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
public Boolean @default(false)
bookmarked Boolean @default(false)
tags String[] @default([])
input Json?
output Json?
sessionId String? @map("session_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
sessionId String? @map("session_id")
session TraceSession? @relation(fields: [sessionId, projectId], references: [id, projectId])
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
@@index([projectId, timestamp])
@@index([sessionId])
@@ -470,24 +468,25 @@ enum ObservationLevel {
}
model Score {
id String @id @default(cuid())
timestamp DateTime @default(now())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
id String @id @default(cuid())
timestamp DateTime @default(now())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
name String
value Float? // always defined if data type is NUMERIC or BOOLEAN, optional for CATEGORICAL
source ScoreSource
authorUserId String? @map("author_user_id")
authorUserId String? @map("author_user_id")
comment String?
traceId String @map("trace_id")
observationId String? @map("observation_id")
configId String? @map("config_id")
stringValue String? @map("string_value") // always defined if data type is CATEGORICAL or BOOLEAN, null for NUMERIC
queueId String? @map("queue_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
dataType ScoreDataType @default(NUMERIC) @map("data_type")
scoreConfig ScoreConfig? @relation(fields: [configId], references: [id], onDelete: SetNull)
traceId String @map("trace_id")
observationId String? @map("observation_id")
configId String? @map("config_id")
stringValue String? @map("string_value") // always defined if data type is CATEGORICAL or BOOLEAN, null for NUMERIC
queueId String? @map("queue_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
dataType ScoreDataType @default(NUMERIC) @map("data_type")
JobExecution JobExecution[]
scoreConfig ScoreConfig? @relation(fields: [configId], references: [id], onDelete: SetNull)
@@unique([id, projectId]) // used for upserts via prisma
@@index(timestamp)
@@ -738,15 +737,14 @@ model Prompt {
createdBy String @map("created_by")
prompt Json
name String
version Int
type String @default("text")
isActive Boolean? @map("is_active") // Deprecated. To be removed once 'production' labels work as expected.
config Json @default("{}") @db.Json
tags String[] @default([])
labels String[] @default([])
commitMessage String? @map("commit_message")
prompt Json
name String
version Int
type String @default("text")
isActive Boolean? @map("is_active") // Deprecated. To be removed once 'production' labels work as expected.
config Json @default("{}") @db.Json
tags String[] @default([])
labels String[] @default([])
@@unique([projectId, name, version])
@@index([projectId, id])
@@ -902,9 +900,10 @@ model JobExecution {
jobInputObservationId String? @map("job_input_observation_id") // no fk constraint - observations in ClickHouse, deletion handled via project cascade
jobInputDatasetItemId String? @map("job_input_dataset_item_id") // no fk constraint - job execution sensible standalone
jobInputDatasetItemId String? @map("job_input_dataset_item_id") // no fk constraint - job execution sensible standalone
jobOutputScoreId String? @map("job_output_score_id")
score Score? @relation(fields: [jobOutputScoreId], references: [id], onDelete: SetNull) // job remains when scores are deleted
@@index([projectId, status])
@@index([projectId, id])
@@ -1019,37 +1018,3 @@ model ObservationMedia {
@@index([projectId, observationId])
@@map("observation_media")
}
model QueueBackUp {
id String @id @default(cuid())
projectId String? @map("project_id")
queueName String @map("queue_name")
content Json
createdAt DateTime @default(now()) @map("created_at")
@@map("queue_backups")
}
model BillingMeterBackup {
// unique
stripeCustomerId String @map("stripe_customer_id")
meterId String @map("meter_id")
startTime DateTime @map("start_time")
endTime DateTime @map("end_time")
// value
aggregatedValue Int @map("aggregated_value")
// labels
eventName String @map("event_name")
orgId String @map("org_id")
// ts
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
@@unique([stripeCustomerId, meterId, startTime, endTime])
@@index([stripeCustomerId, meterId, startTime, endTime])
@@map("billing_meter_backups")
}
-6
View File
@@ -450,9 +450,6 @@ export async function createDatasets(
Math.random() > 0.3
? observations[Math.floor(Math.random() * observations.length)]
: undefined;
if (!sourceObservation) {
continue;
}
const datasetItem = await prisma.datasetItem.create({
data: {
projectId,
@@ -513,9 +510,6 @@ export async function createDatasets(
Math.floor(Math.random() * relevantObservations.length)
];
if (!observation) {
continue;
}
await prisma.datasetRunItems.create({
data: {
projectId,
+15 -31
View File
@@ -96,37 +96,21 @@ export const prepareClickhouse = async (
'version' AS version,
repeat('input', toInt64(randExponential(1 / 100))) AS input,
repeat('output', toInt64(randExponential(1 / 100))) AS output,
if("type" = 'GENERATION',
case
when number % 2 = 0 then 'claude-3-haiku-20240307'
else 'gpt-4'
end,
NULL) as provided_model_name,
if("type" = 'GENERATION',
case
when number % 2 = 0 then 'cltr0w45b000008k1407o9qv1'
else 'clrntkjgy000f08jx79v9g1xj'
end,
NULL) as internal_model_id,
if("type" = 'GENERATION',
'{"temperature": 0.7, "max_tokens": 150}',
'{}') AS model_parameters,
if("type" = 'GENERATION',
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))),
map()) AS provided_usage_details,
if("type" = 'GENERATION',
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))),
map()) AS usage_details,
if("type" = 'GENERATION',
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)),
map()) AS provided_cost_details,
if("type" = 'GENERATION',
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)),
map()) AS cost_details,
if("type" = 'GENERATION',
toDecimal64(randUniform(0, 2000), 12),
NULL) AS total_cost,
addMilliseconds(start_time, if(rand() < 0.6, floor(randUniform(0, 500)), floor(randUniform(0, 600)))) AS completion_start_time,
case
when number % 2 = 0 then 'claude-3-haiku-20240307'
else 'gpt-4'
end as provided_model_name,
case
when number % 2 = 0 then 'cltr0w45b000008k1407o9qv1'
else 'clrntkjgy000f08jx79v9g1xj'
end as internal_model_id,
'{"temperature": 0.7, "max_tokens": 150}' AS model_parameters,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS provided_usage_details,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS usage_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS provided_cost_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS cost_details,
toDecimal64(randUniform(0, 2000), 12) AS total_cost,
start_time AS completion_start_time,
array(${SEED_PROMPTS.map((p) => `concat('${p.id}',project_id)`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_id,
+12 -21
View File
@@ -5,7 +5,6 @@ const EnvSchema = z.object({
NODE_ENV: z
.enum(["development", "test", "production"])
.default("development"),
NEXTAUTH_URL: z.string().url().optional(),
REDIS_HOST: z.string().nullish(),
REDIS_PORT: z.coerce
.number({
@@ -28,12 +27,15 @@ const EnvSchema = z.object({
.optional(),
LANGFUSE_CACHE_PROMPT_ENABLED: z.enum(["true", "false"]).default("false"),
LANGFUSE_CACHE_PROMPT_TTL_SECONDS: z.coerce.number().default(60 * 60),
CLICKHOUSE_URL: z.string().url(),
CLICKHOUSE_CLUSTER_NAME: z.string().default("default"),
CLICKHOUSE_DB: z.string().default("default"),
CLICKHOUSE_USER: z.string(),
CLICKHOUSE_PASSWORD: z.string(),
CLICKHOUSE_URL: z.string().url().optional(),
CLICKHOUSE_USER: z.string().optional(),
CLICKHOUSE_PASSWORD: z.string().optional(),
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_ASYNC_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_INGESTION_QUEUE_DELAY_MS: z.coerce
.number()
.nonnegative()
@@ -46,9 +48,8 @@ const EnvSchema = z.object({
ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: z.string({
required_error: "Langfuse requires a bucket name for S3 Event Uploads.",
}),
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: z.enum(["true", "false"]).default("false"),
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: z.string().default(""),
LANGFUSE_S3_EVENT_UPLOAD_REGION: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: z.string().optional(),
@@ -59,16 +60,6 @@ const EnvSchema = z.object({
.default("false"),
LANGFUSE_USE_AZURE_BLOB: z.enum(["true", "false"]).default("false"),
STRIPE_SECRET_KEY: z.string().optional(),
LANGFUSE_S3_CORE_DATA_EXPORT_IS_ENABLED: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_POSTGRES_METERING_DATA_EXPORT_IS_ENABLED: z
.enum(["true", "false"])
.default("false"),
});
export const env: z.infer<typeof EnvSchema> =
process.env.DOCKER_BUILD === "1"
? (process.env as any)
: EnvSchema.parse(removeEmptyEnvVariables(process.env));
export const env = EnvSchema.parse(removeEmptyEnvVariables(process.env));
+2 -2
View File
@@ -1,7 +1,7 @@
import { BaseError } from "./BaseError";
export class ApiError extends BaseError {
constructor(description = "Api call failed", status = 500) {
super("ApiError", status, description, true);
constructor(description = "Api call failed") {
super("ApiError", 500, description, true);
}
}
@@ -1,7 +1,5 @@
import { z } from "zod";
const MAX_COMMENT_LENGTH = 3000;
const COMMENT_OBJECT_TYPES = [
"TRACE",
"OBSERVATION",
@@ -11,7 +9,7 @@ const COMMENT_OBJECT_TYPES = [
export const CreateCommentData = z.object({
projectId: z.string(),
content: z.string().trim().min(1).max(MAX_COMMENT_LENGTH),
content: z.string().trim().min(1).max(500),
objectId: z.string(),
objectType: z.enum(COMMENT_OBJECT_TYPES),
});
@@ -3,17 +3,15 @@ export const planLabels = {
"cloud:hobby": "Hobby",
"cloud:pro": "Pro",
"cloud:team": "Team",
"self-hosted:pro": "Pro (self-hosted)",
"self-hosted:enterprise": "Enterprise (self-hosted)",
"self-hosted:enterprise": "Enterprise",
} as const;
export type Plan = keyof typeof planLabels;
export const plans = Object.keys(planLabels) as Plan[];
// These functions are kept here to ensure consistency when updating plan names in the future.
// This function is kept here to ensure consistency when updating plan names in the future.
export const isCloudPlan = (plan: Plan) => plan.startsWith("cloud");
export const isSelfHostedPlan = (plan: Plan) => plan.startsWith("self-hosted");
export const isPlan = (value: string): value is Plan =>
plans.includes(value as Plan);
@@ -1,15 +0,0 @@
import { Prisma } from "../../db";
export const datasetItemMatchesVariable = (
input: Prisma.JsonValue,
variable: string,
) => {
if (
input === null ||
input === undefined ||
typeof input !== "object" ||
Array.isArray(input)
)
return false;
return Object.keys(input).includes(variable);
};
@@ -6,7 +6,7 @@ import { isPresent } from "../../utils/typeChecks";
import {
jsonSchema,
paginationMetaResponseZod,
publicApiPaginationZod,
paginationZod,
} from "../../utils/zod";
/**
@@ -17,7 +17,7 @@ export type ValidatedScoreConfig = z.infer<typeof ValidatedScoreConfigSchema>;
const validateCategories = (
categories: ConfigCategory[],
ctx: z.RefinementCtx,
ctx: z.RefinementCtx
) => {
const uniqueNames = new Set<string>();
const uniqueValues = new Set<number>();
@@ -94,7 +94,7 @@ const BooleanScoreConfig = z.object({
return categories.every(
(category, index) =>
category.label === expectedCategories[index].label &&
category.value === expectedCategories[index].value,
category.value === expectedCategories[index].value
);
}),
});
@@ -123,7 +123,7 @@ const ValidatedScoreConfigSchema = z
minValue: z.undefined().nullish(),
dataType: z.literal("CATEGORICAL"),
categories: Categories.superRefine(validateCategories),
}),
})
),
ScoreConfigBase.merge(BooleanScoreConfig),
])
@@ -150,7 +150,7 @@ const ValidatedScoreConfigSchema = z
*/
export const filterAndValidateDbScoreConfigList = (
scoreConfigs: ScoreConfigDbType[],
onParseError?: (error: z.ZodError) => void,
onParseError?: (error: z.ZodError) => void
): ValidatedScoreConfig[] =>
scoreConfigs.reduce((acc, ts) => {
const result = ValidatedScoreConfigSchema.safeParse(ts);
@@ -170,7 +170,7 @@ export const filterAndValidateDbScoreConfigList = (
* @throws error if score fails validation
*/
export const validateDbScoreConfig = (
scoreConfig: ScoreConfigDbType,
scoreConfig: ScoreConfigDbType
): ValidatedScoreConfig => ValidatedScoreConfigSchema.parse(scoreConfig);
/**
@@ -205,7 +205,7 @@ export const PostScoreConfigBody = z
z.object({
dataType: z.literal("BOOLEAN"),
categories: z.undefined().nullish(),
}),
})
),
])
.superRefine((data, ctx) => {
@@ -227,7 +227,7 @@ export const PostScoreConfigResponse = ValidatedScoreConfigSchema;
// GET /score-configs
export const GetScoreConfigsQuery = z.object({
...publicApiPaginationZod,
...paginationZod,
});
export const GetScoreConfigsResponse = z.object({
@@ -6,7 +6,7 @@ import { isPresent, stringDateTime } from "../../utils/typeChecks";
import {
NonEmptyString,
paginationMetaResponseZod,
publicApiPaginationZod,
paginationZod,
} from "../../utils/zod";
import { Category as ConfigCategory } from "./scoreConfigTypes";
@@ -84,13 +84,13 @@ export const ScoreBodyWithoutConfig = z.discriminatedUnion("dataType", [
z.object({
value: z.number(),
dataType: z.literal("NUMERIC"),
}),
})
),
BaseScoreBody.merge(
z.object({
value: z.string(),
dataType: z.literal("CATEGORICAL"),
}),
})
),
BaseScoreBody.merge(
z.object({
@@ -98,7 +98,7 @@ export const ScoreBodyWithoutConfig = z.discriminatedUnion("dataType", [
message: "Value must be either 0 or 1",
}),
dataType: z.literal("BOOLEAN"),
}),
})
),
]);
@@ -162,7 +162,7 @@ export const ScorePropsAgainstConfig = z.union([
*/
export const filterAndValidateDbScoreList = (
scores: Score[],
onParseError?: (error: z.ZodError) => void,
onParseError?: (error: z.ZodError) => void
): APIScore[] =>
scores.reduce((acc, ts) => {
const result = APIScoreSchema.safeParse(ts);
@@ -199,14 +199,14 @@ export const PostScoresBody = z.discriminatedUnion("dataType", [
value: z.number(),
dataType: z.literal("NUMERIC"),
configId: z.string().nullish(),
}),
})
),
BaseScoreBody.merge(
z.object({
value: z.string(),
dataType: z.literal("CATEGORICAL"),
configId: z.string().nullish(),
}),
})
),
BaseScoreBody.merge(
z.object({
@@ -216,14 +216,14 @@ export const PostScoresBody = z.discriminatedUnion("dataType", [
}),
dataType: z.literal("BOOLEAN"),
configId: z.string().nullish(),
}),
})
),
BaseScoreBody.merge(
z.object({
value: z.union([z.string(), z.number()]),
dataType: z.undefined(),
configId: z.string().nullish(),
}),
})
),
]);
@@ -231,7 +231,7 @@ export const PostScoresResponse = z.object({ id: z.string() });
// GET /scores
export const GetScoresQuery = z.object({
...publicApiPaginationZod,
...paginationZod,
userId: z.string().nullish(),
dataType: z.enum(ScoreDataType).nullish(),
configId: z.string().nullish(),
@@ -260,7 +260,7 @@ const LegacyGetScoreResponseDataV1 = z.intersection(
userId: z.string().nullish(),
tags: z.array(z.string()).nullish(),
}),
}),
})
);
export const GetScoresResponse = z.object({
data: z.array(LegacyGetScoreResponseDataV1),
@@ -269,7 +269,7 @@ export const GetScoresResponse = z.object({
export const legacyFilterAndValidateV1GetScoreList = (
scores: unknown[],
onParseError?: (error: z.ZodError) => void,
onParseError?: (error: z.ZodError) => void
): z.infer<typeof LegacyGetScoreResponseDataV1>[] =>
scores.reduce(
(acc: z.infer<typeof LegacyGetScoreResponseDataV1>[], ts) => {
@@ -282,7 +282,7 @@ export const legacyFilterAndValidateV1GetScoreList = (
}
return acc;
},
[] as z.infer<typeof LegacyGetScoreResponseDataV1>[],
[] as z.infer<typeof LegacyGetScoreResponseDataV1>[]
);
// GET /scores/{scoreId}
+2 -3
View File
@@ -7,6 +7,7 @@ export * from "./interfaces/customLLMProviderConfigSchemas";
export * from "./tableDefinitions";
export * from "./types";
export * from "./tableDefinitions/tracesTable";
export * from "./server/auth/apiKeys";
export * from "./observationsTable";
export * from "./utils/zod";
export * from "./utils/json";
@@ -15,6 +16,7 @@ export * from "./utils/objects";
export * from "./utils/typeChecks";
export * from "./features/entitlements/plans";
export * from "./interfaces/rate-limits";
export { env } from "./env";
// llm api
export * from "./server/llm/types";
@@ -32,9 +34,6 @@ export * from "./features/scores";
// comments
export * from "./features/comments/types";
// experiments
export * from "./features/experiments/utils";
// export db types only
export * from "@prisma/client";
export { type DB } from "../prisma/generated/types";
@@ -10,19 +10,3 @@ export const BedrockCredentialSchema = z
})
.optional();
export type BedrockCredential = z.infer<typeof BedrockCredentialSchema>;
export const GCPServiceAccountKeySchema = z.object({
type: z.literal("service_account"),
project_id: z.string(),
private_key_id: z.string(),
private_key: z.string(),
client_email: z.string(),
client_id: z.string(),
auth_uri: z.string(),
token_uri: z.string(),
auth_provider_x509_cert_url: z.string(),
client_x509_cert_url: z.string(),
});
export type GCPServiceAccountKey = z.infer<typeof GCPServiceAccountKeySchema>;
export default GCPServiceAccountKeySchema;
+1 -24
View File
@@ -20,13 +20,6 @@ export const observationsTableCols: ColumnDefinition[] = [
options: [], // to be added at runtime
nullable: true,
},
{
name: "type",
id: "type",
type: "stringOptions",
options: [],
internal: 'o."type"',
},
{ name: "Trace ID", id: "traceId", type: "string", internal: 't."id"' },
{
name: "Trace Name",
@@ -119,14 +112,6 @@ export const observationsTableCols: ColumnDefinition[] = [
options: [], // to be added at runtime
nullable: true,
},
{
name: "Model ID",
id: "modelId",
type: "stringOptions",
internal: 'o."internal_model_id"',
options: [], // to be added at runtime
nullable: true,
},
{
name: "Input Tokens",
id: "inputTokens",
@@ -201,25 +186,20 @@ export const observationsTableCols: ColumnDefinition[] = [
// allows for undefined options, to offer filters while options are still loading
export type ObservationOptions = {
model: Array<OptionsDefinition>;
modelId: Array<OptionsDefinition>;
name: Array<OptionsDefinition>;
traceName: Array<OptionsDefinition>;
scores_avg: Array<string>;
promptName: Array<OptionsDefinition>;
tags: Array<OptionsDefinition>;
type: Array<OptionsDefinition>;
};
export function observationsTableColsWithOptions(
options?: ObservationOptions,
options?: ObservationOptions
): ColumnDefinition[] {
return observationsTableCols.map((col) => {
if (col.id === "model") {
return { ...col, options: options?.model ?? [] };
}
if (col.id === "modelId") {
return { ...col, options: options?.modelId ?? [] };
}
if (col.id === "name") {
return { ...col, options: options?.name ?? [] };
}
@@ -235,9 +215,6 @@ export function observationsTableColsWithOptions(
if (col.id === "tags") {
return { ...col, options: options?.tags ?? [] };
}
if (col.id === "type") {
return { ...col, options: options?.type ?? [] };
}
return col;
});
}
@@ -2,64 +2,64 @@ import type { OAuthConfig, OAuthUserConfig } from "next-auth/providers/oauth";
import type { GithubProfile, GithubEmail } from "next-auth/providers/github";
export function GitHubEnterpriseProvider<P extends GithubProfile>(
options: OAuthUserConfig<P> & {
enterprise?: {
baseUrl?: string;
};
},
options: OAuthUserConfig<P> & {
enterprise?: {
baseUrl?: string;
};
}
): OAuthConfig<P> {
const baseUrl = options?.enterprise?.baseUrl ?? "https://github.com";
const apiBaseUrl = options?.enterprise?.baseUrl
? `${options?.enterprise?.baseUrl}/api/v3`
: "https://api.github.com";
const baseUrl = options?.enterprise?.baseUrl ?? "https://github.com"
const apiBaseUrl = options?.enterprise?.baseUrl
? `${options?.enterprise?.baseUrl}/api/v3`
: "https://api.github.com"
return {
id: "github-enterprise",
name: "GitHub Enterprise",
type: "oauth",
authorization: {
url: `${baseUrl}/login/oauth/authorize`,
params: { scope: "read:user user:email" },
},
token: `${baseUrl}/login/oauth/access_token`,
userinfo: {
url: `${apiBaseUrl}/user`,
async request({ client, tokens }) {
const profile = await client.userinfo(tokens.access_token!);
return {
id: "github-enterprise",
name: "GitHub Enterprise",
type: "oauth",
authorization: {
url: `${baseUrl}/login/oauth/authorize`,
params: { scope: "read:user user:email" },
},
token: `${baseUrl}/login/oauth/access_token`,
userinfo: {
url: `${apiBaseUrl}/user`,
async request({ client, tokens }) {
const profile = await client.userinfo(tokens.access_token!)
if (!profile.email) {
// If the user does not have a public email, get another via the GitHub API
// See https://docs.github.com/en/rest/users/emails#list-email-addresses-for-the-authenticated-user
const res = await fetch(`${apiBaseUrl}/user/emails`, {
headers: { Authorization: `token ${tokens.access_token}` },
});
if (!profile.email) {
// If the user does not have a public email, get another via the GitHub API
// See https://docs.github.com/en/rest/users/emails#list-email-addresses-for-the-authenticated-user
const res = await fetch(`${apiBaseUrl}/user/emails`, {
headers: { Authorization: `token ${tokens.access_token}` },
})
if (res.ok) {
const emails = (await res.json()) as GithubEmail[];
profile.email = (emails.find((e) => e.primary) ?? emails[0]).email;
}
}
if (res.ok) {
const emails: GithubEmail[] = await res.json()
profile.email = (emails.find((e) => e.primary) ?? emails[0]).email
}
}
return profile;
},
},
profile(profile) {
return {
id: profile.id.toString(),
name: profile.name ?? profile.login,
email: profile.email,
image: profile.avatar_url,
};
},
style: {
logo: "https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github.svg",
logoDark:
"https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github-dark.svg",
bg: "#fff",
bgDark: "#000",
text: "#000",
textDark: "#fff",
},
options,
};
}
return profile
},
},
profile(profile) {
return {
id: profile.id.toString(),
name: profile.name ?? profile.login,
email: profile.email,
image: profile.avatar_url,
}
},
style: {
logo: "https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github.svg",
logoDark:
"https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github-dark.svg",
bg: "#fff",
bgDark: "#000",
text: "#000",
textDark: "#fff",
},
options,
}
}
@@ -10,7 +10,7 @@ export const clickhouseClient = (opts?: NodeClickHouseClientConfigOptions) =>
url: env.CLICKHOUSE_URL,
username: env.CLICKHOUSE_USER,
password: env.CLICKHOUSE_PASSWORD,
database: env.CLICKHOUSE_DB,
database: "default",
clickhouse_settings: {
async_insert: 1,
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
+4 -14
View File
@@ -8,7 +8,6 @@ export * from "./auth/apiKeys";
export * from "./auth/customSsoProvider";
export * from "./auth/gitHubEnterpriseProvider";
export * from "./llm/fetchLLMCompletion";
export * from "./llm/utils";
export * from "./llm/types";
export * from "./utils/DatabaseReadStream";
export * from "./utils/transforms";
@@ -23,22 +22,17 @@ export * from "../server/ingestion/types";
export * from "../server/ingestion/validateAndInflateScore";
export * from "./redis/redis";
export * from "./redis/traceUpsert";
export * from "./redis/cloudUsageMeteringQueue";
export * from "./redis/CloudUsageMeteringQueue";
export * from "./redis/getQueue";
export * from "./redis/traceDelete";
export * from "./redis/projectDelete";
export * from "./redis/datasetRunItemUpsert";
export * from "./redis/batchExport";
export * from "./redis/legacyIngestion";
export * from "./redis/ingestionQueue";
export * from "./redis/postHogIntegrationQueue";
export * from "./redis/postHogIntegrationProcessingQueue";
export * from "./redis/dataRetentionQueue";
export * from "./redis/dataRetentionProcessingQueue";
export * from "./redis/coreDataS3ExportQueue";
export * from "./redis/meteringDataPostgresExportQueue";
export * from "./redis/experimentCreateQueue";
export * from "./auth/types";
export * from "./ingestion/legacy/index";
export * from "./queues";
export * from "./ingestion/legacy/EventProcessor";
export * from "./orderByToPrisma";
export * from "./filterToPrisma";
export * from "./instrumentation";
@@ -46,7 +40,3 @@ export * from "./logger";
export * from "./queries";
export * from "./repositories";
export * from "./redis/evalExecutionQueue";
export * from "./services/sessions-ui-table-service";
// test utils
export * from "./test-utils";
@@ -0,0 +1,786 @@
import { v4 } from "uuid";
import { type z } from "zod";
import Decimal from "decimal.js";
import { findModel } from "../modelMatch";
import {
ObservationEvent,
eventTypes,
legacyObservationCreateEvent,
generationCreateEvent,
traceEvent,
scoreEvent,
sdkLogEvent,
ingestionEvent,
} from "../types";
import { validateAndInflateScore } from "../validateAndInflateScore";
import { Trace, Observation, Score, Prisma, Model } from "@prisma/client";
import { ForbiddenError, LangfuseNotFoundError } from "../../../errors";
import { mergeJson } from "../../../utils/json";
import { jsonSchema } from "../../../utils/zod";
import { prisma } from "../../../db";
import { LegacyIngestionAccessScope } from ".";
import { logger } from "../../logger";
import { env } from "../../../env";
import { upsertTrace } from "../../repositories";
import { convertDateToClickhouseDateTime } from "../../clickhouse/client";
export interface EventProcessor {
auth(apiScope: LegacyIngestionAccessScope): void;
process(
apiScope: LegacyIngestionAccessScope,
): Promise<Trace | Observation | Score> | undefined;
}
export const getProcessorForEvent = (
event: z.infer<typeof ingestionEvent>,
calculateTokenDelegate: (p: {
model: Model;
text: unknown;
}) => number | undefined,
): EventProcessor => {
switch (event.type) {
case eventTypes.TRACE_CREATE:
return new TraceProcessor(event);
case eventTypes.OBSERVATION_CREATE:
case eventTypes.OBSERVATION_UPDATE:
case eventTypes.EVENT_CREATE:
case eventTypes.SPAN_CREATE:
case eventTypes.SPAN_UPDATE:
case eventTypes.GENERATION_CREATE:
case eventTypes.GENERATION_UPDATE:
return new ObservationProcessor(event, calculateTokenDelegate);
case eventTypes.SCORE_CREATE: {
return new ScoreProcessor(event);
}
case eventTypes.SDK_LOG:
return new SdkLogProcessor(event);
}
};
export class ObservationProcessor implements EventProcessor {
event: ObservationEvent;
calculateTokenDelegate: (p: {
model: Model;
text: unknown;
}) => number | undefined;
constructor(
event: ObservationEvent,
calculateTokenDelegate: (p: {
model: Model;
text: unknown;
}) => number | undefined,
) {
this.event = event;
this.calculateTokenDelegate = calculateTokenDelegate;
}
async convertToObservation(
apiScope: LegacyIngestionAccessScope,
existingObservation: Omit<Observation, "input" | "output"> | null,
): Promise<{
id: string;
create: Prisma.ObservationUncheckedCreateInput;
update: Prisma.ObservationUncheckedUpdateInput;
}> {
let type: "EVENT" | "SPAN" | "GENERATION";
switch (this.event.type) {
case eventTypes.OBSERVATION_CREATE:
case eventTypes.OBSERVATION_UPDATE:
type = this.event.body.type;
break;
case eventTypes.EVENT_CREATE:
type = "EVENT" as const;
break;
case eventTypes.SPAN_CREATE:
case eventTypes.SPAN_UPDATE:
type = "SPAN" as const;
break;
case eventTypes.GENERATION_CREATE:
case eventTypes.GENERATION_UPDATE:
type = "GENERATION" as const;
break;
}
if (
this.event.type === eventTypes.OBSERVATION_UPDATE &&
!existingObservation
) {
throw new LangfuseNotFoundError(
`Observation with id ${this.event.id} not found`,
);
}
// find matching model definition based on event and existing observation in db
const internalModel: Model | undefined | null =
type === "GENERATION"
? await findModel({
event: {
projectId: apiScope.projectId,
model:
"model" in this.event.body
? (this.event.body.model ?? undefined)
: undefined,
unit:
"usage" in this.event.body
? (this.event.body.usage?.unit ?? undefined)
: undefined,
startTime: this.event.body.startTime
? new Date(this.event.body.startTime)
: undefined,
},
existingDbObservation: existingObservation ?? undefined,
})
: undefined;
// Token counts
const [newInputCount, newOutputCount] =
"usage" in this.event.body
? await this.calculateTokenCounts(
apiScope.projectId,
this.event.body,
this.calculateTokenDelegate,
internalModel ?? undefined,
existingObservation ?? undefined,
)
: [undefined, undefined];
const newTotalCount =
"usage" in this.event.body
? (this.event.body.usage?.total ??
(newInputCount != null || newOutputCount != null
? (newInputCount ?? 0) + (newOutputCount ?? 0)
: undefined))
: undefined;
const userProvidedTokenCosts = {
inputCost:
"usage" in this.event.body && this.event.body.usage?.inputCost != null // inputCost can be explicitly 0. Note only one equal sign to capture null AND undefined
? new Decimal(this.event.body.usage?.inputCost)
: existingObservation?.inputCost,
outputCost:
"usage" in this.event.body && this.event.body.usage?.outputCost != null // outputCost can be explicitly 0. Note only one equal sign to capture null AND undefined
? new Decimal(this.event.body.usage?.outputCost)
: existingObservation?.outputCost,
totalCost:
"usage" in this.event.body && this.event.body.usage?.totalCost != null // totalCost can be explicitly 0. Note only one equal sign to capture null AND undefined
? new Decimal(this.event.body.usage?.totalCost)
: existingObservation?.totalCost,
};
const tokenCounts = {
input: newInputCount ?? existingObservation?.promptTokens,
output: newOutputCount ?? existingObservation?.completionTokens,
total: newTotalCount || existingObservation?.totalTokens,
};
const calculatedCosts = ObservationProcessor.calculateTokenCosts(
internalModel,
userProvidedTokenCosts,
tokenCounts,
);
// merge metadata from existingObservation.metadata and metadata
const mergedMetadata = mergeJson(
existingObservation?.metadata
? jsonSchema.parse(existingObservation.metadata)
: undefined,
this.event.body.metadata ?? undefined,
);
const prompt =
"promptName" in this.event.body &&
typeof this.event.body.promptName === "string" &&
"promptVersion" in this.event.body &&
typeof this.event.body.promptVersion === "number"
? await prisma.prompt.findUnique({
where: {
projectId_name_version: {
projectId: apiScope.projectId,
name: this.event.body.promptName,
version: this.event.body.promptVersion,
},
},
})
: undefined;
// Only null if promptName and promptVersion are set but prompt is not found
if (prompt === null) {
logger.warn("Prompt not found for observation", this.event.body);
}
const observationId =
this.event.body.id ??
(() => {
const newId = v4();
logger.info(
`observation.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
let traceId = this.event.body?.traceId;
if (!this.event.body.traceId && !existingObservation) {
// Create trace if no traceId
traceId = observationId;
// Insert trace into postgres
await prisma.trace.upsert({
where: {
id: observationId,
},
create: {
projectId: apiScope.projectId,
name: this.event.body.name,
id: observationId,
timestamp: this.event.body.startTime || new Date(),
},
update: {},
});
if (env.CLICKHOUSE_URL) {
// Insert trace into clickhouse if enabled
await upsertTrace({
id: observationId,
project_id: apiScope.projectId,
timestamp: convertDateToClickhouseDateTime(
this.event.body.startTime
? new Date(this.event.body.startTime)
: new Date(),
),
created_at: convertDateToClickhouseDateTime(new Date()),
updated_at: convertDateToClickhouseDateTime(new Date()),
});
}
}
return {
id: observationId,
create: {
id: observationId,
traceId,
type: type,
name: this.event.body.name,
startTime: this.event.body.startTime
? new Date(this.event.body.startTime)
: undefined,
endTime:
"endTime" in this.event.body && this.event.body.endTime
? new Date(this.event.body.endTime)
: undefined,
completionStartTime:
"completionStartTime" in this.event.body &&
this.event.body.completionStartTime
? new Date(this.event.body.completionStartTime)
: undefined,
metadata: mergedMetadata ?? this.event.body.metadata ?? undefined,
model: "model" in this.event.body ? this.event.body.model : undefined,
modelParameters:
"modelParameters" in this.event.body
? (this.event.body.modelParameters ?? undefined)
: undefined,
input: this.event.body.input ?? undefined,
output: this.event.body.output ?? undefined,
promptTokens: newInputCount,
completionTokens: newOutputCount,
totalTokens: newTotalCount,
unit:
"usage" in this.event.body
? (this.event.body.usage?.unit ?? internalModel?.unit)
: internalModel?.unit,
level: this.event.body.level ?? undefined,
statusMessage: this.event.body.statusMessage ?? undefined,
parentObservationId: this.event.body.parentObservationId ?? undefined,
version: this.event.body.version ?? undefined,
projectId: apiScope.projectId,
promptId: prompt ? prompt.id : undefined,
...(internalModel
? { internalModel: internalModel.modelName }
: undefined),
inputCost:
"usage" in this.event.body
? this.event.body.usage?.inputCost
: undefined,
outputCost:
"usage" in this.event.body
? this.event.body.usage?.outputCost
: undefined,
totalCost:
"usage" in this.event.body
? this.event.body.usage?.totalCost
: undefined,
calculatedInputCost: calculatedCosts?.inputCost,
calculatedOutputCost: calculatedCosts?.outputCost,
calculatedTotalCost: calculatedCosts?.totalCost,
internalModelId: internalModel?.id,
},
update: {
name: this.event.body.name ?? undefined,
startTime: this.event.body.startTime
? new Date(this.event.body.startTime)
: undefined,
endTime:
"endTime" in this.event.body && this.event.body.endTime
? new Date(this.event.body.endTime)
: undefined,
completionStartTime:
"completionStartTime" in this.event.body &&
this.event.body.completionStartTime
? new Date(this.event.body.completionStartTime)
: undefined,
metadata: mergedMetadata ?? this.event.body.metadata ?? undefined,
model: "model" in this.event.body ? this.event.body.model : undefined,
modelParameters:
"modelParameters" in this.event.body
? (this.event.body.modelParameters ?? undefined)
: undefined,
input: this.event.body.input ?? undefined,
output: this.event.body.output ?? undefined,
promptTokens: newInputCount,
completionTokens: newOutputCount,
totalTokens: newTotalCount,
unit:
"usage" in this.event.body
? (this.event.body.usage?.unit ?? internalModel?.unit)
: internalModel?.unit,
level: this.event.body.level ?? undefined,
statusMessage: this.event.body.statusMessage ?? undefined,
parentObservationId: this.event.body.parentObservationId ?? undefined,
version: this.event.body.version ?? undefined,
promptId: prompt ? prompt.id : undefined,
...(internalModel
? { internalModel: internalModel.modelName }
: undefined),
inputCost:
"usage" in this.event.body
? this.event.body.usage?.inputCost
: undefined,
outputCost:
"usage" in this.event.body
? this.event.body.usage?.outputCost
: undefined,
totalCost:
"usage" in this.event.body
? this.event.body.usage?.totalCost
: undefined,
calculatedInputCost: calculatedCosts?.inputCost,
calculatedOutputCost: calculatedCosts?.outputCost,
calculatedTotalCost: calculatedCosts?.totalCost,
internalModelId: internalModel?.id,
},
};
}
async calculateTokenCounts(
projectId: string,
body:
| z.infer<typeof legacyObservationCreateEvent>["body"]
| z.infer<typeof generationCreateEvent>["body"],
calculateTokenDelegate: (p: {
model: Model;
text: unknown;
}) => number | undefined,
model?: Model,
existingObservation?: Omit<Observation, "input" | "output">,
) {
let newPromptTokens = body.usage?.input;
if (newPromptTokens === undefined && model && model.tokenizerId) {
if (body.input) {
newPromptTokens = calculateTokenDelegate({
model: model,
text: body.input,
});
} else {
logger.debug(
`No input provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationInput = await prisma.observation.findFirst({
where: { id: existingObservation?.id, projectId: projectId },
select: {
input: true,
},
});
newPromptTokens = calculateTokenDelegate({
model: model,
text: observationInput?.input,
});
}
}
let newCompletionTokens = body.usage?.output;
if (newCompletionTokens === undefined && model && model.tokenizerId) {
if (body.output) {
newCompletionTokens = calculateTokenDelegate({
model: model,
text: body.output,
});
} else {
logger.debug(
`No output provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationOutput = await prisma.observation.findFirst({
where: { id: existingObservation?.id, projectId: projectId },
select: {
output: true,
},
});
newCompletionTokens = calculateTokenDelegate({
model: model,
text: observationOutput?.output,
});
}
}
return [newPromptTokens ?? undefined, newCompletionTokens ?? undefined];
}
static calculateTokenCosts(
model: Model | null | undefined,
userProvidedCosts: {
inputCost?: Decimal | null;
outputCost?: Decimal | null;
totalCost?: Decimal | null;
},
tokenCounts: { input?: number; output?: number; total?: number },
): {
inputCost?: Decimal | null;
outputCost?: Decimal | null;
totalCost?: Decimal | null;
} {
// If user has provided any cost point, do not calculate anything else
if (
userProvidedCosts.inputCost ||
userProvidedCosts.outputCost ||
userProvidedCosts.totalCost
) {
return {
...userProvidedCosts,
totalCost:
userProvidedCosts.totalCost ??
(userProvidedCosts.inputCost ?? new Decimal(0)).add(
userProvidedCosts.outputCost ?? new Decimal(0),
),
};
}
const finalInputCost =
tokenCounts.input !== undefined && model?.inputPrice
? model.inputPrice.mul(tokenCounts.input)
: undefined;
const finalOutputCost =
tokenCounts.output !== undefined && model?.outputPrice
? model.outputPrice.mul(tokenCounts.output)
: finalInputCost
? new Decimal(0)
: undefined;
const finalTotalCost =
tokenCounts.total !== undefined && model?.totalPrice
? model.totalPrice.mul(tokenCounts.total)
: (finalInputCost ?? finalOutputCost)
? new Decimal(finalInputCost ?? 0).add(finalOutputCost ?? 0)
: undefined;
return {
inputCost: finalInputCost,
outputCost: finalOutputCost,
totalCost: finalTotalCost,
};
}
auth(apiScope: LegacyIngestionAccessScope): void {
if (apiScope.accessLevel !== "all")
throw new ForbiddenError("Access denied for observation creation");
}
async process(apiScope: LegacyIngestionAccessScope): Promise<Observation> {
this.auth(apiScope);
const existingObservation = this.event.body.id
? await prisma.observation.findFirst({
select: {
// do not select I/O to spare our db
input: false,
output: false,
id: true,
traceId: true,
projectId: true,
type: true,
startTime: true,
endTime: true,
name: true,
metadata: true,
parentObservationId: true,
level: true,
statusMessage: true,
version: true,
createdAt: true,
updatedAt: true,
model: true,
internalModelId: true,
modelParameters: true,
promptTokens: true,
completionTokens: true,
totalTokens: true,
unit: true,
inputCost: true,
outputCost: true,
totalCost: true,
calculatedInputCost: true,
calculatedOutputCost: true,
calculatedTotalCost: true,
completionStartTime: true,
promptId: true,
internalModel: true,
},
where: { id: this.event.body.id, projectId: apiScope.projectId },
})
: null;
if (
existingObservation &&
existingObservation.projectId !== apiScope.projectId
) {
throw new ForbiddenError(
`Access denied for observation creation ${existingObservation.projectId} `,
);
}
const obs = await this.convertToObservation(apiScope, existingObservation);
// Do not use nested upserts or multiple where conditions as this should be a single native database upsert
// https://www.prisma.io/docs/orm/reference/prisma-client-reference#database-upserts
return await prisma.observation.upsert({
where: {
id: obs.id,
},
create: obs.create,
update: obs.update,
});
}
}
export class TraceProcessor implements EventProcessor {
event: z.infer<typeof traceEvent>;
constructor(event: z.infer<typeof traceEvent>) {
this.event = event;
}
auth(apiScope: LegacyIngestionAccessScope): void {
if (apiScope.accessLevel !== "all")
throw new ForbiddenError("Access denied for trace creation");
}
async process(
apiScope: LegacyIngestionAccessScope,
): Promise<Trace | Observation | Score> {
const { body } = this.event;
this.auth(apiScope);
const internalId =
body.id ??
(() => {
const newId = v4();
logger.info(
`trace.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
logger.debug(
`Trying to create trace, project ${apiScope.projectId}, id: ${internalId}`,
);
const existingTrace = await prisma.trace.findFirst({
where: {
id: internalId,
},
});
if (existingTrace && existingTrace.projectId !== apiScope.projectId) {
throw new ForbiddenError(
`Access denied for trace creation ${existingTrace.projectId}`,
);
}
const mergedMetadata = mergeJson(
existingTrace?.metadata
? jsonSchema.parse(existingTrace.metadata)
: undefined,
body.metadata ?? undefined,
);
const mergedTags =
existingTrace?.tags && body.tags
? Array.from(new Set(existingTrace.tags.concat(body.tags ?? []))).sort()
: body.tags
? Array.from(new Set(body.tags)).sort()
: undefined;
if (body.sessionId) {
try {
await prisma.traceSession.upsert({
where: {
id_projectId: {
id: body.sessionId,
projectId: apiScope.projectId,
},
},
create: {
id: body.sessionId,
projectId: apiScope.projectId,
},
update: {},
});
} catch (e) {
if (
e instanceof Prisma.PrismaClientKnownRequestError &&
e.code === "P2002"
) {
logger.warn(
`Failed to upsert session. Session ${body.sessionId} in project ${apiScope.projectId} already exists`,
);
} else {
throw e;
}
}
}
// Do not use nested upserts or multiple where conditions as this should be a single native database upsert
// https://www.prisma.io/docs/orm/reference/prisma-client-reference#database-upserts
const upsertedTrace = await prisma.trace.upsert({
where: {
id: internalId,
},
create: {
id: internalId,
timestamp: this.event.body.timestamp
? new Date(this.event.body.timestamp)
: undefined,
name: body.name ?? undefined,
userId: body.userId ?? undefined,
input: body.input ?? undefined,
output: body.output ?? undefined,
metadata: mergedMetadata ?? body.metadata ?? undefined,
release: body.release ?? undefined,
version: body.version ?? undefined,
sessionId: body.sessionId ?? undefined,
public: body.public ?? undefined,
projectId: apiScope.projectId,
tags: mergedTags ?? undefined,
},
update: {
name: body.name ?? undefined,
timestamp: this.event.body.timestamp
? new Date(this.event.body.timestamp)
: undefined,
userId: body.userId ?? undefined,
input: body.input ?? undefined,
output: body.output ?? undefined,
metadata: mergedMetadata ?? body.metadata ?? undefined,
release: body.release ?? undefined,
version: body.version ?? undefined,
sessionId: body.sessionId ?? undefined,
public: body.public ?? undefined,
tags: mergedTags ?? undefined,
},
});
return upsertedTrace;
}
}
export class ScoreProcessor implements EventProcessor {
event: z.infer<typeof scoreEvent>;
constructor(event: z.infer<typeof scoreEvent>) {
this.event = event;
}
auth(apiScope: LegacyIngestionAccessScope) {
if (apiScope.accessLevel !== "scores" && apiScope.accessLevel !== "all")
throw new ForbiddenError(
`Access denied for score creation, ${apiScope.accessLevel}`,
);
}
async process(
apiScope: LegacyIngestionAccessScope,
): Promise<Trace | Observation | Score> {
const { body } = this.event;
this.auth(apiScope);
const id =
body.id ??
(() => {
const newId = v4();
logger.info(
`score.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
const existingScore = await prisma.score.findFirst({
where: {
id: id,
},
select: {
projectId: true,
},
});
if (existingScore && existingScore.projectId !== apiScope.projectId) {
throw new ForbiddenError(
`Access denied for score creation ${existingScore.projectId}`,
);
}
const validatedScore = await validateAndInflateScore({
body,
scoreId: id,
projectId: apiScope.projectId,
});
return await prisma.score.upsert({
where: {
id_projectId: {
id,
projectId: apiScope.projectId,
},
},
create: {
...validatedScore,
},
update: {
...validatedScore,
},
});
}
}
export class SdkLogProcessor implements EventProcessor {
event: z.infer<typeof sdkLogEvent>;
constructor(event: z.infer<typeof sdkLogEvent>) {
this.event = event;
}
auth(apiScope: LegacyIngestionAccessScope) {
return;
}
process() {
try {
logger.info("SDK Log", this.event);
return undefined;
} catch (error) {
return undefined;
}
}
}
@@ -0,0 +1,161 @@
import { env } from "node:process";
import z from "zod";
import { ForbiddenError, UnauthorizedError } from "../../../errors";
import { eventTypes, ingestionApiSchema, IngestionEventType } from "../types";
import { getProcessorForEvent } from "./EventProcessor";
import { ApiAccessScope } from "../../auth/types";
import { backOff } from "exponential-backoff";
import { Model } from "../../..";
import { logger } from "../../logger";
export type BatchResult = {
result: unknown;
id: string;
type: string;
};
type TokenCountInput = {
model: Model;
text: unknown;
};
export type LegacyIngestionAccessScope = Omit<
ApiAccessScope,
"orgId" | "plan" | "rateLimitOverrides"
>;
type LegacyIngestionAuthHeaderVerificationResult =
| {
validKey: true;
scope: LegacyIngestionAccessScope;
}
| {
validKey: false;
error: string;
};
export const handleBatch = async (
events: z.infer<typeof ingestionApiSchema>["batch"],
authCheck: LegacyIngestionAuthHeaderVerificationResult,
calculateTokenDelegate: (p: TokenCountInput) => number | undefined,
) => {
logger.debug(`handling ingestion ${events.length} events`);
if (!authCheck.validKey) throw new UnauthorizedError(authCheck.error);
const results: BatchResult[] = []; // Array to store the results
const errors: {
error: unknown;
id: string;
type: string;
}[] = []; // Array to store the errors
for (const singleEvent of events) {
try {
const result = await retry(async () => {
return await handleSingleEvent(
singleEvent,
authCheck.scope,
calculateTokenDelegate,
);
});
results.push({
result: result,
id: singleEvent.id,
type: singleEvent.type,
}); // Push each result into the array
} catch (error) {
// Handle or log the error if `handleSingleEvent` fails
logger.error("Error handling event:", error);
// Decide how to handle the error: rethrow, continue, or push an error object to results
// For example, push an error object:
errors.push({
error,
id: singleEvent.id,
type: singleEvent.type,
});
}
}
return { results, errors };
};
async function retry<T>(request: () => Promise<T>): Promise<T> {
return await backOff(request, {
numOfAttempts: env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" ? 5 : 3,
retry: (e: Error, attemptNumber: number) => {
if (e instanceof UnauthorizedError || e instanceof ForbiddenError) {
logger.info("not retrying auth error");
return false;
}
logger.info(`retrying processing events ${attemptNumber}`);
return true;
},
});
}
const handleSingleEvent = async (
event: IngestionEventType,
apiScope: LegacyIngestionAccessScope,
calculateTokenDelegate: (p: {
model: Model;
text: unknown;
}) => number | undefined,
) => {
const { body } = event;
let restEvent = body;
if ("input" in body) {
// eslint-disable-next-line @typescript-eslint/no-unused-vars
const { input, ...rest } = body;
restEvent = rest;
}
if ("output" in restEvent) {
// eslint-disable-next-line @typescript-eslint/no-unused-vars
const { output, ...rest } = restEvent;
restEvent = rest;
}
logger.debug(
`handling single event ${event.id} of type ${event.type}: ${JSON.stringify({ body: restEvent })}`,
);
const cleanedEvent = cleanEvent(event) as IngestionEventType;
// Deny access to non-score events if the access level is not "all"
// This is an additional safeguard to auth checks in EventProcessor
if (
apiScope.accessLevel !== "all" &&
cleanedEvent.type !== eventTypes.SCORE_CREATE
) {
throw new ForbiddenError("Access denied. Event type not allowed.");
}
return getProcessorForEvent(cleanedEvent, calculateTokenDelegate).process(
apiScope,
);
};
// cleans NULL characters from the event
export function cleanEvent(obj: unknown): unknown {
if (typeof obj === "string") {
return obj.replace(/\u0000/g, "");
} else if (typeof obj === "object" && obj !== null) {
if (Array.isArray(obj)) {
return obj.map(cleanEvent);
} else {
// Here we assert that obj is a Record<string, unknown>
const objAsRecord = obj as Record<string, unknown>;
const newObj: Record<string, unknown> = {};
for (const key in objAsRecord) {
newObj[key] = cleanEvent(objAsRecord[key]);
}
return newObj;
}
} else {
return obj;
}
}
export const isUndefinedOrNull = <T>(val?: T | null): val is undefined | null =>
val === undefined || val === null;
@@ -4,7 +4,6 @@ import { z } from "zod";
import { type Model } from "../../db";
import { env } from "../../env";
import {
ForbiddenError,
InvalidRequestError,
LangfuseNotFoundError,
UnauthorizedError,
@@ -19,13 +18,16 @@ import {
traceException,
} from "../instrumentation";
import { logger } from "../logger";
import { QueueJobs } from "../queues";
import { LegacyIngestionEventType, QueueJobs } from "../queues";
import { IngestionQueue } from "../redis/ingestionQueue";
import { LegacyIngestionQueue } from "../redis/legacyIngestion";
import { redis } from "../redis/redis";
import { handleBatch } from "./legacy";
import {
StorageService,
StorageServiceFactory,
} from "../services/StorageService";
import { getProcessorForEvent } from "./legacy/EventProcessor";
import { eventTypes, ingestionEvent, IngestionEventType } from "./types";
export type TokenCountDelegate = (p: {
@@ -52,6 +54,7 @@ const getS3StorageServiceClient = (bucketName: string): StorageService => {
export const processEventBatch = async (
input: unknown[],
authCheck: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): Promise<{
successes: { id: string; status: number }[];
errors: {
@@ -96,7 +99,7 @@ export const processEventBatch = async (
});
return [];
}
if (!isAuthorized(parsed.data, authCheck)) {
if (!isAuthorized(parsed.data, authCheck, tokenCountDelegate)) {
authenticationErrors.push({
id: parsed.data.id,
error: new UnauthorizedError("Access Scope Denied"),
@@ -153,95 +156,157 @@ export const processEventBatch = async (
* ASYNC PROCESSING *
********************/
let s3UploadErrored = false;
await instrumentAsync({ name: "s3-upload-events" }, async () => {
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
// S3 Event Upload is blocking, but non-failing.
// If a promise rejects, we log it below, but do not throw an error.
// In this case, we upload the full batch into the Redis queue.
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
await instrumentAsync({ name: "s3-upload-events" }, async () => {
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
throw new Error("S3 event store is enabled but no bucket is set");
}
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
// S3 Event Upload is currently blocking, but non-failing.
// If a promise rejects, we log it below, but do not throw an error.
// In this case, we upload the full batch into the Redis queue.
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) => {
// We upload the event in an array to the S3 bucket grouped by the eventBodyId.
// That way we batch updates from the same invocation into a single file and reduce
// write operations on S3.
const { data, key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${authCheck.scope.projectId}/${getClickhouseEntityType(type)}/${eventBodyId}/${key}.json`,
data,
);
}),
);
results.forEach((result) => {
if (result.status === "rejected") {
s3UploadErrored = true;
logger.error("Failed to upload event to S3", {
error: result.reason,
});
}
});
});
}
// Send each event individually to IngestionQueue for new processing
if (
env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" &&
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" &&
env.LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING === "true" &&
redis &&
!s3UploadErrored
) {
const queue = IngestionQueue.getInstance();
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) => {
// We upload the event in an array to the S3 bucket grouped by the eventBodyId.
// That way we batch updates from the same invocation into a single file and reduce
// write operations on S3.
const { data, key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${authCheck.scope.projectId}/${getClickhouseEntityType(type)}/${eventBodyId}/${key}.json`,
data,
);
}),
Object.keys(sortedBatchByEventBodyId).map(async (id) =>
queue
? queue.add(
QueueJobs.IngestionJob,
{
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.IngestionJob as const,
payload: {
data: {
type: sortedBatchByEventBodyId[id].type,
eventBodyId: sortedBatchByEventBodyId[id].eventBodyId,
},
authCheck,
},
},
{
delay: env.LANGFUSE_INGESTION_QUEUE_DELAY_MS,
},
)
: Promise.reject("Failed to instantiate queue"),
),
);
results.forEach((result) => {
if (result.status === "rejected") {
s3UploadErrored = true;
logger.error("Failed to upload event to S3", {
logger.error("Failed to add event to IngestionQueue", {
error: result.reason,
});
}
});
});
// Send each event individually to IngestionQueue for ClickHouse processing
if (s3UploadErrored) {
throw new Error(
"Failed to upload events to blob storage, aborting event processing",
);
}
if (!redis) {
throw new Error("Redis not initialized, aborting event processing");
// As part of the legacy processing we sent the entire batch to the worker.
if (env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" && redis) {
const queue = LegacyIngestionQueue.getInstance();
if (queue) {
let addToQueueFailed = false;
const queuePayload: LegacyIngestionEventType =
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" && !s3UploadErrored
? {
data: Object.keys(sortedBatchByEventBodyId).map((id) => {
const { key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return {
type,
eventBodyId,
eventId: key,
};
}),
authCheck,
useS3EventStore: true,
}
: { data: sortedBatch, authCheck, useS3EventStore: false };
try {
await queue.add(QueueJobs.LegacyIngestionJob, {
payload: queuePayload,
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.LegacyIngestionJob as const,
});
} catch (e: unknown) {
logger.warn(
"Failed to add batch to queue, falling back to sync processing",
e,
);
addToQueueFailed = true;
}
if (!addToQueueFailed) {
return aggregateBatchResult(
// we are not sending additional server errors to the client in case of early return
[...validationErrors, ...authenticationErrors],
sortedBatch.map((event) => ({ id: event.id, result: event })),
);
}
} else {
logger.error(
"Ingestion queue not initialized, falling back to sync processing",
);
}
}
const queue = IngestionQueue.getInstance();
await Promise.all(
Object.keys(sortedBatchByEventBodyId).map(async (id) =>
queue
? queue.add(
QueueJobs.IngestionJob,
{
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.IngestionJob as const,
payload: {
data: {
type: sortedBatchByEventBodyId[id].type,
eventBodyId: sortedBatchByEventBodyId[id].eventBodyId,
},
authCheck,
},
},
{
delay: env.LANGFUSE_INGESTION_QUEUE_DELAY_MS,
},
)
: Promise.reject("Failed to instantiate queue"),
),
);
/*******************
* SYNC PROCESSING *
*******************/
const result = await handleBatch(sortedBatch, authCheck, tokenCountDelegate);
// in case we did not return early, we return the result here
return aggregateBatchResult(
[...validationErrors, ...authenticationErrors],
sortedBatch.map((event) => ({ id: event.id, result: event })),
authCheck.scope.projectId,
[...validationErrors, ...authenticationErrors, ...result.errors],
result.results,
);
};
const isAuthorized = (
event: IngestionEventType,
authScope: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): boolean => {
if (event.type === eventTypes.SDK_LOG) {
try {
getProcessorForEvent(event, tokenCountDelegate).auth(authScope.scope);
return true;
} catch (error) {
return false;
}
if (event.type === eventTypes.SCORE_CREATE) {
return (
authScope.scope.accessLevel === "scores" ||
authScope.scope.accessLevel === "all"
);
}
return authScope.scope.accessLevel === "all";
};
/**
@@ -271,7 +336,6 @@ const sortBatch = (batch: Array<z.infer<typeof ingestionEvent>>) => {
export const aggregateBatchResult = (
errors: Array<{ id: string; error: unknown }>,
results: Array<{ id: string; result: unknown }>,
projectId?: string,
) => {
const returnedErrors: {
id: string;
@@ -318,10 +382,7 @@ export const aggregateBatchResult = (
if (returnedErrors.length > 0) {
traceException(errors);
logger.error("Error processing events", {
errors: returnedErrors,
"langfuse.project.id": projectId,
});
logger.error("Error processing events", returnedErrors);
}
results.forEach((result) => {
+5 -74
View File
@@ -56,76 +56,13 @@ export const usage = MixedUsage.nullish()
// ensure output is always of new usage model
.pipe(Usage.nullish());
const RawUsageOrCostDetails = z.record(
z.string(),
z.number().nonnegative().nullish(),
);
const OpenAIUsageSchema = z
.object({
prompt_tokens: z.number().nonnegative(),
completion_tokens: z.number().nonnegative(),
total_tokens: z.number().nonnegative(),
prompt_tokens_details: z
.record(z.string(), z.number().nonnegative())
.nullish(),
completion_tokens_details: z
.record(z.string(), z.number().nonnegative())
.nullish(),
})
.strict()
.transform((v) => {
if (!v) return;
const {
prompt_tokens,
completion_tokens,
total_tokens,
prompt_tokens_details,
completion_tokens_details,
} = v;
const result: z.infer<typeof RawUsageOrCostDetails> & {
input: number;
output: number;
total: number;
} = {
input: prompt_tokens,
output: completion_tokens,
total: total_tokens,
};
if (prompt_tokens_details) {
for (const [key, value] of Object.entries(prompt_tokens_details)) {
result[`input_${key}`] = value;
result.input = Math.max(result.input - (value ?? 0), 0);
}
}
if (completion_tokens_details) {
for (const [key, value] of Object.entries(completion_tokens_details)) {
result[`output_${key}`] = value;
result.output = Math.max(result.output - (value ?? 0), 0);
}
}
return result;
})
.pipe(RawUsageOrCostDetails);
export const UsageOrCostDetails = z
.union([OpenAIUsageSchema, RawUsageOrCostDetails])
.nullish();
// Using z.any instead of jsonSchema for input/output as we saw huge CPU overhead for large numeric arrays.
// With this setup parsing should be more lightweight and doesn't block other requests.
// As we allow plain values, arrays, and objects the JSON parse via bodyParser should suffice.
export const TraceBody = z.object({
id: z.string().nullish(),
timestamp: stringDateTime,
name: z.string().max(1000).nullish(),
name: z.string().nullish(),
externalId: z.string().nullish(),
input: z.any().nullish(),
output: z.any().nullish(),
input: jsonSchema.nullish(),
output: jsonSchema.nullish(),
sessionId: z.string().nullish(),
userId: z.string().nullish(),
metadata: jsonSchema.nullish(),
@@ -140,8 +77,8 @@ export const OptionalObservationBody = z.object({
name: z.string().nullish(),
startTime: stringDateTime,
metadata: jsonSchema.nullish(),
input: z.any().nullish(),
output: z.any().nullish(),
input: jsonSchema.nullish(),
output: jsonSchema.nullish(),
level: z.nativeEnum(ObservationLevel).nullish(),
statusMessage: z.string().nullish(),
parentObservationId: z.string().nullish(),
@@ -182,8 +119,6 @@ export const CreateGenerationBody = CreateSpanBody.extend({
)
.nullish(),
usage: usage,
usageDetails: UsageOrCostDetails,
costDetails: UsageOrCostDetails,
promptName: z.string().nullish(),
promptVersion: z.number().int().nullish(),
}).refine((value) => {
@@ -212,8 +147,6 @@ export const UpdateGenerationBody = UpdateSpanBody.extend({
)
.nullish(),
usage: usage,
usageDetails: UsageOrCostDetails,
costDetails: UsageOrCostDetails,
promptName: z.string().nullish(),
promptVersion: z.number().int().nullish(),
}).refine((value) => {
@@ -365,8 +298,6 @@ export const LegacyObservationBody = z.object({
input: jsonSchema.nullish(),
output: jsonSchema.nullish(),
usage: usage,
usageDetails: UsageOrCostDetails,
costDetails: UsageOrCostDetails,
metadata: jsonSchema.nullish(),
parentObservationId: z.string().nullish(),
level: z.nativeEnum(ObservationLevel).nullish(),
@@ -17,7 +17,7 @@ type ValidateAndInflateScoreParams = {
export async function validateAndInflateScore(
params: ValidateAndInflateScoreParams,
): Promise<Score> {
const { body, projectId, scoreId } = params;
const { body, projectId } = params;
if (body.configId) {
const config = await prisma.scoreConfig.findFirst({
@@ -32,22 +32,10 @@ export async function validateAndInflateScore(
"The configId you provided does not match a valid config in this project",
);
// Override some fields in the score body with config fields
// We ignore the set fields in the body
const bodyWithConfigOverrides = {
...body,
name: config.name,
};
validateConfigAgainstBody(
bodyWithConfigOverrides,
config as ValidatedScoreConfig,
);
validateConfigAgainstBody(body, config as ValidatedScoreConfig);
return inflateScoreBody({
projectId,
scoreId,
body: bodyWithConfigOverrides,
...params,
config: config as ValidatedScoreConfig,
});
}
@@ -226,14 +226,6 @@ export const recordHistogram = (
dd.dogstatsd.histogram(stat, value, tags);
};
export const recordDistribution = (
stat: string,
value?: number | undefined,
tags?: { [tag: string]: string | number } | undefined,
) => {
dd.dogstatsd.distribution(stat, value, tags);
};
/**
* Converts a queue name to the matching datadog metric name.
* Consumer only needs to append the relevant suffix.
@@ -1,11 +1,11 @@
import type { ZodSchema } from "zod";
import { CallbackHandler } from "langfuse-langchain";
import { ChatAnthropic } from "@langchain/anthropic";
import { ChatVertexAI } from "@langchain/google-vertexai";
import { ChatBedrockConverse } from "@langchain/aws";
import {
AIMessage,
BaseMessage,
HumanMessage,
SystemMessage,
} from "@langchain/core/messages";
@@ -15,24 +15,31 @@ import {
} from "@langchain/core/output_parsers";
import { IterableReadableStream } from "@langchain/core/utils/stream";
import { ChatOpenAI } from "@langchain/openai";
import GCPServiceAccountKeySchema, {
import {
BedrockConfigSchema,
BedrockCredentialSchema,
} from "../../interfaces/customLLMProviderConfigSchemas";
import { processEventBatch } from "../ingestion/processEventBatch";
import { logger } from "../logger";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import {
ChatMessage,
ChatMessageRole,
LLMAdapter,
ModelParams,
TraceParams,
} from "./types";
import { CallbackHandler } from "langfuse-langchain";
processEventBatch,
type TokenCountDelegate,
} from "../ingestion/processEventBatch";
import { logger } from "../logger";
import { ChatMessage, ChatMessageRole, LLMAdapter, ModelParams } from "./types";
import type { BaseCallbackHandler } from "@langchain/core/callbacks/base";
type ProcessTracedEvents = () => Promise<void>;
export type TraceParams = {
traceName: string;
traceId: string;
projectId: string;
tags: string[];
tokenCountDelegate: TokenCountDelegate;
authCheck: AuthHeaderValidVerificationResult;
};
type LLMCompletionParams = {
messages: ChatMessage[];
modelParams: ModelParams;
@@ -40,11 +47,9 @@ type LLMCompletionParams = {
callbacks?: BaseCallbackHandler[];
baseURL?: string;
apiKey: string;
extraHeaders?: Record<string, string>;
maxRetries?: number;
config?: Record<string, string> | null;
traceParams?: TraceParams;
throwOnError?: boolean; // default is true
};
type FetchLLMCompletionParams = LLMCompletionParams & {
@@ -93,8 +98,6 @@ export async function fetchLLMCompletion(
maxRetries,
config,
traceParams,
extraHeaders,
throwOnError = true,
} = params;
let finalCallbacks: BaseCallbackHandler[] | undefined = callbacks ?? [];
@@ -106,6 +109,7 @@ export async function fetchLLMCompletion(
_isLocalEventExportEnabled: true,
tags: traceParams.tags,
});
finalCallbacks.push(handler);
processTracedEvents = async () => {
@@ -116,6 +120,7 @@ export async function fetchLLMCompletion(
await processEventBatch(
JSON.parse(JSON.stringify(events)), // stringify to emulate network event batch from network call
traceParams.authCheck,
traceParams.tokenCountDelegate,
);
} catch (e) {
logger.error("Failed to process traced events", { error: e });
@@ -125,28 +130,16 @@ export async function fetchLLMCompletion(
finalCallbacks = finalCallbacks.length > 0 ? finalCallbacks : undefined;
let finalMessages: BaseMessage[];
// VertexAI requires at least 1 user message
if (modelParams.adapter === LLMAdapter.VertexAI && messages.length === 1) {
finalMessages = [new HumanMessage(messages[0].content)];
} else {
finalMessages = messages.map((message) => {
if (message.role === ChatMessageRole.User)
return new HumanMessage(message.content);
if (message.role === ChatMessageRole.System)
return new SystemMessage(message.content);
const finalMessages = messages.map((message) => {
if (message.role === ChatMessageRole.User)
return new HumanMessage(message.content);
if (message.role === ChatMessageRole.System)
return new SystemMessage(message.content);
return new AIMessage(message.content);
});
}
return new AIMessage(message.content);
});
finalMessages = finalMessages.filter((m) => m.content.length > 0);
let chatModel:
| ChatOpenAI
| ChatAnthropic
| ChatBedrockConverse
| ChatVertexAI;
let chatModel: ChatOpenAI | ChatAnthropic | ChatBedrockConverse;
if (modelParams.adapter === LLMAdapter.Anthropic) {
chatModel = new ChatAnthropic({
anthropicApiKey: apiKey,
@@ -156,7 +149,7 @@ export async function fetchLLMCompletion(
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks: finalCallbacks,
clientOptions: { maxRetries, timeout: 1000 * 60 * 2 }, // 2 minutes timeout
clientOptions: { maxRetries },
});
} else if (modelParams.adapter === LLMAdapter.OpenAI) {
chatModel = new ChatOpenAI({
@@ -170,9 +163,7 @@ export async function fetchLLMCompletion(
maxRetries,
configuration: {
baseURL,
defaultHeaders: extraHeaders,
},
timeout: 1000 * 60 * 2, // 2 minutes timeout
});
} else if (modelParams.adapter === LLMAdapter.Azure) {
chatModel = new ChatOpenAI({
@@ -185,7 +176,6 @@ export async function fetchLLMCompletion(
topP: modelParams.top_p,
callbacks: finalCallbacks,
maxRetries,
timeout: 1000 * 60 * 2, // 2 minutes timeout
});
} else if (modelParams.adapter === LLMAdapter.Bedrock) {
const { region } = BedrockConfigSchema.parse(config);
@@ -200,24 +190,6 @@ export async function fetchLLMCompletion(
topP: modelParams.top_p,
callbacks: finalCallbacks,
maxRetries,
timeout: 1000 * 60 * 2, // 2 minutes timeout
});
} else if (modelParams.adapter === LLMAdapter.VertexAI) {
const credentials = GCPServiceAccountKeySchema.parse(JSON.parse(apiKey));
// Requests time out after 60 seconds for both public and private endpoints by default
// Reference: https://cloud.google.com/vertex-ai/docs/predictions/get-online-predictions#send-request
chatModel = new ChatVertexAI({
modelName: modelParams.model,
temperature: modelParams.temperature,
maxOutputTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks: finalCallbacks,
maxRetries,
authOptions: {
projectId: credentials.project_id,
credentials,
},
});
} else {
// eslint-disable-next-line no-unused-vars
@@ -231,17 +203,16 @@ export async function fetchLLMCompletion(
runName: traceParams?.traceName,
};
try {
if (params.structuredOutputSchema) {
return {
completion: await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
if (params.structuredOutputSchema) {
return {
completion: await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
/*
/*
Workaround OpenAI o1 while in beta:
This is a temporary workaround to avoid sending system messages to OpenAI's O1 models.
@@ -253,49 +224,42 @@ export async function fetchLLMCompletion(
Reference: https://platform.openai.com/docs/guides/reasoning/beta-limitations
*/
if (modelParams.model.startsWith("o1-")) {
return {
completion: await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
timeout: 1000 * 60 * 2, // 2 minutes timeout
})
.pipe(new StringOutputParser())
.invoke(
finalMessages.filter((message) => message._getType() !== "system"),
runConfig,
),
processTracedEvents,
};
}
if (streaming) {
return {
completion: await chatModel
.pipe(new BytesOutputParser())
.stream(finalMessages, runConfig),
processTracedEvents,
};
}
if (modelParams.model.startsWith("o1-")) {
return {
completion: await chatModel
completion: await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
})
.pipe(new StringOutputParser())
.invoke(finalMessages, runConfig),
.invoke(
finalMessages.filter((message) => message._getType() !== "system"),
runConfig,
),
processTracedEvents,
};
} catch (error) {
if (throwOnError) {
throw error;
}
return { completion: null, processTracedEvents };
}
if (streaming) {
return {
completion: await chatModel
.pipe(new BytesOutputParser())
.stream(finalMessages, runConfig),
processTracedEvents,
};
}
return {
completion: await chatModel
.pipe(new StringOutputParser())
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
-24
View File
@@ -1,8 +1,6 @@
import { LlmApiKeys } from "@prisma/client";
import z from "zod";
import { BedrockConfigSchema } from "../../interfaces/customLLMProviderConfigSchemas";
import { TokenCountDelegate } from "../ingestion/processEventBatch";
import { AuthHeaderValidVerificationResult } from "../auth/types";
export type PromptVariable = { name: string; value: string; isUsed: boolean };
@@ -18,7 +16,6 @@ export enum LLMAdapter {
OpenAI = "openai",
Azure = "azure",
Bedrock = "bedrock",
VertexAI = "google-vertex-ai",
}
export enum ChatMessageRole {
@@ -73,7 +70,6 @@ export const ExperimentMetadataSchema = z
provider: z.string(),
model: z.string(),
model_params: ZodModelConfig,
error: z.string().optional(),
})
.strict();
export type ExperimentMetadata = z.infer<typeof ExperimentMetadataSchema>;
@@ -118,19 +114,10 @@ export const anthropicModels = [
"claude-instant-1.2",
] as const;
export const vertexAIModels = [
"gemini-2.0-flash-exp",
"gemini-1.5-pro",
"gemini-1.5-flash",
"gemini-1.0-pro",
] as const;
export type AnthropicModel = (typeof anthropicModels)[number];
export type VertexAIModel = (typeof vertexAIModels)[number];
export const supportedModels = {
[LLMAdapter.Anthropic]: anthropicModels,
[LLMAdapter.OpenAI]: openAIModels,
[LLMAdapter.VertexAI]: vertexAIModels,
[LLMAdapter.Azure]: [],
[LLMAdapter.Bedrock]: [],
} as const;
@@ -151,8 +138,6 @@ export const LLMApiKeySchema = z
provider: z.string(),
displaySecretKey: z.string(),
secretKey: z.string(),
extraHeaders: z.string().nullish(),
extraHeaderKeys: z.array(z.string()),
baseURL: z.string().nullable(),
customModels: z.array(z.string()),
withDefaultModels: z.boolean(),
@@ -166,12 +151,3 @@ export type LLMApiKey =
z.infer<typeof LLMApiKeySchema> extends LlmApiKeys
? z.infer<typeof LLMApiKeySchema>
: never;
export type TraceParams = {
traceName: string;
traceId: string;
projectId: string;
tags: string[];
tokenCountDelegate: TokenCountDelegate;
authCheck: AuthHeaderValidVerificationResult;
};
-13
View File
@@ -1,13 +0,0 @@
import { z } from "zod";
import { decrypt } from "../../encryption";
const ExtraHeaderSchema = z.record(z.string(), z.string());
export function decryptAndParseExtraHeaders(
extraHeaders: string | null | undefined,
) {
if (!extraHeaders) return;
return ExtraHeaderSchema.parse(JSON.parse(decrypt(extraHeaders)));
}
@@ -161,8 +161,8 @@ export class StringOptionsFilter implements Filter {
return {
query:
this.operator === "any of"
? `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} IN ({${varName}: Array(String)})`
: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} NOT IN ({${varName}: Array(String)})`,
? `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`
: `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`,
params: { [varName]: this.values },
};
}
@@ -383,10 +383,6 @@ export class FilterList {
return this.filters.find(predicate);
}
some(predicate: (filter: Filter) => boolean) {
return this.filters.some(predicate);
}
length() {
return this.filters.length;
}
@@ -3,53 +3,31 @@ import { OrderByState } from "../../../interfaces/orderBy";
import { UiColumnMapping } from "../../../tableDefinitions";
import { logger } from "../../logger";
type OrderByStateNotNull = Exclude<OrderByState, null>;
export function orderByToClickhouseSql(
orderBy: OrderByState | OrderByState[] = [],
orderBy: OrderByState,
tableColumns: UiColumnMapping[],
): string {
if (
!orderBy ||
(Array.isArray(orderBy) && orderBy.filter(Boolean).length === 0)
) {
if (!orderBy) {
return "";
}
// Get column definition to map column to internal name, e.g. "t.id"
const col = tableColumns.find(
(c) => c.uiTableName === orderBy.column || c.uiTableId === orderBy.column,
);
if (!Array.isArray(orderBy)) {
orderBy = [orderBy];
}
// Initialize an array to hold order by clauses
const orderByClauses: string[] = [];
// Loop through each orderBy entry
for (const ob of orderBy.filter((o): o is OrderByStateNotNull =>
Boolean(o),
)) {
// Get column definition to map column to internal name, e.g. "t.id"
const col = tableColumns.find(
(c) => c.uiTableName === ob.column || c.uiTableId === ob.column,
);
if (!col) {
logger.warn("Invalid order by column", ob.column);
throw new Error("Invalid order by column: " + ob.column);
}
// Assert that ob.order is either "asc" or "desc"
const orderByOrder = z.enum(["ASC", "DESC"]);
const order = orderByOrder.safeParse(ob.order);
if (!order.success) {
logger.warn("Invalid order", ob.order);
throw new Error("Invalid order: " + ob.order);
}
// Append the order by clause to the array
orderByClauses.push(
`${col.queryPrefix ? col.queryPrefix + "." : ""}${col.clickhouseSelect} ${order.data}`,
);
if (!col) {
logger.warn("Invalid order by column", orderBy.column);
throw new Error("Invalid order by column: " + orderBy.column);
}
// Join all order by clauses with a comma and return
return `ORDER BY ${orderByClauses.join(", ")}`;
// Assert that orderBy.order is either "asc" or "desc"
const orderByOrder = z.enum(["ASC", "DESC"]);
const order = orderByOrder.safeParse(orderBy.order);
if (!order.success) {
logger.warn("Invalid order", orderBy.order);
throw new Error("Invalid order: " + orderBy.order);
}
// Both column and order are safe, can use raw SQL
return `ORDER BY ${col.queryPrefix ? col.queryPrefix + "." : ""}${col.clickhouseSelect} ${order.data}`;
}
@@ -1,10 +1,15 @@
import { ObservationView } from "@prisma/client";
import { ObservationView, Prisma } from "@prisma/client";
import {
datetimeFilterToPrismaSql,
tableColumnsToSqlFilterAndPrefix,
} from "../filterToPrisma";
import { orderByToPrismaSql } from "../orderByToPrisma";
import { observationsTableCols } from "../../observationsTable";
import { TableFilters } from "./types";
type AdditionalObservationFields = {
traceName: string | null;
traceTags: Array<string>;
usageDetails: Record<string, number>;
costDetails: Record<string, number>;
};
export type FullObservation = AdditionalObservationFields & ObservationView;
@@ -19,3 +24,147 @@ export type IOAndMetadataOmittedObservations = Array<
Omit<ObservationView, "input" | "output" | "metadata"> &
AdditionalObservationFields
>;
export function parseGetAllGenerationsInput(filters: TableFilters) {
const searchCondition = filters.searchQuery
? Prisma.sql`AND (
o."id" ILIKE ${`%${filters.searchQuery}%`} OR
o."name" ILIKE ${`%${filters.searchQuery}%`} OR
o."model" ILIKE ${`%${filters.searchQuery}%`} OR
t."name" ILIKE ${`%${filters.searchQuery}%`}
)`
: Prisma.empty;
const filterCondition = tableColumnsToSqlFilterAndPrefix(
filters.filter ?? [],
observationsTableCols,
"observations",
);
const orderByCondition = orderByToPrismaSql(
filters.orderBy,
observationsTableCols,
);
// to improve query performance, add timeseries filter to observation queries as well
const startTimeFilter = filters.filter?.find(
(f) => f.column === "Start Time" && f.type === "datetime",
);
const datetimeFilter =
startTimeFilter && startTimeFilter.type === "datetime"
? datetimeFilterToPrismaSql(
"start_time",
startTimeFilter.operator,
startTimeFilter.value,
)
: Prisma.empty;
return {
searchCondition,
filterCondition,
orderByCondition,
datetimeFilter,
};
}
export function createGenerationsQuery({
projectId,
datetimeFilter = Prisma.empty,
page,
limit,
searchCondition = Prisma.empty,
filterCondition = Prisma.empty,
orderByCondition = Prisma.empty,
selectIOAndMetadata = false,
selectScoreValues = false,
}: {
projectId: string;
datetimeFilter?: Prisma.Sql;
page?: number;
limit?: number;
searchCondition?: Prisma.Sql;
filterCondition?: Prisma.Sql;
orderByCondition?: Prisma.Sql;
selectIOAndMetadata?: boolean;
selectScoreValues?: boolean;
}) {
return Prisma.sql`
WITH scores_avg AS (
SELECT
trace_id,
observation_id,
${selectScoreValues ? Prisma.sql`jsonb_object_agg(name::text, "values") AS "scores_values",` : Prisma.empty}
jsonb_object_agg(name::text, avg_value::double precision) AS "scores_avg"
FROM (
SELECT
trace_id,
observation_id,
name,
${selectScoreValues ? Prisma.sql`array_agg(COALESCE(string_value, value::text)) AS "values",` : Prisma.empty}
avg(value) avg_value,
comment
FROM
scores
WHERE
project_id = ${projectId}
${selectScoreValues ? Prisma.empty : Prisma.sql`AND scores."data_type" IN ('NUMERIC', 'BOOLEAN')`}
GROUP BY
trace_id,
observation_id,
name,
comment
ORDER BY
trace_id
) tmp
GROUP BY
trace_id,
observation_id
)
SELECT
${selectScoreValues ? Prisma.sql`s_avg."scores_values" AS "scores",` : Prisma.empty}
o.id,
o.name,
o.model,
o."modelParameters",
o.start_time as "startTime",
o.end_time as "endTime",
${selectIOAndMetadata ? Prisma.sql`o.input, o.output, o.metadata,` : Prisma.empty}
o.trace_id as "traceId",
t.name as "traceName",
o.completion_start_time as "completionStartTime",
o.time_to_first_token as "timeToFirstToken",
o.prompt_tokens as "promptTokens",
o.completion_tokens as "completionTokens",
o.total_tokens as "totalTokens",
o.unit,
o.level,
o.status_message as "statusMessage",
o.version,
o.model_id as "modelId",
o.input_price as "inputPrice",
o.output_price as "outputPrice",
o.total_price as "totalPrice",
o.calculated_input_cost as "calculatedInputCost",
o.calculated_output_cost as "calculatedOutputCost",
o.calculated_total_cost as "calculatedTotalCost",
o."latency",
o.prompt_id as "promptId",
p.name as "promptName",
p.version as "promptVersion",
t.tags as "traceTags"
FROM observations_view o
JOIN traces t ON t.id = o.trace_id AND t.project_id = ${projectId}
LEFT JOIN scores_avg AS s_avg ON s_avg.trace_id = t.id and s_avg.observation_id = o.id
LEFT JOIN prompts p ON p.id = o.prompt_id AND p.project_id = ${projectId}
WHERE
o.project_id = ${projectId}
AND o.type = 'GENERATION'
${datetimeFilter}
${searchCondition}
${filterCondition}
${orderByCondition}
${limit ? Prisma.sql`LIMIT ${limit}` : Prisma.empty}
${page && limit ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
`;
}
@@ -0,0 +1,135 @@
import { Prisma } from "@prisma/client";
import { TableFilters } from "./types";
import {
datetimeFilterToPrismaSql,
tableColumnsToSqlFilterAndPrefix,
} from "../filterToPrisma";
import { tracesTableCols } from "../../tableDefinitions/tracesTable";
import { orderByToPrismaSql } from "../orderByToPrisma";
export function parseTraceAllFilters(input: TableFilters) {
const filterCondition = tableColumnsToSqlFilterAndPrefix(
input.filter ?? [],
tracesTableCols,
"traces",
);
const orderByCondition = orderByToPrismaSql(input.orderBy, tracesTableCols);
// to improve query performance, add timeseries filter to observation queries as well
const timeseriesFilter = input.filter?.find(
(f) => f.column === "Timestamp" && f.type === "datetime",
);
const observationTimeseriesFilter =
timeseriesFilter && timeseriesFilter.type === "datetime"
? datetimeFilterToPrismaSql(
"start_time",
timeseriesFilter.operator,
timeseriesFilter.value,
)
: Prisma.empty;
const searchCondition = input.searchQuery
? Prisma.sql`AND (
t."id" ILIKE ${`%${input.searchQuery}%`} OR
t."external_id" ILIKE ${`%${input.searchQuery}%`} OR
t."user_id" ILIKE ${`%${input.searchQuery}%`} OR
t."name" ILIKE ${`%${input.searchQuery}%`}
)`
: Prisma.empty;
return {
filterCondition,
orderByCondition,
observationTimeseriesFilter,
searchCondition,
};
}
export function createTracesQuery({
select,
projectId,
observationTimeseriesFilter = Prisma.empty,
page,
limit,
searchCondition = Prisma.empty,
filterCondition = Prisma.empty,
orderByCondition = Prisma.empty,
selectScoreValues = false,
}: {
select: Prisma.Sql;
projectId: string;
observationTimeseriesFilter?: Prisma.Sql;
page?: number;
limit?: number;
searchCondition?: Prisma.Sql;
filterCondition?: Prisma.Sql;
orderByCondition?: Prisma.Sql;
selectScoreValues?: boolean;
}) {
return Prisma.sql`
SELECT
${select}
FROM
"traces" AS t
LEFT JOIN LATERAL (
SELECT
SUM(prompt_tokens) AS "promptTokens",
SUM(completion_tokens) AS "completionTokens",
SUM(total_tokens) AS "totalTokens",
SUM(calculated_total_cost) AS "calculatedTotalCost",
SUM(calculated_input_cost) AS "calculatedInputCost",
SUM(calculated_output_cost) AS "calculatedOutputCost"
FROM
"observations_view"
WHERE
trace_id = t.id
AND "type" = 'GENERATION'
AND "project_id" = ${projectId}
${observationTimeseriesFilter}
) AS generation_metrics ON true
LEFT JOIN LATERAL (
SELECT
COUNT(*) AS "observationCount",
EXTRACT(EPOCH FROM COALESCE(MAX("end_time"), MAX("start_time"))) - EXTRACT(EPOCH FROM MIN("start_time"))::double precision AS "latency",
COALESCE(
MAX(CASE WHEN level = 'ERROR' THEN 'ERROR' END),
MAX(CASE WHEN level = 'WARNING' THEN 'WARNING' END),
MAX(CASE WHEN level = 'DEFAULT' THEN 'DEFAULT' END),
'DEBUG'
) AS "level"
FROM
"observations"
WHERE
trace_id = t.id
AND "project_id" = ${projectId}
${observationTimeseriesFilter}
) AS observation_metrics ON true
LEFT JOIN LATERAL (
SELECT
${selectScoreValues ? Prisma.sql`jsonb_object_agg(name::text, "values") AS "scores_values",` : Prisma.empty}
jsonb_object_agg(name::text, avg_value::double precision) AS "scores_avg"
FROM (
SELECT
name,
${selectScoreValues ? Prisma.sql`array_agg(COALESCE(string_value, value::text)) AS "values",` : Prisma.empty}
AVG(value) avg_value
FROM
scores
WHERE
trace_id = t.id
AND t."project_id" = ${projectId}
${selectScoreValues ? Prisma.empty : Prisma.sql`AND scores."data_type" IN ('NUMERIC', 'BOOLEAN')`}
GROUP BY
name
) tmp
) AS s_avg ON true
WHERE
t."project_id" = ${projectId}
${searchCondition}
${filterCondition}
${orderByCondition}
${limit ? Prisma.sql`LIMIT ${limit}` : Prisma.empty}
${page !== undefined && limit !== undefined ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
`;
}
+3 -1
View File
@@ -1,5 +1,8 @@
export { createSessionsAllQuery } from "./createSessionsAllQuery";
export { createTracesQuery, parseTraceAllFilters } from "./createTracesQuery";
export {
createGenerationsQuery,
parseGetAllGenerationsInput,
type FullObservations,
type FullObservationsWithScores,
type IOAndMetadataOmittedObservations,
@@ -17,4 +20,3 @@ export {
NullFilter,
type ClickhouseOperator,
} from "./clickhouse-sql/clickhouse-filter";
export { orderByToClickhouseSql } from "./clickhouse-sql/orderby-factory";
+76 -72
View File
@@ -1,6 +1,50 @@
import { z } from "zod";
import { eventTypes, ingestionBatchEvent } from ".";
export enum EventName {
TraceUpsert = "TraceUpsert",
BatchExport = "BatchExport",
EvaluationExecution = "EvaluationExecution",
LegacyIngestion = "LegacyIngestion",
CloudUsageMetering = "CloudUsageMetering",
ExperimentCreate = "ExperimentCreate",
}
export const LegacyIngestionEventFull = z.object({
useS3EventStore: z.literal(false),
data: ingestionBatchEvent,
authCheck: z.object({
validKey: z.literal(true),
scope: z.object({
projectId: z.string(),
accessLevel: z.enum(["all", "scores"]),
}),
}),
});
export const LegacyIngestionEventMeta = z.object({
useS3EventStore: z.literal(true),
data: z.array(
z.object({
type: z.nativeEnum(eventTypes),
eventBodyId: z.string(),
eventId: z.string(),
}),
),
authCheck: z.object({
validKey: z.literal(true),
scope: z.object({
projectId: z.string(),
accessLevel: z.enum(["all", "scores"]),
}),
}),
});
export const LegacyIngestionEvent = z.discriminatedUnion("useS3EventStore", [
LegacyIngestionEventFull,
LegacyIngestionEventMeta,
]);
export const IngestionEvent = z.object({
data: z.object({
type: z.nativeEnum(eventTypes),
@@ -19,18 +63,10 @@ export const BatchExportJobSchema = z.object({
projectId: z.string(),
batchExportId: z.string(),
});
export const TraceQueueEventSchema = z.object({
export const TraceUpsertEventSchema = z.object({
projectId: z.string(),
traceId: z.string(),
});
export const TracesQueueEventSchema = z.object({
projectId: z.string(),
traceIds: z.array(z.string()),
});
export const ProjectQueueEventSchema = z.object({
projectId: z.string(),
orgId: z.string(),
});
export const DatasetRunItemUpsertEventSchema = z.object({
projectId: z.string(),
datasetItemId: z.string(),
@@ -42,96 +78,76 @@ export const EvalExecutionEvent = z.object({
jobExecutionId: z.string(),
delay: z.number().nullish(),
});
export const PostHogIntegrationProcessingEventSchema = z.object({
projectId: z.string(),
});
export const ExperimentCreateEventSchema = z.object({
projectId: z.string(),
datasetId: z.string(),
runId: z.string(),
description: z.string().optional(),
});
export const DataRetentionProcessingEventSchema = z.object({
projectId: z.string(),
retention: z.number(),
});
export type BatchExportJobType = z.infer<typeof BatchExportJobSchema>;
export type TraceQueueEventType = z.infer<typeof TraceQueueEventSchema>;
export type TracesQueueEventType = z.infer<typeof TracesQueueEventSchema>;
export type ProjectQueueEventType = z.infer<typeof ProjectQueueEventSchema>;
export type TraceUpsertEventType = z.infer<typeof TraceUpsertEventSchema>;
export type DatasetRunItemUpsertEventType = z.infer<
typeof DatasetRunItemUpsertEventSchema
>;
export type EvalExecutionEventType = z.infer<typeof EvalExecutionEvent>;
export type LegacyIngestionEventType = z.infer<typeof LegacyIngestionEvent>;
export type IngestionEventQueueType = z.infer<typeof IngestionEvent>;
export type ExperimentCreateEventType = z.infer<
typeof ExperimentCreateEventSchema
>;
export type PostHogIntegrationProcessingEventType = z.infer<
typeof PostHogIntegrationProcessingEventSchema
>;
export type DataRetentionProcessingEventType = z.infer<
typeof DataRetentionProcessingEventSchema
>;
export const EventBodySchema = z.union([
z.object({
name: z.literal(EventName.TraceUpsert),
payload: z.array(TraceUpsertEventSchema),
}),
z.object({
name: z.literal(EventName.EvaluationExecution),
payload: EvalExecutionEvent,
}),
z.object({
name: z.literal(EventName.BatchExport),
payload: BatchExportJobSchema,
}),
z.object({
name: z.literal(EventName.ExperimentCreate),
payload: ExperimentCreateEventSchema,
}),
]);
export type EventBodyType = z.infer<typeof EventBodySchema>;
export enum QueueName {
TraceUpsert = "trace-upsert", // Ingestion pipeline adds events on each Trace upsert
TraceDelete = "trace-delete",
ProjectDelete = "project-delete",
EvaluationExecution = "evaluation-execution-queue", // Worker executes Evals
DatasetRunItemUpsert = "dataset-run-item-upsert-queue",
BatchExport = "batch-export-queue",
IngestionQueue = "ingestion-queue", // Process single events with S3-merge
IngestionSecondaryQueue = "secondary-ingestion-queue", // Separates high priority + high throughput projects from other projects.
LegacyIngestionQueue = "legacy-ingestion-queue", // Used for batch processing of Ingestion
CloudUsageMeteringQueue = "cloud-usage-metering-queue",
ExperimentCreate = "experiment-create-queue",
PostHogIntegrationQueue = "posthog-integration-queue",
PostHogIntegrationProcessingQueue = "posthog-integration-processing-queue",
CoreDataS3ExportQueue = "core-data-s3-export-queue",
MeteringDataPostgresExportQueue = "metering-data-postgres-export-queue",
DataRetentionQueue = "data-retention-queue",
DataRetentionProcessingQueue = "data-retention-processing-queue",
}
export enum QueueJobs {
TraceUpsert = "trace-upsert",
TraceDelete = "trace-delete",
ProjectDelete = "project-delete",
DatasetRunItemUpsert = "dataset-run-item-upsert",
EvaluationExecution = "evaluation-execution-job",
BatchExportJob = "batch-export-job",
EnqueueBatchExportJobs = "enqueue-batch-export-jobs",
LegacyIngestionJob = "legacy-ingestion-job",
CloudUsageMeteringJob = "cloud-usage-metering-job",
IngestionJob = "ingestion-job",
IngestionSecondaryJob = "secondary-ingestion-job",
ExperimentCreateJob = "experiment-create-job",
PostHogIntegrationJob = "posthog-integration-job",
PostHogIntegrationProcessingJob = "posthog-integration-processing-job",
CoreDataS3ExportJob = "core-data-s3-export-job",
MeteringDataPostgresExportJob = "metering-data-postgres-export-job",
DataRetentionJob = "data-retention-job",
DataRetentionProcessingJob = "data-retention-processing-job",
}
export type TQueueJobTypes = {
[QueueName.TraceUpsert]: {
timestamp: Date;
id: string;
payload: TraceQueueEventType;
payload: TraceUpsertEventType;
name: QueueJobs.TraceUpsert;
};
[QueueName.TraceDelete]: {
timestamp: Date;
id: string;
payload: TracesQueueEventType | TraceQueueEventType;
name: QueueJobs.TraceDelete;
};
[QueueName.ProjectDelete]: {
timestamp: Date;
id: string;
payload: ProjectQueueEventType;
name: QueueJobs.ProjectDelete;
};
[QueueName.DatasetRunItemUpsert]: {
timestamp: Date;
id: string;
@@ -150,13 +166,13 @@ export type TQueueJobTypes = {
payload: BatchExportJobType;
name: QueueJobs.BatchExportJob;
};
[QueueName.IngestionQueue]: {
[QueueName.LegacyIngestionQueue]: {
timestamp: Date;
id: string;
payload: IngestionEventQueueType;
name: QueueJobs.IngestionJob;
payload: LegacyIngestionEventType;
name: QueueJobs.LegacyIngestionJob;
};
[QueueName.IngestionSecondaryQueue]: {
[QueueName.IngestionQueue]: {
timestamp: Date;
id: string;
payload: IngestionEventQueueType;
@@ -168,16 +184,4 @@ export type TQueueJobTypes = {
payload: ExperimentCreateEventType;
name: QueueJobs.ExperimentCreateJob;
};
[QueueName.PostHogIntegrationProcessingQueue]: {
timestamp: Date;
id: string;
payload: PostHogIntegrationProcessingEventType;
name: QueueJobs.PostHogIntegrationProcessingJob;
};
[QueueName.DataRetentionProcessingQueue]: {
timestamp: Date;
id: string;
payload: DataRetentionProcessingEventType;
name: QueueJobs.DataRetentionProcessingJob;
};
};
@@ -1,8 +1,8 @@
import { Queue } from "bullmq";
import { env } from "../../env";
import { env } from "../..";
import { logger } from "@azure/storage-blob";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class CloudUsageMeteringQueue {
private static instance: Queue | null = null;
@@ -45,7 +45,6 @@ export class CloudUsageMeteringQueue {
QueueJobs.CloudUsageMeteringJob,
{},
{
// Run at minute 5 of every hour (e.g. 1:05, 2:05, 3:05, etc)
repeat: { pattern: "5 * * * *" },
},
);
@@ -1,60 +0,0 @@
import { Queue } from "bullmq";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
import { env } from "../../env";
export class CoreDataS3ExportQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (env.LANGFUSE_S3_CORE_DATA_EXPORT_IS_ENABLED !== "true") {
return null;
}
if (CoreDataS3ExportQueue.instance) {
return CoreDataS3ExportQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
CoreDataS3ExportQueue.instance = newRedis
? new Queue(QueueName.CoreDataS3ExportQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
CoreDataS3ExportQueue.instance?.on("error", (err) => {
logger.error("CoreDataS3ExportQueue error", err);
});
if (CoreDataS3ExportQueue.instance) {
logger.debug("Scheduling jobs for CoreDataS3ExportQueue");
CoreDataS3ExportQueue.instance
.add(
QueueJobs.CoreDataS3ExportJob,
{},
{
repeat: { pattern: "15 3 * * *" }, // every day at 3:15am
},
)
.catch((err) => {
logger.error("Error adding CoreDataS3ExportJob schedule", err);
});
}
return CoreDataS3ExportQueue.instance;
}
}
@@ -1,40 +0,0 @@
import { Queue } from "bullmq";
import { QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class DataRetentionProcessingQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (DataRetentionProcessingQueue.instance) {
return DataRetentionProcessingQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
DataRetentionProcessingQueue.instance = newRedis
? new Queue(QueueName.DataRetentionProcessingQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10000,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
DataRetentionProcessingQueue.instance?.on("error", (err) => {
logger.error("DataRetentionProcessingQueue error", err);
});
return DataRetentionProcessingQueue.instance;
}
}
@@ -1,55 +0,0 @@
import { Queue } from "bullmq";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class DataRetentionQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (DataRetentionQueue.instance) {
return DataRetentionQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
DataRetentionQueue.instance = newRedis
? new Queue(QueueName.DataRetentionQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
DataRetentionQueue.instance?.on("error", (err) => {
logger.error("DataRetentionQueue error", err);
});
if (DataRetentionQueue.instance) {
logger.debug("Scheduling jobs for DataRetentionQueue");
DataRetentionQueue.instance
.add(
QueueJobs.DataRetentionJob,
{},
{
repeat: { pattern: "15 3 * * *" }, // every day at 3:15am
},
)
.catch((err) => {
logger.error("Error adding DataRetentionQueue schedule", err);
});
}
return DataRetentionQueue.instance;
}
}
@@ -29,7 +29,7 @@ export class EvalExecutionQueue {
attempts: 10,
backoff: {
type: "exponential",
delay: 1000,
delay: 5000,
},
},
},
@@ -26,10 +26,10 @@ export class ExperimentCreateQueue {
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 10,
attempts: 2,
backoff: {
type: "exponential",
delay: 1000,
delay: 5000,
},
},
},
+5 -28
View File
@@ -1,23 +1,18 @@
import { Queue } from "bullmq";
import { QueueName } from "../queues";
import { BatchExportQueue } from "./batchExport";
import { CloudUsageMeteringQueue } from "./cloudUsageMeteringQueue";
import { CloudUsageMeteringQueue } from "./CloudUsageMeteringQueue";
import { DatasetRunItemUpsertQueue } from "./datasetRunItemUpsert";
import { EvalExecutionQueue } from "./evalExecutionQueue";
import { ExperimentCreateQueue } from "./experimentCreateQueue";
import { IngestionQueue, SecondaryIngestionQueue } from "./ingestionQueue";
import { IngestionQueue } from "./ingestionQueue";
import { LegacyIngestionQueue } from "./legacyIngestion";
import { TraceUpsertQueue } from "./traceUpsert";
import { TraceDeleteQueue } from "./traceDelete";
import { ProjectDeleteQueue } from "./projectDelete";
import { PostHogIntegrationQueue } from "./postHogIntegrationQueue";
import { PostHogIntegrationProcessingQueue } from "./postHogIntegrationProcessingQueue";
import { CoreDataS3ExportQueue } from "./coreDataS3ExportQueue";
import { MeteringDataPostgresExportQueue } from "./meteringDataPostgresExportQueue";
import { DataRetentionQueue } from "./dataRetentionQueue";
import { DataRetentionProcessingQueue } from "./dataRetentionProcessingQueue";
export function getQueue(queueName: QueueName): Queue | null {
switch (queueName) {
case QueueName.LegacyIngestionQueue:
return LegacyIngestionQueue.getInstance();
case QueueName.BatchExport:
return BatchExportQueue.getInstance();
case QueueName.CloudUsageMeteringQueue:
@@ -30,26 +25,8 @@ export function getQueue(queueName: QueueName): Queue | null {
return ExperimentCreateQueue.getInstance();
case QueueName.TraceUpsert:
return TraceUpsertQueue.getInstance();
case QueueName.TraceDelete:
return TraceDeleteQueue.getInstance();
case QueueName.IngestionQueue:
return IngestionQueue.getInstance();
case QueueName.ProjectDelete:
return ProjectDeleteQueue.getInstance();
case QueueName.PostHogIntegrationQueue:
return PostHogIntegrationQueue.getInstance();
case QueueName.PostHogIntegrationProcessingQueue:
return PostHogIntegrationProcessingQueue.getInstance();
case QueueName.IngestionSecondaryQueue:
return SecondaryIngestionQueue.getInstance();
case QueueName.CoreDataS3ExportQueue:
return CoreDataS3ExportQueue.getInstance();
case QueueName.MeteringDataPostgresExportQueue:
return MeteringDataPostgresExportQueue.getInstance();
case QueueName.DataRetentionQueue:
return DataRetentionQueue.getInstance();
case QueueName.DataRetentionProcessingQueue:
return DataRetentionProcessingQueue.getInstance();
default:
const exhaustiveCheckDefault: never = queueName;
throw new Error(`Queue ${queueName} not found`);
@@ -43,45 +43,3 @@ export class IngestionQueue {
return IngestionQueue.instance;
}
}
export class SecondaryIngestionQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.IngestionSecondaryQueue]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.IngestionSecondaryQueue]
> | null {
if (SecondaryIngestionQueue.instance)
return SecondaryIngestionQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
SecondaryIngestionQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.IngestionSecondaryQueue]>(
QueueName.IngestionSecondaryQueue,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
SecondaryIngestionQueue.instance?.on("error", (err) => {
logger.error("SecondaryIngestionQueue error", err);
});
return SecondaryIngestionQueue.instance;
}
}
@@ -1,31 +1,31 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import { QueueName, TQueueJobTypes } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class ProjectDeleteQueue {
export class LegacyIngestionQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.ProjectDelete]
TQueueJobTypes[QueueName.LegacyIngestionQueue]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.ProjectDelete]
TQueueJobTypes[QueueName.LegacyIngestionQueue]
> | null {
if (ProjectDeleteQueue.instance) return ProjectDeleteQueue.instance;
if (LegacyIngestionQueue.instance) return LegacyIngestionQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
ProjectDeleteQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.ProjectDelete]>(
QueueName.ProjectDelete,
LegacyIngestionQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.LegacyIngestionQueue]>(
QueueName.LegacyIngestionQueue,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
removeOnFail: 500_000,
attempts: 5,
backoff: {
type: "exponential",
@@ -36,10 +36,10 @@ export class ProjectDeleteQueue {
)
: null;
ProjectDeleteQueue.instance?.on("error", (err) => {
logger.error("ProjectDeleteQueue error", err);
LegacyIngestionQueue.instance?.on("error", (err) => {
logger.error("LegacyIngestionQueue error", err);
});
return ProjectDeleteQueue.instance;
return LegacyIngestionQueue.instance;
}
}
@@ -1,63 +0,0 @@
import { Queue } from "bullmq";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
import { env } from "../../env";
export class MeteringDataPostgresExportQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (env.LANGFUSE_POSTGRES_METERING_DATA_EXPORT_IS_ENABLED !== "true") {
return null;
}
if (MeteringDataPostgresExportQueue.instance) {
return MeteringDataPostgresExportQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
MeteringDataPostgresExportQueue.instance = newRedis
? new Queue(QueueName.MeteringDataPostgresExportQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
MeteringDataPostgresExportQueue.instance?.on("error", (err) => {
logger.error("MeteringDataPostgresExportQueue error", err);
});
if (MeteringDataPostgresExportQueue.instance) {
logger.debug("Scheduling jobs for MeteringDataPostgresExportQueue");
MeteringDataPostgresExportQueue.instance
.add(
QueueJobs.MeteringDataPostgresExportJob,
{},
{
repeat: { pattern: "30 2 * * *" }, // every day at 2:30am UTC
},
)
.catch((err) => {
logger.error(
"Error adding MeteringDataPostgresExportJob schedule",
err,
);
});
}
return MeteringDataPostgresExportQueue.instance;
}
}
@@ -1,40 +0,0 @@
import { Queue } from "bullmq";
import { QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class PostHogIntegrationProcessingQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (PostHogIntegrationProcessingQueue.instance) {
return PostHogIntegrationProcessingQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
PostHogIntegrationProcessingQueue.instance = newRedis
? new Queue(QueueName.PostHogIntegrationProcessingQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
PostHogIntegrationProcessingQueue.instance?.on("error", (err) => {
logger.error("PostHogIntegrationProcessingQueue error", err);
});
return PostHogIntegrationProcessingQueue.instance;
}
}
@@ -1,55 +0,0 @@
import { Queue } from "bullmq";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class PostHogIntegrationQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (PostHogIntegrationQueue.instance) {
return PostHogIntegrationQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
PostHogIntegrationQueue.instance = newRedis
? new Queue(QueueName.PostHogIntegrationQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
PostHogIntegrationQueue.instance?.on("error", (err) => {
logger.error("PostHogIntegrationQueue error", err);
});
if (PostHogIntegrationQueue.instance) {
logger.debug("Scheduling jobs for PostHogIntegrationQueue");
PostHogIntegrationQueue.instance
.add(
QueueJobs.PostHogIntegrationJob,
{},
{
repeat: { pattern: "30 * * * *" }, // every hour at 30 minutes past
},
)
.catch((err) => {
logger.error("Error adding PostHogIntegrationJob schedule", err);
});
}
return PostHogIntegrationQueue.instance;
}
}
@@ -1,44 +0,0 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class TraceDeleteQueue {
private static instance: Queue<TQueueJobTypes[QueueName.TraceDelete]> | null =
null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.TraceDelete]
> | null {
if (TraceDeleteQueue.instance) return TraceDeleteQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
TraceDeleteQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.TraceDelete]>(
QueueName.TraceDelete,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
TraceDeleteQueue.instance?.on("error", (err) => {
logger.error("TraceDeleteQueue error", err);
});
return TraceDeleteQueue.instance;
}
}
@@ -1,4 +1,10 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { randomUUID } from "crypto";
import {
QueueJobs,
QueueName,
TQueueJobTypes,
TraceUpsertEventType,
} from "../queues";
import { Queue } from "bullmq";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
@@ -26,7 +32,7 @@ export class TraceUpsertQueue {
removeOnComplete: 100, // Important: If not true, new jobs for that ID would be ignored as jobs in the complete set are still considered as part of the queue
removeOnFail: 100_000,
attempts: 5,
delay: 15_000, // 15 seconds
delay: 10_000, // 10 seconds
backoff: {
type: "exponential",
delay: 5000,
@@ -4,7 +4,7 @@ import {
convertDateToClickhouseDateTime,
} from "../clickhouse/client";
import { logger } from "../logger";
import { getTracer, instrumentAsync } from "../instrumentation";
import { instrumentAsync } from "../instrumentation";
import {
StorageService,
StorageServiceFactory,
@@ -12,7 +12,6 @@ import {
import { randomUUID } from "crypto";
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
import { NodeClickHouseClientConfigOptions } from "@clickhouse/client/dist/config";
import { context, trace } from "@opentelemetry/api";
let s3StorageServiceClient: StorageService;
@@ -41,31 +40,37 @@ export async function upsertClickhouse<
// https://opentelemetry.io/docs/specs/semconv/database/database-spans/
span.setAttribute("ch.query.table", opts.table);
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
await Promise.all(
opts.records.map((record) => {
// drop trailing s and pretend it's always a create.
// Only applicable to scores and traces.
let eventType = `${opts.table.slice(0, -1)}-create`;
if (opts.table === "observations") {
// @ts-ignore - If it's an observation we now that `type` is a string
eventType = `${record["type"].toLowerCase()}-create`;
}
s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${record.project_id}/${getClickhouseEntityType(eventType)}/${record.id}/${randomUUID()}.json`,
[
{
id: randomUUID(),
timestamp: new Date().toISOString(),
type: eventType,
body: opts.eventBodyMapper(record),
},
],
);
}),
);
// If event upload is enabled, we store all rows in S3 to have a backup
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
throw new Error("S3 event store is enabled but no bucket is set");
}
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
await Promise.all(
opts.records.map((record) => {
// drop trailing s and pretend it's always a create.
// Only applicable to scores and traces.
let eventType = `${opts.table.slice(0, -1)}-create`;
if (opts.table === "observations") {
// @ts-ignore - If it's an observation we now that `type` is a string
eventType = `${record["type"].toLowerCase()}-create`;
}
s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${record.project_id}/${getClickhouseEntityType(eventType)}/${record.id}/${randomUUID()}.json`,
[
{
id: randomUUID(),
timestamp: new Date().toISOString(),
type: eventType,
body: opts.eventBodyMapper(record),
},
],
);
}),
);
}
const res = await clickhouseClient().insert({
table: opts.table,
@@ -102,64 +107,6 @@ export async function upsertClickhouse<
});
}
export async function* queryClickhouseStream<T>(opts: {
query: string;
params?: Record<string, unknown> | undefined;
clickhouseConfigs?: NodeClickHouseClientConfigOptions;
}): AsyncGenerator<T> {
const tracer = getTracer("clickhouse-query-stream");
const span = tracer.startSpan("clickhouse-query-stream");
try {
const res = await context.with(
trace.setSpan(context.active(), span),
async () => {
// https://opentelemetry.io/docs/specs/semconv/database/database-spans/
span.setAttribute("ch.query.text", opts.query);
const res = await clickhouseClient(opts.clickhouseConfigs).query({
query: opts.query,
format: "JSONEachRow",
query_params: opts.params,
});
// same logic as for prisma. we want to see queries in development
if (env.NODE_ENV === "development") {
logger.info(`clickhouse:query ${res.query_id} ${opts.query}`);
}
span.setAttribute("ch.queryId", res.query_id);
// add summary headers to the span. Helps to tune performance
const summaryHeader = res.response_headers["x-clickhouse-summary"];
if (summaryHeader) {
try {
const summary = Array.isArray(summaryHeader)
? JSON.parse(summaryHeader[0])
: JSON.parse(summaryHeader);
for (const key in summary) {
span.setAttribute(`ch.${key}`, summary[key]);
}
} catch (error) {
logger.debug(
`Failed to parse clickhouse summary header ${summaryHeader}`,
error,
);
}
}
return res;
},
);
for await (const rows of res.stream<T>()) {
for (const row of rows) {
yield row.json();
}
}
} finally {
span.end();
}
}
export async function queryClickhouse<T>(opts: {
query: string;
params?: Record<string, unknown> | undefined;
@@ -1,7 +1,4 @@
import {
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
} from "./clickhouse";
import { queryClickhouse } from "./clickhouse";
import { createFilterFromFilterState } from "../queries/clickhouse-sql/factory";
import { FilterState } from "../../types";
import {
@@ -145,26 +142,23 @@ export const getScoreAggregate = async (
export const groupTracesByTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
).apply();
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"timestamp",
);
const query = `
SELECT
${selectTimeseriesColumn(bucketSizeInSeconds, "timestamp", "timestamp")},
${selectTimeseriesColumn(groupBy, "timestamp", "timestamp")},
count(*) as count
FROM traces t FINAL
WHERE project_id = {projectId: String}
AND ${chFilter.query}
GROUP BY timestamp
${orderByQuery}
${orderByTimeSeries(groupBy, "timestamp")}
`;
const result = await queryClickhouse<{
timestamp: string;
count: string;
@@ -173,12 +167,11 @@ export const groupTracesByTime = async (
params: {
projectId,
...chFilter.params,
...orderByParams,
},
});
return result.map((row) => ({
timestamp: parseClickhouseUTCDateTimeFormat(row.timestamp),
timestamp: new Date(row.timestamp),
countTraceId: Number(row.count),
}));
};
@@ -186,6 +179,7 @@ export const groupTracesByTime = async (
export const getObservationUsageByTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
@@ -203,16 +197,11 @@ export const getObservationUsageByTime = async (
) as DateTimeFilter | undefined)
: undefined;
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"start_time",
);
const query = `
SELECT
${selectTimeseriesColumn(bucketSizeInSeconds, "start_time", "start_time")},
sumMap(usage_details) as units,
sumMap(cost_details) as cost,
${selectTimeseriesColumn(groupBy, "start_time", "start_time")},
sumMap(usage_details)['total'] as sum_usage_details,
sumMap(cost_details)['total'] as sum_cost_details,
provided_model_name
FROM observations o FINAL
${tracesFilter ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
@@ -220,20 +209,19 @@ export const getObservationUsageByTime = async (
AND ${appliedFilter.query}
${timeFilter ? `AND t.timestamp >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
GROUP BY start_time, provided_model_name
${orderByQuery}
${orderByTimeSeries(groupBy, "start_time")}
`;
const result = await queryClickhouse<{
start_time: string;
units: Record<string, number>;
cost: Record<string, number>;
sum_usage_details: string;
sum_cost_details: number;
provided_model_name: string;
}>({
query,
params: {
projectId,
...appliedFilter.params,
...orderByParams,
...(timeFilter
? { traceTimestamp: convertDateToClickhouseDateTime(timeFilter.value) }
: {}),
@@ -241,19 +229,9 @@ export const getObservationUsageByTime = async (
});
return result.map((row) => ({
start_time: parseClickhouseUTCDateTimeFormat(row.start_time),
units: Object.fromEntries(
Object.entries(row.units ?? {}).map(([key, value]) => [
key,
Number(value),
]),
),
cost: Object.fromEntries(
Object.entries(row.cost ?? {}).map(([key, value]) => [
key,
Number(value),
]),
),
start_time: new Date(row.start_time),
sum_usage_details: Number(row.sum_usage_details),
sum_cost_details: row.sum_cost_details,
provided_model_name: row.provided_model_name,
}));
};
@@ -305,6 +283,7 @@ export const getDistinctModels = async (
export const getScoresAggregateOverTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
@@ -315,14 +294,9 @@ export const getScoresAggregateOverTime = async (
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
// TODO: Validate whether we can filter traces on timestamp here.
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"timestamp",
);
const query = `
SELECT
${selectTimeseriesColumn(bucketSizeInSeconds, "timestamp", "timestamp")},
${selectTimeseriesColumn(groupBy, "timestamp", "timestamp")},
name,
data_type,
source,
@@ -337,7 +311,7 @@ export const getScoresAggregateOverTime = async (
name,
data_type,
source
${orderByQuery};
${orderByTimeSeries(groupBy, "timestamp")};
`;
const result = await queryClickhouse<{
@@ -351,12 +325,11 @@ export const getScoresAggregateOverTime = async (
params: {
projectId,
...appliedFilter.params,
...orderByParams,
},
});
return result.map((row) => ({
scoreTimestamp: parseClickhouseUTCDateTimeFormat(row.timestamp),
scoreTimestamp: new Date(row.timestamp),
scoreName: row.name,
scoreDataType: row.data_type,
scoreSource: row.source,
@@ -432,7 +405,7 @@ export const getObservationLatencies = async (
// Skipping FINAL here, as the quantiles are approximate to begin with.
const query = `
SELECT
quantiles(0.5, 0.9, 0.95, 0.99)(date_diff('millisecond', o.start_time, o.end_time)) as quantiles,
quantiles(0.5, 0.9, 0.95, 0.99)(date_diff('milliseconds', o.start_time, o.end_time)) as quantiles,
name
FROM observations o
${chFilter.find((f) => f.clickhouseTable === "traces") ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
@@ -479,7 +452,7 @@ export const getTracesLatencies = async (
select o.trace_id,
t.name,
o.project_id,
date_diff('millisecond', min(o.start_time), coalesce(max(o.end_time), max(o.start_time))) as duration
date_diff('milliseconds', min(o.start_time), coalesce(max(o.end_time), max(o.start_time))) as duration
FROM traces t
JOIN observations o
ON o.trace_id = t.id AND o.project_id = t.project_id
@@ -520,6 +493,7 @@ export const getTracesLatencies = async (
export const getModelLatenciesOverTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
@@ -529,33 +503,25 @@ export const getModelLatenciesOverTime = async (
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"start_time_bucket",
);
// Skipping FINAL here, as the quantiles are approximate to begin with.
const query = `
SELECT
${selectTimeseriesColumn(bucketSizeInSeconds, "o.start_time", "start_time_bucket")},
${selectTimeseriesColumn(groupBy, "o.start_time", "start_time_bucket")},
provided_model_name,
quantiles(0.5, 0.75, 0.9, 0.95, 0.99)(date_diff('millisecond', o.start_time, o.end_time)) as quantiles
quantiles(0.5, 0.75, 0.9, 0.95, 0.99)(date_diff('milliseconds', o.start_time, o.end_time)) as quantiles
FROM observations o
${traceFilter ? "JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
GROUP BY provided_model_name, start_time_bucket
${orderByQuery};
${orderByTimeSeries(groupBy, "start_time_bucket")};
`;
const result = await queryClickhouse<{
start_time_bucket: string;
provided_model_name: string;
quantiles: string[];
}>({
query,
params: { projectId, ...appliedFilter.params, ...orderByParams },
});
}>({ query, params: { projectId, ...appliedFilter.params } });
return result.map((row) => ({
p50: Number(row.quantiles[0]) / 1000,
@@ -564,13 +530,14 @@ export const getModelLatenciesOverTime = async (
p95: Number(row.quantiles[3]) / 1000,
p99: Number(row.quantiles[4]) / 1000,
model: row.provided_model_name,
start_time: parseClickhouseUTCDateTimeFormat(row.start_time_bucket),
start_time: new Date(row.start_time_bucket),
}));
};
export const getNumericScoreTimeSeries = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
@@ -579,14 +546,9 @@ export const getNumericScoreTimeSeries = async (
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"score_timestamp",
);
const query = `
SELECT
${selectTimeseriesColumn(bucketSizeInSeconds, "s.timestamp", "score_timestamp")},
${selectTimeseriesColumn(groupBy, "s.timestamp", "score_timestamp")},
s.name as score_name,
AVG(s.value) as avg_value
FROM scores s final
@@ -594,11 +556,11 @@ export const getNumericScoreTimeSeries = async (
WHERE s.project_id = {projectId: String}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
GROUP BY score_name, score_timestamp
${orderByQuery}
${orderByTimeSeries(groupBy, "score_timestamp")}
`;
const result = await queryClickhouse<{
score_timestamp: string;
return queryClickhouse<{
score_timestamp: Date;
score_name: string;
avg_value: number;
}>({
@@ -606,20 +568,14 @@ export const getNumericScoreTimeSeries = async (
params: {
projectId,
...(chFilterRes ? chFilterRes.params : {}),
...orderByParams,
},
});
return result.map((row) => ({
scoreTimestamp: parseClickhouseUTCDateTimeFormat(row.score_timestamp),
scoreName: row.score_name,
avgValue: Number(row.avg_value),
}));
};
export const getCategoricalScoreTimeSeries = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc | undefined,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
@@ -628,14 +584,9 @@ export const getCategoricalScoreTimeSeries = async (
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"score_timestamp",
);
const query = `
SELECT
${bucketSizeInSeconds ? selectTimeseriesColumn(bucketSizeInSeconds, "s.timestamp", "score_timestamp") + ", " : ""}
${groupBy ? selectTimeseriesColumn(groupBy, "s.timestamp", "score_timestamp") + ", " : ""}
s.name as score_name,
s.data_type as score_data_type,
s.source as score_source,
@@ -645,12 +596,12 @@ export const getCategoricalScoreTimeSeries = async (
${traceFilter ? "JOIN traces t ON s.trace_id = t.id AND s.project_id = t.project_id" : ""}
WHERE s.project_id = {projectId: String}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
GROUP BY score_name, score_data_type, score_source, score_value ${bucketSizeInSeconds ? ", score_timestamp" : ""}
${orderByQuery}
GROUP BY score_name, score_data_type, score_source, score_value ${groupBy ? ", score_timestamp" : ""}
${groupBy ? orderByTimeSeries(groupBy, "score_timestamp") : ""}
`;
const result = await queryClickhouse<{
score_timestamp?: string;
return queryClickhouse<{
score_timestamp?: Date;
score_name: string;
score_data_type: string;
score_source: string;
@@ -661,138 +612,65 @@ export const getCategoricalScoreTimeSeries = async (
params: {
projectId,
...(chFilterRes ? chFilterRes.params : {}),
...orderByParams,
},
});
return result.map((row) => ({
scoreTimestamp: row.score_timestamp
? parseClickhouseUTCDateTimeFormat(row.score_timestamp)
: undefined,
scoreName: row.score_name,
scoreDataType: row.score_data_type,
scoreSource: row.score_source,
scoreValue: row.score_value,
count: Number(row.count),
}));
};
export const getObservationsStatusTimeSeries = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const chFilterRes = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const [orderByQuery, orderByParams, bucketSizeInSeconds] = orderByTimeSeries(
filter,
"start_time_bucket",
);
const query = `
SELECT
${bucketSizeInSeconds ? selectTimeseriesColumn(bucketSizeInSeconds, "o.start_time", "start_time_bucket") + ", " : ""}
count(*) as observation_count,
level as level
FROM observations o
${traceFilter ? "JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND o.level IS NOT NULL
AND ${chFilterRes?.query}
GROUP BY level ${bucketSizeInSeconds ? ", start_time_bucket" : ""}
${orderByQuery}
`;
const result = await queryClickhouse<{
start_time_bucket?: string;
observation_count: string;
level: string;
}>({
query,
params: {
projectId,
...(chFilterRes ? chFilterRes.params : {}),
...orderByParams,
},
});
return result.map((row) => ({
start_time_bucket: row.start_time_bucket
? parseClickhouseUTCDateTimeFormat(row.start_time_bucket)
: undefined,
count: Number(row.observation_count),
level: row.level,
}));
};
export const orderByTimeSeries = (
filter: FilterState,
col: string,
): [string, { fromTime: number; toTime: number }, number] => {
const potentialBucketSizesSeconds = [
5, 10, 30, 60, 300, 600, 1800, 3600, 18000, 36000, 86400, 604800, 2592000,
];
// Calculate time difference in seconds
const [from, to] = extractFromAndToTimestampsFromFilter(filter);
if (!from || !to) {
throw new Error("Time Filter is required for time series queries");
const orderByTimeSeries = (dateTrunc: DateTrunc, col: string) => {
let interval;
switch (dateTrunc) {
case "year":
interval = "toIntervalYear(1)";
break;
case "month":
interval = "toIntervalMonth(1)";
break;
case "week":
interval = "toIntervalWeek(1)";
break;
case "day":
interval = "toIntervalDay(1)";
break;
case "hour":
interval = "toIntervalHour(1)";
break;
case "minute":
interval = "toIntervalMinute(1)";
break;
default:
return undefined;
}
const fromDate = new Date(from.value as Date);
const toDate = new Date(to.value as Date);
const diffInSeconds = Math.abs(toDate.getTime() - fromDate.getTime()) / 1000;
// choose the bucket size that is the closest to the desired number of buckets
const bucketSizeInSeconds = potentialBucketSizesSeconds.reduce(
(closest, size) => {
const diffFromDesiredBuckets = Math.abs(diffInSeconds / size - 50);
return diffFromDesiredBuckets < closest.diffFromDesiredBuckets
? { size, diffFromDesiredBuckets }
: closest;
},
{ size: 0, diffFromDesiredBuckets: Infinity },
).size;
// Convert to interval string
const interval = `toIntervalSecond(${bucketSizeInSeconds})`;
return [
`ORDER BY ${col} ASC
WITH FILL
FROM toStartOfInterval(toDateTime({fromTime: DateTime64(3)}), INTERVAL ${bucketSizeInSeconds} SECOND)
TO toDateTime({toTime: DateTime64(3)}) + INTERVAL ${bucketSizeInSeconds} SECOND
STEP ${interval}`,
{ fromTime: fromDate.getTime(), toTime: toDate.getTime() },
bucketSizeInSeconds,
];
return `ORDER BY ${col} ASC WITH FILL STEP ${interval}`;
};
export const selectTimeseriesColumn = (
bucketSizeInSeconds: number,
const selectTimeseriesColumn = (
dateTrunc: DateTrunc,
col: string,
as: String,
) => {
return `toStartOfInterval(${col}, INTERVAL ${bucketSizeInSeconds} SECOND) as ${as}`;
};
export const extractFromAndToTimestampsFromFilter = (filter?: FilterState) => {
if (!filter)
throw new Error("Time Filter is required for time series queries");
const fromTimestamp = filter.filter(
(f) => f.type === "datetime" && (f.operator === ">" || f.operator === ">="),
);
const toTimestamp = filter.filter(
(f) => f.type === "datetime" && (f.operator === "<" || f.operator === "<="),
);
return [fromTimestamp[0], toTimestamp[0]];
let interval;
switch (dateTrunc) {
case "year":
interval = "toStartOfYear";
break;
case "month":
interval = "toStartOfMonth";
break;
case "week":
interval = "toStartOfWeek";
break;
case "day":
interval = "toStartOfDay";
break;
case "hour":
interval = "toStartOfHour";
break;
case "minute":
interval = "toStartOfMinute";
break;
default:
return undefined;
}
return `${interval}(${col}) as ${as}`;
};
@@ -7,5 +7,3 @@ export * from "./traces_converters";
export * from "./scores_converters";
export * from "./observations_converters";
export * from "./clickhouse";
export * from "./constants";
export * from "./trace-sessions";
@@ -2,7 +2,6 @@ import {
commandClickhouse,
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
queryClickhouseStream,
upsertClickhouse,
} from "./clickhouse";
import { ObservationLevel } from "@prisma/client";
@@ -39,7 +38,6 @@ import {
OBSERVATIONS_TO_TRACE_INTERVAL,
TRACE_TO_OBSERVATIONS_INTERVAL,
} from "./constants";
import { env } from "../../env";
export const checkObservationExists = async (
projectId: string,
@@ -212,13 +210,11 @@ export const getObservationById = async (
id: string,
projectId: string,
fetchWithInputOutput: boolean = false,
startTime?: Date,
) => {
const records = await getObservationByIdInternal(
id,
projectId,
fetchWithInputOutput,
startTime,
);
const mapped = records.map(convertObservation);
@@ -315,7 +311,6 @@ const getObservationByIdInternal = async (
id: string,
projectId: string,
fetchWithInputOutput: boolean = false,
startTime?: Date,
) => {
const query = `
SELECT
@@ -350,18 +345,11 @@ const getObservationByIdInternal = async (
FROM observations
WHERE id = {id: String}
AND project_id = {projectId: String}
${startTime ? `AND start_time = {startTime: DateTime64(3)}` : ""}
ORDER BY event_ts desc
LIMIT 1 by id, project_id`;
return await queryClickhouse<ObservationRecordReadType>({
query,
params: {
id,
projectId,
...(startTime
? { startTime: convertDateToClickhouseDateTime(startTime) }
: {}),
},
params: { id, projectId },
});
};
@@ -401,7 +389,7 @@ export type ObservationsTableQueryResult = ObservationRecordReadType & {
export const getObservationsTableCount = async (opts: ObservationTableQuery) =>
getObservationsTableInternal<TableCount>({
...opts,
select: "count",
select: "count(*) as count",
});
export type ObservationsTableRow = Omit<
@@ -415,11 +403,33 @@ export const getObservationsTable = async (
const observationRecords = await getObservationsTableInternal<
Omit<
ObservationsTableQueryResult,
"trace_tags" | "trace_name" | "trace_user_id"
"trace_tags" | "trace_name" | "trace_user_id" | "type"
>
>({
...opts,
select: "rows",
select: `
o.id as id,
o.name as name,
o."model_parameters" as model_parameters,
o.start_time as "start_time",
o.end_time as "end_time",
o.trace_id as "trace_id",
o.completion_start_time as "completion_start_time",
o.provided_usage_details as "provided_usage_details",
o.usage_details as "usage_details",
o.provided_cost_details as "provided_cost_details",
o.cost_details as "cost_details",
o.level as level,
o.status_message as "status_message",
o.version as version,
o.parent_observation_id as "parent_observation_id",
o.created_at as "created_at",
o.updated_at as "updated_at",
o.provided_model_name as "provided_model_name",
o.total_cost as "total_cost",
internal_model_id as "internal_model_id",
if(isNull(end_time), NULL, date_diff('milliseconds', start_time, end_time)) as latency,
if(isNull(completion_start_time), NULL, date_diff('milliseconds', start_time, completion_start_time)) as "time_to_first_token"`,
});
const traces = await getTracesByIds(
@@ -432,7 +442,7 @@ export const getObservationsTable = async (
return observationRecords.map((o) => {
const trace = traces.find((t) => t.id === o.trace_id);
return {
...convertObservationToView(o),
...convertObservationToView({ ...o, type: "GENERATION" }),
latency: o.latency ? Number(o.latency) / 1000 : null,
timeToFirstToken: o.time_to_first_token
? Number(o.time_to_first_token) / 1000
@@ -450,11 +460,33 @@ export const getObservationsTableWithModelData = async (
const observationRecords = await getObservationsTableInternal<
Omit<
ObservationsTableQueryResult,
"trace_tags" | "trace_name" | "trace_user_id"
"trace_tags" | "trace_name" | "trace_user_id" | "type"
>
>({
...opts,
select: "rows",
select: `
o.id as id,
o.name as name,
o."model_parameters" as model_parameters,
o.start_time as "start_time",
o.end_time as "end_time",
o.trace_id as "trace_id",
o.completion_start_time as "completion_start_time",
o.provided_usage_details as "provided_usage_details",
o.usage_details as "usage_details",
o.provided_cost_details as "provided_cost_details",
o.cost_details as "cost_details",
o.level as level,
o.status_message as "status_message",
o.version as version,
o.parent_observation_id as "parent_observation_id",
o.created_at as "created_at",
o.updated_at as "updated_at",
o.provided_model_name as "provided_model_name",
o.total_cost as "total_cost",
internal_model_id as "internal_model_id",
if(isNull(end_time), NULL, date_diff('milliseconds', start_time, end_time)) as latency,
if(isNull(completion_start_time), NULL, date_diff('milliseconds', start_time, completion_start_time)) as "time_to_first_token"`,
});
const uniqueModels: string[] = Array.from(
@@ -491,7 +523,7 @@ export const getObservationsTableWithModelData = async (
const trace = traces.find((t) => t.id === o.trace_id);
const model = models.find((m) => m.id === o.internal_model_id);
return {
...convertObservationToView(o),
...convertObservationToView({ ...o, type: "GENERATION" }),
latency: o.latency ? Number(o.latency) / 1000 : null,
timeToFirstToken: o.time_to_first_token
? Number(o.time_to_first_token) / 1000
@@ -511,49 +543,17 @@ export const getObservationsTableWithModelData = async (
};
const getObservationsTableInternal = async <T>(
opts: ObservationTableQuery & { select: "count" | "rows" },
opts: ObservationTableQuery & { select: string },
): Promise<Array<T>> => {
const select =
opts.select === "count"
? "count(*) as count"
: `
o.id as id,
o.type as type,
o.project_id as "project_id",
o.name as name,
o."model_parameters" as model_parameters,
o.start_time as "start_time",
o.end_time as "end_time",
o.trace_id as "trace_id",
o.completion_start_time as "completion_start_time",
o.provided_usage_details as "provided_usage_details",
o.usage_details as "usage_details",
o.provided_cost_details as "provided_cost_details",
o.cost_details as "cost_details",
o.level as level,
o.status_message as "status_message",
o.version as version,
o.parent_observation_id as "parent_observation_id",
o.created_at as "created_at",
o.updated_at as "updated_at",
o.provided_model_name as "provided_model_name",
o.total_cost as "total_cost",
o.prompt_id as "prompt_id",
o.prompt_name as "prompt_name",
o.prompt_version as "prompt_version",
internal_model_id as "internal_model_id",
if(isNull(end_time), NULL, date_diff('millisecond', start_time, end_time)) as latency,
if(isNull(completion_start_time), NULL, date_diff('millisecond', start_time, completion_start_time)) as "time_to_first_token"`;
const { projectId, filter, selectIOAndMetadata, limit, offset, orderBy } =
opts;
const selectString = selectIOAndMetadata
? `
${select},
${opts.select},
${selectIOAndMetadata ? `o.input, o.output, o.metadata` : ""}
`
: select;
: opts.select;
const scoresFilter = new FilterList([
new StringFilter({
@@ -580,10 +580,6 @@ const getObservationsTableInternal = async <T>(
.includes(f.column),
);
const hasScoresFilter = filter.some(
(f) => f.column === "Scores" || f.column === "scores",
);
const orderByTraces = opts.orderBy
? observationsTableTraceUiColumnDefinitions
.map((c) => c.uiTableId)
@@ -654,58 +650,63 @@ const getObservationsTableInternal = async <T>(
observation_id
)`;
// if we have default ordering by time, we order by toDate(o.start_time) first and then by
// o.start_time. This way, clickhouse is able to read more efficiently directly from disk without ordering
const newDefaultOrder =
orderBy?.column === "startTime"
? [{ column: "order_by_date", order: orderBy.order }, orderBy]
: [orderBy ?? null];
if (traceTableFilter.length > 0 || orderByTraces) {
// joins with traces are very expensive. We need to filter by time as well.
// We assume that a trace has to have been within the last 2 days to be relevant.
const chOrderBy = orderByToClickhouseSql(newDefaultOrder, [
...observationsTableUiColumnDefinitions,
{
uiTableName: "order_by_date",
uiTableId: "order_by_date",
clickhouseTableName: "observation",
clickhouseSelect: "toDate(o.start_time)",
},
]);
// joins with traces are very expensive. We need to filter by time as well.
// We assume that a trace has to have been within the last 2 days to be relevant.
const query = `
const query = `
${scoresCte}
SELECT
${selectString}
FROM observations o
${traceTableFilter.length > 0 || orderByTraces || search.query ? "LEFT JOIN traces t FINAL ON t.id = o.trace_id AND t.project_id = o.project_id" : ""}
${hasScoresFilter ? `LEFT JOIN scores_avg AS s_avg ON s_avg.trace_id = o.trace_id and s_avg.observation_id = o.id` : ""}
FROM observations o FINAL
LEFT JOIN traces t FINAL ON t.id = o.trace_id AND t.project_id = o.project_id
LEFT JOIN scores_avg AS s_avg ON s_avg.trace_id = o.trace_id and s_avg.observation_id = o.id
WHERE ${appliedObservationsFilter.query}
${timeFilter && (traceTableFilter.length > 0 || orderByTraces) ? `AND t.timestamp > {tracesTimestampFilter: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
AND o.type = 'GENERATION'
${timeFilter ? `AND t.timestamp > {tracesTimestampFilter: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
${search.query}
${chOrderBy}
${opts.select === "rows" ? "LIMIT 1 BY o.id, o.project_id" : ""}
${orderByToClickhouseSql(orderBy ?? null, observationsTableUiColumnDefinitions)}
${limit !== undefined && offset !== undefined ? `LIMIT ${limit} OFFSET ${offset}` : ""};`;
const res = await queryClickhouse<T>({
query,
params: {
...appliedScoresFilter.params,
...appliedObservationsFilter.params,
...(timeFilter
? {
tracesTimestampFilter: convertDateToClickhouseDateTime(
timeFilter.value as Date,
),
}
: {}),
...search.params,
},
});
const res = await queryClickhouse<T>({
query,
params: {
...appliedScoresFilter.params,
...appliedObservationsFilter.params,
...(timeFilter
? {
tracesTimestampFilter: convertDateToClickhouseDateTime(
timeFilter.value as Date,
),
}
: {}),
...search.params,
},
});
return res;
return res;
} else {
// we query by T, which could also be {count: string}.
const query = `
${scoresCte}
SELECT
${selectString}
FROM observations o FINAL
LEFT JOIN scores_avg AS s_avg ON s_avg.trace_id = o.trace_id and s_avg.observation_id = o.id
WHERE ${appliedObservationsFilter.query}
${orderByToClickhouseSql(orderBy ?? null, observationsTableUiColumnDefinitions)}
${limit !== undefined && offset !== undefined ? `LIMIT ${limit} OFFSET ${offset}` : ""};`;
const res = await queryClickhouse<T>({
query,
params: {
...appliedScoresFilter.params,
...appliedObservationsFilter.params,
},
});
return res;
}
};
export const getObservationsGroupedByModel = async (
@@ -752,50 +753,6 @@ export const getObservationsGroupedByModel = async (
return res.map((r) => ({ model: r.name }));
};
export const getObservationsGroupedByModelId = async (
projectId: string,
filter: FilterState,
) => {
const observationsFilter = new FilterList([
new StringFilter({
clickhouseTable: "observations",
field: "project_id",
operator: "=",
value: projectId,
tablePrefix: "o",
}),
]);
observationsFilter.push(
...createFilterFromFilterState(
filter,
observationsTableUiColumnDefinitions,
),
);
const appliedObservationsFilter = observationsFilter.apply();
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
SELECT o.internal_model_id as modelId
FROM observations o
WHERE ${appliedObservationsFilter.query}
AND o.type = 'GENERATION'
GROUP BY o.internal_model_id
ORDER BY count() DESC
LIMIT 1000;
`;
const res = await queryClickhouse<{ modelId: string }>({
query,
params: {
...appliedObservationsFilter.params,
},
});
return res.map((r) => ({ modelId: r.modelId }));
};
export const getObservationsGroupedByName = async (
projectId: string,
filter: FilterState,
@@ -950,46 +907,6 @@ export const deleteObservationsByTraceIds = async (
projectId,
traceIds,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
export const deleteObservationsByProjectId = async (projectId: string) => {
const query = `
DELETE FROM observations
WHERE project_id = {projectId: String};
`;
await commandClickhouse({
query: query,
params: {
projectId,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
export const deleteObservationsOlderThanDays = async (
projectId: string,
days: number,
) => {
const query = `
DELETE FROM observations
WHERE project_id = {projectId: String}
AND start_time < now() - INTERVAL {numDays: Int} DAYS;
`;
await commandClickhouse({
query: query,
params: {
projectId,
numDays: days,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
@@ -1033,7 +950,7 @@ export const getObservationMetricsForPrompts = async (
end_time,
usage_details,
cost_details,
dateDiff('millisecond', start_time, end_time) AS latency_ms
dateDiff('milliseconds', start_time, end_time) AS latency_ms
FROM observations
FINAL
WHERE (type = 'GENERATION')
@@ -1047,8 +964,8 @@ export const getObservationMetricsForPrompts = async (
prompt_version,
min(start_time) AS first_observation,
max(start_time) AS last_observation,
medianExact(arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'input') > 0, usage_details)))) AS median_input_usage,
medianExact(arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'output') > 0, usage_details)))) AS median_output_usage,
medianExact(usage_details['input']) AS median_input_usage,
medianExact(usage_details['output']) AS median_output_usage,
medianExact(cost_details['total']) AS median_total_cost,
medianExact(latency_ms) AS median_latency_ms
FROM latencies
@@ -1096,7 +1013,7 @@ export const getLatencyAndTotalCostForObservations = async (
SELECT
id,
cost_details['total'] AS total_cost,
dateDiff('millisecond', start_time, end_time) AS latency_ms
dateDiff('milliseconds', start_time, end_time) AS latency_ms
FROM observations FINAL
WHERE project_id = {projectId: String}
AND id IN ({observationIds: Array(String)})
@@ -1128,7 +1045,7 @@ export const getLatencyAndTotalCostForObservationsByTraces = async (
SELECT
trace_id,
sumMap(cost_details)['total'] AS total_cost,
dateDiff('millisecond', min(start_time), max(end_time)) AS latency_ms
dateDiff('milliseconds', min(start_time), max(end_time)) AS latency_ms
FROM observations FINAL
WHERE project_id = {projectId: String}
AND trace_id IN ({traceIds: Array(String)})
@@ -1152,167 +1069,3 @@ export const getLatencyAndTotalCostForObservationsByTraces = async (
latency: Number(r.latency_ms) / 1000,
}));
};
export const getObservationCountsByProjectInCreationInterval = async ({
start,
end,
}: {
start: Date;
end: Date;
}) => {
const query = `
SELECT
project_id,
count(*) as count
FROM observations
WHERE created_at >= {start: DateTime64(3)}
AND created_at < {end: DateTime64(3)}
GROUP BY project_id
`;
const rows = await queryClickhouse<{ project_id: string; count: string }>({
query,
params: {
start: convertDateToClickhouseDateTime(start),
end: convertDateToClickhouseDateTime(end),
},
});
return rows.map((row) => ({
projectId: row.project_id,
count: Number(row.count),
}));
};
export const getObservationCountOfProjectsSinceCreationDate = async ({
projectIds,
start,
}: {
projectIds: string[];
start: Date;
}) => {
const query = `
SELECT
count(*) as count
FROM observations
WHERE project_id IN ({projectIds: Array(String)})
AND created_at >= {start: DateTime64(3)}
`;
const rows = await queryClickhouse<{ count: string }>({
query,
params: {
projectIds,
start: convertDateToClickhouseDateTime(start),
},
});
return Number(rows[0]?.count ?? 0);
};
export const getTraceIdsForObservations = async (
projectId: string,
observationIds: string[],
) => {
const query = `
SELECT
trace_id,
id
FROM observations
WHERE project_id = {projectId: String}
AND id IN ({observationIds: Array(String)})
`;
const rows = await queryClickhouse<{ id: string; trace_id: string }>({
query,
params: {
projectId,
observationIds,
},
});
return rows.map((row) => ({
id: row.id,
traceId: row.trace_id,
}));
};
export const getGenerationsForPostHog = async function* (
projectId: string,
minTimestamp: Date,
maxTimestamp: Date,
) {
const query = `
SELECT
o.name as name,
o.start_time as start_time,
o.id as id,
o.total_cost as total_cost,
if(isNull(completion_start_time), NULL, date_diff('millisecond', start_time, completion_start_time)) as time_to_first_token,
o.usage_details['total'] as input_tokens,
o.usage_details['output'] as output_tokens,
o.cost_details['total'] as total_tokens,
o.project_id as project_id,
if(isNull(end_time), NULL, date_diff('millisecond', start_time, end_time) / 1000) as latency,
o.provided_model_name as model,
o.level as level,
o.version as version,
t.id as trace_id,
t.name as trace_name,
t.session_id as trace_session_id,
t.user_id as trace_user_id,
t.release as trace_release,
t.tags as trace_tags,
t.metadata['$posthog_session_id'] as posthog_session_id
FROM observations o FINAL
LEFT JOIN traces t FINAL ON o.trace_id = t.id AND o.project_id = t.project_id
WHERE o.project_id = {projectId: String}
AND t.project_id = {projectId: String}
AND o.start_time >= {minTimestamp: DateTime64(3)}
AND o.start_time <= {maxTimestamp: DateTime64(3)}
AND t.timestamp >= {minTimestamp: DateTime64(3)} - INTERVAL 7 DAY
AND t.timestamp <= {maxTimestamp: DateTime64(3)}
AND o.type = 'GENERATION'
`;
const records = queryClickhouseStream<Record<string, unknown>>({
query,
params: {
projectId,
minTimestamp: convertDateToClickhouseDateTime(minTimestamp),
maxTimestamp: convertDateToClickhouseDateTime(maxTimestamp),
},
});
const baseUrl = env.NEXTAUTH_URL?.replace("/api/auth", "");
for await (const record of records) {
yield {
timestamp: record.start_time,
langfuse_generation_name: record.name,
langfuse_trace_name: record.trace_name,
langfuse_url: `${baseUrl}/project/${projectId}/traces/${encodeURIComponent(record.trace_id as string)}?observation=${encodeURIComponent(record.id as string)}`,
langfuse_id: record.id,
langfuse_cost_usd: record.total_cost,
langfuse_input_units: record.input_tokens,
langfuse_output_units: record.output_tokens,
langfuse_total_units: record.total_tokens,
langfuse_session_id: record.trace_session_id,
langfuse_project_id: projectId,
langfuse_user_id: record.trace_user_id || "langfuse_unknown_user",
langfuse_latency: record.latency,
langfuse_time_to_first_token: record.time_to_first_token,
langfuse_release: record.trace_release,
langfuse_version: record.version,
langfuse_model: record.model,
langfuse_level: record.level,
langfuse_tags: record.trace_tags,
langfuse_event_version: "1.0.0",
$session_id: record.posthog_session_id ?? null,
$set: {
langfuse_user_url: record.user_id
? `${baseUrl}/project/${projectId}/users/${encodeURIComponent(record.user_id as string)}`
: null,
},
};
}
};
@@ -9,13 +9,11 @@ import Decimal from "decimal.js";
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { ObservationRecordReadType } from "./definitions";
import { parseJsonPrioritised } from "../../utils/json";
import { jsonSchema } from "../../utils/zod";
export const convertObservationToView = (
record: ObservationRecordReadType,
): Omit<ObservationView, "inputPrice" | "outputPrice" | "totalPrice"> & {
usageDetails: Record<string, number>;
costDetails: Record<string, number>;
} => {
): Omit<ObservationView, "inputPrice" | "outputPrice" | "totalPrice"> => {
// these cost are not used from the view. They are in the select statement but not in the
// Prisma file. We will not clean this up but keep it as it is for now.
// eslint-disable-next-line no-unused-vars
@@ -27,7 +25,10 @@ export const convertObservationToView = (
? parseClickhouseUTCDateTimeFormat(record.end_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.start_time).getTime()
: null,
timeToFirstToken: record.completion_start_time
? parseClickhouseUTCDateTimeFormat(record.start_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.completion_start_time).getTime()
: null,
promptName: record.prompt_name ?? null,
promptVersion: record.prompt_version ?? null,
modelId: record.internal_model_id ?? null,
@@ -41,15 +42,7 @@ export const convertObservation = (
promptVersion: number | null;
latency: number | null;
timeToFirstToken: number | null;
usageDetails: Record<string, number>;
costDetails: Record<string, number>;
} => {
const reducedUsageDetails = reduceUsageOrCostDetails(record.usage_details);
const reducedCostDetails = reduceUsageOrCostDetails(record.cost_details);
const reducedProvidedCostDetails = reduceUsageOrCostDetails(
record.provided_cost_details,
);
return {
id: record.id,
traceId: record.trace_id ?? null,
@@ -61,22 +54,15 @@ export const convertObservation = (
? parseClickhouseUTCDateTimeFormat(record.end_time)
: null,
name: record.name ?? null,
metadata:
record.metadata &&
Object.fromEntries(
Object.entries(record.metadata ?? {}).map(([key, val]) => [
key,
val && parseJsonPrioritised(val),
]),
),
metadata: record.metadata,
level: record.level as ObservationLevel,
statusMessage: record.status_message ?? null,
version: record.version ?? null,
input: (record.input
? parseJsonPrioritised(record.input)
? jsonSchema.parse(parseJsonPrioritised(record.input))
: null) as Prisma.JsonValue | null,
output: (record.output
? parseJsonPrioritised(record.output)
? jsonSchema.parse(parseJsonPrioritised(record.output))
: null) as Prisma.JsonValue | null,
modelParameters: record.model_parameters
? JSON.parse(record.model_parameters)
@@ -87,41 +73,31 @@ export const convertObservation = (
promptId: record.prompt_id ?? null,
createdAt: parseClickhouseUTCDateTimeFormat(record.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(record.updated_at),
promptTokens: reducedUsageDetails.input ?? 0,
completionTokens: reducedUsageDetails.output ?? 0,
totalTokens: reducedUsageDetails.total ?? 0,
calculatedInputCost:
reducedCostDetails.input != null
? new Decimal(reducedCostDetails.input)
: null,
calculatedOutputCost:
reducedCostDetails.output != null
? new Decimal(reducedCostDetails.output)
: null,
promptTokens: record.usage_details?.input
? Number(record.usage_details?.input)
: 0,
completionTokens: record.usage_details?.output
? Number(record.usage_details?.output)
: 0,
totalTokens: record.usage_details?.total
? Number(record.usage_details?.total)
: 0,
calculatedInputCost: record.cost_details?.input
? new Decimal(record.cost_details.input)
: null,
calculatedOutputCost: record.cost_details?.output
? new Decimal(record.cost_details.output)
: null,
calculatedTotalCost: record.cost_details?.total
? new Decimal(record.cost_details.total)
: null,
inputCost:
reducedProvidedCostDetails.input != null
? new Decimal(reducedProvidedCostDetails.input)
: null,
outputCost:
reducedProvidedCostDetails.output != null
? new Decimal(reducedProvidedCostDetails.output)
: null,
inputCost: record.cost_details?.input
? new Decimal(record.cost_details?.input)
: null,
outputCost: record.cost_details?.output
? new Decimal(record.cost_details?.output)
: null,
totalCost: record.total_cost ? new Decimal(record.total_cost) : null,
usageDetails: Object.fromEntries(
Object.entries(record.usage_details ?? {}).map(([key, value]) => [
key,
Number(value),
]),
),
costDetails: Object.fromEntries(
Object.entries(record.cost_details ?? {}).map(([key, value]) => [
key,
Number(value),
]),
),
model: record.provided_model_name ?? null,
internalModelId: record.internal_model_id ?? null,
unit: "TOKENS", // to be removed.
@@ -132,35 +108,8 @@ export const convertObservation = (
parseClickhouseUTCDateTimeFormat(record.start_time).getTime()
: null,
timeToFirstToken: record.completion_start_time
? (parseClickhouseUTCDateTimeFormat(
record.completion_start_time,
).getTime() -
parseClickhouseUTCDateTimeFormat(record.start_time).getTime()) /
1000
? parseClickhouseUTCDateTimeFormat(record.start_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.completion_start_time).getTime()
: null,
};
};
export const reduceUsageOrCostDetails = (
details: Record<string, number> | null | undefined,
): {
input: number | null;
output: number | null;
total: number | null;
} => {
return {
input: Object.entries(details ?? {})
.filter(([usageType]) => usageType.startsWith("input"))
.reduce(
(acc, [_, value]) => (acc ?? 0) + Number(value),
null as number | null, // default to null if no input usage is found
),
output: Object.entries(details ?? {})
.filter(([usageType]) => usageType.startsWith("output"))
.reduce(
(acc, [_, value]) => (acc ?? 0) + Number(value),
null as number | null, // default to null if no output usage is found
),
total: Number(details?.total ?? 0),
};
};
+19 -229
View File
@@ -3,11 +3,10 @@ import {
commandClickhouse,
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
queryClickhouseStream,
upsertClickhouse,
} from "./clickhouse";
import { FilterList } from "../queries/clickhouse-sql/clickhouse-filter";
import { FilterCondition, FilterState, TimeFilter } from "../../types";
import { FilterState } from "../../types";
import {
createFilterFromFilterState,
getProjectIdDefaultFilter,
@@ -26,7 +25,6 @@ import {
import { SCORE_TO_TRACE_OBSERVATIONS_INTERVAL } from "./constants";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { ScoreRecordReadType } from "./definitions";
import { env } from "../../env";
export const searchExistingAnnotationScore = async (
projectId: string,
@@ -95,32 +93,6 @@ export const getScoreById = async (
return rows.map(convertToScore).shift();
};
export const getScoresByIds = async (
projectId: string,
scoreId: string[],
source?: ScoreSource,
) => {
const query = `
SELECT *
FROM scores s
WHERE s.project_id = {projectId: String}
AND s.id IN ({scoreId: Array(String)})
${source ? `AND s.source = {source: String}` : ""}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
`;
const rows = await queryClickhouse<ScoreRecordReadType>({
query,
params: {
projectId,
scoreId,
...(source !== undefined ? { source } : {}),
},
});
return rows.map(convertToScore);
};
/**
* Accepts a score in a Clickhouse-ready format.
* id, project_id, name, and timestamp must always be provided.
@@ -136,16 +108,13 @@ export const upsertScore = async (score: Partial<ScoreRecordReadType>) => {
});
};
export type GetScoresForTracesProps = {
projectId: string;
traceIds: string[];
timestamp?: Date;
limit?: number;
offset?: number;
};
export const getScoresForTraces = async (props: GetScoresForTracesProps) => {
const { projectId, traceIds, timestamp, limit, offset } = props;
export const getScoresForTraces = async (
projectId: string,
traceIds: string[],
timestamp?: Date,
limit?: number,
offset?: number,
) => {
const query = `
select
*
@@ -204,24 +173,20 @@ export const getScoresForObservations = async (
return rows.map(convertToScore);
};
export const getScoresGroupedByNameSourceType = async (
projectId: string,
timestamp: Date | undefined,
) => {
export const getScoresGroupedByNameSourceType = async (projectId: string) => {
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
name,
source,
data_type
from scores s
WHERE s.project_id = {projectId: String}
${timestamp ? `AND s.timestamp >= {timestamp: DateTime64(3)}` : ""}
GROUP BY name, source, data_type
ORDER BY count() desc
LIMIT 1000;
`;
select
name,
source,
data_type
from scores s
WHERE s.project_id = {projectId: String}
GROUP BY name, source, data_type
ORDER BY count() desc
LIMIT 1000;
`;
const rows = await queryClickhouse<{
name: string;
@@ -231,9 +196,6 @@ export const getScoresGroupedByNameSourceType = async (
query: query,
params: {
projectId: projectId,
...(timestamp
? { timestamp: convertDateToClickhouseDateTime(timestamp) }
: {}),
},
});
@@ -514,46 +476,6 @@ export const deleteScoresByTraceIds = async (
projectId,
traceIds,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
export const deleteScoresByProjectId = async (projectId: string) => {
const query = `
DELETE FROM scores
WHERE project_id = {projectId: String};
`;
await commandClickhouse({
query: query,
params: {
projectId,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
export const deleteScoresOlderThanDays = async (
projectId: string,
days: number,
) => {
const query = `
DELETE FROM scores
WHERE project_id = {projectId: String}
AND timestamp < now() - INTERVAL {numDays: Int} DAYS;
`;
await commandClickhouse({
query: query,
params: {
projectId,
numDays: days,
},
clickhouseConfigs: {
request_timeout: 120_000, // 2 minutes
},
});
};
@@ -567,14 +489,10 @@ export const getNumericScoreHistogram = async (
);
const chFilterRes = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const query = `
select s.value
from scores s
${traceFilter ? `LEFT JOIN traces t ON s.trace_id = t.id AND t.project_id = s.project_id` : ""}
WHERE s.project_id = {projectId: String}
${traceFilter ? `AND t.project_id = {projectId: String}` : ""}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
@@ -631,131 +549,3 @@ export const getAggregatedScoresForPrompts = async (
promptId: row.prompt_id,
}));
};
export const getScoreCountsByProjectInCreationInterval = async ({
start,
end,
}: {
start: Date;
end: Date;
}) => {
const query = `
SELECT
project_id,
count(*) as count
FROM scores
WHERE created_at >= {start: DateTime64(3)}
AND created_at < {end: DateTime64(3)}
GROUP BY project_id
`;
const rows = await queryClickhouse<{ project_id: string; count: string }>({
query,
params: {
start: convertDateToClickhouseDateTime(start),
end: convertDateToClickhouseDateTime(end),
},
});
return rows.map((row) => ({
projectId: row.project_id,
count: Number(row.count),
}));
};
export const getDistinctScoreNames = async (
projectId: string,
cutoffCreatedAt: Date,
filter: FilterState,
isTimestampFilter: (filter: FilterCondition) => filter is TimeFilter,
) => {
const scoreTimestampFilter = filter?.find(isTimestampFilter);
const query = `
SELECT DISTINCT
name
FROM scores s
WHERE s.project_id = {projectId: String}
AND s.created_at <= {cutoffCreatedAt: DateTime64(3)}
${scoreTimestampFilter ? `AND s.timestamp >= {filterTimestamp: DateTime64(3)}` : ""}
`;
const rows = await queryClickhouse<{ name: string }>({
query,
params: {
projectId,
cutoffCreatedAt: convertDateToClickhouseDateTime(cutoffCreatedAt),
...(scoreTimestampFilter
? {
filterTimestamp: convertDateToClickhouseDateTime(
scoreTimestampFilter.value,
),
}
: {}),
},
});
return rows.map((row) => row.name);
};
export const getScoresForPostHog = async function* (
projectId: string,
minTimestamp: Date,
maxTimestamp: Date,
) {
const query = `
SELECT
s.id as id,
s.timestamp as timestamp,
s.name as name,
s.value as value,
s.comment as comment,
t.name as trace_name,
t.session_id as trace_session_id,
t.user_id as trace_user_id,
t.release as trace_release,
t.tags as trace_tags,
t.metadata['$posthog_session_id'] as posthog_session_id
FROM scores s FINAL
LEFT JOIN traces t FINAL ON s.trace_id = t.id AND s.project_id = t.project_id
WHERE s.project_id = {projectId: String}
AND t.project_id = {projectId: String}
AND s.timestamp >= {minTimestamp: DateTime64(3)}
AND s.timestamp <= {maxTimestamp: DateTime64(3)}
AND t.timestamp >= {minTimestamp: DateTime64(3)} - INTERVAL 7 DAY
AND t.timestamp <= {maxTimestamp: DateTime64(3)}
`;
const records = queryClickhouseStream<Record<string, unknown>>({
query,
params: {
projectId,
minTimestamp: convertDateToClickhouseDateTime(minTimestamp),
maxTimestamp: convertDateToClickhouseDateTime(maxTimestamp),
},
});
const baseUrl = env.NEXTAUTH_URL?.replace("/api/auth", "");
for await (const record of records) {
yield {
timestamp: record.timestamp,
langfuse_score_name: record.name,
langfuse_score_value: record.value,
langfuse_score_comment: record.comment,
langfuse_trace_name: record.trace_name,
langfuse_id: record.id,
langfuse_session_id: record.trace_session_id,
langfuse_project_id: projectId,
langfuse_user_id: record.trace_user_id || "langfuse_unknown_user",
langfuse_release: record.trace_release,
langfuse_tags: record.trace_tags,
langfuse_event_version: "1.0.0",
$session_id: record.posthog_session_id ?? null,
$set: {
langfuse_user_url: record.user_id
? `${baseUrl}/project/${projectId}/users/${encodeURIComponent(record.user_id as string)}`
: null,
},
};
}
};

Some files were not shown because too many files have changed in this diff Show More