Compare commits

...
225 Commits
Author SHA1 Message Date
Marc Klingen 2dd5c8a8aa chore: release v3.97.2 2025-08-12 17:31:22 +02:00
Marc Klingen d17b21f04d chore: specify development node version 2025-08-12 17:30:31 +02:00
Marc KlingenandGitHub c5a0f44bdd fix: fail silently when /api/latest-releases returns invalid schema (#8476)
* fix: fail silently when /api/latest-releases returns invalid schema

* push
2025-08-12 15:18:27 +00:00
Steffen SchmitzandGitHub 011b4912f2 chore: limit traces to trace AMT migration to only write into traces_all_amt (#8458)
* chore: limit traces to trace AMT migration to only write into traces_all_amt

* chore: revert validation changes

* chore: simplify query

* chore: tune both queries

* chore: avoid full aggregation during migration

* chore: handle IO coalescing to skip aggregations fully

* chore: remove obsolete query parts
2025-08-12 14:44:55 +00:00
Steffen SchmitzandGitHub a4d773066f chore: move health check to new traces AMTs (#8473) 2025-08-12 14:44:09 +00:00
Marlies Mayerhofer 25bdb1690b chore: release v3.97.1 2025-08-12 11:33:34 +02:00
marliessophieandGitHub f8565d7772 fix(evals): update trace deletion logic to remove job executions directly (#8466) 2025-08-12 09:15:34 +00:00
marliessophieandGitHub a912057082 fix(evals): ensure default model supports langfuse evals at set up (#8411)
* fix(evals): ensure default model supports langfuse evals at set up

* fix: await upsertDefaultModel

* chore: provide actionable error messages in case of misconfiguration
2025-08-12 09:11:00 +00:00
marliessophieandGitHub e0d3270f50 fix(evals): json parse preview to properly handle doubleEncoded IO (#8464)
* fixup(evals): json parse preview

* chore: add comment
2025-08-12 07:47:09 +00:00
Hassieb Pakzad 99e5a0b224 chore: release v3.97.0 2025-08-11 19:22:53 +02:00
bccdee5410 feat: add HTTPS proxy support for LLM API calls (#8461)
* feat: Add HTTPS proxy support for LLM API calls (#7932)

* remove tests

---------

Co-authored-by: suhwan <52690419+suhwan-cheon@users.noreply.github.com>
2025-08-11 17:18:54 +00:00
marliessophieandGitHub 738cbbc8cd feat(annotation-assignment): allow removing user assignments in UI (#8370)
* feat(annotation-assignment): allow removing user assignments in UI

* chore: push
2025-08-11 15:57:02 +00:00
marliessophieandGitHub 920c52bb06 chore(evals): cancel job executions in case of underlying trace delete (#8443)
* chore(evals): cancel job executions in case of underlying trace delete

* chore: lint
2025-08-11 15:56:27 +00:00
Leo WeigandandGitHub 4f134790af chore: add edit button to dataset items dot menu (#8444)
* chore: add edit button to dataset items dot menu

* fix: disable button based on access

* fix: DatasetActionButton not forwarding refs

* fix: eslint
2025-08-11 15:33:03 +00:00
Steffen SchmitzandGitHub 21b3ce3c82 Revert "perf: stream new records to clickhouse in writer (#8421)" (#8459)
This reverts commit b35583056b.
2025-08-11 15:30:42 +00:00
Leo WeigandandGitHub 0531b57e1a refactor: update create org form validation (#8455) 2025-08-11 14:26:45 +00:00
Leo WeigandandGitHub 7b857a0dc4 chore: surface required models for Azure/Bedrock (#8453)
- LLM connections: surface required model selection
for Azure/Bedrock; simplify advanced settings and validation
- Move custom model names to main form for Azure and Bedrock; require at least one model
- Remove default models toggle for Azure/Bedrock (they don’t support defaults)
- Keep Azure base URL and extra headers in main form; remove Azure advanced panel entirely
- Show advanced settings only for OpenAI, Anthropic, Vertex AI, and Google AI Studio
- Improve validation order to avoid confusing errors when defaults aren’t supported
- Add adapter-based placeholder for provider name; clarify copy
- Consolidate adapter logic to a single helper used in UI and schema
2025-08-11 14:26:27 +00:00
Steffen SchmitzandGitHub 6c97fe3c04 chore: drop "remove tag" capability from UI (#8454) 2025-08-11 14:24:08 +00:00
Steffen SchmitzandGitHub 1d69cbd41b chore: migrate remaining analytics and export queries to traces AMTs (#8451)
* chore: migrate remaining analytics and export queries to traces AMTs

* chore: patch

* chore: patches

* dummy

* chore: replace start_time with timestamp

* chore: adjust timeshift

* chore: switch to startTime

* chore: updat ereadme
2025-08-11 13:48:48 +00:00
Steffen SchmitzandGitHub ab26692913 fix(dashboards): patch invalid tags filter (#8445)
* fix(dashboards): patch invalid tags filter

* chore: move tests

* chore: remove unnecessary auth overwrite

* chore: test update
2025-08-11 12:14:56 +00:00
Hassieb PakzadandGitHub b791544864 fix(model-prices): gpt-5 prices with model date (#8442)
* fix(model-prices): gpt-5 prices with model date

* add to playground and evals

* add cache clearance
2025-08-11 11:49:32 +00:00
Steffen SchmitzandGitHub b35583056b perf: stream new records to clickhouse in writer (#8421)
* perf: stream new records to clickhouse in writer

* chore: update unit tests
2025-08-11 08:23:16 +00:00
Steffen SchmitzandGitHub 4504530e3d chore: skip unavailable shards in traces AMT background migration (#8438) 2025-08-11 08:21:50 +00:00
marliessophieandGitHub 45eed4b5c0 feat(scores-exports): add author to score exports (#8419) 2025-08-08 16:29:45 +00:00
marliessophieandGitHub 9fa31ba68e chore(datasets-csv-upload): no longer nest column values if single valid json object (#8416) 2025-08-08 16:04:55 +00:00
Hassieb PakzadandGitHub ea35c25269 fix: gemini-2.5-flash name (#8412) 2025-08-08 14:47:05 +00:00
steffen911 c427791070 chore: release v3.96.2 2025-08-08 16:35:11 +02:00
Steffen SchmitzandGitHub 25220aef55 perf: allow sharding for trace upsert queue (#8396)
* perf: allow sharding for trace upsert queue

* chore: handle metrics for traceupsertqueue

* chore: remove outdated comment

* chore: test util upgrade

* chore: patch typing
2025-08-08 14:16:27 +00:00
Max Deichmann 8c02d5dc22 chore: release v3.96.1 2025-08-08 15:57:05 +02:00
Max DeichmannandGitHub 0b60076871 chore: upgrade form data (#8408) 2025-08-08 13:17:12 +00:00
marliessophieandGitHub b28a83531b fix(evals): display final status in peek view (#8406) 2025-08-08 13:12:38 +00:00
Steffen SchmitzandGitHub aa4d34ae66 chore: skip S3 list and legacy trace insert based on env flag (#8298)
* chore: skip S3 list and legacy trace insert based on env flag

* chore: adjust skip s3 list conditions

* chore; refactor traceUpserQueue entity assoication

* chore: reverse
2025-08-08 12:47:04 +00:00
Steffen SchmitzandGitHub cca352ad13 chore: remove trace null deletions (#8405) 2025-08-08 15:02:37 +02:00
Steffen SchmitzandGitHub 58c96cada7 chore: include boolean scores in the reference of scores-numeric (#8086) 2025-08-08 12:39:56 +00:00
Steffen SchmitzandGitHub 6f93389936 chore: adjust devcontainer setup to install deps via dockerfile (#8399)
* chore: adjust devcontainer setup to install deps via dockerfile

* Update Dockerfile
2025-08-08 12:10:05 +00:00
Max DeichmannandGitHub b8d9586490 chore: fix dev command (#8404) 2025-08-08 14:38:21 +02:00
Leo 534f5696ad chore: release v3.96.0 2025-08-08 14:20:05 +02:00
Leo WeigandandGitHub 4cebe2831c fix(onboarding): only show org onboarding for Cloud and more UX touches (#8392) 2025-08-08 14:17:49 +02:00
45a5f3b8a2 chore: upgrade slack dependency (#8402)
* chore: upgrade slack dependencies

* chore: upgrade slack deps and remove unnecessary deps from web/worker

---------

Co-authored-by: Max Deichmann <m.deichmann@tum.de>
2025-08-08 14:05:41 +02:00
Max DeichmannandGitHub 8dea9b6a3a chore: remove explicit axios dependency (#8397)
remove axios
2025-08-08 11:38:42 +00:00
marliessophieandGitHub c5b781d3f6 fix(llm-connections): ensure dialog remains open when switching tabs (#8391)
* fix(llm-connections): ensure dialog remains open when switching tabs

* chore: lint
2025-08-08 09:52:48 +00:00
Leo WeigandandGitHub 07469c928d feat(cloud): new onboarding survey to better understand our users (#8374) 2025-08-08 10:37:13 +02:00
Marc KlingenandGitHub 5b10110dfe fix(auth): honor NEXT_PUBLIC_BASE_PATH in password reset verification (#8385) 2025-08-07 23:14:32 +00:00
Thorsten SpiekerandGitHub 0eb5d1c3c0 feat: pivot table sorting (#8214) 2025-08-08 00:12:09 +02:00
Max Deichmann 1585bf5e14 chore: release v3.95.2 2025-08-07 20:11:10 +02:00
Max DeichmannandGitHub dec6d6f975 chore: add openai 5 models (#8378) 2025-08-07 20:09:36 +02:00
Max DeichmannandGitHub 36fa7ea15e chore: improve typing of users (#8376)
push
2025-08-07 17:28:10 +00:00
Marlies Mayerhofer 6a4d56a5a9 chore: release v3.95.1 2025-08-07 18:32:29 +02:00
Marc KlingenandGitHub 9f0a40949e fix: check projectMembers:read for trpc members.byProjectId (#8373) 2025-08-07 17:55:57 +02:00
Steffen SchmitzandGitHub 10fdd9e172 chore: add env to move short-term reads to AMTs (#8367) 2025-08-07 15:26:11 +00:00
marliessophieandGitHub fc58add111 chore(rbac): fix types for resolveProjectRole (#8371)
* chore(rbac): fix types for resolveProjectRole

* chore: eslint
2025-08-07 15:18:22 +00:00
Steffen SchmitzandGitHub 10993838d5 perf: truncate IO in table overviews and skip internal JSON parse on TRPC (#8364)
* perf: truncate IO in table overviews and skip internal JSON parse on TRPC

* chore: fix imports

* chore: align test case
2025-08-07 15:09:35 +00:00
Steffen SchmitzGitHubellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
148cd29e9f fix: handle invalid string lengths on traces_null table gracefully (#8369)
* fix: handle invalid string lengths on traces_null table gracefully

* Update worker/src/services/ClickhouseWriter/index.ts

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>

* chore: test dropped record truncation

---------

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
2025-08-07 14:25:00 +00:00
Steffen SchmitzandGitHub 7a353e99af chore: disable usage alert configuration temporarily (#8360) 2025-08-07 10:22:08 +00:00
Marc KlingenandGitHub dc52106feb chore: improve multi-tenant sso error message (#8363) 2025-08-07 12:44:06 +02:00
Marlies Mayerhofer b2f8eb7078 chore: release v3.95.0 2025-08-07 11:55:36 +02:00
026d8e2ba0 feat(annotation): support assigning users to queues (#8325)
* chore(annotation-queue): add AnnotationQueueMembership model and related database schema updates

* chore(annotation-queue): add AnnotationQueueMembership POST and DEL APIs

* feat(annotation-queue): implement user assignment functionality for annotation queues

* fix: role logic for users

* chore: rn to assignments

* fixup: rn to assignments

* chore(annotation-queue): enhance user assignment section with debounced search and display of assigned users

* feat(annotation-queue): add MultiSelectCombobox for user assignment and enhance DataTable with row class name functionality

* refactor(rbac): replace generateUserQuery with generateUserProjectRolesQuery and introduce utility functions for project role resolution

* fixup: filter query for user ids

* fixup: resolve project role test

* fix: validate user project role before adding to annotation assignment

* chore: remove code

* chore: fix

* chore: fix filter

* chore: fix filter

* tests for auth logic

* push: fix api route test

* tests for auth logic

* fix

* fix

* chore: push

* chore: rename

* docs: api types

* fix: formatting

* fix: formatting

* chore: add disabled prop to MultiSelectCombobox

* chore: rename migration

---------

Co-authored-by: Max Deichmann <m.deichmann@tum.de>
2025-08-07 09:29:45 +00:00
Hassieb Pakzad 9a9f0b1743 chore: release v3.94.0 2025-08-07 10:30:22 +02:00
Hassieb PakzadandGitHub 74a510fbde feat(otel): add parsing for Vercel AI SDK (#8347)
* feat(otel): add parsing for Vercel AI SDK

* push

* push
2025-08-07 07:43:34 +00:00
Steffen SchmitzandGitHub c23b226e62 chore: move dashboard queries to new AMTs (#8308)
* chore: create background migrations for traces amt tables

* chore: add database migrations

* chore: move tests to use AMTs in one config

* chore: shift backfill from {} to map()

* chore: modify ingestion test timeouts

* chore: keep minDate updates tracked

* chore: remove android lib to save swap space

* chore: check disk space

* chore: check mnt usage

* chore: skip swap to check runner performance

* chore: add timestamp lines

* chore: handle deletions

* chore: update test cases

* chore: linting

* chore: allow opt-in creation for background migratoin record

* chore: update comparison query

* chore: concurrency config

* chore: update names

* chore: update names

* chore: update tiemout

* chore: add pre-query logs

* chore: refactor long running query into util function

* chore: refactor query exists checks

* chore: set correct tables for clustered configs

* chore: increase number of retries

* chore: extend logging

* chore: gracefully handle aborted query

* chore: handle errors during execution and raise them

* chore; skip in progress header updates

* Update migrateTracesToTracesAMTs.ts

* chore: use default clickhouse settings for ingestion

* chore: add trace_id index on traces_all_amt

* chore: update sorting behaviour of batchAction test

* chore: drop logger

* chore: update test

* chore: lint and renames

* chore: update test cases

* chore: patch score tests

* chore: add final for the checkTraceExists behaviour

* chore: remove background migration insert

* chore: move dashboard queries to new AMTs

* chore: make sure to replace all references to old traces table

* chore: revert replacement behaviour

* chore: formatting

* chore: typo

* chore: lint and refactor

* chore: update import path

* chore: patch test behaviour

* chore: remove outdated comment
2025-08-06 13:57:15 +00:00
Steffen SchmitzandGitHub 2f50b38a68 chore: adjust stalled settings for eval execution queue (#8343)
* chore: tune queue stalled checks for higher reliability

* chore: only apply stalled to eval execution queue
2025-08-06 12:29:47 +00:00
marliessophieandGitHub 451ae15e00 feat(evaluator-form): add tooltips for evaluator usage and target in edit mode (#8340)
* feat(evaluator-form): add tooltip for existing evaluator usage in edit mode

* feat(evaluator-form): enhance target data label with tooltip for edit mode

* chore: lint
2025-08-06 10:35:43 +00:00
marliessophieandGitHub e34d81578b chore(evaluator-form): prevent form submission on Enter key press; hide controls in preview (#8337)
* chore(evaluator-form): prevent form submission on Enter key press

* chore: disable select control on evaluation traces preview
2025-08-06 08:28:13 +00:00
Max DeichmannandGitHub 214c9c7ed2 perf: increase clickhouse timeouts for blob storage exports (#8328)
push
2025-08-05 22:45:49 +00:00
Max DeichmannandGitHub 961b3bfa8d chore: truncate traces i/o for the traces table (#8326)
* push

* push

* push

* something

* something
2025-08-05 20:53:40 +00:00
Marc KlingenandGitHub a1dd5b22a2 docs: improve api reference for llm-connections api (#8327) 2025-08-05 20:43:45 +02:00
Marc KlingenandGitHub 49950f9706 chore: add validation to okta issuer (#8323) 2025-08-05 17:22:40 +00:00
Max DeichmannandGitHub cfdd0fdb73 chore: refactor IOTableCell (#8324) 2025-08-05 19:26:16 +02:00
Marc KlingenandGitHub 891e7e9716 feat(models): add claude-opus-4-1-20250805 (#8322) 2025-08-05 17:19:34 +00:00
Marc KlingenandGitHub 5b407dad53 feat(posthog-integration): add langfuse_trace_id to generations and scores (#8319)
feat(posthog-integration): add langfuse_trace_id to generations and scores
2025-08-05 15:25:31 +00:00
steffen911 26e3bf9a44 chore: release v3.93.0 2025-08-05 16:56:10 +02:00
Steffen SchmitzandGitHub 1f02e364e1 chore: create background migrations for traces amt tables (#7689)
* chore: create background migrations for traces amt tables

* chore: add database migrations

* chore: move tests to use AMTs in one config

* chore: shift backfill from {} to map()

* chore: modify ingestion test timeouts

* chore: keep minDate updates tracked

* chore: remove android lib to save swap space

* chore: check disk space

* chore: check mnt usage

* chore: skip swap to check runner performance

* chore: add timestamp lines

* chore: handle deletions

* chore: update test cases

* chore: linting

* chore: allow opt-in creation for background migratoin record

* chore: update comparison query

* chore: concurrency config

* chore: update names

* chore: update names

* chore: update tiemout

* chore: add pre-query logs

* chore: refactor long running query into util function

* chore: refactor query exists checks

* chore: set correct tables for clustered configs

* chore: increase number of retries

* chore: extend logging

* chore: gracefully handle aborted query

* chore: handle errors during execution and raise them

* chore; skip in progress header updates

* Update migrateTracesToTracesAMTs.ts

* chore: use default clickhouse settings for ingestion

* chore: add trace_id index on traces_all_amt

* chore: update sorting behaviour of batchAction test

* chore: drop logger

* chore: update test

* chore: lint and renames

* chore: update test cases

* chore: patch score tests

* chore: add final for the checkTraceExists behaviour

* chore: remove background migration insert

* chore: convert traces_mt to traces_null

* chore: patch tests

* chore: switch from minMap to maxMap

* chore: mark aggregation columns on traces as do not use

* chore: patch new test

* chore: switch to maxMap

* consistent naming
2025-08-05 14:20:49 +00:00
Leo WeigandandGitHub 90dca15d86 feat(settings): improve UX for managing LLM connections (#8312) 2025-08-05 14:30:23 +02:00
Steffen SchmitzandGitHub e3f8dc8bb3 fix(blob-storage): disable auto container check for Azure buckets for costsavings (#8317) 2025-08-05 11:44:03 +00:00
Hassieb Pakzad 585ede0919 chore: release v3.92.1 2025-08-05 11:24:29 +02:00
Hassieb PakzadandGitHub 0db425d120 fix(pagination): allow limit zero again (#8315) 2025-08-05 09:22:34 +00:00
Hassieb Pakzad 09e33c3059 chore: release v3.92.0 2025-08-05 10:01:57 +02:00
Hassieb PakzadandGitHub 66226011af feat(llm-api-keys): add public api methods (#8301)
* feat(llm-api-keys): add public api LIST and PATCH endpoints

* puhs

* push

* push

* add post endpoint

* add PUT endpoint

* push

* push
2025-08-05 07:32:18 +00:00
marliessophieandGitHub 571698ab2e fix(pretty-json-view): prevent infinite recursion and optimize child row processing (#8302) 2025-08-04 18:05:45 +00:00
Steffen SchmitzandGitHub 312066f735 chore: add operation_name tag to AMT query experiments (#8294) 2025-08-04 11:25:32 +00:00
marliessophieandGitHub 5abeaf8adb fix(dataset-run-items): fix pagination for consistent ordering across sources (#8292) 2025-08-04 10:23:48 +00:00
Steffen SchmitzandGitHub fac3c732de chore: optimize queries for session AMT accesses (#8289) 2025-08-04 09:55:17 +00:00
marliessophieandGitHub 5e2e3bb5fc feat(sessions-ui): allow filtering table by scores (#7858)
* fixup(sessions-ui): allow filtering table by scores

* fix(table-definitions): update scores label from "Scores (avg)" to "Scores (numeric)"

* feat: support scores filters on sessions

* chore

* chore: fix duplicate in query

* chore: hide whitespace

* chore: fix
2025-08-04 09:36:48 +00:00
Max Deichmann 0564df8e51 chore: release v3.91.0 2025-08-04 11:33:20 +02:00
Steffen SchmitzandGitHub a2801a3be9 chore: add experiment tags to clickhouse queries (#8288) 2025-08-04 09:17:07 +00:00
Hassieb PakzadGitHubellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
075eb58ecb chore: add fern JS SDK generator (#8095)
* chore: add fern JS SDK generator

* push

* return java generator

* push

* push

* push

* Update worker/src/services/IngestionService/tests/IngestionService.integration.test.ts

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>

* Update worker/src/services/IngestionService/tests/IngestionService.integration.test.ts

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>

* puhs

---------

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
2025-08-04 08:49:43 +00:00
Steffen SchmitzandGitHub 674d66d179 chore: record AMT duration distribution (#8287) 2025-08-04 10:32:10 +02:00
marliessophieandGitHub 163f2a02ff fix(dataset-run-items): include run metadata for runs without dataset_run_items (#8267)
* fix(dataset-run-items): include run metadata for runs without dataset_run_items

* chore: drop name from metrics query
2025-08-04 08:12:46 +00:00
Steffen SchmitzandGitHub 075836f210 chore: upsert traces into both traces tables (#8268) 2025-08-04 09:28:46 +02:00
marliessophieandGitHub 543b6ee0f2 feat(peek-evaluator-config): add edit mode functionality (#8285) 2025-08-04 07:12:59 +00:00
marliessophieandGitHub 8c8c488e4d feat(annotation-queues): add support for batch adding observations to annotation queues (#8261) 2025-08-04 07:10:48 +00:00
Hassieb PakzadandGitHub 6a0e0a4221 feat(llm-connections): remove atla integration (#8059) 2025-08-04 05:53:24 +00:00
Max DeichmannandGitHub 1b01a267df chore: only show 10 traces for evals (#8283) 2025-08-03 20:39:51 +00:00
Thorsten SpiekerandGitHub 69a9146894 Revert "refactor: break up SlackService and move to web/worker packages" (#8276)
Revert "refactor: break up SlackService and move to web/worker packages (#8247)"

This reverts commit e452004a6a.
2025-08-03 18:59:20 +00:00
380403e8ed fix(cloud): add csp headers needed for attachment upload to Plain support chat (#7813)
Update CSP headers to allow additional S3 image and connect sources

Co-authored-by: Cursor Agent <cursoragent@cursor.com>
2025-08-03 18:52:25 +02:00
Max DeichmannGitHubellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
296a6c3ee6 chore: limit batch sizes of trace deletions (#8274)
* push

* Update worker/src/env.ts

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>

---------

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
2025-08-02 07:07:45 +00:00
Steffen SchmitzandGitHub 62904a9563 ci: debug docker build issues (#8269) 2025-08-01 18:07:42 +02:00
Steffen SchmitzandGitHub 90247ab64c chore: migrate additional queries to traces amts (#8249) 2025-08-01 15:06:16 +02:00
marliessophieandGitHub 5cf91c60d9 fix(layout): highlight nav items when viewing child routes (#8211) 2025-08-01 11:50:11 +00:00
Steffen SchmitzandGitHub 559ba6d05d chore: remove unused sessions query (#8264) 2025-08-01 13:17:32 +02:00
Steffen SchmitzandGitHub f071be69b6 build: add always() condition to run notifications on failure (#8265) 2025-08-01 12:54:30 +02:00
Max DeichmannandGitHub 52c261b422 perf: delete traces in batches (#8250) 2025-08-01 11:33:32 +02:00
Steffen SchmitzandGitHub 6f0a43ec65 build: setup slack alerts on failing CI (#8255) 2025-08-01 11:12:02 +02:00
marliessophieandGitHub 301bd6b569 feat(dataset-run-items): reads router (#8207)
* chore(dataset-run-items): GET /dataset-run-items

* fixup(dataset-run-items): GET /dataset-run-items

* chore(dataset-run-items): GET datasetRun/[runName]

* fix(dataset-run-items): return 0 for count when no records found

* fix(dataset-run-items): ensure offset is calculated only when page and limit are present

* feat(dataset-run-items): enhance dataset filtering by adding datasetId support and refactor related functions

* chore: rm comment

* refactor(dataset-run-items): rename dataset run name to ID across API and table definitions

* feat(dataset-run-items): add dataset_run_items to Clickhouse table names and refactor dataset filtering to use camelCase

* fix(dataset-run-items): update dataset filtering to ensure only the latest version of each dataset run item is retrieved

* chore: imports

* chore: fix de-duplication

* chore(dataset-router): mark runitemsByRunIdOrItemId as deprecated

* chore: refactor only, split out routes into byItemId and byRunId

* Revert "chore: refactor only, split out routes into byItemId and byRunId"

This reverts commit 2e1f2e3c14b9ed1173e2fb73b7f721c72d57bfbb.

* Revert "chore(dataset-router): mark runitemsByRunIdOrItemId as deprecated"

This reverts commit 5cb2232198d84baf7f027b00603c6fb3b9489c52.

* fixup: runitemsbyrunidoritemid

* chore: rewrite runitemsByRunIdOrItemId

* chore: rewrite runsByDatasetIdMetrics

* chore: ordering

* chore: order by created at and remove base data from query

* chore: add back name

* chore: push
2025-08-01 08:10:22 +00:00
NimarandGitHub 56fd3df2d2 fix(json-view): truncate cell after 2k chacacters (#8227)
* fix(json-view): truncate cell after 2k chacacters

* real value for truncated chars

* just rely on regex
2025-07-31 20:12:44 +00:00
Max Deichmann db433e2b72 chore: release v3.90.0 2025-07-31 20:52:35 +02:00
Thorsten SpiekerandGitHub e452004a6a refactor: break up SlackService and move to web/worker packages (#8247)
* refactor: break up SlackService and move to web/worker packages

* fix: pnpm lock file

* fix: tests

* fix
2025-07-31 18:33:03 +00:00
NimarandGitHub bce026d8b2 fix(dataset-items): add placeholder text for new items (#8228) 2025-07-31 19:11:48 +02:00
marliessophieandGitHub fbdf12bfa3 chore(dataset-run-items): add background migration for DRI (#8245) 2025-07-31 18:07:43 +02:00
marliessophieandGitHub 78f59bc543 chore: add dataset-run-items tables to clickhouse (#8243) 2025-07-31 16:50:25 +02:00
James IdzikandGitHub ea338ed83d fix: coerce LANGFUSE_INIT_PROJECT_RETENTION and other envs correctly to number (#8103)
fix: coerce init project retention, historic max eval limit envs
2025-07-31 16:01:53 +02:00
Steffen SchmitzandGitHub 8cf67fe329 chore: correctly parse input and output on traces_mt ingestion (#8242) 2025-07-31 15:21:36 +02:00
Steffen SchmitzandGitHub e3b23ea5ec feat: allow PUT operations on SCIM users endpoint (#8236)
* feat: allow PUT operations on SCIM users endpoint

* chore: make tests pass

* cleanup
2025-07-31 12:24:45 +00:00
Steffen SchmitzandGitHub 7be2791105 chore: patch docs for ingestion replay (#8238)
* chore: patch docs for ingestion replay

* chore: increase s3 parallelization
2025-07-31 11:47:10 +00:00
NimarandGitHub 683aae0069 fix(json-view): tighter ChatML validation (#8225) 2025-07-31 10:02:58 +00:00
Steffen SchmitzandGitHub 1c8ffc607d chore: write single rows into AMTs instead of pre-merged record (#8218)
* chore: write single rows into AMTs instead of pre-merged record

* chore: adjust assertion

* chore: drop debug log
2025-07-31 09:27:14 +00:00
Steffen SchmitzandGitHub 9c203e8b8a chore: allow return from aggregatingmergetrees without experiments (#8215) 2025-07-31 09:07:39 +00:00
Steffen SchmitzandGitHub c32cccce52 feat: allow redis username overwrites via env variables (#8234) 2025-07-31 09:52:03 +02:00
NimarandGitHub f1f8da5b74 fix(json-view): store formatted/json preference in local storage (#8222) 2025-07-30 19:47:22 +00:00
Max DeichmannandGitHub c23447a624 chore: improve slack links (#8219) 2025-07-30 18:31:44 +02:00
Thorsten SpiekerandGitHub 5e2225de46 fix: use different env variable in worker to construct links (#8212) 2025-07-30 17:02:11 +02:00
Nimar bb9d118853 chore: release v3.89.0 2025-07-30 16:09:51 +02:00
marliessophieandGitHub df57ff60d6 feat(dataset-run-items): support reads for public APIs (#8179)
* chore(dataset-run-items): GET /dataset-run-items

* fixup(dataset-run-items): GET /dataset-run-items

* chore(dataset-run-items): GET datasetRun/[runName]

* fix(dataset-run-items): return 0 for count when no records found

* fix(dataset-run-items): ensure offset is calculated only when page and limit are present

* feat(dataset-run-items): enhance dataset filtering by adding datasetId support and refactor related functions

* chore: rm comment

* refactor(dataset-run-items): rename dataset run name to ID across API and table definitions

* feat(dataset-run-items): add dataset_run_items to Clickhouse table names and refactor dataset filtering to use camelCase

* fix(dataset-run-items): update dataset filtering to ensure only the latest version of each dataset run item is retrieved

* chore: imports

* chore: fix de-duplication
2025-07-30 13:52:09 +00:00
NimarandGitHub bae2c5a65d fix(json-table): render links clickable (#8209)
* fix(json-table): render links clickable

* export
2025-07-30 15:48:01 +02:00
Steffen SchmitzandGitHub 3efa696513 fix(sessions): allow filter combination of environment and session_id (#8210) 2025-07-30 13:20:04 +00:00
Hassieb PakzadandGitHub 58ebf006b8 fix(llm-conenctions): drop system message from test call (#8189) 2025-07-30 11:49:31 +00:00
Thorsten SpiekerandGitHub 16743363dc fix: slack channel search component (#8208) 2025-07-30 13:56:44 +02:00
NimarandGitHub ebf0e35073 fix(json-view): smart retain collapsed/expanded state of json table view (#8191)
* handle user set expansion state in context provider

* update

* update2

* fix collapse state

* fix state bool

* final cleanup

* fix collapse all

* fix lint

* cleanup

* simplify

* remove log

* fix expand

* fix types

* lint
2025-07-30 11:21:42 +00:00
Thorsten SpiekerandGitHub 2c4799340f fix: set slack redirect uri specifically per environment (#8205) 2025-07-30 12:50:15 +02:00
Steffen SchmitzGitHubellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
5612c6a9a5 feat(cloud): add billing alerts (#7903)
* feat(cloud): add billing alerts

* chore: checkpoint state

* chore: cleanup

* chore: remove implementation plan

* chore: update send email behaviour

* chore: add usage alert config options in UI

* chore: implement stripe hook and email

* chore: optimize email

* chore: linting

* chore: use invoice hooks to retrigger usage alerts in each cycle

* chore: add log statement if usage alert got recreated

* chore: update log messages

* chore: replace cloud-billing-alerts with cloud-usage-alerts

* Update web/src/ee/features/billing/components/UsageAlerts.tsx

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>

---------

Co-authored-by: ellipsis-dev[bot] <65095814+ellipsis-dev[bot]@users.noreply.github.com>
2025-07-30 09:48:59 +00:00
Max DeichmannandGitHub b62962ca08 chore: move slack database migration in right order (#8203)
push
2025-07-30 11:12:28 +02:00
Thorsten SpiekerandGitHub 3a46283bf2 feat: add slack integration (#8073) 2025-07-30 10:52:57 +02:00
Marlies Mayerhofer 07801180f8 chore: release v3.88.1 2025-07-30 10:49:42 +02:00
marliessophieandGitHub 14831902ef fix(annotation-queues): remove unnecessary refetchOnMount option and ensure current item type is defined (#8199) 2025-07-30 08:28:46 +00:00
marliessophieandGitHub d377f02a49 fix(dataset-items): ensure deterministic ordering given pagination for GET dataset items api (#8200)
* fix(dataset-items): ensure deterministic ordering given pagination for GET dataset items api

* chore: push
2025-07-30 08:28:22 +00:00
Nimar e07120b163 chore: release v3.88.0 2025-07-29 20:27:11 +02:00
NimarandGitHub 506482dbe4 fix(playground): enable jumpt to playground for gemini (#8188)
* fix(playground): enable jumpt to playground for gemini

* fix exhaustive check

* correct types
2025-07-29 17:34:02 +00:00
Xichen PanandGitHub ec61f4a420 fix: set explicit ES2021 target in tsconfig to support Unicode regex (#8174)
* fix(worker): set explicit ES2021 target in tsconfig to support Unicode regex

* fix(worker): remove redundant lib config from tsconfig.json
2025-07-29 17:30:58 +00:00
marliessophieandGitHub 4db08a2960 fix(eval): trace filter preview enable dynamic sorting based on control visibility (#8187) 2025-07-29 17:11:37 +00:00
NimarandGitHub 2351d0d370 fix(markdown): don't break after list item number (#8181) 2025-07-29 12:03:49 +00:00
marliessophieandGitHub 99b3401549 feat(dataset-run-items): implement experiment service writes (#8090)
* feat(dataset-run-items): implement experiment service writes

* chore: add

* fix(experiments): update error messages for missing API key using constants

* chore(experimentCreateQueue): increase backoff delay to 10 seconds

* chore(auth): update ingestion types for improved flexibility

* chore: create traces & error-level generations for DRIs

* chore: unify timestamps

* chore: revert test to PG implementation until we add DRI migration across environments

* chore: typo

* chore: implement unified trace ID generation for ClickHouse and PostgreSQL executions to prevent duplicate traces

* chore: refactor trace creation logic for ClickHouse and PostgreSQL executions to use a unified approach

* chore: eslint

* chore: comment

* chore: rebase

* chore: rebase

* chore: rename function for clarity and fix typo in comments

* refactor: remove fetchDatasetRun function and replace with direct Prisma query for dataset run retrieval

* chore: eslint

* chore: fix test

* chore: fix test

* refactor: rename function and update comments for clarity in experimentServiceClickhouse
2025-07-29 11:14:30 +00:00
Nimar 6fb796df40 chore: release v3.87.1 2025-07-29 12:04:14 +02:00
NimarandGitHub 846d6e6c37 fix(markdown-rendering): tighter rendering (#8177)
* fix(trace-view): better markdown spacing for density

* better spacing
2025-07-29 09:55:05 +00:00
marliessophieandGitHub 1fa4fc6329 fix(annotation): mobile responsivness behaviour (#8154)
* fix(annotation): mobile behavior of annotation queue item view

* fix(ItemBadge): enhance label display with truncation and overflow handling

* fix(annotation-processing): improve layout by adjusting panel height properties for better overflow handling
2025-07-29 08:29:31 +00:00
Marlies Mayerhofer a1a48d5e7b chore: release v3.87.0 2025-07-29 00:15:18 +02:00
NimarandGitHub fcb7563763 chore: upgrade turbo to 2.5.5 (#8171) 2025-07-28 21:00:45 +00:00
Thorsten SpiekerandGitHub ae754b146d fix: scroll area for large lists of variables (#8170) 2025-07-28 21:13:56 +02:00
marliessophieandGitHub 49343b9a0c feat(annotation-queues): add support for batch adding sessions to annotation queues (#8169)
* feat(annotation-queues): add support for batch adding sessions to annotation queues

* chore: eslint types
2025-07-28 18:44:06 +00:00
Steffen SchmitzandGitHub 76bf5c0f4e fix: add queue prefix to webhook queue (#8165) 2025-07-28 16:12:48 +00:00
marliessophieandGitHub e91a29be61 fix(members-table): update table column visibility and order logic based on project context (#8164)
refactor(members-table): update column visibility and order logic based on project context
2025-07-28 14:38:14 +00:00
marliessophieandGitHub 81694363a2 fix(metrics): convert latency from milliseconds to seconds for accurate display (#8162) 2025-07-28 12:38:43 +00:00
marliessophieandGitHub 5998266680 feat(dataset-run-items): implement deletion for CH writes (#8074)
* feat(dataset): implement dataset run items deletion queue and processing

* feat(dataset): add dataset run items deletion functionality and integrate with deletion queue

* chore: revert imports

* chore: handle DRI deletion on project deletion conditionally

* fix(dataset-router): await Promise.all for dataset run items deletion

* refactor(dataset-router): streamline dataset deletion process and ensure async handling of dataset run items

* fix(env): add new environment variable for dataset run items deletion concurrency duration

* chore: rm comment

* refactor(dataset-run-items): update delete functions to use object destructuring for parameters

* fix(dataset-run-items): replace hardcoded request timeout with environment variable for deletion timeout

* refactor(dataset): rename and restructure dataset deletion functionality, replacing dataset run items deletion with a unified dataset deletion queue
2025-07-28 11:36:05 +00:00
Nimar 462e8e847d chore: release v3.86.1 2025-07-28 12:53:59 +02:00
NimarandGitHub d7c186858b fix(observations): if value is 0 show it instead of null (#8158) 2025-07-28 10:22:03 +00:00
NimarandGitHub e686aedac9 fix(trace-table): smart expand json table if not many rows (#8142)
* fix(trace-table): smart expand json table if not many rows

* clean up comments

* better performance

* cleanup

* fix lint

* clean
2025-07-28 09:58:02 +00:00
Max DeichmannandGitHub 85f75a5e8e chore: improve webhook retries (#8098)
* chore: improve webhook retries

* chore: improve webhook retries

* chore: improve webhook retries

* chore: improve webhook retries

* fixes

* push

* fixes

* fixes
2025-07-26 13:26:15 +00:00
Max DeichmannandGitHub 1ea643300d chore: remove ingestion via bullmq (#8143) 2025-07-26 00:13:53 +02:00
596486c1ec chore: ingest secondary ingestion queue data (#8137)
* push

* add duration tracking

* push

---------

Co-authored-by: Hassieb Pakzad <68423100+hassiebp@users.noreply.github.com>
2025-07-25 17:13:03 +00:00
Hassieb PakzadandGitHub cb65391c2d fix(model-selector): correct combinations provider model (#8135)
* fix(model-selector): correct combinations provider model

* push

* push

* push

* push

* push
2025-07-25 16:29:06 +00:00
marliessophieandGitHub 419e07260d fix(evaluator): update button label based on form mode (#8134) 2025-07-25 14:06:59 +00:00
marliessophieandGitHub 825f030e81 feat(annotation): support annotation on sessions (#8130)
* feat(session): integrate CreateNewAnnotationQueueItem component into session page

* refactor(annotation-queues): enhance QueueItemRowData structure to support multiple object types

* chore: refactor to extract common functionality

* fixup: refactor to extract common functionality

* chore: handle session annotation type in annotation processing UI

* feat(annotation-queues): add object type column and enhance source display in AnnotationQueueItemsTable

* chore: add session annotation ui

* feat(annotation-queues): add tooltip for detailed view toggle and enhance session item handling

* chore: small improvement

* feat(annotation-queues): implement intersection observer for trace visibility and update pagination logic

* fix(annotation-queues): update current item type retrieval and improve session item handling logic

* chore: eslint

* chore: nits
2025-07-25 13:32:54 +00:00
Max DeichmannandGitHub 34ded9c927 chore: adjust sampling metric (#8132) 2025-07-25 11:53:15 +02:00
Hassieb PakzadandGitHub c935d4ae73 chore: add sampling to ingestion pipeline (#8128) 2025-07-25 11:26:30 +02:00
Hassieb PakzadandGitHub e7283ac06e fix(ingestion): cached model not found (#8122) 2025-07-25 06:51:18 +00:00
Max DeichmannandGitHub 20d0106626 chore: fix date parsing for eval retries (#8119) 2025-07-25 02:00:48 +02:00
Max DeichmannandGitHub 849591fdf2 chore: improve evals logging (#8118) 2025-07-25 01:34:52 +02:00
Max DeichmannandGitHub 0c7b50e564 chore: fix eval event envelope timestmp (#8117) 2025-07-25 01:10:40 +02:00
Max DeichmannandGitHub 37b4f43351 chore: increase observability and delay of eval executions (#8116) 2025-07-25 00:08:54 +02:00
marliessophieandGitHub 7a3b0379e5 fix(experiments): update loading condition in CreateExperimentsForm to check for datasetId (#8115) 2025-07-24 21:11:42 +00:00
marliessophieandGitHub f458d7626c chore: add sessions type for annotation queue (#8107) 2025-07-24 17:11:47 +00:00
Max DeichmannandGitHub 258dde4691 fix: fix automation updates (#8108)
* fix: fix automation updates

* fix: fix automation updates

* fix: fix automation updates
2025-07-24 16:35:33 +00:00
marliessophieandGitHub f15246d9df feat(experiments): support remote experiment trigger (#8094)
* fixup(experiments): support trigger for remote experiment run

* chore: remove instructions

* chore: rename webhook > remote experiment server side

* chore: rename webhook > remote experiment client side

* style: design review

* chore: lint

* fix(experiments): change URL validation to use z.url() in RemoteExperimentUpsertForm

* chore: rename

* feat(datasets): add transformation function for datasets and update API responses
2025-07-24 16:12:59 +00:00
NimarandGitHub edb43a24ab fix(trace): show linebreaks in only text markdown view (#8106) 2025-07-24 16:08:00 +00:00
Steffen SchmitzandGitHub c45ae1b140 chore: adjust getSessionsTable query for AMTs (#8102)
* chore: adjust getSessionsTable query for AMTs

* chore: add additional index on traces_all_amt
2025-07-24 14:56:43 +00:00
Ted KimandGitHub 90eeeeaf9d fix: remove 'ON CLUSTER' instruction from unclustered migration (#7959) 2025-07-24 16:49:00 +02:00
Nimar 66accb563c chore: release v3.86.0 2025-07-24 16:30:43 +02:00
NimarandGitHub 9f2e8906d0 fix(trace-table): expanded state and make it fast! (#8099)
* preserve expanded state

* lazy load table for better performance

* fix broken commit

* pending: verbose state tracking working again

* fix states

* fix lint

* cleanup

* cleaner even

* fix build
2025-07-24 14:24:47 +00:00
c5d61b7ed2 chore: Remove add window button and its imports (#8097)
* Remove Add Window button from playground page

Co-authored-by: max <max@langfuse.com>

* feat(playground): remove Add Window button from header

- Remove Add Window button from playground page header
- Clean up unused imports (Plus icon, MULTI_WINDOW_CONFIG)
- Remove unused isAddWindowDisabled variable
- Preserve addWindow function for copy functionality
- Fix linting warnings for unused imports

Users can still add windows via the copy button on individual windows.

---------

Co-authored-by: Cursor Agent <cursoragent@cursor.com>
2025-07-24 13:05:27 +00:00
Max DeichmannandGitHub f95dd872e6 chore: fix output box for playground (#8096) 2025-07-24 12:38:11 +00:00
marliessophieandGitHub 04339741a9 chore(dataset-run-items): adjust converter for background migration (#8075)
* chore: reorder converter

* chore: treat numbers and booleans as strings

* chore: adjust metadata conversion

* Revert "chore: adjust metadata conversion"

This reverts commit d5ca3a5cb1b6e5972e5a321ab4884c589b938087.
2025-07-24 11:38:48 +00:00
marliessophieandGitHub 9c85e46996 feat(table): persist column sizes in local storage (#7816)
* feat(table): persist column sizing in localStorage

* chore: apply to all tables

* chore: push

* chore: add string
2025-07-24 11:37:16 +00:00
Steffen SchmitzandGitHub 4a7236451f chore: bump clickhouse client to 1.12.0 (#8092) 2025-07-24 11:20:21 +00:00
Steffen SchmitzandGitHub 5cb5204181 chore: migrate remaining trace reads to AMT (#8056)
* chore: migrate remaining trace reads to AMT

* chore: revert box ticks

* chore: revert streamed exports

* chore: confirm trace count by project in creation interval

* chore: confirm trace count since created

* chore: confirm getTracesByIdsForAnyProject

* chore: session table simplification

* chore: refactor user metrics

* chore: update
2025-07-24 10:49:44 +00:00
NimarandGitHub c19066afa0 fix(playground): api key link should show llm connection (#8080) 2025-07-24 09:43:03 +00:00
Max DeichmannandGitHub bbb9dc285d chore: remove delay for otel (#8085) 2025-07-24 11:16:59 +02:00
Steffen SchmitzandGitHub b750acba06 feat(admin-api): add okta role assignment capabilities (#8084)
* feat(admin-api): add okta role assignment capabilities

* chore: lint
2025-07-24 08:21:36 +00:00
Max DeichmannandGitHub bb06e079eb chore: upgrade prism (#8083)
* chore: upgrade prismjs

* fix
2025-07-24 07:49:50 +00:00
Max DeichmannandGitHub 5f711780e0 chore: upgrade prismjs (#8077) 2025-07-23 17:20:53 +00:00
Max DeichmannandGitHub 0f58ea1ebf chore: remove trivy (#8076) 2025-07-23 17:11:59 +00:00
Nimar ec70ca3951 chore: release v3.85.2 2025-07-23 18:20:34 +02:00
NimarandGitHub 142d3a1612 fix(trace-view): render new lines in json table view (#8072) 2025-07-23 16:09:32 +00:00
Max DeichmannandGitHub 1dee7092b3 chore: configure posthog SDK (#8067)
* chore: configure posthog SDK

* chore: add score values to scores

* update posthog integration

* update posthog integration
2025-07-23 15:27:06 +00:00
Steffen SchmitzandGitHub 83b64f5e77 perf: extend model cache duration and invalidate on model changes (#8053) 2025-07-23 16:39:04 +02:00
Steffen SchmitzandGitHub 48b05247c8 fix(admin-api): handle Okta provisioning for POST and PATCH (#8064) 2025-07-23 14:15:10 +00:00
Max DeichmannandGitHub 9732262466 chore: fix node types discrepancies across packages (#8060)
* chore: fix node types discrepancies across packages

* chore: fix node types discrepancies across packages
2025-07-23 12:23:43 +00:00
Max DeichmannandGitHub e51e2df2dc chore: fix snyk again (#8050)
* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes
2025-07-23 12:04:01 +00:00
Nimar 74d1dfe5ac chore: release v3.85.1 2025-07-23 13:55:34 +02:00
NimarandGitHub 5633a8867e fix(datasets-table): don't table render json here (#8058) 2025-07-23 11:52:50 +00:00
6a8bdae22e feat: multi window playground (#7661)
* docs: add technical specs for multi window/prompt playground

* docs: add requirements and tech summary for sharing

* feat: add multi-window playground architecture and state isolation

* feat: make playground UI multi-window capable

* fix: multi window playground caching

* fix: jump to playground button adds windows to playground using stable id

* fix: avoid race condition in writing to cache before navigating to playground

* feat: add collapsible, more compact sections to non-model config

* feat: add compact version of model picker/params component

* feat: add ability to copy window states to multi playground

* fix: solve header crowding for thin windows in playground

* feat: make entire playground config section collapsible

* feat: change SaveToPrompt button to align with other window buttons

* feat: switch buttons in page header to icon-only when too small

* fix: adjust button labels to fit button size

* fix: update the execution window status correctly

* docs: complete remaining task items

* refactor: simplify playground state management hooks

* fix: make the model configuration section prettier

* docs: remove agent work files

* refactor: call register functions on the registry directly

* chore: refactored some components and compacted the design even more

* chore: move add message buttons

* chore: start align model param settings popover when compact

* chore: small UI spacing improvements

* fix: scrolling bug

* fix: handle config headers when window gets small

* chore: rename button hover

* feat: move CMD+Enter to execute all button

* feat: give user the choice to jump to fresh/existing playground

* push

* chore: fix

* chore: fix

* push

* push

* fixes

---------

Co-authored-by: Max Deichmann <m.deichmann@tum.de>
2025-07-23 10:01:43 +00:00
Steffen SchmitzandGitHub dbf994f9bb perf: replace prisma upsert with raw query (#8036)
* perf: replace prisma upsert with raw query

* chore: drop prsima import

* chore: lint
2025-07-23 10:00:06 +00:00
Steffen SchmitzandGitHub 4d5c288d1e perf: initialize clickhouse skip read map on startup (#8039)
* perf: initialize clickhouse skip read map on startup

* chore: linting

* chore: patch skip cache init
2025-07-23 09:29:44 +00:00
Hassieb PakzadandGitHub cf119fb9cc fix(otel): parse trace name if as_root is true (#8008)
* fix(otel): parse trace name if as_root is true

* push

* push

* push
2025-07-23 09:13:54 +00:00
Hassieb PakzadandGitHub 5af8c4e400 fix(otel): parse public attribute for traces (#8049) 2025-07-23 10:18:31 +02:00
Max DeichmannandGitHub 835a19bbd2 fix: fix snyk uploads (#8044) 2025-07-22 20:48:38 +00:00
Max DeichmannandGitHub ae91a98ad9 fix: use drop down for prompt full text search (#8043)
* fix: use drop down for prompt full text search

* fix: use drop down for prompt full text search

* fix: use drop down for prompt full text search

* fix: use drop down for prompt full text search

* fix: use drop down for prompt full text search
2025-07-22 20:33:25 +00:00
NimarandGitHub e38cc24205 chore: add default claude settings (#8042) 2025-07-22 19:12:19 +00:00
marliessophieandGitHub 7ab45ec685 feat(dataset-run-items): migrate from postgres to clickhouse; write to CH API-only (#7912)
* feat(migrations): add dataset_run_items table in clickhouse

* feat(dataset): add environment variables and utility for pg/ch execution

* chore: re-write `runitemsByRunIdOrItemId` to support both clickhouse and postgres execution

* chore: use `runItemsByRunId` or `runItemsByItemId` rather than  `runitemsByRunIdOrItemId`

* chore: rename env variables

* chore: reflect dri as ch data is seeder

* fixup: dataset run item types and converters

* chore: background migration

* chore: rewrite dataset-router routes

* chore(dataset): dual-write strategy for dataset run items to support ClickHouse and PostgreSQL

* chore: add utilities and converters

* chore: rewrite GET api/public/dataset-run-items

* feat(dataset): implement dataset run items deletion queue and processing

* chore: rewrite DELETE api/public/datasets/{datasetName}/runs/{runName}

* chore: rewrite GET api/public/datasets/{datasetName}/runs/{runName}

* fixup: rewrite GET api/public/datasets/{datasetName}/runs/{runName}

* chore(schema): add error to DRI schema

* fixup(ingestion): support dataset run items

* fixup: rewrite POST /api/public/dataset-run-items​

* fixup: rewrite POST /api/public/dataset-run-items​ --amend

* fixup(ingestion): support dataset run items --amend

* fixup(experiments): preliminary implementation, for illustrative purposes only

* fixup(schema): indexes

* fixup(schema): remove indexes, add `output`

* chore: support DRI ingestion

* chore: eslint, writes related

* chore: rewrite experiment service

* chore: rm file

* chore: cascade delete dataset run items

* chore: add clustered migrations

* chore: update background migration

* chore: handle experiment job errors

* chore: remove CH read implementation

* chore: remove md file

* chore: remove CH read implementation II

* chore: fix typing

* chore: fix migration types

* fix: lint/imports

* fix: import errors

* chore: imports and comments

* chore: remove only from test

* fix: validateDatasetRunAndFetch

* fix: validateDatasetRunAndFetch id

* fixup: attempt fix of test

* fix: test

* chore: test validate logging

* chore: adjust logs

* chore: revert cascade delete

* Revert "feat(dataset): implement dataset run items deletion queue and processing"

This reverts commit 9f725f053386ded41e3617efaacee128d5ac023f.

* chore: delete imports

* revert: experiment service changes

* chore: adjust comment

* rename: env variables to incl experiment

* chore: remove unused variable eslint disables from datasetExecution and types files

* refactor: remove enrichedDatasetRunItem function and integrate its logic directly into IngestionService

* chore: remove TODO comment regarding dataset run item authorization

* refactor: replace Date constructor with parseClickhouseUTCDateTimeFormat for date parsing in dataset run item conversion

* chore: remove deprecated flag for runitemsByRunIdOrItemId method

* refactor: remove getDatasetRunItemsByRunId function to streamline dataset run item retrieval

* refactor: add back run properties in dataset runs mapping

* chore: revert changes to delete-dataset-run API

* refactor: remove validateDatasetItemAndFetch function and replace its usage with direct Prisma query in IngestionService

* refactor: delete validateCreateDatasetRunItemBodyAndFetch function and replace its logic with direct Prisma queries in dataset-run-items API

* refactor: move createOrFetchDatasetRun function to a new dataset-runs API file and maintain unique constraint handling

* refactor: remove validateDatasetRunAndFetch function and replace its usage with direct Prisma queries in dataset-run-items and IngestionService

* chore: nits

* refactor: streamline dataset run item creation by consolidating validation and execution logic

* chore: fix imports

* fix: types

* fix: simplify ingestion schema

* refactor: enhance ingestion schema creation to support public and internal environments

* chore: extend immutable keys list DRI

* chore: use same id for clickhouse and postgres

* chore: remove background migration trigger

* chore: move dataset run items migration into readme

* chore: remove creation of DRI in seeder

* chore: lint
2025-07-22 16:20:47 +00:00
Nimar b101bf48d3 chore: release v3.85.0 2025-07-22 16:47:56 +02:00
NimarandGitHub 4c4f0ff3c0 feat(tracing): pretty display json as collapsible table (#7938)
* feat(tracing): pretty display json as collapsible table

* show empty list and unwrap single item containing objects

* enable toggle for code or pretty view

* clean up

* fix type casting

* some more fixes

* expand row with click anywhere

* fix alignment

* remove dead code

* cleanup

* make it more clean

* don't unwrap initial items

* collapse expand all

* fixing collapse button

* don't render for chatml

* better markdown check

* reduce col width

* make table borderless

* display null / empty string for empty value

* smaller text

* make table more compact

* make preview items grey & italic

* linebreak path column

* table headers inherit color

* cleanup

* external collapsed state; also move button into code view

* remove code view switch

* show empty dicts also as empty

* always show objects as tables

* extract markdown into helper function

* review fixes

* update
2025-07-22 14:41:11 +00:00
Steffen SchmitzandGitHub a42ca96225 chore: expose email_from_address and smtp_connection_url in docker-compose (#8033)
chore: expose email_from_address and smtp_connection_url in docker-compose.yml
2025-07-22 13:26:03 +00:00
marliessophieandGitHub f63420827b feat(prompts/playground): allow arbitrary message sorting in messages (#8031) 2025-07-22 12:41:39 +00:00
Steffen SchmitzandGitHub e6fd056ae5 chore: migrate analytic routes to new AMTs (#8029)
* chore: migrate analytic routes to new AMTs

* chore: reduce new shared code

* chore: cleanup

* chore: cleanup
2025-07-22 12:39:48 +00:00
Steffen SchmitzandGitHub 4708fe4f47 chore: migrate additional session routes to new AMTs (#8027)
* chore: migrate additional session routes to new AMTs

* chore: remove unused function

* chore: remove unused code

* chore: remove outdated row in readme

* chore: update assertions
2025-07-22 12:24:57 +00:00
marliessophieandGitHub 362246b616 chore(datasets): update button label from "Rename" to "Edit" in DatasetActionButton component (#8030) 2025-07-22 12:21:45 +00:00
Steffen SchmitzandGitHub b3de5fd71a chore: migrate traces table to AMTs (#7853)
* chore: migrate traces table to AMTs

* chore: fix linting errors

* chore: ensure inplace update

* chore: add updated query

* chore: use final on query
2025-07-21 14:12:25 +00:00
Hassieb PakzadandGitHub ad236ecc99 fix(model-params): increase max tokens limit (#8005) 2025-07-21 15:12:37 +02:00
NimarandGitHub beb2063dcb fix: add select all to multi select dropdown (#8004) 2025-07-21 12:53:09 +00:00
Hassieb PakzadandGitHub 4de10298dc fix(otel): parse only trace io on trace update (#8003) 2025-07-21 12:11:04 +00:00
Danushka FernandoandGitHub f9924975fa feat: expose a key prefix option for Redis (#7958) 2025-07-21 14:01:14 +02:00
Steffen SchmitzandGitHub 98774cbd61 fix(sessions): infer session env from trace (#7996) 2025-07-21 10:44:17 +00:00
Steffen SchmitzandGitHub 80361ef53f fix(sessions): align duration unit with UI (#7998)
* fix(sessions): align duration unit with UI

* chore: update tests
2025-07-21 09:40:17 +00:00
Steffen SchmitzandGitHub 8350507434 chore: add missing blob-storage export entitlement (#7990) 2025-07-21 08:09:22 +00:00
marliessophieandGitHub 0a976edb70 fix(prompts): commitMessage error handling in NewPromptForm (#7984) 2025-07-20 16:13:39 +00:00
dependabot[bot]GitHubdependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>Max Deichmann
6a67c93010 chore(deps): bump exponential-backoff from 3.1.1 to 3.1.2 (#7412)
Bumps [exponential-backoff](https://github.com/coveooss/exponential-backoff) from 3.1.1 to 3.1.2.
- [Commits](https://github.com/coveooss/exponential-backoff/compare/v3.1.1...v3.1.2)

---
updated-dependencies:
- dependency-name: exponential-backoff
  dependency-version: 3.1.2
  dependency-type: direct:production
  update-type: version-update:semver-patch
...

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
Co-authored-by: Max Deichmann <m.deichmann@tum.de>
2025-07-20 09:33:39 +00:00
Max DeichmannandGitHub e51ee05d9c chore: upgrade express 5 (#7978)
* chore: upgrade express 5

* chore: upgrade express 5
2025-07-20 09:13:22 +00:00
418 changed files with 33314 additions and 7214 deletions
+14
View File
@@ -0,0 +1,14 @@
{
"permissions": {
"allow": [
"Bash(find:*)",
"Bash(rg:*)",
"Bash(grep:*)",
"Bash(ls:*)",
"Bash(cat:*)",
"Bash(head:*)",
"Bash(tail:*)"
],
"deny": []
}
}
-6
View File
@@ -1,6 +0,0 @@
{
"name": "langfuse-development",
"forwardPorts": [3000, 5432, 6379, 8123, 9000],
"onCreateCommand": "npm install -g pnpm@9.5.0",
"postCreateCommand": "curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz && git restore LICENSE README.md && chmod +x migrate && sudo mv migrate /usr/bin && cp .env.dev.example .env && npm install -g @anthropic-ai/claude-code && pnpm i"
}
+13
View File
@@ -0,0 +1,13 @@
# Dev container Dockerfile
FROM mcr.microsoft.com/vscode/devcontainers/typescript-node:20-bookworm
# Install golang-migrate for database migrations
RUN curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz && \
chmod +x migrate && \
mv migrate /usr/local/bin/migrate
# Install pnpm globally
RUN npm install -g pnpm@9.5.0
# Install Claude Code CLI
RUN npm install -g @anthropic-ai/claude-code
+8
View File
@@ -0,0 +1,8 @@
{
"name": "langfuse-development",
"build": {
"dockerfile": "Dockerfile"
},
"forwardPorts": [3000, 5432, 6379, 8123, 9000],
"postCreateCommand": "cp .env.dev.example .env && pnpm i"
}
+1
View File
@@ -68,6 +68,7 @@ LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
LANGFUSE_USE_AZURE_BLOB=true
LANGFUSE_AZURE_SKIP_CONTAINER_CHECK=false
# Set during docker build of application
# Used to disable environment verification at build time
+9
View File
@@ -76,9 +76,18 @@ REDIS_AUTH="bitnami"
REDIS_CLUSTER_ENABLED="true"
REDIS_CLUSTER_NODES="127.0.0.1:6370,127.0.0.1:6371,127.0.0.1:6372,127.0.0.1:6373,127.0.0.1:6374,127.0.0.1:6375"
LANGFUSE_INGESTION_QUEUE_SHARD_COUNT=8
LANGFUSE_TRACE_UPSERT_QUEUE_SHARD_COUNT=4
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
# Use the following settings to enforce running the new AMTs during the tests
LANGFUSE_EXPERIMENT_INSERT_INTO_AGGREGATING_MERGE_TREES="true"
LANGFUSE_EXPERIMENT_COMPARE_READ_FROM_AGGREGATING_MERGE_TREES="true"
LANGFUSE_EXPERIMENT_WHITELISTED_PROJECT_IDS="7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
LANGFUSE_EXPERIMENT_ADD_QUERY_RESULT_TO_SPAN_PROJECT_IDS="7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT="true"
LANGFUSE_EXPERIMENT_SAMPLING_RATE=1
+5
View File
@@ -83,3 +83,8 @@ NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
# For SDK integration tests to pass, decrease the ingestion queue delay by uncommenting the env vars:
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=10
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=10
# Slack credentials for development
SLACK_CLIENT_ID=your_slack_client_id
SLACK_CLIENT_SECRET=your_slack_client_secret
SLACK_STATE_SECRET=your_slack_state_secret
+1
View File
@@ -172,6 +172,7 @@ OTEL_SERVICE_NAME="langfuse"
# REDIS_HOST=
# REDIS_PORT=
# REDIS_AUTH=
# REDIS_USERNAME=default
# REDIS_CONNECTION_STRING=
# REDIS_ENABLE_AUTO_PIPELINING=
+1
View File
@@ -36,6 +36,7 @@ updates:
patterns:
- "express"
- "@types/express"
- "@types/express-serve-static-core"
observability:
patterns:
- "dd-trace"
+11 -1
View File
@@ -114,7 +114,9 @@ jobs:
run: echo "NEXT_PUBLIC_BUILD_ID=$(git rev-parse --short HEAD)" >> $GITHUB_ENV
- name: Build and run both images from compose
run: |
docker compose -f docker-compose.build.yml up -d
docker compose --progress plain --verbose -f docker-compose.build.yml build --print > /tmp/bake.json
docker buildx bake -f /tmp/bake.json
docker compose --progress plain -f docker-compose.build.yml up -d
sleep 5 # Wait for PostgreSQL to accept connections
- name: Ensure no unhealthy status
run: |
@@ -490,6 +492,14 @@ jobs:
if: ${{ contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') }}
run: exit 1
working-directory: .
- name: Notify Slack
uses: ravsamhq/notify-slack-action@v2
if: always() && github.event_name == 'push' && (github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/'))
with:
status: ${{ job.status }}
notify_when: "failure"
env:
SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL }}
push-docker-image:
needs: all-ci-passed
+10 -42
View File
@@ -1,60 +1,28 @@
name: Snyk Container
on:
push:
branches:
- "main"
# Snyk cannot upload results in merge group. Hence, we only run when pushing https://github.com/github/codeql-action/issues/1572
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
branches: ["main"]
jobs:
snyk:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v2
- uses: actions/checkout@v3
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-web)
continue-on-error: true
- name: Scan web image with Snyk
uses: snyk/actions/docker@master
continue-on-error: true # let upload step always run
env:
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
with:
image: langfuse/langfuse
args: --file=web/Dockerfile
image: langfuse/langfuse # pulled from Docker Hub
args: --severity-threshold=high # (any extra CLI flags, but NO --file)
# Workaround for https://github.com/github/codeql-action/issues/2187
- name: Replace security-severity undefined for license-related findings
run: 'sed -i ''s/"security-severity": "undefined"/"security-severity": "0"/g'' snyk.sarif'
- name: Replace security-severity null for license-related findings
run: 'sed -i ''s/"security-severity": "null"/"security-severity": "0"/g'' snyk.sarif'
- name: Upload result to GitHub Code Scanning
uses: github/codeql-action/upload-sarif@v3
with:
sarif_file: snyk.sarif
category: web
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-worker)
continue-on-error: true
- name: Scan worker image with Snyk
uses: snyk/actions/docker@master
continue-on-error: true # let upload step always run
env:
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
with:
image: langfuse/langfuse-worker
args: --file=worker/Dockerfile
# Workaround for https://github.com/github/codeql-action/issues/2187
- name: Replace security-severity undefined for license-related findings
run: 'sed -i ''s/"security-severity": "undefined"/"security-severity": "0"/g'' snyk.sarif'
- name: Replace security-severity null for license-related findings
run: 'sed -i ''s/"security-severity": "null"/"security-severity": "0"/g'' snyk.sarif'
- name: Upload result to GitHub Code Scanning
uses: github/codeql-action/upload-sarif@v3
with:
sarif_file: snyk.sarif
category: worker
image: langfuse/langfuse-worker # pulled from Docker Hub
args: --severity-threshold=high # (any extra CLI flags, but NO --file)
+1 -3
View File
@@ -50,15 +50,13 @@ yarn-error.log*
# typescript
*.tsbuildinfo
/generated/typescript-server
/generated
# openapi spec that is copied during build
/public/openapi*.yml
# vscode
.devcontainer
node_modules
**/node_modules
**/dist
+1 -1
View File
@@ -1 +1 @@
v20
v20.19.2
+3
View File
@@ -192,3 +192,6 @@ To get a project, use the `get_project` capability with the full project name as
## General Coding Guidelines
- For easier code reviews, prefer not to move functions etc around within a file unless necessary or instructed to do so
## Development Tips
- Before trying to build the package, try running the linter once first
+3 -1
View File
@@ -19,6 +19,7 @@ services:
ports:
- 127.0.0.1:3030:3030
environment: &langfuse-worker-env
NEXTAUTH_URL: http://localhost:3000
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres # CHANGEME
SALT: "mysalt" # CHANGEME
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # CHANGEME: generate via `openssl rand -hex 32`
@@ -62,6 +63,8 @@ services:
REDIS_TLS_CA: ${REDIS_TLS_CA:-/certs/ca.crt}
REDIS_TLS_CERT: ${REDIS_TLS_CERT:-/certs/redis.crt}
REDIS_TLS_KEY: ${REDIS_TLS_KEY:-/certs/redis.key}
EMAIL_FROM_ADDRESS: ${EMAIL_FROM_ADDRESS:-}
SMTP_CONNECTION_URL: ${SMTP_CONNECTION_URL:-}
langfuse-web:
image: docker.io/langfuse/langfuse:3
@@ -71,7 +74,6 @@ services:
- 3000:3000
environment:
<<: *langfuse-worker-env
NEXTAUTH_URL: http://localhost:3000
NEXTAUTH_SECRET: mysecret # CHANGEME
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
-1
View File
@@ -26,7 +26,6 @@
"dependencies": {
"@langfuse/shared": "workspace:*",
"@opentelemetry/api": ">=1.0.0 <1.10.0",
"axios": "^1.8.2",
"https-proxy-agent": "^7.0.6",
"next": "^14.2.30",
"next-auth": "^4.24.11",
@@ -105,6 +105,28 @@ service:
docs: The unique identifier of the annotation queue item
response: DeleteAnnotationQueueItemResponse
createQueueAssignment:
docs: Create an assignment for a user to an annotation queue
method: POST
path: /annotation-queues/{queueId}/assignments
path-parameters:
queueId:
type: string
docs: The unique identifier of the annotation queue
request: AnnotationQueueAssignmentRequest
response: CreateAnnotationQueueAssignmentResponse
deleteQueueAssignment:
docs: Delete an assignment for a user to an annotation queue
method: DELETE
path: /annotation-queues/{queueId}/assignments
path-parameters:
queueId:
type: string
docs: The unique identifier of the annotation queue
request: AnnotationQueueAssignmentRequest
response: DeleteAnnotationQueueAssignmentResponse
types:
AnnotationQueueStatus:
enum:
@@ -163,3 +185,17 @@ types:
properties:
success: boolean
message: string
AnnotationQueueAssignmentRequest:
properties:
userId: string
DeleteAnnotationQueueAssignmentResponse:
properties:
success: boolean
CreateAnnotationQueueAssignmentResponse:
properties:
userId: string
queueId: string
projectId: string
@@ -0,0 +1,102 @@
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
imports:
commons: ./commons.yml
pagination: ./utils/pagination.yml
service:
auth: true
base-path: /api/public
endpoints:
list:
method: GET
docs: Get all LLM connections in a project
path: /llm-connections
request:
name: GetLlmConnectionsRequest
query-parameters:
page:
type: optional<integer>
docs: page number, starts at 1
limit:
type: optional<integer>
docs: limit of items per page
response: PaginatedLlmConnections
upsert:
method: PUT
docs: Create or update an LLM connection. The connection is upserted on provider.
path: /llm-connections
request: UpsertLlmConnectionRequest
response: LlmConnection
types:
LlmConnection:
docs: LLM API connection configuration (secrets excluded)
properties:
id: string
provider:
type: string
docs: Provider name (e.g., 'openai', 'my-gateway'). Must be unique in project, used for upserting.
adapter:
type: string
docs: The adapter used to interface with the LLM
displaySecretKey:
type: string
docs: Masked version of the secret key for display purposes
baseURL:
type: optional<string>
docs: Custom base URL for the LLM API
customModels:
type: list<string>
docs: List of custom model names available for this connection
withDefaultModels:
type: boolean
docs: Whether to include default models for this adapter
extraHeaderKeys:
type: list<string>
docs: Keys of extra headers sent with requests (values excluded for security)
createdAt: datetime
updatedAt: datetime
PaginatedLlmConnections:
properties:
data: list<LlmConnection>
meta: pagination.MetaResponse
UpsertLlmConnectionRequest:
docs: Request to create or update an LLM connection (upsert)
properties:
provider:
type: string
docs: Provider name (e.g., 'openai', 'my-gateway'). Must be unique in project, used for upserting.
adapter:
type: LlmAdapter
docs: The adapter used to interface with the LLM
secretKey:
type: string
docs: Secret key for the LLM API.
baseURL:
type: optional<string>
docs: Custom base URL for the LLM API
customModels:
type: optional<list<string>>
docs: List of custom model names
withDefaultModels:
type: optional<boolean>
docs: Whether to include default models. Default is true.
extraHeaders:
type: optional<map<string, string>>
docs: Extra headers to send with requests
LlmAdapter:
enum:
- value: anthropic
name: Anthropic
- value: openai
name: OpenAI
- value: azure
name: Azure
- value: bedrock
name: Bedrock
- value: google-vertex-ai
name: GoogleVertexAI
- value: google-ai-studio
name: GoogleAIStudio
+26 -27
View File
@@ -1,3 +1,4 @@
# yaml-language-server: $schema=https://schema.buildwithfern.dev/generators-yml.json
default-group: local
groups:
local:
@@ -7,6 +8,7 @@ groups:
output:
location: local-file-system
path: ../../../web/public/generated/api
- name: fernapi/fern-python-sdk
version: 2.16.0
output:
@@ -19,35 +21,32 @@ groups:
pydantic_config:
require_optional_fields: false
use_str_enums: false
# - name: fernapi/fern-java-sdk
# version: 2.20.1
# output:
# location: local-file-system
# path: ../../../../langfuse-java/src/main/java/com/langfuse/client/
# config:
# client-class-name: LangfuseClient
# - name: fernapi/fern-java-sdk
# version: 2.20.1
# output:
# location: local-file-system
# path: ../../../../langfuse-java/src/main/java/com/langfuse/client/
# config:
# client-class-name: LangfuseClient
- name: fernapi/fern-typescript-node-sdk
version: 2.6.1
output:
location: local-file-system
path: ../../../generated/typescript
config:
namespaceExport: LangfuseAPI
outputSourceFiles: true
skipResponseValidation: true
fetchSupport: native
formDataSupport: Node18
fileResponseType: binary-response
streamType: web
omitFernHeaders: true
- name: fernapi/fern-postman
version: 0.0.45
output:
location: local-file-system
path: ../../../web/public/generated/postman
# published:
# generators:
# - name: fernapi/fern-python-sdk
# version: 0.3.7
# output:
# location: pypi
# url: pypi.buildwithfern.com
# package-name: finto-fern-langfuse
# config:
# namespaceExport: Langfuse
# allowCustomFetcher: true
# - name: fernapi/fern-typescript-node-sdk
# version: 0.7.1
# output:
# location: npm
# url: npm.buildwithfern.com
# package-name: "@finto-fern/langfuse-node"
# config:
# namespaceExport: Langfuse
# allowCustomFetcher: true
-1
View File
@@ -1 +0,0 @@
python/
+4 -3
View File
@@ -1,6 +1,6 @@
{
"name": "langfuse",
"version": "3.84.0",
"version": "3.97.2",
"author": "engineering@langfuse.com",
"license": "MIT",
"private": true,
@@ -40,7 +40,7 @@
"husky": "^9.0.11",
"prettier": "^3.6.2",
"release-it": "^19.0.3",
"turbo": "^2.5.4"
"turbo": "^2.5.5"
},
"release-it": {
"git": {
@@ -91,7 +91,8 @@
"nanoid": "^3.3.8",
"katex": "^0.16.21",
"tar-fs": "^2.1.2",
"rollup@^4.0.0": "^4.22.4"
"rollup@^4.0.0": "^4.22.4",
"@types/node-fetch": "^2.6.13"
},
"patchedDependencies": {
"next-auth@4.24.11": "patches/next-auth@4.24.11.patch"
+1 -1
View File
@@ -13,7 +13,7 @@
"@vercel/style-guide": "^6.0.0",
"eslint-config-next": "^14.2.15",
"eslint-config-prettier": "^9.1.0",
"eslint-config-turbo": "^2.5.4",
"eslint-config-turbo": "^2.5.5",
"eslint-plugin-only-warn": "^1.1.0",
"typescript": "^5.4.5"
}
@@ -0,0 +1 @@
DROP TABLE dataset_run_items ON CLUSTER default;
@@ -0,0 +1,36 @@
CREATE TABLE dataset_run_items ON CLUSTER default (
-- primary identifiers
`id` String,
`project_id` String,
`dataset_run_id` String,
`dataset_item_id` String,
`dataset_id` String,
`trace_id` String,
`observation_id` Nullable(String),
-- error field
`error` Nullable(String),
-- timestamps
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
-- denormalized immutable dataset run fields
`dataset_run_name` String,
`dataset_run_description` Nullable(String),
`dataset_run_metadata` Map(LowCardinality(String), String),
`dataset_run_created_at` DateTime64(3),
-- denormalized dataset item fields (mutable, but snapshots are relevant)
`dataset_item_input` Nullable(String) CODEC(ZSTD(3)), -- json
`dataset_item_expected_output` Nullable(String) CODEC(ZSTD(3)), -- json
`dataset_item_metadata` Map(LowCardinality(String), String),
-- clickhouse engine fields
`event_ts` DateTime64(3),
`is_deleted` UInt8,
-- For dataset item lookups
INDEX idx_dataset_item dataset_item_id TYPE bloom_filter(0.001) GRANULARITY 1,
) ENGINE = ReplacingMergeTree(event_ts, is_deleted)
ORDER BY (project_id, dataset_id, dataset_run_id, id);
@@ -0,0 +1,12 @@
-- Drop materialized views first
DROP VIEW IF EXISTS traces_30d_amt_mv ON CLUSTER default;
DROP VIEW IF EXISTS traces_7d_amt_mv ON CLUSTER default;
DROP VIEW IF EXISTS traces_all_amt_mv ON CLUSTER default;
-- Drop AMT tables
DROP TABLE IF EXISTS traces_30d_amt ON CLUSTER default;
DROP TABLE IF EXISTS traces_7d_amt ON CLUSTER default;
DROP TABLE IF EXISTS traces_all_amt ON CLUSTER default;
-- Drop the Null table
DROP TABLE IF EXISTS traces_null ON CLUSTER default;
@@ -0,0 +1,300 @@
-- Create a Null table that serves as a trigger for all materialized views.
-- We use a Null engine here to avoid storing intermediate results and save on storage.
CREATE TABLE traces_null ON CLUSTER default
(
-- Identifiers
`project_id` String,
`id` String,
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`name` Nullable(String),
-- Metadata properties
`metadata` Map(LowCardinality(String), String),
`user_id` Nullable(String),
`session_id` Nullable(String),
`environment` String,
`tags` Array(String),
`version` Nullable(String),
`release` Nullable(String),
-- UI properties - We make them nullable to prevent absent values being interpreted as overwrites.
`bookmarked` Nullable(Bool),
`public` Nullable(Bool),
-- Aggregations -- DO NOT USE
`observation_ids` Array(String),
`score_ids` Array(String),
`cost_details` Map(String, Decimal64(12)),
`usage_details` Map(String, UInt64),
-- TODO: Do we want to aggregate/collect `levels` seen within the trace?
-- Input/Output
`input` String,
`output` String,
`created_at` DateTime64(3),
`updated_at` DateTime64(3),
`event_ts` DateTime64(3)
) Engine = Null();
-- Create the all AMT
CREATE TABLE traces_all_amt ON CLUSTER default
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations -- DO NOT USE
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_trace_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id);
-- Create materialized view for all_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_all_amt_mv ON CLUSTER default TO traces_all_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations -- DO NOT USE
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
-- Create the 7-day TTL AMT
CREATE TABLE traces_7d_amt ON CLUSTER default
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations -- DO NOT USE
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id)
TTL toDate(start_time) + INTERVAL 7 DAY;
-- Create materialized view for 7d_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_7d_amt_mv ON CLUSTER default TO traces_7d_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations -- DO NOT USE
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
-- Create the 30-day TTL AMT
CREATE TABLE traces_30d_amt ON CLUSTER default
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations -- DO NOT USE
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id)
TTL toDate(start_time) + INTERVAL 30 DAY;
-- Create materialized view for 30d_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_30d_amt_mv ON CLUSTER default TO traces_30d_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations -- DO NOT USE
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
@@ -1 +1 @@
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_user_id;
ALTER TABLE traces DROP INDEX IF EXISTS idx_user_id;
@@ -0,0 +1 @@
DROP TABLE dataset_run_items;
@@ -0,0 +1,36 @@
CREATE TABLE dataset_run_items (
-- primary identifiers
`id` String,
`project_id` String,
`dataset_run_id` String,
`dataset_item_id` String,
`dataset_id` String,
`trace_id` String,
`observation_id` Nullable(String),
-- error field
`error` Nullable(String),
-- timestamps
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
-- denormalized immutable dataset run fields
`dataset_run_name` String,
`dataset_run_description` Nullable(String),
`dataset_run_metadata` Map(LowCardinality(String), String),
`dataset_run_created_at` DateTime64(3),
-- denormalized dataset item fields (mutable, but snapshots are relevant)
`dataset_item_input` Nullable(String) CODEC(ZSTD(3)), -- json
`dataset_item_expected_output` Nullable(String) CODEC(ZSTD(3)), -- json
`dataset_item_metadata` Map(LowCardinality(String), String),
-- clickhouse engine fields
`event_ts` DateTime64(3),
`is_deleted` UInt8,
-- For dataset item lookups
INDEX idx_dataset_item dataset_item_id TYPE bloom_filter(0.001) GRANULARITY 1,
) ENGINE = ReplacingMergeTree(event_ts, is_deleted)
ORDER BY (project_id, dataset_id, dataset_run_id, id);
@@ -0,0 +1,12 @@
-- Drop materialized views first
DROP VIEW IF EXISTS traces_30d_amt_mv;
DROP VIEW IF EXISTS traces_7d_amt_mv;
DROP VIEW IF EXISTS traces_all_amt_mv;
-- Drop AMT tables
DROP TABLE IF EXISTS traces_30d_amt;
DROP TABLE IF EXISTS traces_7d_amt;
DROP TABLE IF EXISTS traces_all_amt;
-- Drop the Null table
DROP TABLE IF EXISTS traces_null;
@@ -0,0 +1,300 @@
-- Create a Null table that serves as a trigger for all materialized views.
-- We use a Null engine here to avoid storing intermediate results and save on storage.
CREATE TABLE traces_null
(
-- Identifiers
`project_id` String,
`id` String,
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`name` Nullable(String),
-- Metadata properties
`metadata` Map(LowCardinality(String), String),
`user_id` Nullable(String),
`session_id` Nullable(String),
`environment` String,
`tags` Array(String),
`version` Nullable(String),
`release` Nullable(String),
-- UI properties - We make them nullable to prevent absent values being interpreted as overwrites.
`bookmarked` Nullable(Bool),
`public` Nullable(Bool),
-- Aggregations
`observation_ids` Array(String),
`score_ids` Array(String),
`cost_details` Map(String, Decimal64(12)),
`usage_details` Map(String, UInt64),
-- TODO: Do we want to aggregate/collect `levels` seen within the trace?
-- Input/Output
`input` String,
`output` String,
`created_at` DateTime64(3),
`updated_at` DateTime64(3),
`event_ts` DateTime64(3)
) Engine = Null();
-- Create the all AMT
CREATE TABLE traces_all_amt
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_trace_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id);
-- Create materialized view for all_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_all_amt_mv TO traces_all_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
-- Create the 7-day TTL AMT
CREATE TABLE traces_7d_amt
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id)
TTL toDate(start_time) + INTERVAL 7 DAY;
-- Create materialized view for 7d_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_7d_amt_mv TO traces_7d_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
-- Create the 30-day TTL AMT
CREATE TABLE traces_30d_amt
(
-- Identifiers
`project_id` String,
`id` String,
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
-- Metadata properties
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
`environment` SimpleAggregateFunction(anyLast, String),
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
-- UI properties
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
-- Aggregations
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
-- Input/Output -> prefer correctness via argMax
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
-- Indexes
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
) Engine = AggregatingMergeTree()
ORDER BY (project_id, id)
TTL toDate(start_time) + INTERVAL 30 DAY;
-- Create materialized view for 30d_amt
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_30d_amt_mv TO traces_30d_amt AS
SELECT
-- Identifiers
tn.project_id as project_id,
tn.id as id,
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
min(tn.start_time) as start_time,
max(coalesce(tn.end_time, tn.start_time)) as end_time,
anyLast(tn.name) as name,
-- Metadata properties
maxMap(tn.metadata) as metadata,
anyLast(tn.user_id) as user_id,
anyLast(tn.session_id) as session_id,
anyLast(tn.environment) as environment,
groupUniqArrayArray(tn.tags) as tags,
anyLast(tn.version) as version,
anyLast(tn.release) as release,
-- UI properties
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
-- Aggregations
groupUniqArrayArray(tn.observation_ids) as observation_ids,
groupUniqArrayArray(tn.score_ids) as score_ids,
sumMap(tn.cost_details) as cost_details,
sumMap(tn.usage_details) as usage_details,
-- Input/Output
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
min(tn.created_at) as created_at,
max(tn.updated_at) as updated_at
FROM traces_null tn
GROUP BY project_id, id;
+4 -3
View File
@@ -61,7 +61,7 @@
"@aws-sdk/lib-storage": "^3.675.0",
"@aws-sdk/s3-request-presigner": "^3.679.0",
"@azure/storage-blob": "^12.26.0",
"@clickhouse/client": "^1.11.2",
"@clickhouse/client": "^1.12.0",
"@google-cloud/storage": "^7.15.2",
"@langchain/anthropic": "^0.3.22",
"@langchain/aws": "^0.1.11",
@@ -73,13 +73,14 @@
"@prisma/client": "^6.10.1",
"@react-email/components": "^0.1.0",
"@react-email/render": "^1.1.2",
"@slack/oauth": "^3.0.3",
"@slack/web-api": "^7.9.3",
"@types/bcryptjs": "^2.4.6",
"axios": "^1.8.2",
"bcryptjs": "^2.4.3",
"bullmq": "^5.34.10",
"dd-trace": "^5.36.0",
"decimal.js": "^10.4.3",
"exponential-backoff": "^3.1.1",
"exponential-backoff": "^3.1.2",
"https-proxy-agent": "^7.0.6",
"ioredis": "^5.4.1",
"jsonpath-plus": "10.3.0",
+49
View File
@@ -55,6 +55,7 @@ export type AnnotationQueueStatus =
export const AnnotationQueueObjectType = {
TRACE: "TRACE",
OBSERVATION: "OBSERVATION",
SESSION: "SESSION",
} as const;
export type AnnotationQueueObjectType =
(typeof AnnotationQueueObjectType)[keyof typeof AnnotationQueueObjectType];
@@ -138,6 +139,7 @@ export type DashboardWidgetChartType =
(typeof DashboardWidgetChartType)[keyof typeof DashboardWidgetChartType];
export const ActionType = {
WEBHOOK: "WEBHOOK",
SLACK: "SLACK",
} as const;
export type ActionType = (typeof ActionType)[keyof typeof ActionType];
export const ActionExecutionStatus = {
@@ -148,6 +150,11 @@ export const ActionExecutionStatus = {
} as const;
export type ActionExecutionStatus =
(typeof ActionExecutionStatus)[keyof typeof ActionExecutionStatus];
export const SurveyName = {
ORG_ONBOARDING: "org_onboarding",
USER_ONBOARDING: "user_onboarding",
} as const;
export type SurveyName = (typeof SurveyName)[keyof typeof SurveyName];
export type Account = {
id: string;
user_id: string;
@@ -183,6 +190,14 @@ export type AnnotationQueue = {
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type AnnotationQueueAssignment = {
id: string;
project_id: string;
user_id: string;
queue_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type AnnotationQueueItem = {
id: string;
queue_id: string;
@@ -359,6 +374,8 @@ export type Dataset = {
name: string;
description: string | null;
metadata: unknown | null;
remote_experiment_url: string | null;
remote_experiment_payload: unknown | null;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
@@ -624,6 +641,15 @@ export type OrganizationMembership = {
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type PendingDeletion = {
id: string;
project_id: string;
object: string;
object_id: string;
is_deleted: Generated<boolean>;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type PosthogIntegration = {
project_id: string;
encrypted_posthog_api_key: string;
@@ -711,6 +737,16 @@ export type Session = {
user_id: string;
expires: Timestamp;
};
export type SlackIntegration = {
id: string;
project_id: string;
team_id: string;
team_name: string;
bot_token: string;
bot_user_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type SsoConfig = {
domain: string;
created_at: Generated<Timestamp>;
@@ -718,6 +754,15 @@ export type SsoConfig = {
auth_provider: string;
auth_config: unknown | null;
};
export type Survey = {
id: string;
created_at: Generated<Timestamp>;
survey_name: SurveyName;
response: unknown;
user_id: string | null;
user_email: string | null;
org_id: string | null;
};
export type TableViewPreset = {
id: string;
created_at: Generated<Timestamp>;
@@ -781,6 +826,7 @@ export type VerificationToken = {
export type DB = {
Account: Account;
actions: Action;
annotation_queue_assignments: AnnotationQueueAssignment;
annotation_queue_items: AnnotationQueueItem;
annotation_queues: AnnotationQueue;
api_keys: ApiKey;
@@ -813,6 +859,7 @@ export type DB = {
observations: LegacyPrismaObservation;
organization_memberships: OrganizationMembership;
organizations: Organization;
pending_deletions: PendingDeletion;
posthog_integrations: PosthogIntegration;
prices: Price;
project_memberships: ProjectMembership;
@@ -823,7 +870,9 @@ export type DB = {
score_configs: ScoreConfig;
scores: LegacyPrismaScore;
Session: Session;
slack_integrations: SlackIntegration;
sso_configs: SsoConfig;
surveys: Survey;
table_view_presets: TableViewPreset;
trace_media: TraceMedia;
trace_sessions: TraceSession;
@@ -0,0 +1,3 @@
-- AlterTable
ALTER TABLE "datasets" ADD COLUMN "remote_experiment_payload" JSONB,
ADD COLUMN "remote_experiment_url" TEXT;
@@ -0,0 +1,2 @@
-- AlterEnum
ALTER TYPE "AnnotationQueueObjectType" ADD VALUE 'SESSION';
@@ -0,0 +1,30 @@
-- Migration: Add Slack Integration Support
-- This migration adds support for Slack automation actions by:
-- 1. Adding SLACK to the ActionType enum
-- 2. Creating slack_integrations table for centralized token storage
-- AlterEnum
ALTER TYPE "ActionType" ADD VALUE 'SLACK';
-- CreateTable
CREATE TABLE "slack_integrations" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"team_id" TEXT NOT NULL,
"team_name" TEXT NOT NULL,
"bot_token" TEXT NOT NULL,
"bot_user_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
CONSTRAINT "slack_integrations_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE UNIQUE INDEX "slack_integrations_project_id_key" ON "slack_integrations"("project_id");
-- CreateIndex
CREATE INDEX "slack_integrations_team_id_idx" ON "slack_integrations"("team_id");
-- AddForeignKey
ALTER TABLE "slack_integrations" ADD CONSTRAINT "slack_integrations_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
@@ -0,0 +1,2 @@
INSERT INTO background_migrations (id, name, script, args)
VALUES ('8d47f91b-3e5c-4a26-9f85-c12d6e4b9a3d', '20250731_1001_migrate_dataset_run_items_pg_to_ch', 'migrateDatasetRunItemsFromPostgresToClickhouse', '{}');
@@ -0,0 +1,21 @@
-- CreateTable
CREATE TABLE "pending_deletions" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"object" TEXT NOT NULL,
"object_id" TEXT NOT NULL,
"is_deleted" BOOLEAN NOT NULL DEFAULT false,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
CONSTRAINT "pending_deletions_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE INDEX "pending_deletions_project_id_object_is_deleted_idx" ON "pending_deletions"("project_id", "object", "is_deleted");
-- CreateIndex
CREATE INDEX "pending_deletions_object_id_object_idx" ON "pending_deletions"("object_id", "object");
-- AddForeignKey
ALTER TABLE "pending_deletions" ADD CONSTRAINT "pending_deletions_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
@@ -0,0 +1,23 @@
-- CreateTable
CREATE TABLE "annotation_queue_assignments" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"user_id" TEXT NOT NULL,
"queue_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
CONSTRAINT "annotation_queue_assignments_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE UNIQUE INDEX "annotation_queue_assignments_project_id_queue_id_key" ON "annotation_queue_assignments"("project_id", "queue_id", "user_id");
-- AddForeignKey
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_user_id_fkey" FOREIGN KEY ("user_id") REFERENCES "users"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_queue_id_fkey" FOREIGN KEY ("queue_id") REFERENCES "annotation_queues"("id") ON DELETE CASCADE ON UPDATE CASCADE;
@@ -0,0 +1,24 @@
-- CreateEnum
CREATE TYPE "SurveyName" AS ENUM ('org_onboarding', 'user_onboarding');
-- CreateTable
CREATE TABLE "surveys" (
"id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"survey_name" "SurveyName" NOT NULL,
"response" JSONB NOT NULL,
"user_id" TEXT,
"user_email" TEXT,
"org_id" TEXT,
CONSTRAINT "surveys_pkey" PRIMARY KEY ("id")
);
-- AddForeignKey
ALTER TABLE "surveys" ADD CONSTRAINT "surveys_org_id_fkey" FOREIGN KEY ("org_id") REFERENCES "organizations"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "surveys" ADD CONSTRAINT "surveys_user_id_fkey" FOREIGN KEY ("user_id") REFERENCES "users"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- RenameIndex
ALTER INDEX "annotation_queue_assignments_project_id_queue_id_key" RENAME TO "annotation_queue_assignments_project_id_queue_id_user_id_key";
+173 -88
View File
@@ -65,29 +65,31 @@ model Session {
}
model User {
id String @id @default(cuid())
name String?
email String? @unique
emailVerified DateTime? @map("email_verified")
password String?
image String?
admin Boolean @default(false)
accounts Account[]
sessions Session[]
organizationMemberships OrganizationMembership[]
projectMemberships ProjectMembership[]
invitations MembershipInvitation[]
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
featureFlags String[] @default([]) @map("feature_flags")
annotatedLockedItem AnnotationQueueItem[] @relation("LockedByUser")
annotatedCompletedItem AnnotationQueueItem[] @relation("AnnotatorUser")
dashboardWidgetsCreated DashboardWidget[] @relation("CreatedByUser")
dashboardWidgetsUpdated DashboardWidget[] @relation("UpdatedByUser")
dashboardCreated Dashboard[] @relation("CreatedByUser")
dashboardUpdated Dashboard[] @relation("UpdatedByUser")
tableViewPresetCreated TableViewPreset[] @relation("CreatedByUser")
tableViewPresetUpdated TableViewPreset[] @relation("UpdatedByUser")
id String @id @default(cuid())
name String?
email String? @unique
emailVerified DateTime? @map("email_verified")
password String?
image String?
admin Boolean @default(false)
accounts Account[]
sessions Session[]
organizationMemberships OrganizationMembership[]
projectMemberships ProjectMembership[]
invitations MembershipInvitation[]
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
featureFlags String[] @default([]) @map("feature_flags")
annotatedLockedItem AnnotationQueueItem[] @relation("LockedByUser")
annotatedCompletedItem AnnotationQueueItem[] @relation("AnnotatorUser")
dashboardWidgetsCreated DashboardWidget[] @relation("CreatedByUser")
dashboardWidgetsUpdated DashboardWidget[] @relation("UpdatedByUser")
dashboardCreated Dashboard[] @relation("CreatedByUser")
dashboardUpdated Dashboard[] @relation("UpdatedByUser")
tableViewPresetCreated TableViewPreset[] @relation("CreatedByUser")
tableViewPresetUpdated TableViewPreset[] @relation("UpdatedByUser")
annotationQueueAssignment AnnotationQueueAssignment[]
surveys Survey[]
@@map("users")
}
@@ -112,57 +114,61 @@ model Organization {
projects Project[]
MembershipInvitation MembershipInvitation[]
ApiKey ApiKey[]
surveys Survey[]
@@map("organizations")
}
model Project {
id String @id @default(cuid())
orgId String @map("org_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
deletedAt DateTime? @map("deleted_at")
name String
retentionDays Int? @map("retention_days")
metadata Json?
projectMembers ProjectMembership[]
organization Organization @relation(fields: [orgId], references: [id], onUpdate: Cascade, onDelete: Cascade)
apiKeys ApiKey[]
dataset Dataset[]
invitations MembershipInvitation[]
sessions TraceSession[]
Prompt Prompt[]
Model Model[]
EvalTemplate EvalTemplate[]
JobConfiguration JobConfiguration[]
JobExecution JobExecution[]
LlmApiKeys LlmApiKeys[]
PosthogIntegration PosthogIntegration[]
BlobStorageIntegration BlobStorageIntegration[]
scoreConfig ScoreConfig[]
BatchExport BatchExport[]
comment Comment[]
annotationQueue AnnotationQueue[]
annotationQueueItem AnnotationQueueItem[]
TraceMedia TraceMedia[]
Media Media[]
ObservationMedia ObservationMedia[]
LegacyTrace LegacyPrismaTrace[]
LegacyObservation LegacyPrismaObservation[]
LegacyScore LegacyPrismaScore[]
PromptDependency PromptDependency[]
LlmSchema LlmSchema[]
LlmTool LlmTool[]
PromptProtectedLabels PromptProtectedLabels[]
Dashboard Dashboard[]
DashboardWidget DashboardWidget[]
TableViewPreset TableViewPreset[]
actions Action[]
triggers Trigger[]
automationExecutions AutomationExecution[]
Automation Automation[]
DefaultLlmModel DefaultLlmModel[]
Price Price[]
id String @id @default(cuid())
orgId String @map("org_id")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
deletedAt DateTime? @map("deleted_at")
name String
retentionDays Int? @map("retention_days")
metadata Json?
projectMembers ProjectMembership[]
organization Organization @relation(fields: [orgId], references: [id], onUpdate: Cascade, onDelete: Cascade)
apiKeys ApiKey[]
dataset Dataset[]
invitations MembershipInvitation[]
sessions TraceSession[]
Prompt Prompt[]
Model Model[]
EvalTemplate EvalTemplate[]
JobConfiguration JobConfiguration[]
JobExecution JobExecution[]
LlmApiKeys LlmApiKeys[]
PosthogIntegration PosthogIntegration[]
BlobStorageIntegration BlobStorageIntegration[]
scoreConfig ScoreConfig[]
BatchExport BatchExport[]
comment Comment[]
annotationQueue AnnotationQueue[]
annotationQueueItem AnnotationQueueItem[]
TraceMedia TraceMedia[]
Media Media[]
ObservationMedia ObservationMedia[]
LegacyTrace LegacyPrismaTrace[]
LegacyObservation LegacyPrismaObservation[]
LegacyScore LegacyPrismaScore[]
PromptDependency PromptDependency[]
LlmSchema LlmSchema[]
LlmTool LlmTool[]
PromptProtectedLabels PromptProtectedLabels[]
Dashboard Dashboard[]
DashboardWidget DashboardWidget[]
TableViewPreset TableViewPreset[]
actions Action[]
triggers Trigger[]
automationExecutions AutomationExecution[]
Automation Automation[]
DefaultLlmModel DefaultLlmModel[]
Price Price[]
SlackIntegration SlackIntegration?
PendingDeletion PendingDeletion[]
AnnotationQueueAssignment AnnotationQueueAssignment[]
@@index([orgId])
@@map("projects")
@@ -489,15 +495,16 @@ enum ScoreDataType {
}
model AnnotationQueue {
id String @id @default(cuid())
name String
description String?
scoreConfigIds String[] @default([]) @map("score_config_ids")
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
annotationQueueItem AnnotationQueueItem[]
id String @id @default(cuid())
name String
description String?
scoreConfigIds String[] @default([]) @map("score_config_ids")
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
annotationQueueItem AnnotationQueueItem[]
annotationQueueAssignment AnnotationQueueAssignment[]
@@unique([projectId, name])
@@index([id, projectId])
@@ -539,6 +546,22 @@ enum AnnotationQueueStatus {
enum AnnotationQueueObjectType {
TRACE
OBSERVATION
SESSION
}
model AnnotationQueueAssignment {
id String @id @default(cuid())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
userId String @map("user_id")
user User @relation(fields: [userId], references: [id], onDelete: Cascade)
queueId String @map("queue_id")
queue AnnotationQueue @relation(fields: [queueId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
@@unique([projectId, queueId, userId])
@@map("annotation_queue_assignments")
}
model CronJobs {
@@ -551,16 +574,18 @@ model CronJobs {
}
model Dataset {
id String @default(cuid())
projectId String @map("project_id")
name String
description String?
metadata Json?
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
datasetItems DatasetItem[]
datasetRuns DatasetRuns[]
id String @default(cuid())
projectId String @map("project_id")
name String
description String?
metadata Json?
remoteExperimentUrl String? @map("remote_experiment_url")
remoteExperimentPayload Json? @map("remote_experiment_payload")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
datasetItems DatasetItem[]
datasetRuns DatasetRuns[]
@@id([id, projectId])
@@unique([projectId, name])
@@ -1295,6 +1320,7 @@ model Automation {
enum ActionType {
WEBHOOK
SLACK
// More action types can be added as needed
}
@@ -1335,3 +1361,62 @@ model AutomationExecution {
@@index([projectId])
@@map("automation_executions")
}
// Slack Integration: Stores centralized Slack workspace connection for each project
// One project can connect to one Slack workspace, supporting multiple channel automations
model SlackIntegration {
id String @id @default(cuid())
projectId String @unique @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
// Installation details (encrypted using shared encryption utilities)
teamId String @map("team_id") // Slack workspace ID
teamName String @map("team_name") // Human-readable workspace name
botToken String @map("bot_token") // Encrypted bot token for API calls
botUserId String @map("bot_user_id") // Bot user ID for workspace
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
@@index([teamId])
@@map("slack_integrations")
}
// Pending Deletions: Tracks objects (like traces) that are scheduled for batch deletion
model PendingDeletion {
id String @id @default(cuid())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
object String @map("object") // e.g., "trace", "observation", etc.
objectId String @map("object_id") // The ID of the object to be deleted
isDeleted Boolean @default(false) @map("is_deleted")
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
@@index([projectId, object, isDeleted])
@@index([objectId, object])
@@map("pending_deletions")
}
model Survey {
id String @id @default(cuid())
createdAt DateTime @default(now()) @map("created_at")
surveyName SurveyName @map("survey_name")
response Json
userId String? @map("user_id")
userEmail String? @map("user_email")
orgId String? @map("org_id")
org Organization? @relation(fields: [orgId], references: [id], onDelete: Cascade)
user User? @relation(fields: [userId], references: [id], onDelete: Cascade)
@@map("surveys")
}
enum SurveyName {
ORG_ONBOARDING @map("org_onboarding")
USER_ONBOARDING @map("user_onboarding")
@@map("SurveyName")
}
@@ -23,6 +23,7 @@ import {
SEED_TEXT_PROMPTS,
} from "./utils/postgres-seed-constants";
import {
generateDatasetItemId,
generateDatasetRunTraceId,
generateEvalObservationId,
generateEvalScoreId,
@@ -517,6 +518,7 @@ export async function createDatasets(
description: data.description,
projectId,
metadata: data.metadata,
id: `${datasetName}-${projectId.slice(-8)}`,
},
}));
@@ -532,13 +534,23 @@ export async function createDatasets(
const datasetItem = await prisma.datasetItem.upsert({
where: {
id_projectId: {
id: `${dataset.id}-${index}`,
id: generateDatasetItemId(
datasetName,
index,
projectId,
SEED_DATASETS.indexOf(data) || 0,
),
projectId,
},
},
create: {
projectId,
id: `${dataset.id}-${index}`,
id: generateDatasetItemId(
datasetName,
index,
projectId,
SEED_DATASETS.indexOf(data) || 0,
),
datasetId: dataset.id,
sourceTraceId: sourceTraceId ?? null,
sourceObservationId: null,
@@ -554,14 +566,14 @@ export async function createDatasets(
for (let datasetRunNumber = 0; datasetRunNumber < 3; datasetRunNumber++) {
const datasetRun = await prisma.datasetRuns.upsert({
where: {
datasetId_projectId_name: {
datasetId: dataset.id,
id_projectId: {
id: `demo-dataset-run-${datasetRunNumber}-${projectId.slice(-8)}`,
projectId,
name: `demo-dataset-run-${datasetRunNumber}`,
},
},
create: {
projectId,
id: `demo-dataset-run-${datasetRunNumber}-${projectId.slice(-8)}`,
name: `demo-dataset-run-${datasetRunNumber}`,
description: Math.random() > 0.5 ? "Dataset run description" : "",
datasetId: dataset.id,
@@ -2,6 +2,8 @@ import {
TraceRecordInsertType,
ObservationRecordInsertType,
ScoreRecordInsertType,
DatasetRunItemRecordInsertType,
createDatasetRunItemsCh,
} from "../../../src/server";
import { SEED_TEXT_PROMPTS } from "./postgres-seed-constants";
import {
@@ -42,6 +44,16 @@ export class ClickHouseQueryBuilder {
return await createObservationsCh(observations);
}
/**
* Creates INSERT query for dataset run items data using VALUES syntax.
* Use for: Small datasets, dataset run items that link to postgres data (e.g. dataset runs)
*/
async executeDatasetRunItemsInsert(
datasetRunItems: DatasetRunItemRecordInsertType[],
): Promise<InsertResult> {
return await createDatasetRunItemsCh(datasetRunItems);
}
/**
* Creates INSERT query for score data using VALUES syntax.
* Use for: Small datasets, scores with custom values and metadata.
@@ -6,6 +6,8 @@ import {
REALISTIC_MODELS,
} from "./clickhouse-seed-constants";
import {
generateDatasetItemId,
generateDatasetRunItemId,
generateDatasetRunTraceId,
generateEvalObservationId,
generateEvalScoreId,
@@ -23,6 +25,8 @@ import {
ObservationRecordInsertType,
ScoreRecordInsertType,
TraceRecordInsertType,
DatasetRunItemRecordInsertType,
createDatasetRunItem,
} from "../../../src/server";
/**
@@ -60,6 +64,49 @@ export class DataGenerator {
return Math.floor(Math.random() * (max - min + 1)) + min;
}
/**
* Creates dataset run items for dataset runs.
* Use for: Dataset experiment scenarios.
*/
generateDatasetRunItem(
input: DatasetItemInput & { runCreatedAt: number },
projectId: string,
): DatasetRunItemRecordInsertType {
const datasetRunItemId = generateDatasetRunItemId(
input.datasetName,
input.itemIndex,
projectId,
input.runNumber || 0,
);
// TODO: there are too many dataset run items in the postgres database?
return createDatasetRunItem({
id: datasetRunItemId,
project_id: projectId,
trace_id: generateDatasetRunTraceId(
input.datasetName,
input.itemIndex,
projectId,
input.runNumber || 0,
),
dataset_id: `${input.datasetName}-${projectId.slice(-8)}`,
dataset_run_id: `demo-dataset-run-${input.runNumber}-${projectId.slice(-8)}`,
dataset_run_name: `demo-dataset-run-${input.runNumber}-${projectId.slice(-8)}`,
dataset_run_created_at: input.runCreatedAt,
dataset_run_description:
(input.runNumber || 0) % 2 === 0 ? "Dataset run description" : "",
dataset_run_metadata: { key: "value" },
dataset_item_id: generateDatasetItemId(
input.datasetName,
input.itemIndex,
projectId,
input.runNumber || 0,
),
dataset_item_input: input.item.input,
dataset_item_expected_output: input.item.output,
});
}
/**
* Creates traces from dataset items for experiment runs.
* Use for: Dataset experiments scenarios.
+143 -143
View File
@@ -11,29 +11,29 @@
## 🎯 Getting Started
### Prerequisites
\`\`\`bash
npm install @langfuse/core
pip install langfuse
\`\`\`bash
npm install @langfuse/core
pip install langfuse
\`\`\`
### Quick Setup
1. **Initialize your project**
\`\`\`typescript
import { Langfuse } from 'langfuse'
const langfuse = new Langfuse({
secretKey: process.env.LANGFUSE_SECRET_KEY,
publicKey: process.env.LANGFUSE_PUBLIC_KEY,
baseUrl: 'https://cloud.langfuse.com'
})
\`\`\`typescript
import { Langfuse } from 'langfuse'
const langfuse = new Langfuse({
secretKey: process.env.LANGFUSE_SECRET_KEY,
publicKey: process.env.LANGFUSE_PUBLIC_KEY,
baseUrl: 'https://cloud.langfuse.com'
})
\`\`\`
2. **Create your first trace**
\`\`\`python
from langfuse import Langfuse
langfuse = Langfuse()
trace = langfuse.trace(name="chat-application")
\`\`\`python
from langfuse import Langfuse
langfuse = Langfuse()
trace = langfuse.trace(name="chat-application")
\`\`\`
---
@@ -72,75 +72,75 @@ graph TD
> **Note:** Traces are the foundation of observability in LLM applications.
#### Creating Traces
\`\`\`typescript
// Basic trace creation
const trace = langfuse.trace({
name: "user-query-processing",
userId: "user-123",
sessionId: "session-456",
metadata: {
environment: "production",
version: "2.1.0"
}
})
\`\`\`typescript
// Basic trace creation
const trace = langfuse.trace({
name: "user-query-processing",
userId: "user-123",
sessionId: "session-456",
metadata: {
environment: "production",
version: "2.1.0"
}
})
// Nested observations
const span = trace.span({
name: "document-retrieval",
input: { query: "What is machine learning?" },
metadata: { vectorStore: "pinecone" }
})
// Nested observations
const span = trace.span({
name: "document-retrieval",
input: { query: "What is machine learning?" },
metadata: { vectorStore: "pinecone" }
})
const generation = span.generation({
name: "answer-generation",
model: "gpt-4",
input: retrievedDocs,
output: generatedAnswer,
usage: {
promptTokens: 1250,
completionTokens: 420,
totalTokens: 1670
}
})
const generation = span.generation({
name: "answer-generation",
model: "gpt-4",
input: retrievedDocs,
output: generatedAnswer,
usage: {
promptTokens: 1250,
completionTokens: 420,
totalTokens: 1670
}
})
\`\`\`
### Advanced Features
#### 🔄 Async Processing
\`\`\`python
import asyncio
from langfuse import Langfuse
\`\`\`python
import asyncio
from langfuse import Langfuse
async def process_batch():
langfuse = Langfuse()
tasks = []
for item in batch_items:
task = asyncio.create_task(
process_item_with_tracing(langfuse, item)
)
tasks.append(task)
results = await asyncio.gather(*tasks)
return results
async def process_batch():
langfuse = Langfuse()
tasks = []
for item in batch_items:
task = asyncio.create_task(
process_item_with_tracing(langfuse, item)
)
tasks.append(task)
results = await asyncio.gather(*tasks)
return results
\`\`\`
#### 🎯 Custom Scoring
\`\`\`typescript
// Automated scoring
trace.score({
name: "relevance",
value: 0.95,
comment: "Highly relevant response"
})
\`\`\`typescript
// Automated scoring
trace.score({
name: "relevance",
value: 0.95,
comment: "Highly relevant response"
})
// Human feedback scoring
trace.score({
name: "user-satisfaction",
value: 1,
source: "user-feedback",
comment: "User rated 5/5 stars"
})
// Human feedback scoring
trace.score({
name: "user-satisfaction",
value: 1,
source: "user-feedback",
comment: "User rated 5/5 stars"
})
\`\`\`
---
@@ -156,20 +156,20 @@ trace.score({
- **User Satisfaction**: Quality metrics
#### Dashboard Setup
\`\`\`yaml
# monitoring-config.yml
dashboards:
- name: "LLM Performance"
panels:
- type: "time-series"
title: "Response Latency"
query: "avg(response_time) by (model)"
- type: "stat"
title: "Daily Token Usage"
query: "sum(tokens_used)"
- type: "table"
title: "Top Errors"
query: "topk(10, count by (error_type))"
\`\`\`yaml
# monitoring-config.yml
dashboards:
- name: "LLM Performance"
panels:
- type: "time-series"
title: "Response Latency"
query: "avg(response_time) by (model)"
- type: "stat"
title: "Daily Token Usage"
query: "sum(tokens_used)"
- type: "table"
title: "Top Errors"
query: "topk(10, count by (error_type))"
\`\`\`
### 🔐 Security Considerations
@@ -177,40 +177,40 @@ dashboards:
> ⚠️ **Important**: Never log sensitive user data in traces
#### Data Sanitization
\`\`\`python
def sanitize_input(data):
"""Remove PII from trace data"""
sanitized = data.copy()
# Remove email addresses
sanitized = re.sub(r'\\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\\.[A-Z|a-z]{2,}\\b',
'[EMAIL_REDACTED]', sanitized)
# Remove phone numbers
sanitized = re.sub(r'\\b\\d{3}-\\d{3}-\\d{4}\\b',
'[PHONE_REDACTED]', sanitized)
return sanitized
\`\`\`python
def sanitize_input(data):
"""Remove PII from trace data"""
sanitized = data.copy()
# Remove email addresses
sanitized = re.sub(r'\\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\\.[A-Z|a-z]{2,}\\b',
'[EMAIL_REDACTED]', sanitized)
# Remove phone numbers
sanitized = re.sub(r'\\b\\d{3}-\\d{3}-\\d{4}\\b',
'[PHONE_REDACTED]', sanitized)
return sanitized
\`\`\`
### 🚀 Performance Optimization
#### Batch Processing
\`\`\`typescript
// Efficient batch uploads
const batchSize = 100
const traces = []
\`\`\`typescript
// Efficient batch uploads
const batchSize = 100
const traces = []
for (let i = 0; i < data.length; i += batchSize) {
const batch = data.slice(i, i + batchSize)
const processedBatch = await Promise.all(
batch.map(item => processWithLangfuse(item))
)
traces.push(...processedBatch)
}
for (let i = 0; i < data.length; i += batchSize) {
const batch = data.slice(i, i + batchSize)
const processedBatch = await Promise.all(
batch.map(item => processWithLangfuse(item))
)
traces.push(...processedBatch)
}
// Flush all traces at once
await langfuse.flushAsync()
// Flush all traces at once
await langfuse.flushAsync()
\`\`\`
---
@@ -219,34 +219,34 @@ await langfuse.flushAsync()
### Multi-Agent System Tracing
\`\`\`python
class MultiAgentTracer:
def __init__(self):
self.langfuse = Langfuse()
async def orchestrate_agents(self, task):
# Main orchestration trace
main_trace = self.langfuse.trace(
name="multi-agent-orchestration",
input={"task": task}
)
# Agent 1: Research
research_span = main_trace.span(name="research-agent")
research_result = await self.research_agent.process(task)
research_span.end(output=research_result)
class MultiAgentTracer:
def __init__(self):
self.langfuse = Langfuse()
async def orchestrate_agents(self, task):
# Main orchestration trace
main_trace = self.langfuse.trace(
name="multi-agent-orchestration",
input={"task": task}
)
# Agent 1: Research
research_span = main_trace.span(name="research-agent")
research_result = await self.research_agent.process(task)
research_span.end(output=research_result)
# Agent 2: Analysis
analysis_span = main_trace.span(name="analysis-agent")
analysis_result = await self.analysis_agent.process(research_result)
analysis_span.end(output=analysis_result)
# Agent 3: Synthesis
synthesis_span = main_trace.span(name="synthesis-agent")
final_result = await self.synthesis_agent.process(analysis_result)
synthesis_span.end(output=final_result)
main_trace.end(output=final_result)
return final_result
analysis_span = main_trace.span(name="analysis-agent")
analysis_result = await self.analysis_agent.process(research_result)
analysis_span.end(output=analysis_result)
# Agent 3: Synthesis
synthesis_span = main_trace.span(name="synthesis-agent")
final_result = await self.synthesis_agent.process(analysis_result)
synthesis_span.end(output=final_result)
main_trace.end(output=final_result)
return final_result
\`\`\`
---
@@ -256,11 +256,11 @@ class MultiAgentTracer:
With proper implementation of Langfuse tracing, you can:
- ✅ **Monitor** your LLM applications in real-time
- ✅ **Debug** issues with detailed trace information
- ✅ **Debug** issues with detailed trace information
- ✅ **Optimize** performance and costs
- ✅ **Scale** your applications with confidence
### Next Steps
1. Review the [official documentation](https://langfuse.com/docs)
2. Join our [Discord community](https://discord.gg/langfuse)
3. Check out [example projects](https://github.com/langfuse/langfuse)
3. Check out [example projects](https://github.com/langfuse/langfuse)
@@ -1,3 +1,21 @@
export const generateDatasetRunItemId = (
datasetName: string,
itemIndex: number,
projectId: string,
runNumber: number,
) => {
return `dataset-run-item-${datasetName}-${itemIndex}-${projectId.slice(-8)}-${runNumber}`;
};
export const generateDatasetItemId = (
datasetName: string,
itemIndex: number,
projectId: string,
runNumber: number,
) => {
return `dataset-item-${datasetName}-${itemIndex}-${projectId.slice(-8)}-${runNumber}`;
};
export const generateDatasetRunTraceId = (
datasetName: string,
itemIndex: number,
@@ -92,12 +92,26 @@ export class SeederOrchestrator {
logger.info(
`Processing run ${runNumber + 1}/${numberOfRuns} for project ${projectId}`,
);
// const now = Date.now();
const traces: TraceRecordInsertType[] = [];
const observations: ObservationRecordInsertType[] = [];
// const datasetRunItems: DatasetRunItemRecordInsertType[] = [];
for (const seedDataset of SEED_DATASETS) {
for (const [itemIndex, datasetItem] of seedDataset.items.entries()) {
// // Generate dataset run item data
// const datasetRunItem = this.dataGenerator.generateDatasetRunItem(
// {
// datasetName: seedDataset.name,
// itemIndex,
// item: datasetItem,
// runNumber,
// runCreatedAt: now,
// },
// projectId,
// );
// Generate trace data
const trace = this.dataGenerator.generateDatasetTrace(
{
@@ -123,12 +137,14 @@ export class SeederOrchestrator {
traces.push(trace);
observations.push(observation);
// datasetRunItems.push(datasetRunItem);
}
}
try {
await this.queryBuilder.executeTracesInsert(traces);
await this.queryBuilder.executeObservationsInsert(observations);
// await this.queryBuilder.executeDatasetRunItemsInsert(datasetRunItems);
} catch (error) {
logger.error(`✗ Insert failed:`, error);
throw error;
+83 -6
View File
@@ -30,14 +30,14 @@ export type AutomationDomain = {
};
export type ActionDomain = Omit<Action, "config"> & {
config: SafeWebhookActionConfig;
config: SafeActionConfig;
};
export type ActionDomainWithSecrets = Omit<Action, "config"> & {
config: WebhookActionConfigWithSecrets;
config: ActionConfigWithSecrets;
};
export const ActionTypeSchema = z.enum(["WEBHOOK"]);
export const ActionTypeSchema = z.enum(["WEBHOOK", "SLACK"]);
export const AvailableWebhookApiSchema = z.record(
z.enum(["prompt"]),
@@ -52,12 +52,13 @@ export const RequestHeaderSchema = z.object({
export const WebhookActionConfigSchema = z.object({
type: z.literal("WEBHOOK"),
url: z.url(),
headers: z.record(z.string(), z.string()),
requestHeaders: z.record(z.string(), RequestHeaderSchema),
displayHeaders: z.record(z.string(), RequestHeaderSchema),
headers: z.record(z.string(), z.string()).optional(), // deprecated field, use requestHeaders instead
requestHeaders: z.record(z.string(), RequestHeaderSchema).optional(), // might not exist on legacy webhooks
displayHeaders: z.record(z.string(), RequestHeaderSchema).optional(), // might not exist on legacy webhooks
apiVersion: AvailableWebhookApiSchema,
secretKey: z.string(),
displaySecretKey: z.string(),
lastFailingExecutionId: z.string().nullish(),
});
export const SafeWebhookActionConfigSchema = WebhookActionConfigSchema.omit({
@@ -77,18 +78,94 @@ export const WebhookActionCreateSchema = WebhookActionConfigSchema.omit({
displayHeaders: true,
});
export const SlackActionConfigSchema = z.object({
type: z.literal("SLACK"),
channelId: z.string(),
channelName: z.string(),
messageTemplate: z.string().optional(),
});
export type SlackActionConfig = z.infer<typeof SlackActionConfigSchema>;
export const ActionConfigSchema = z.discriminatedUnion("type", [
WebhookActionConfigSchema,
SlackActionConfigSchema,
]);
export const ActionCreateSchema = z.discriminatedUnion("type", [
WebhookActionCreateSchema,
SlackActionConfigSchema,
]);
export const SafeActionConfigSchema = z.discriminatedUnion("type", [
SafeWebhookActionConfigSchema,
SlackActionConfigSchema,
]);
export type ActionTypes = z.infer<typeof ActionTypeSchema>;
export type ActionConfig = z.infer<typeof ActionConfigSchema>;
export type ActionCreate = z.infer<typeof ActionCreateSchema>;
export type SafeActionConfig = z.infer<typeof SafeActionConfigSchema>;
export type WebhookActionCreate = z.infer<typeof WebhookActionCreateSchema>;
export type WebhookActionConfigWithSecrets = z.infer<
typeof WebhookActionConfigSchema
>;
export type ActionConfigWithSecrets = z.infer<typeof ActionConfigSchema>;
// Type Guards for Runtime Validation
// Using existing Zod schemas to provide both compile-time and runtime type safety
/**
* Type guard to check if a config is a valid webhook configuration with secrets
*/
export function isWebhookActionConfig(
config: unknown,
): config is WebhookActionConfigWithSecrets {
return WebhookActionConfigSchema.safeParse(config).success;
}
/**
* Type guard to check if a config is a valid Slack configuration
*/
export function isSlackActionConfig(
config: unknown,
): config is SlackActionConfig {
return SlackActionConfigSchema.safeParse(config).success;
}
/**
* Type guard to check if an entire action has valid webhook configuration
*/
export function isWebhookAction(action: {
type: string;
config: unknown;
}): action is { type: "WEBHOOK"; config: WebhookActionConfigWithSecrets } {
return action.type === "WEBHOOK" && isWebhookActionConfig(action.config);
}
/**
* Type guard for safe webhook config (without secrets)
*/
export function isSafeWebhookActionConfig(
config: unknown,
): config is SafeWebhookActionConfig {
return SafeWebhookActionConfigSchema.safeParse(config).success;
}
/**
* Converts webhook config with secrets to safe config by only including allowed fields
*/
export function convertToSafeWebhookConfig(
webhookConfig: WebhookActionConfigWithSecrets,
): SafeWebhookActionConfig {
return {
type: webhookConfig.type,
url: webhookConfig.url,
displayHeaders: webhookConfig.displayHeaders,
apiVersion: webhookConfig.apiVersion,
displaySecretKey: webhookConfig.displaySecretKey,
lastFailingExecutionId: webhookConfig.lastFailingExecutionId,
};
}
@@ -0,0 +1,28 @@
import z from "zod/v4";
import { jsonSchema } from "../utils/zod";
import { MetadataDomain } from "./traces";
export const DatasetRunItemSchema = z.object({
id: z.string(),
projectId: z.string(),
datasetRunId: z.string(),
datasetItemId: z.string(),
datasetId: z.string(),
traceId: z.string(),
observationId: z.string().nullable(),
error: z.string().nullable(),
// timestamps
createdAt: z.date(),
updatedAt: z.date(),
// dataset run fields
datasetRunName: z.string(),
datasetRunDescription: z.string().nullable(),
datasetRunMetadata: MetadataDomain,
datasetRunCreatedAt: z.date(),
// dataset item fields
datasetItemInput: jsonSchema,
datasetItemExpectedOutput: jsonSchema,
datasetItemMetadata: MetadataDomain,
});
export type DatasetRunItemDomain = z.infer<typeof DatasetRunItemSchema>;
+77 -2
View File
@@ -15,7 +15,9 @@ const EnvSchema = z.object({
.default(6379)
.nullable(),
REDIS_AUTH: z.string().nullish(),
REDIS_USERNAME: z.string().nullish(),
REDIS_CONNECTION_STRING: z.string().nullish(),
REDIS_KEY_PREFIX: z.string().nullish(),
REDIS_TLS_ENABLED: z.enum(["true", "false"]).default("false"),
REDIS_TLS_CA_PATH: z.string().optional(),
REDIS_TLS_CERT_PATH: z.string().optional(),
@@ -31,8 +33,10 @@ const EnvSchema = z.object({
"ENCRYPTION_KEY must be 256 bits, 64 string characters in hex format, generate via: openssl rand -hex 32",
)
.optional(),
LANGFUSE_CACHE_MODEL_MATCH_ENABLED: z.enum(["true", "false"]).default("true"),
LANGFUSE_CACHE_MODEL_MATCH_TTL_SECONDS: z.coerce.number().default(86400), // 24 hours
LANGFUSE_CACHE_PROMPT_ENABLED: z.enum(["true", "false"]).default("true"),
LANGFUSE_CACHE_PROMPT_TTL_SECONDS: z.coerce.number().default(300),
LANGFUSE_CACHE_PROMPT_TTL_SECONDS: z.coerce.number().default(300), // 5 minutes
CLICKHOUSE_URL: z.string().url(),
CLICKHOUSE_CLUSTER_NAME: z.string().default("default"),
CLICKHOUSE_DB: z.string().default("default"),
@@ -46,6 +50,14 @@ const EnvSchema = z.object({
.nonnegative()
.default(15_000),
LANGFUSE_INGESTION_QUEUE_SHARD_COUNT: z.coerce.number().positive().default(1),
LANGFUSE_TRACE_UPSERT_QUEUE_SHARD_COUNT: z.coerce
.number()
.positive()
.default(1),
LANGFUSE_TRACE_DELETE_DELAY_MS: z.coerce
.number()
.nonnegative()
.default(5_000),
SALT: z.string().optional(), // used by components imported by web package
LANGFUSE_LOG_LEVEL: z
.enum(["trace", "debug", "info", "warn", "error", "fatal"])
@@ -84,6 +96,9 @@ const EnvSchema = z.object({
LANGFUSE_S3_MEDIA_UPLOAD_SSE: z.enum(["AES256", "aws:kms"]).optional(),
LANGFUSE_S3_MEDIA_UPLOAD_SSE_KMS_KEY_ID: z.string().optional(),
LANGFUSE_USE_AZURE_BLOB: z.enum(["true", "false"]).default("false"),
LANGFUSE_AZURE_SKIP_CONTAINER_CHECK: z
.enum(["true", "false"])
.default("true"),
LANGFUSE_USE_GOOGLE_CLOUD_STORAGE: z.enum(["true", "false"]).default("false"),
LANGFUSE_GOOGLE_CLOUD_STORAGE_CREDENTIALS: z.string().optional(),
STRIPE_SECRET_KEY: z.string().optional(),
@@ -107,7 +122,13 @@ const EnvSchema = z.object({
LANGFUSE_CLICKHOUSE_DELETION_TIMEOUT_MS: z.coerce.number().default(240_000), // 4 minutes
LANGFUSE_CLICKHOUSE_QUERY_MAX_ATTEMPTS: z.coerce.number().default(3), // Maximum attempts for socket hang up errors
LANGFUSE_SKIP_S3_LIST_FOR_OBSERVATIONS_PROJECT_IDS: z.string().optional(),
// Dataset Run Items Migration Environment Variables
LANGFUSE_EXPERIMENT_DATASET_RUN_ITEMS_WRITE_CH: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_EXPERIMENT_DATASET_RUN_ITEMS_READ_CH: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_EXPERIMENT_COMPARE_READ_FROM_AGGREGATING_MERGE_TREES: z
.enum(["true", "false"])
.default("false"),
@@ -131,6 +152,60 @@ const EnvSchema = z.object({
LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT_SHORT_TERM: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_EXPERIMENT_INSERT_INTO_AGGREGATING_MERGE_TREES: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_INGESTION_PROCESSING_SAMPLED_PROJECTS: z
.string()
.optional()
.transform((val) => {
try {
if (!val) return new Map<string, number>();
const map = new Map<string, number>();
const parts = val.split(",");
for (const part of parts) {
const [projectId, sampleRateStr] = part.split(":");
if (!projectId || sampleRateStr === undefined) {
throw new Error(`Invalid format: ${part}`);
}
// Validate sample rate is between 0 and 1
const sampleRate = z.coerce
.number()
.min(0)
.max(1)
.parse(sampleRateStr);
map.set(projectId, sampleRate);
}
return map;
} catch (err) {
return new Map<string, number>();
}
}),
SLACK_CLIENT_ID: z.string().optional(),
SLACK_CLIENT_SECRET: z.string().optional(),
SLACK_STATE_SECRET: z.string().optional(),
HTTPS_PROXY: z.string().optional(),
LANGFUSE_SERVER_SIDE_IO_CHAR_LIMIT: z.coerce
.number()
.int()
.positive()
.default(1_000),
LANGFUSE_CLICKHOUSE_DATA_EXPORT_REQUEST_TIMEOUT_MS: z.coerce
.number()
.int()
.positive()
.default(600_000), // 10 minutes
});
export const env: z.infer<typeof EnvSchema> =
@@ -91,4 +91,11 @@ export const CreateQueueData = z.object({
}),
});
export const CreateQueueWithAssignmentsData = CreateQueueData.extend({
newAssignmentUserIds: z.array(z.string()),
});
export type CreateQueue = z.infer<typeof CreateQueueData>;
export type CreateQueueWithAssignments = z.infer<
typeof CreateQueueWithAssignmentsData
>;
@@ -14,6 +14,8 @@ const ActionIdSchema = z.enum([
"score-delete",
"trace-delete",
"trace-add-to-annotation-queue",
"session-add-to-annotation-queue",
"observation-add-to-annotation-queue",
]);
export type ActionId = z.infer<typeof ActionIdSchema>;
@@ -27,16 +27,42 @@ export const parseUnknownToString = (value: unknown): string => {
return String(value);
};
/**
* Recursively parses JSON strings that may have been encoded multiple times.
* This handles cases where data has been JSON.stringify'd multiple times.
*
* @param value - The potentially multi-encoded JSON string
* @returns The final parsed object or the original value if parsing fails
*/
function parseMultiEncodedJson(value: unknown): unknown {
if (typeof value !== "string") {
return value;
}
try {
const parsed = JSON.parse(value);
// If result is still a string, it might be double-encoded - recurse
if (typeof parsed === "string") {
return parseMultiEncodedJson(parsed);
}
return parsed;
} catch {
// If parsing fails, return original value
return value;
}
}
function parseJsonDefault(selectedColumn: unknown, jsonSelector: string) {
// selectedColumn should already be preprocessed by preprocessObjectWithJsonFields
// so we can directly use it with JSONPath
const result = JSONPath({
path: jsonSelector,
json:
typeof selectedColumn === "string"
? JSON.parse(selectedColumn)
: selectedColumn,
json: selectedColumn as any, // JSONPath accepts unknown but types are strict
});
return result.length > 0 ? result[0] : undefined;
return Array.isArray(result) && result.length > 0 ? result[0] : undefined;
}
export function extractValueFromObject(
@@ -44,7 +70,13 @@ export function extractValueFromObject(
mapping: z.infer<typeof variableMapping>,
parseJson?: (selectedColumn: unknown, jsonSelector: string) => unknown, // eslint-disable-line no-unused-vars
): { value: string; error: Error | null } {
const selectedColumn = obj[mapping.selectedColumnId];
let selectedColumn = obj[mapping.selectedColumnId];
// Simple preprocessing: attempt to parse to valid JSON object
if (typeof selectedColumn === "string") {
selectedColumn = parseMultiEncodedJson(selectedColumn);
}
const jsonParser = parseJson || parseJsonDefault;
let jsonSelectedColumn;
+1
View File
@@ -18,6 +18,7 @@ export * from "./features/entitlements/plans";
export * from "./interfaces/rate-limits";
export * from "./tableDefinitions/typeHelpers";
export * from "./domain/webhooks";
export * from "./domain/dataset-run-items";
// llm api
export * from "./server/llm/types";
@@ -18,6 +18,21 @@ export const CloudConfigSchema = z.object({
// custom rate limits for an organization
rateLimitOverrides: CloudConfigRateLimit.optional(),
// billing alert configuration
usageAlerts: z
.object({
enabled: z.boolean().default(true),
type: z.enum(["STRIPE"]).default("STRIPE"),
threshold: z.number().int().positive(),
alertId: z.string(), // Alert ID for tracking
meterId: z.string(), // Meter ID for usage tracking
notifications: z.object({
email: z.boolean().default(true),
recipients: z.array(z.string().email()).default([]),
}),
})
.optional(),
});
export type CloudConfigSchema = z.infer<typeof CloudConfigSchema>;
+15 -2
View File
@@ -1,7 +1,7 @@
import z from "zod/v4";
import { Plan, plans } from "../../features/entitlements/plans";
import { CloudConfigRateLimit } from "../../interfaces/rate-limits";
import { ApiKeyScope } from "../../";
import { ApiKeyScope, MakeOptional } from "../../";
const ApiKeyBaseSchema = z.object({
id: z.string(),
@@ -48,12 +48,25 @@ export type AuthHeaderValidVerificationResult = {
scope: ApiAccessScope;
};
export type ApiAccessScope = {
export type AuthHeaderValidVerificationResultIngestion = {
validKey: true;
scope: ApiAccessScopeIngestion;
};
type BaseApiAccessScope = {
projectId: string | null;
accessLevel: "organization" | "project" | "scores";
};
type ApiAccessScopeMetadata = {
orgId: string;
plan: Plan;
rateLimitOverrides: z.infer<typeof CloudConfigRateLimit>;
apiKeyId: string;
publicKey: string;
};
export type ApiAccessScopeIngestion = BaseApiAccessScope &
MakeOptional<ApiAccessScopeMetadata>;
export type ApiAccessScope = BaseApiAccessScope & ApiAccessScopeMetadata;
@@ -1,4 +1,4 @@
import { instrumentAsync } from "../instrumentation";
import { instrumentAsync, recordDistribution } from "../instrumentation";
import * as opentelemetry from "@opentelemetry/api";
import { env } from "../../env";
import { logger } from "../logger";
@@ -19,12 +19,17 @@ const executionWrapper = async <T, Y>(
return [res, duration];
};
/**
* Measures the execution time of two functions and returns the result based on the experiment configuration.
* This is used to compare the execution of AggregatingMergeTrees with the existing ReplacingMergeTree execution.
*/
export const measureAndReturn = async <T, Y>(args: {
operationName: string;
projectId: string;
input: T;
existingExecution: (input: T) => Promise<Y>; // eslint-disable-line no-unused-vars
newExecution: (input: T) => Promise<Y>; // eslint-disable-line no-unused-vars
minStartTime?: Date;
}): Promise<Y> => {
return instrumentAsync(
{
@@ -32,14 +37,34 @@ export const measureAndReturn = async <T, Y>(args: {
spanKind: opentelemetry.SpanKind.CLIENT,
},
async (currentSpan) => {
const { input, existingExecution, newExecution } = args;
const { input, existingExecution, newExecution, minStartTime } = args;
if (
env.LANGFUSE_EXPERIMENT_COMPARE_READ_FROM_AGGREGATING_MERGE_TREES !==
"true"
) {
currentSpan.setAttribute(`langfuse.experiment.amts.run`, "disabled");
return existingExecution(input);
// Check for short-term new result experiment
if (
env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT_SHORT_TERM === "true" &&
minStartTime
) {
const thirtyDaysAgo = new Date();
thirtyDaysAgo.setDate(thirtyDaysAgo.getDate() - 30);
if (minStartTime >= thirtyDaysAgo) {
currentSpan.setAttribute(
`langfuse.experiment.amts.short-term`,
"true",
);
return newExecution(input);
}
}
return env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT === "true"
? newExecution(input)
: existingExecution(input);
}
// If not whitelisted, apply sampling logic
@@ -68,6 +93,14 @@ export const measureAndReturn = async <T, Y>(args: {
durationDifference,
);
recordDistribution(
"langfuse.experiment.amts.duration_difference_distribution",
durationDifference,
{
operation: args.operationName,
},
);
if (
env.LANGFUSE_EXPERIMENT_ADD_QUERY_RESULT_TO_SPAN_PROJECT_IDS.some(
(p) => p === args.projectId,
@@ -83,6 +116,23 @@ export const measureAndReturn = async <T, Y>(args: {
);
}
// Check for short-term new result experiment
if (
env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT_SHORT_TERM === "true" &&
minStartTime
) {
const thirtyDaysAgo = new Date();
thirtyDaysAgo.setDate(thirtyDaysAgo.getDate() - 30);
if (minStartTime >= thirtyDaysAgo) {
currentSpan.setAttribute(
`langfuse.experiment.amts.short-term`,
"true",
);
return newResult;
}
}
return env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT === "true"
? newResult
: existingResult;
@@ -2,6 +2,7 @@ export const ClickhouseTableNames = {
traces: "traces",
observations: "observations",
scores: "scores",
dataset_run_items: "dataset_run_items",
// Virtual tables for dashboards
// TODO: Check if we can do this more elegantly
@@ -11,7 +11,8 @@ export type IngestionEntityTypes =
| "trace"
| "observation"
| "score"
| "sdk_log";
| "sdk_log"
| "dataset_run_item";
export const getClickhouseEntityType = (
eventType: string,
@@ -29,6 +30,8 @@ export const getClickhouseEntityType = (
return "observation";
case eventTypes.SCORE_CREATE:
return "score";
case eventTypes.DATASET_RUN_ITEM_CREATE:
return "dataset_run_item";
case eventTypes.SDK_LOG:
return "sdk_log";
default:
@@ -0,0 +1,34 @@
import { DatasetDeleteQueue } from "../redis/datasetDelete";
import { QueueJobs } from "../queues";
import { redis } from "../redis/redis";
import { randomUUID } from "crypto";
type DatasetDeletionType = "dataset" | "dataset-runs";
type DatasetDeletionPayload = {
deletionType: DatasetDeletionType;
projectId: string;
datasetId: string;
datasetRunIds?: string[];
};
export const addToDeleteDatasetQueue = async ({
deletionType,
projectId,
datasetId,
datasetRunIds = [],
}: DatasetDeletionPayload) => {
if (redis) {
await DatasetDeleteQueue.getInstance()?.add(QueueJobs.DatasetDelete, {
payload: {
deletionType,
projectId,
datasetId,
datasetRunIds,
},
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.DatasetDelete,
});
}
};
@@ -0,0 +1,94 @@
import { env } from "../../env";
import { logger } from "../../server/logger";
import {
DatasetRunItemsExecutionStrategy,
DatasetRunItemsOperationType,
} from "./types";
/**
* Returns the execution strategy for dataset run items based on environment variables.
*
* Two-phase migration approach:
* 1. Dual-write phase: DATASET_RUN_ITEMS_WRITE_TO_CLICKHOUSE=true (write to both databases)
* 2. Read migration phase: DATASET_RUN_ITEMS_READ_FROM_CLICKHOUSE=true (read from ClickHouse)
*/
function getDatasetRunItemsExecutionStrategy(): DatasetRunItemsExecutionStrategy {
return {
shouldWriteToClickHouse:
env.LANGFUSE_EXPERIMENT_DATASET_RUN_ITEMS_WRITE_CH === "true",
shouldReadFromClickHouse:
env.LANGFUSE_EXPERIMENT_DATASET_RUN_ITEMS_READ_CH === "true",
};
}
// Re-export the enum for backward compatibility
/**
* Executes the appropriate database operation based on the execution strategy.
*
* @param postgresExecution - Function to execute PostgreSQL operation
* @param clickhouseExecution - Function to execute ClickHouse operation
* @param operationType - Type of operation ("read" or "write")
* @returns Result from the selected execution strategy
*/
export async function executeWithDatasetRunItemsStrategy<TInput, TOutput>({
input,
operationType,
postgresExecution,
clickhouseExecution,
}: {
input: TInput;
operationType: DatasetRunItemsOperationType;
// eslint-disable-next-line no-unused-vars
postgresExecution: (input: TInput) => Promise<TOutput>;
// eslint-disable-next-line no-unused-vars
clickhouseExecution: (input: TInput) => Promise<TOutput>;
}): Promise<TOutput> {
const strategy = getDatasetRunItemsExecutionStrategy();
if (operationType === DatasetRunItemsOperationType.WRITE) {
// For write operations, implement dual-write strategy
if (strategy.shouldWriteToClickHouse) {
// Dual-write phase: write to both databases
const postgresResult = await postgresExecution(input);
try {
await clickhouseExecution(input);
logger.debug("Successfully wrote to both PostgreSQL and ClickHouse", {
operation: `dataset_run_items_${operationType}`,
});
} catch (error) {
logger.error("ClickHouse write failed during dual-write phase", {
error: error instanceof Error ? error.message : String(error),
operation: `dataset_run_items_${operationType}`,
});
// Continue with PostgreSQL result since it succeeded
}
return postgresResult;
} else {
// Write only to PostgreSQL
return await postgresExecution(input);
}
} else {
// For read operations, rely on the strategy
const shouldExecuteClickhouse = strategy.shouldReadFromClickHouse;
if (shouldExecuteClickhouse) {
try {
return await clickhouseExecution(input);
} catch (error) {
logger.error(
"ClickHouse execution failed, falling back to PostgreSQL",
{
error: error instanceof Error ? error.message : String(error),
operation: `dataset_run_items_${operationType}`,
},
);
// Fallback to PostgreSQL for reliability
return await postgresExecution(input);
}
} else {
return await postgresExecution(input);
}
}
}
@@ -0,0 +1,16 @@
/**
* Types and enums for dataset run items execution.
* This file is frontend-safe and doesn't import server-side dependencies.
*/
export enum DatasetRunItemsOperationType {
// eslint-disable-next-line no-unused-vars
READ = "read",
// eslint-disable-next-line no-unused-vars
WRITE = "write",
}
export type DatasetRunItemsExecutionStrategy = {
shouldWriteToClickHouse: boolean;
shouldReadFromClickHouse: boolean;
};
+12
View File
@@ -2,6 +2,7 @@ export * from "./services/StorageService";
export * from "./services/email/organizationInvitation/sendMembershipInvitationEmail";
export * from "./services/email/batchExportSuccess/sendBatchExportSuccessEmail";
export * from "./services/email/passwordReset/sendResetPasswordVerificationRequest";
export * from "./services/email/billingAlert/sendBillingAlertEmail";
export * from "./services/PromptService";
export * from "./services/PromptService/types";
export * from "./services/traces-ui-table-service";
@@ -13,6 +14,7 @@ export * from "./llm/fetchLLMCompletion";
export * from "./llm/utils";
export * from "./llm/types";
export * from "./llm/compileChatMessages";
export * from "./llm/testModelCall";
export * from "./utils/DatabaseReadStream";
export * from "./utils/transforms";
export * from "./clickhouse/client";
@@ -20,6 +22,7 @@ export * from "./clickhouse/schemaUtils";
export * from "./clickhouse/schema";
export * from "./repositories/definitions";
export * from "../server/ingestion/types";
export * from "../server/ingestion/modelMatch";
export * from "./ingestion/processEventBatch";
export * from "../server/ingestion/validateAndInflateScore";
export * from "./redis/redis";
@@ -31,6 +34,7 @@ export * from "./redis/webhookQueue";
export * from "./redis/traceDelete";
export * from "./redis/projectDelete";
export * from "./redis/scoreDelete";
export * from "./redis/datasetDelete";
export * from "./redis/datasetRunItemUpsert";
export * from "./redis/batchExport";
export * from "./redis/batchActionQueue";
@@ -55,6 +59,7 @@ export * from "./logger";
export * from "./headerPropagation";
export * from "./queries";
export * from "./repositories";
export * from "./utils/rendering";
export * from "./redis/evalExecutionQueue";
export * from "./services/sessions-ui-table-service";
export * from "./services/datasets-ui-table-service";
@@ -62,10 +67,17 @@ export * from "./services/DashboardService";
export * from "./services/TableViewService";
export * from "./services/DefaultEvaluationModelService";
export * from "./clickhouse/measureAndReturn";
export * from "./services/SlackService";
export * from "./data-deletion/ingestionFileDeletion";
export * from "./s3";
// dataset run items
export * from "./dataset-run-items/datasetExecution";
export * from "./dataset-run-items/types";
export * from "./dataset-run-items/addToDeleteQueue";
// test utils
export * from "./test-utils";
export * from "./utils/headerUtils";
export * from "./traceDeletionProcessor";
@@ -1,19 +1,23 @@
import { Model, Prisma } from "@langfuse/shared";
import { Model, Prisma } from "../../";
import {
instrumentAsync,
logger,
recordIncrement,
} from "@langfuse/shared/src/server";
import { env } from "../env";
import { redis } from "@langfuse/shared/src/server";
redis,
safeMultiDel,
} from "../";
import { type Cluster } from "ioredis";
import { env } from "../../env";
import { Decimal } from "decimal.js";
import { prisma } from "@langfuse/shared/src/db";
import { prisma } from "../../db";
export type ModelMatchProps = {
projectId: string;
model: string;
};
const MODEL_MATCH_CACHE_LOCKED_KEY = "LOCK:model-match-clear";
export async function findModel(p: ModelMatchProps): Promise<Model | null> {
return instrumentAsync(
{
@@ -76,12 +80,20 @@ const getModelFromRedis = async (
}
try {
if (await isModelMatchCacheLocked()) {
logger.info(
"Model match cache is locked. Skipping model lookup from Redis.",
);
return null;
}
const key = getRedisModelKey(p);
const redisModel = await redis?.get(key);
if (redisModel) {
recordIncrement("langfuse.model_match.cache_hit", 1);
if (redisModel === NOT_FOUND_TOKEN) {
return null;
return NOT_FOUND_TOKEN;
}
const model = redisModelToPrismaModel(redisModel);
return model;
@@ -178,6 +190,11 @@ export const getRedisModelKey = (p: ModelMatchProps) => {
};
const getModelMatchKeyPrefix = () => {
if (env.REDIS_CLUSTER_ENABLED === "true") {
// Use hash tags for Redis cluster compatibility
// This ensures all model cache keys are placed on the same hash slot
return "{model-match}";
}
return "model-match";
};
@@ -205,3 +222,99 @@ export const redisModelToPrismaModel = (redisModel: string): Model => {
: null,
};
};
export async function clearModelCacheForProject(
projectId: string,
): Promise<void> {
if (env.LANGFUSE_CACHE_MODEL_MATCH_ENABLED === "false" || !redis) {
return;
}
try {
const pattern = `${getModelMatchKeyPrefix()}:${projectId}:*`;
const keys =
env.REDIS_CLUSTER_ENABLED === "true"
? (
await Promise.all(
(redis as Cluster)
.nodes("master")
.map((node) => node.keys(pattern) || []),
)
).flat()
: await redis.keys(pattern);
if (keys.length > 0) {
await safeMultiDel(redis, keys);
logger.info(
`Cleared ${keys.length} model cache entries for project ${projectId}`,
);
}
} catch (error) {
logger.error(
`Error clearing model cache for project ${projectId}: ${error}`,
);
}
}
export async function isModelMatchCacheLocked() {
try {
return Boolean(await redis?.exists(MODEL_MATCH_CACHE_LOCKED_KEY));
} catch (err) {
logger.error("Failed to check whether model match is locked", err);
return false;
}
}
export async function clearFullModelCache() {
if (env.LANGFUSE_CACHE_MODEL_MATCH_ENABLED === "false" || !redis) {
return;
}
try {
// Use lock to protect for concurrent executions
// This function is called on worker startup, so we want to avoid all workers triggering this delete
if (await isModelMatchCacheLocked()) {
logger.info("Model cache clearing already in progress; skipping.");
return;
}
const startTime = Date.now();
logger.info("Clearing full model cache...");
const tenMinutesInSeconds = 60 * 10;
await redis.setex(
MODEL_MATCH_CACHE_LOCKED_KEY,
tenMinutesInSeconds,
"locked",
);
const pattern = getModelMatchKeyPrefix() + "*";
const keys =
env.REDIS_CLUSTER_ENABLED === "true"
? (
await Promise.all(
(redis as Cluster)
.nodes("master")
.map((node) => node.keys(pattern) || []),
)
).flat()
: await redis.keys(pattern);
if (keys.length > 0) {
await safeMultiDel(redis, keys);
logger.info(
`Cleared full model cache with ${keys.length} keys in ${Date.now() - startTime}ms.`,
);
} else {
logger.info(`No keys found for match pattern '${pattern}'`);
}
} catch (error) {
logger.error(`Error clearing full model cache: ${error}`);
} finally {
await redis?.del(MODEL_MATCH_CACHE_LOCKED_KEY);
}
}
@@ -8,7 +8,7 @@ import {
LangfuseNotFoundError,
UnauthorizedError,
} from "../../errors";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import { AuthHeaderValidVerificationResultIngestion } from "../auth/types";
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
import {
getCurrentSpan,
@@ -30,6 +30,7 @@ import {
StorageService,
StorageServiceFactory,
} from "../services/StorageService";
import { isTraceIdInSample } from "./sampling";
let s3StorageServiceClient: StorageService;
@@ -61,7 +62,7 @@ export type TokenCountDelegate = (p: {
* We need the delay around date boundaries to avoid duplicates for out-of-order processing of events.
* @param delay - Delay overwrite. Used if non-null.
*/
const getDelay = (delay: number | null) => {
const getDelay = (delay: number | null, source: "api" | "otel") => {
if (delay !== null) {
return delay;
}
@@ -73,6 +74,10 @@ const getDelay = (delay: number | null) => {
return env.LANGFUSE_INGESTION_QUEUE_DELAY_MS;
}
if (source === "otel") {
return 0;
}
// Use 5s here to avoid duplicate processing on the worker. If the ingestion delay is set to a lower value,
// we use this instead.
// Values should be revisited based on a cost/performance trade-off.
@@ -94,12 +99,12 @@ type ProcessEventBatchOptions = {
/**
* Processes a batch of events.
* @param input - Batch of IngestionEventType. Will validate the types first thing and return errors if they are invalid.
* @param authCheck - AuthHeaderValidVerificationResult
* @param authCheck - AuthHeaderValidVerificationResultIngestion
* @param options - (Optional) Options for the event batch processing.
*/
export const processEventBatch = async (
input: unknown[],
authCheck: AuthHeaderValidVerificationResult,
authCheck: AuthHeaderValidVerificationResultIngestion,
options: ProcessEventBatchOptions = {},
): Promise<{
successes: { id: string; status: number }[];
@@ -124,8 +129,10 @@ export const processEventBatch = async (
"langfuse.project.id",
authCheck.scope.projectId ?? "",
);
currentSpan?.setAttribute("langfuse.org.id", authCheck.scope.orgId);
currentSpan?.setAttribute("langfuse.org.plan", authCheck.scope.plan);
if (authCheck.scope.orgId)
currentSpan?.setAttribute("langfuse.org.id", authCheck.scope.orgId);
if (authCheck.scope.plan)
currentSpan?.setAttribute("langfuse.org.plan", authCheck.scope.plan);
/**************
* VALIDATION *
@@ -256,11 +263,39 @@ export const processEventBatch = async (
const shardingKey = `${authCheck.scope.projectId}-${eventData.eventBodyId}`;
const queue = IngestionQueue.getInstance({ shardingKey });
const shouldSkipS3List =
getClickhouseEntityType(eventData.type) === "observation" &&
const isDatasetRunItemEvent =
getClickhouseEntityType(eventData.type) === "dataset_run_item";
const isObservationEvent =
getClickhouseEntityType(eventData.type) === "observation";
const isOtelOrSkipS3Project =
authCheck.scope.projectId !== null &&
(projectIdsToSkipS3List.includes(authCheck.scope.projectId) ||
source === "otel");
(source === "otel" ||
projectIdsToSkipS3List.includes(authCheck.scope.projectId));
const shouldSkipS3List =
isDatasetRunItemEvent || (isObservationEvent && isOtelOrSkipS3Project);
const { isSampled, isSamplingConfigured } = isTraceIdInSample({
projectId: authCheck.scope.projectId,
event: eventData.data[0],
});
if (!isSampled) {
recordIncrement("langfuse.ingestion.sampling", eventData.data.length, {
projectId: authCheck.scope.projectId ?? "<not set>",
sampling_decision: "out",
});
return;
}
if (isSamplingConfigured) {
recordIncrement("langfuse.ingestion.sampling", eventData.data.length, {
projectId: authCheck.scope.projectId ?? "<not set>",
sampling_decision: "in",
});
}
return queue
? queue.add(
@@ -285,7 +320,7 @@ export const processEventBatch = async (
},
},
},
{ delay: getDelay(delay) },
{ delay: getDelay(delay, source) },
)
: Promise.reject("Failed to instantiate queue");
}),
@@ -300,7 +335,7 @@ export const processEventBatch = async (
const isAuthorized = (
event: IngestionEventType,
authScope: AuthHeaderValidVerificationResult,
authScope: AuthHeaderValidVerificationResultIngestion,
): boolean => {
if (event.type === eventTypes.SDK_LOG) {
return true;
@@ -0,0 +1,59 @@
import crypto from "node:crypto";
import { logger } from "../logger";
import { env } from "../../env";
import { IngestionEventType } from "./types";
export function isTraceIdInSample(params: {
projectId: string | null;
event: IngestionEventType;
}): { isSampled: boolean; isSamplingConfigured: boolean } {
const { projectId, event } = params;
const sampledProjects = env.LANGFUSE_INGESTION_PROCESSING_SAMPLED_PROJECTS;
if (!projectId || !sampledProjects.has(projectId))
return { isSampled: true, isSamplingConfigured: false };
const sampleRate = sampledProjects.get(projectId);
if (sampleRate === undefined)
return { isSampled: true, isSamplingConfigured: true };
const traceId = parseTraceId(event);
if (!traceId) return { isSampled: true, isSamplingConfigured: true };
return {
isSampled: isInSample(traceId, sampleRate),
isSamplingConfigured: true,
};
}
function isInSample(traceId: string, sampleRate: number) {
if (sampleRate < 0 || sampleRate > 1) {
logger.error(`Invalid sample rate ${sampleRate}`);
// Be conservative and keep the trace ID in sample for invalid configs
return true;
}
if (sampleRate === 0) return false;
if (sampleRate === 1) return true;
// Create SHA-256 hash of the input
const hash = crypto.createHash("sha256").update(traceId).digest("hex");
// Take first 8 characters and convert to integer
// Equivalent to 4 bytes, 32 bit integer
const hashInt = parseInt(hash.substring(0, 8), 16);
// Convert to a value between 0 and 1 by dividing by largest integer
const normalizedHash = hashInt / 0xffffffff;
// Return true if normalized hash is less than sample rate
return normalizedHash < sampleRate;
}
function parseTraceId(event: IngestionEventType): string | null | undefined {
if (event.type === "trace-create") return event.body.id;
return "traceId" in event.body ? event.body.traceId : null;
}
+44 -5
View File
@@ -224,6 +224,7 @@ export const eventTypes = {
GENERATION_CREATE: "generation-create",
GENERATION_UPDATE: "generation-update",
SDK_LOG: "sdk-log",
DATASET_RUN_ITEM_CREATE: "dataset-run-item-create",
// LEGACY, only required for backwards compatibility
OBSERVATION_CREATE: "observation-create",
OBSERVATION_UPDATE: "observation-update",
@@ -343,9 +344,15 @@ export const SdkLogEvent = z.object({
// As we allow plain values, arrays, and objects the JSON parse via bodyParser should suffice.
// Complete schema factory - single source of truth for ALL schemas
const createAllIngestionSchemas = (
environmentSchema: z.ZodDefault<z.ZodString>,
) => {
const createAllIngestionSchemas = ({
isPublic = true,
}: {
isPublic: boolean;
}) => {
const environmentSchema = isPublic
? PublicEnvironmentName
: InternalEnvironmentName;
// Base schemas with environment
const TraceBody = z.object({
id: idSchema.nullish(),
@@ -504,6 +511,22 @@ const createAllIngestionSchemas = (
]),
);
const DatasetRunItemBody = z.object({
// Core identifiers
id: idSchema.nullish(),
traceId: z.string(),
observationId: z.string().nullish(),
error: z.string().nullish(),
// Metadata (optional)
createdAt: stringDateTime.nullish(),
// Dataset identification
datasetId: z.string(),
// Run identification
runId: z.string(),
// Dataset item identification
datasetItemId: z.string(),
});
// Event schemas
const base = z.object({
id: idSchema,
@@ -546,6 +569,17 @@ const createAllIngestionSchemas = (
body: ScoreBody,
});
const baseDatasetRunItemCreateEvent = base.extend({
type: z.literal(eventTypes.DATASET_RUN_ITEM_CREATE),
body: DatasetRunItemBody,
});
const datasetRunItemCreateEvent = isPublic
? baseDatasetRunItemCreateEvent.refine(() => false, {
message: "Dataset run item creation is only allowed for internal usage",
})
: baseDatasetRunItemCreateEvent;
const sdkLogEvent = base.extend({
type: z.literal(eventTypes.SDK_LOG),
body: SdkLogEvent,
@@ -570,6 +604,7 @@ const createAllIngestionSchemas = (
generationCreateEvent,
generationUpdateEvent,
sdkLogEvent,
datasetRunItemCreateEvent,
// LEGACY, only required for backwards compatibility
legacyObservationCreateEvent,
legacyObservationUpdateEvent,
@@ -595,6 +630,7 @@ const createAllIngestionSchemas = (
generationCreateEvent,
generationUpdateEvent,
scoreEvent,
datasetRunItemCreateEvent,
sdkLogEvent,
legacyObservationCreateEvent,
legacyObservationUpdateEvent,
@@ -604,8 +640,8 @@ const createAllIngestionSchemas = (
};
// Create both public and internal schema instances
const publicSchemas = createAllIngestionSchemas(PublicEnvironmentName);
const internalSchemas = createAllIngestionSchemas(InternalEnvironmentName);
const publicSchemas = createAllIngestionSchemas({ isPublic: true });
const internalSchemas = createAllIngestionSchemas({ isPublic: false });
// Export individual schemas for backwards compatibility
export const TraceBody = publicSchemas.TraceBody;
@@ -632,6 +668,8 @@ export const generationCreateEvent = publicSchemas.generationCreateEvent;
export const generationUpdateEvent = publicSchemas.generationUpdateEvent;
export const scoreEvent = publicSchemas.scoreEvent;
export const sdkLogEvent = publicSchemas.sdkLogEvent;
export const datasetRunItemCreateEvent =
publicSchemas.datasetRunItemCreateEvent;
export const legacyObservationCreateEvent =
publicSchemas.legacyObservationCreateEvent;
export const legacyObservationUpdateEvent =
@@ -649,6 +687,7 @@ export const ingestionEvent = publicSchemas.ingestionEvent;
export type IngestionEventType = z.infer<typeof ingestionEvent>;
export type TraceEventType = z.infer<typeof traceEvent>;
export type ScoreEventType = z.infer<typeof scoreEvent>;
export type DatasetRunItemEventType = z.infer<typeof datasetRunItemCreateEvent>;
/**
* Creates an ingestion event schema with appropriate environment validation.
@@ -42,6 +42,7 @@ import {
} from "./types";
import { CallbackHandler } from "langfuse-langchain";
import type { BaseCallbackHandler } from "@langchain/core/callbacks/base";
import { HttpsProxyAgent } from "https-proxy-agent";
const isLangfuseCloud = Boolean(env.NEXT_PUBLIC_LANGFUSE_CLOUD_REGION);
@@ -217,6 +218,10 @@ export async function fetchLLMCompletion(
(m) => m.content.length > 0 || "tool_calls" in m,
);
// Common proxy configuration for all adapters
const proxyUrl = env.HTTPS_PROXY;
const proxyAgent = proxyUrl ? new HttpsProxyAgent(proxyUrl) : undefined;
let chatModel:
| ChatOpenAI
| ChatAnthropic
@@ -232,7 +237,11 @@ export async function fetchLLMCompletion(
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks: finalCallbacks,
clientOptions: { maxRetries, timeout: 1000 * 60 * 2 }, // 2 minutes timeout
clientOptions: {
maxRetries,
timeout: 1000 * 60 * 2, // 2 minutes timeout
...(proxyAgent && { httpAgent: proxyAgent }),
},
});
} else if (modelParams.adapter === LLMAdapter.OpenAI) {
chatModel = new ChatOpenAI({
@@ -247,6 +256,7 @@ export async function fetchLLMCompletion(
configuration: {
baseURL,
defaultHeaders: extraHeaders,
...(proxyAgent && { httpAgent: proxyAgent }),
},
timeout: 1000 * 60 * 2, // 2 minutes timeout
});
@@ -264,6 +274,7 @@ export async function fetchLLMCompletion(
timeout: 1000 * 60 * 2, // 2 minutes timeout
configuration: {
defaultHeaders: extraHeaders,
...(proxyAgent && { httpAgent: proxyAgent }),
},
});
} else if (modelParams.adapter === LLMAdapter.Bedrock) {
@@ -316,22 +327,6 @@ export async function fetchLLMCompletion(
maxRetries,
apiKey,
});
} else if (modelParams.adapter === LLMAdapter.Atla) {
// Atla models do not support:
// - temperature
// - max_tokens
// - top_p
chatModel = new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
callbacks: finalCallbacks,
maxRetries,
configuration: {
baseURL: baseURL,
defaultHeaders: extraHeaders,
},
timeout: 1000 * 60, // 1 minute timeout
});
} else {
// eslint-disable-next-line no-unused-vars
const _exhaustiveCheck: never = modelParams.adapter;
@@ -392,6 +387,7 @@ export async function fetchLLMCompletion(
},
configuration: {
baseURL,
...(proxyAgent && { httpAgent: proxyAgent }),
},
timeout: 1000 * 60 * 2, // 2 minutes timeout
})
@@ -0,0 +1,52 @@
import { z as zodV3 } from "zod/v3";
import {
ChatMessageRole,
ChatMessageType,
LLMApiKeySchema,
type ModelConfig,
} from "./types";
import { decrypt } from "../../encryption";
import { fetchLLMCompletion } from "./fetchLLMCompletion";
import { decryptAndParseExtraHeaders } from "./utils";
import z from "zod/v4";
export const testModelCall = async ({
provider,
model,
apiKey,
prompt,
modelConfig,
}: {
provider: string;
model: string;
apiKey: z.infer<typeof LLMApiKeySchema>;
prompt?: string;
modelConfig?: ModelConfig | null;
}): Promise<void> => {
(
await fetchLLMCompletion({
streaming: false,
apiKey: decrypt(apiKey.secretKey), // decrypt the secret key
extraHeaders: decryptAndParseExtraHeaders(apiKey.extraHeaders),
baseURL: apiKey.baseURL ?? undefined,
messages: [
{
role: ChatMessageRole.User,
content: prompt ?? "mock content",
type: ChatMessageType.User,
},
],
modelParams: {
provider: provider,
model: model,
adapter: apiKey.adapter,
...modelConfig,
},
structuredOutputSchema: zodV3.object({
score: zodV3.string(),
reasoning: zodV3.string(),
}),
config: apiKey.config,
})
).completion;
};
+17 -9
View File
@@ -113,6 +113,7 @@ export enum ChatMessageRole {
User = "user",
Assistant = "assistant",
Tool = "tool",
Model = "model", // Google Gemini assistant format
}
// Thought: should placeholder not semantically be part of this, because it can be
@@ -124,6 +125,7 @@ export enum ChatMessageType {
AssistantText = "assistant-text",
AssistantToolCall = "assistant-tool-call",
ToolResult = "tool-result",
ModelText = "model-text",
PublicAPICreated = "public-api-created",
Placeholder = "placeholder",
}
@@ -156,6 +158,13 @@ export const AssistantTextMessageSchema = z.object({
});
export type AssistantTextMessage = z.infer<typeof AssistantTextMessageSchema>;
export const ModelMessageSchema = z.object({
type: z.literal(ChatMessageType.ModelText),
role: z.literal(ChatMessageRole.Model),
content: z.string(),
});
export type ModelMessage = z.infer<typeof ModelMessageSchema>;
export const AssistantToolCallMessageSchema = z.object({
type: z.literal(ChatMessageType.AssistantToolCall),
role: z.literal(ChatMessageRole.Assistant),
@@ -193,6 +202,7 @@ export const ChatMessageSchema = z.union([
AssistantTextMessageSchema,
AssistantToolCallMessageSchema,
ToolResultMessageSchema,
ModelMessageSchema,
z
.object({
role: z.union([ChatMessageDefaultRoleSchema, z.string()]), // Users may ingest any string as role via API/SDK
@@ -226,18 +236,12 @@ export type PromptVariable = { name: string; value: string; isUsed: boolean };
export enum LLMAdapter {
Anthropic = "anthropic",
OpenAI = "openai",
Atla = "atla",
Azure = "azure",
Bedrock = "bedrock",
VertexAI = "google-vertex-ai",
GoogleAIStudio = "google-ai-studio",
}
export const SYSTEM_ROLES: string[] = [
ChatMessageRole.System,
ChatMessageRole.Developer,
];
export const TextPromptContentSchema = z.string().min(1, "Enter a prompt");
export const PromptContentSchema = z.union([
@@ -290,6 +294,12 @@ export const openAIModels = [
"gpt-4.1-mini-2025-04-14",
"gpt-4.1-nano",
"gpt-4.1-nano-2025-04-14",
"gpt-5",
"gpt-5-2025-08-07",
"gpt-5-mini",
"gpt-5-mini-2025-08-07",
"gpt-5-nano",
"gpt-5-nano-2025-08-07",
"o3",
"o3-2025-04-16",
"o4-mini",
@@ -327,6 +337,7 @@ export type OpenAIModel = (typeof openAIModels)[number];
// WARNING: The first entry in the array is chosen as the default model to add LLM API keys
export const anthropicModels = [
"claude-sonnet-4-20250514",
"claude-opus-4-1-20250805",
"claude-opus-4-20250514",
"claude-3-7-sonnet-20250219",
"claude-3-5-sonnet-20241022",
@@ -373,8 +384,6 @@ export const googleAIStudioModels = [
"gemini-1.5-flash-8b",
] as const;
export const atlaModels = ["atla-selene", "atla-selene-20250214"] as const;
export type AnthropicModel = (typeof anthropicModels)[number];
export type VertexAIModel = (typeof vertexAIModels)[number];
export const supportedModels = {
@@ -384,7 +393,6 @@ export const supportedModels = {
[LLMAdapter.GoogleAIStudio]: googleAIStudioModels,
[LLMAdapter.Azure]: [],
[LLMAdapter.Bedrock]: [],
[LLMAdapter.Atla]: atlaModels,
} as const;
export type LLMFunctionCall = {
@@ -6,6 +6,7 @@ export const clickhouseSearchCondition = (
query?: string,
searchType?: TracingSearchType[],
tablePrefix?: string,
useTracesAmtCompatMode: boolean = false,
) => {
const prefix = tablePrefix ? `${tablePrefix}.` : "";
@@ -13,9 +14,11 @@ export const clickhouseSearchCondition = (
!searchType || searchType.includes("id")
? `${prefix}id ILIKE {searchString: String} OR user_id ILIKE {searchString: String} OR ${prefix}name ILIKE {searchString: String}`
: null,
searchType && searchType.includes("content")
searchType && searchType.includes("content") && !useTracesAmtCompatMode
? `${prefix}input ILIKE {searchString: String} OR ${prefix}output ILIKE {searchString: String}`
: null,
: searchType && searchType.includes("content") && useTracesAmtCompatMode
? `finalizeAggregation(${prefix}input) ILIKE {searchString: String} OR finalizeAggregation(${prefix}output) ILIKE {searchString: String}`
: null,
].filter(Boolean);
return {
@@ -1,88 +0,0 @@
import { z } from "zod/v4";
import { Prisma } from "@prisma/client";
import { tableColumnsToSqlFilterAndPrefix } from "../filterToPrisma";
import { singleFilter } from "../../interfaces/filters";
import { orderBy } from "../../interfaces/orderBy";
import { orderByToPrismaSql } from "../orderByToPrisma";
import { sessionsViewCols } from "../../tableDefinitions";
const GetSessionTableSQLParamsSchema = z.object({
projectId: z.string(),
filter: z.array(singleFilter).nullable(),
orderBy: orderBy,
page: z.number(),
limit: z.number(),
});
type GetSessionTableSQLParams = z.infer<typeof GetSessionTableSQLParamsSchema>;
export const createSessionsAllQuery = (
select: Prisma.Sql,
params: GetSessionTableSQLParams,
options?: {
ignoreOrderBy?: boolean; // used by session.metrics and session.all.totalCount
sessionIdList?: string[]; // used by session.metrics
},
): Prisma.Sql => {
const { projectId, filter, orderBy, page, limit } =
GetSessionTableSQLParamsSchema.parse(params);
const filterCondition = tableColumnsToSqlFilterAndPrefix(
filter ?? [],
sessionsViewCols,
"sessions",
);
const orderByCondition = orderByToPrismaSql(orderBy, sessionsViewCols);
const sessionIdFilter = options?.sessionIdList
? Prisma.sql`AND s.id IN (${Prisma.join(options?.sessionIdList)})`
: Prisma.sql``;
const sql = Prisma.sql`
SELECT
${select}
FROM
trace_sessions AS s
LEFT JOIN LATERAL (
SELECT
t.session_id,
MAX(t. "timestamp") AS "max_timestamp",
MIN(t. "timestamp") AS "min_timestamp",
array_agg(t.id) AS "traceIds",
array_agg(DISTINCT t.user_id) AS "userIds",
count(t.id)::int AS "countTraces",
array_agg(DISTINCT u.tag) AS "tags"
FROM
traces t
LEFT JOIN LATERAL (
SELECT DISTINCT UNNEST(t.tags) AS tag) AS u ON TRUE
WHERE
t.project_id = ${projectId}
AND t.session_id = s.id
GROUP BY
t.session_id) AS t ON TRUE
LEFT JOIN LATERAL (
SELECT
EXTRACT(EPOCH FROM COALESCE(MAX(o. "end_time"), MAX(o. "start_time"), t. "max_timestamp")) - EXTRACT(EPOCH FROM COALESCE(MIN(o. "start_time"), t. "min_timestamp"))::double precision AS "sessionDuration",
SUM(COALESCE(o. "calculated_input_cost", 0)) AS "inputCost",
SUM(COALESCE(o. "calculated_output_cost", 0)) AS "outputCost",
SUM(COALESCE(o. "calculated_total_cost", 0)) AS "totalCost",
SUM(o.prompt_tokens) AS "promptTokens",
SUM(o.completion_tokens) AS "completionTokens",
SUM(o.total_tokens) AS "totalTokens"
FROM
observations_view o
WHERE
o.project_id = ${projectId}
AND o.trace_id = ANY (t. "traceIds")) AS o ON TRUE
WHERE
s. "project_id" = ${projectId}
${filterCondition}
${sessionIdFilter}
${options?.ignoreOrderBy ? Prisma.sql`` : orderByCondition}
LIMIT ${limit}
OFFSET ${page * limit}
`;
return sql;
};
@@ -1,4 +1,3 @@
export { createSessionsAllQuery } from "./createSessionsAllQuery";
export {
type FullObservations,
type FullObservationsWithScores,
+52 -2
View File
@@ -40,6 +40,21 @@ export const ScoresQueueEventSchema = z.object({
projectId: z.string(),
scoreIds: z.array(z.string()),
});
export const DatasetQueueEventSchema = z.discriminatedUnion("deletionType", [
// Delete all run items for a specific dataset
z.object({
deletionType: z.literal("dataset"),
projectId: z.string(),
datasetId: z.string(),
}),
// Delete all run items for multiple dataset runs (also used for single run deletion)
z.object({
deletionType: z.literal("dataset-runs"),
projectId: z.string(),
datasetId: z.string(),
datasetRunIds: z.array(z.string()),
}),
]);
export const ProjectQueueEventSchema = z.object({
projectId: z.string(),
orgId: z.string(),
@@ -101,6 +116,24 @@ export const BatchActionProcessingEventSchema = z.discriminatedUnion(
targetId: z.string().optional(),
type: z.enum(BatchActionType),
}),
z.object({
actionId: z.literal("session-add-to-annotation-queue"),
projectId: z.string(),
query: BatchActionQuerySchema,
tableName: z.enum(BatchTableNames),
cutoffCreatedAt: z.date(),
targetId: z.string().optional(),
type: z.enum(BatchActionType),
}),
z.object({
actionId: z.literal("observation-add-to-annotation-queue"),
projectId: z.string(),
query: BatchActionQuerySchema,
tableName: z.enum(BatchTableNames),
cutoffCreatedAt: z.date(),
targetId: z.string().optional(),
type: z.enum(BatchActionType),
}),
z.object({
actionId: z.literal("eval-create"),
targetObject: z.enum(["trace", "dataset"]),
@@ -144,6 +177,7 @@ export const WebhookInputSchema = z.object({
payload: WebhookOutboundEnvelopeSchema,
});
export type WebhookInput = z.infer<typeof WebhookInputSchema>;
export const EntityChangeEventSchema = z.discriminatedUnion("entityType", [
z.object({
entityType: z.literal("prompt-version"),
@@ -154,8 +188,6 @@ export const EntityChangeEventSchema = z.discriminatedUnion("entityType", [
}),
// Add other entity types here in the future
]);
export type WebhookInput = z.infer<typeof WebhookInputSchema>;
export type EntityChangeEventType = z.infer<typeof EntityChangeEventSchema>;
export type CreateEvalQueueEventType = z.infer<
@@ -165,6 +197,7 @@ export type BatchExportJobType = z.infer<typeof BatchExportJobSchema>;
export type TraceQueueEventType = z.infer<typeof TraceQueueEventSchema>;
export type TracesQueueEventType = z.infer<typeof TracesQueueEventSchema>;
export type ScoresQueueEventType = z.infer<typeof ScoresQueueEventSchema>;
export type DatasetQueueEventType = z.infer<typeof DatasetQueueEventSchema>;
export type ProjectQueueEventType = z.infer<typeof ProjectQueueEventSchema>;
export type DatasetRunItemUpsertEventType = z.infer<
typeof DatasetRunItemUpsertEventSchema
@@ -192,6 +225,13 @@ export type DeadLetterRetryQueueEventType = z.infer<
export type WebhookQueueEventType = z.infer<typeof WebhookInputSchema>;
export const RetryBaggage = z.object({
originalJobTimestamp: z.date(),
attempt: z.number(),
});
export type RetryBaggage = z.infer<typeof RetryBaggage>;
export enum QueueName {
TraceUpsert = "trace-upsert", // Ingestion pipeline adds events on each Trace upsert
TraceDelete = "trace-delete",
@@ -214,6 +254,7 @@ export enum QueueName {
BatchActionQueue = "batch-action-queue",
CreateEvalQueue = "create-eval-queue",
ScoreDelete = "score-delete",
DatasetDelete = "dataset-delete-queue",
DeadLetterRetryQueue = "dead-letter-retry-queue",
WebhookQueue = "webhook-queue",
EntityChangeQueue = "entity-change-queue",
@@ -241,6 +282,7 @@ export enum QueueJobs {
BatchActionProcessingJob = "batch-action-processing-job",
CreateEvalJob = "create-eval-job",
ScoreDelete = "score-delete",
DatasetDelete = "dataset-delete-job",
DeadLetterRetryJob = "dead-letter-retry-job",
WebhookJob = "webhook-job",
EntityChangeJob = "entity-change-job",
@@ -265,6 +307,12 @@ export type TQueueJobTypes = {
payload: ScoresQueueEventType;
name: QueueJobs.ScoreDelete;
};
[QueueName.DatasetDelete]: {
timestamp: Date;
id: string;
payload: DatasetQueueEventType;
name: QueueJobs.DatasetDelete;
};
[QueueName.ProjectDelete]: {
timestamp: Date;
id: string;
@@ -282,6 +330,7 @@ export type TQueueJobTypes = {
id: string;
payload: EvalExecutionEventType;
name: QueueJobs.EvaluationExecution;
retryBaggage?: RetryBaggage;
};
[QueueName.BatchExport]: {
timestamp: Date;
@@ -306,6 +355,7 @@ export type TQueueJobTypes = {
id: string;
payload: ExperimentCreateEventType;
name: QueueJobs.ExperimentCreateJob;
retryBaggage?: RetryBaggage;
};
[QueueName.PostHogIntegrationProcessingQueue]: {
timestamp: Date;
@@ -0,0 +1,50 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import {
createNewRedisInstance,
redisQueueRetryOptions,
getQueuePrefix,
} from "./redis";
import { logger } from "../logger";
export class DatasetDeleteQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.DatasetDelete]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.DatasetDelete]
> | null {
if (DatasetDeleteQueue.instance) return DatasetDeleteQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
DatasetDeleteQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.DatasetDelete]>(
QueueName.DatasetDelete,
{
connection: newRedis,
prefix: getQueuePrefix(QueueName.DatasetDelete),
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
attempts: 2,
backoff: {
type: "exponential",
delay: 30_000,
},
},
},
)
: null;
DatasetDeleteQueue.instance?.on("error", (err) => {
logger.error("DatasetDeleteQueue error", err);
});
return DatasetDeleteQueue.instance;
}
}
@@ -34,7 +34,7 @@ export class ExperimentCreateQueue {
attempts: 10,
backoff: {
type: "exponential",
delay: 1000,
delay: 10_000, // 10 seconds
},
},
},
+9 -6
View File
@@ -6,7 +6,6 @@ import { DatasetRunItemUpsertQueue } from "./datasetRunItemUpsert";
import { EvalExecutionQueue } from "./evalExecutionQueue";
import { ExperimentCreateQueue } from "./experimentCreateQueue";
import { SecondaryIngestionQueue } from "./ingestionQueue";
import { TraceUpsertQueue } from "./traceUpsert";
import { TraceDeleteQueue } from "./traceDelete";
import { ProjectDeleteQueue } from "./projectDelete";
import { PostHogIntegrationQueue } from "./postHogIntegrationQueue";
@@ -23,11 +22,15 @@ import { ScoreDeleteQueue } from "./scoreDelete";
import { DeadLetterRetryQueue } from "./dlqRetryQueue";
import { WebhookQueue } from "./webhookQueue";
import { EntityChangeQueue } from "./entityChangeQueue";
import { DatasetDeleteQueue } from "./datasetDelete";
// IngestionQueue is sharded and requires a sharding key
// Use IngestionQueue.getInstance({ shardName: queueName }) directly instead
// IngestionQueue and TraceUpsert are sharded and require a sharding key
// Use IngestionQueue.getInstance({ shardName: queueName }) or TraceUpsertQueue.getInstance({ shardName: queueName }) directly instead
export function getQueue(
queueName: Exclude<QueueName, QueueName.IngestionQueue>,
queueName: Exclude<
QueueName,
QueueName.IngestionQueue | QueueName.TraceUpsert
>,
): Queue | null {
switch (queueName) {
case QueueName.BatchExport:
@@ -36,12 +39,12 @@ export function getQueue(
return CloudUsageMeteringQueue.getInstance();
case QueueName.DatasetRunItemUpsert:
return DatasetRunItemUpsertQueue.getInstance();
case QueueName.DatasetDelete:
return DatasetDeleteQueue.getInstance();
case QueueName.EvaluationExecution:
return EvalExecutionQueue.getInstance();
case QueueName.ExperimentCreate:
return ExperimentCreateQueue.getInstance();
case QueueName.TraceUpsert:
return TraceUpsertQueue.getInstance();
case QueueName.TraceDelete:
return TraceDeleteQueue.getInstance();
case QueueName.ProjectDelete:
@@ -6,6 +6,7 @@ import { logger } from "../logger";
const defaultRedisOptions: Partial<RedisOptions> = {
maxRetriesPerRequest: null,
enableAutoPipelining: env.REDIS_ENABLE_AUTO_PIPELINING === "true",
keyPrefix: env.REDIS_KEY_PREFIX ?? undefined,
};
export const redisQueueRetryOptions: Partial<RedisOptions> = {
@@ -76,6 +77,7 @@ const createRedisClusterInstance = (
callback(null, address);
},
redisOptions: {
username: env.REDIS_USERNAME || undefined,
password: env.REDIS_AUTH || undefined,
...defaultRedisOptions,
...additionalOptions,
@@ -128,6 +130,7 @@ export const createNewRedisInstance = (
? new Redis({
host: String(env.REDIS_HOST),
port: Number(env.REDIS_PORT),
username: env.REDIS_USERNAME || undefined,
password: String(env.REDIS_AUTH),
...defaultRedisOptions,
...additionalOptions,
+72 -25
View File
@@ -6,45 +6,92 @@ import {
getQueuePrefix,
} from "./redis";
import { logger } from "../logger";
import { getShardIndex } from "./sharding";
import { env } from "../../env";
export class TraceUpsertQueue {
private static instance: Queue<TQueueJobTypes[QueueName.TraceUpsert]> | null =
null;
private static instances: Map<
number,
Queue<TQueueJobTypes[QueueName.TraceUpsert]> | null
> = new Map();
public static getInstance(): Queue<
TQueueJobTypes[QueueName.TraceUpsert]
> | null {
if (TraceUpsertQueue.instance) return TraceUpsertQueue.instance;
public static getShardNames() {
return Array.from(
{ length: env.LANGFUSE_TRACE_UPSERT_QUEUE_SHARD_COUNT },
(_, i) => `${QueueName.TraceUpsert}${i > 0 ? `-${i}` : ""}`,
);
}
static getShardIndexFromShardName(
shardName: string | undefined,
): number | null {
if (!shardName) return null;
// Extract shard index from shard name
const shardIndex =
shardName === QueueName.TraceUpsert
? 0
: parseInt(shardName.replace(`${QueueName.TraceUpsert}-`, ""), 10);
if (isNaN(shardIndex)) return null;
return shardIndex;
}
/**
* Get the trace upsert queue instance for the given sharding key or shard name.
* @param shardingKey - ShardingKey is being hashed and randomly allocated to a shard. Should be `projectId-traceId`.
* @param shardName - Name of the shard. Should be `trace-upsert-queue-${shardIndex}` or plainly `trace-upsert-queue` for the first shard.
*/
public static getInstance({
shardingKey,
shardName,
}: {
shardingKey?: string;
shardName?: string;
} = {}): Queue<TQueueJobTypes[QueueName.TraceUpsert]> | null {
const shardIndex =
TraceUpsertQueue.getShardIndexFromShardName(shardName) ??
(env.REDIS_CLUSTER_ENABLED === "true" && shardingKey
? getShardIndex(
shardingKey,
env.LANGFUSE_TRACE_UPSERT_QUEUE_SHARD_COUNT,
)
: 0);
// Check if we already have an instance for this shard
if (TraceUpsertQueue.instances.has(shardIndex)) {
return TraceUpsertQueue.instances.get(shardIndex) || null;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
TraceUpsertQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.TraceUpsert]>(
QueueName.TraceUpsert,
{
connection: newRedis,
prefix: getQueuePrefix(QueueName.TraceUpsert),
defaultJobOptions: {
removeOnComplete: 100, // Important: If not true, new jobs for that ID would be ignored as jobs in the complete set are still considered as part of the queue
removeOnFail: 100_000,
attempts: 5,
delay: 15_000, // 15 seconds
backoff: {
type: "exponential",
delay: 5000,
},
const name = `${QueueName.TraceUpsert}${shardIndex > 0 ? `-${shardIndex}` : ""}`;
const queueInstance = newRedis
? new Queue<TQueueJobTypes[QueueName.TraceUpsert]>(name, {
connection: newRedis,
prefix: getQueuePrefix(name),
defaultJobOptions: {
removeOnComplete: 100,
removeOnFail: 100_000,
attempts: 5,
delay: 15_000, // 15 seconds
backoff: {
type: "exponential",
delay: 5000,
},
},
)
})
: null;
TraceUpsertQueue.instance?.on("error", (err) => {
logger.error("TraceUpsertQueue error", err);
queueInstance?.on("error", (err) => {
logger.error(`TraceUpsertQueue shard ${shardIndex} error`, err);
});
return TraceUpsertQueue.instance;
TraceUpsertQueue.instances.set(shardIndex, queueInstance);
return queueInstance;
}
}
@@ -1,6 +1,10 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import {
createNewRedisInstance,
getQueuePrefix,
redisQueueRetryOptions,
} from "./redis";
import { logger } from "../logger";
export class WebhookQueue {
@@ -23,6 +27,7 @@ export class WebhookQueue {
QueueName.WebhookQueue,
{
connection: newRedis,
prefix: getQueuePrefix(QueueName.WebhookQueue),
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100_000,
@@ -2,17 +2,22 @@ import {
Action,
ActionExecutionStatus,
JobConfigState,
Prisma,
prisma,
Trigger,
} from "../../db";
import {
TriggerEventSource,
WebhookActionConfigWithSecrets,
TriggerDomain,
TriggerEventAction,
ActionDomain,
AutomationDomain,
SafeWebhookActionConfig,
ActionDomainWithSecrets,
SafeActionConfig,
isWebhookActionConfig,
WebhookActionConfigWithSecrets,
isSafeWebhookActionConfig,
convertToSafeWebhookConfig,
} from "../../domain/automations";
import { FilterState } from "../../types";
import { decryptSecretHeaders, mergeHeaders } from "../utils/headerUtils";
@@ -23,7 +28,7 @@ export const getActionByIdWithSecrets = async ({
}: {
projectId: string;
actionId: string;
}) => {
}): Promise<ActionDomainWithSecrets | null> => {
const actionConfig = await prisma.action.findFirst({
where: {
id: actionId,
@@ -35,31 +40,41 @@ export const getActionByIdWithSecrets = async ({
return null;
}
const config = actionConfig.config as WebhookActionConfigWithSecrets;
if (isWebhookActionConfig(actionConfig.config)) {
const config = actionConfig.config; // Type guard ensures this is WebhookActionConfigWithSecrets
// Decrypt secret headers for webhook execution using new structure
const decryptedHeaders = config.requestHeaders
? decryptSecretHeaders(mergeHeaders(config.headers, config.requestHeaders))
: Object.entries(config.headers).reduce(
(acc, [key, value]) => {
acc[key] = { secret: false, value };
return acc;
},
{} as Record<string, { secret: boolean; value: string }>,
);
// Decrypt secret headers for webhook execution using new structure
const decryptedHeaders = config.requestHeaders
? decryptSecretHeaders(
mergeHeaders(config.headers, config.requestHeaders),
)
: config.headers
? Object.entries(config.headers).reduce(
(acc, [key, value]) => {
acc[key] = { secret: false, value };
return acc;
},
{} as Record<string, { secret: boolean; value: string }>,
)
: {};
return {
...actionConfig,
config: {
type: config.type,
url: config.url,
requestHeaders: decryptedHeaders,
displayHeaders: config.displayHeaders,
apiVersion: config.apiVersion,
displaySecretKey: config.displaySecretKey,
secretKey: config.secretKey,
},
};
return {
...actionConfig,
config: {
type: config.type,
url: config.url,
requestHeaders: decryptedHeaders,
displayHeaders: getDisplayHeaders(config),
apiVersion: config.apiVersion,
displaySecretKey: config.displaySecretKey,
secretKey: config.secretKey,
lastFailingExecutionId: config.lastFailingExecutionId,
},
};
}
// For SLACK and others, return as stored (already safe)
return actionConfig as ActionDomainWithSecrets;
};
export const getActionById = async ({
@@ -128,10 +143,7 @@ const convertTriggerToDomain = (trigger: Trigger): TriggerDomain => {
};
};
const convertActionToDomain = (action: Action): ActionDomain => {
const config = action.config as WebhookActionConfigWithSecrets;
// Handle legacy headers - convert them to displayHeaders format if displayHeaders is undefined
const getDisplayHeaders = (config: WebhookActionConfigWithSecrets) => {
let displayHeaders = config.displayHeaders;
if (!displayHeaders && config.headers) {
// Convert legacy headers to displayHeaders format
@@ -143,17 +155,25 @@ const convertActionToDomain = (action: Action): ActionDomain => {
{} as Record<string, { secret: boolean; value: string }>,
);
}
return displayHeaders;
};
const convertActionToDomain = (action: Action): ActionDomain => {
if (isWebhookActionConfig(action.config)) {
const config = action.config;
config.displayHeaders = getDisplayHeaders(config);
return {
...action,
config: convertToSafeWebhookConfig(config),
};
}
// For SLACK (or future types) return config as-is
return {
...action,
config: {
type: config.type,
url: config.url,
displayHeaders,
apiVersion: config.apiVersion,
displaySecretKey: config.displaySecretKey,
} as SafeWebhookActionConfig,
};
config: action.config as SafeActionConfig,
} as ActionDomain;
};
export const getAutomationById = async ({
@@ -225,28 +245,49 @@ export const getConsecutiveAutomationFailures = async ({
automationId: string;
projectId: string;
}): Promise<number> => {
// First get the automation to extract triggerId and actionId
const automation = await prisma.automation.findFirst({
where: {
id: automationId,
projectId,
},
const automation = await getAutomationById({
automationId,
projectId,
});
if (!automation) {
return 0;
}
const { triggerId, actionId } = automation;
const executions = await prisma.automationExecution.findMany({
where: {
triggerId,
actionId,
projectId,
status: {
in: [ActionExecutionStatus.ERROR, ActionExecutionStatus.COMPLETED],
},
// Build where clause - if lastFailingExecutionId is set, only consider executions newer than it
const whereClause: Prisma.AutomationExecutionWhereInput = {
triggerId: automation.trigger.id,
actionId: automation.action.id,
projectId,
status: {
in: [ActionExecutionStatus.ERROR, ActionExecutionStatus.COMPLETED],
},
};
// If there's a lastFailingExecutionId, we need to get executions that are newer than that execution
if (
isSafeWebhookActionConfig(automation.action.config) &&
automation.action.config.lastFailingExecutionId
) {
// First get the timestamp of the last failing execution
const lastFailingExecution = await prisma.automationExecution.findUnique({
where: {
id: automation.action.config.lastFailingExecutionId,
},
select: {
createdAt: true,
},
});
if (lastFailingExecution) {
whereClause.createdAt = {
gt: lastFailingExecution.createdAt,
};
}
}
const executions = await prisma.automationExecution.findMany({
where: whereClause,
orderBy: {
createdAt: "desc",
},
@@ -36,7 +36,7 @@ const getS3StorageServiceClient = (bucketName: string): StorageService => {
export async function upsertClickhouse<
T extends Record<string, unknown>,
>(opts: {
table: "scores" | "traces" | "observations";
table: "scores" | "traces" | "observations" | "traces_null";
records: T[];
eventBodyMapper: (body: T) => Record<string, unknown>; // eslint-disable-line no-unused-vars
tags?: Record<string, string>;
@@ -0,0 +1,86 @@
import { DatasetRunItemDomain } from "../../domain/dataset-run-items";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { parseMetadataCHRecordToDomain } from "../utils/metadata_conversion";
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { DatasetRunItemRecordReadType } from "./definitions";
export const convertToDatasetRunMetrics = (row: any) => {
return {
id: row.dataset_run_id,
projectId: row.project_id,
createdAt: new Date(row.created_at),
updatedAt: new Date(row.updated_at),
name: row.datasetRunName,
description: row.datasetRunDescription ?? "",
metadata: row.datasetRunMetadata,
countRunItems: row.count_run_items,
avgTotalCost: undefined,
avgLatency: undefined,
scores: undefined,
datasetId: row.dataset_id,
};
};
export const convertDatasetRunItemDomainToClickhouse = (
datasetRunItem: DatasetRunItemDomain,
): DatasetRunItemRecordReadType => {
return {
id: datasetRunItem.id,
project_id: datasetRunItem.projectId,
trace_id: datasetRunItem.traceId,
observation_id: datasetRunItem.observationId,
dataset_id: datasetRunItem.datasetId,
dataset_run_id: datasetRunItem.datasetRunId,
dataset_run_name: datasetRunItem.datasetRunName,
dataset_run_description: datasetRunItem.datasetRunDescription,
dataset_run_metadata: datasetRunItem.datasetRunMetadata as Record<
string,
string
>,
dataset_item_id: datasetRunItem.datasetItemId,
dataset_item_input: datasetRunItem.datasetItemInput as string,
dataset_item_expected_output:
datasetRunItem.datasetItemExpectedOutput as string,
dataset_item_metadata: datasetRunItem.datasetItemMetadata as Record<
string,
string
>,
created_at: convertDateToClickhouseDateTime(datasetRunItem.createdAt),
updated_at: convertDateToClickhouseDateTime(datasetRunItem.updatedAt),
event_ts: convertDateToClickhouseDateTime(new Date()),
is_deleted: 0,
dataset_run_created_at: convertDateToClickhouseDateTime(
datasetRunItem.datasetRunCreatedAt,
),
error: datasetRunItem.error,
};
};
export const convertDatasetRunItemClickhouseToDomain = (
row: DatasetRunItemRecordReadType,
): DatasetRunItemDomain => {
return {
id: row.id,
projectId: row.project_id,
traceId: row.trace_id,
observationId: row.observation_id ?? null,
datasetRunId: row.dataset_run_id,
datasetRunName: row.dataset_run_name,
datasetRunDescription: row.dataset_run_description ?? null,
datasetRunCreatedAt: parseClickhouseUTCDateTimeFormat(
row.dataset_run_created_at,
),
datasetRunMetadata:
parseMetadataCHRecordToDomain(row.dataset_run_metadata) ?? null,
datasetItemId: row.dataset_item_id,
datasetItemInput: row.dataset_item_input,
datasetItemExpectedOutput: row.dataset_item_expected_output,
datasetItemMetadata: parseMetadataCHRecordToDomain(
row.dataset_item_metadata,
),
createdAt: parseClickhouseUTCDateTimeFormat(row.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(row.updated_at),
datasetId: row.dataset_id,
error: row.error ?? null,
};
};
@@ -0,0 +1,458 @@
import { DatasetRunItemDomain } from "../../domain/dataset-run-items";
import { type OrderByState } from "../../interfaces/orderBy";
import { datasetRunItemsTableUiColumnDefinitions } from "../../tableDefinitions";
import { FilterState } from "../../types";
import {
createFilterFromFilterState,
FilterList,
orderByToClickhouseSql,
StringFilter,
} from "../queries";
import {
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
} from "./clickhouse";
import { convertDatasetRunItemClickhouseToDomain } from "./dataset-run-items-converters";
import { DatasetRunItemRecordReadType } from "./definitions";
import { env } from "../../env";
import { commandClickhouse } from "./clickhouse";
import Decimal from "decimal.js";
type DatasetRunItemsTableQuery = {
projectId: string;
datasetId: string;
filter: FilterState;
orderBy?: OrderByState | OrderByState[];
limit?: number;
offset?: number;
};
type DatasetRunsMetricsTableQuery = {
projectId: string;
datasetId: string;
orderBy?: OrderByState;
limit?: number;
offset?: number;
};
export type DatasetRunsMetrics = {
id: string;
projectId: string;
createdAt: Date;
datasetId: string;
countRunItems: number;
avgTotalCost: Decimal;
avgLatency: number;
};
type DatasetRunsMetricsRecordType = {
dataset_run_id: string;
project_id: string;
dataset_run_created_at: string;
dataset_id: string;
count_run_items: number;
avg_latency_seconds: number;
avg_total_cost: number;
};
const convertDatasetRunsMetricsRecord = (
record: DatasetRunsMetricsRecordType,
): DatasetRunsMetrics => {
return {
id: record.dataset_run_id,
projectId: record.project_id,
createdAt: parseClickhouseUTCDateTimeFormat(record.dataset_run_created_at),
datasetId: record.dataset_id,
countRunItems: record.count_run_items,
avgTotalCost: record.avg_total_cost
? new Decimal(record.avg_total_cost)
: new Decimal(0),
avgLatency: record.avg_latency_seconds ?? 0,
};
};
const getProjectDatasetIdDefaultFilter = (
projectId: string,
datasetId: string,
) => {
return {
datasetRunItemsFilter: new FilterList([
new StringFilter({
clickhouseTable: "dataset_run_items",
field: "project_id",
operator: "=",
value: projectId,
}),
new StringFilter({
clickhouseTable: "dataset_run_items",
field: "dataset_id",
operator: "=",
value: datasetId,
}),
]),
};
};
const getDatasetRunsTableInternal = async <T>(
opts: DatasetRunsMetricsTableQuery & {
tags: Record<string, string>;
},
): Promise<Array<T>> => {
const { projectId, datasetId, orderBy, limit, offset } = opts;
const { datasetRunItemsFilter } = getProjectDatasetIdDefaultFilter(
projectId,
datasetId,
);
const appliedFilter = datasetRunItemsFilter.apply();
// Build ORDER BY array - conditionally add event_ts DESC for rows
const orderByArray: OrderByState[] = [];
// Add user ordering if provided
if (orderBy) {
orderByArray.push(orderBy);
}
const orderByClause = orderByToClickhouseSql(
orderByArray,
datasetRunItemsTableUiColumnDefinitions,
);
const query = `
WITH observations_filtered AS (
SELECT
o.id,
o.trace_id,
o.project_id,
o.start_time,
o.end_time,
o.total_cost
FROM observations o FINAL
WHERE o.project_id = {projectId: String}
AND o.start_time >= (
SELECT min(dri.dataset_run_created_at) - INTERVAL 1 DAY
FROM dataset_run_items dri
WHERE dri.project_id = {projectId: String}
AND dri.dataset_id = {datasetId: String}
)
AND o.start_time <= (
SELECT max(dri.dataset_run_created_at) + INTERVAL 1 DAY
FROM dataset_run_items dri
WHERE dri.project_id = {projectId: String}
AND dri.dataset_id = {datasetId: String}
)
),
traces_aggregated AS (
SELECT
of.trace_id,
of.project_id,
dateDiff('millisecond', min(of.start_time), max(of.end_time)) as latency_ms,
sum(of.total_cost) as total_cost
FROM observations_filtered of
JOIN dataset_run_items dri ON dri.trace_id = of.trace_id
AND dri.project_id = of.project_id
AND dri.observation_id IS NULL -- Only for trace-level dataset run items
WHERE dri.dataset_id = {datasetId: String}
GROUP BY of.trace_id, of.project_id
),
observations_direct AS (
SELECT
dri.observation_id,
dri.project_id,
dri.trace_id,
of.total_cost,
dateDiff('millisecond', of.start_time, of.end_time) as latency_ms
FROM dataset_run_items dri
JOIN observations_filtered of ON dri.observation_id = of.id
AND dri.project_id = of.project_id
AND dri.trace_id = of.trace_id
WHERE dri.dataset_id = {datasetId: String}
AND dri.observation_id IS NOT NULL -- Only for observation-level dataset run items
)
SELECT DISTINCT
dri.dataset_run_id as dataset_run_id,
dri.project_id as project_id,
dri.dataset_id as dataset_id,
dri.dataset_run_created_at as dataset_run_created_at,
count(DISTINCT dri.project_id, dri.dataset_id, dri.dataset_run_id, dri.dataset_item_id) as count_run_items,
-- Latency metrics (priority: observation > trace)
AVG(CASE
WHEN dri.observation_id IS NOT NULL AND od.latency_ms IS NOT NULL
THEN od.latency_ms / 1000.0
ELSE COALESCE(ta.latency_ms / 1000.0, 0)
END) as avg_latency_seconds,
-- Cost metrics (priority: observation > trace)
AVG(CASE
WHEN dri.observation_id IS NOT NULL AND od.total_cost IS NOT NULL
THEN od.total_cost
ELSE COALESCE(ta.total_cost, 0)
END) as avg_total_cost
FROM dataset_run_items dri
LEFT JOIN traces_aggregated ta
ON dri.trace_id = ta.trace_id
AND dri.project_id = ta.project_id
LEFT JOIN observations_direct od
ON dri.observation_id = od.observation_id
AND dri.project_id = od.project_id
AND dri.trace_id = od.trace_id
WHERE ${appliedFilter.query}
GROUP BY dri.project_id, dri.dataset_id, dri.dataset_run_id, dri.dataset_run_created_at
ORDER BY dri.dataset_run_created_at DESC
${orderByClause}
${limit !== undefined && offset !== undefined ? `LIMIT ${limit} OFFSET ${offset}` : ""};`;
const res = await queryClickhouse<T>({
query,
params: {
projectId,
datasetId,
...appliedFilter.params,
},
tags: {
...(opts.tags ?? {}),
feature: "datasets",
type: "dataset-run-items",
projectId,
datasetId,
},
});
return res;
};
export const getDatasetRunsTableMetricsCh = async (
opts: DatasetRunsMetricsTableQuery,
): Promise<DatasetRunsMetrics[]> => {
// First get the metrics (latency, cost, counts)
const rows = await getDatasetRunsTableInternal<DatasetRunsMetricsRecordType>({
...opts,
tags: { kind: "list" },
});
return rows.map(convertDatasetRunsMetricsRecord);
};
const getDatasetRunItemsTableInternal = async <T>(
opts: DatasetRunItemsTableQuery & {
select: "count" | "rows";
tags: Record<string, string>;
},
): Promise<Array<T>> => {
const { projectId, datasetId, filter, orderBy, limit, offset } = opts;
let selectString = "";
switch (opts.select) {
case "count":
selectString =
"count(DISTINCT dri.project_id, dri.dataset_id, dri.dataset_run_id, dri.dataset_item_id) as count";
break;
case "rows":
selectString = `
dri.id as id,
dri.project_id as project_id,
dri.trace_id as trace_id,
dri.observation_id as observation_id,
dri.dataset_id as dataset_id,
dri.dataset_run_id as dataset_run_id,
dri.dataset_item_id as dataset_item_id,
dri.error as error,
dri.created_at as created_at,
dri.updated_at as updated_at,
dri.dataset_run_name as dataset_run_name,
dri.dataset_run_description as dataset_run_description,
dri.dataset_run_metadata as dataset_run_metadata,
dri.dataset_run_created_at as dataset_run_created_at,
dri.dataset_item_input as dataset_item_input,
dri.dataset_item_expected_output as dataset_item_expected_output,
dri.dataset_item_metadata as dataset_item_metadata,
dri.is_deleted as is_deleted,
dri.event_ts as event_ts`;
break;
default:
throw new Error(`Unknown select type: ${opts.select}`);
}
const { datasetRunItemsFilter } = getProjectDatasetIdDefaultFilter(
projectId,
datasetId,
);
datasetRunItemsFilter.push(
...createFilterFromFilterState(
filter,
datasetRunItemsTableUiColumnDefinitions,
),
);
const appliedFilter = datasetRunItemsFilter.apply();
// Build ORDER BY array - conditionally add event_ts DESC for rows
const orderByArray: OrderByState[] = [];
// Add user ordering if provided
if (orderBy) {
if (Array.isArray(orderBy)) {
orderByArray.push(...orderBy);
} else {
orderByArray.push(orderBy);
}
}
// Add event_ts DESC for row queries (for deduplication)
if (opts.select === "rows") {
orderByArray.push({
column: "eventTs",
order: "DESC",
});
}
const orderByClause = orderByToClickhouseSql(
orderByArray,
datasetRunItemsTableUiColumnDefinitions,
);
const query = `
SELECT
${selectString}
FROM dataset_run_items dri
WHERE ${appliedFilter.query}
${orderByClause}
${opts.select === "rows" ? "LIMIT 1 BY dri.project_id, dri.dataset_id, dri.dataset_run_id, dri.dataset_item_id" : ""}
${limit !== undefined && offset !== undefined ? `LIMIT ${limit} OFFSET ${offset}` : ""};`;
const res = await queryClickhouse<T>({
query,
params: {
...appliedFilter.params,
},
tags: {
...(opts.tags ?? {}),
feature: "datasets",
type: "dataset-run-items",
projectId,
datasetId,
},
});
return res;
};
export const getDatasetRunItemsByDatasetIdCh = async (
opts: DatasetRunItemsTableQuery,
): Promise<DatasetRunItemDomain[]> => {
const rows =
await getDatasetRunItemsTableInternal<DatasetRunItemRecordReadType>({
...opts,
select: "rows",
tags: { kind: "list" },
});
return rows.map(convertDatasetRunItemClickhouseToDomain);
};
export const getDatasetRunItemsCountByDatasetIdCh = async (
opts: DatasetRunItemsTableQuery,
): Promise<number> => {
const rows = await getDatasetRunItemsTableInternal<{ count: string }>({
...opts,
select: "count",
tags: { kind: "list" },
});
return Number(rows[0]?.count);
};
export const deleteDatasetRunItemsByProjectId = async ({
projectId,
}: {
projectId: string;
}) => {
const query = `
DELETE FROM dataset_run_items
WHERE project_id = {projectId: String};
`;
await commandClickhouse({
query: query,
params: {
projectId,
},
clickhouseConfigs: {
request_timeout: env.LANGFUSE_CLICKHOUSE_DELETION_TIMEOUT_MS,
},
tags: {
feature: "datasets",
type: "dataset-run-items",
kind: "delete",
projectId,
},
});
};
export const deleteDatasetRunItemsByDatasetId = async ({
projectId,
datasetId,
}: {
projectId: string;
datasetId: string;
}) => {
const query = `
DELETE FROM dataset_run_items
WHERE project_id = {projectId: String}
AND dataset_id = {datasetId: String}
`;
await commandClickhouse({
query,
params: {
projectId,
datasetId,
},
clickhouseConfigs: {
request_timeout: env.LANGFUSE_CLICKHOUSE_DELETION_TIMEOUT_MS,
},
tags: {
feature: "datasets",
type: "dataset-run-items",
kind: "delete",
projectId,
},
});
};
export const deleteDatasetRunItemsByDatasetRunIds = async ({
projectId,
datasetRunIds,
datasetId,
}: {
projectId: string;
datasetRunIds: string[];
datasetId: string;
}) => {
const query = `
DELETE FROM dataset_run_items
WHERE project_id = {projectId: String}
AND dataset_id = {datasetId: String}
AND dataset_run_id IN ({datasetRunIds: Array(String)})
`;
await commandClickhouse({
query,
params: {
projectId,
datasetRunIds,
datasetId,
},
clickhouseConfigs: {
request_timeout: env.LANGFUSE_CLICKHOUSE_DELETION_TIMEOUT_MS,
},
tags: {
feature: "datasets",
type: "dataset-run-items",
kind: "delete",
projectId,
},
});
};
@@ -122,7 +122,7 @@ export const traceRecordInsertSchema = traceRecordBaseSchema.extend({
});
export type TraceRecordInsertType = z.infer<typeof traceRecordInsertSchema>;
export const traceMtRecordInsertSchema = z.object({
export const traceNullRecordInsertSchema = z.object({
// Identifiers
project_id: z.string(),
id: z.string(),
@@ -157,7 +157,9 @@ export const traceMtRecordInsertSchema = z.object({
updated_at: z.number(),
event_ts: z.number(),
});
export type TraceMtRecordInsertType = z.infer<typeof traceMtRecordInsertSchema>;
export type TraceNullRecordInsertType = z.infer<
typeof traceNullRecordInsertSchema
>;
export const scoreRecordBaseSchema = z.object({
id: z.string(),
@@ -196,6 +198,45 @@ export const scoreRecordInsertSchema = scoreRecordBaseSchema.extend({
});
export type ScoreRecordInsertType = z.infer<typeof scoreRecordInsertSchema>;
const datasetRunItemRecordBaseSchema = z.object({
id: z.string(),
project_id: z.string(),
trace_id: z.string(),
observation_id: z.string().nullish(),
dataset_id: z.string(),
dataset_run_id: z.string(),
dataset_item_id: z.string(),
dataset_run_name: z.string(),
dataset_run_description: z.string().nullish(),
dataset_run_metadata: z.record(z.string(), z.string()),
dataset_item_input: z.string(),
dataset_item_expected_output: z.string(),
dataset_item_metadata: z.record(z.string(), z.string()),
is_deleted: z.number(),
error: z.string().nullish(),
});
const datasetRunItemRecordReadSchema = datasetRunItemRecordBaseSchema.extend({
dataset_run_created_at: clickhouseStringDateSchema,
created_at: clickhouseStringDateSchema,
updated_at: clickhouseStringDateSchema,
event_ts: clickhouseStringDateSchema,
});
export type DatasetRunItemRecordReadType = z.infer<
typeof datasetRunItemRecordReadSchema
>;
export const datasetRunItemRecordInsertSchema =
datasetRunItemRecordBaseSchema.extend({
created_at: z.number(),
updated_at: z.number(),
event_ts: z.number(),
dataset_run_created_at: z.number(),
});
export type DatasetRunItemRecordInsertType = z.infer<
typeof datasetRunItemRecordInsertSchema
>;
export const blobStorageFileLogRecordBaseSchema = z.object({
id: z.string(),
project_id: z.string(),
@@ -308,6 +349,50 @@ export const convertPostgresTraceToInsert = (
};
};
export const convertPostgresDatasetRunItemToInsert = (
datasetRunItem: Record<string, any>,
): DatasetRunItemRecordInsertType => {
return {
id: datasetRunItem.id,
project_id: datasetRunItem.project_id,
dataset_run_id: datasetRunItem.dataset_run_id,
dataset_item_id: datasetRunItem.dataset_item_id,
dataset_id: datasetRunItem.dataset_id,
trace_id: datasetRunItem.trace_id,
observation_id: datasetRunItem.observation_id,
error: datasetRunItem.error,
created_at: datasetRunItem.created_at?.getTime(),
updated_at: datasetRunItem.updated_at?.getTime(),
// denormalized run data
dataset_run_name: datasetRunItem.dataset_run_name,
dataset_run_description: datasetRunItem.dataset_run_description,
dataset_run_metadata:
typeof datasetRunItem.dataset_run_metadata === "string" ||
typeof datasetRunItem.dataset_run_metadata === "number" ||
typeof datasetRunItem.dataset_run_metadata === "boolean"
? { metadata: datasetRunItem.dataset_run_metadata }
: Array.isArray(datasetRunItem.dataset_run_metadata)
? { metadata: datasetRunItem.dataset_run_metadata }
: (datasetRunItem.dataset_run_metadata ?? {}),
dataset_run_created_at: datasetRunItem.dataset_run_created_at?.getTime(),
// denormalized item data
dataset_item_input: JSON.stringify(datasetRunItem.dataset_item_input),
dataset_item_expected_output: JSON.stringify(
datasetRunItem.dataset_item_expected_output,
),
dataset_item_metadata:
typeof datasetRunItem.dataset_item_metadata === "string" ||
typeof datasetRunItem.dataset_item_metadata === "number" ||
typeof datasetRunItem.dataset_item_metadata === "boolean"
? { metadata: datasetRunItem.dataset_item_metadata }
: Array.isArray(datasetRunItem.dataset_item_metadata)
? { metadata: datasetRunItem.dataset_item_metadata }
: (datasetRunItem.dataset_item_metadata ?? {}),
event_ts: datasetRunItem.created_at?.getTime(),
is_deleted: 0,
};
};
/**
* Expects a single record from a
* `select o.*,
@@ -412,9 +497,9 @@ export const convertPostgresScoreToInsert = (
};
};
export const convertTraceToTraceMt = (
export const convertTraceToTraceNull = (
traceRecord: TraceRecordInsertType,
): TraceMtRecordInsertType => {
): TraceNullRecordInsertType => {
return {
// Identifiers
project_id: traceRecord.project_id,
@@ -452,13 +537,13 @@ export const convertTraceToTraceMt = (
};
};
export const convertObservationToTraceMt = (
export const convertObservationToTraceNull = (
observationRecord: ObservationRecordInsertType,
): TraceMtRecordInsertType => {
): TraceNullRecordInsertType => {
return {
// Identifiers
project_id: observationRecord.project_id,
// Use trace_id as the id in traces_mt. Always set given the conditions around calling the function
// Use trace_id as the id in traces_null. Always set given the conditions around calling the function
id: observationRecord.trace_id || "",
start_time: observationRecord.start_time,
end_time: observationRecord.end_time || null,
@@ -496,13 +581,13 @@ export const convertObservationToTraceMt = (
};
};
export const convertScoreToTraceMt = (
export const convertScoreToTraceNull = (
scoreRecord: ScoreRecordInsertType,
): TraceMtRecordInsertType => {
): TraceNullRecordInsertType => {
return {
// Identifiers
project_id: scoreRecord.project_id,
// Use trace_id as the id in traces_mt. Always set given the conditions around calling the function
// Use trace_id as the id in traces_null. Always set given the conditions around calling the function
id: scoreRecord.trace_id || "",
start_time: scoreRecord.timestamp,
end_time: null, // scores don't have end_time
@@ -13,3 +13,5 @@ export * from "./scores-utils";
export * from "./blobStorageLog";
export * from "./environments";
export * from "./automation-repository";
export * from "./dataset-run-items-converters";
export * from "./dataset-run-items";
@@ -23,7 +23,8 @@ import {
observationsTableUiColumnDefinitions,
} from "../../tableDefinitions";
import { OrderByState } from "../../interfaces/orderBy";
import { getTracesByIds } from "./traces";
import { getTimeframesTracesAMT, getTracesByIds } from "./traces";
import { measureAndReturn } from "../clickhouse/measureAndReturn";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { convertObservation } from "./observations_converters";
import { clickhouseSearchCondition } from "../queries/clickhouse-sql/search";
@@ -36,6 +37,7 @@ import { TracingSearchType } from "../../interfaces/search";
import { ClickHouseClientConfigOptions } from "@clickhouse/client";
import { ObservationType } from "../../domain";
import { recordDistribution } from "../instrumentation";
import { DEFAULT_RENDERING_PROPS, RenderingProps } from "../utils/rendering";
/**
* Checks if observation exists in clickhouse.
@@ -288,7 +290,7 @@ export const getObservationForTraceIdByName = async (
},
});
return records.map(convertObservation);
return records.map((record) => convertObservation(record));
};
export const getObservationById = async ({
@@ -298,6 +300,7 @@ export const getObservationById = async ({
startTime,
type,
traceId,
renderingProps = DEFAULT_RENDERING_PROPS,
}: {
id: string;
projectId: string;
@@ -305,6 +308,7 @@ export const getObservationById = async ({
startTime?: Date;
type?: ObservationType;
traceId?: string;
renderingProps?: RenderingProps;
}) => {
const records = await getObservationByIdInternal({
id,
@@ -313,8 +317,11 @@ export const getObservationById = async ({
startTime,
type,
traceId,
renderingProps,
});
const mapped = records.map(convertObservation);
const mapped = records.map((record) =>
convertObservation(record, renderingProps),
);
mapped.forEach((observation) => {
recordDistribution(
@@ -384,7 +391,7 @@ export const getObservationsById = async (
query,
params: { ids, projectId },
});
return records.map(convertObservation);
return records.map((record) => convertObservation(record));
};
const getObservationByIdInternal = async ({
@@ -394,6 +401,7 @@ const getObservationByIdInternal = async ({
startTime,
type,
traceId,
renderingProps = DEFAULT_RENDERING_PROPS,
}: {
id: string;
projectId: string;
@@ -401,6 +409,7 @@ const getObservationByIdInternal = async ({
startTime?: Date;
type?: ObservationType;
traceId?: string;
renderingProps?: RenderingProps;
}) => {
const query = `
SELECT
@@ -417,7 +426,7 @@ const getObservationByIdInternal = async ({
level,
status_message,
version,
${fetchWithInputOutput ? "input, output," : ""}
${fetchWithInputOutput ? (renderingProps.truncated ? `left(input, ${env.LANGFUSE_SERVER_SIDE_IO_CHAR_LIMIT}) as input, left(output, ${env.LANGFUSE_SERVER_SIDE_IO_CHAR_LIMIT}) as output,` : "input, output,") : ""}
provided_model_name,
internal_model_id,
model_parameters,
@@ -612,12 +621,14 @@ const getObservationsTableInternal = async <T>(
} = opts;
const selectString = selectIOAndMetadata
? `
${select},
${selectIOAndMetadata ? `o.input, o.output, o.metadata` : ""}
`
? `${select}, o.input, o.output, o.metadata`
: select;
const timeFilter = filter.find(
(f) =>
f.column === "Start Time" && (f.operator === ">=" || f.operator === ">"),
);
const scoresFilter = new FilterList([
new StringFilter({
clickhouseTable: "scores",
@@ -627,33 +638,22 @@ const getObservationsTableInternal = async <T>(
}),
]);
const timeFilter = opts.filter.find(
(f) =>
f.column === "Start Time" && (f.operator === ">=" || f.operator === ">"),
);
// query optimisation: joining traces onto observations is expensive. Hence, only join if the UI table contains filters on traces.
const traceTableFilter = opts.filter.filter(
(f) =>
observationsTableTraceUiColumnDefinitions
.map((c) => c.uiTableId)
.includes(f.column) ||
observationsTableTraceUiColumnDefinitions
.map((c) => c.uiTableName)
.includes(f.column),
);
const hasScoresFilter = filter.some((f) =>
f.column.toLowerCase().includes("scores"),
);
const orderByTraces = opts.orderBy
? observationsTableTraceUiColumnDefinitions
.map((c) => c.uiTableId)
.includes(opts.orderBy.column) ||
observationsTableTraceUiColumnDefinitions
.map((c) => c.uiTableName)
.includes(opts.orderBy.column)
// query optimisation: joining traces onto observations is expensive. Hence, only join if the UI table contains filters on traces.
const traceTableFilter = filter.filter((f) =>
observationsTableTraceUiColumnDefinitions.some(
(c) => c.uiTableId === f.column || c.uiTableName === f.column,
),
);
const orderByTraces = orderBy
? observationsTableTraceUiColumnDefinitions.some(
(c) =>
c.uiTableId === orderBy.column || c.uiTableName === orderBy.column,
)
: undefined;
timeFilter
@@ -759,7 +759,7 @@ const getObservationsTableInternal = async <T>(
SELECT
${selectString}
FROM observations o
${traceTableFilter.length > 0 || orderByTraces || search.query ? "LEFT JOIN traces t FINAL ON t.id = o.trace_id AND t.project_id = o.project_id" : ""}
${traceTableFilter.length > 0 || orderByTraces || search.query ? "LEFT JOIN __TRACE_TABLE__ t FINAL ON t.id = o.trace_id AND t.project_id = o.project_id" : ""}
${hasScoresFilter ? `LEFT JOIN scores_agg AS s ON s.trace_id = o.trace_id and s.observation_id = o.id` : ""}
WHERE ${appliedObservationsFilter.query}
@@ -769,30 +769,52 @@ const getObservationsTableInternal = async <T>(
${opts.select === "rows" ? "LIMIT 1 BY o.id, o.project_id" : ""}
${limit !== undefined && offset !== undefined ? `LIMIT ${limit} OFFSET ${offset}` : ""};`;
const res = await queryClickhouse<T>({
query,
params: {
...appliedScoresFilter.params,
...appliedObservationsFilter.params,
...(timeFilter
? {
tracesTimestampFilter: convertDateToClickhouseDateTime(
timeFilter.value as Date,
),
}
: {}),
...search.params,
return measureAndReturn({
operationName: "getObservationsTableInternal",
projectId,
minStartTime: (timeFilter?.value as Date) || undefined,
input: {
params: {
...appliedScoresFilter.params,
...appliedObservationsFilter.params,
...(timeFilter
? {
tracesTimestampFilter: convertDateToClickhouseDateTime(
timeFilter.value as Date,
),
}
: {}),
...search.params,
},
tags: {
...(opts.tags ?? {}),
feature: "tracing",
type: "observation",
projectId,
kind: opts.select,
operation_name: "getObservationsTableInternal",
},
},
tags: {
...(opts.tags ?? {}),
feature: "tracing",
type: "observation",
projectId,
existingExecution: async (input) => {
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", "traces"),
params: input.params,
tags: { ...input.tags, experiment_amt: "original" },
clickhouseConfigs,
});
},
newExecution: async (input) => {
const traceAmt = getTimeframesTracesAMT(
(timeFilter?.value as Date) || undefined,
);
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", traceAmt),
params: input.params,
tags: { ...input.tags, experiment_amt: "new" },
clickhouseConfigs,
});
},
clickhouseConfigs,
});
return res;
};
export const getObservationsGroupedByModel = async (
@@ -1473,6 +1495,9 @@ export const getObservationsForBlobStorageExport = function (
kind: "analytic",
projectId,
},
clickhouseConfigs: {
request_timeout: env.LANGFUSE_CLICKHOUSE_DATA_EXPORT_REQUEST_TIMEOUT_MS,
},
});
return records;
@@ -1483,6 +1508,15 @@ export const getGenerationsForPostHog = async function* (
minTimestamp: Date,
maxTimestamp: Date,
) {
// Determine which trace table to use based on experiment flag
const useAMT = env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT === "true";
// Subtract 7d from minTimestamp to account for shift in query
const traceTable = useAMT
? getTimeframesTracesAMT(
new Date(minTimestamp.getTime() - 7 * 24 * 60 * 60 * 1000),
)
: "traces";
const query = `
SELECT
o.name as name,
@@ -1507,7 +1541,7 @@ export const getGenerationsForPostHog = async function* (
t.tags as trace_tags,
t.metadata['$posthog_session_id'] as posthog_session_id
FROM observations o FINAL
LEFT JOIN traces t FINAL ON o.trace_id = t.id AND o.project_id = t.project_id
LEFT JOIN ${traceTable} t FINAL ON o.trace_id = t.id AND o.project_id = t.project_id
WHERE o.project_id = {projectId: String}
AND t.project_id = {projectId: String}
AND o.start_time >= {minTimestamp: DateTime64(3)}
@@ -1529,9 +1563,10 @@ export const getGenerationsForPostHog = async function* (
type: "observation",
kind: "analytic",
projectId,
experiment_amt: useAMT ? "new" : "original",
},
clickhouseConfigs: {
request_timeout: 300_000, // 5 minutes
request_timeout: env.LANGFUSE_CLICKHOUSE_DATA_EXPORT_REQUEST_TIMEOUT_MS,
clickhouse_settings: {
join_algorithm: "grace_hash",
grace_hash_join_initial_buckets: "32",
@@ -1545,6 +1580,7 @@ export const getGenerationsForPostHog = async function* (
timestamp: record.start_time,
langfuse_generation_name: record.name,
langfuse_trace_name: record.trace_name,
langfuse_trace_id: record.trace_id,
langfuse_url: `${baseUrl}/project/${projectId}/traces/${encodeURIComponent(record.trace_id as string)}?observation=${encodeURIComponent(record.id as string)}`,
langfuse_id: record.id,
langfuse_cost_usd: record.total_cost,
@@ -1,15 +1,20 @@
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { ObservationRecordReadType } from "./definitions";
import { parseJsonPrioritised } from "../../utils/json";
import {
Observation,
ObservationLevelType,
ObservationType,
} from "../../domain";
import { parseMetadataCHRecordToDomain } from "../utils/metadata_conversion";
import {
RenderingProps,
DEFAULT_RENDERING_PROPS,
applyInputOutputRendering,
} from "../utils/rendering";
export const convertObservation = (
record: ObservationRecordReadType,
renderingProps: RenderingProps = DEFAULT_RENDERING_PROPS,
): Observation => {
const reducedCostDetails = reduceUsageOrCostDetails(record.cost_details);
const reducedUsageDetails = reduceUsageOrCostDetails(record.usage_details);
@@ -30,10 +35,8 @@ export const convertObservation = (
level: record.level as ObservationLevelType,
statusMessage: record.status_message ?? null,
version: record.version ?? null,
input: record.input ? (parseJsonPrioritised(record.input) ?? null) : null,
output: record.output
? (parseJsonPrioritised(record.output) ?? null)
: null,
input: applyInputOutputRendering(record.input, renderingProps),
output: applyInputOutputRendering(record.output, renderingProps),
modelParameters: record.model_parameters
? (JSON.parse(record.model_parameters) ?? null)
: null,
+145 -30
View File
@@ -32,6 +32,8 @@ import { parseMetadataCHRecordToDomain } from "../utils/metadata_conversion";
import { ClickHouseClientConfigOptions } from "@clickhouse/client";
import { recordDistribution } from "../instrumentation";
import { prisma } from "../../db";
import { measureAndReturn } from "../clickhouse/measureAndReturn";
import { getTimeframesTracesAMT } from "./traces";
export const searchExistingAnnotationScore = async (
projectId: string,
@@ -284,6 +286,53 @@ export const getScoresForDatasetRuns = async <
return rows.map(convertToScore);
};
export const getTraceScoresForDatasetRuns = async (
projectId: string,
datasetRunIds: string[],
): Promise<Array<{ dataset_run_id: string } & any>> => {
if (datasetRunIds.length === 0) return [];
const query = `
SELECT
s.* EXCEPT (metadata),
length(mapKeys(s.metadata)) > 0 AS has_metadata,
dri.dataset_run_id as run_id
FROM dataset_run_items dri
JOIN scores s FINAL ON dri.trace_id = s.trace_id
AND dri.project_id = s.project_id
WHERE dri.project_id = {projectId: String}
AND dri.dataset_run_id IN {datasetRunIds: Array(String)}
AND s.project_id = {projectId: String}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id, dri.dataset_run_id
`;
const rows = await queryClickhouse<
Omit<ScoreRecordReadType, "metadata"> & {
has_metadata: 0 | 1;
run_id: string;
}
>({
query,
params: {
projectId,
datasetRunIds,
},
tags: {
feature: "dataset-run-items",
type: "trace-scores",
kind: "list",
projectId,
},
});
return rows.map((row) => ({
...convertToScore({ ...row, metadata: {} }),
datasetRunId: row.run_id,
hasMetadata: !!row.has_metadata,
}));
};
// Used in multiple places, including the public API, hence the non-default exclusion of metadata via excludeMetadata flag
export const getScoresForTraces = async <
ExcludeMetadata extends boolean,
@@ -895,31 +944,49 @@ const getScoresUiGeneric = async <T>(props: {
SELECT
${select}
FROM scores s final
${performTracesJoin ? "LEFT JOIN traces t ON s.trace_id = t.id AND t.project_id = s.project_id" : ""}
${performTracesJoin ? "LEFT JOIN __TRACE_TABLE__ t ON s.trace_id = t.id AND t.project_id = s.project_id" : ""}
WHERE s.project_id = {projectId: String}
${scoresFilterRes?.query ? `AND ${scoresFilterRes.query}` : ""}
${orderByToClickhouseSql(orderBy ?? null, scoresTableUiColumnDefinitions)}
${limit !== undefined && offset !== undefined ? `limit {limit: Int32} offset {offset: Int32}` : ""}
`;
const rows = await queryClickhouse<T>({
query: query,
params: {
projectId: projectId,
...(scoresFilterRes ? scoresFilterRes.params : {}),
limit: limit,
offset: offset,
return measureAndReturn({
operationName: "getScoresUiGeneric",
projectId,
input: {
params: {
projectId: projectId,
...(scoresFilterRes ? scoresFilterRes.params : {}),
limit: limit,
offset: offset,
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "score",
projectId,
select: props.select,
operation_name: "getScoresUiGeneric",
},
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "score",
projectId,
existingExecution: async (input) => {
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", "traces"),
params: input.params,
tags: { ...input.tags, experiment_amt: "original" },
clickhouseConfigs,
});
},
newExecution: async (input) => {
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", "traces_all_amt"),
params: input.params,
tags: { ...input.tags, experiment_amt: "new" },
clickhouseConfigs,
});
},
clickhouseConfigs,
});
return rows;
};
export const getScoreNames = async (
@@ -1086,7 +1153,7 @@ export const getNumericScoreHistogram = async (
const query = `
select s.value
from scores s
${traceFilter ? `LEFT JOIN traces t ON s.trace_id = t.id AND t.project_id = s.project_id` : ""}
${traceFilter ? `LEFT JOIN __TRACE_TABLE__ t ON s.trace_id = t.id AND t.project_id = s.project_id` : ""}
WHERE s.project_id = {projectId: String}
${traceFilter ? `AND t.project_id = {projectId: String}` : ""}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
@@ -1095,18 +1162,45 @@ export const getNumericScoreHistogram = async (
${limit !== undefined ? `limit {limit: Int32}` : ""}
`;
return queryClickhouse<{ value: number }>({
query,
params: {
projectId,
limit,
...(chFilterRes ? chFilterRes.params : {}),
// Extract timestamp from filter for AMT table selection
const timestampFilter = chFilter.find(
(f) => f.clickhouseTable === "traces" && f.field === "timestamp",
) as TimeFilter | undefined;
const timestamp = timestampFilter?.value;
return measureAndReturn({
operationName: "getNumericScoreHistogram",
projectId,
minStartTime: timestamp,
input: {
params: {
projectId,
limit,
...(chFilterRes ? chFilterRes.params : {}),
},
tags: {
feature: "tracing",
type: "score",
kind: "analytic",
projectId,
operation_name: "getNumericScoreHistogram",
},
timestamp,
},
tags: {
feature: "tracing",
type: "score",
kind: "analytic",
projectId,
existingExecution: async (input) => {
return queryClickhouse<{ value: number }>({
query: query.replace("__TRACE_TABLE__", "traces"),
params: input.params,
tags: { ...input.tags, experiment_amt: "original" },
});
},
newExecution: async (input) => {
const traceAmt = getTimeframesTracesAMT(input.timestamp);
return queryClickhouse<{ value: number }>({
query: query.replace("__TRACE_TABLE__", traceAmt),
params: input.params,
tags: { ...input.tags, experiment_amt: "new" },
});
},
});
};
@@ -1320,6 +1414,9 @@ export const getScoresForBlobStorageExport = function (
kind: "analytic",
projectId,
},
clickhouseConfigs: {
request_timeout: env.LANGFUSE_CLICKHOUSE_DATA_EXPORT_REQUEST_TIMEOUT_MS,
},
});
return records;
@@ -1330,21 +1427,34 @@ export const getScoresForPostHog = async function* (
minTimestamp: Date,
maxTimestamp: Date,
) {
// Determine which trace table to use based on experiment flag
const useAMT = env.LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT === "true";
// Subtract 7d from minTimestamp to account for shift in query
const traceTable = useAMT
? getTimeframesTracesAMT(
new Date(minTimestamp.getTime() - 7 * 24 * 60 * 60 * 1000),
)
: "traces";
const query = ` SELECT
s.id as id,
s.timestamp as timestamp,
s.name as name,
s.value as value,
s.string_value as string_value,
s.data_type as data_type,
s.comment as comment,
s.environment as environment,
t.id as trace_id,
t.name as trace_name,
t.session_id as trace_session_id,
t.user_id as trace_user_id,
t.release as trace_release,
t.tags as trace_tags,
s.metadata as metadata,
t.metadata['$posthog_session_id'] as posthog_session_id
FROM scores s FINAL
LEFT JOIN traces t FINAL ON s.trace_id = t.id AND s.project_id = t.project_id
LEFT JOIN ${traceTable} t FINAL ON s.trace_id = t.id AND s.project_id = t.project_id
WHERE s.project_id = {projectId: String}
AND t.project_id = {projectId: String}
AND s.timestamp >= {minTimestamp: DateTime64(3)}
@@ -1365,9 +1475,10 @@ export const getScoresForPostHog = async function* (
type: "score",
kind: "analytic",
projectId,
experiment_amt: useAMT ? "new" : "original",
},
clickhouseConfigs: {
request_timeout: 300_000, // 5 minutes
request_timeout: env.LANGFUSE_CLICKHOUSE_DATA_EXPORT_REQUEST_TIMEOUT_MS,
clickhouse_settings: {
join_algorithm: "grace_hash",
grace_hash_join_initial_buckets: "32",
@@ -1382,7 +1493,11 @@ export const getScoresForPostHog = async function* (
langfuse_score_name: record.name,
langfuse_score_value: record.value,
langfuse_score_comment: record.comment,
langfuse_score_metadata: record.metadata,
langfuse_score_string_value: record.string_value,
langfuse_score_data_type: record.data_type,
langfuse_trace_name: record.trace_name,
langfuse_trace_id: record.trace_id,
langfuse_id: record.id,
langfuse_session_id: record.trace_session_id,
langfuse_project_id: projectId,
File diff suppressed because it is too large Load Diff
@@ -1,9 +1,13 @@
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { TraceRecordReadType } from "./definitions";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { parseJsonPrioritised } from "../../utils/json";
import { TraceDomain } from "../../domain";
import { parseMetadataCHRecordToDomain } from "../utils/metadata_conversion";
import {
RenderingProps,
DEFAULT_RENDERING_PROPS,
applyInputOutputRendering,
} from "../utils/rendering";
export const convertTraceDomainToClickhouse = (
trace: TraceDomain,
@@ -33,6 +37,7 @@ export const convertTraceDomainToClickhouse = (
export const convertClickhouseToDomain = (
record: TraceRecordReadType,
renderingProps: RenderingProps = DEFAULT_RENDERING_PROPS,
): TraceDomain => {
return {
id: record.id,
@@ -47,10 +52,8 @@ export const convertClickhouseToDomain = (
userId: record.user_id ?? null,
sessionId: record.session_id ?? null,
public: record.public,
input: record.input ? (parseJsonPrioritised(record.input) ?? null) : null,
output: record.output
? (parseJsonPrioritised(record.output) ?? null)
: null,
input: applyInputOutputRendering(record.input, renderingProps),
output: applyInputOutputRendering(record.output, renderingProps),
metadata: parseMetadataCHRecordToDomain(record.metadata),
createdAt: parseClickhouseUTCDateTimeFormat(record.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(record.updated_at),
@@ -35,6 +35,12 @@ export const HistogramChartConfig = BaseTotalValueChartConfig.extend({
export const PivotTableChartConfig = BaseTotalValueChartConfig.extend({
type: z.literal("PIVOT_TABLE"),
defaultSort: z
.object({
column: z.string(),
order: z.enum(["ASC", "DESC"]),
})
.optional(),
});
// Define dimension schema
@@ -1,7 +1,12 @@
import z from "zod/v4";
import { prisma } from "../../../db";
import { LangfuseNotFoundError, QUEUE_ERROR_MESSAGES } from "../../../errors";
import {
ForbiddenError,
LangfuseNotFoundError,
QUEUE_ERROR_MESSAGES,
} from "../../../errors";
import { LLMApiKeySchema, ZodModelConfig } from "../../llm/types";
import { testModelCall } from "../../llm/testModelCall";
type ValidConfig = {
provider: string;
@@ -47,6 +52,23 @@ export class DefaultEvalModelService {
);
}
try {
if (LLMApiKeySchema.safeParse(llmApiKey).success) {
// Make a test structured output call to validate the LLM key
await testModelCall({
provider,
model,
apiKey: llmApiKey as z.infer<typeof LLMApiKeySchema>,
modelConfig: modelParams,
});
}
} catch (err) {
const message = err instanceof Error ? err.message : "Unknown error";
throw new ForbiddenError(
`Model configuration not valid for evaluation. ${message}`,
);
}
// Create or update the default model
return prisma.defaultLlmModel.upsert({
where: {
@@ -0,0 +1,386 @@
/**
* Slack Integration Service
*
* Simplified service that properly uses the official Slack SDK libraries:
* - @slack/oauth InstallProvider for OAuth flow management
* - @slack/web-api WebClient for Slack API operations
* - Metadata-based project-to-team mapping
*/
import { WebClient } from "@slack/web-api";
import { InstallProvider } from "@slack/oauth";
import { logger } from "../logger";
import { env } from "../../env";
import { prisma } from "../../db";
import { encrypt, decrypt } from "../../encryption";
// Types for Slack integration
export interface SlackChannel {
id: string;
name: string;
isPrivate: boolean;
isMember: boolean;
}
export interface SlackMessageParams {
client: WebClient;
channelId: string;
blocks: any[];
text?: string;
}
export interface SlackMessageResponse {
messageTs: string;
channel: string;
}
// Interface for Slack installation metadata
export interface SlackInstallationMetadata {
projectId: string;
}
/**
* Type guard to validate Slack installation metadata
*/
function isSlackInstallationMetadata(
metadata: unknown,
): metadata is SlackInstallationMetadata {
return (
typeof metadata === "object" &&
metadata !== null &&
"projectId" in metadata &&
typeof metadata.projectId === "string" &&
metadata.projectId.length > 0
);
}
/**
* Helper function to safely parse and validate Slack installation metadata
*/
export function parseSlackInstallationMetadata(
metadata: unknown,
): SlackInstallationMetadata {
if (typeof metadata !== "string") {
throw new Error("Installation metadata must be a string");
}
let parsedMetadata: unknown;
try {
parsedMetadata = JSON.parse(metadata);
} catch {
throw new Error("Failed to parse installation metadata as JSON");
}
if (!isSlackInstallationMetadata(parsedMetadata)) {
throw new Error(
"Invalid installation metadata: missing or invalid projectId",
);
}
return parsedMetadata;
}
/**
* Slack Service Class
*
* Uses InstallProvider for OAuth flow and metadata-based project mapping.
* Much simpler than the previous implementation while maintaining all functionality.
*/
export class SlackService {
private static instance: SlackService | null = null;
private installer: InstallProvider;
private constructor() {
this.installer = new InstallProvider({
clientId: env.SLACK_CLIENT_ID!,
clientSecret: env.SLACK_CLIENT_SECRET!,
stateSecret: env.SLACK_STATE_SECRET!,
installUrlOptions: {
scopes: ["channels:read", "chat:write", "chat:write.public"],
},
installationStore: {
storeInstallation: async (installation) => {
try {
const metadata = parseSlackInstallationMetadata(
installation.metadata,
);
const projectId = metadata.projectId;
logger.info("Storing Slack installation for project", {
projectId,
teamId: installation.team?.id,
teamName: installation.team?.name,
});
// Store by projectId (one integration per project)
await prisma.slackIntegration.upsert({
where: { projectId },
create: {
projectId,
teamId: installation.team?.id!,
teamName: installation.team?.name!,
botToken: encrypt(installation.bot?.token!),
botUserId: installation.bot?.userId!,
},
update: {
teamId: installation.team?.id!,
teamName: installation.team?.name!,
botToken: encrypt(installation.bot?.token!),
botUserId: installation.bot?.userId!,
},
});
logger.info("Slack installation stored successfully", {
projectId,
teamId: installation.team?.id,
});
} catch (error) {
logger.error("Failed to store Slack installation", { error });
throw error;
}
},
fetchInstallation: async (installQuery) => {
try {
// Handle both teamId and projectId lookups
// When SDK calls with teamId, we treat it as projectId
const lookupId = installQuery.teamId;
if (!lookupId) {
throw new Error("No lookup ID provided");
}
const integration = await prisma.slackIntegration.findFirst({
where: {
OR: [
{ teamId: lookupId }, // Actual team ID lookup
{ projectId: lookupId }, // Project ID lookup (our custom usage)
],
},
});
if (!integration) {
throw new Error("Slack integration not found");
}
// Return full Installation interface as expected by SDK
return {
team: {
id: integration.teamId,
name: integration.teamName,
},
bot: {
id: integration.botUserId,
token: decrypt(integration.botToken),
userId: integration.botUserId,
scopes: [],
},
enterprise: undefined,
user: {
token: undefined,
refreshToken: undefined,
expiresAt: undefined,
scopes: undefined,
id: integration.botUserId,
},
};
} catch (error) {
logger.error("Failed to fetch Slack installation", { error });
throw error;
}
},
deleteInstallation: async (installQuery) => {
try {
const lookupId = installQuery.teamId;
if (!lookupId) {
throw new Error("No lookup ID provided for deletion");
}
await prisma.slackIntegration.deleteMany({
where: {
OR: [{ teamId: lookupId }, { projectId: lookupId }],
},
});
logger.info("Slack installation deleted successfully", {
lookupId,
});
} catch (error) {
logger.error("Failed to delete Slack installation", { error });
throw error;
}
},
},
});
}
/**
* Get singleton instance of SlackService
*/
static getInstance(): SlackService {
if (!SlackService.instance) {
SlackService.instance = new SlackService();
}
return SlackService.instance;
}
/**
* Get the configured InstallProvider instance for OAuth handling
*/
getInstaller(): InstallProvider {
return this.installer;
}
/**
* Reset the singleton instance (useful for testing)
*/
static resetInstance(): void {
SlackService.instance = null;
}
/**
* Delete Slack integration for a project
*/
async deleteIntegration(projectId: string): Promise<void> {
try {
if (!this.installer.installationStore?.deleteInstallation) {
throw new Error("Installation store not configured");
}
await this.installer.installationStore.deleteInstallation({
teamId: projectId,
isEnterpriseInstall: false,
enterpriseId: undefined,
});
logger.info("Slack integration deleted for project", { projectId });
} catch (error) {
logger.error("Failed to delete Slack integration", { error, projectId });
throw new Error(
`Failed to delete integration: ${error instanceof Error ? error.message : "Unknown error"}`,
);
}
}
/**
* Get WebClient for a specific project
*/
async getWebClientForProject(projectId: string): Promise<WebClient> {
try {
// Use projectId as the teamId parameter (handled by our fetchInstallation)
const auth = await this.installer.authorize({
teamId: projectId,
isEnterpriseInstall: false,
enterpriseId: undefined,
});
if (!auth.botToken) {
throw new Error("No bot token found for project");
}
const client = new WebClient(auth.botToken);
logger.debug("Created WebClient for project", { projectId });
return client;
} catch (error) {
logger.error("Failed to create WebClient for project", {
error,
projectId,
});
throw new Error(
`Failed to create WebClient: ${error instanceof Error ? error.message : "Unknown error"}`,
);
}
}
/**
* Get channels accessible to the bot
*/
async getChannels(client: WebClient): Promise<SlackChannel[]> {
try {
const result = await client.conversations.list({
exclude_archived: true,
types: "public_channel",
limit: 200,
});
if (!result.ok) {
throw new Error(`Slack API error: ${result.error}`);
}
const channels: SlackChannel[] = (result.channels || []).map(
(channel) => ({
id: channel.id!,
name: channel.name!,
isPrivate: channel.is_private || false,
isMember: channel.is_member || false,
}),
);
logger.debug("Retrieved channels from Slack", {
channelCount: channels.length,
});
return channels;
} catch (error) {
logger.error("Failed to fetch channels", { error });
throw new Error(
`Failed to fetch channels: ${error instanceof Error ? error.message : "Unknown error"}`,
);
}
}
/**
* Send a message to a Slack channel
*/
async sendMessage(params: SlackMessageParams): Promise<SlackMessageResponse> {
try {
const result = await params.client.chat.postMessage({
channel: params.channelId,
blocks: params.blocks,
text: params.text || "Langfuse Notification",
unfurl_links: false,
unfurl_media: false,
});
if (!result.ok) {
throw new Error(`Failed to send message: ${result.error}`);
}
const response = {
messageTs: result.ts!,
channel: result.channel!,
};
logger.info("Message sent successfully to Slack", {
channel: params.channelId,
messageTs: response.messageTs,
});
return response;
} catch (error) {
logger.error("Failed to send message", {
error,
channelId: params.channelId,
});
throw new Error(
`Failed to send message: ${error instanceof Error ? error.message : "Unknown error"}`,
);
}
}
/**
* Validate a WebClient instance
*/
async validateClient(client: WebClient): Promise<boolean> {
try {
const result = await client.auth.test();
return result.ok || false;
} catch (error) {
logger.warn("Client validation failed", { error });
return false;
}
}
}
@@ -104,6 +104,7 @@ export class StorageServiceFactory {
}
}
let azureContainersExists: Record<string, boolean> = {};
class AzureBlobStorageService implements StorageService {
private client: ContainerClient;
private container: string;
@@ -139,8 +140,18 @@ class AzureBlobStorageService implements StorageService {
}
private async createContainerIfNotExists(): Promise<void> {
// Skip container existence check if environment variable is set
if (env.LANGFUSE_AZURE_SKIP_CONTAINER_CHECK === "true") {
return;
}
try {
if (azureContainersExists[this.container]) {
return; // Container already exists, no need to create it again
}
await this.client.createIfNotExists();
azureContainersExists[this.container] = true; // Mark container as created
logger.info(`Azure Blob Storage container ${this.container} created`);
} catch (err) {
logger.error(
`Failed to create Azure Blob Storage container ${this.container}`,
@@ -9,7 +9,7 @@ type FetchDatasetItemsTableProps = {
filter: FilterState;
};
const getDatasetRunItemsTableGeneric = async <T>(
const getDatasetRunItemsTableGenericPg = async <T>(
props: FetchDatasetItemsTableProps,
) => {
const { select, projectId, filter } = props;
@@ -45,11 +45,11 @@ const getDatasetRunItemsTableGeneric = async <T>(
return res;
};
export const getDatasetRunItemsTableCount = async (props: {
export const getDatasetRunItemsTableCountPg = async (props: {
projectId: string;
filter: FilterState;
}) => {
const res = await getDatasetRunItemsTableGeneric<Array<{ count: bigint }>>({
const res = await getDatasetRunItemsTableGenericPg<Array<{ count: bigint }>>({
select: "count",
projectId: props.projectId,
filter: props.filter,
@@ -0,0 +1,134 @@
import React from "react";
import {
Body,
Button,
Container,
Head,
Heading,
Hr,
Html,
Img,
Preview,
Section,
Text,
Tailwind,
Row,
Column,
} from "@react-email/components";
interface BillingAlertEmailProps {
organizationName: string;
currentUsage: number;
threshold: number;
billingUrl: string;
receiverEmail: string;
}
export const BillingAlertEmailTemplate = ({
organizationName,
currentUsage,
threshold,
billingUrl,
receiverEmail,
}: BillingAlertEmailProps) => {
return (
<Html>
<Head />
<Preview>
Your Langfuse Cloud usage is {`${currentUsage}`} events for the current
billing period
</Preview>
<Tailwind>
<Body className="bg-background my-auto mx-auto font-sans">
<Container className="mx-auto my-10 w-[465px] rounded border border-solid border-[#eaeaea] p-5">
<Section className="mt-8">
<Img
src="https://static.langfuse.com/langfuse_logo_transactional_email.png"
width="40"
height="40"
alt="Langfuse"
className="mx-auto my-0"
/>
</Section>
<Section>
<Heading className="mx-0 my-[30px] p-0 text-center text-2xl font-normal text-black">
Usage Threshold Exceeded
</Heading>
<Text className="text-gray-700 text-sm leading-6">
Your organization &quot;{organizationName}&quot; has exceeded
the configured billing threshold
</Text>
</Section>
<Section className="mt-8">
<div className="bg-gray-50 border border-gray-200 rounded-lg p-4">
<Row>
<Column className="text-center">
<Text className="text-gray-600 text-sm font-medium m-0 mb-1">
Current Usage (# Events)
</Text>
<Text className="text-2xl font-bold text-gray-900 m-0">
{currentUsage}
</Text>
</Column>
<Column className="text-center">
<Text className="text-gray-600 text-sm font-medium m-0 mb-1">
Alert Threshold (# Events)
</Text>
<Text className="text-2xl font-bold text-gray-900 m-0">
{threshold}
</Text>
</Column>
</Row>
</div>
</Section>
<Section className="mt-8 text-center">
<Button
className="rounded bg-black px-5 py-3 text-center text-xs font-semibold text-white no-underline"
href={billingUrl}
>
View Billing Page and Manage Alerts
</Button>
</Section>
<Section className="mt-8">
<Heading className="text-black text-[18px] font-semibold">
What happens next?
</Heading>
<Text className="text-gray-700 text-sm leading-6">
Your current billing cycle continues normally
<br />
Charges will appear on your next invoice
<br />
You can adjust usage or modify alert thresholds
<br /> Contact support if you have questions about your bill
</Text>
</Section>
<Hr className="border border-solid border-[#eaeaea] my-[26px] mx-0 w-full" />
<Section>
<Text className="text-[#666666] text-[12px] leading-[24px]">
This email was sent to {receiverEmail} regarding billing alerts
for &quot;{organizationName}&quot;.
</Text>
<Text className="text-[#666666] text-[12px] leading-[24px]">
Questions? Contact us at{" "}
<a
href="mailto:support@langfuse.com"
className="text-blue-600 no-underline"
>
support@langfuse.com
</a>
</Text>
</Section>
</Container>
</Body>
</Tailwind>
</Html>
);
};
export default BillingAlertEmailTemplate;
@@ -0,0 +1,59 @@
import { createTransport } from "nodemailer";
import { parseConnectionUrl } from "nodemailer/lib/shared/index.js";
import { render } from "@react-email/render";
import { BillingAlertEmailTemplate } from "./BillingAlertEmailTemplate";
import { logger } from "../../../logger";
export interface BillingAlertEmailProps {
env: Partial<
Record<"EMAIL_FROM_ADDRESS" | "SMTP_CONNECTION_URL", string | undefined>
>;
organizationName: string;
currentUsage: number;
threshold: number;
billingUrl: string;
receiverEmail: string;
}
export const sendBillingAlertEmail = async ({
env,
organizationName,
currentUsage,
threshold,
billingUrl,
receiverEmail,
}: BillingAlertEmailProps) => {
if (!env.EMAIL_FROM_ADDRESS || !env.SMTP_CONNECTION_URL) {
logger.error(
"Missing environment variables for sending billing alert email.",
);
return;
}
try {
const mailer = createTransport(parseConnectionUrl(env.SMTP_CONNECTION_URL));
const emailSubject = `Langfuse Cloud Billing Alert: ${organizationName} usage exceeded ${threshold} events`;
const emailHtml = await render(
BillingAlertEmailTemplate({
organizationName,
currentUsage,
threshold,
billingUrl,
receiverEmail,
}),
);
await mailer.sendMail({
to: receiverEmail,
from: {
address: env.EMAIL_FROM_ADDRESS,
name: "Langfuse",
},
subject: emailSubject,
html: emailHtml,
});
} catch (error) {
logger.error(`Failed to send billing alert email`, error);
}
};
@@ -3,6 +3,7 @@ import { OrderByState } from "../../interfaces/orderBy";
import { sessionCols } from "../../tableDefinitions/mapSessionTable";
import { FilterState } from "../../types";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { measureAndReturn } from "../clickhouse/measureAndReturn";
import { DateTimeFilter, FilterList, orderByToClickhouseSql } from "../queries";
import {
getProjectIdDefaultFilter,
@@ -11,6 +12,7 @@ import {
import {
TRACE_TO_OBSERVATIONS_INTERVAL,
queryClickhouse,
getTimeframesTracesAMT,
} from "../repositories";
export type SessionDataReturnType = {
@@ -22,6 +24,8 @@ export type SessionDataReturnType = {
trace_count: number;
trace_tags: string[];
trace_environment?: string;
scores_avg?: Array<Array<[string, number]>>;
score_categories?: Array<Array<string>>;
};
export type SessionWithMetricsReturnType = SessionDataReturnType & {
@@ -157,7 +161,9 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
session_total_cost,
session_input_usage,
session_output_usage,
session_total_usage`;
session_total_usage,
scores_avg,
score_categories`;
break;
default: {
const exhaustiveCheckDefault: never = select;
@@ -165,7 +171,7 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
}
}
const { tracesFilter } = getProjectIdDefaultFilter(projectId, {
const { tracesFilter, scoresFilter } = getProjectIdDefaultFilter(projectId, {
tracesPrefix: "s",
});
@@ -174,6 +180,7 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
const tracesFilterRes = tracesFilter
.filter((f) => f.field !== "environment")
.apply();
const scoresFilterRes = scoresFilter.apply();
const traceTimestampFilter: DateTimeFilter | undefined = tracesFilter.find(
(f) =>
@@ -193,20 +200,25 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
);
}
const additionalSingleTraceFilter = tracesFilter.find(
(f) =>
f.field === "bookmarked" ||
f.field === "session_id" ||
f.field === "environment",
);
if (additionalSingleTraceFilter) {
filters.push(additionalSingleTraceFilter);
}
tracesFilter
.filter(
(f) =>
f.field === "bookmarked" ||
f.field === "session_id" ||
f.field === "environment",
)
.forEach((f) => filters.push(f));
const singleTraceFilter =
filters.length > 0 ? new FilterList(filters).apply() : undefined;
const requiresScoresJoin =
tracesFilter.find((f) => f.clickhouseTable === "scores") !== undefined ||
sessionCols.find(
(c) =>
c.uiTableName === orderBy?.column || c.uiTableId === orderBy?.column,
)?.clickhouseTableName === "scores";
const hasMetricsFilter =
tracesFilter.find((f) =>
[
@@ -217,6 +229,8 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
"session_total_usage",
"session_output_usage",
"session_input_usage",
"scores_avg",
"score_categories",
].includes(f.field),
) ||
(orderBy &&
@@ -233,115 +247,188 @@ const getSessionsTableGeneric = async <T>(props: FetchSessionsTableProps) => {
const selectMetrics = select === "metrics" || hasMetricsFilter;
const scoresCte = `scores_agg AS (
SELECT
project_id,
session_id AS score_session_id,
-- For numeric scores, use tuples of (name, avg_value)
groupArrayIf(
tuple(name, avg_value),
data_type IN ('NUMERIC', 'BOOLEAN')
) AS scores_avg,
-- For categorical scores, use name:value format for improved query performance
groupArrayIf(
concat(name, ':', string_value),
data_type = 'CATEGORICAL' AND notEmpty(string_value)
) AS score_categories
FROM (
SELECT
project_id,
session_id,
name,
data_type,
string_value,
avg(value) avg_value
FROM scores s FINAL
WHERE
project_id = {projectId: String}
${scoresFilterRes ? `AND ${scoresFilterRes.query}` : ""}
GROUP BY
project_id,
session_id,
name,
data_type,
string_value
) tmp
GROUP BY
project_id, session_id
)`;
// We use deduplicated traces and observations CTEs instead of final to be able to use Skip indices in Clickhouse.
const query = `
WITH deduplicated_traces AS (
SELECT * EXCEPT input, output, metadata
FROM traces t
WHERE t.session_id IS NOT NULL
AND t.project_id = {projectId: String}
${singleTraceFilter?.query ? ` AND ${singleTraceFilter.query}` : ""}
ORDER BY event_ts DESC
LIMIT 1 BY id, project_id
),
deduplicated_observations AS (
SELECT *
FROM observations o
WHERE o.project_id = {projectId: String}
${traceTimestampFilter ? `AND o.start_time >= {observationsStartTime: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
AND o.trace_id IN (
SELECT id
FROM deduplicated_traces
)
ORDER BY event_ts DESC
LIMIT 1 BY id, project_id
),
observations_agg AS (
SELECT o.trace_id,
count(*) as obs_count,
min(o.start_time) as min_start_time,
max(o.end_time) as max_end_time,
sumMap(usage_details) as sum_usage_details,
sumMap(cost_details) as sum_cost_details,
anyLast(project_id) as project_id
FROM deduplicated_observations o
WHERE o.project_id = {projectId: String}
${traceTimestampFilter ? `AND o.start_time >= {observationsStartTime: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
GROUP BY o.trace_id
),
session_data AS (
SELECT
t.session_id,
anyLast(t.project_id) as project_id,
max(t.timestamp) as max_timestamp,
min(t.timestamp) as min_timestamp,
groupArray(t.id) AS trace_ids,
groupUniqArray(t.user_id) AS user_ids,
count(*) as trace_count,
groupUniqArrayArray(t.tags) as trace_tags,
anyLast(t.environment) as trace_environment
-- Aggregate observations data at session level
${
selectMetrics
? `
,
sum(o.obs_count) as total_observations,
-- Use minIf, because ClickHouse fills 1970-01-01 on left joins. We assume that no
-- LLM session started on that date so this behaviour should yield better results.
date_diff('millisecond', minIf(min_start_time, min_start_time > '1970-01-01'), max(max_end_time)) as duration,
sumMap(o.sum_usage_details) as session_usage_details,
sumMap(o.sum_cost_details) as session_cost_details,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'input') > 0, sumMap(o.sum_cost_details)))) as session_input_cost,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'output') > 0, sumMap(o.sum_cost_details)))) as session_output_cost,
sumMap(o.sum_cost_details)['total'] as session_total_cost,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'input') > 0, sumMap(o.sum_usage_details)))) as session_input_usage,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'output') > 0, sumMap(o.sum_usage_details)))) as session_output_usage,
sumMap(o.sum_usage_details)['total'] as session_total_usage`
: ""
}
FROM deduplicated_traces t
${
selectMetrics
? `LEFT JOIN observations_agg o
ON t.id = o.trace_id AND t.project_id = o.project_id`
: ""
}
WHERE t.session_id IS NOT NULL
WITH ${select === "metrics" || requiresScoresJoin ? `${scoresCte},` : ""}
deduplicated_traces AS (
SELECT * EXCEPT input, output, metadata
FROM __TRACE_TABLE__ t
WHERE t.session_id IS NOT NULL
AND t.project_id = {projectId: String}
${singleTraceFilter?.query ? ` AND ${singleTraceFilter.query}` : ""}
GROUP BY t.session_id
)
SELECT ${sqlSelect}
FROM session_data s
WHERE ${tracesFilterRes.query ? tracesFilterRes.query : ""}
${orderByToClickhouseSql(orderBy ?? null, sessionCols)}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
LIMIT 1 BY id, project_id
),
deduplicated_observations AS (
SELECT *
FROM observations o
WHERE o.project_id = {projectId: String}
${traceTimestampFilter ? `AND o.start_time >= {observationsStartTime: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
AND o.trace_id IN (
SELECT id
FROM deduplicated_traces
)
ORDER BY event_ts DESC
LIMIT 1 BY id, project_id
),
observations_agg AS (
SELECT o.trace_id,
count(*) as obs_count,
min(o.start_time) as min_start_time,
max(o.end_time) as max_end_time,
sumMap(usage_details) as sum_usage_details,
sumMap(cost_details) as sum_cost_details,
anyLast(project_id) as project_id
FROM deduplicated_observations o
WHERE o.project_id = {projectId: String}
${traceTimestampFilter ? `AND o.start_time >= {observationsStartTime: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
GROUP BY o.trace_id
),
session_data AS (
SELECT
t.session_id,
anyLast(t.project_id) as project_id,
max(t.timestamp) as max_timestamp,
min(t.timestamp) as min_timestamp,
groupArray(t.id) AS trace_ids,
groupUniqArray(t.user_id) AS user_ids,
count(*) as trace_count,
groupUniqArrayArray(t.tags) as trace_tags,
anyLast(t.environment) as trace_environment
-- Aggregate observations data at session level
${
selectMetrics
? `,
sum(o.obs_count) as total_observations,
-- Use minIf, because ClickHouse fills 1970-01-01 on left joins. We assume that no
-- LLM session started on that date so this behaviour should yield better results.
date_diff('second', minIf(min_start_time, min_start_time > '1970-01-01'), max(max_end_time)) as duration,
sumMap(o.sum_usage_details) as session_usage_details,
sumMap(o.sum_cost_details) as session_cost_details,
${
select === "metrics" || requiresScoresJoin
? `groupUniqArrayArray(s.scores_avg) as scores_avg,
groupUniqArrayArray(s.score_categories) as score_categories,`
: ""
}
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'input') > 0, sumMap(o.sum_cost_details)))) as session_input_cost,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'output') > 0, sumMap(o.sum_cost_details)))) as session_output_cost,
sumMap(o.sum_cost_details)['total'] as session_total_cost,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'input') > 0, sumMap(o.sum_usage_details)))) as session_input_usage,
arraySum(mapValues(mapFilter(x -> positionCaseInsensitive(x.1, 'output') > 0, sumMap(o.sum_usage_details)))) as session_output_usage,
sumMap(o.sum_usage_details)['total'] as session_total_usage`
: ""
}
FROM deduplicated_traces t
${
selectMetrics
? `LEFT JOIN observations_agg o
ON t.id = o.trace_id AND t.project_id = o.project_id`
: ""
}
${select === "metrics" || requiresScoresJoin ? `LEFT JOIN scores_agg s on s.project_id = t.project_id and t.session_id = s.score_session_id` : ""}
WHERE t.session_id IS NOT NULL
AND t.project_id = {projectId: String}
${singleTraceFilter?.query ? ` AND ${singleTraceFilter.query}` : ""}
GROUP BY t.session_id
)
SELECT ${sqlSelect}
FROM session_data s
WHERE ${tracesFilterRes.query ? tracesFilterRes.query : ""}
${orderByToClickhouseSql(orderBy ?? null, sessionCols)}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const obsStartTimeValue = traceTimestampFilter
? convertDateToClickhouseDateTime(traceTimestampFilter.value)
: null;
const res = await queryClickhouse<T>({
query: query,
params: {
projectId,
limit: limit,
offset: limit && page ? limit * page : 0,
...tracesFilterRes.params,
...singleTraceFilter?.params,
...(obsStartTimeValue
? { observationsStartTime: obsStartTimeValue }
: {}),
return measureAndReturn({
operationName: "getSessionsTableGeneric",
projectId,
minStartTime: filter?.find(
(f) =>
f.column === "min_timestamp" &&
(f.operator === ">=" || f.operator === ">"),
)?.value as Date | undefined,
input: {
params: {
projectId,
limit: limit,
offset: limit && page ? limit * page : 0,
...tracesFilterRes.params,
...singleTraceFilter?.params,
...scoresFilterRes.params,
...(traceTimestampFilter
? {
observationsStartTime: convertDateToClickhouseDateTime(
traceTimestampFilter.value,
),
}
: {}),
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "sessions-table",
projectId,
operation_name: "getSessionsTableGeneric",
},
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "sessions-table",
projectId,
existingExecution: async (input) => {
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", "traces"),
params: input.params,
tags: { ...input.tags, experiment_amt: "original" },
clickhouseConfigs,
});
},
newExecution: async (input) => {
// Extract the timestamp from filter for AMT table selection
const fromTimestamp = filter?.find(
(f) =>
f.column === "min_timestamp" &&
(f.operator === ">=" || f.operator === ">"),
)?.value as Date | undefined;
const traceAmt = getTimeframesTracesAMT(fromTimestamp);
return queryClickhouse<T>({
query: query.replace("__TRACE_TABLE__", traceAmt),
params: input.params,
tags: { ...input.tags, experiment_amt: "new" },
clickhouseConfigs,
});
},
clickhouseConfigs,
});
return res;
};
@@ -21,7 +21,9 @@ import {
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
reduceUsageOrCostDetails,
getTimeframesTracesAMT,
} from "../repositories";
import { measureAndReturn } from "../clickhouse/measureAndReturn";
import { TracingSearchType } from "../../interfaces/search";
import { ObservationLevelType, TraceDomain } from "../../domain";
import { ClickHouseClientConfigOptions } from "@clickhouse/client";
@@ -217,54 +219,6 @@ async function getTracesTableGeneric(props: FetchTracesTableProps) {
clickhouseConfigs,
} = props;
let sqlSelect: string;
switch (select) {
case "count":
sqlSelect = "count(*) as count";
break;
case "metrics":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
os.latency_milliseconds / 1000 as latency,
os.cost_details as cost_details,
os.usage_details as usage_details,
os.aggregated_level as level,
os.error_count as error_count,
os.warning_count as warning_count,
os.default_count as default_count,
os.debug_count as debug_count,
os.observation_count as observation_count,
s.scores_avg as scores_avg,
s.score_categories as score_categories,
t.public as public`;
break;
case "rows":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
t.tags as tags,
t.bookmarked as bookmarked,
t.name as name,
t.release as release,
t.version as version,
t.user_id as user_id,
t.environment as environment,
t.session_id as session_id,
t.public as public`;
break;
case "identifiers":
sqlSelect = `
t.id as id,
t.project_id as projectId,
t.timestamp as timestamp`;
break;
default:
throw new Error(`Unknown select type: ${select}`);
}
const { tracesFilter, scoresFilter, observationsFilter } =
getProjectIdDefaultFilter(projectId, { tracesPrefix: "t" });
@@ -329,149 +283,319 @@ async function getTracesTableGeneric(props: FetchTracesTableProps) {
const scoresFilterRes = scoresFilter.apply();
const observationFilterRes = observationsFilter.apply();
const search = clickhouseSearchCondition(searchQuery, searchType, "t");
const defaultOrder = orderBy?.order && orderBy?.column === "timestamp";
const orderByCols = [
...tracesTableUiColumnDefinitions,
{
clickhouseSelect: "toDate(t.timestamp)",
uiTableName: "timestamp_to_date",
uiTableId: "timestamp_to_date",
clickhouseTableName: "traces",
},
{
clickhouseSelect: "t.event_ts",
uiTableName: "event_ts",
uiTableId: "event_ts",
clickhouseTableName: "traces",
},
];
const chOrderBy = orderByToClickhouseSql(
[
defaultOrder
? [
{
column: "timestamp_to_date",
order: orderBy.order,
},
{ column: "timestamp", order: orderBy.order },
{ column: "event_ts", order: "DESC" as "DESC" },
]
: null,
orderBy ?? null,
].flat(),
orderByCols,
);
// complex query ahead:
// - we only join scores and observations if we really need them to speed up default views
// - we use FINAL on traces only in case we not need to order by something different than time. Otherwise we cannot guarantee correct reads.
// - we filter the observations and scores as much as possible before joining them to traces.
// - we order by todate(timestamp), event_ts desc per default and do not use FINAL.
// In this case, CH is able to read the data only from the latest date from disk and filtering them in memory. No need to read all data e.g. for 1 month from disk.
const query = `
const observationsAndScoresCTE = `
WITH observations_stats AS (
SELECT
COUNT(*) AS observation_count,
sumMap(usage_details) as usage_details,
SUM(total_cost) AS total_cost,
date_diff('millisecond', least(min(start_time), min(end_time)), greatest(max(start_time), max(end_time))) as latency_milliseconds,
countIf(level = 'ERROR') as error_count,
countIf(level = 'WARNING') as warning_count,
countIf(level = 'DEFAULT') as default_count,
countIf(level = 'DEBUG') as debug_count,
multiIf(
arrayExists(x -> x = 'ERROR', groupArray(level)), 'ERROR',
arrayExists(x -> x = 'WARNING', groupArray(level)), 'WARNING',
arrayExists(x -> x = 'DEFAULT', groupArray(level)), 'DEFAULT',
'DEBUG'
) AS aggregated_level,
sumMap(cost_details) as cost_details,
trace_id,
project_id
FROM observations o FINAL
sumMap(usage_details) as usage_details,
SUM(total_cost) AS total_cost,
date_diff('millisecond', least(min(start_time), min(end_time)), greatest(max(start_time), max(end_time))) as latency_milliseconds,
countIf(level = 'ERROR') as error_count,
countIf(level = 'WARNING') as warning_count,
countIf(level = 'DEFAULT') as default_count,
countIf(level = 'DEBUG') as debug_count,
multiIf(
arrayExists(x -> x = 'ERROR', groupArray(level)), 'ERROR',
arrayExists(x -> x = 'WARNING', groupArray(level)), 'WARNING',
arrayExists(x -> x = 'DEFAULT', groupArray(level)), 'DEFAULT',
'DEBUG'
) AS aggregated_level,
sumMap(cost_details) as cost_details,
trace_id,
project_id
FROM observations o FINAL
WHERE o.project_id = {projectId: String}
${timeStampFilter ? `AND o.start_time >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
${observationsFilter ? `AND ${observationFilterRes.query}` : ""}
${timeStampFilter ? `AND o.start_time >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
${observationsFilter ? `AND ${observationFilterRes.query}` : ""}
GROUP BY trace_id, project_id
),
scores_avg AS (
SELECT
project_id,
trace_id,
-- For numeric scores, use tuples of (name, avg_value)
groupArrayIf(
tuple(name, avg_value),
data_type IN ('NUMERIC', 'BOOLEAN')
) AS scores_avg,
-- For categorical scores, use name:value format for improved query performance
groupArrayIf(
concat(name, ':', string_value),
data_type = 'CATEGORICAL' AND notEmpty(string_value)
) AS score_categories
FROM (
SELECT
project_id,
trace_id,
name,
data_type,
string_value,
avg(value) as avg_value
FROM scores s FINAL
WHERE
project_id = {projectId: String}
${timeStampFilter ? `AND s.timestamp >= {traceTimestamp: DateTime64(3)} - ${SCORE_TO_TRACE_OBSERVATIONS_INTERVAL}` : ""}
${scoresFilterRes ? `AND ${scoresFilterRes.query}` : ""}
GROUP BY
project_id,
trace_id,
name,
data_type,
string_value
) tmp
GROUP BY project_id, trace_id
)
SELECT ${sqlSelect}
-- FINAL is used for non default ordering and count.
FROM traces t ${["metrics", "rows", "identifiers"].includes(select) && defaultOrder ? "" : "FINAL"}
${select === "metrics" || requiresObservationsJoin ? `LEFT JOIN observations_stats os on os.project_id = t.project_id and os.trace_id = t.id` : ""}
${select === "metrics" || requiresScoresJoin ? `LEFT JOIN scores_avg s on s.project_id = t.project_id and s.trace_id = t.id` : ""}
WHERE t.project_id = {projectId: String}
${tracesFilterRes ? `AND ${tracesFilterRes.query}` : ""}
${search.query}
${chOrderBy}
-- This is used for metrics and row queries. Count has only one result.
-- This is only used for default ordering. Otherwise, we use final.
${["metrics", "rows", "identifiers"].includes(select) && defaultOrder ? "LIMIT 1 BY id, project_id" : ""}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
scores_avg AS (
SELECT
project_id,
trace_id,
-- For numeric scores, use tuples of (name, avg_value)
groupArrayIf(
tuple(name, avg_value),
data_type IN ('NUMERIC', 'BOOLEAN')
) AS scores_avg,
-- For categorical scores, use name:value format for improved query performance
groupArrayIf(
concat(name, ':', string_value),
data_type = 'CATEGORICAL' AND notEmpty(string_value)
) AS score_categories
FROM (
SELECT
project_id,
trace_id,
name,
data_type,
string_value,
avg(value) as avg_value
FROM scores s FINAL
WHERE
project_id = {projectId: String}
${timeStampFilter ? `AND s.timestamp >= {traceTimestamp: DateTime64(3)} - ${SCORE_TO_TRACE_OBSERVATIONS_INTERVAL}` : ""}
${scoresFilterRes ? `AND ${scoresFilterRes.query}` : ""}
GROUP BY
project_id,
trace_id,
name,
data_type,
string_value
) tmp
GROUP BY project_id, trace_id
)
`;
const res = await queryClickhouse<
SelectReturnTypeMap[keyof SelectReturnTypeMap]
>({
query: query,
params: {
limit: limit,
offset: limit && page ? limit * page : 0,
traceTimestamp: timeStampFilter?.value.getTime(),
projectId: projectId,
...tracesFilterRes.params,
...observationFilterRes.params,
...scoresFilterRes.params,
...search.params,
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "traces-table",
projectId,
},
clickhouseConfigs,
});
return measureAndReturn({
operationName: "getTracesTableGeneric",
projectId: props.projectId,
minStartTime: select !== "metrics" ? timeStampFilter?.value : undefined,
input: props,
existingExecution: async (props) => {
let sqlSelect: string;
switch (select) {
case "count":
sqlSelect = "count(*) as count";
break;
case "metrics":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
os.latency_milliseconds / 1000 as latency,
os.cost_details as cost_details,
os.usage_details as usage_details,
os.aggregated_level as level,
os.error_count as error_count,
os.warning_count as warning_count,
os.default_count as default_count,
os.debug_count as debug_count,
os.observation_count as observation_count,
s.scores_avg as scores_avg,
s.score_categories as score_categories,
t.public as public`;
break;
case "rows":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
t.tags as tags,
t.bookmarked as bookmarked,
t.name as name,
t.release as release,
t.version as version,
t.user_id as user_id,
t.environment as environment,
t.session_id as session_id,
t.public as public`;
break;
case "identifiers":
sqlSelect = `
t.id as id,
t.project_id as projectId,
t.timestamp as timestamp`;
break;
default:
throw new Error(`Unknown select type: ${select}`);
}
return res;
const search = clickhouseSearchCondition(searchQuery, searchType, "t");
const defaultOrder = orderBy?.order && orderBy?.column === "timestamp";
const orderByCols = [
...tracesTableUiColumnDefinitions,
{
clickhouseSelect: "toDate(t.timestamp)",
uiTableName: "timestamp_to_date",
uiTableId: "timestamp_to_date",
clickhouseTableName: "traces",
},
{
clickhouseSelect: "t.event_ts",
uiTableName: "event_ts",
uiTableId: "event_ts",
clickhouseTableName: "traces",
},
];
const chOrderBy = orderByToClickhouseSql(
[
defaultOrder
? [
{
column: "timestamp_to_date",
order: orderBy.order,
},
{ column: "timestamp", order: orderBy.order },
{ column: "event_ts", order: "DESC" as "DESC" },
]
: null,
orderBy ?? null,
].flat(),
orderByCols,
);
// complex query ahead:
// - we only join scores and observations if we really need them to speed up default views
// - we use FINAL on traces only in case we not need to order by something different than time. Otherwise we cannot guarantee correct reads.
// - we filter the observations and scores as much as possible before joining them to traces.
// - we order by todate(timestamp), event_ts desc per default and do not use FINAL.
// In this case, CH is able to read the data only from the latest date from disk and filtering them in memory. No need to read all data e.g. for 1 month from disk.
const query = `
${observationsAndScoresCTE}
SELECT ${sqlSelect}
-- FINAL is used for non default ordering and count.
FROM traces t ${["metrics", "rows", "identifiers"].includes(select) && defaultOrder ? "" : "FINAL"}
${select === "metrics" || requiresObservationsJoin ? `LEFT JOIN observations_stats os on os.project_id = t.project_id and os.trace_id = t.id` : ""}
${select === "metrics" || requiresScoresJoin ? `LEFT JOIN scores_avg s on s.project_id = t.project_id and s.trace_id = t.id` : ""}
WHERE t.project_id = {projectId: String}
${tracesFilterRes ? `AND ${tracesFilterRes.query}` : ""}
${search.query}
${chOrderBy}
-- This is used for metrics and row queries. Count has only one result.
-- This is only used for default ordering. Otherwise, we use final.
${["metrics", "rows", "identifiers"].includes(select) && defaultOrder ? "LIMIT 1 BY id, project_id" : ""}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const res = await queryClickhouse<
SelectReturnTypeMap[keyof SelectReturnTypeMap]
>({
query: query,
params: {
limit: limit,
offset: limit && page ? limit * page : 0,
traceTimestamp: timeStampFilter?.value.getTime(),
projectId: projectId,
...tracesFilterRes.params,
...observationFilterRes.params,
...scoresFilterRes.params,
...search.params,
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "traces-table",
projectId,
experiment_amt: "original",
operation_name: "getTracesTableGeneric",
},
clickhouseConfigs,
});
return res;
},
newExecution: async () => {
let sqlSelect: string;
switch (select) {
case "count":
sqlSelect = "count(*) as count";
break;
case "metrics":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
os.latency_milliseconds / 1000 as latency,
os.cost_details as cost_details,
os.usage_details as usage_details,
os.aggregated_level as level,
os.error_count as error_count,
os.warning_count as warning_count,
os.default_count as default_count,
os.debug_count as debug_count,
os.observation_count as observation_count,
s.scores_avg as scores_avg,
s.score_categories as score_categories,
t.public as public`;
break;
case "rows":
sqlSelect = `
t.id as id,
t.project_id as project_id,
t.timestamp as timestamp,
t.tags as tags,
finalizeAggregation(t.bookmarked) as bookmarked,
t.name as name,
t.release as release,
t.version as version,
t.user_id as user_id,
t.environment as environment,
t.session_id as session_id,
finalizeAggregation(t.public) as public`;
break;
case "identifiers":
sqlSelect = `
t.id as id,
t.project_id as projectId,
t.timestamp as timestamp`;
break;
default:
throw new Error(`Unknown select type: ${select}`);
}
const search = clickhouseSearchCondition(
searchQuery,
searchType,
"t",
true,
);
const chOrderBy = orderByToClickhouseSql(
[orderBy ?? null].flat(),
tracesTableUiColumnDefinitions,
);
const tracesAmt =
select === "metrics"
? "traces_all_amt"
: getTimeframesTracesAMT(timeStampFilter?.value);
const query = `
${observationsAndScoresCTE}
SELECT ${sqlSelect}
FROM ${tracesAmt} t FINAL
${select === "metrics" || requiresObservationsJoin ? `LEFT JOIN observations_stats os on os.project_id = t.project_id and os.trace_id = t.id` : ""}
${select === "metrics" || requiresScoresJoin ? `LEFT JOIN scores_avg s on s.project_id = t.project_id and s.trace_id = t.id` : ""}
WHERE t.project_id = {projectId: String}
${tracesFilterRes ? `AND ${tracesFilterRes.query}` : ""}
${search.query}
${chOrderBy}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const res = await queryClickhouse<
SelectReturnTypeMap[keyof SelectReturnTypeMap]
>({
query: query,
params: {
limit: limit,
offset: limit && page ? limit * page : 0,
traceTimestamp: timeStampFilter?.value.getTime(),
projectId: projectId,
...tracesFilterRes.params,
...observationFilterRes.params,
...scoresFilterRes.params,
...search.params,
},
tags: {
...(props.tags ?? {}),
feature: "tracing",
type: "traces-table",
projectId,
experiment_amt: "new",
operation_name: "getTracesTableGeneric",
},
clickhouseConfigs,
});
return res;
},
});
}
export const getTracesTableCount = async (props: {

Some files were not shown because too many files have changed in this diff Show More