Compare commits

...
288 Commits
Author SHA1 Message Date
Marc Klingen db5c575ae0 chore: release v2.93.2
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 17:53:55 +01:00
Marc Klingen 2a0f482578 fix: remove LANGFUSE_CSP_DISABLE (did not work) and disable csp headers on HF Spaces (#4545) 2024-12-03 17:51:50 +01:00
Marc Klingen c041cf371a chore: release v2.93.1
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 01:25:53 +01:00
Marc Klingen c6daf09cd2 [cherry-pick][2.93.x] chore: add LANGFUSE_CSP_DISABLE to .env.prod.example (#4533) 2024-12-03 01:25:04 +01:00
Marc KlingenandGitHub 7296e2e012 [cherry-pick][2.93.x] feat: optionally disable csp headers via LANGFUSE_CSP_DISABLE=true (#4532) 2024-12-03 01:22:11 +01:00
steffen911 43bf176ef7 chore: release v2.93.0 2024-11-26 10:04:45 +01:00
Steffen SchmitzandGitHub 6b48d7771c build: adjust docker actions to handle v2 branch (#4416)
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-11-26 09:59:57 +01:00
Max DeichmannandGitHub cd0d39b3c2 fix: fix bullmq api (#4425) 2024-11-25 23:18:16 +00:00
Max DeichmannandGitHub 8fcbcdd29d fix: order datasets API correctly (#4423) 2024-11-25 22:09:49 +00:00
Max DeichmannandGitHub e24f65c51d feat: add all queues to get api for bullmq api (#4422) 2024-11-25 21:53:27 +00:00
Marc KlingenandGitHub 38e6464219 chore: improve logging of cloud metering queue (#4420) 2024-11-25 21:22:13 +00:00
Steffen SchmitzandGitHub 86f5c885a1 chore: skip string parsing for metadata string values (#4415) 2024-11-25 18:04:49 +01:00
Hassieb PakzadandGitHub c0ebd06c0c feat(evaluators): add updates (#4360) 2024-11-25 16:51:46 +01:00
Steffen SchmitzandGitHub 0ce5ddcd00 feat: add azure storage provider support (#4386) 2024-11-25 14:57:40 +00:00
Steffen SchmitzandGitHub 7539af3bd9 chore: overwrite numerical type for second filters on observations table (#4412) 2024-11-25 14:39:04 +00:00
Steffen SchmitzandGitHub 60f9147873 fix: use larger numeric type for latency filters (#4385) 2024-11-25 13:53:27 +00:00
Steffen SchmitzandGitHub a594c142e0 chore: use clickhouse in evaluations queues (#4319) 2024-11-25 13:42:17 +00:00
Max DeichmannandGitHub 18905a9872 feat: add all scores to run items UI (#4409) 2024-11-25 14:31:11 +01:00
Hassieb PakzadandGitHub 27f5dd837d chore(costs): add chatgpt-4o-latest prices (#4407) 2024-11-25 13:55:58 +01:00
Max DeichmannandGitHub a921f9db94 feat: exclude env defined operations from clickhouse (#4405) 2024-11-25 12:56:12 +01:00
Marc Klingen b3cd940c73 chore(ui): remove ph badge (#4403) 2024-11-25 10:41:33 +01:00
Max DeichmannandGitHub f91cf1ca65 feat: refactor clickhouse filter builder for SDK APIs (#4402) 2024-11-24 22:45:29 +00:00
Max DeichmannandGitHub c438e89bd0 feat: support scores list endpoint from clickhouse (#4400) 2024-11-24 21:08:44 +00:00
Max DeichmannandGitHub 9b1746b7ed feat: fetch observations via api from clickhouse (#4395) 2024-11-24 09:48:17 +00:00
Steffen SchmitzandGitHub 0611a72c40 fix: allow special characters in openai tokenization (#4396)
* fix: allow special characters in openai tokenization

* chore: update test
2024-11-24 05:26:45 +00:00
Max DeichmannandGitHub 0ba66526e9 fix: do not use memory tables in clickhouse (#4394) 2024-11-23 18:21:30 +01:00
Max DeichmannandGitHub 425202dd1e fix: correctly retrieve scores for datasets (#4387) 2024-11-22 17:20:48 +00:00
Hassieb PakzadandGitHub 882ad83aad chore: upgrade langfuse-langchain (#4384) 2024-11-22 12:12:00 +01:00
Max DeichmannandGitHub 9b12783d08 fix: fix dataset run table to show all scores (#4383) 2024-11-22 10:44:23 +00:00
Marc KlingenandGitHub c6817ae32f chore: add Product Hunt Launch note (#4364) 2024-11-22 08:50:23 +01:00
marliessophieandGitHub 42711a8cc9 fix: don't show error message to user if trace is not yet available for dataset (#4382) 2024-11-22 08:27:54 +01:00
Marc KlingenandGitHub 0344ab2cf9 fix: experimentation should not use structured output (#4379) 2024-11-22 04:57:47 +01:00
Marc KlingenandGitHub df0d12af43 chore: rename default experiment tag langfuse-prompt-experiment (#4378) 2024-11-22 01:01:40 +00:00
marliessophieandGitHub d058ba8861 docs: add link to experimentation docs (#4371) 2024-11-21 22:42:36 +00:00
marliessophieandGitHub d1309905d0 chore: create dataset run items in order of dataset items timestamp (#4370)
* chore: create dataset run items in order of dataset items timestamp
2024-11-21 22:36:47 +01:00
Max DeichmannandGitHub ca6ac3912f fix: fix dataset run item api to always return traceid (#4367) 2024-11-21 20:25:30 +00:00
Max DeichmannandGitHub 9eabc32681 fix: return i/o from single object APIs (#4368) 2024-11-21 20:14:21 +00:00
marliessophieandGitHub 8798e4a499 feat(experiments): trigger experiments via UI (#4333) 2024-11-21 20:00:07 +01:00
Max Deichmann ea8b2da37c chore: release v2.92.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-21 18:44:30 +01:00
Max DeichmannandGitHub b7f7d10fea fix: fix dataset item api for self-hosters (#4366) 2024-11-21 17:43:48 +00:00
Marc KlingenandGitHub 0279c0f48e docs: add notification for lw 2 - day 4 (#4362) 2024-11-21 17:16:58 +01:00
Max DeichmannandGitHub c29cf3855f feat: SDK API for observations by id (#4295) 2024-11-21 15:28:34 +00:00
Max DeichmannandGitHub c73901369c fix: fix cost filters for traces table (#4325) 2024-11-21 14:21:01 +00:00
Hassieb PakzadandGitHub 0e70f03fc3 fix(media): lookup env var by exact string (#4356) 2024-11-21 10:48:33 +01:00
Max DeichmannandGitHub 74d1f7f717 feat: fetch single trace via API from clickhouse (#4344) 2024-11-21 10:38:24 +01:00
Marc KlingenandGitHub dfcf67184d docs: add lw2-3 info for multi-modal (#4351) 2024-11-20 23:15:19 +01:00
Max DeichmannandGitHub ce641d1e2b fix: do not fail on invalid prompt templates for evals (#4349) 2024-11-20 19:45:12 +00:00
Max DeichmannandGitHub 166b84a81a fix: do not use temp tables for dataset runs (#4347)
fix
2024-11-20 17:56:55 +00:00
Max DeichmannandGitHub f4097a1ae5 fix: dataset item api fix for large dataset runs (#4313) 2024-11-20 16:33:51 +00:00
Max DeichmannandGitHub 8374e889cf security: upgrade cross-spawn (#4339) 2024-11-20 13:32:19 +00:00
Max DeichmannandGitHub 62deaf7e00 sec: upgrade sentry (#4338) 2024-11-20 13:10:54 +00:00
Max DeichmannandGitHub a4cd68b1ad security: upgrade json path (#4335) 2024-11-20 13:32:01 +01:00
marliessophieandGitHub dde28f21fa chore: don't create traces in evaluation service (#4334) 2024-11-20 13:18:56 +01:00
Hassieb PakzadandGitHub a4c040b314 feat(core): add server side tracing of LLM invocations (#4317) 2024-11-20 12:00:31 +01:00
Max DeichmannandGitHub 87995625ca fix: correct traces pagination (#4329) 2024-11-20 02:31:34 +01:00
Marc KlingenandGitHub e717854909 docs: add launch week day 2 note (#4328) 2024-11-20 00:20:47 +01:00
Max DeichmannandGitHub 9e37f2cb17 feat: add more bullmq helpers (#4324) 2024-11-19 21:51:51 +00:00
Max DeichmannandGitHub fc08db08f3 fix: fix bullmq retry (#4323) 2024-11-19 21:24:55 +01:00
Max DeichmannandGitHub 270bc793ba feat: add service endpoint for failed bull events (#4321) 2024-11-19 21:18:40 +01:00
Max DeichmannandGitHub ced2d339ee fix: increase dataset item delay (#4318) 2024-11-19 17:46:49 +01:00
Max Deichmann 43acee5a29 chore: release v2.91.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-19 15:43:51 +01:00
Max DeichmannandGitHub 723098994b feat: move trace not found to 404 (#4270) 2024-11-19 14:40:22 +00:00
Max DeichmannandGitHub 684159abd1 chore: add logging to ingestion API (#4306) 2024-11-19 14:06:40 +00:00
Steffen SchmitzandGitHub dbb482553f chore: move invalid tokenizer errors to warning (#4311) 2024-11-19 13:21:53 +00:00
Hassieb PakzadandGitHub c99a43559d fix(model-prices): upsert if prices are missing (#4312) 2024-11-19 14:08:30 +01:00
Steffen SchmitzandGitHub 3cbe56861f chore: upgrade prisma to 5.22.0 (#4308) 2024-11-19 10:54:13 +00:00
Steffen SchmitzandGitHub 2ef9aad41c chore: only warn on upsert failures if traceSession exists (#4307) 2024-11-19 10:42:40 +00:00
marliessophieandGitHub 087f4a526b feat(datasets): show linked evaluators on dataset (#4277) 2024-11-19 09:40:17 +00:00
Steffen SchmitzandGitHub f876cf6e34 chore: make migrations self-host replica compatible (#4287) 2024-11-19 08:37:08 +00:00
Marc KlingenandGitHub 75fbd56c67 docs: add sidebar notification for dataset comparison view (#4300) 2024-11-19 01:23:39 +00:00
Max DeichmannandGitHub 0c326e56ba feat: fetch sessions by id for public API (#4292) 2024-11-18 22:39:30 +01:00
Max DeichmannandGitHub 2068ba7314 feat: fetch scores by id via API from clickhouse (#4291) 2024-11-18 20:33:05 +00:00
Max DeichmannandGitHub 9841973b16 chore: add web async test setup (#4294) 2024-11-18 21:05:32 +01:00
Max DeichmannandGitHub 1c44000933 perf: do not fetch models for UI (#4288) 2024-11-18 15:56:45 +00:00
Hassieb PakzadandGitHub f71e1aebed feat(media): add ui support (#4285) 2024-11-18 16:22:43 +01:00
marliessophieandGitHub 252fe184b6 feat(evals_on_datasets): remove feature flag (#4289) 2024-11-18 14:35:35 +00:00
marliessophieandGitHub 8034148313 fix(dataset_compare): show latency only if not null (#4282) 2024-11-18 11:41:33 +00:00
marliessophieandGitHub f6746e8f84 feat(datasets): add header to peek view (#4278)
* refactor: extract peek view

* feat: add consistent header view

* link to dataset item detail view
2024-11-18 10:46:05 +01:00
marliessophieandGitHub a48817300b feat(ui): peek view on dataset run compare table (#4131) 2024-11-18 08:28:18 +01:00
Max DeichmannandGitHub a74a820e32 fix: fix dataset runs api (#4275) 2024-11-17 21:49:11 +00:00
Max DeichmannandGitHub e0f9333d2f fix: fix dataset id api (#4274) 2024-11-17 17:54:21 +01:00
Max DeichmannandGitHub 34d4feae27 fix: fix dataset run items api (#4273) 2024-11-17 17:34:33 +01:00
Max DeichmannandGitHub f09ff1f070 feat: serve datasets from clickhouse (#4272) 2024-11-17 17:06:12 +01:00
Max DeichmannandGitHub 9a89c3f002 fix: fix dataset route (#4271) 2024-11-17 16:33:19 +01:00
Max DeichmannandGitHub f1e55b8ac7 feat: move datasets results to logs only (#4269) 2024-11-17 13:39:45 +00:00
Max DeichmannandGitHub d908fb13b3 feat: read dataset fetures from clickhouse (#4264) 2024-11-17 14:18:20 +01:00
Max DeichmannandGitHub 3c02e525bc fix: session id filter on session table (#4267) 2024-11-16 19:09:23 +01:00
Max DeichmannandGitHub 2cf04e4ab0 feat: configure to read from clickhouse only (#4266) 2024-11-16 13:02:01 +00:00
Max DeichmannandGitHub 24a549dc54 feat: move traces.hasany to clickhouse (#4265) 2024-11-16 12:51:25 +00:00
marliessophieandGitHub 63749970c6 chore(ui): enable all feature flags for admins (#4256) 2024-11-15 19:10:19 +01:00
Marc KlingenandGitHub 0d1b4f384a fix(cloud): usage indicator only on hobby plan (#4257) 2024-11-15 18:10:03 +01:00
marliessophieandGitHub 8517670ac2 feat: support evals triggered via dataset run (#4086) 2024-11-15 16:37:55 +01:00
Steffen SchmitzandGitHub 6157287549 chore: upsert tracesessions from clickhouse ingestion pipeline (#4254) 2024-11-15 14:43:39 +00:00
Max DeichmannandGitHub 041345ad3d perf: improve prompt metrics performance (#4251) 2024-11-15 14:28:08 +00:00
Hassieb PakzadandGitHub 7ed865cb4a chore: add media env to v3preview docker compose (#4250) 2024-11-15 14:32:52 +01:00
steffen911 b4d59960c0 chore: release v2.90.1
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-15 14:16:38 +01:00
Steffen SchmitzandGitHub 611b32e965 chore: add tzdata package (#4253) 2024-11-15 14:15:58 +01:00
Steffen SchmitzandGitHub fc94c02058 fix: fail on errors in database migrations (#4252) 2024-11-15 13:49:42 +01:00
Max Deichmann 853ca3c252 chore: release v2.90.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-15 11:25:56 +01:00
Steffen SchmitzandGitHub 74e378fe4d chore: fill clickhouse trace for observations with missing traceId (#4246) 2024-11-15 10:14:35 +00:00
Steffen SchmitzandGitHub a4d1288b00 chore: add note for future score create removal (#4248) 2024-11-15 11:03:59 +01:00
Steffen SchmitzandGitHub 91d27dd231 chore: add clickhouse migrations to web entrypoint (#4245) 2024-11-15 09:26:59 +01:00
Max DeichmannandGitHub 6d90635476 feat: exclude project_ids from clickhouse experimentation (#4242) 2024-11-14 21:47:01 +00:00
Max DeichmannandGitHub e9be67efb4 feat: read metrics for prompts table (#4224) 2024-11-14 20:39:47 +00:00
Max DeichmannandGitHub 71d1f2d8e6 feat: add reverse feature flag for clickhouse reads (#4241) 2024-11-14 18:43:38 +00:00
Marc KlingenandGitHub 56233d52eb feat(cloud): move hobby plan usage info to sidebar (#4240) 2024-11-14 18:06:21 +00:00
Max DeichmannandGitHub 5b5c3ba439 chore: improve clickhouse experimentation (#4239) 2024-11-14 18:23:02 +01:00
Steffen SchmitzandGitHub a0c6acf4e4 chore: add query parameter to healthcheck which enables failure on unavailable database (#4238) 2024-11-14 17:08:16 +00:00
Max DeichmannandGitHub bf54c123c1 fix: sessions metrics query (#4237) 2024-11-14 17:43:37 +01:00
Max DeichmannandGitHub 6a30fed1c6 fix: use todate for trace timestamps (#4235) 2024-11-14 15:30:54 +00:00
Steffen SchmitzandGitHub 8c2f8e6247 chore: reduce use of FINAL in clickhouse queries (#4233) 2024-11-14 16:16:34 +01:00
Steffen SchmitzandGitHub 593d1a087d chore: add timestamp filter for dashboard queries (#4234) 2024-11-14 15:10:39 +01:00
Hassieb PakzadandGitHub 633f76f64a feat(multimodal): add media upload / download endpoints (#3989) 2024-11-14 10:26:30 +01:00
Max DeichmannandGitHub ec89fabc5d fix: fix trace filter on score analytics charts (#4225) 2024-11-13 22:45:08 +00:00
Hassieb PakzadandGitHub 26c02ad49c fix(playground): jump to playground for clickhouse reads (#4222) 2024-11-13 20:10:01 +01:00
Max DeichmannandGitHub e1841e016f fix: fix order by in clickhouse (#4221) 2024-11-13 18:38:47 +01:00
Max DeichmannandGitHub 8ebfb3d8a2 fix: fix observations order by (#4220) 2024-11-13 17:25:05 +00:00
Max DeichmannandGitHub 4c0320c565 feat: fetch number of generations for prompts table form clickhouse (#4218) 2024-11-13 17:09:15 +00:00
Max DeichmannandGitHub 5c1f2202cd chore: add experimentation execution time metrics (#4219) 2024-11-13 17:55:03 +01:00
Steffen SchmitzandGitHub e1abbecabd chore: migrate users table to clickhouse (#4191) 2024-11-13 15:40:41 +00:00
Steffen SchmitzandGitHub 52466da955 chore: use event_ts instead of final for obs and score retrieval (#4214) 2024-11-13 15:19:13 +00:00
Max DeichmannandGitHub ede7ad49bf revert: "security: upgrade dd-trace (#4211)" (#4216) 2024-11-13 16:08:07 +01:00
Max DeichmannandGitHub c61acee8e8 security: upgrade dd-trace (#4211) 2024-11-13 13:41:45 +00:00
Steffen SchmitzandGitHub 7ea0a368af chore: handle clickhouse errors gracefully and correct formatting for getTraceById timestamp (#4210) 2024-11-13 13:10:32 +00:00
Steffen SchmitzandGitHub bb07b3508d chore: apply trace timestamp filter correctly on user model query (#4209) 2024-11-13 12:36:35 +00:00
Steffen SchmitzandGitHub 4df24cfde5 chore: remove superfluous isClickhouseAdminEligible checks (#4207) 2024-11-13 10:21:51 +01:00
Hassieb PakzadandGitHub 47c06a77bd fix(playground): disable stream_options for openai adapter (#4206) 2024-11-13 08:45:42 +00:00
Steffen SchmitzandGitHub 7a95b1e1da chore: migrate score analytics dashboard to clickhouse (#4196) 2024-11-13 08:18:05 +00:00
Max DeichmannandGitHub b09f9cbf5a chore: add tests for traces table UI (#4202) 2024-11-12 22:40:50 +00:00
Max Deichmann 780880cf07 chore: release v2.89.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-12 19:34:07 +01:00
Max DeichmannandGitHub eb6f93bd9c fix: fix public trace access (#4195) 2024-11-12 18:32:45 +00:00
Max DeichmannandGitHub a0d4a2e599 fix: fix indefinite table queries (#4194) 2024-11-12 17:04:23 +00:00
Max DeichmannandGitHub 8458443a2e chore: remove logs (#4193) 2024-11-12 16:37:49 +01:00
Max DeichmannandGitHub feafc1b7d3 feat: move scores table apis to experimentation setup (#4192) 2024-11-12 15:14:33 +00:00
Max DeichmannandGitHub fcd54180bc chore: centralize lookback guarantees (#4188) 2024-11-12 15:56:33 +01:00
Max DeichmannandGitHub d73c3ec8d3 fix: fix clickhouse filters (#4190) 2024-11-12 14:48:45 +00:00
Max DeichmannandGitHub ca7210b26f feat: use timestamp to link single traces from navigation (#4187) 2024-11-12 11:50:51 +00:00
Max DeichmannandGitHub d9e22324d4 feat: up/down detail navigation support params (#4186) 2024-11-12 12:13:44 +01:00
Max DeichmannandGitHub 0a18e59274 perf: add timestamp to traces table link (#4185) 2024-11-12 10:38:00 +00:00
Max DeichmannandGitHub ffc5f81fb5 chore: add error logs in trpc route (#4184) 2024-11-12 09:56:03 +00:00
marliessophieandGitHub b4668ed851 Revert "fix(dataset_runs): ensure consistent ordering to fetch score data for dataset run aggregation metrics table" (#4182)
Revert "fix(dataset_runs): ensure consistent ordering to fetch score data for…"

This reverts commit 2b2022ae85.
2024-11-12 09:20:12 +00:00
Steffen SchmitzandGitHub 999ff06c58 chore: migrate spammy log messages to debug level (#4181) 2024-11-12 07:41:01 +00:00
Steffen SchmitzandGitHub e494b2c215 chore: remove uninstantiated projectId from clickhouse seeder (#4180) 2024-11-12 06:56:42 +00:00
marliessophieandGitHub 2b2022ae85 fix(dataset_runs): ensure consistent ordering to fetch score data for dataset run aggregation metrics table (#4176) 2024-11-11 16:07:56 +00:00
Steffen SchmitzandGitHub 3afa137130 chore: update latency calculation for clickhouse dashboard queries (#4175) 2024-11-11 16:28:43 +01:00
Max DeichmannandGitHub 0b88ddf438 fix: correctly filter observations by generations for generations ui table (#4173) 2024-11-11 14:16:01 +00:00
Marc KlingenandGitHub f878bd1317 perf(cloud): add 60 min stale time for usage indicator (#4171) 2024-11-11 14:10:58 +01:00
Marc KlingenandGitHub 72075b29de fix(ui): width of date range time input (#4170) 2024-11-11 14:06:58 +01:00
Max DeichmannandGitHub edf1a0b879 fix: fix aggregated user consumption (#4167) 2024-11-11 11:21:03 +00:00
Max DeichmannandGitHub df763ff09c perf: improve performance for latency dashboard (#4164) 2024-11-11 10:23:49 +00:00
Max DeichmannandGitHub 83e09bcf47 chore: read dashboards form clickhouse flag (#4162) 2024-11-10 20:24:50 +00:00
Max DeichmannandGitHub 10e9aef044 feat: move sessions api to experimentation setup (#4161) 2024-11-10 20:09:59 +00:00
Max DeichmannandGitHub 45a0c2eb3d feat: move generations api to experimentation setup (#4160)
chore: move generations api to experimentation setup
2024-11-10 19:09:44 +00:00
Max DeichmannandGitHub 99c2a75964 perf: add time filter for sessions (#4159) 2024-11-10 18:49:11 +00:00
Max DeichmannandGitHub 9cc61eb851 perf: add timestamp to filteroptions api (#4158) 2024-11-10 18:35:06 +00:00
Max DeichmannandGitHub 62e51b8cf6 feat: add traces experimentation (#4157) 2024-11-10 17:45:31 +00:00
Max DeichmannandGitHub b147560a80 feat: steer reading from CH and PG (#4153) 2024-11-10 18:14:30 +01:00
Max DeichmannandGitHub 97a0b1f44c perf: increase traces metrics performance (#4155) 2024-11-10 17:08:00 +01:00
Max DeichmannandGitHub 8c1e6647e0 feat: query model latencies from clickhouse (#4152) 2024-11-10 14:37:56 +00:00
Max DeichmannandGitHub 8432327127 fix: traces aggregate trace filter (#4151) 2024-11-10 14:12:29 +00:00
Max DeichmannandGitHub 231f8e8f0c feat: read latency tables from clickhouse (#4150) 2024-11-10 13:47:58 +00:00
Max DeichmannandGitHub 28e43eb74c feat: query scores over time (#4149) 2024-11-10 12:57:30 +00:00
Steffen SchmitzandGitHub e2d3647d76 chore: set source: API if no source provided (#4148) 2024-11-10 13:41:01 +01:00
Max DeichmannandGitHub 876509d9e5 feat: add user charts (#4146) 2024-11-10 12:18:38 +00:00
Max DeichmannandGitHub 1647a79198 perf: scores dashboard performance improvement (#4144)
fix
2024-11-10 12:01:16 +00:00
Steffen SchmitzandGitHub 622d8a89a7 chore: update bookmarked, tags, and public for clickhouse traces (#4132) 2024-11-10 11:29:48 +00:00
Max DeichmannandGitHub d5e66e10d6 feat: add table search for clickhouse (#4142) 2024-11-09 15:55:11 +01:00
Max DeichmannandGitHub 5b1776710f chore: clean ups of clickhouse read logic (#4141) 2024-11-09 14:35:04 +01:00
Max DeichmannandGitHub 7e7e3b77d4 fix: correctly parse model params from clickhouse (#4140) 2024-11-09 12:48:45 +00:00
Max DeichmannandGitHub 28ecd9600a fix: fix latency calculations when reading from clickhouse (#4139) 2024-11-09 10:59:52 +00:00
Max DeichmannandGitHub 1ef6834a31 feat: delete traces from clickhouse from the UI (#4138) 2024-11-09 11:49:44 +01:00
Marc KlingenandGitHub b88f3bed40 feat(ui): add notification card in sidebar (#4137) 2024-11-09 00:30:36 +00:00
Steffen SchmitzandGitHub eaa885f5f5 chore: inject eval and annotation scores into clickhouse (#4081) 2024-11-08 17:42:41 +00:00
Max DeichmannandGitHub f7f48c2509 fix: reduce ingestion concurrency (#4130) 2024-11-08 18:18:40 +01:00
Steffen SchmitzandGitHub 07344e89b1 chore: move legacy api endpoints to async ingestion (#4109) 2024-11-08 18:00:59 +01:00
Max DeichmannandGitHub e6bc717612 perf: improve generations table on clickhouse (#4124) 2024-11-08 11:18:41 +00:00
Max DeichmannandGitHub 11521fc24d perf: improve generations performance (#4111) 2024-11-07 20:54:49 +00:00
Max DeichmannandGitHub 213ec6027d perf: add sessions index in clickhouse (#4108) 2024-11-07 18:09:41 +01:00
Max DeichmannandGitHub 259d832d37 feat: support sessions API from Clickhouse (#4065) 2024-11-07 16:52:07 +00:00
Marc KlingenandGitHub 62deb83520 chore(cloud): stripe checkout settings (#4107) 2024-11-07 16:34:07 +00:00
Max DeichmannandGitHub edc4cd7873 perf: query traces via timestamps (#4105) 2024-11-07 14:52:33 +01:00
Steffen SchmitzandGitHub 05b75fbee0 chore: merge observation backfill states in table (#4104) 2024-11-07 13:20:18 +00:00
Steffen SchmitzandGitHub d6065e8b89 chore: add stateSuffix to observation migration for potential parallelization (#4101) 2024-11-07 12:46:58 +00:00
Steffen SchmitzandGitHub 617d129542 chore: swap negative provided usage for null (#4100) 2024-11-07 12:29:44 +00:00
Max Deichmann 057424b269 chore: release v2.88.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-07 11:44:46 +01:00
Max DeichmannandGitHub 63d1b85a6c fix: pagination for clickhouse tables (#4098) 2024-11-07 11:41:46 +01:00
Max DeichmannandGitHub 9481c3b424 feat: build scores table for clickhouse (#4092) 2024-11-07 10:18:14 +00:00
Steffen SchmitzandGitHub 15e174fc2f chore: convert starttime filter to DateTime in Ingestion pipeline (#4095) 2024-11-07 09:51:35 +01:00
Steffen SchmitzandGitHub 5efd4ec0a4 chore: include backgroundmigration state default in schema.prisma (#4094) 2024-11-07 08:25:32 +00:00
marliessophieandGitHub 895b61fd62 fix(ui): render dataset run description and metadata on screen given table (#4089)
fix(ui): render dataset run description and metadata on screen without cutting off table
2024-11-06 19:47:01 +00:00
Steffen SchmitzandGitHub 5d4125208c chore: restrict domain timestamp updates to one day in clickhouse ingestion (#4085) 2024-11-06 17:27:25 +00:00
Steffen SchmitzandGitHub ff3bfe75ec chore: drop redundant observation project_id index on Clickhouse (#4084) 2024-11-06 16:45:53 +01:00
Steffen SchmitzandGitHub 89c31fdbc2 chore: add type filter to observation lookup in ingestion queue (#4083) 2024-11-06 15:18:19 +00:00
Steffen SchmitzandGitHub 9445c7ca78 chore: forward /api/public/scores to /api/public/ingestion (#4064) 2024-11-06 11:32:38 +01:00
marliessophieandGitHub b3d5833520 fix: unify ordering of dataset_items and dataset_run_items (#4063)
* fix: unify ordering of `dataset_items` and `dataset_run_items`

* fix: filter trace scores by observation id NULL
2024-11-05 19:03:30 +00:00
Steffen SchmitzandGitHub 20ed40510a chore: update cost calculation to handle null values from DB (#4057) 2024-11-05 18:22:10 +01:00
Steffen SchmitzandGitHub 9c5c57f134 chore: add test case for different input/output values (#4061) 2024-11-05 16:46:17 +00:00
Steffen SchmitzandGitHub 09b916aac9 chore: do not send auth errors to DLQ in bull (#4062) 2024-11-05 16:23:17 +00:00
marliessophieandGitHub 92843fef28 feat(datasets): view to compare different dataset runs (#3971) 2024-11-05 16:10:28 +01:00
Steffen SchmitzandGitHub 433951afda chore: retry legacy ingestion batches on errors (#4058) 2024-11-05 15:25:25 +01:00
Steffen SchmitzandGitHub aa1f861e4e chore: ignore all .env files aside from examples (#4059) 2024-11-05 13:25:27 +00:00
Hassieb PakzadandGitHub 66b6a09400 feat(models): add claude haiku 3.5 support (#4055) 2024-11-05 12:14:25 +01:00
Max DeichmannandGitHub edf0dc8399 feat: add metadata filter for all UI tables using clickhouse (#4022)
* feat: add metadata filter for all UI tables using clickhouse

* feat: add metadata filter for all UI tables using clickhouse

* push
2024-11-05 10:05:22 +00:00
Marc KlingenandGitHub 8d0c1219ca fix: env configuration for telemetry (#4052) 2024-11-04 23:04:07 +00:00
Max DeichmannandGitHub 33b3f69076 feat: query distinct models for charts (#4051) 2024-11-04 22:42:46 +01:00
Max DeichmannandGitHub f582941ad3 perf: fix timeseries clickouse (#4049) 2024-11-04 21:14:14 +00:00
Max DeichmannandGitHub 9fd57b4fbf feat: query second row of dashboards using clickhouse (#4048) 2024-11-04 20:18:37 +00:00
Max DeichmannandGitHub f61f69c238 perf: only join in dashboard if required (#4047) 2024-11-04 18:17:22 +00:00
Steffen SchmitzandGitHub 55c626d627 chore: reduce startTime missing logs to create events (#4044) 2024-11-04 14:52:09 +00:00
Max DeichmannandGitHub c6d3257916 feat: add first three dashboards in clickhouse (#4023) 2024-11-04 15:02:29 +01:00
Marc KlingenandGitHub 34f91e9872 chore(ui): slightly lighter bg of darkmode sidebar (#4042) 2024-11-04 15:01:58 +01:00
Steffen SchmitzandGitHub 5acd509fc0 chore: init migration scripts with state on restarts (#4039) 2024-11-04 13:37:28 +00:00
marliessophieandGitHub 765756b002 fix: scores tab in observation preview to only show scores liked to observation and trace id (#4030) 2024-11-04 12:54:21 +00:00
Steffen SchmitzandGitHub 64525744de chore: include usagedetails from postgres in ingestion pipeline (#4034) 2024-11-04 12:41:47 +00:00
Steffen SchmitzandGitHub 2b89f62add chore: remove temporary column from PG to CH migrations (#4031) 2024-11-04 13:26:42 +01:00
marliessophieandGitHub 0a67adf3e7 fix: show generations count on prompt metrics table (#4003) 2024-11-04 09:56:18 +00:00
marliessophieandGitHub 4ef7296e37 fix: multi-select header partially cut off on Firefox (#4029) 2024-11-04 09:45:18 +00:00
Max Deichmann 0cef46cd78 chore: release v2.87.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-03 18:34:51 +01:00
Max DeichmannandGitHub 243d72772c perf: improve select by id observations performance (#4021) 2024-11-03 16:45:43 +00:00
Max DeichmannandGitHub 91214cc6a6 feat: add order by to clickhouse tables (#4020) 2024-11-03 16:35:05 +00:00
Max DeichmannandGitHub da85e80689 feat: add generations table clickhouse support (#4015) 2024-11-03 14:48:10 +01:00
Max DeichmannandGitHub e22827d380 perf: improve traces metrics API performance (#4017) 2024-11-02 21:03:51 +00:00
Marc KlingenandGitHub 7161643574 docs: add star history to readme 2024-11-02 21:52:31 +01:00
Max DeichmannandGitHub 77f1c6cd0c fix: respect limit and offset for traces table query (#4016) 2024-11-02 19:38:06 +01:00
Max DeichmannandGitHub 0a4b56eb28 chore: improve clickhouse observability (#4014) 2024-11-02 17:50:26 +00:00
Max DeichmannandGitHub b5f94d78f3 fix: fix clickhouse untis and conversions (#4013) 2024-11-02 17:39:52 +00:00
Max DeichmannandGitHub a9fd1e1832 fix: fix timestamps in clickhouse reading and refactor clickhouse reads (#4011) 2024-11-02 15:53:37 +01:00
Max DeichmannandGitHub 762a7e3489 chore: fix fetching model for clickhouse reads (#4008) 2024-11-02 11:02:13 +00:00
Max DeichmannandGitHub 0f3717db0d fix: fix eval execution retries (#4010) 2024-11-02 10:22:19 +00:00
Max DeichmannandGitHub 8590ab84ba fix: fix traces table latency (#4005) 2024-11-01 16:49:48 +01:00
marliessophieandGitHub 9d17903392 fix: select all functionality on prompt metrics table (#4004) 2024-11-01 15:16:24 +00:00
Max DeichmannandGitHub 7d94831c01 feat: support for traces detail view with clickhouse (#3982) 2024-11-01 14:51:59 +00:00
Marc KlingenandGitHub b97900f5b8 perf(ui): do not refetch filter options on page mount (#4002) 2024-11-01 14:38:58 +00:00
marliessophieandGitHub 887947a967 fix: opening public langfuse urls (#3999)
* fix: opening public langfuse urls

* push
2024-11-01 14:14:07 +00:00
Max DeichmannandGitHub 85661c0e58 fix: remove final from ingestion (#3998) 2024-11-01 13:21:46 +00:00
fc2ba97792 feat(ui): add new sidebar and main navigation (#3837)
Co-authored-by: Marlies Mayerhofer <74332854+marliessophie@users.noreply.github.com>
2024-11-01 12:12:47 +01:00
Max DeichmannandGitHub 4b6cd3fd2b revert: revert clickhouse trace check (#3993) 2024-10-31 23:54:08 +01:00
Max DeichmannandGitHub d758821817 feat: support traces.countAll api with clickhouse (#3974) 2024-10-31 20:35:18 +00:00
Steffen SchmitzandGitHub 4685d0b1b6 chore: ensure model parameters are cast correctly in clickhouse ingestion (#3992) 2024-10-31 20:14:13 +00:00
Max DeichmannandGitHub e88946f0b5 fix: fix zod schema (#3991) 2024-10-31 19:45:56 +01:00
Steffen SchmitzandGitHub f14c611aac chore: correct number parsing from postgres in clickhouse merge (#3987) 2024-10-31 17:07:58 +00:00
Max DeichmannandGitHub 84ccdc24ed feat: log all project_ids with missing domain id (#3984) 2024-10-31 16:29:39 +00:00
Steffen SchmitzandGitHub 04a0366af5 chore: add fallback to event timestamp again, log warning (#3983) 2024-10-31 15:43:19 +00:00
Max DeichmannandGitHub 1a956c909c feat: add additional traces table filter (#3973) 2024-10-31 16:28:09 +01:00
Steffen SchmitzandGitHub 8e150701cb chore: join postgres state into Clickhouse results and add event_ts (#3978) 2024-10-31 15:44:26 +01:00
Max DeichmannandGitHub b9d8b5026e feat: add clickhouse support for traces filter options endpoint (#3969) 2024-10-30 21:05:31 +00:00
Max DeichmannandGitHub a6dfc58c72 feat: read traces and observations from clickhouse (#3810) 2024-10-30 21:32:08 +01:00
Steffen SchmitzandGitHub 6266cd8f0b chore: update traces and observations tables to use map type, full mapping script for migration (#3960) 2024-10-30 19:51:24 +00:00
Steffen SchmitzandGitHub e6e6328594 chore: remove remaining zod parses in worker pipeline (#3968) 2024-10-30 11:05:24 +00:00
Steffen SchmitzandGitHub 8b8826437a chore: remove zod parsing from worker (#3967) 2024-10-30 09:21:07 +00:00
marliessophieandGitHub b77c51512e feat(evals): option to update ref eval configs when creating new version (#3919) 2024-10-30 09:54:44 +01:00
marliessophieandGitHub 10d2299fae chore: global error handling for trpc routes (#3959)
* refactor: extract middleware for trpc routers to consistently throw trpc error

* chore: drop user-facing message in case of trpc error to not expose internals

* drop cause from error object
2024-10-29 19:02:59 +01:00
marliessophieandGitHub 83e1eb50bd Revert "chore(deps): bump react-hook-form from 7.51.5 to 7.53.0 (#3917)" (#3958)
Revert "chore(deps): bump react-hook-form from 7.51.5 to 7.53.0 (#3613)"
2024-10-29 14:24:26 +00:00
Hassieb PakzadandGitHub c7546bcafd fix(models): upsert prices on model drift; (#3955) 2024-10-29 14:39:10 +01:00
b26a537181 refactor: full-screen-page to drop magic in height calculation (#3944)
* fix: remove magic from full screen page

* fix: wrap prompts table as full page

* style: account for sticky header layout for mobile
---------

Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-10-29 10:34:45 +00:00
marliessophieandGitHub e5cbca7f95 style(evals): add table metadata view for eval config detail (#3916) 2024-10-29 09:18:54 +00:00
Max DeichmannandGitHub 73c7a554c2 chore: add external packages to next config (#3950) 2024-10-29 10:07:14 +01:00
Steffen SchmitzandGitHub c460029704 chore: add additional requirements for background migrations (#3951) 2024-10-29 08:45:20 +00:00
Max DeichmannandGitHub 2808157200 chore: add bullmq to external package (#3947) 2024-10-28 22:21:11 +00:00
Max DeichmannandGitHub 79913e3816 chore: adjust eval execution retries (#3946) 2024-10-28 22:58:31 +01:00
Marc KlingenandGitHub c9b79f4277 chore: pnpm dx pulls latest infra containers, waits for healthcheck, prunes old volumes (#3945) 2024-10-28 20:03:51 +01:00
Steffen SchmitzandGitHub 5897db3880 chore: add background migrations for long-running tasks (#3895) 2024-10-28 14:24:08 +00:00
Steffen SchmitzandGitHub 4a4d08d9f1 chore: create v3beta docker compose file (#3936) 2024-10-28 14:10:33 +00:00
marliessophieandGitHub f383a3f50f chore: support table-level-tab navigation through eval configs, templates, log (#3899) 2024-10-28 14:55:51 +01:00
Max DeichmannandGitHub 79954cf188 fix: treat api call failures for eval executions the same way as parsing errors. (#3942) 2024-10-28 12:59:54 +00:00
Max DeichmannandGitHub f87adc209a fix: do not fail eval queue on LLM Api errors (#3939) 2024-10-28 11:50:02 +00:00
marliessophieandGitHub 3a71bfd763 fix: populate default metadata when adding observation as dataset item (#3938) 2024-10-28 11:22:53 +00:00
Steffen SchmitzandGitHub c8ff61057e chore: remove hasCreateEvent check in ingestion pipeline (#3937) 2024-10-28 10:01:28 +00:00
Steffen SchmitzandGitHub 10d5975e26 chore: add is_deleted column to clickhouse seeder (#3934) 2024-10-28 08:14:03 +00:00
Steffen SchmitzandGitHub 7b34c6fc32 chore(deps): dependency updates (#3917) 2024-10-25 15:34:22 +02:00
Steffen SchmitzandGitHub 1ac6e51271 chore: validate clickhouse data quality (#3809) 2024-10-25 12:31:45 +00:00
Steffen SchmitzandGitHub 8435080ec7 chore: remove prod-eu-temp deploy environment (#3849) 2024-10-25 09:22:06 +00:00
Marc KlingenandGitHub 3b474e596d chore: faster healthcheck of minio in dev docker compose (#3914) 2024-10-25 11:11:45 +02:00
marliessophieandGitHub 16b40917c6 feat(evals): allow creating template from eval config form (#3896) 2024-10-25 08:39:00 +00:00
Hassieb Pakzad d0e5a6bfe3 chore(models): increase upsert diff tolerance 2024-10-24 19:48:52 +02:00
Hassieb Pakzad 0048dbfae2 docs(models): add note on how to update default models 2024-10-24 19:01:25 +02:00
Hassieb PakzadandGitHub bc612a8384 fix(models): allow for time tolerance in updateat diff (#3903) 2024-10-24 18:47:11 +02:00
Hassieb PakzadandGitHub 541765f8a0 feat(cost): make cost tracking flexible (#3811) 2024-10-24 18:12:38 +02:00
Marc KlingenandGitHub 7b4663e373 feat(cloud): add event usage metering (#3898)
push
2024-10-24 15:25:37 +02:00
marliessophieandGitHub 5eb792e2f5 fix(evals): show eval filters in eval-configs table (#3897) 2024-10-24 13:03:49 +00:00
Max DeichmannandGitHub 5499da8cc3 feat: add is_deleted to clickhouse (#3887) 2024-10-24 10:01:49 +00:00
Hassieb PakzadandGitHub 35b212954c fix(models): update sonnet model name (#3894) 2024-10-24 11:48:34 +02:00
Max DeichmannandGitHub 11c791ae38 fix: improve ingestion pipeline by id query (#3883) 2024-10-23 20:50:53 +02:00
Max DeichmannandGitHub 4ab1cbd03c perf: remove query cache in ingestion pipeline (#3882) 2024-10-23 15:48:34 +00:00
Max DeichmannandGitHub d33fb015d5 feat: adjust clickhouse writer (#3881) 2024-10-23 16:45:18 +02:00
Max DeichmannandGitHub 5b1704c95e perf: improve CH primary keys and order by (#3880) 2024-10-23 14:26:50 +00:00
Max DeichmannandGitHub 03659f3e82 feat: remove io from wide tables (#3878) 2024-10-23 15:45:28 +02:00
Max DeichmannandGitHub cf3b7f4603 chore: add clickhouse writer attributes (#3867) 2024-10-22 21:31:49 +00:00
413 changed files with 35722 additions and 11447 deletions
+92
View File
@@ -0,0 +1,92 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
# Clickhouse
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_MIGRATION_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
# openssl rand -base64 32
# https://next-auth.js.org/configuration/options#secret
# NEXTAUTH_SECRET=""
NEXTAUTH_URL="http://localhost:3000"
NEXTAUTH_SECRET="secret"
# Langfuse Cloud Environment
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
# Langfuse experimental features
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
# Salt for API key hashing
SALT="salt"
# Email
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# DON'T PANIC: The Azurite Secrets are well-known and meant to be hard-coded
# S3 storage
S3_ENDPOINT=http://localhost:10000/devstoreaccount1
S3_ACCESS_KEY_ID=devstoreaccount1
S3_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
S3_BUCKET_NAME=langfuse
S3_REGION=auto
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
LANGFUSE_S3_MEDIA_UPLOAD_REGION=auto
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
LANGFUSE_S3_EVENT_UPLOAD_REGION=auto
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
## Necessary for minio compatibility
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
LANGFUSE_USE_AZURE_BLOB=true
# Set during docker build of application
# Used to disable environment verification at build time
# DOCKER_BUILD=1
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+24 -4
View File
@@ -11,6 +11,7 @@ CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
@@ -37,15 +38,26 @@ SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
S3_ENDPOINT=http://localhost:9090
S3_ACCESS_KEY_ID=minio
S3_SECRET_ACCESS_KEY=miniosecret
S3_BUCKET_NAME=mybucket
S3_BUCKET_NAME=langfuse
S3_REGION=us-east-1
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=false
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=mybucket
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
@@ -66,4 +78,12 @@ REDIS_AUTH="myredissecret"
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+81
View File
@@ -0,0 +1,81 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
# Clickhouse
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_ENABLED="false"
# Next Auth
# You can generate a new secret on the command line with:
# openssl rand -base64 32
# https://next-auth.js.org/configuration/options#secret
# NEXTAUTH_SECRET=""
NEXTAUTH_URL="http://localhost:3000"
NEXTAUTH_SECRET="secret"
# Langfuse Cloud Environment
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
# Langfuse experimental features
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
# Salt for API key hashing
SALT="salt"
# Email
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# S3 storage
S3_ENDPOINT=http://localhost:9090
S3_ACCESS_KEY_ID=minio
S3_SECRET_ACCESS_KEY=miniosecret
S3_BUCKET_NAME=langfuse
S3_REGION=us-east-1
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# # S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
# Set during docker build of application
# Used to disable environment verification at build time
# DOCKER_BUILD=1
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
+1 -1
View File
@@ -237,7 +237,7 @@ OTEL_SERVICE_NAME="langfuse"
# CLICKHOUSE_PASSWORD=
# Ingestion
# LANGFUSE_INGESTION_QUEUE_DELAY_SECONDS=
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_BATCH_SIZE=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_MAX_ATTEMPTS=
+1 -2
View File
@@ -19,7 +19,6 @@ on:
type: choice
options:
- staging
- prod-eu-temp
- prod-eu
- prod-us
required: true
@@ -79,7 +78,7 @@ jobs:
return `["staging"]`
}
if (context.ref === "refs/heads/production") {
return `["prod-eu-temp", "prod-eu", "prod-us"]`
return `["prod-eu", "prod-us"]`
}
}
return "[]"
+110 -9
View File
@@ -66,10 +66,10 @@ jobs:
run: |
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
tests-web:
tests-web-sync:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-web (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
name: tests-web-sync (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
strategy:
matrix:
node-version: [20]
@@ -80,6 +80,11 @@ jobs:
with:
swap-size-gb: 10
- uses: actions/checkout@v4
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- uses: pnpm/action-setup@v3
with:
version: 9.5.0
@@ -100,7 +105,7 @@ jobs:
- name: Load default env
run: |
cp .env.dev.example .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.example > .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.legacy.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev.yml up -d
@@ -111,6 +116,76 @@ jobs:
- name: Seed DB
run: |
pnpm run db:migrate
pnpm --filter=shared ch:up
- name: Build
run: pnpm run build
- name: Start Langfuse
run: (pnpm run start&)
env:
LANGFUSE_INIT_ORG_ID: "seed-org-id"
LANGFUSE_INIT_ORG_NAME: "Seed Org"
LANGFUSE_INIT_PROJECT_ID: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
LANGFUSE_INIT_PROJECT_NAME: "Seed Project"
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: "pk-lf-1234567890"
LANGFUSE_INIT_PROJECT_SECRET_KEY: "sk-lf-1234567890"
LANGFUSE_INIT_USER_EMAIL: "demo@langfuse.com"
LANGFUSE_INIT_USER_NAME: "Demo User"
LANGFUSE_INIT_USER_PASSWORD: "password"
- name: run test-sync
run: pnpm --filter=web run test-sync
tests-web-async:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
node-version: [20]
postgres-version: [12, 15]
blob-provider: ["", "-azure"]
steps:
- name: Set Swap Space
uses: pierotofy/set-swap-space@master
with:
swap-size-gb: 10
- uses: actions/checkout@v4
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- uses: pnpm/action-setup@v3
with:
version: 9.5.0
- name: Login to Docker Hub
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
password: ${{ secrets.DOCKERHUB_TOKEN_READ }}
- name: Use Node.js ${{ matrix.node-version }}
uses: actions/setup-node@v4
with:
node-version: ${{ matrix.node-version }}
cache: "pnpm"
cache-dependency-path: "pnpm-lock.yaml"
- name: install dependencies
run: |
pnpm install
- name: Load default env
run: |
cp .env.dev${{ matrix.blob-provider }}.example .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
sleep 5 # Wait for PostgreSQL to accept connections
docker compose ps
env:
POSTGRES_VERSION: ${{ matrix.postgres-version }}
- name: Seed DB
run: |
pnpm run db:migrate
pnpm --filter=shared ch:up
- name: Build
run: pnpm run build
- name: Start Langfuse
@@ -131,11 +206,12 @@ jobs:
tests-worker:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
node-version: [20]
postgres-version: [12, 15]
blob-provider: ["", "-azure"]
steps:
- name: Set Swap Space
uses: pierotofy/set-swap-space@master
@@ -166,12 +242,12 @@ jobs:
which migrate
- name: Load default env
run: |
cp .env.dev.example .env
cp .env.dev.example web/.env
cp .env.dev.example worker/.env
cp .env.dev${{ matrix.blob-provider }}.example .env
cp .env.dev${{ matrix.blob-provider }}.example web/.env
cp .env.dev${{ matrix.blob-provider }}.example worker/.env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev.yml up -d
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
sleep 5 # Wait for PostgreSQL to accept connections
docker compose ps
- name: Ensure no unhealthy status
@@ -252,9 +328,15 @@ jobs:
- name: install dependencies
run: |
pnpm install
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- name: Load default env
run: |
cp .env.dev.example .env
echo "LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_ASYNC_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_CACHE_API_KEY_ENABLED=true" >> .env
echo "LANGFUSE_CACHE_PROMPT_ENABLED=true" >> .env
@@ -263,14 +345,29 @@ jobs:
docker compose -f docker-compose.dev.yml up -d
docker compose ps
sleep 5 # Wait for PostgreSQL to accept connections
- name: Ensure Docker dependencies are healthy
run: |
if docker-compose ps | grep "(unhealthy)"; then
echo "One or more services are unhealthy"
exit 1
else
echo "All services are healthy"
fi
- name: Seed DB
run: |
pnpm run db:migrate
pnpm --filter=shared run ch:up
pnpm run db:seed:examples
- name: Build
run: pnpm run build
- name: Run server
run: (pnpm run start&)
- name: Check worker health
run: |
timeout 10 bash -c 'until curl -f http://localhost:3030/api/health; do sleep 2; done'
- name: Check server health
run: |
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
- name: Run e2e tests
run: pnpm --filter=web run test:e2e:server
@@ -280,11 +377,12 @@ jobs:
needs:
[
lint,
tests-web,
tests-web-sync,
tests-worker,
e2e-tests,
test-docker-build,
e2e-server-tests,
tests-web-async,
]
if: always()
steps:
@@ -343,6 +441,8 @@ jobs:
images: |
ghcr.io/langfuse/langfuse # GitHub
langfuse/langfuse # Docker Hub
flavor: |
latest=false
tags: |
type=ref,event=branch
type=ref,event=pr
@@ -350,6 +450,7 @@ jobs:
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') }}
- name: Build and push Docker image (web)
uses: docker/build-push-action@v4
with:
+1 -1
View File
@@ -3,7 +3,7 @@ on:
push:
# Pattern matched against refs/tags
tags:
- "v[0-9]+.[0-9]+.[0-9]+" # Semantic version tags
- "v3.[0-9]+.[0-9]+" # Semantic version tags
jobs:
release:
+6 -3
View File
@@ -35,9 +35,12 @@ yarn-error.log*
# local env files
# do not commit any .env files to git, except for the .env.example file. https://create.t3.gg/en/usage/env-variables#using-environment-variables
.env
.env.local
.env*.local
.env*
!.env.dev.example
!.env.dev-azure.example
!.env.local.example
!.env.prod.example
!.env.dev.legacy.example
# vercel
.vercel
+1
View File
@@ -20,6 +20,7 @@
"prettier.documentSelectors": [
"**/*.{cjs,mjs,ts,tsx,astro,md,mdx,json,yaml,yml}"
],
"prettier.trailingComma": "all",
"mdx.experimentalLanguageServer": true,
"typescript.preferences.importModuleSpecifier": "non-relative",
"docwriter.style": "JSDoc",
+14
View File
@@ -432,6 +432,20 @@ Example:
op run --env-file="./.env" -- pnpm --filter=shared run db:deploy
```
### Editing default models and prices
You can update the default AI models and prices by adding or updating an entry in `worker/src/constants/default-model-prices.json`.
Please note that
- prices are in USD
- the list is ordered by ID, so make sure to keep this order
- the `updated_at` field must be updated with the current date in ISO 8601 format. Otherwise, the change will be ignored.
### Transition period until V3 release
Until the V3 release, both the JSON record must be updated **and** a migration must be created to continue supporting self-hosted users. Note that the migration must updated both the `models` as well as the `prices` table accordingly.
## License
Langfuse is MIT licensed, except for `ee/` folder. See [LICENSE](LICENSE) and [docs](https://langfuse.com/docs/open-source) for more details.
+10
View File
@@ -181,3 +181,13 @@ This helps us to:
None of the data is shared with third parties and does not include any sensitive information. We want to be super transparent about this and you can find the exact data we collect [here](/web/src/features/telemetry/index.ts).
You can opt-out by setting `TELEMETRY_ENABLED=false`.
### Star History
<a href="https://star-history.com/#langfuse/langfuse&Date">
<picture>
<source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date&theme=dark" />
<source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date" />
<img alt="Star History Chart" src="https://api.star-history.com/svg?repos=langfuse/langfuse&type=Date" />
</picture>
</a>
+63
View File
@@ -0,0 +1,63 @@
services:
clickhouse:
image: clickhouse/clickhouse-server
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
depends_on:
- postgres
azurite:
image: mcr.microsoft.com/azure-storage/azurite
container_name: azurite
command: azurite-blob --blobHost 0.0.0.0
ports:
- "10000:10000"
volumes:
- langfuse_azurite_data:/data
redis:
image: redis:7.2.4
restart: always
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
ports:
- 6379:6379
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
command: ["postgres", "-c", "log_statement=all"]
environment:
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=postgres
- POSTGRES_DB=postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_azurite_data:
driver: local
+10 -16
View File
@@ -20,7 +20,9 @@ services:
minio:
image: minio/minio
container_name: minio
command: server /data --console-address ":9001"
entrypoint: sh
# create the 'langfuse' bucket before starting the service
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
environment:
MINIO_ACCESS_KEY: minio
MINIO_SECRET_KEY: miniosecret
@@ -31,23 +33,10 @@ services:
- langfuse_minio_data:/data
healthcheck:
test: ["CMD", "mc", "ready", "local"]
interval: 10s
interval: 1s
timeout: 5s
retries: 5
start_period: 5s
miniocreatebucket:
image: minio/mc
container_name: miniocreatebucket
entrypoint: ["/bin/sh", "-c"]
command: >
"mc alias set minio http://minio:9000 minio miniosecret &&
mc rm -r --force minio/mybucket || true &&
mc mb minio/mybucket &&
mc policy set download minio/mybucket"
depends_on:
minio:
condition: service_healthy
start_period: 1s
redis:
image: redis:7.2.4
@@ -60,6 +49,11 @@ services:
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
command: ["postgres", "-c", "log_statement=all"]
environment:
- POSTGRES_USER=postgres
+145
View File
@@ -0,0 +1,145 @@
services:
langfuse-worker:
image: langfuse/langfuse-worker:latest
depends_on: &langfuse-depends-on
postgres:
condition: service_healthy
minio:
condition: service_healthy
redis:
condition: service_healthy
ports:
- "3030:3030"
environment: &langfuse-worker-env
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres
SALT: "mysalt"
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # generate via `openssl rand -hex 32`
TELEMETRY_ENABLED: ${TELEMETRY_ENABLED:-true}
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-true}
LANGFUSE_ASYNC_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_INGESTION_PROCESSING:-true}
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING:-true}
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE: ${LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE:-true}
LANGFUSE_READ_FROM_POSTGRES_ONLY: ${LANGFUSE_READ_FROM_POSTGRES_ONLY:-false}
LANGFUSE_RETURN_FROM_CLICKHOUSE: ${LANGFUSE_RETURN_FROM_CLICKHOUSE:-true}
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: ${LANGFUSE_S3_EVENT_UPLOAD_ENABLED:-true}
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED: ${LANGFUSE_S3_MEDIA_UPLOAD_ENABLED:-true}
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET: ${LANGFUSE_S3_MEDIA_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_MEDIA_UPLOAD_REGION: ${LANGFUSE_S3_MEDIA_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT: ${LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX: ${LANGFUSE_S3_MEDIA_UPLOAD_PREFIX:-media/}
REDIS_HOST: ${REDIS_HOST:-redis}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
langfuse-web:
image: langfuse/langfuse:latest
depends_on: *langfuse-depends-on
ports:
- "3000:3000"
environment:
<<: *langfuse-worker-env
NEXTAUTH_URL: http://localhost:3000
NEXTAUTH_SECRET: mysecret
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
LANGFUSE_INIT_PROJECT_ID: ${LANGFUSE_INIT_PROJECT_ID:-}
LANGFUSE_INIT_PROJECT_NAME: ${LANGFUSE_INIT_PROJECT_NAME:-}
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: ${LANGFUSE_INIT_PROJECT_PUBLIC_KEY:-}
LANGFUSE_INIT_PROJECT_SECRET_KEY: ${LANGFUSE_INIT_PROJECT_SECRET_KEY:-}
LANGFUSE_INIT_USER_EMAIL: ${LANGFUSE_INIT_USER_EMAIL:-}
LANGFUSE_INIT_USER_NAME: ${LANGFUSE_INIT_USER_NAME:-}
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
clickhouse:
image: clickhouse/clickhouse-server
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
depends_on:
- postgres
minio:
image: minio/minio
container_name: minio
entrypoint: sh
# create the 'langfuse' bucket before starting the service
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
environment:
MINIO_ROOT_USER: minio
MINIO_ROOT_PASSWORD: miniosecret
ports:
- "9090:9000"
- "9091:9001"
volumes:
- langfuse_minio_data:/data
healthcheck:
test: ["CMD", "mc", "ready", "local"]
interval: 1s
timeout: 5s
retries: 5
start_period: 1s
redis:
image: redis:7
restart: always
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
ports:
- 6379:6379
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 10s
retries: 10
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
environment:
POSTGRES_USER: postgres
POSTGRES_PASSWORD: postgres
POSTGRES_DB: postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_minio_data:
driver: local
+1 -1
View File
@@ -47,7 +47,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.0.7"
}
}
}
+103
View File
@@ -0,0 +1,103 @@
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
imports:
commons: ./commons.yml
service:
auth: true
base-path: /api/public
endpoints:
get:
docs: Get a media record
method: GET
path: /media/{mediaId}
path-parameters:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
response: GetMediaResponse
patch:
docs: Patch a media record
method: PATCH
path: /media/{mediaId}
path-parameters:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
request: PatchMediaBody
getUploadUrl:
docs: Get a presigned upload URL for a media record
method: POST
path: /media
request: GetMediaUploadUrlRequest
response: GetMediaUploadUrlResponse
types:
GetMediaResponse:
properties:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
contentType:
type: string
docs: The MIME type of the media record
contentLength:
type: integer
docs: The size of the media record in bytes
uploadedAt:
type: datetime
docs: The date and time when the media record was uploaded
url:
type: string
docs: The download URL of the media record
urlExpiry:
type: string
docs: The expiry date and time of the media record download URL
PatchMediaBody:
properties:
uploadedAt:
type: datetime
docs: The date and time when the media record was uploaded
uploadHttpStatus:
type: integer
docs: The HTTP status code of the upload
uploadHttpError:
type: optional<string>
docs: The HTTP error message of the upload
uploadTimeMs:
type: optional<integer>
docs: The time in milliseconds it took to upload the media record
GetMediaUploadUrlRequest:
properties:
traceId:
type: string
docs: The trace ID associated with the media record
observationId:
type: optional<string>
docs: The observation ID associated with the media record. If the media record is associated directly with a trace, this will be null.
contentType: MediaContentType
contentLength:
type: integer
docs: The size of the media record in bytes
sha256Hash:
type: string
docs: The SHA-256 hash of the media record
field:
type: string
docs: The trace / observation field the media record is associated with. This can be one of `input`, `output`, `metadata`
GetMediaUploadUrlResponse:
properties:
uploadUrl:
type: optional<string>
docs: The presigned upload URL. If the asset is already uploaded, this will be null
mediaId:
type: string
docs: The unique langfuse identifier of a media record
MediaContentType:
type: literal<"image/png","image/jpeg","image/jpg","image/webp","audio/mpeg","audio/mp3","audio/wav","text/plain","application/pdf">
docs: The MIME type of the media record
+21 -2
View File
@@ -147,10 +147,29 @@ types:
tags:
type: optional<list<string>>
docs: A list of tags associated with the trace referenced by score
GetScoresResponseData:
GetScoresResponseDataNumeric:
extends: commons.NumericScore
properties:
<<: commons.Score
trace: GetScoresResponseTraceData
GetScoresResponseDataCategorical:
extends: commons.CategoricalScore
properties:
trace: GetScoresResponseTraceData
GetScoresResponseDataBoolean:
extends: commons.BooleanScore
properties:
trace: GetScoresResponseTraceData
GetScoresResponseData:
discriminant: dataType
union:
NUMERIC: GetScoresResponseDataNumeric
CATEGORICAL: GetScoresResponseDataCategorical
BOOLEAN: GetScoresResponseDataBoolean
GetScoresResponse:
properties:
data: list<GetScoresResponseData>
+6 -5
View File
@@ -1,6 +1,6 @@
{
"name": "langfuse",
"version": "2.86.0",
"version": "2.93.2",
"author": "engineering@langfuse.com",
"license": "MIT",
"private": true,
@@ -9,15 +9,16 @@
},
"scripts": {
"preinstall": "npx only-allow pnpm",
"infra:dev:up": "docker compose -f ./docker-compose.dev.yml up -d",
"infra:dev:up": "docker compose -f ./docker-compose.dev.yml up -d --wait",
"infra:dev:down": "docker compose -f ./docker-compose.dev.yml down",
"infra:dev:prune": "docker compose -f ./docker-compose.dev.yml down -v",
"db:generate": "turbo run db:generate",
"db:migrate": "turbo run db:migrate",
"db:seed": "turbo run db:seed",
"db:seed:examples": "turbo run db:seed:examples",
"nuke": "bash ./scripts/nuke.sh",
"dx": "pnpm i && pnpm run infra:dev:up && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
"dx-f": "pnpm i && pnpm run infra:dev:up && pnpm --filter=shared run db:reset -f && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
"dx": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
"dx-f": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset -f && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
"dx:skip-infra": "pnpm i && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
"build": "turbo run build",
"start": "turbo run start",
@@ -82,7 +83,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.0.7"
}
},
"packageManager": "pnpm@9.5.0"
@@ -1,3 +0,0 @@
DROP TABLE IF EXISTS mv_observations_to_observations_wide;
DROP TABLE IF EXISTS mv_traces_to_observations_wide;
DROP TABLE IF EXISTS observations_wide;
@@ -1,175 +0,0 @@
CREATE TABLE observations_wide
(
`id` String,
trace_id Nullable(String),
`name` Nullable(String),
`project_id` String,
`user_id` Nullable(String),
`metadata` Map(String, String),
`release` Nullable(String),
`version` Nullable(String),
`public` Bool,
`bookmarked` Bool,
`tags` Array(String),
`input` Nullable(String),
`output` Nullable(String),
`session_id` Nullable(String),
`created_at` DateTime64(3),
`updated_at` DateTime64(3),
event_ts DateTime64(3),
`type` LowCardinality(String),
`parent_observation_id` Nullable(String),
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`level` LowCardinality(String),
`status_message` Nullable(String),
`provided_model_name` Nullable(String),
`internal_model_id` Nullable(String),
`model_parameters` Nullable(String),
`provided_input_usage_units` Nullable(Decimal64(12)),
`provided_output_usage_units` Nullable(Decimal64(12)),
`provided_total_usage_units` Nullable(Decimal64(12)),
`input_usage_units` Nullable(Decimal64(12)),
`output_usage_units` Nullable(Decimal64(12)),
`total_usage_units` Nullable(Decimal64(12)),
`unit` Nullable(String),
`provided_input_cost` Nullable(Decimal64(12)),
`provided_output_cost` Nullable(Decimal64(12)),
`provided_total_cost` Nullable(Decimal64(12)),
`input_cost` Nullable(Decimal64(12)),
`output_cost` Nullable(Decimal64(12)),
`total_cost` Nullable(Decimal64(12)),
`completion_start_time` Nullable(DateTime64(3)),
`prompt_id` Nullable(String),
`prompt_name` Nullable(String),
`prompt_version` Nullable(UInt16),
trace_timestamp DateTime64(3),
trace_name String,
trace_user_id Nullable(String),
trace_metadata Map(String, String),
trace_release Nullable(String),
trace_version Nullable(String),
trace_public Bool,
trace_bookmarked Bool,
trace_tags Array(String),
trace_input Nullable(String),
trace_output Nullable(String),
trace_session_id Nullable(String),
trace_event_ts DateTime64(3)
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(start_time)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
CREATE MATERIALIZED VIEW mv_traces_to_observations_wide TO observations_wide AS
SELECT
argMax(t.`name`, o.event_ts) as trace_name,
argMax(t.timestamp, o.event_ts) as trace_timestamp,
argMax(t.user_id, o.event_ts) as trace_user_id,
argMax(t.metadata, o.event_ts) as trace_metadata,
argMax(t.release, o.event_ts) as trace_release,
argMax(t.version, o.event_ts) as trace_version,
argMax(t.project_id, o.event_ts) as trace_project_id,
argMax(t.public, o.event_ts) as trace_public,
argMax(t.bookmarked, o.event_ts) as trace_bookmarked,
argMax(t.tags, o.event_ts) as trace_tags,
argMax(t.input, o.event_ts) as trace_input,
argMax(t.output, o.event_ts) as trace_output,
argMax(t.session_id, o.event_ts) as trace_session_id,
argMax(t.event_ts, o.event_ts) as trace_event_ts,
o.id as id,
argMax(o.trace_id, o.event_ts) as trace_id,
argMax(o.name, o.event_ts) as `name`,
o.project_id as project_id,
argMax(o.metadata, o.event_ts) as metadata,
argMax(o.type, o.event_ts) as type,
argMax(o.parent_observation_id, o.event_ts) as parent_observation_id,
argMax(o.start_time, o.event_ts) as start_time,
argMax(o.end_time, o.event_ts) as end_time,
argMax(o.level, o.event_ts) as level,
argMax(o.status_message, o.event_ts) as status_message,
argMax(o.provided_model_name, o.event_ts) as provided_model_name,
argMax(o.internal_model_id, o.event_ts) as internal_model_id,
argMax(o.model_parameters, o.event_ts) as model_parameters,
argMax(o.provided_input_usage_units, o.event_ts) as provided_input_usage_units,
argMax(o.provided_output_usage_units, o.event_ts) as provided_output_usage_units,
argMax(o.provided_total_usage_units, o.event_ts) as provided_total_usage_units,
argMax(o.input_usage_units, o.event_ts) as input_usage_units,
argMax(o.output_usage_units, o.event_ts) as output_usage_units,
argMax(o.total_usage_units, o.event_ts) as total_usage_units,
argMax(o.unit, o.event_ts) as unit,
argMax(o.provided_input_cost, o.event_ts) as provided_input_cost,
argMax(o.provided_output_cost, o.event_ts) as provided_output_cost,
argMax(o.provided_total_cost, o.event_ts) as provided_total_cost,
argMax(o.input_cost, o.event_ts) as input_cost,
argMax(o.output_cost, o.event_ts) as output_cost,
argMax(o.total_cost, o.event_ts) as total_cost,
argMax(o.completion_start_time, o.event_ts) as completion_start_time,
argMax(o.prompt_id, o.event_ts) as prompt_id,
argMax(o.prompt_name, o.event_ts) as prompt_name,
argMax(o.prompt_version, o.event_ts) as prompt_version,
argMax(o.created_at, o.event_ts) as created_at,
argMax(o.updated_at, o.event_ts) as updated_at,
argMax(o.event_ts, o.event_ts) as event_ts
FROM traces t
INNER JOIN observations o ON t.id = o.trace_id
GROUP BY o.id, o.project_id;
CREATE MATERIALIZED VIEW mv_observations_to_observations_wide TO observations_wide AS
SELECT
argMax(t.timestamp, o.event_ts) as trace_timestamp,
argMax(t.name, o.event_ts) as trace_name,
argMax(t.user_id, o.event_ts) as trace_user_id,
argMax(t.metadata, o.event_ts) as trace_metadata,
argMax(t.release, o.event_ts) as trace_release,
argMax(t.version, o.event_ts) as trace_version,
argMax(t.project_id, o.event_ts) as trace_project_id,
argMax(t.public, o.event_ts) as trace_public,
argMax(t.bookmarked, o.event_ts) as trace_bookmarked,
argMax(t.tags, o.event_ts) as trace_tags,
argMax(t.input, o.event_ts) as trace_input,
argMax(t.output, o.event_ts) as trace_output,
argMax(t.session_id, o.event_ts) as trace_session_id,
argMax(t.event_ts, o.event_ts) as trace_event_ts,
o.id as id,
argMax(o.trace_id, o.event_ts) as trace_id,
argMax(o.name, o.event_ts) as name,
o.project_id as project_id,
argMax(o.metadata, o.event_ts) as metadata,
argMax(o.type, o.event_ts) as type,
argMax(o.parent_observation_id, o.event_ts) as parent_observation_id,
argMax(o.start_time, o.event_ts) as start_time,
argMax(o.end_time, o.event_ts) as end_time,
argMax(o.level, o.event_ts) as level,
argMax(o.status_message, o.event_ts) as status_message,
argMax(o.provided_model_name, o.event_ts) as provided_model_name,
argMax(o.internal_model_id, o.event_ts) as internal_model_id,
argMax(o.model_parameters, o.event_ts) as model_parameters,
argMax(o.provided_input_usage_units, o.event_ts) as provided_input_usage_units,
argMax(o.provided_output_usage_units, o.event_ts) as provided_output_usage_units,
argMax(o.provided_total_usage_units, o.event_ts) as provided_total_usage_units,
argMax(o.input_usage_units, o.event_ts) as input_usage_units,
argMax(o.output_usage_units, o.event_ts) as output_usage_units,
argMax(o.total_usage_units, o.event_ts) as total_usage_units,
argMax(o.unit, o.event_ts) as unit,
argMax(o.provided_input_cost, o.event_ts) as provided_input_cost,
argMax(o.provided_output_cost, o.event_ts) as provided_output_cost,
argMax(o.provided_total_cost, o.event_ts) as provided_total_cost,
argMax(o.input_cost, o.event_ts) as input_cost,
argMax(o.output_cost, o.event_ts) as output_cost,
argMax(o.total_cost, o.event_ts) as total_cost,
argMax(o.completion_start_time, o.event_ts) as completion_start_time,
argMax(o.prompt_id, o.event_ts) as prompt_id,
argMax(o.prompt_name, o.event_ts) as prompt_name,
argMax(o.prompt_version, o.event_ts) as prompt_version,
argMax(o.created_at, o.event_ts) as created_at,
argMax(o.updated_at, o.event_ts) as updated_at,
argMax(o.event_ts, o.event_ts) as event_ts
FROM observations o
LEFT OUTER JOIN traces t ON t.id = o.trace_id
GROUP BY o.id, o.project_id;
@@ -1,3 +0,0 @@
DROP TABLE IF EXISTS mv_observations_to_traces_wide;
DROP TABLE IF EXISTS mv_traces_to_traces_wide;
DROP TABLE IF EXISTS traces_wide;
@@ -1,172 +0,0 @@
CREATE TABLE traces_wide
(
`id` String,
trace_id Nullable(String),
`name` Nullable(String),
`project_id` String,
`user_id` Nullable(String),
`metadata` Map(String, String),
`release` Nullable(String),
`version` Nullable(String),
`public` Bool,
`bookmarked` Bool,
`tags` Array(String),
`input` Nullable(String),
`output` Nullable(String),
`session_id` Nullable(String),
`created_at` DateTime64(3),
`updated_at` DateTime64(3),
event_ts DateTime64(3),
`type` LowCardinality(String),
`parent_observation_id` Nullable(String),
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`level` LowCardinality(String),
`status_message` Nullable(String),
`provided_model_name` Nullable(String),
`internal_model_id` Nullable(String),
`model_parameters` Nullable(String),
`provided_input_usage_units` Nullable(Decimal64(12)),
`provided_output_usage_units` Nullable(Decimal64(12)),
`provided_total_usage_units` Nullable(Decimal64(12)),
`input_usage_units` Nullable(Decimal64(12)),
`output_usage_units` Nullable(Decimal64(12)),
`total_usage_units` Nullable(Decimal64(12)),
`unit` Nullable(String),
`provided_input_cost` Nullable(Decimal64(12)),
`provided_output_cost` Nullable(Decimal64(12)),
`provided_total_cost` Nullable(Decimal64(12)),
`input_cost` Nullable(Decimal64(12)),
`output_cost` Nullable(Decimal64(12)),
`total_cost` Nullable(Decimal64(12)),
`completion_start_time` Nullable(DateTime64(3)),
`prompt_id` Nullable(String),
`prompt_name` Nullable(String),
`prompt_version` Nullable(UInt16),
trace_timestamp DateTime64(3),
trace_name String,
trace_user_id Nullable(String),
trace_metadata Map(String, String),
trace_release Nullable(String),
trace_version Nullable(String),
trace_public Bool,
trace_bookmarked Bool,
trace_tags Array(String),
trace_input Nullable(String),
trace_output Nullable(String),
trace_session_id Nullable(String),
trace_event_ts DateTime64(3)
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(start_time)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
CREATE MATERIALIZED VIEW mv_traces_to_traces_wide TO traces_wide AS
SELECT
argMax(t.`name`, t.event_ts) as trace_name,
argMax(t.timestamp, t.event_ts) as trace_timestamp,
argMax(t.user_id, t.event_ts) as trace_user_id,
argMax(t.metadata, t.event_ts) as trace_metadata,
argMax(t.release, t.event_ts) as trace_release,
argMax(t.version, t.event_ts) as trace_version,
argMax(t.public, t.event_ts) as trace_public,
argMax(t.bookmarked, t.event_ts) as trace_bookmarked,
argMax(t.tags, t.event_ts) as trace_tags,
argMax(t.input, t.event_ts) as trace_input,
argMax(t.output, t.event_ts) as trace_output,
argMax(t.session_id, t.event_ts) as trace_session_id,
argMax(t.event_ts, t.event_ts) as trace_event_ts,
o.id as id,
argMax(o.trace_id, t.event_ts) as trace_id,
argMax(o.name, t.event_ts) as `name`,
o.project_id as project_id,
argMax(o.metadata, t.event_ts) as metadata,
argMax(o.type, t.event_ts) as type,
argMax(o.parent_observation_id, t.event_ts) as parent_observation_id,
argMax(o.start_time, t.event_ts) as start_time,
argMax(o.end_time, t.event_ts) as end_time,
argMax(o.level, t.event_ts) as level,
argMax(o.status_message, t.event_ts) as status_message,
argMax(o.provided_model_name, t.event_ts) as provided_model_name,
argMax(o.internal_model_id, t.event_ts) as internal_model_id,
argMax(o.model_parameters, t.event_ts) as model_parameters,
argMax(o.provided_input_usage_units, t.event_ts) as provided_input_usage_units,
argMax(o.provided_output_usage_units, t.event_ts) as provided_output_usage_units,
argMax(o.provided_total_usage_units, t.event_ts) as provided_total_usage_units,
argMax(o.input_usage_units, t.event_ts) as input_usage_units,
argMax(o.output_usage_units, t.event_ts) as output_usage_units,
argMax(o.total_usage_units, t.event_ts) as total_usage_units,
argMax(o.unit, t.event_ts) as unit,
argMax(o.provided_input_cost, t.event_ts) as provided_input_cost,
argMax(o.provided_output_cost, t.event_ts) as provided_output_cost,
argMax(o.provided_total_cost, t.event_ts) as provided_total_cost,
argMax(o.input_cost, t.event_ts) as input_cost,
argMax(o.output_cost, t.event_ts) as output_cost,
argMax(o.total_cost, t.event_ts) as total_cost,
argMax(o.completion_start_time, t.event_ts) as completion_start_time,
argMax(o.prompt_id, t.event_ts) as prompt_id,
argMax(o.prompt_name, t.event_ts) as prompt_name,
argMax(o.prompt_version, t.event_ts) as prompt_version,
argMax(o.created_at, t.event_ts) as created_at,
argMax(o.updated_at, t.event_ts) as updated_at,
argMax(o.event_ts, t.event_ts) as event_ts
FROM traces t
LEFT JOIN observations o ON t.id = o.trace_id
GROUP BY o.id, o.project_id;
CREATE MATERIALIZED VIEW mv_observations_to_traces_wide TO traces_wide AS
SELECT
argMax(t.timestamp, o.event_ts) as trace_timestamp,
argMax(t.name, o.event_ts) as trace_name,
argMax(t.user_id, o.event_ts) as trace_user_id,
argMax(t.metadata, o.event_ts) as trace_metadata,
argMax(t.release, o.event_ts) as trace_release,
argMax(t.version, o.event_ts) as trace_version,
argMax(t.public, o.event_ts) as trace_public,
argMax(t.bookmarked, o.event_ts) as trace_bookmarked,
argMax(t.tags, o.event_ts) as trace_tags,
argMax(t.input, o.event_ts) as trace_input,
argMax(t.output, o.event_ts) as trace_output,
argMax(t.session_id, o.event_ts) as trace_session_id,
argMax(t.event_ts, o.event_ts) as trace_event_ts,
o.id as id,
argMax(o.trace_id, o.event_ts) as trace_id,
argMax(o.name, o.event_ts) as name,
o.project_id as project_id,
argMax(o.metadata, o.event_ts) as metadata,
argMax(o.type, o.event_ts) as type,
argMax(o.parent_observation_id, o.event_ts) as parent_observation_id,
argMax(o.start_time, o.event_ts) as start_time,
argMax(o.end_time, o.event_ts) as end_time,
argMax(o.level, o.event_ts) as level,
argMax(o.status_message, o.event_ts) as status_message,
argMax(o.provided_model_name, o.event_ts) as provided_model_name,
argMax(o.internal_model_id, o.event_ts) as internal_model_id,
argMax(o.model_parameters, o.event_ts) as model_parameters,
argMax(o.provided_input_usage_units, o.event_ts) as provided_input_usage_units,
argMax(o.provided_output_usage_units, o.event_ts) as provided_output_usage_units,
argMax(o.provided_total_usage_units, o.event_ts) as provided_total_usage_units,
argMax(o.input_usage_units, o.event_ts) as input_usage_units,
argMax(o.output_usage_units, o.event_ts) as output_usage_units,
argMax(o.total_usage_units, o.event_ts) as total_usage_units,
argMax(o.unit, o.event_ts) as unit,
argMax(o.provided_input_cost, o.event_ts) as provided_input_cost,
argMax(o.provided_output_cost, o.event_ts) as provided_output_cost,
argMax(o.provided_total_cost, o.event_ts) as provided_total_cost,
argMax(o.input_cost, o.event_ts) as input_cost,
argMax(o.output_cost, o.event_ts) as output_cost,
argMax(o.total_cost, o.event_ts) as total_cost,
argMax(o.completion_start_time, o.event_ts) as completion_start_time,
argMax(o.prompt_id, o.event_ts) as prompt_id,
argMax(o.prompt_name, o.event_ts) as prompt_name,
argMax(o.prompt_version, o.event_ts) as prompt_version,
argMax(o.created_at, o.event_ts) as created_at,
argMax(o.updated_at, o.event_ts) as updated_at,
argMax(o.event_ts, o.event_ts) as event_ts
FROM observations o
INNER JOIN traces t ON t.id = o.trace_id
WHERE t.id IS NOT NULL
GROUP BY o.id, o.project_id;
@@ -0,0 +1 @@
DROP TABLE traces ON CLUSTER default;
@@ -0,0 +1,32 @@
CREATE TABLE traces ON CLUSTER default (
`id` String,
`timestamp` DateTime64(3),
`name` String,
`user_id` Nullable(String),
`metadata` Map(LowCardinality(String), String),
`release` Nullable(String),
`version` Nullable(String),
`project_id` String,
`public` Bool,
`bookmarked` Bool,
`tags` Array(String),
`input` Nullable(String) CODEC(ZSTD(3)),
`output` Nullable(String) CODEC(ZSTD(3)),
`session_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
updated_at DateTime64(3) DEFAULT now(),
`event_ts` DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter(0.01) GRANULARITY 1,
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp)
)
ORDER BY (
project_id,
toDate(timestamp),
id
);
@@ -0,0 +1 @@
DROP TABLE observations ON CLUSTER default;
@@ -0,0 +1,47 @@
CREATE TABLE observations ON CLUSTER default (
`id` String,
`trace_id` String,
`project_id` String,
`type` LowCardinality(String),
`parent_observation_id` Nullable(String),
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`name` String,
`metadata` Map(LowCardinality(String), String),
`level` LowCardinality(String),
`status_message` Nullable(String),
`version` Nullable(String),
`input` Nullable(String) CODEC(ZSTD(3)),
`output` Nullable(String) CODEC(ZSTD(3)),
`provided_model_name` Nullable(String),
`internal_model_id` Nullable(String),
`model_parameters` Nullable(String),
`provided_usage_details` Map(LowCardinality(String), UInt64),
`usage_details` Map(LowCardinality(String), UInt64),
`provided_cost_details` Map(LowCardinality(String), Decimal64(12)),
`cost_details` Map(LowCardinality(String), Decimal64(12)),
`total_cost` Nullable(Decimal64(12)),
`completion_start_time` Nullable(DateTime64(3)),
`prompt_id` Nullable(String),
`prompt_name` Nullable(String),
`prompt_version` Nullable(UInt16),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
is_deleted UInt8,
INDEX idx_id id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_trace_id trace_id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
PRIMARY KEY (
project_id,
`type`,
toDate(start_time)
)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
@@ -0,0 +1 @@
DROP TABLE scores ON CLUSTER default;
@@ -0,0 +1,33 @@
CREATE TABLE scores ON CLUSTER default (
`id` String,
`timestamp` DateTime64(3),
`project_id` String,
`trace_id` String,
`observation_id` Nullable(String),
`name` String,
`value` Float64,
`source` String,
`comment` Nullable(String) CODEC(ZSTD(1)),
`author_user_id` Nullable(String),
`config_id` Nullable(String),
`data_type` String,
`string_value` Nullable(String),
`queue_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp),
name
)
ORDER BY (
project_id,
toDate(timestamp),
name,
id
)
@@ -0,0 +1 @@
ALTER TABLE observations ON CLUSTER default ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
@@ -0,0 +1 @@
ALTER TABLE observations ON CLUSTER default DROP INDEX IF EXISTS idx_project_id;
@@ -0,0 +1 @@
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_session_id;
@@ -0,0 +1,2 @@
ALTER TABLE traces ON CLUSTER default ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces ON CLUSTER default MATERIALIZE INDEX IF EXISTS idx_session_id;
@@ -3,25 +3,30 @@ CREATE TABLE traces (
`timestamp` DateTime64(3),
`name` String,
`user_id` Nullable(String),
`metadata` Map(String, String) CODEC(ZSTD(1)),
`metadata` Map(LowCardinality(String), String),
`release` Nullable(String),
`version` Nullable(String),
`project_id` String,
`public` Bool,
`bookmarked` Bool,
`tags` Array(String),
`input` Nullable(String) CODEC(ZSTD(1)),
`output` Nullable(String) CODEC(ZSTD(1)),
`input` Nullable(String) CODEC(ZSTD(3)),
`output` Nullable(String) CODEC(ZSTD(3)),
`session_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
updated_at DateTime64(3) DEFAULT now(),
`event_ts` DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter(0.01) GRANULARITY 1,
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(timestamp)
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp)
)
ORDER BY (
project_id,
toDate(timestamp),
id
);
project_id,
toDate(timestamp),
id
);
@@ -7,7 +7,7 @@ CREATE TABLE observations (
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`name` String,
`metadata` Map(LowCardinality(String), String) CODEC(ZSTD(1)),
`metadata` Map(LowCardinality(String), String),
`level` LowCardinality(String),
`status_message` Nullable(String),
`version` Nullable(String),
@@ -16,18 +16,10 @@ CREATE TABLE observations (
`provided_model_name` Nullable(String),
`internal_model_id` Nullable(String),
`model_parameters` Nullable(String),
`provided_input_usage_units` Nullable(Decimal64(12)),
`provided_output_usage_units` Nullable(Decimal64(12)),
`provided_total_usage_units` Nullable(Decimal64(12)),
`input_usage_units` Nullable(Decimal64(12)),
`output_usage_units` Nullable(Decimal64(12)),
`total_usage_units` Nullable(Decimal64(12)),
`unit` Nullable(String),
`provided_input_cost` Nullable(Decimal64(12)),
`provided_output_cost` Nullable(Decimal64(12)),
`provided_total_cost` Nullable(Decimal64(12)),
`input_cost` Nullable(Decimal64(12)),
`output_cost` Nullable(Decimal64(12)),
`provided_usage_details` Map(LowCardinality(String), UInt64),
`usage_details` Map(LowCardinality(String), UInt64),
`provided_cost_details` Map(LowCardinality(String), Decimal64(12)),
`cost_details` Map(LowCardinality(String), Decimal64(12)),
`total_cost` Nullable(Decimal64(12)),
`completion_start_time` Nullable(DateTime64(3)),
`prompt_id` Nullable(String),
@@ -36,16 +28,20 @@ CREATE TABLE observations (
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
is_deleted UInt8,
INDEX idx_id id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_trace_id trace_id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter() GRANULARITY 1,
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter() GRANULARITY 1
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(start_time)
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
PRIMARY KEY (
project_id,
`type`,
toDate(start_time)
)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
project_id,
`type`,
toDate(start_time),
id
);
@@ -12,15 +12,22 @@ CREATE TABLE scores (
`config_id` Nullable(String),
`data_type` String,
`string_value` Nullable(String),
`queue_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
) ENGINE = ReplacingMergeTree Partition by toYYYYMM(timestamp)
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp),
name
)
ORDER BY (
project_id,
toDate(timestamp),
name,
id
)
project_id,
toDate(timestamp),
name,
id
)
@@ -0,0 +1 @@
ALTER TABLE observations ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
@@ -0,0 +1 @@
ALTER TABLE observations DROP INDEX IF EXISTS idx_project_id;
@@ -0,0 +1 @@
ALTER TABLE traces DROP INDEX IF EXISTS idx_session_id;
@@ -0,0 +1,2 @@
ALTER TABLE traces ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces MATERIALIZE INDEX IF EXISTS idx_session_id;
+25 -7
View File
@@ -1,7 +1,13 @@
#!/bin/bash
# Load environment variables
source ../../.env
[ -f ../../.env ] && source ../../.env
# Check if CLICKHOUSE_URL is configured
if [ -z "${CLICKHOUSE_URL}" ]; then
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
exit 0
fi
# Check if golang-migrate is installed
if ! command -v migrate &> /dev/null
@@ -13,10 +19,22 @@ then
fi
# Construct the database URL
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the down command
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" down
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
fi
+55 -8
View File
@@ -1,24 +1,71 @@
import { clickhouseClient } from "@langfuse/shared/src/server";
import { prisma } from "../../src/db";
import {
clickhouseClient,
ObservationRecordReadType,
} from "@langfuse/shared/src/server";
import { Prisma, prisma } from "../../src/db";
import { redis } from "@langfuse/shared/src/server";
import { prepareClickhouse } from "../../scripts/prepareClickhouse";
import { createDatasets } from "../../prisma/seed";
import { queryClickhouse } from "../../src/server/repositories/clickhouse";
import { convertObservation } from "../../src/server/repositories/observations_converters";
async function main() {
try {
const projectIds = [
"7a88fb47-b4e2-43b8-a06c-a5ce950dc53a",
"239ad00f-562f-411d-af14-831c75ddd875",
]; // Example project IDs
const projectIds = ["7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"]; // Example project IDs
if (
await prisma.project.findFirst({
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
})
) {
projectIds.push("239ad00f-562f-411d-af14-831c75ddd875");
}
await prepareClickhouse(projectIds, {
numberOfDays: 3,
totalObservations: 1000,
totalObservations: 10000,
});
const project1 = await prisma.project.findFirst({
where: { id: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a" },
});
const project2 =
projectIds.length > 1
? await prisma.project.findFirst({
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
})
: await prisma.project.findFirst();
const query = `
SELECT *
FROM observations o
WHERE o.project_id IN ({projectIds: Array(String)})
LIMIT 2000;
`;
const res = await queryClickhouse<ObservationRecordReadType>({
query,
params: {
projectIds,
},
});
await createDatasets(
project1!,
project2!,
(await Promise.all(res.map(convertObservation))).map((o) => ({
...o,
metadata: {},
modelParameters: {},
input: {},
output: {},
})),
);
console.log("Clickhouse preparation completed successfully.");
} catch (error) {
console.error("Error during Clickhouse preparation:", error);
} finally {
await clickhouseClient.close();
await clickhouseClient().close();
await prisma.$disconnect();
redis?.disconnect();
console.log("Disconnected from Clickhouse.");
+25 -8
View File
@@ -1,7 +1,13 @@
#!/bin/bash
# Load environment variables
source ../../.env
[ -f ../../.env ] && source ../../.env
# Check if CLICKHOUSE_URL is configured
if [ -z "${CLICKHOUSE_URL}" ]; then
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
exit 0
fi
# Check if golang-migrate is installed
if ! command -v migrate &> /dev/null
@@ -13,11 +19,22 @@ then
fi
# Construct the database URL
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" up
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" up
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" up
fi
+13 -12
View File
@@ -43,7 +43,6 @@
"db:generate": "dotenv -e ../../.env -- npx prisma generate",
"db:seed:examples": "dotenv -e ../../.env -- npx prisma db seed -- --environment examples",
"db:seed:load": "dotenv -e ../../.env -- npx prisma db seed -- --environment load",
"ch:status": "dotenv -e ../../.env -- goose -dir './clickhouse/migrations/' status",
"ch:up": "bash clickhouse/scripts/up.sh",
"ch:down": "bash clickhouse/scripts/down.sh",
"ch:drop": "bash clickhouse/scripts/drop.sh",
@@ -59,14 +58,15 @@
"@aws-sdk/client-cloudwatch": "^3.675.0",
"@aws-sdk/client-s3": "^3.675.0",
"@aws-sdk/lib-storage": "^3.675.0",
"@aws-sdk/s3-request-presigner": "^3.675.0",
"@aws-sdk/s3-request-presigner": "^3.679.0",
"@azure/storage-blob": "^12.26.0",
"@clickhouse/client": "^1.4.0",
"@langchain/anthropic": "^0.3.1",
"@langchain/aws": "^0.1.0",
"@langchain/core": "^0.3.9",
"@langchain/openai": "^0.3.0",
"@langchain/anthropic": "^0.3.8",
"@langchain/aws": "^0.1.2",
"@langchain/core": "^0.3.18",
"@langchain/openai": "^0.3.14",
"@opentelemetry/api": ">=1.0.0 <1.10.0",
"@prisma/client": "^5.20.0",
"@prisma/client": "^5.22.0",
"@react-email/components": "^0.0.19",
"@react-email/render": "^0.0.15",
"@types/bcryptjs": "^2.4.6",
@@ -78,15 +78,16 @@
"exponential-backoff": "^3.1.1",
"ioredis": "^5.4.1",
"kysely": "^0.27.4",
"langchain": "^0.3.2",
"langchain": "^0.3.6",
"langfuse-langchain": "3.30.3",
"lodash": "^4.17.21",
"next-auth": "^4.24.7",
"nodemailer": "^6.9.15",
"prisma-extension-kysely": "^2.1.0",
"uuid": "^9.0.1",
"winston": "^3.14.2",
"winston": "^3.15.0",
"zod": "^3.23.8",
"zod-to-json-schema": "^3.23.2"
"zod-to-json-schema": "^3.23.5"
},
"devDependencies": {
"@repo/eslint-config": "workspace:*",
@@ -104,7 +105,7 @@
"kysely-codegen": "^0.16.8",
"nodemon": "^3.1.7",
"prettier": "^3.3.3",
"prisma": "^5.20.0",
"prisma": "^5.22.0",
"prisma-erd-generator": "^1.11.2",
"prisma-kysely": "^1.8.0",
"ts-node": "^10.9.2",
@@ -119,7 +120,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.0.7"
}
}
}
+61 -1
View File
@@ -143,6 +143,18 @@ export type AuditLog = {
before: string | null;
after: string | null;
};
export type BackgroundMigration = {
id: string;
name: string;
script: string;
args: unknown;
state: Generated<unknown>;
finished_at: Timestamp | null;
failed_at: Timestamp | null;
failed_reason: string | null;
worker_id: string | null;
locked_at: Timestamp | null;
};
export type BatchExport = {
id: string;
created_at: Generated<Timestamp>;
@@ -266,6 +278,8 @@ export type JobExecution = {
end_time: Timestamp | null;
error: string | null;
job_input_trace_id: string | null;
job_input_observation_id: string | null;
job_input_dataset_item_id: string | null;
job_output_score_id: string | null;
};
export type LlmApiKeys = {
@@ -282,6 +296,20 @@ export type LlmApiKeys = {
config: unknown | null;
project_id: string;
};
export type Media = {
id: string;
sha_256_hash: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
uploaded_at: Timestamp | null;
upload_http_status: number | null;
upload_http_error: string | null;
bucket_path: string;
bucket_name: string;
content_type: string;
content_length: string;
};
export type MembershipInvitation = {
id: string;
email: string;
@@ -304,7 +332,7 @@ export type Model = {
input_price: string | null;
output_price: string | null;
total_price: string | null;
unit: string;
unit: string | null;
tokenizer_id: string | null;
tokenizer_config: unknown | null;
};
@@ -342,6 +370,16 @@ export type Observation = {
completion_start_time: Timestamp | null;
prompt_id: string | null;
};
export type ObservationMedia = {
id: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
media_id: string;
trace_id: string;
observation_id: string;
field: string;
};
export type ObservationView = {
id: string;
trace_id: string | null;
@@ -402,6 +440,14 @@ export type PosthogIntegration = {
enabled: boolean;
created_at: Generated<Timestamp>;
};
export type Price = {
id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
model_id: string;
usage_type: string;
price: string;
};
export type Project = {
id: string;
org_id: string;
@@ -495,6 +541,15 @@ export type Trace = {
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type TraceMedia = {
id: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
media_id: string;
trace_id: string;
field: string;
};
export type TraceSession = {
id: string;
created_at: Generated<Timestamp>;
@@ -546,6 +601,7 @@ export type DB = {
annotation_queues: AnnotationQueue;
api_keys: ApiKey;
audit_logs: AuditLog;
background_migrations: BackgroundMigration;
batch_exports: BatchExport;
comments: Comment;
cron_jobs: CronJobs;
@@ -558,13 +614,16 @@ export type DB = {
job_configurations: JobConfiguration;
job_executions: JobExecution;
llm_api_keys: LlmApiKeys;
media: Media;
membership_invitations: MembershipInvitation;
models: Model;
observation_media: ObservationMedia;
observations: Observation;
observations_view: ObservationView;
organization_memberships: OrganizationMembership;
organizations: Organization;
posthog_integrations: PosthogIntegration;
prices: Price;
project_memberships: ProjectMembership;
projects: Project;
prompts: Prompt;
@@ -572,6 +631,7 @@ export type DB = {
scores: Score;
Session: Session;
sso_configs: SsoConfig;
trace_media: TraceMedia;
trace_sessions: TraceSession;
traces: Trace;
traces_view: TraceView;
@@ -0,0 +1,3 @@
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
UPDATE models SET model_name = 'claude-3.5-sonnet-20241022' WHERE id = 'cm2krz1uf000208jjg5653iud';
UPDATE models SET model_name = 'claude-3.5-sonnet-latest' WHERE id = 'cm2ks2vzn000308jjh4ze1w7q';
@@ -0,0 +1,62 @@
-- CreateTable
CREATE TABLE "prices" (
"id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"model_id" TEXT NOT NULL,
"usage_type" TEXT NOT NULL,
"price" DECIMAL(65,30) NOT NULL,
CONSTRAINT "prices_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE INDEX "prices_model_id_idx" ON "prices"("model_id");
-- CreateIndex
CREATE UNIQUE INDEX "prices_model_id_usage_type_key" ON "prices"("model_id", "usage_type");
-- AddForeignKey
ALTER TABLE "prices" ADD CONSTRAINT "prices_model_id_fkey" FOREIGN KEY ("model_id") REFERENCES "models"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- Alter Table to make unit column nullable
ALTER TABLE "models" ALTER COLUMN "unit" DROP NOT NULL;
-- Create a temporary table to store the new prices for user-defined models
CREATE TEMPORARY TABLE temp_prices AS
SELECT
id AS model_id,
'input' AS usage_type,
input_price AS price
FROM models
WHERE project_id IS NOT NULL AND input_price IS NOT NULL
UNION ALL
SELECT
id AS model_id,
'output' AS usage_type,
output_price AS price
FROM models
WHERE project_id IS NOT NULL AND output_price IS NOT NULL
UNION ALL
SELECT
id AS model_id,
'total' AS usage_type,
total_price AS price
FROM models
WHERE project_id IS NOT NULL AND total_price IS NOT NULL;
-- Insert into prices table
INSERT INTO prices (id, created_at, updated_at, model_id, usage_type, price)
SELECT
md5(random()::text || clock_timestamp()::text || model_id::text || usage_type::text)::uuid AS id,
NOW() AS created_at,
NOW() AS updated_at,
model_id,
usage_type,
price
FROM temp_prices
ON CONFLICT (model_id, usage_type) DO NOTHING;
-- Drop the temporary table
DROP TABLE temp_prices;
@@ -0,0 +1,18 @@
-- CreateTable
CREATE TABLE "background_migrations" (
"id" TEXT NOT NULL,
"name" TEXT NOT NULL,
"script" TEXT NOT NULL,
"args" JSONB NOT NULL,
"finished_at" TIMESTAMP(3),
"failed_at" TIMESTAMP(3),
"failed_reason" TEXT,
"worker_id" TEXT,
"locked_at" TIMESTAMP(3),
CONSTRAINT "background_migrations_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE UNIQUE INDEX "background_migrations_name_key" ON "background_migrations"("name");
@@ -0,0 +1,2 @@
INSERT INTO background_migrations (id, name, script, args)
VALUES ('32859a35-98f5-4a4a-b438-ebc579349e00', '20241024_1216_add_generations_cost_backfill', 'addGenerationsCostBackfill', '{}');
@@ -0,0 +1,2 @@
INSERT INTO background_migrations (id, name, script, args)
VALUES ('5960f22a-748f-480c-b2f3-bc4f9d5d84bc', '20241024_1730_migrate_traces_from_pg_to_ch', 'migrateTracesFromPostgresToClickhouse', '{}');
@@ -0,0 +1,2 @@
INSERT INTO background_migrations (id, name, script, args)
VALUES ('7526e7c9-0026-4595-af2c-369dfd9176ec', '20241024_1737_migrate_observations_from_pg_to_ch', 'migrateObservationsFromPostgresToClickhouse', '{}');
@@ -0,0 +1,2 @@
INSERT INTO background_migrations (id, name, script, args)
VALUES ('94e50334-50d3-4e49-ad2e-9f6d92c85ef7', '20241024_1738_migrate_scores_from_pg_to_ch', 'migrateScoresFromPostgresToClickhouse', '{}');
@@ -0,0 +1,2 @@
-- DropIndex
DROP INDEX CONCURRENTLY "prices_model_id_idx";
@@ -0,0 +1,2 @@
-- AlterTable
ALTER TABLE "background_migrations" ADD COLUMN "state" jsonb NOT NULL DEFAULT '{}';
@@ -0,0 +1,29 @@
INSERT INTO models (
id,
project_id,
model_name,
match_pattern,
start_date,
input_price,
output_price,
total_price,
unit,
tokenizer_id,
tokenizer_config
)
VALUES
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
('cm34aq60d000207ml0j1h31ar', NULL, 'claude-3-5-haiku-20241022', '(?i)^(claude-3-5-haiku-20241022|anthropic\.claude-3-5-haiku-20241022-v1:0|claude-3-5-haiku-V1@20241022)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL),
('cm34aqb9h000307ml6nypd618', NULL, 'claude-3.5-haiku-latest', '(?i)^(claude-3-5-haiku-latest)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL);
INSERT INTO prices (
id,
model_id,
usage_type,
price
)
VALUES
('cm34ax6mc000008jkfqed92mb', 'cm34aq60d000207ml0j1h31ar', 'input', 0.000001),
('cm34axb2o000108jk09wn9b47', 'cm34aqb9h000307ml6nypd618', 'input', 0.000001),
('cm34axeie000208jk8b2ke2t8', 'cm34aq60d000207ml0j1h31ar', 'output', 0.000005),
('cm34axi67000308jk7x1a7qko', 'cm34aqb9h000307ml6nypd618', 'output', 0.000005);
@@ -0,0 +1,71 @@
-- CreateTable
CREATE TABLE "media" (
"id" TEXT NOT NULL,
"sha_256_hash" CHAR(44) NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"uploaded_at" TIMESTAMP(3),
"upload_http_status" INTEGER,
"upload_http_error" TEXT,
"bucket_path" TEXT NOT NULL,
"bucket_name" TEXT NOT NULL,
"content_type" TEXT NOT NULL,
"content_length" BIGINT NOT NULL,
CONSTRAINT "media_pkey" PRIMARY KEY ("id")
);
-- CreateTable
CREATE TABLE "trace_media" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"media_id" TEXT NOT NULL,
"trace_id" TEXT NOT NULL,
"field" TEXT NOT NULL,
CONSTRAINT "trace_media_pkey" PRIMARY KEY ("id")
);
-- CreateTable
CREATE TABLE "observation_media" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"media_id" TEXT NOT NULL,
"trace_id" TEXT NOT NULL,
"observation_id" TEXT NOT NULL,
"field" TEXT NOT NULL,
CONSTRAINT "observation_media_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE UNIQUE INDEX "media_project_id_sha_256_hash_key" ON "media"("project_id", "sha_256_hash");
-- CreateIndex
CREATE UNIQUE INDEX "trace_media_project_id_trace_id_media_id_field_key" ON "trace_media"("project_id", "trace_id", "media_id", "field");
-- CreateIndex
CREATE INDEX "observation_media_project_id_observation_id_idx" ON "observation_media"("project_id", "observation_id");
-- CreateIndex
CREATE UNIQUE INDEX "observation_media_project_id_trace_id_observation_id_media__key" ON "observation_media"("project_id", "trace_id", "observation_id", "media_id", "field");
-- AddForeignKey
ALTER TABLE "media" ADD CONSTRAINT "media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
@@ -0,0 +1,6 @@
-- DropForeignKey
ALTER TABLE "job_executions" DROP CONSTRAINT "job_executions_job_input_trace_id_fkey";
-- AlterTable
ALTER TABLE "job_executions" ADD COLUMN "job_input_dataset_item_id" TEXT,
ADD COLUMN "job_input_observation_id" TEXT;
@@ -0,0 +1,25 @@
INSERT INTO models (
id,
project_id,
model_name,
match_pattern,
start_date,
input_price,
output_price,
total_price,
unit,
tokenizer_id,
tokenizer_config
)
VALUES
('cm3x0p8ev000008kyd96800c8', NULL, 'chatgpt-4o-latest', '(?i)^(chatgpt-4o-latest)$', NULL, 0.000005, 0.000015, NULL, 'TOKENS', 'openai', '{ "tokensPerMessage": 3, "tokensPerName": 1, "tokenizerModel": "gpt-4o" }');
INSERT INTO prices (
id,
model_id,
usage_type,
price
)
VALUES
('cm3x0psrz000108kydpxg9o2k', 'cm3x0p8ev000008kyd96800c8', 'input', 0.000005),
('cm3x0pyt7000208ky8737gdla', 'cm3x0p8ev000008kyd96800c8', 'output', 0.000015);
File diff suppressed because it is too large Load Diff
+157 -118
View File
@@ -256,7 +256,7 @@ async function main() {
const queueIds = await generateQueuesForProject(
[project1, project2],
configIdsAndNames
configIdsAndNames,
);
const promptIds = await generatePromptsForProject([project1, project2]);
@@ -282,11 +282,11 @@ async function main() {
project2,
promptIds,
queueIds,
configIdsAndNames
configIdsAndNames,
);
logger.info(
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`,
);
await uploadObjects(
@@ -296,7 +296,7 @@ async function main() {
sessions,
events,
comments,
queueItems
queueItems,
);
// If openai key is in environment, add it to the projects LLM API keys
@@ -314,7 +314,7 @@ async function main() {
});
} else {
logger.warn(
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key."
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key.",
);
}
@@ -387,16 +387,62 @@ async function main() {
update: {},
});
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
const dataset = await prisma.dataset.create({
data: {
name: `demo-dataset-${datasetNumber}`,
description:
datasetNumber === 0 ? "Dataset test description" : undefined,
projectId: project2.id,
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
},
});
await createDatasets(project1, project2, observations);
}
}
main()
.then(async () => {
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
})
.catch(async (e) => {
logger.error(e);
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
process.exit(1);
});
export async function createDatasets(
project1: {
id: string;
orgId: string;
createdAt: Date;
updatedAt: Date;
name: string;
},
project2: {
id: string;
orgId: string;
createdAt: Date;
updatedAt: Date;
name: string;
},
observations: Prisma.ObservationCreateManyInput[],
) {
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
for (const projectId of [project1.id, project2.id]) {
const datasetName = `demo-dataset-${datasetNumber}`;
// check if ds already exists
const dataset =
(await prisma.dataset.findFirst({
where: {
projectId,
name: datasetName,
},
})) ??
(await prisma.dataset.create({
data: {
name: datasetName,
description:
datasetNumber === 0 ? "Dataset test description" : undefined,
projectId,
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
},
}));
const datasetItemIds = [];
for (let i = 0; i < 18; i++) {
@@ -406,7 +452,7 @@ async function main() {
: undefined;
const datasetItem = await prisma.datasetItem.create({
data: {
projectId: project2.id,
projectId,
datasetId: dataset.id,
sourceTraceId: sourceObservation?.traceId,
sourceObservationId:
@@ -431,9 +477,16 @@ async function main() {
}
for (let datasetRunNumber = 0; datasetRunNumber < 5; datasetRunNumber++) {
const datasetRun = await prisma.datasetRuns.create({
data: {
projectId: project2.id,
const datasetRun = await prisma.datasetRuns.upsert({
where: {
datasetId_projectId_name: {
datasetId: dataset.id,
projectId,
name: `demo-dataset-run-${datasetRunNumber}`,
},
},
create: {
projectId,
name: `demo-dataset-run-${datasetRunNumber}`,
description: Math.random() > 0.5 ? "Dataset run description" : "",
datasetId: dataset.id,
@@ -445,11 +498,12 @@ async function main() {
["tag1", "tag2"],
][datasetRunNumber % 5],
},
update: {},
});
for (const datasetItemId of datasetItemIds) {
const relevantObservations = observations.filter(
(o) => o.projectId === project2.id
(o) => o.projectId === projectId,
);
const observation =
relevantObservations[
@@ -458,7 +512,7 @@ async function main() {
await prisma.datasetRunItems.create({
data: {
projectId: project2.id,
projectId,
datasetItemId,
traceId: observation.traceId as string,
observationId: Math.random() > 0.5 ? observation.id : undefined,
@@ -471,20 +525,6 @@ async function main() {
}
}
main()
.then(async () => {
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
})
.catch(async (e) => {
logger.error(e);
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
process.exit(1);
});
async function uploadObjects(
traces: Prisma.TraceCreateManyInput[],
observations: Prisma.ObservationCreateManyInput[],
@@ -492,7 +532,7 @@ async function uploadObjects(
sessions: Prisma.TraceSessionCreateManyInput[],
events: Prisma.ObservationCreateManyInput[],
comments: Prisma.CommentCreateManyInput[],
queueItems: Prisma.AnnotationQueueItemCreateManyInput[]
queueItems: Prisma.AnnotationQueueItemCreateManyInput[],
) {
let promises: Prisma.PrismaPromise<unknown>[] = [];
@@ -506,14 +546,14 @@ async function uploadObjects(
},
create: chunk[0]!,
update: {},
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -524,13 +564,13 @@ async function uploadObjects(
promises.push(
prisma.trace.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -540,14 +580,14 @@ async function uploadObjects(
promises.push(
prisma.observation.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -557,14 +597,14 @@ async function uploadObjects(
promises.push(
prisma.observation.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -574,13 +614,13 @@ async function uploadObjects(
promises.push(
prisma.score.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -590,13 +630,13 @@ async function uploadObjects(
promises.push(
prisma.comment.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -606,13 +646,13 @@ async function uploadObjects(
promises.push(
prisma.annotationQueueItem.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -634,7 +674,7 @@ function createObjects(
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
>
>,
) {
const traces: Prisma.TraceCreateManyInput[] = [];
const observations: Prisma.ObservationCreateManyInput[] = [];
@@ -649,7 +689,7 @@ function createObjects(
// print progress to console with a progress bar that refreshes every 10 iterations
// random date within last 90 days, with a linear bias towards more recent dates
const traceTs = new Date(
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000)
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000),
);
const envTag = envTags[Math.floor(Math.random() * envTags.length)];
@@ -805,11 +845,11 @@ function createObjects(
for (let j = 0; j < Math.floor(Math.random() * 10) + 1; j++) {
// add between 1 and 30 ms to trace timestamp
const spanTsStart = new Date(
traceTs.getTime() + Math.floor(Math.random() * 30)
traceTs.getTime() + Math.floor(Math.random() * 30),
);
// random duration of upto 5000ms
const spanTsEnd = new Date(
spanTsStart.getTime() + Math.floor(Math.random() * 5000)
spanTsStart.getTime() + Math.floor(Math.random() * 5000),
);
const span = {
@@ -844,22 +884,22 @@ function createObjects(
const generationTsStart = new Date(
spanTsStart.getTime() +
Math.floor(
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
)
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
),
);
const generationTsEnd = new Date(
generationTsStart.getTime() +
Math.floor(
Math.random() *
(spanTsEnd.getTime() - generationTsStart.getTime())
)
(spanTsEnd.getTime() - generationTsStart.getTime()),
),
);
// somewhere in the middle
const generationTsCompletionStart = new Date(
generationTsStart.getTime() +
Math.floor(
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3
)
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3,
),
);
const promptTokens = Math.floor(Math.random() * 1000) + 300;
@@ -880,7 +920,7 @@ function createObjects(
const promptId =
promptIds.get(projectId)![
Math.floor(
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2)
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2),
)
];
@@ -962,8 +1002,8 @@ function createObjects(
const eventTs = new Date(
spanTsStart.getTime() +
Math.floor(
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
)
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
),
);
events.push({
@@ -985,7 +1025,7 @@ function createObjects(
}
// find unique sessions by id and projectid
const uniqueSessions: Prisma.TraceSessionCreateManyInput[] = Array.from(
new Set(sessions.map((session) => JSON.stringify(session)))
new Set(sessions.map((session) => JSON.stringify(session))),
).map((session) => JSON.parse(session) as Prisma.TraceSessionCreateManyInput);
return {
@@ -1007,65 +1047,64 @@ async function generatePromptsForProject(projects: Project[]) {
projects.map(async (project) => {
const promptIdsForProject = await generatePrompts(project);
promptIds.set(project.id, promptIdsForProject);
})
}),
);
return promptIds;
}
async function generatePrompts(project: Project) {
const promptIds: string[] = [];
const prompts = [
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 1 content",
name: "Prompt 1",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 2 content",
name: "Prompt 2",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "API",
prompt: "Prompt 3 content",
name: "Prompt 3 by API",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 4 content",
name: "Prompt 4",
version: 1,
labels: ["production", "latest"],
tags: ["tag1", "tag2"],
},
];
export const SEED_PROMPTS = [
{
id: `prompt-123`,
createdBy: "user-1",
prompt: "Prompt 1 content",
name: "Prompt 1",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-456`,
createdBy: "user-1",
prompt: "Prompt 2 content",
name: "Prompt 2",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-789`,
createdBy: "API",
prompt: "Prompt 3 content",
name: "Prompt 3 by API",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-abc`,
createdBy: "user-1",
prompt: "Prompt 4 content",
name: "Prompt 4",
version: 1,
labels: ["production", "latest"],
tags: ["tag1", "tag2"],
},
];
for (const prompt of prompts) {
export const PROMPT_IDS: string[] = [];
async function generatePrompts(project: Project) {
const promptIds = [];
for (const prompt of SEED_PROMPTS) {
await prisma.prompt.upsert({
where: {
projectId_name_version: {
projectId: prompt.projectId,
projectId: prompt.id + project.id,
name: prompt.name,
version: prompt.version,
},
id: prompt.id + project.id,
},
create: {
id: prompt.id,
projectId: prompt.projectId,
id: prompt.id + project.id,
projectId: project.id,
createdBy: prompt.createdBy,
prompt: prompt.prompt,
name: prompt.name,
@@ -1073,9 +1112,7 @@ async function generatePrompts(project: Project) {
labels: prompt.labels,
tags: prompt.tags,
},
update: {
id: prompt.id,
},
update: {},
});
promptIds.push(prompt.id);
}
@@ -1129,6 +1166,7 @@ async function generatePrompts(project: Project) {
name: version.name,
version: version.version,
},
id: version.id,
},
create: {
id: version.id,
@@ -1159,6 +1197,7 @@ async function generatePrompts(project: Project) {
name: promptName,
version: i,
},
id: promptId,
},
create: {
id: promptId,
@@ -1193,7 +1232,7 @@ async function generateConfigsForProject(projects: Project[]) {
projects.map(async (project) => {
const configNameAndId = await generateConfigs(project);
projectIdsToConfigs.set(project.id, configNameAndId);
})
}),
);
return projectIdsToConfigs;
}
@@ -1343,7 +1382,7 @@ async function generateQueuesForProject(
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
>
>,
) {
const projectIdsToQueues: Map<string, string[]> = new Map();
@@ -1351,10 +1390,10 @@ async function generateQueuesForProject(
projects.map(async (project) => {
const queueIds = await generateQueues(
project,
configIdsAndNames.get(project.id) ?? []
configIdsAndNames.get(project.id) ?? [],
);
projectIdsToQueues.set(project.id, queueIds);
})
}),
);
return projectIdsToQueues;
}
@@ -1366,7 +1405,7 @@ async function generateQueues(
id: string;
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
}[],
) {
const queue = {
id: `queue-${v4()}`,
+15 -14
View File
@@ -7,13 +7,13 @@ import {
logger,
} from "../src/server";
import { prepareClickhouse } from "./prepareClickhouse";
import { redis } from "@langfuse/shared/src/server";
import { redis } from "../src/server";
const createRandomProjectId = () => randomUUID().toString();
const prepareProjectsAndApiKeys = async (
numOfProjects: number,
opts: { requiredProjectIds: string[] }
opts: { requiredProjectIds: string[] },
) => {
const { requiredProjectIds } = opts;
const projectsToCreate = numOfProjects - requiredProjectIds.length;
@@ -51,7 +51,7 @@ const prepareProjectsAndApiKeys = async (
});
if (!apiKeyExists) {
const sk = await hashSecretKey(
`sk-${Math.random().toString(36).substr(2, 9)}`
`sk-${Math.random().toString(36).substr(2, 9)}`,
);
await prisma.apiKey.create({
data: {
@@ -75,35 +75,36 @@ const prepareProjectsAndApiKeys = async (
};
async function main() {
let numOfProjects = parseInt(process.argv[3], 10);
let numberOfDays = parseInt(process.argv[4], 10);
let totalObservations = parseInt(process.argv[5], 10);
let numOfProjects = parseInt(process.argv[2], 10);
let numberOfDays = parseInt(process.argv[3], 10);
let totalObservations = parseInt(process.argv[4], 10);
logger.info(process.argv);
logger.info(
`Preparing Clickhouse for ${numOfProjects} projects and ${numberOfDays} days with max Observations ${totalObservations}.`,
);
if (isNaN(totalObservations)) {
logger.warn(
"Total observations not provided or invalid. Defaulting to 1000 observations."
"Total observations not provided or invalid. Defaulting to 1000 observations.",
);
totalObservations = 1000;
}
if (isNaN(numOfProjects)) {
logger.warn(
"Number of projects not provided or invalid. Defaulting to 10 projects."
"Number of projects not provided or invalid. Defaulting to 10 projects.",
);
numOfProjects = 10;
}
if (isNaN(numberOfDays)) {
logger.warn(
"Number of days not provided or invalid. Defaulting to 3 days."
"Number of days not provided or invalid. Defaulting to 3 days.",
);
numberOfDays = 3;
}
logger.info(
`Preparing Clickhouse for ${numOfProjects} projects and ${numberOfDays} days with max Observations ${totalObservations}.`
);
try {
const projectIds = [
"7a88fb47-b4e2-43b8-a06c-a5ce950dc53a",
@@ -123,7 +124,7 @@ async function main() {
} catch (error) {
logger.error("Error during Clickhouse preparation:", error);
} finally {
await clickhouseClient.close();
await clickhouseClient().close();
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from Clickhouse.");
+69 -35
View File
@@ -1,3 +1,5 @@
import { SEED_PROMPTS } from "../prisma/seed";
import { prisma } from "../src/db";
import { clickhouseClient, logger } from "../src/server";
function randn_bm(min: number, max: number, skew: number) {
@@ -23,15 +25,15 @@ export const prepareClickhouse = async (
opts: {
numberOfDays: number;
totalObservations: number;
}
},
) => {
logger.info(
`Preparing Clickhouse for ${projectIds.length} projects and ${opts.numberOfDays} days.`
`Preparing Clickhouse for ${projectIds.length} projects and ${opts.numberOfDays} days.`,
);
const projectData = projectIds.map((projectId) => {
const observationsPerProject = Math.ceil(
randn_bm(0, opts.totalObservations, 2)
randn_bm(0, opts.totalObservations, 2),
); // Skew the number of observations
const tracesPerProject = Math.floor(observationsPerProject / 6); // On average, one trace should have 6 observations
@@ -52,7 +54,7 @@ export const prepareClickhouse = async (
scoresPerProject,
} = data;
logger.info(
`Preparing Clickhouse for ${projectId}: Traces: ${tracesPerProject}, Scores: ${scoresPerProject}, Observations: ${observationsPerProject}`
`Preparing Clickhouse for ${projectId}: Traces: ${tracesPerProject}, Scores: ${scoresPerProject}, Observations: ${observationsPerProject}`,
);
const tracesQuery = `
@@ -60,7 +62,7 @@ export const prepareClickhouse = async (
SELECT toString(number) AS id,
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
concat('name_', toString(rand() % 100)) AS name,
concat('user_id_', toString(randUniform(0, 100))) AS user_id,
concat('user_id_', toInt64(randExponential(1 / 100))) AS user_id,
map('key', 'value') AS metadata,
concat('release_', toString(randUniform(0, 100))) AS release,
concat('version_', toString(randUniform(0, 100))) AS version,
@@ -70,10 +72,11 @@ export const prepareClickhouse = async (
array('tag1', 'tag2') as tags,
repeat('input', toInt64(randExponential(1 / 100))) AS input,
repeat('output', toInt64(randExponential(1 / 100))) AS output,
concat('session_', toString(rand() % 100)) AS session_id,
if(randUniform(0, 1) < 0.2, NULL, concat('session_', toString(rand() % 1000))) AS session_id,
timestamp AS created_at,
timestamp AS updated_at,
timestamp AS event_ts
timestamp AS event_ts,
0 AS is_deleted
FROM numbers(${tracesPerProject});
`;
@@ -82,13 +85,13 @@ export const prepareClickhouse = async (
SELECT toString(number) AS id,
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
'${projectId}' AS project_id,
if(rand() < 0.47, 'GENERATION', if(rand() < 0.94, 'SPAN', 'EVENT')) AS type,
if(randUniform(0, 1) < 0.47, 'GENERATION', if(randUniform(0, 1) < 0.94, 'SPAN', 'EVENT')) AS type,
toString(rand()) AS parent_observation_id,
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS start_time,
addSeconds(start_time, if(rand() < 0.6, floor(randUniform(0, 20)), floor(randUniform(0, 3600)))) AS end_time,
concat('name', toString(rand() % 100)) AS name,
map('key', 'value') AS metadata,
if(rand() < 0.9, 'DEFAULT', if(rand() < 0.5, 'ERROR', if(rand() < 0.5, 'DEBUG', 'WANING'))) AS level,
if(randUniform(0, 1) < 0.9, 'DEFAULT', if(randUniform(0, 1) < 0.5, 'ERROR', if(randUniform(0, 1) < 0.5, 'DEBUG', 'WARNING'))) AS level,
'status_message' AS status_message,
'version' AS version,
repeat('input', toInt64(randExponential(1 / 100))) AS input,
@@ -101,30 +104,31 @@ export const prepareClickhouse = async (
when number % 2 = 0 then 'cltr0w45b000008k1407o9qv1'
else 'clrntkjgy000f08jx79v9g1xj'
end as internal_model_id,
'model_parameters' AS model_parameters,
toInt32(randUniform(0, 1000)) AS provided_input_usage_units,
toInt32(randUniform(0, 1000)) AS provided_output_usage_units,
toInt32(randUniform(0, 1000)) AS provided_total_usage_units,
toInt32(randUniform(0, 1000)) AS input_usage_units,
toInt32(randUniform(0, 1000)) AS output_usage_units,
toInt32(randUniform(0, 1000)) AS total_usage_units,
toInt32(randUniform(0, 1000)) AS provided_input_cost,
toInt32(randUniform(0, 1000)) AS provided_output_cost,
toInt32(randUniform(0, 1000)) AS provided_total_cost,
'TOKENS' AS unit,
toInt32(randUniform(0, 1000)) AS input_cost,
toInt32(randUniform(0, 1000)) AS output_cost,
toInt32(randUniform(0, 1000)) AS total_cost,
'{"temperature": 0.7, "max_tokens": 150}' AS model_parameters,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS provided_usage_details,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS usage_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS provided_cost_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS cost_details,
toDecimal64(randUniform(0, 2000), 12) AS total_cost,
start_time AS completion_start_time,
toString(rand()) AS prompt_id,
toString(rand()) AS prompt_name,
1000 AS prompt_version,
array(${SEED_PROMPTS.map((p) => `concat('${p.id}',project_id)`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_id,
array(${SEED_PROMPTS.map((p) => `'${p.name}'`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_name,
array(${SEED_PROMPTS.map((p) => `'${p.version}'`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_version,
start_time AS created_at,
start_time AS updated_at,
start_time AS event_ts
start_time AS event_ts,
0 AS is_deleted
FROM numbers(${observationsPerProject});
`;
console.log(observationsQuery);
const scoresQuery = `
INSERT INTO scores
SELECT toString(floor(randUniform(0, 100))) AS id,
@@ -136,17 +140,19 @@ export const prepareClickhouse = async (
toString(floor(randUniform(0, ${observationsPerProject}))),
NULL
) AS observation_id,
concat('name_', toString(rand() % 100)) AS name,
concat('name_', toString(rand() % 10)) AS name,
randUniform(0, 100) as value,
'API' as source,
'comment' as comment,
toString(rand() % 100) as author_user_id,
toString(rand() % 100) as config_id,
toString(rand() % 100) as data_type,
if (rand() < 0.33, 'NUMERIC', if (rand() < 0.5, 'CATEGORICAL', 'BOOLEAN')) as data_type,
toString(rand() % 100) as string_value,
NULL as queue_id,
timestamp AS created_at,
timestamp AS updated_at,
timestamp AS event_ts
timestamp AS event_ts,
0 AS is_deleted
FROM numbers(${scoresPerProject});
`;
@@ -154,13 +160,41 @@ export const prepareClickhouse = async (
for (const query of queries) {
logger.info(`Executing query: ${query}`);
await clickhouseClient.command({
await clickhouseClient().command({
query,
clickhouse_settings: {
wait_end_of_query: 1,
},
});
}
// we also need to upsert trace sessions in postgres
const sessionQuery = `
SELECT session_id, project_id
FROM traces
WHERE session_id IS NOT NULL;
`;
const sessionResult = await clickhouseClient().query({
query: sessionQuery,
format: "JSONEachRow",
});
const sessionData = await sessionResult.json<{
session_id: string;
project_id: string;
}>();
const idProjectIdCombinations = sessionData.map((session) => ({
id: session.session_id,
projectId: session.project_id,
public: Math.random() < 0.1,
bookmarked: Math.random() < 0.1,
}));
await prisma.traceSession.createMany({
data: idProjectIdCombinations,
skipDuplicates: true,
});
}
const tables = ["traces", "scores", "observations"];
@@ -178,14 +212,14 @@ export const prepareClickhouse = async (
ORDER BY count() desc
`;
const result = await clickhouseClient.query({
const result = await clickhouseClient().query({
query,
format: "TabSeparated",
});
logger.info(
`${table.charAt(0).toUpperCase() + table.slice(1)} per Project: \n` +
(await result.text())
(await result.text()),
);
}
@@ -209,14 +243,14 @@ export const prepareClickhouse = async (
ORDER BY event_date desc
`;
const result = await clickhouseClient.query({
const result = await clickhouseClient().query({
query,
format: "TabSeparated",
});
logger.info(
`${table.charAt(0).toUpperCase() + table.slice(1)} per Date: \n` +
(await result.text())
(await result.text()),
);
}
};
+2 -2
View File
@@ -76,7 +76,7 @@ export class KyselySingleton {
createQueryCompiler: () => new PostgresQueryCompiler(),
},
}),
})
}),
);
return KyselySingleton.instance;
@@ -103,7 +103,7 @@ if (process.env.NODE_ENV === "development") {
createQueryCompiler: () => new PostgresQueryCompiler(),
},
}),
})
}),
);
}
+22
View File
@@ -30,6 +30,16 @@ const EnvSchema = z.object({
CLICKHOUSE_URL: z.string().url().optional(),
CLICKHOUSE_USER: z.string().optional(),
CLICKHOUSE_PASSWORD: z.string().optional(),
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_ASYNC_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_INGESTION_QUEUE_DELAY_MS: z.coerce
.number()
.nonnegative()
.default(15_000),
SALT: z.string().optional(), // used by components imported by web package
LANGFUSE_LOG_LEVEL: z
.enum(["trace", "debug", "info", "warn", "error", "fatal"])
@@ -38,6 +48,18 @@ const EnvSchema = z.object({
ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: z.enum(["true", "false"]).default("false"),
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: z.string().default(""),
LANGFUSE_S3_EVENT_UPLOAD_REGION: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_USE_AZURE_BLOB: z.enum(["true", "false"]).default("false"),
STRIPE_SECRET_KEY: z.string().optional(),
});
export const env = EnvSchema.parse(removeEmptyEnvVariables(process.env));
+26 -2
View File
@@ -5,6 +5,7 @@ export const langfuseObjects = [
"span",
"generation",
"event",
"dataset_item",
] as const;
// variable mapping stored in the db for eval templates
@@ -21,7 +22,7 @@ export const variableMapping = z
(value) => value.langfuseObject === "trace" || value.objectName !== null,
{
message: "objectName is required for langfuseObjects other than trace",
}
},
);
export const variableMappingList = z.array(variableMapping);
@@ -44,7 +45,7 @@ const observationCols = [
{ name: "Output", id: "output", internal: 'o."output"' },
];
export const availableEvalVariables = [
export const availableTraceEvalVariables = [
{
id: "trace",
display: "Trace",
@@ -76,6 +77,28 @@ export const availableEvalVariables = [
},
];
export const availableDatasetEvalVariables = [
{
id: "dataset_item",
display: "Dataset item",
availableColumns: [
{
name: "Metadata",
id: "metadata",
type: "stringObject",
internal: 'd."metadata"',
},
{ name: "Input", id: "input", internal: 'd."input"' },
{
name: "Expected output",
id: "expected_output",
internal: 'd."expected_output"',
},
],
},
...availableTraceEvalVariables,
];
export const OutputSchema = z.object({
reasoning: z.string(),
score: z.string(),
@@ -83,6 +106,7 @@ export const OutputSchema = z.object({
export enum EvalTargetObject {
Trace = "trace",
Dataset = "dataset",
}
export const DEFAULT_TRACE_JOB_DELAY = 10_000;
@@ -0,0 +1,3 @@
export * from "./scoreTypes";
export * from "./types";
export * from "./scoreConfigTypes";
@@ -0,0 +1,37 @@
import { ScoreDataType, ScoreSource } from "@prisma/client";
export type CategoricalAggregate = {
type: "CATEGORICAL";
values: string[];
valueCounts: { value: string; count: number }[];
comment?: string | null;
};
export type NumericAggregate = {
type: "NUMERIC";
values: number[];
average: number;
comment?: string | null;
};
export type ScoreAggregate = Record<
string,
CategoricalAggregate | NumericAggregate
>;
export type ScoreSimplified = {
name: string;
dataType: ScoreDataType;
source: ScoreSource;
value?: number | null;
comment?: string | null;
stringValue?: string | null;
};
export type LastUserScore = ScoreSimplified & {
timestamp: string;
traceId: string;
observationId?: string | null;
userId: string;
};
+3 -3
View File
@@ -6,11 +6,12 @@ export * from "./interfaces/parseDbOrg";
export * from "./interfaces/customLLMProviderConfigSchemas";
export * from "./tableDefinitions";
export * from "./types";
export * from "./tracesTable";
export * from "./tableDefinitions/tracesTable";
export * from "./server/auth/apiKeys";
export * from "./observationsTable";
export * from "./utils/zod";
export * from "./utils/json";
export * from "./utils/stringChecks";
export * from "./utils/objects";
export * from "./utils/typeChecks";
export * from "./features/entitlements/plans";
@@ -28,8 +29,7 @@ export * from "./features/batchExport/types";
export * from "./features/annotation/types";
// scores
export * from "./features/scores/scoreConfigTypes";
export * from "./features/scores/scoreTypes";
export * from "./features/scores";
// comments
export * from "./features/comments/types";
@@ -15,6 +15,7 @@ export const filterOperators = {
],
numberObject: ["=", ">", "<", ">=", "<="],
boolean: ["=", "<>"],
null: ["is null", "is not null"],
} as const;
export const timeFilter = z.object({
@@ -68,6 +69,12 @@ export const booleanFilter = z.object({
operator: z.enum(filterOperators.boolean),
value: z.boolean(),
});
export const nullFilter = z.object({
type: z.literal("null"),
column: z.string(),
operator: z.enum(filterOperators.null),
value: z.literal(""),
});
export const singleFilter = z.discriminatedUnion("type", [
timeFilter,
stringFilter,
@@ -77,4 +84,5 @@ export const singleFilter = z.discriminatedUnion("type", [
stringObjectFilter,
numberObjectFilter,
booleanFilter,
nullFilter,
]);
-48
View File
@@ -1,48 +0,0 @@
import { createClient } from "@clickhouse/client";
import { env } from "../env";
import { eventTypes } from "./ingestion/types";
import { LangfuseNotFoundError } from "../errors";
export enum ClickhouseEntityType {
Trace = "trace",
Score = "score",
Observation = "observation",
SdkLog = "sdk-log",
}
export function getClickhouseEntityType(
eventType: string,
): ClickhouseEntityType {
switch (eventType) {
case eventTypes.TRACE_CREATE:
return ClickhouseEntityType.Trace;
case eventTypes.OBSERVATION_CREATE:
case eventTypes.OBSERVATION_UPDATE:
case eventTypes.EVENT_CREATE:
case eventTypes.SPAN_CREATE:
case eventTypes.SPAN_UPDATE:
case eventTypes.GENERATION_CREATE:
case eventTypes.GENERATION_UPDATE:
return ClickhouseEntityType.Observation;
case eventTypes.SCORE_CREATE:
return ClickhouseEntityType.Score;
case eventTypes.SDK_LOG:
return ClickhouseEntityType.SdkLog;
default:
throw new LangfuseNotFoundError(`Unknown event type: ${eventType}`);
}
}
export type ClickhouseClientType = ReturnType<typeof createClient>;
export const clickhouseClient = createClient({
url: env.CLICKHOUSE_URL,
username: env.CLICKHOUSE_USER,
password: env.CLICKHOUSE_PASSWORD,
database: "default",
clickhouse_settings: {
async_insert: 1,
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
},
});
@@ -0,0 +1,28 @@
import { createClient } from "@clickhouse/client";
import { env } from "../../env";
import { NodeClickHouseClientConfigOptions } from "@clickhouse/client/dist/config";
export type ClickhouseClientType = ReturnType<typeof createClient>;
export const clickhouseClient = (opts?: NodeClickHouseClientConfigOptions) =>
createClient({
...opts,
url: env.CLICKHOUSE_URL,
username: env.CLICKHOUSE_USER,
password: env.CLICKHOUSE_PASSWORD,
database: "default",
clickhouse_settings: {
async_insert: 1,
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
},
});
export const defaultClickhouseClient = clickhouseClient();
/**
* Accepts a JavaScript date and returns the DateTime in format YYYY-MM-DD HH:MM:SS
*/
export const convertDateToClickhouseDateTime = (date: Date): string => {
// 2024-11-06T20:37:00.123Z -> 2024-11-06 21:37:00.123
return date.toISOString().replace("T", " ").replace("Z", "");
};
@@ -0,0 +1,7 @@
export const ClickhouseTableNames = {
traces: "traces",
observations: "observations",
scores: "scores",
} as const;
export type ClickhouseTableName = keyof typeof ClickhouseTableNames;
@@ -0,0 +1,37 @@
import { LangfuseNotFoundError } from "../../errors";
import { eventTypes } from "../ingestion/types";
import { ClickhouseTableName, ClickhouseTableNames } from "./schema";
export const isValidTableName = (
tableName: string,
): tableName is ClickhouseTableName =>
Object.keys(ClickhouseTableNames).includes(tableName);
export type IngestionEntityTypes =
| "trace"
| "observation"
| "score"
| "sdk_log";
export const getClickhouseEntityType = (
eventType: string,
): IngestionEntityTypes => {
switch (eventType) {
case eventTypes.TRACE_CREATE:
return "trace";
case eventTypes.OBSERVATION_CREATE:
case eventTypes.OBSERVATION_UPDATE:
case eventTypes.EVENT_CREATE:
case eventTypes.SPAN_CREATE:
case eventTypes.SPAN_UPDATE:
case eventTypes.GENERATION_CREATE:
case eventTypes.GENERATION_UPDATE:
return "observation";
case eventTypes.SCORE_CREATE:
return "score";
case eventTypes.SDK_LOG:
return "sdk_log";
default:
throw new LangfuseNotFoundError(`Unknown event type: ${eventType}`);
}
};
-176
View File
@@ -1,176 +0,0 @@
import z from "zod";
export const clickhouseStringDateSchema = z
.string()
// clickhouse stores UTC like '2024-05-23 18:33:41.602000'
// we need to convert it to '2024-05-23T18:33:41.602000Z'
.transform((str) => str.replace(" ", "T") + "Z")
.pipe(z.string().datetime());
export const observationRecordBaseSchema = z.object({
id: z.string(),
trace_id: z.string().nullish(),
project_id: z.string(),
type: z.string(),
parent_observation_id: z.string().nullish(),
name: z.string().nullish(),
metadata: z.record(z.string()),
level: z.string().nullish(),
status_message: z.string().nullish(),
version: z.string().nullish(),
input: z.string().nullish(),
output: z.string().nullish(),
provided_model_name: z.string().nullish(),
internal_model_id: z.string().nullish(),
model_parameters: z.string().nullish(),
unit: z.string().nullish(),
input_usage_units: z.number().nullish(),
output_usage_units: z.number().nullish(),
total_usage_units: z.number().nullish(),
input_cost: z.number().nullish(),
output_cost: z.number().nullish(),
total_cost: z.number().nullish(),
provided_input_usage_units: z.number().nullish(),
provided_output_usage_units: z.number().nullish(),
provided_total_usage_units: z.number().nullish(),
provided_input_cost: z.number().nullish(),
provided_output_cost: z.number().nullish(),
provided_total_cost: z.number().nullish(),
prompt_id: z.string().nullish(),
prompt_name: z.string().nullish(),
prompt_version: z.number().nullish(),
});
export type ObservationRecordBaseType = z.infer<
typeof observationRecordBaseSchema
>;
export const observationRecordReadSchema = observationRecordBaseSchema.extend({
created_at: clickhouseStringDateSchema,
updated_at: clickhouseStringDateSchema,
start_time: clickhouseStringDateSchema,
end_time: clickhouseStringDateSchema.nullish(),
completion_start_time: clickhouseStringDateSchema.nullish(),
event_ts: clickhouseStringDateSchema,
});
export type ObservationRecordReadType = z.infer<
typeof observationRecordReadSchema
>;
export const observationRecordInsertSchema = observationRecordBaseSchema.extend(
{
created_at: z.number(),
updated_at: z.number(),
start_time: z.number(),
end_time: z.number().nullish(),
completion_start_time: z.number().nullish(),
event_ts: z.number(),
}
);
export type ObservationRecordInsertType = z.infer<
typeof observationRecordInsertSchema
>;
export const traceRecordBaseSchema = z.object({
id: z.string(),
name: z.string().nullish(),
user_id: z.string().nullish(),
metadata: z.record(z.string()),
release: z.string().nullish(),
version: z.string().nullish(),
project_id: z.string(),
public: z.boolean(),
bookmarked: z.boolean(),
tags: z.array(z.string()),
input: z.string().nullish(),
output: z.string().nullish(),
session_id: z.string().nullish(),
});
export type TraceRecordBaseType = z.infer<typeof traceRecordBaseSchema>;
export const traceRecordReadSchema = traceRecordBaseSchema.extend({
timestamp: clickhouseStringDateSchema,
created_at: clickhouseStringDateSchema,
updated_at: clickhouseStringDateSchema,
event_ts: clickhouseStringDateSchema,
});
export type TraceRecordReadType = z.infer<typeof traceRecordReadSchema>;
export const traceRecordInsertSchema = traceRecordBaseSchema.extend({
timestamp: z.number(),
created_at: z.number(),
updated_at: z.number(),
event_ts: z.number(),
});
export type TraceRecordInsertType = z.infer<typeof traceRecordInsertSchema>;
export const scoreRecordBaseSchema = z.object({
id: z.string(),
project_id: z.string(),
trace_id: z.string(),
observation_id: z.string().nullish(),
name: z.string().nullish(),
value: z.union([z.number(), z.string()]).nullish(),
source: z.string(),
comment: z.string().nullish(),
author_user_id: z.string().nullish(),
config_id: z.string().nullish(),
data_type: z.enum(["NUMERIC", "CATEGORICAL", "BOOLEAN"]).nullish(),
string_value: z.string().nullish(),
});
export type ScoreRecordBaseType = z.infer<typeof scoreRecordBaseSchema>;
export const scoreRecordReadSchema = scoreRecordBaseSchema.extend({
created_at: clickhouseStringDateSchema,
updated_at: clickhouseStringDateSchema,
timestamp: clickhouseStringDateSchema,
event_ts: clickhouseStringDateSchema,
});
export type ScoreRecordReadType = z.infer<typeof scoreRecordReadSchema>;
export const scoreRecordInsertSchema = scoreRecordBaseSchema.extend({
created_at: z.number(),
updated_at: z.number(),
timestamp: z.number(),
event_ts: z.number(),
});
export type ScoreRecordInsertType = z.infer<typeof scoreRecordInsertSchema>;
export const convertTraceReadToInsert = (
record: TraceRecordReadType
): TraceRecordInsertType => {
return {
...record,
created_at: new Date(record.created_at).getTime(),
updated_at: new Date(record.created_at).getTime(),
timestamp: new Date(record.timestamp).getTime(),
event_ts: new Date(record.event_ts).getTime(),
};
};
export const convertObservationReadToInsert = (
record: ObservationRecordReadType
): ObservationRecordInsertType => {
return {
...record,
created_at: new Date(record.created_at).getTime(),
updated_at: new Date(record.created_at).getTime(),
start_time: new Date(record.start_time).getTime(),
end_time: record.end_time ? new Date(record.end_time).getTime() : undefined,
completion_start_time: record.completion_start_time
? new Date(record.completion_start_time).getTime()
: undefined,
event_ts: new Date(record.event_ts).getTime(),
};
};
export const convertScoreReadToInsert = (
record: ScoreRecordReadType
): ScoreRecordInsertType => {
return {
...record,
created_at: new Date(record.created_at).getTime(),
updated_at: new Date(record.updated_at).getTime(),
timestamp: new Date(record.timestamp).getTime(),
event_ts: new Date(record.event_ts).getTime(),
};
};
+16 -14
View File
@@ -26,7 +26,7 @@ const arrayOperatorReplacements = {
export function tableColumnsToSqlFilterAndPrefix(
filters: FilterState,
tableColumns: ColumnDefinition[],
table: TableNames
table: TableNames,
): Prisma.Sql {
const sql = tableColumnsToSqlFilter(filters, tableColumns, table);
if (sql === Prisma.empty) {
@@ -42,14 +42,14 @@ export function tableColumnsToSqlFilterAndPrefix(
export function tableColumnsToSqlFilter(
filters: FilterState,
tableColumns: ColumnDefinition[],
table: TableNames
table: TableNames,
): Prisma.Sql {
const internalFilters = filters.map((filter) => {
// Get column definition to map column to internal name, e.g. "t.id"
const col = tableColumns.find(
(c) =>
// TODO: Only use id instead of name
c.name === filter.column || c.id === filter.column
c.name === filter.column || c.id === filter.column,
);
if (!col) {
logger.error("Invalid filter column", filter.column);
@@ -71,13 +71,13 @@ export function tableColumnsToSqlFilter(
? Prisma.raw(
arrayOperatorReplacements[
filter.operator as keyof typeof arrayOperatorReplacements
]
],
)
: filter.operator in operatorReplacements
? Prisma.raw(
operatorReplacements[
filter.operator as keyof typeof operatorReplacements
]
],
)
: Prisma.raw(filter.operator); //checked by zod
@@ -97,19 +97,21 @@ export function tableColumnsToSqlFilter(
break;
case "stringOptions":
valuePrisma = Prisma.sql`(${Prisma.join(
filter.value.map((v) => Prisma.sql`${v}`)
filter.value.map((v) => Prisma.sql`${v}`),
)})`;
break;
case "arrayOptions":
valuePrisma = Prisma.sql`ARRAY[${Prisma.join(
filter.value.map((v) => Prisma.sql`${v}`),
", "
", ",
)}] `;
break;
case "boolean":
valuePrisma = Prisma.sql`${filter.value}`;
break;
case "null":
valuePrisma = Prisma.sql``;
break;
}
const jsonKeyPrisma =
filter.type === "stringObject" || filter.type === "numberObject"
@@ -123,12 +125,12 @@ export function tableColumnsToSqlFilter(
filter.type === "string" || filter.type === "stringObject"
? [
["contains", "does not contain", "ends with"].includes(
filter.operator
filter.operator,
)
? Prisma.raw("'%' || ")
: Prisma.empty,
["contains", "does not contain", "starts with"].includes(
filter.operator
filter.operator,
)
? Prisma.raw(" || '%'")
: Prisma.empty,
@@ -152,7 +154,7 @@ export function tableColumnsToSqlFilter(
const castValueToPostgresTypes = (
column: ColumnDefinition,
table: TableNames
table: TableNames,
) => {
return column.name === "type" &&
(table === "observations" ||
@@ -168,7 +170,7 @@ const dateOperators = filterOperators["datetime"];
export const datetimeFilterToPrismaSql = (
safeColumn: string,
operator: (typeof dateOperators)[number],
value: Date
value: Date,
) => {
if (!dateOperators.includes(operator)) {
throw new Error("Invalid operator: " + operator);
@@ -178,12 +180,12 @@ export const datetimeFilterToPrismaSql = (
}
return Prisma.sql`AND ${Prisma.raw(safeColumn)} ${Prisma.raw(
operator
operator,
)} ${value}::timestamp with time zone at time zone 'UTC'`;
};
export const datetimeFilterToPrisma = (
timestampFilter: z.infer<typeof timeFilter>
timestampFilter: z.infer<typeof timeFilter>,
) => {
const prismaTimestampFilter =
timestampFilter.operator === ">="
+13 -3
View File
@@ -1,25 +1,33 @@
export * from "./services/S3StorageService";
export * from "./services/StorageService";
export * from "./services/email/organizationInvitation/sendMembershipInvitationEmail";
export * from "./services/email/batchExportSuccess/sendBatchExportSuccessEmail";
export * from "./services/email/passwordReset/sendResetPasswordVerificationRequest";
export * from "./services/PromptService";
export * from "./services/traces-ui-table-service";
export * from "./auth/apiKeys";
export * from "./auth/customSsoProvider";
export * from "./llm/fetchLLMCompletion";
export * from "./llm/types";
export * from "./utils/DatabaseReadStream";
export * from "./utils/transforms";
export * from "./clickhouse";
export * from "../server/definitions";
export * from "./clickhouse/client";
export * from "./clickhouse/schemaUtils";
export * from "./clickhouse/schema";
export * from "./repositories/definitions";
export * from "../server/ingestion/types";
export * from "./ingestion/modelMatch";
export * from "./ingestion/processEventBatch";
export * from "../server/ingestion/types";
export * from "../server/ingestion/validateAndInflateScore";
export * from "./redis/redis";
export * from "./redis/traceUpsert";
export * from "./redis/CloudUsageMeteringQueue";
export * from "./redis/getQueue";
export * from "./redis/datasetRunItemUpsert";
export * from "./redis/batchExport";
export * from "./redis/legacyIngestion";
export * from "./redis/ingestionQueue";
export * from "./redis/experimentCreateQueue";
export * from "./auth/types";
export * from "./ingestion/legacy/index";
export * from "./queues";
@@ -29,3 +37,5 @@ export * from "./filterToPrisma";
export * from "./instrumentation";
export * from "./logger";
export * from "./queries";
export * from "./repositories";
export * from "./redis/evalExecutionQueue";
@@ -20,6 +20,9 @@ import { jsonSchema } from "../../../utils/zod";
import { prisma } from "../../../db";
import { LegacyIngestionAccessScope } from ".";
import { logger } from "../../logger";
import { env } from "../../../env";
import { upsertTrace } from "../../repositories";
import { convertDateToClickhouseDateTime } from "../../clickhouse/client";
export interface EventProcessor {
auth(apiScope: LegacyIngestionAccessScope): void;
@@ -131,19 +134,6 @@ export class ObservationProcessor implements EventProcessor {
})
: undefined;
const traceId =
!this.event.body.traceId && !existingObservation
? // Create trace if no traceid
(
await prisma.trace.create({
data: {
projectId: apiScope.projectId,
name: this.event.body.name,
},
})
).id
: this.event.body.traceId;
// Token counts
const [newInputCount, newOutputCount] =
"usage" in this.event.body
@@ -220,13 +210,56 @@ export class ObservationProcessor implements EventProcessor {
logger.warn("Prompt not found for observation", this.event.body);
}
const observationId = this.event.body.id ?? v4();
const observationId =
this.event.body.id ??
(() => {
const newId = v4();
logger.info(
`observation.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
let traceId = this.event.body?.traceId;
if (!this.event.body.traceId && !existingObservation) {
// Create trace if no traceId
traceId = observationId;
// Insert trace into postgres
await prisma.trace.upsert({
where: {
id: observationId,
},
create: {
projectId: apiScope.projectId,
name: this.event.body.name,
id: observationId,
timestamp: this.event.body.startTime || new Date(),
},
update: {},
});
if (env.CLICKHOUSE_URL) {
// Insert trace into clickhouse if enabled
await upsertTrace({
id: observationId,
project_id: apiScope.projectId,
timestamp: convertDateToClickhouseDateTime(
this.event.body.startTime
? new Date(this.event.body.startTime)
: new Date(),
),
created_at: convertDateToClickhouseDateTime(new Date()),
updated_at: convertDateToClickhouseDateTime(new Date()),
});
}
}
return {
id: observationId,
create: {
id: observationId,
traceId: traceId,
traceId,
type: type,
name: this.event.body.name,
startTime: this.event.body.startTime
@@ -359,7 +392,7 @@ export class ObservationProcessor implements EventProcessor {
text: body.input,
});
} else {
logger.info(
logger.debug(
`No input provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationInput = await prisma.observation.findFirst({
@@ -385,7 +418,7 @@ export class ObservationProcessor implements EventProcessor {
text: body.output,
});
} else {
logger.info(
logger.debug(
`No output provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationOutput = await prisma.observation.findFirst({
@@ -551,7 +584,15 @@ export class TraceProcessor implements EventProcessor {
this.auth(apiScope);
const internalId = body.id ?? v4();
const internalId =
body.id ??
(() => {
const newId = v4();
logger.info(
`trace.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
logger.debug(
`Trying to create trace, project ${apiScope.projectId}, id: ${internalId}`,
@@ -584,19 +625,32 @@ export class TraceProcessor implements EventProcessor {
: undefined;
if (body.sessionId) {
await prisma.traceSession.upsert({
where: {
id_projectId: {
try {
await prisma.traceSession.upsert({
where: {
id_projectId: {
id: body.sessionId,
projectId: apiScope.projectId,
},
},
create: {
id: body.sessionId,
projectId: apiScope.projectId,
},
},
create: {
id: body.sessionId,
projectId: apiScope.projectId,
},
update: {},
});
update: {},
});
} catch (e) {
if (
e instanceof Prisma.PrismaClientKnownRequestError &&
e.code === "P2002"
) {
logger.warn(
`Failed to upsert session. Session ${body.sessionId} in project ${apiScope.projectId} already exists`,
);
} else {
throw e;
}
}
}
// Do not use nested upserts or multiple where conditions as this should be a single native database upsert
@@ -663,7 +717,15 @@ export class ScoreProcessor implements EventProcessor {
this.auth(apiScope);
const id = body.id ?? v4();
const id =
body.id ??
(() => {
const newId = v4();
logger.info(
`score.id is null. Generating for projectId: ${apiScope.projectId}, id: ${newId}`,
);
return newId;
})();
const existingScore = await prisma.score.findFirst({
where: {
@@ -1,15 +1,9 @@
import { env } from "node:process";
import z from "zod";
import { ForbiddenError, UnauthorizedError } from "../../../errors";
import { eventTypes, ingestionApiSchema, ingestionEvent } from "../types";
import { eventTypes, ingestionApiSchema, IngestionEventType } from "../types";
import { getProcessorForEvent } from "./EventProcessor";
import { TraceUpsertEventType } from "../../queues";
import {
convertTraceUpsertEventsToRedisEvents,
TraceUpsertQueue,
} from "../../redis/traceUpsert";
import { ApiAccessScope } from "../../auth/types";
import { redis } from "../../redis/redis";
import { backOff } from "exponential-backoff";
import { Model } from "../../..";
import { logger } from "../../logger";
@@ -77,7 +71,7 @@ export const handleBatch = async (
// Decide how to handle the error: rethrow, continue, or push an error object to results
// For example, push an error object:
errors.push({
error: error,
error,
id: singleEvent.id,
type: singleEvent.type,
});
@@ -102,7 +96,7 @@ async function retry<T>(request: () => Promise<T>): Promise<T> {
}
const handleSingleEvent = async (
event: z.infer<typeof ingestionEvent>,
event: IngestionEventType,
apiScope: LegacyIngestionAccessScope,
calculateTokenDelegate: (p: {
model: Model;
@@ -122,11 +116,11 @@ const handleSingleEvent = async (
restEvent = rest;
}
logger.info(
logger.debug(
`handling single event ${event.id} of type ${event.type}: ${JSON.stringify({ body: restEvent })}`,
);
const cleanedEvent = ingestionEvent.parse(cleanEvent(event));
const cleanedEvent = cleanEvent(event) as IngestionEventType;
// Deny access to non-score events if the access level is not "all"
// This is an additional safeguard to auth checks in EventProcessor
@@ -163,42 +157,5 @@ export function cleanEvent(obj: unknown): unknown {
}
}
export const isNotNullOrUndefined = <T>(
val?: T | null,
): val is Exclude<T, null | undefined> => !isUndefinedOrNull(val);
export const isUndefinedOrNull = <T>(val?: T | null): val is undefined | null =>
val === undefined || val === null;
export const addTracesToTraceUpsertQueue = async (
batchResults: BatchResult[],
projectId: string,
): Promise<void> => {
const traceEvents: TraceUpsertEventType[] = batchResults
.filter((result) => result.type === eventTypes.TRACE_CREATE) // we only have create, no update.
.map((result) =>
result.result &&
typeof result.result === "object" &&
"id" in result.result
? // ingestion API only gets traces for one projectId
{ traceId: result.result.id as string, projectId }
: null,
)
.filter(isNotNullOrUndefined);
try {
if (env.NEXT_PUBLIC_LANGFUSE_CLOUD_REGION && redis) {
logger.info(`Sending ${traceEvents.length} events to worker via Redis`);
const queue = TraceUpsertQueue.getInstance();
if (!queue) {
logger.error("TraceUpsertQueue not initialized");
return;
}
await queue.addBulk(convertTraceUpsertEventsToRedisEvents(traceEvents));
}
} catch (error) {
logger.error("Error sending events to worker", error);
}
};
@@ -0,0 +1,396 @@
import { randomUUID } from "crypto";
import { z } from "zod";
import { type Model } from "../../db";
import { env } from "../../env";
import {
InvalidRequestError,
LangfuseNotFoundError,
UnauthorizedError,
} from "../../errors";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
import {
getCurrentSpan,
instrumentAsync,
instrumentSync,
recordIncrement,
traceException,
} from "../instrumentation";
import { logger } from "../logger";
import { LegacyIngestionEventType, QueueJobs } from "../queues";
import { IngestionQueue } from "../redis/ingestionQueue";
import { LegacyIngestionQueue } from "../redis/legacyIngestion";
import { redis } from "../redis/redis";
import { handleBatch } from "./legacy";
import {
StorageService,
StorageServiceFactory,
} from "../services/StorageService";
import { getProcessorForEvent } from "./legacy/EventProcessor";
import { eventTypes, ingestionEvent, IngestionEventType } from "./types";
export type TokenCountDelegate = (p: {
model: Model;
text: unknown;
}) => number | undefined;
let s3StorageServiceClient: StorageService;
const getS3StorageServiceClient = (bucketName: string): StorageService => {
if (!s3StorageServiceClient) {
s3StorageServiceClient = StorageServiceFactory.getInstance({
bucketName,
accessKeyId: env.LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID,
secretAccessKey: env.LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY,
endpoint: env.LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT,
region: env.LANGFUSE_S3_EVENT_UPLOAD_REGION,
forcePathStyle: env.LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE === "true",
});
}
return s3StorageServiceClient;
};
export const processEventBatch = async (
input: unknown[],
authCheck: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): Promise<{
successes: { id: string; status: number }[];
errors: {
id: string;
status: number;
message?: string;
error?: string;
}[];
}> => {
// add context of api call to the span
const currentSpan = getCurrentSpan();
recordIncrement("langfuse.ingestion.event", input.length);
currentSpan?.setAttribute("event_count", input.length);
/**************
* VALIDATION *
**************/
const validationErrors: { id: string; error: unknown }[] = [];
const authenticationErrors: { id: string; error: unknown }[] = [];
const batch: z.infer<typeof ingestionEvent>[] = input
.flatMap((event) => {
const parsed = instrumentSync(
{ name: "ingestion-zod-parse-individual-event" },
(span) => {
const parsedBody = ingestionEvent.safeParse(event);
if (parsedBody.data?.id !== undefined) {
span.setAttribute("object.id", parsedBody.data.id);
}
return parsedBody;
},
);
if (!parsed.success) {
validationErrors.push({
id:
typeof event === "object" && event && "id" in event
? typeof event.id === "string"
? event.id
: "unknown"
: "unknown",
error: new InvalidRequestError(parsed.error.message),
});
return [];
}
if (!isAuthorized(parsed.data, authCheck, tokenCountDelegate)) {
authenticationErrors.push({
id: parsed.data.id,
error: new UnauthorizedError("Access Scope Denied"),
});
return [];
}
return [parsed.data];
})
.flatMap((event) => {
if (event.type === eventTypes.SDK_LOG) {
// Log SDK_LOG events, but remove them from further processing
logger.info("SDK Log Event", { event });
return [];
}
return [event];
});
const sortedBatch = sortBatch(batch);
// We group events by eventBodyId which allows us to store and process them
// as one which reduces infra interactions per event. Only used in the S3 case.
const sortedBatchByEventBodyId = sortedBatch.reduce(
(
acc: Record<
string,
{
data: IngestionEventType[];
key: string;
eventBodyId: string;
type: (typeof eventTypes)[keyof typeof eventTypes];
}
>,
event,
) => {
if (!event.body?.id) {
return acc;
}
const key = `${getClickhouseEntityType(event.type)}-${event.body.id}`;
if (!acc[key]) {
acc[key] = {
data: [],
key: event.id,
type: event.type,
eventBodyId: event.body.id,
};
}
acc[key].data.push(event);
return acc;
},
{},
);
/********************
* ASYNC PROCESSING *
********************/
let s3UploadErrored = false;
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
await instrumentAsync({ name: "s3-upload-events" }, async () => {
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
throw new Error("S3 event store is enabled but no bucket is set");
}
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
// S3 Event Upload is currently blocking, but non-failing.
// If a promise rejects, we log it below, but do not throw an error.
// In this case, we upload the full batch into the Redis queue.
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) => {
// We upload the event in an array to the S3 bucket grouped by the eventBodyId.
// That way we batch updates from the same invocation into a single file and reduce
// write operations on S3.
const { data, key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${authCheck.scope.projectId}/${getClickhouseEntityType(type)}/${eventBodyId}/${key}.json`,
data,
);
}),
);
results.forEach((result) => {
if (result.status === "rejected") {
s3UploadErrored = true;
logger.error("Failed to upload event to S3", {
error: result.reason,
});
}
});
});
}
// Send each event individually to IngestionQueue for new processing
if (
env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" &&
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" &&
env.LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING === "true" &&
redis &&
!s3UploadErrored
) {
const queue = IngestionQueue.getInstance();
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) =>
queue
? queue.add(
QueueJobs.IngestionJob,
{
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.IngestionJob as const,
payload: {
data: {
type: sortedBatchByEventBodyId[id].type,
eventBodyId: sortedBatchByEventBodyId[id].eventBodyId,
},
authCheck,
},
},
{
delay: env.LANGFUSE_INGESTION_QUEUE_DELAY_MS,
},
)
: Promise.reject("Failed to instantiate queue"),
),
);
results.forEach((result) => {
if (result.status === "rejected") {
logger.error("Failed to add event to IngestionQueue", {
error: result.reason,
});
}
});
}
// As part of the legacy processing we sent the entire batch to the worker.
if (env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" && redis) {
const queue = LegacyIngestionQueue.getInstance();
if (queue) {
let addToQueueFailed = false;
const queuePayload: LegacyIngestionEventType =
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" && !s3UploadErrored
? {
data: Object.keys(sortedBatchByEventBodyId).map((id) => {
const { key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return {
type,
eventBodyId,
eventId: key,
};
}),
authCheck,
useS3EventStore: true,
}
: { data: sortedBatch, authCheck, useS3EventStore: false };
try {
await queue.add(QueueJobs.LegacyIngestionJob, {
payload: queuePayload,
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.LegacyIngestionJob as const,
});
} catch (e: unknown) {
logger.warn(
"Failed to add batch to queue, falling back to sync processing",
e,
);
addToQueueFailed = true;
}
if (!addToQueueFailed) {
return aggregateBatchResult(
// we are not sending additional server errors to the client in case of early return
[...validationErrors, ...authenticationErrors],
sortedBatch.map((event) => ({ id: event.id, result: event })),
);
}
} else {
logger.error(
"Ingestion queue not initialized, falling back to sync processing",
);
}
}
/*******************
* SYNC PROCESSING *
*******************/
const result = await handleBatch(sortedBatch, authCheck, tokenCountDelegate);
// in case we did not return early, we return the result here
return aggregateBatchResult(
[...validationErrors, ...authenticationErrors, ...result.errors],
result.results,
);
};
const isAuthorized = (
event: IngestionEventType,
authScope: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): boolean => {
try {
getProcessorForEvent(event, tokenCountDelegate).auth(authScope.scope);
return true;
} catch (error) {
return false;
}
};
/**
* Sorts a batch of ingestion events. Orders by: updating events last, sorted by timestamp asc.
*/
const sortBatch = (batch: Array<z.infer<typeof ingestionEvent>>) => {
const updateEvents: (typeof eventTypes)[keyof typeof eventTypes][] = [
eventTypes.GENERATION_UPDATE,
eventTypes.SPAN_UPDATE,
eventTypes.OBSERVATION_UPDATE, // legacy event type
];
const updates = batch
.filter((event) => updateEvents.includes(event.type))
.sort((a, b) => {
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
});
const others = batch
.filter((event) => !updateEvents.includes(event.type))
.sort((a, b) => {
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
});
// Return the array with non-update events first, followed by update events
return [...others, ...updates];
};
export const aggregateBatchResult = (
errors: Array<{ id: string; error: unknown }>,
results: Array<{ id: string; result: unknown }>,
) => {
const returnedErrors: {
id: string;
status: number;
message?: string;
error?: string;
}[] = [];
const successes: {
id: string;
status: number;
}[] = [];
errors.forEach((error) => {
if (error.error instanceof InvalidRequestError) {
returnedErrors.push({
id: error.id,
status: 400,
message: "Invalid request data",
error: error.error.message,
});
} else if (error.error instanceof UnauthorizedError) {
returnedErrors.push({
id: error.id,
status: 401,
message: "Authentication error",
error: error.error.message,
});
} else if (error.error instanceof LangfuseNotFoundError) {
returnedErrors.push({
id: error.id,
status: 404,
message: "Resource not found",
error: error.error.message,
});
} else {
returnedErrors.push({
id: error.id,
status: 500,
error: "Internal Server Error",
});
}
});
if (returnedErrors.length > 0) {
traceException(errors);
logger.error("Error processing events", returnedErrors);
}
results.forEach((result) => {
successes.push({
id: result.id,
status: 201,
});
});
return { successes, errors: returnedErrors };
};
@@ -3,7 +3,7 @@ import { z } from "zod";
import { NonEmptyString, jsonSchema } from "../../utils/zod";
import { ModelUsageUnit } from "../../constants";
import { ObservationLevel } from "@prisma/client";
import { ObservationLevel, ScoreSource } from "@prisma/client";
export const Usage = z.object({
input: z.number().int().nullish(),
@@ -163,6 +163,7 @@ const BaseScoreBody = z.object({
traceId: z.string(),
observationId: z.string().nullish(),
comment: z.string().nullish(),
source: z.nativeEnum(ScoreSource).default(ScoreSource.API),
});
/**
@@ -76,7 +76,7 @@ function inflateScoreBody(
const { body, projectId, scoreId, config } = params;
const relevantDataType = config?.dataType ?? body.dataType;
const scoreProps = { ...body, id: scoreId, projectId, source: "API" };
const scoreProps = { source: "API", ...body, id: scoreId, projectId };
if (typeof body.value === "number") {
if (relevantDataType && relevantDataType === ScoreDataType.BOOLEAN) {
@@ -1,10 +1,10 @@
import * as opentelemetry from "@opentelemetry/api";
import * as dd from "dd-trace";
import { env } from "../../env";
import {
CloudWatchClient,
PutMetricDataCommand,
} from "@aws-sdk/client-cloudwatch";
import * as opentelemetry from "@opentelemetry/api";
import * as dd from "dd-trace";
import { env } from "../../env";
import { logger } from "../logger";
// type CallbackFn<T> = () => T;
@@ -38,7 +38,7 @@ export async function instrumentAsync<T>(
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
ctx.name,
{
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
root: !ctx.traceContext && ctx.rootSpan,
kind: ctx.spanKind,
},
activeContext,
@@ -72,7 +72,7 @@ export function instrumentSync<T>(
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
ctx.name,
{
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
root: !ctx.traceContext && ctx.rootSpan,
kind: ctx.spanKind,
},
activeContext,
@@ -1,5 +1,7 @@
import type { ZodSchema } from "zod";
import { CallbackHandler } from "langfuse-langchain";
import { ChatAnthropic } from "@langchain/anthropic";
import { ChatBedrockConverse } from "@langchain/aws";
import {
@@ -17,11 +19,27 @@ import {
BedrockConfigSchema,
BedrockCredentialSchema,
} from "../../interfaces/customLLMProviderConfigSchemas";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import {
processEventBatch,
type TokenCountDelegate,
} from "../ingestion/processEventBatch";
import { logger } from "../logger";
import { ChatMessage, ChatMessageRole, LLMAdapter, ModelParams } from "./types";
import type { BaseCallbackHandler } from "@langchain/core/callbacks/base";
type ProcessTracedEvents = () => Promise<void>;
export type TraceParams = {
traceName: string;
traceId: string;
projectId: string;
tags: string[];
tokenCountDelegate: TokenCountDelegate;
authCheck: AuthHeaderValidVerificationResult;
};
type LLMCompletionParams = {
messages: ChatMessage[];
modelParams: ModelParams;
@@ -31,6 +49,7 @@ type LLMCompletionParams = {
apiKey: string;
maxRetries?: number;
config?: Record<string, string> | null;
traceParams?: TraceParams;
};
type FetchLLMCompletionParams = LLMCompletionParams & {
@@ -40,25 +59,34 @@ type FetchLLMCompletionParams = LLMCompletionParams & {
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: true;
}
): Promise<IterableReadableStream<Uint8Array>>;
},
): Promise<{
completion: IterableReadableStream<Uint8Array>;
processTracedEvents: ProcessTracedEvents;
}>;
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: false;
}
): Promise<string>;
},
): Promise<{ completion: string; processTracedEvents: ProcessTracedEvents }>;
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: false;
structuredOutputSchema: ZodSchema;
}
): Promise<unknown>;
},
): Promise<{
completion: unknown;
processTracedEvents: ProcessTracedEvents;
}>;
export async function fetchLLMCompletion(
params: FetchLLMCompletionParams
): Promise<string | IterableReadableStream<Uint8Array> | unknown> {
params: FetchLLMCompletionParams,
): Promise<{
completion: string | IterableReadableStream<Uint8Array> | unknown;
processTracedEvents: ProcessTracedEvents;
}> {
// the apiKey must never be printed to the console
const {
messages,
@@ -69,8 +97,39 @@ export async function fetchLLMCompletion(
baseURL,
maxRetries,
config,
traceParams,
} = params;
let finalCallbacks: BaseCallbackHandler[] | undefined = callbacks ?? [];
let processTracedEvents: ProcessTracedEvents = () => Promise.resolve();
if (traceParams) {
const handler = new CallbackHandler({
_projectId: traceParams.projectId,
_isLocalEventExportEnabled: true,
tags: traceParams.tags,
});
finalCallbacks.push(handler);
processTracedEvents = async () => {
try {
const events = await handler.langfuse._exportLocalEvents(
traceParams.projectId,
);
await processEventBatch(
JSON.parse(JSON.stringify(events)), // stringify to emulate network event batch from network call
traceParams.authCheck,
traceParams.tokenCountDelegate,
);
} catch (e) {
logger.error("Failed to process traced events", { error: e });
}
};
}
finalCallbacks = finalCallbacks.length > 0 ? finalCallbacks : undefined;
const finalMessages = messages.map((message) => {
if (message.role === ChatMessageRole.User)
return new HumanMessage(message.content);
@@ -89,7 +148,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
clientOptions: { maxRetries },
});
} else if (modelParams.adapter === LLMAdapter.OpenAI) {
@@ -99,7 +158,8 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
streamUsage: false, // https://github.com/langchain-ai/langchainjs/issues/6533
callbacks: finalCallbacks,
maxRetries,
configuration: {
baseURL,
@@ -114,7 +174,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
maxRetries,
});
} else if (modelParams.adapter === LLMAdapter.Bedrock) {
@@ -128,7 +188,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
maxRetries,
});
} else {
@@ -137,10 +197,19 @@ export async function fetchLLMCompletion(
throw new Error("This model provider is not supported.");
}
const runConfig = {
callbacks: finalCallbacks,
runId: traceParams?.traceId,
runName: traceParams?.traceName,
};
if (params.structuredOutputSchema) {
return await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages);
return {
completion: await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
/*
@@ -156,27 +225,41 @@ export async function fetchLLMCompletion(
Reference: https://platform.openai.com/docs/guides/reasoning/beta-limitations
*/
if (modelParams.model.startsWith("o1-")) {
return await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
})
.pipe(new StringOutputParser())
.invoke(
finalMessages.filter((message) => message._getType() !== "system")
);
return {
completion: await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
})
.pipe(new StringOutputParser())
.invoke(
finalMessages.filter((message) => message._getType() !== "system"),
runConfig,
),
processTracedEvents,
};
}
if (streaming) {
return chatModel.pipe(new BytesOutputParser()).stream(finalMessages);
return {
completion: await chatModel
.pipe(new BytesOutputParser())
.stream(finalMessages, runConfig),
processTracedEvents,
};
}
return await chatModel.pipe(new StringOutputParser()).invoke(finalMessages);
return {
completion: await chatModel
.pipe(new StringOutputParser())
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
+29 -2
View File
@@ -26,6 +26,20 @@ export enum ChatMessageRole {
export const ChatMessageDefaultRoleSchema = z.nativeEnum(ChatMessageRole);
const ChatMessageSchema = z.object({
role: z.union([ChatMessageDefaultRoleSchema, z.string()]), // Users may ingest any string as role via API/SDK
content: z.string(),
});
export const ChatMessageListSchema = z.array(ChatMessageSchema);
export const TextPromptSchema = z.string().min(1, "Enter a prompt");
export const PromptContentSchema = z.union([
ChatMessageListSchema,
TextPromptSchema,
]);
export type PromptContent = z.infer<typeof PromptContentSchema>;
export type ModelParams = {
provider: string;
adapter: LLMAdapter;
@@ -49,7 +63,18 @@ export const ZodModelConfig = z.object({
top_p: z.coerce.number().optional(),
});
// NOTE: Update docs page when changing this!
// Experiment config
export const ExperimentMetadataSchema = z
.object({
prompt_id: z.string(),
provider: z.string(),
model: z.string(),
model_params: ZodModelConfig,
})
.strict();
export type ExperimentMetadata = z.infer<typeof ExperimentMetadataSchema>;
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
export const openAIModels = [
"gpt-4o",
"gpt-4o-2024-08-06",
@@ -76,11 +101,13 @@ export const openAIModels = [
export type OpenAIModel = (typeof openAIModels)[number];
// NOTE: Update docs page when changing this!
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
export const anthropicModels = [
"claude-3-5-sonnet-20241022",
"claude-3-5-sonnet-20240620",
"claude-3-opus-20240229",
"claude-3-sonnet-20240229",
"claude-3-5-haiku-20241022",
"claude-3-haiku-20240307",
"claude-2.1",
"claude-2.0",
@@ -0,0 +1,411 @@
import { filterOperators } from "../../../interfaces/filters";
import { clickhouseCompliantRandomCharacters } from "../../repositories";
export type ClickhouseOperator =
| (typeof filterOperators)[keyof typeof filterOperators][number]
| "!=";
export interface Filter {
apply(): ClickhouseFilter;
clickhouseTable: string;
operator: ClickhouseOperator;
field: string;
}
type ClickhouseFilter = {
query: string;
params: { [x: string]: any } | {};
};
export class StringFilter implements Filter {
public clickhouseTable: string;
public field: string;
public value: string;
public operator: (typeof filterOperators)["string"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["string"][number];
value: string;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
const varName = `stringFilter${clickhouseCompliantRandomCharacters()}`;
const fieldWithPrefix = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
let query: string;
switch (this.operator) {
case "=":
query = `${fieldWithPrefix} = {${varName}: String}`;
break;
case "contains":
query = `position(${fieldWithPrefix}, {${varName}: String}) > 0`;
break;
case "does not contain":
query = `position(${fieldWithPrefix}, {${varName}: String}) = 0`;
break;
case "starts with":
query = `startsWith(${fieldWithPrefix}, {${varName}: String})`;
break;
case "ends with":
query = `endsWith(${fieldWithPrefix}, {${varName}: String})`;
break;
default:
throw new Error(`Unsupported operator: ${this.operator}`);
}
return {
query: query,
params: { [varName]: this.value },
};
}
}
export class NumberFilter implements Filter {
public clickhouseTable: string;
public field: string;
public value: number;
public operator: (typeof filterOperators)["number"][number] | "!=";
public clickhouseTypeOverwrite?: string;
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["number"][number] | "!=";
value: number;
tablePrefix?: string;
clickhouseTypeOverwrite?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
this.clickhouseTypeOverwrite = opts.clickhouseTypeOverwrite;
}
apply(): ClickhouseFilter {
const uid = clickhouseCompliantRandomCharacters();
const varName = `numberFilter${uid}`;
const type = this.clickhouseTypeOverwrite ?? "Decimal64(12)";
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: ${type}}`,
params: { [varName]: this.value.toString() },
};
}
}
export class DateTimeFilter implements Filter {
public clickhouseTable: string;
public field: string;
public value: Date;
public operator: (typeof filterOperators)["datetime"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["datetime"][number];
value: Date;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
const uid = clickhouseCompliantRandomCharacters();
const varName = `dateTimeFilter${uid}`;
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: DateTime64(3)}`,
params: { [varName]: new Date(this.value).getTime() },
};
}
}
export class StringOptionsFilter implements Filter {
public clickhouseTable: string;
public field: string;
public values: string[];
public operator: (typeof filterOperators.stringOptions)[number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators.stringOptions)[number];
values: string[];
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.values = opts.values;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
const uid = clickhouseCompliantRandomCharacters();
const varName = `stringOptionsFilter${uid}`;
return {
query:
this.operator === "any of"
? `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`
: `has({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`,
params: { [varName]: this.values },
};
}
}
// stringObject filter is used when we want to filter on a key value pair in a clickhouse map.
// As we use the MAP form clickhouse, we can only filter efficiently on the first level of a json obj.
export class StringObjectFilter implements Filter {
public clickhouseTable: string;
public field: string;
public key: string;
public value: string;
public operator: (typeof filterOperators)["stringObject"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["stringObject"][number];
key: string;
value: string;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
this.key = opts.key;
}
apply(): ClickhouseFilter {
const varKeyName = `stringObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
const varValueName = `stringObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
// const query: `${column}['{varKeyName: String}'] ${this.operator} {${varValueName}: String}`,
let query: string;
switch (this.operator) {
case "=":
query = `${column}[{${varKeyName}: String}] = {${varValueName}: String}`;
break;
case "contains":
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) > 0`;
break;
case "does not contain":
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) = 0`;
break;
case "starts with":
query = `startsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
break;
case "ends with":
query = `endsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
break;
default:
throw new Error(`Unsupported operator: ${this.operator}`);
}
return {
query,
params: { [varKeyName]: this.key, [varValueName]: this.value },
};
}
}
// this is used when we want to filter multiple values on a clickhouse column which is also an array
export class ArrayOptionsFilter implements Filter {
public clickhouseTable: string;
public field: string;
public values: string[];
public operator: (typeof filterOperators.arrayOptions)[number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators.arrayOptions)[number];
values: string[];
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.values = opts.values;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
const uid = clickhouseCompliantRandomCharacters();
const varName = `arrayOptionsFilter${uid}`;
let query: string;
switch (this.operator) {
case "any of":
query = `hasAny({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`;
break;
case "none of":
query = `hasAny({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`;
break;
case "all of":
query = `hasAll(${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}, {${varName}: Array(String)}) = True`;
break;
default:
throw new Error(`Unsupported operator: ${this.operator}`);
}
return {
query,
params: { [varName]: this.values },
};
}
}
export class NullFilter implements Filter {
public clickhouseTable: string;
public field: string;
public operator: (typeof filterOperators)["null"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["null"][number];
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator}`,
params: {},
};
}
}
export class NumberObjectFilter implements Filter {
public clickhouseTable: string;
public field: string;
public key: string;
public value: number;
public operator: (typeof filterOperators)["numberObject"][number] | "!=";
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["numberObject"][number] | "!=";
key: string;
value: number;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
this.key = opts.key;
}
apply(): ClickhouseFilter {
const varKeyName = `numberObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
const varValueName = `numberObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
return {
query: `empty(arrayFilter(x -> (((x.1) = {${varKeyName}: String}) AND ((x.2) ${this.operator} {${varValueName}: Decimal64(12)})), ${column})) = 0`,
params: { [varKeyName]: this.key, [varValueName]: this.value },
};
}
}
export class BooleanFilter implements Filter {
public clickhouseTable: string;
public field: string;
public operator: (typeof filterOperators)["boolean"][number];
public value: boolean;
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["boolean"][number];
value: boolean;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.tablePrefix = opts.tablePrefix;
this.operator = opts.operator;
}
apply(): ClickhouseFilter {
const uid = clickhouseCompliantRandomCharacters();
const varName = `booleanFilter${uid}`;
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: Boolean}`,
params: { [varName]: this.value },
};
}
}
export class FilterList {
private filters: Filter[];
constructor(filters: Filter[] = []) {
this.filters = filters;
}
push(...filter: Filter[]) {
this.filters.push(...filter);
}
find(predicate: (filter: Filter) => boolean) {
return this.filters.find(predicate);
}
length() {
return this.filters.length;
}
public apply(): ClickhouseFilter {
if (this.filters.length === 0) {
return {
query: "",
params: {},
};
}
const compiledQueries = this.filters.map((filter) => filter.apply());
const { params, queries } = compiledQueries.reduce(
(acc, { params, query }) => {
acc.params = { ...acc.params, ...params };
acc.queries.push(query);
return acc;
},
{ params: {}, queries: [] as string[] },
);
return {
query: queries.join(" AND "),
params,
};
}
}
@@ -0,0 +1,182 @@
import z from "zod";
import { singleFilter } from "../../../interfaces/filters";
import { FilterCondition } from "../../../types";
import { isValidTableName } from "../../clickhouse/schemaUtils";
import { logger } from "../../logger";
import { UiColumnMapping } from "../../../tableDefinitions";
import {
StringFilter,
DateTimeFilter,
StringOptionsFilter,
FilterList,
NumberFilter,
ArrayOptionsFilter,
BooleanFilter,
NumberObjectFilter,
StringObjectFilter,
NullFilter,
} from "./clickhouse-filter";
export class QueryBuilderError extends Error {
constructor(message: string) {
super(message);
this.name = "QueryBuilderError";
}
}
// This function ensures that the user only selects valid columns from the clickhouse schema.
// The filter property in this column needs to be zod verified.
// User input for values (e.g. project_id = <value>) are sent to Clickhouse as parameters to prevent SQL injection
export const createFilterFromFilterState = (
filter: FilterCondition[],
columnMapping: UiColumnMapping[],
) => {
return filter.map((frontEndFilter) => {
// checks if the column exists in the clickhouse schema
const column = matchAndVerifyTracesUiColumn(frontEndFilter, columnMapping);
switch (frontEndFilter.type) {
case "string":
return new StringFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "datetime":
return new DateTimeFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "stringOptions":
return new StringOptionsFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
values: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "number":
return new NumberFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
clickhouseTypeOverwrite: column.clickhouseTypeOverwrite,
});
case "arrayOptions":
return new ArrayOptionsFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
values: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "boolean":
return new BooleanFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
value: frontEndFilter.value,
operator: frontEndFilter.operator,
tablePrefix: column.queryPrefix,
});
case "numberObject":
return new NumberObjectFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
key: frontEndFilter.key,
operator: frontEndFilter.operator,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "stringObject":
return new StringObjectFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
key: frontEndFilter.key,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "null":
return new NullFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
tablePrefix: column.queryPrefix,
});
default:
const exhaustiveCheck: never = frontEndFilter;
logger.error(`Invalid filter type: ${JSON.stringify(exhaustiveCheck)}`);
throw new QueryBuilderError(`Invalid filter type`);
}
});
};
const matchAndVerifyTracesUiColumn = (
filter: z.infer<typeof singleFilter>,
uiTableDefinitions: UiColumnMapping[],
) => {
// tries to match the column name to the clickhouse table name
logger.debug(`Filter to match: ${JSON.stringify(filter)}`);
const uiTable = uiTableDefinitions.find(
(col) =>
col.uiTableName === filter.column || col.uiTableId === filter.column, // matches on the NAME of the column in the UI.
);
if (!uiTable) {
throw new QueryBuilderError(
`Column ${filter.column} does not match a UI / CH table mapping.`,
);
}
if (!isValidTableName(uiTable.clickhouseTableName)) {
throw new QueryBuilderError(
`Invalid clickhouse table name: ${uiTable.clickhouseTableName}`,
);
}
return uiTable;
};
export function getProjectIdDefaultFilter(
projectId: string,
opts: { tracesPrefix: string },
): {
tracesFilter: FilterList;
scoresFilter: FilterList;
observationsFilter: FilterList;
} {
return {
tracesFilter: new FilterList([
new StringFilter({
clickhouseTable: "traces",
field: "project_id",
operator: "=",
value: projectId,
tablePrefix: opts.tracesPrefix,
}),
]),
scoresFilter: new FilterList([
new StringFilter({
clickhouseTable: "scores",
field: "project_id",
operator: "=",
value: projectId,
}),
]),
observationsFilter: new FilterList([
new StringFilter({
clickhouseTable: "observations",
field: "project_id",
operator: "=",
value: projectId,
}),
]),
};
}
@@ -0,0 +1,33 @@
import z from "zod";
import { OrderByState } from "../../../interfaces/orderBy";
import { UiColumnMapping } from "../../../tableDefinitions";
import { logger } from "../../logger";
export function orderByToClickhouseSql(
orderBy: OrderByState,
tableColumns: UiColumnMapping[],
): string {
if (!orderBy) {
return "";
}
// Get column definition to map column to internal name, e.g. "t.id"
const col = tableColumns.find(
(c) => c.uiTableName === orderBy.column || c.uiTableId === orderBy.column,
);
if (!col) {
logger.warn("Invalid order by column", orderBy.column);
throw new Error("Invalid order by column: " + orderBy.column);
}
// Assert that orderBy.order is either "asc" or "desc"
const orderByOrder = z.enum(["ASC", "DESC"]);
const order = orderByOrder.safeParse(orderBy.order);
if (!order.success) {
logger.warn("Invalid order", orderBy.order);
throw new Error("Invalid order: " + orderBy.order);
}
// Both column and order are safe, can use raw SQL
return `ORDER BY ${col.queryPrefix ? col.queryPrefix + "." : ""}${col.clickhouseSelect} ${order.data}`;
}
@@ -0,0 +1,20 @@
const regexIndefiniteCharacters = "%";
export const clickhouseSearchCondition = (query?: string) => {
return {
query: query
? `
AND (
id ILIKE {searchString: String} OR
user_id ILIKE {searchString: String} OR
name ILIKE {searchString: String}
)
`
: "",
params: query
? {
searchString: `${regexIndefiniteCharacters}${query}${regexIndefiniteCharacters}`,
}
: {},
};
};
@@ -9,12 +9,10 @@ import { TableFilters } from "./types";
type AdditionalObservationFields = {
traceName: string | null;
promptName: string | null;
promptVersion: string | null;
traceTags: Array<string>;
};
type FullObservation = AdditionalObservationFields & ObservationView;
export type FullObservation = AdditionalObservationFields & ObservationView;
export type FullObservations = Array<FullObservation>;
@@ -40,17 +38,17 @@ export function parseGetAllGenerationsInput(filters: TableFilters) {
const filterCondition = tableColumnsToSqlFilterAndPrefix(
filters.filter ?? [],
observationsTableCols,
"observations"
"observations",
);
const orderByCondition = orderByToPrismaSql(
filters.orderBy,
observationsTableCols
observationsTableCols,
);
// to improve query performance, add timeseries filter to observation queries as well
const startTimeFilter = filters.filter?.find(
(f) => f.column === "Start Time" && f.type === "datetime"
(f) => f.column === "Start Time" && f.type === "datetime",
);
const datetimeFilter =
@@ -58,7 +56,7 @@ export function parseGetAllGenerationsInput(filters: TableFilters) {
? datetimeFilterToPrismaSql(
"start_time",
startTimeFilter.operator,
startTimeFilter.value
startTimeFilter.value,
)
: Prisma.empty;
@@ -4,20 +4,20 @@ import {
datetimeFilterToPrismaSql,
tableColumnsToSqlFilterAndPrefix,
} from "../filterToPrisma";
import { tracesTableCols } from "../../tracesTable";
import { tracesTableCols } from "../../tableDefinitions/tracesTable";
import { orderByToPrismaSql } from "../orderByToPrisma";
export function parseTraceAllFilters(input: TableFilters) {
const filterCondition = tableColumnsToSqlFilterAndPrefix(
input.filter ?? [],
tracesTableCols,
"traces"
"traces",
);
const orderByCondition = orderByToPrismaSql(input.orderBy, tracesTableCols);
// to improve query performance, add timeseries filter to observation queries as well
const timeseriesFilter = input.filter?.find(
(f) => f.column === "Timestamp" && f.type === "datetime"
(f) => f.column === "Timestamp" && f.type === "datetime",
);
const observationTimeseriesFilter =
@@ -25,7 +25,7 @@ export function parseTraceAllFilters(input: TableFilters) {
? datetimeFilterToPrismaSql(
"start_time",
timeseriesFilter.operator,
timeseriesFilter.value
timeseriesFilter.value,
)
: Prisma.empty;
@@ -130,6 +130,6 @@ export function createTracesQuery({
${filterCondition}
${orderByCondition}
${limit ? Prisma.sql`LIMIT ${limit}` : Prisma.empty}
${page && limit ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
${page !== undefined && limit !== undefined ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
`;
}
@@ -7,3 +7,16 @@ export {
type FullObservationsWithScores,
type IOAndMetadataOmittedObservations,
} from "./createGenerationsQuery";
export {
FilterList,
StringFilter,
DateTimeFilter,
StringOptionsFilter,
NumberFilter,
ArrayOptionsFilter,
BooleanFilter,
NumberObjectFilter,
StringObjectFilter,
NullFilter,
type ClickhouseOperator,
} from "./clickhouse-sql/clickhouse-filter";
+40
View File
@@ -7,6 +7,7 @@ export enum EventName {
EvaluationExecution = "EvaluationExecution",
LegacyIngestion = "LegacyIngestion",
CloudUsageMetering = "CloudUsageMetering",
ExperimentCreate = "ExperimentCreate",
}
export const LegacyIngestionEventFull = z.object({
@@ -66,17 +67,36 @@ export const TraceUpsertEventSchema = z.object({
projectId: z.string(),
traceId: z.string(),
});
export const DatasetRunItemUpsertEventSchema = z.object({
projectId: z.string(),
datasetItemId: z.string(),
traceId: z.string(),
observationId: z.string().optional(),
});
export const EvalExecutionEvent = z.object({
projectId: z.string(),
jobExecutionId: z.string(),
delay: z.number().nullish(),
});
export const ExperimentCreateEventSchema = z.object({
projectId: z.string(),
datasetId: z.string(),
runId: z.string(),
description: z.string().optional(),
});
export type BatchExportJobType = z.infer<typeof BatchExportJobSchema>;
export type TraceUpsertEventType = z.infer<typeof TraceUpsertEventSchema>;
export type DatasetRunItemUpsertEventType = z.infer<
typeof DatasetRunItemUpsertEventSchema
>;
export type EvalExecutionEventType = z.infer<typeof EvalExecutionEvent>;
export type LegacyIngestionEventType = z.infer<typeof LegacyIngestionEvent>;
export type IngestionEventQueueType = z.infer<typeof IngestionEvent>;
export type ExperimentCreateEventType = z.infer<
typeof ExperimentCreateEventSchema
>;
export const EventBodySchema = z.union([
z.object({
@@ -91,26 +111,34 @@ export const EventBodySchema = z.union([
name: z.literal(EventName.BatchExport),
payload: BatchExportJobSchema,
}),
z.object({
name: z.literal(EventName.ExperimentCreate),
payload: ExperimentCreateEventSchema,
}),
]);
export type EventBodyType = z.infer<typeof EventBodySchema>;
export enum QueueName {
TraceUpsert = "trace-upsert", // Ingestion pipeline adds events on each Trace upsert
EvaluationExecution = "evaluation-execution-queue", // Worker executes Evals
DatasetRunItemUpsert = "dataset-run-item-upsert-queue",
BatchExport = "batch-export-queue",
IngestionQueue = "ingestion-queue", // Process single events with S3-merge
LegacyIngestionQueue = "legacy-ingestion-queue", // Used for batch processing of Ingestion
CloudUsageMeteringQueue = "cloud-usage-metering-queue",
ExperimentCreate = "experiment-create-queue",
}
export enum QueueJobs {
TraceUpsert = "trace-upsert",
DatasetRunItemUpsert = "dataset-run-item-upsert",
EvaluationExecution = "evaluation-execution-job",
BatchExportJob = "batch-export-job",
EnqueueBatchExportJobs = "enqueue-batch-export-jobs",
LegacyIngestionJob = "legacy-ingestion-job",
CloudUsageMeteringJob = "cloud-usage-metering-job",
IngestionJob = "ingestion-job",
ExperimentCreateJob = "experiment-create-job",
}
export type TQueueJobTypes = {
@@ -120,6 +148,12 @@ export type TQueueJobTypes = {
payload: TraceUpsertEventType;
name: QueueJobs.TraceUpsert;
};
[QueueName.DatasetRunItemUpsert]: {
timestamp: Date;
id: string;
payload: DatasetRunItemUpsertEventType;
name: QueueJobs.DatasetRunItemUpsert;
};
[QueueName.EvaluationExecution]: {
timestamp: Date;
id: string;
@@ -144,4 +178,10 @@ export type TQueueJobTypes = {
payload: IngestionEventQueueType;
name: QueueJobs.IngestionJob;
};
[QueueName.ExperimentCreate]: {
timestamp: Date;
id: string;
payload: ExperimentCreateEventType;
name: QueueJobs.ExperimentCreateJob;
};
};
@@ -0,0 +1,61 @@
import { Queue } from "bullmq";
import { env } from "../..";
import { logger } from "@azure/storage-blob";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class CloudUsageMeteringQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (!env.STRIPE_SECRET_KEY) {
return null;
}
if (CloudUsageMeteringQueue.instance) {
return CloudUsageMeteringQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
CloudUsageMeteringQueue.instance = newRedis
? new Queue(QueueName.CloudUsageMeteringQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
CloudUsageMeteringQueue.instance?.on("error", (err) => {
logger.error("CloudUsageMeteringQueue error", err);
});
if (CloudUsageMeteringQueue.instance) {
CloudUsageMeteringQueue.instance.add(
QueueJobs.CloudUsageMeteringJob,
{},
{
repeat: { pattern: "5 * * * *" },
},
);
CloudUsageMeteringQueue.instance.add(
QueueJobs.CloudUsageMeteringJob,
{},
{},
);
}
return CloudUsageMeteringQueue.instance;
}
}
@@ -0,0 +1,47 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class DatasetRunItemUpsertQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.DatasetRunItemUpsert]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.DatasetRunItemUpsert]
> | null {
if (DatasetRunItemUpsertQueue.instance)
return DatasetRunItemUpsertQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
DatasetRunItemUpsertQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.DatasetRunItemUpsert]>(
QueueName.DatasetRunItemUpsert,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 5,
delay: 30_000, // 30 seconds
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
DatasetRunItemUpsertQueue.instance?.on("error", (err) => {
logger.error("DatasetRunItemUpsertQueue error", err);
});
return DatasetRunItemUpsertQueue.instance;
}
}
@@ -0,0 +1,45 @@
import { Queue } from "bullmq";
import { logger } from "../logger";
import { TQueueJobTypes, QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class EvalExecutionQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.EvaluationExecution]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.EvaluationExecution]
> | null {
if (EvalExecutionQueue.instance) return EvalExecutionQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
EvalExecutionQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.EvaluationExecution]>(
QueueName.EvaluationExecution,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 10,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
EvalExecutionQueue.instance?.on("error", (err) => {
logger.error("EvalExecutionQueue error", err);
});
return EvalExecutionQueue.instance;
}
}
@@ -0,0 +1,45 @@
import { Queue } from "bullmq";
import { logger } from "../logger";
import { TQueueJobTypes, QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class ExperimentCreateQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.ExperimentCreate]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.ExperimentCreate]
> | null {
if (ExperimentCreateQueue.instance) return ExperimentCreateQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
ExperimentCreateQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.ExperimentCreate]>(
QueueName.ExperimentCreate,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 2,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
ExperimentCreateQueue.instance?.on("error", (err) => {
logger.error("ExperimentCreateQueue error", err);
});
return ExperimentCreateQueue.instance;
}
}
@@ -0,0 +1,34 @@
import { Queue } from "bullmq";
import { QueueName } from "../queues";
import { BatchExportQueue } from "./batchExport";
import { CloudUsageMeteringQueue } from "./CloudUsageMeteringQueue";
import { DatasetRunItemUpsertQueue } from "./datasetRunItemUpsert";
import { EvalExecutionQueue } from "./evalExecutionQueue";
import { ExperimentCreateQueue } from "./experimentCreateQueue";
import { IngestionQueue } from "./ingestionQueue";
import { LegacyIngestionQueue } from "./legacyIngestion";
import { TraceUpsertQueue } from "./traceUpsert";
export function getQueue(queueName: QueueName): Queue | null {
switch (queueName) {
case QueueName.LegacyIngestionQueue:
return LegacyIngestionQueue.getInstance();
case QueueName.BatchExport:
return BatchExportQueue.getInstance();
case QueueName.CloudUsageMeteringQueue:
return CloudUsageMeteringQueue.getInstance();
case QueueName.DatasetRunItemUpsert:
return DatasetRunItemUpsertQueue.getInstance();
case QueueName.EvaluationExecution:
return EvalExecutionQueue.getInstance();
case QueueName.ExperimentCreate:
return ExperimentCreateQueue.getInstance();
case QueueName.TraceUpsert:
return TraceUpsertQueue.getInstance();
case QueueName.IngestionQueue:
return IngestionQueue.getInstance();
default:
const exhaustiveCheckDefault: never = queueName;
throw new Error(`Queue ${queueName} not found`);
}
}
@@ -32,6 +32,7 @@ export class TraceUpsertQueue {
removeOnComplete: 100, // Important: If not true, new jobs for that ID would be ignored as jobs in the complete set are still considered as part of the queue
removeOnFail: 100_000,
attempts: 5,
delay: 10_000, // 10 seconds
backoff: {
type: "exponential",
delay: 5000,
@@ -48,43 +49,3 @@ export class TraceUpsertQueue {
return TraceUpsertQueue.instance;
}
}
export function convertTraceUpsertEventsToRedisEvents(
events: TraceUpsertEventType[],
) {
const uniqueTracesPerProject = events.reduce((acc, event) => {
if (!acc.get(event.projectId)) {
acc.set(event.projectId, new Set());
}
acc.get(event.projectId)?.add(event.traceId);
return acc;
}, new Map<string, Set<string>>());
return [...uniqueTracesPerProject.entries()]
.map((tracesPerProject) => {
const [projectId, traceIds] = tracesPerProject;
return [...traceIds].map((traceId) => ({
name: QueueJobs.TraceUpsert,
data: {
payload: {
projectId,
traceId,
},
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.TraceUpsert as const,
},
opts: {
removeOnFail: 1_000,
removeOnComplete: true,
attempts: 5,
backoff: {
type: "exponential",
delay: 1000,
},
},
}));
})
.flat();
}

Some files were not shown because too many files have changed in this diff Show More