Compare commits

..
245 Commits
Author SHA1 Message Date
Max Deichmann e21102664e chore: release v2.95.2 2025-02-15 13:29:57 +01:00
Max DeichmannandGitHub d31b0eaddc security: upgrade dompurify v2 (#5570)
security: upgrade dompurify
2025-02-15 12:26:51 +00:00
Max Deichmann f53ad4de5c chore: release v2.95.1
Codespell / Check for spelling errors (push) Waiting to run
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-02-11 14:19:25 +01:00
Max DeichmannandGitHub 053d7d668d security: upgrade sentry 8.52.0 (#5477)
fix
2025-02-11 14:18:38 +01:00
Max DeichmannandGitHub 21e3ed2b39 security: upgrade clickhouse migration package (#5478)
push
2025-02-11 14:18:26 +01:00
Max DeichmannandGitHub 16ca4e9293 security: upgrade json path (#5475) 2025-02-11 13:44:51 +01:00
Marc KlingenandGitHub 75f82be88d ci(v2): run codespell also on v2 branch and prs (#5224) (#5225) 2025-01-27 13:25:31 +01:00
Marc Klingen 22f6a02b08 chore: release v2.95.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-27 13:16:16 +01:00
Marc KlingenandGitHub 11aa1dbbb1 feat(v2-auth): make checks and auth method configurable across SSO providers (#5203) (#5219) 2025-01-27 13:15:33 +01:00
steffen911 d961d85a28 chore: release v2.94.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-24 16:35:20 +01:00
Steffen SchmitzandGitHub d0c1ad5144 feat: add proxy support for oauth flows (#5198) (#5201)
(cherry picked from commit c442c4290e)
2025-01-24 16:34:53 +01:00
Marc Klingen 0d30b2fe83 chore: release v2.93.9
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-24 14:12:59 +01:00
Marc KlingenandGitHub f33bae6683 feat(auth-v2): Add AUTH_CUSTOM_ID_TOKEN environment variable (#5193) (#5196) 2025-01-24 14:12:00 +01:00
Baptiste Mille-MathiasandGitHub eaa0df125b feat: add support for DATABASE_ARGS config (cherry-pick) (#5152) 2025-01-24 13:50:17 +01:00
Max Deichmann b0e01b7127 chore: release v2.93.8
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2025-01-06 10:54:29 +01:00
Max DeichmannandGitHub 2a421e7406 security: upgrade next (#4891)
push
2025-01-06 10:45:23 +01:00
Marc Klingen 23150b68db chore: release v2.93.7
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-19 00:57:59 +01:00
Ildar IapparovandMarc Klingen e5c46010a4 feat(auth): add AUTH_IGNORE_ACCOUNT_FIELDS to sanitize IDP fields before creating an account (#4728)
* feat: Field sanitization before creating an Account

* add comments

---------

Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-19 00:57:07 +01:00
Marc Klingen b2bf68d7a4 chore: release v2.93.6
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-13 02:57:59 +01:00
Marc Klingen 31cec4f5c9 feat: in HF Spaces, prompt opening in new tab when running in iframe (#4713) 2024-12-13 02:45:32 +01:00
Max Deichmann 84a0ad8dfb chore: release v2.93.5
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-12 14:06:33 +01:00
Max DeichmannandGitHub 69466fd43b security: upgrade next-auth (#4702) 2024-12-12 14:06:12 +01:00
Marc Klingen 324e078c85 chore: release v2.93.4
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-04 00:07:38 +01:00
Marc Klingen 7385fc4529 fix: disable x frame options header on Hugging Face (#4558) 2024-12-04 00:07:02 +01:00
Marc Klingen 66d1fa427f chore: release v2.93.3
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 23:37:22 +01:00
bee396a433 feat(auth): add KeyCloak authentication option (#2866)
---------

Co-authored-by: RTae <natthanan.bhu@doctorasa.co>
Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-03 23:35:01 +01:00
jay0129andMarc Klingen 8727a52931 feat(auth): add GitHub Enterprise Authentication Provider (#4463)
feat: Add GitHub Enterprise Authentication Provider

Co-authored-by: Marc Klingen <git@marcklingen.com>
2024-12-03 23:34:31 +01:00
Marc Klingen ed5c076a5a fix(auth): add nonce check for Cognito NextAuth provider (#4401) 2024-12-03 23:34:15 +01:00
Marc Klingen db5c575ae0 chore: release v2.93.2
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 17:53:55 +01:00
Marc Klingen 2a0f482578 fix: remove LANGFUSE_CSP_DISABLE (did not work) and disable csp headers on HF Spaces (#4545) 2024-12-03 17:51:50 +01:00
Marc Klingen c041cf371a chore: release v2.93.1
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-12-03 01:25:53 +01:00
Marc Klingen c6daf09cd2 [cherry-pick][2.93.x] chore: add LANGFUSE_CSP_DISABLE to .env.prod.example (#4533) 2024-12-03 01:25:04 +01:00
Marc KlingenandGitHub 7296e2e012 [cherry-pick][2.93.x] feat: optionally disable csp headers via LANGFUSE_CSP_DISABLE=true (#4532) 2024-12-03 01:22:11 +01:00
steffen911 43bf176ef7 chore: release v2.93.0 2024-11-26 10:04:45 +01:00
Steffen SchmitzandGitHub 6b48d7771c build: adjust docker actions to handle v2 branch (#4416)
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode) (push) Waiting to run
CI/CD / tests-worker (node20, pg12, mode-azure) (push) Waiting to run
CI/CD / tests-worker (node20, pg15, mode-azure) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
2024-11-26 09:59:57 +01:00
Max DeichmannandGitHub cd0d39b3c2 fix: fix bullmq api (#4425) 2024-11-25 23:18:16 +00:00
Max DeichmannandGitHub 8fcbcdd29d fix: order datasets API correctly (#4423) 2024-11-25 22:09:49 +00:00
Max DeichmannandGitHub e24f65c51d feat: add all queues to get api for bullmq api (#4422) 2024-11-25 21:53:27 +00:00
Marc KlingenandGitHub 38e6464219 chore: improve logging of cloud metering queue (#4420) 2024-11-25 21:22:13 +00:00
Steffen SchmitzandGitHub 86f5c885a1 chore: skip string parsing for metadata string values (#4415) 2024-11-25 18:04:49 +01:00
Hassieb PakzadandGitHub c0ebd06c0c feat(evaluators): add updates (#4360) 2024-11-25 16:51:46 +01:00
Steffen SchmitzandGitHub 0ce5ddcd00 feat: add azure storage provider support (#4386) 2024-11-25 14:57:40 +00:00
Steffen SchmitzandGitHub 7539af3bd9 chore: overwrite numerical type for second filters on observations table (#4412) 2024-11-25 14:39:04 +00:00
Steffen SchmitzandGitHub 60f9147873 fix: use larger numeric type for latency filters (#4385) 2024-11-25 13:53:27 +00:00
Steffen SchmitzandGitHub a594c142e0 chore: use clickhouse in evaluations queues (#4319) 2024-11-25 13:42:17 +00:00
Max DeichmannandGitHub 18905a9872 feat: add all scores to run items UI (#4409) 2024-11-25 14:31:11 +01:00
Hassieb PakzadandGitHub 27f5dd837d chore(costs): add chatgpt-4o-latest prices (#4407) 2024-11-25 13:55:58 +01:00
Max DeichmannandGitHub a921f9db94 feat: exclude env defined operations from clickhouse (#4405) 2024-11-25 12:56:12 +01:00
Marc Klingen b3cd940c73 chore(ui): remove ph badge (#4403) 2024-11-25 10:41:33 +01:00
Max DeichmannandGitHub f91cf1ca65 feat: refactor clickhouse filter builder for SDK APIs (#4402) 2024-11-24 22:45:29 +00:00
Max DeichmannandGitHub c438e89bd0 feat: support scores list endpoint from clickhouse (#4400) 2024-11-24 21:08:44 +00:00
Max DeichmannandGitHub 9b1746b7ed feat: fetch observations via api from clickhouse (#4395) 2024-11-24 09:48:17 +00:00
Steffen SchmitzandGitHub 0611a72c40 fix: allow special characters in openai tokenization (#4396)
* fix: allow special characters in openai tokenization

* chore: update test
2024-11-24 05:26:45 +00:00
Max DeichmannandGitHub 0ba66526e9 fix: do not use memory tables in clickhouse (#4394) 2024-11-23 18:21:30 +01:00
Max DeichmannandGitHub 425202dd1e fix: correctly retrieve scores for datasets (#4387) 2024-11-22 17:20:48 +00:00
Hassieb PakzadandGitHub 882ad83aad chore: upgrade langfuse-langchain (#4384) 2024-11-22 12:12:00 +01:00
Max DeichmannandGitHub 9b12783d08 fix: fix dataset run table to show all scores (#4383) 2024-11-22 10:44:23 +00:00
Marc KlingenandGitHub c6817ae32f chore: add Product Hunt Launch note (#4364) 2024-11-22 08:50:23 +01:00
marliessophieandGitHub 42711a8cc9 fix: don't show error message to user if trace is not yet available for dataset (#4382) 2024-11-22 08:27:54 +01:00
Marc KlingenandGitHub 0344ab2cf9 fix: experimentation should not use structured output (#4379) 2024-11-22 04:57:47 +01:00
Marc KlingenandGitHub df0d12af43 chore: rename default experiment tag langfuse-prompt-experiment (#4378) 2024-11-22 01:01:40 +00:00
marliessophieandGitHub d058ba8861 docs: add link to experimentation docs (#4371) 2024-11-21 22:42:36 +00:00
marliessophieandGitHub d1309905d0 chore: create dataset run items in order of dataset items timestamp (#4370)
* chore: create dataset run items in order of dataset items timestamp
2024-11-21 22:36:47 +01:00
Max DeichmannandGitHub ca6ac3912f fix: fix dataset run item api to always return traceid (#4367) 2024-11-21 20:25:30 +00:00
Max DeichmannandGitHub 9eabc32681 fix: return i/o from single object APIs (#4368) 2024-11-21 20:14:21 +00:00
marliessophieandGitHub 8798e4a499 feat(experiments): trigger experiments via UI (#4333) 2024-11-21 20:00:07 +01:00
Max Deichmann ea8b2da37c chore: release v2.92.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-21 18:44:30 +01:00
Max DeichmannandGitHub b7f7d10fea fix: fix dataset item api for self-hosters (#4366) 2024-11-21 17:43:48 +00:00
Marc KlingenandGitHub 0279c0f48e docs: add notification for lw 2 - day 4 (#4362) 2024-11-21 17:16:58 +01:00
Max DeichmannandGitHub c29cf3855f feat: SDK API for observations by id (#4295) 2024-11-21 15:28:34 +00:00
Max DeichmannandGitHub c73901369c fix: fix cost filters for traces table (#4325) 2024-11-21 14:21:01 +00:00
Hassieb PakzadandGitHub 0e70f03fc3 fix(media): lookup env var by exact string (#4356) 2024-11-21 10:48:33 +01:00
Max DeichmannandGitHub 74d1f7f717 feat: fetch single trace via API from clickhouse (#4344) 2024-11-21 10:38:24 +01:00
Marc KlingenandGitHub dfcf67184d docs: add lw2-3 info for multi-modal (#4351) 2024-11-20 23:15:19 +01:00
Max DeichmannandGitHub ce641d1e2b fix: do not fail on invalid prompt templates for evals (#4349) 2024-11-20 19:45:12 +00:00
Max DeichmannandGitHub 166b84a81a fix: do not use temp tables for dataset runs (#4347)
fix
2024-11-20 17:56:55 +00:00
Max DeichmannandGitHub f4097a1ae5 fix: dataset item api fix for large dataset runs (#4313) 2024-11-20 16:33:51 +00:00
Max DeichmannandGitHub 8374e889cf security: upgrade cross-spawn (#4339) 2024-11-20 13:32:19 +00:00
Max DeichmannandGitHub 62deaf7e00 sec: upgrade sentry (#4338) 2024-11-20 13:10:54 +00:00
Max DeichmannandGitHub a4cd68b1ad security: upgrade json path (#4335) 2024-11-20 13:32:01 +01:00
marliessophieandGitHub dde28f21fa chore: don't create traces in evaluation service (#4334) 2024-11-20 13:18:56 +01:00
Hassieb PakzadandGitHub a4c040b314 feat(core): add server side tracing of LLM invocations (#4317) 2024-11-20 12:00:31 +01:00
Max DeichmannandGitHub 87995625ca fix: correct traces pagination (#4329) 2024-11-20 02:31:34 +01:00
Marc KlingenandGitHub e717854909 docs: add launch week day 2 note (#4328) 2024-11-20 00:20:47 +01:00
Max DeichmannandGitHub 9e37f2cb17 feat: add more bullmq helpers (#4324) 2024-11-19 21:51:51 +00:00
Max DeichmannandGitHub fc08db08f3 fix: fix bullmq retry (#4323) 2024-11-19 21:24:55 +01:00
Max DeichmannandGitHub 270bc793ba feat: add service endpoint for failed bull events (#4321) 2024-11-19 21:18:40 +01:00
Max DeichmannandGitHub ced2d339ee fix: increase dataset item delay (#4318) 2024-11-19 17:46:49 +01:00
Max Deichmann 43acee5a29 chore: release v2.91.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web-sync (node20, pg12) (push) Waiting to run
CI/CD / tests-web-sync (node20, pg15) (push) Waiting to run
CI/CD / tests-web-async (node20, pg12) (push) Waiting to run
CI/CD / tests-web-async (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-19 15:43:51 +01:00
Max DeichmannandGitHub 723098994b feat: move trace not found to 404 (#4270) 2024-11-19 14:40:22 +00:00
Max DeichmannandGitHub 684159abd1 chore: add logging to ingestion API (#4306) 2024-11-19 14:06:40 +00:00
Steffen SchmitzandGitHub dbb482553f chore: move invalid tokenizer errors to warning (#4311) 2024-11-19 13:21:53 +00:00
Hassieb PakzadandGitHub c99a43559d fix(model-prices): upsert if prices are missing (#4312) 2024-11-19 14:08:30 +01:00
Steffen SchmitzandGitHub 3cbe56861f chore: upgrade prisma to 5.22.0 (#4308) 2024-11-19 10:54:13 +00:00
Steffen SchmitzandGitHub 2ef9aad41c chore: only warn on upsert failures if traceSession exists (#4307) 2024-11-19 10:42:40 +00:00
marliessophieandGitHub 087f4a526b feat(datasets): show linked evaluators on dataset (#4277) 2024-11-19 09:40:17 +00:00
Steffen SchmitzandGitHub f876cf6e34 chore: make migrations self-host replica compatible (#4287) 2024-11-19 08:37:08 +00:00
Marc KlingenandGitHub 75fbd56c67 docs: add sidebar notification for dataset comparison view (#4300) 2024-11-19 01:23:39 +00:00
Max DeichmannandGitHub 0c326e56ba feat: fetch sessions by id for public API (#4292) 2024-11-18 22:39:30 +01:00
Max DeichmannandGitHub 2068ba7314 feat: fetch scores by id via API from clickhouse (#4291) 2024-11-18 20:33:05 +00:00
Max DeichmannandGitHub 9841973b16 chore: add web async test setup (#4294) 2024-11-18 21:05:32 +01:00
Max DeichmannandGitHub 1c44000933 perf: do not fetch models for UI (#4288) 2024-11-18 15:56:45 +00:00
Hassieb PakzadandGitHub f71e1aebed feat(media): add ui support (#4285) 2024-11-18 16:22:43 +01:00
marliessophieandGitHub 252fe184b6 feat(evals_on_datasets): remove feature flag (#4289) 2024-11-18 14:35:35 +00:00
marliessophieandGitHub 8034148313 fix(dataset_compare): show latency only if not null (#4282) 2024-11-18 11:41:33 +00:00
marliessophieandGitHub f6746e8f84 feat(datasets): add header to peek view (#4278)
* refactor: extract peek view

* feat: add consistent header view

* link to dataset item detail view
2024-11-18 10:46:05 +01:00
marliessophieandGitHub a48817300b feat(ui): peek view on dataset run compare table (#4131) 2024-11-18 08:28:18 +01:00
Max DeichmannandGitHub a74a820e32 fix: fix dataset runs api (#4275) 2024-11-17 21:49:11 +00:00
Max DeichmannandGitHub e0f9333d2f fix: fix dataset id api (#4274) 2024-11-17 17:54:21 +01:00
Max DeichmannandGitHub 34d4feae27 fix: fix dataset run items api (#4273) 2024-11-17 17:34:33 +01:00
Max DeichmannandGitHub f09ff1f070 feat: serve datasets from clickhouse (#4272) 2024-11-17 17:06:12 +01:00
Max DeichmannandGitHub 9a89c3f002 fix: fix dataset route (#4271) 2024-11-17 16:33:19 +01:00
Max DeichmannandGitHub f1e55b8ac7 feat: move datasets results to logs only (#4269) 2024-11-17 13:39:45 +00:00
Max DeichmannandGitHub d908fb13b3 feat: read dataset fetures from clickhouse (#4264) 2024-11-17 14:18:20 +01:00
Max DeichmannandGitHub 3c02e525bc fix: session id filter on session table (#4267) 2024-11-16 19:09:23 +01:00
Max DeichmannandGitHub 2cf04e4ab0 feat: configure to read from clickhouse only (#4266) 2024-11-16 13:02:01 +00:00
Max DeichmannandGitHub 24a549dc54 feat: move traces.hasany to clickhouse (#4265) 2024-11-16 12:51:25 +00:00
marliessophieandGitHub 63749970c6 chore(ui): enable all feature flags for admins (#4256) 2024-11-15 19:10:19 +01:00
Marc KlingenandGitHub 0d1b4f384a fix(cloud): usage indicator only on hobby plan (#4257) 2024-11-15 18:10:03 +01:00
marliessophieandGitHub 8517670ac2 feat: support evals triggered via dataset run (#4086) 2024-11-15 16:37:55 +01:00
Steffen SchmitzandGitHub 6157287549 chore: upsert tracesessions from clickhouse ingestion pipeline (#4254) 2024-11-15 14:43:39 +00:00
Max DeichmannandGitHub 041345ad3d perf: improve prompt metrics performance (#4251) 2024-11-15 14:28:08 +00:00
Hassieb PakzadandGitHub 7ed865cb4a chore: add media env to v3preview docker compose (#4250) 2024-11-15 14:32:52 +01:00
steffen911 b4d59960c0 chore: release v2.90.1
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-15 14:16:38 +01:00
Steffen SchmitzandGitHub 611b32e965 chore: add tzdata package (#4253) 2024-11-15 14:15:58 +01:00
Steffen SchmitzandGitHub fc94c02058 fix: fail on errors in database migrations (#4252) 2024-11-15 13:49:42 +01:00
Max Deichmann 853ca3c252 chore: release v2.90.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-15 11:25:56 +01:00
Steffen SchmitzandGitHub 74e378fe4d chore: fill clickhouse trace for observations with missing traceId (#4246) 2024-11-15 10:14:35 +00:00
Steffen SchmitzandGitHub a4d1288b00 chore: add note for future score create removal (#4248) 2024-11-15 11:03:59 +01:00
Steffen SchmitzandGitHub 91d27dd231 chore: add clickhouse migrations to web entrypoint (#4245) 2024-11-15 09:26:59 +01:00
Max DeichmannandGitHub 6d90635476 feat: exclude project_ids from clickhouse experimentation (#4242) 2024-11-14 21:47:01 +00:00
Max DeichmannandGitHub e9be67efb4 feat: read metrics for prompts table (#4224) 2024-11-14 20:39:47 +00:00
Max DeichmannandGitHub 71d1f2d8e6 feat: add reverse feature flag for clickhouse reads (#4241) 2024-11-14 18:43:38 +00:00
Marc KlingenandGitHub 56233d52eb feat(cloud): move hobby plan usage info to sidebar (#4240) 2024-11-14 18:06:21 +00:00
Max DeichmannandGitHub 5b5c3ba439 chore: improve clickhouse experimentation (#4239) 2024-11-14 18:23:02 +01:00
Steffen SchmitzandGitHub a0c6acf4e4 chore: add query parameter to healthcheck which enables failure on unavailable database (#4238) 2024-11-14 17:08:16 +00:00
Max DeichmannandGitHub bf54c123c1 fix: sessions metrics query (#4237) 2024-11-14 17:43:37 +01:00
Max DeichmannandGitHub 6a30fed1c6 fix: use todate for trace timestamps (#4235) 2024-11-14 15:30:54 +00:00
Steffen SchmitzandGitHub 8c2f8e6247 chore: reduce use of FINAL in clickhouse queries (#4233) 2024-11-14 16:16:34 +01:00
Steffen SchmitzandGitHub 593d1a087d chore: add timestamp filter for dashboard queries (#4234) 2024-11-14 15:10:39 +01:00
Hassieb PakzadandGitHub 633f76f64a feat(multimodal): add media upload / download endpoints (#3989) 2024-11-14 10:26:30 +01:00
Max DeichmannandGitHub ec89fabc5d fix: fix trace filter on score analytics charts (#4225) 2024-11-13 22:45:08 +00:00
Hassieb PakzadandGitHub 26c02ad49c fix(playground): jump to playground for clickhouse reads (#4222) 2024-11-13 20:10:01 +01:00
Max DeichmannandGitHub e1841e016f fix: fix order by in clickhouse (#4221) 2024-11-13 18:38:47 +01:00
Max DeichmannandGitHub 8ebfb3d8a2 fix: fix observations order by (#4220) 2024-11-13 17:25:05 +00:00
Max DeichmannandGitHub 4c0320c565 feat: fetch number of generations for prompts table form clickhouse (#4218) 2024-11-13 17:09:15 +00:00
Max DeichmannandGitHub 5c1f2202cd chore: add experimentation execution time metrics (#4219) 2024-11-13 17:55:03 +01:00
Steffen SchmitzandGitHub e1abbecabd chore: migrate users table to clickhouse (#4191) 2024-11-13 15:40:41 +00:00
Steffen SchmitzandGitHub 52466da955 chore: use event_ts instead of final for obs and score retrieval (#4214) 2024-11-13 15:19:13 +00:00
Max DeichmannandGitHub ede7ad49bf revert: "security: upgrade dd-trace (#4211)" (#4216) 2024-11-13 16:08:07 +01:00
Max DeichmannandGitHub c61acee8e8 security: upgrade dd-trace (#4211) 2024-11-13 13:41:45 +00:00
Steffen SchmitzandGitHub 7ea0a368af chore: handle clickhouse errors gracefully and correct formatting for getTraceById timestamp (#4210) 2024-11-13 13:10:32 +00:00
Steffen SchmitzandGitHub bb07b3508d chore: apply trace timestamp filter correctly on user model query (#4209) 2024-11-13 12:36:35 +00:00
Steffen SchmitzandGitHub 4df24cfde5 chore: remove superfluous isClickhouseAdminEligible checks (#4207) 2024-11-13 10:21:51 +01:00
Hassieb PakzadandGitHub 47c06a77bd fix(playground): disable stream_options for openai adapter (#4206) 2024-11-13 08:45:42 +00:00
Steffen SchmitzandGitHub 7a95b1e1da chore: migrate score analytics dashboard to clickhouse (#4196) 2024-11-13 08:18:05 +00:00
Max DeichmannandGitHub b09f9cbf5a chore: add tests for traces table UI (#4202) 2024-11-12 22:40:50 +00:00
Max Deichmann 780880cf07 chore: release v2.89.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-12 19:34:07 +01:00
Max DeichmannandGitHub eb6f93bd9c fix: fix public trace access (#4195) 2024-11-12 18:32:45 +00:00
Max DeichmannandGitHub a0d4a2e599 fix: fix indefinite table queries (#4194) 2024-11-12 17:04:23 +00:00
Max DeichmannandGitHub 8458443a2e chore: remove logs (#4193) 2024-11-12 16:37:49 +01:00
Max DeichmannandGitHub feafc1b7d3 feat: move scores table apis to experimentation setup (#4192) 2024-11-12 15:14:33 +00:00
Max DeichmannandGitHub fcd54180bc chore: centralize lookback guarantees (#4188) 2024-11-12 15:56:33 +01:00
Max DeichmannandGitHub d73c3ec8d3 fix: fix clickhouse filters (#4190) 2024-11-12 14:48:45 +00:00
Max DeichmannandGitHub ca7210b26f feat: use timestamp to link single traces from navigation (#4187) 2024-11-12 11:50:51 +00:00
Max DeichmannandGitHub d9e22324d4 feat: up/down detail navigation support params (#4186) 2024-11-12 12:13:44 +01:00
Max DeichmannandGitHub 0a18e59274 perf: add timestamp to traces table link (#4185) 2024-11-12 10:38:00 +00:00
Max DeichmannandGitHub ffc5f81fb5 chore: add error logs in trpc route (#4184) 2024-11-12 09:56:03 +00:00
marliessophieandGitHub b4668ed851 Revert "fix(dataset_runs): ensure consistent ordering to fetch score data for dataset run aggregation metrics table" (#4182)
Revert "fix(dataset_runs): ensure consistent ordering to fetch score data for…"

This reverts commit 2b2022ae85.
2024-11-12 09:20:12 +00:00
Steffen SchmitzandGitHub 999ff06c58 chore: migrate spammy log messages to debug level (#4181) 2024-11-12 07:41:01 +00:00
Steffen SchmitzandGitHub e494b2c215 chore: remove uninstantiated projectId from clickhouse seeder (#4180) 2024-11-12 06:56:42 +00:00
marliessophieandGitHub 2b2022ae85 fix(dataset_runs): ensure consistent ordering to fetch score data for dataset run aggregation metrics table (#4176) 2024-11-11 16:07:56 +00:00
Steffen SchmitzandGitHub 3afa137130 chore: update latency calculation for clickhouse dashboard queries (#4175) 2024-11-11 16:28:43 +01:00
Max DeichmannandGitHub 0b88ddf438 fix: correctly filter observations by generations for generations ui table (#4173) 2024-11-11 14:16:01 +00:00
Marc KlingenandGitHub f878bd1317 perf(cloud): add 60 min stale time for usage indicator (#4171) 2024-11-11 14:10:58 +01:00
Marc KlingenandGitHub 72075b29de fix(ui): width of date range time input (#4170) 2024-11-11 14:06:58 +01:00
Max DeichmannandGitHub edf1a0b879 fix: fix aggregated user consumption (#4167) 2024-11-11 11:21:03 +00:00
Max DeichmannandGitHub df763ff09c perf: improve performance for latency dashboard (#4164) 2024-11-11 10:23:49 +00:00
Max DeichmannandGitHub 83e09bcf47 chore: read dashboards form clickhouse flag (#4162) 2024-11-10 20:24:50 +00:00
Max DeichmannandGitHub 10e9aef044 feat: move sessions api to experimentation setup (#4161) 2024-11-10 20:09:59 +00:00
Max DeichmannandGitHub 45a0c2eb3d feat: move generations api to experimentation setup (#4160)
chore: move generations api to experimentation setup
2024-11-10 19:09:44 +00:00
Max DeichmannandGitHub 99c2a75964 perf: add time filter for sessions (#4159) 2024-11-10 18:49:11 +00:00
Max DeichmannandGitHub 9cc61eb851 perf: add timestamp to filteroptions api (#4158) 2024-11-10 18:35:06 +00:00
Max DeichmannandGitHub 62e51b8cf6 feat: add traces experimentation (#4157) 2024-11-10 17:45:31 +00:00
Max DeichmannandGitHub b147560a80 feat: steer reading from CH and PG (#4153) 2024-11-10 18:14:30 +01:00
Max DeichmannandGitHub 97a0b1f44c perf: increase traces metrics performance (#4155) 2024-11-10 17:08:00 +01:00
Max DeichmannandGitHub 8c1e6647e0 feat: query model latencies from clickhouse (#4152) 2024-11-10 14:37:56 +00:00
Max DeichmannandGitHub 8432327127 fix: traces aggregate trace filter (#4151) 2024-11-10 14:12:29 +00:00
Max DeichmannandGitHub 231f8e8f0c feat: read latency tables from clickhouse (#4150) 2024-11-10 13:47:58 +00:00
Max DeichmannandGitHub 28e43eb74c feat: query scores over time (#4149) 2024-11-10 12:57:30 +00:00
Steffen SchmitzandGitHub e2d3647d76 chore: set source: API if no source provided (#4148) 2024-11-10 13:41:01 +01:00
Max DeichmannandGitHub 876509d9e5 feat: add user charts (#4146) 2024-11-10 12:18:38 +00:00
Max DeichmannandGitHub 1647a79198 perf: scores dashboard performance improvement (#4144)
fix
2024-11-10 12:01:16 +00:00
Steffen SchmitzandGitHub 622d8a89a7 chore: update bookmarked, tags, and public for clickhouse traces (#4132) 2024-11-10 11:29:48 +00:00
Max DeichmannandGitHub d5e66e10d6 feat: add table search for clickhouse (#4142) 2024-11-09 15:55:11 +01:00
Max DeichmannandGitHub 5b1776710f chore: clean ups of clickhouse read logic (#4141) 2024-11-09 14:35:04 +01:00
Max DeichmannandGitHub 7e7e3b77d4 fix: correctly parse model params from clickhouse (#4140) 2024-11-09 12:48:45 +00:00
Max DeichmannandGitHub 28ecd9600a fix: fix latency calculations when reading from clickhouse (#4139) 2024-11-09 10:59:52 +00:00
Max DeichmannandGitHub 1ef6834a31 feat: delete traces from clickhouse from the UI (#4138) 2024-11-09 11:49:44 +01:00
Marc KlingenandGitHub b88f3bed40 feat(ui): add notification card in sidebar (#4137) 2024-11-09 00:30:36 +00:00
Steffen SchmitzandGitHub eaa885f5f5 chore: inject eval and annotation scores into clickhouse (#4081) 2024-11-08 17:42:41 +00:00
Max DeichmannandGitHub f7f48c2509 fix: reduce ingestion concurrency (#4130) 2024-11-08 18:18:40 +01:00
Steffen SchmitzandGitHub 07344e89b1 chore: move legacy api endpoints to async ingestion (#4109) 2024-11-08 18:00:59 +01:00
Max DeichmannandGitHub e6bc717612 perf: improve generations table on clickhouse (#4124) 2024-11-08 11:18:41 +00:00
Max DeichmannandGitHub 11521fc24d perf: improve generations performance (#4111) 2024-11-07 20:54:49 +00:00
Max DeichmannandGitHub 213ec6027d perf: add sessions index in clickhouse (#4108) 2024-11-07 18:09:41 +01:00
Max DeichmannandGitHub 259d832d37 feat: support sessions API from Clickhouse (#4065) 2024-11-07 16:52:07 +00:00
Marc KlingenandGitHub 62deb83520 chore(cloud): stripe checkout settings (#4107) 2024-11-07 16:34:07 +00:00
Max DeichmannandGitHub edc4cd7873 perf: query traces via timestamps (#4105) 2024-11-07 14:52:33 +01:00
Steffen SchmitzandGitHub 05b75fbee0 chore: merge observation backfill states in table (#4104) 2024-11-07 13:20:18 +00:00
Steffen SchmitzandGitHub d6065e8b89 chore: add stateSuffix to observation migration for potential parallelization (#4101) 2024-11-07 12:46:58 +00:00
Steffen SchmitzandGitHub 617d129542 chore: swap negative provided usage for null (#4100) 2024-11-07 12:29:44 +00:00
Max Deichmann 057424b269 chore: release v2.88.0
CI/CD / lint (push) Waiting to run
CI/CD / test-docker-build (push) Waiting to run
CI/CD / tests-web (node20, pg12) (push) Waiting to run
CI/CD / tests-web (node20, pg15) (push) Waiting to run
CI/CD / tests-worker (node20, pg12) (push) Waiting to run
CI/CD / tests-worker (node20, pg15) (push) Waiting to run
CI/CD / e2e-tests (push) Waiting to run
CI/CD / e2e-server-tests (push) Waiting to run
CI/CD / all-ci-passed (push) Blocked by required conditions
CI/CD / push-docker-image (push) Blocked by required conditions
release.yml / release (push) Waiting to run
2024-11-07 11:44:46 +01:00
Max DeichmannandGitHub 63d1b85a6c fix: pagination for clickhouse tables (#4098) 2024-11-07 11:41:46 +01:00
Max DeichmannandGitHub 9481c3b424 feat: build scores table for clickhouse (#4092) 2024-11-07 10:18:14 +00:00
Steffen SchmitzandGitHub 15e174fc2f chore: convert starttime filter to DateTime in Ingestion pipeline (#4095) 2024-11-07 09:51:35 +01:00
Steffen SchmitzandGitHub 5efd4ec0a4 chore: include backgroundmigration state default in schema.prisma (#4094) 2024-11-07 08:25:32 +00:00
marliessophieandGitHub 895b61fd62 fix(ui): render dataset run description and metadata on screen given table (#4089)
fix(ui): render dataset run description and metadata on screen without cutting off table
2024-11-06 19:47:01 +00:00
Steffen SchmitzandGitHub 5d4125208c chore: restrict domain timestamp updates to one day in clickhouse ingestion (#4085) 2024-11-06 17:27:25 +00:00
Steffen SchmitzandGitHub ff3bfe75ec chore: drop redundant observation project_id index on Clickhouse (#4084) 2024-11-06 16:45:53 +01:00
Steffen SchmitzandGitHub 89c31fdbc2 chore: add type filter to observation lookup in ingestion queue (#4083) 2024-11-06 15:18:19 +00:00
Steffen SchmitzandGitHub 9445c7ca78 chore: forward /api/public/scores to /api/public/ingestion (#4064) 2024-11-06 11:32:38 +01:00
marliessophieandGitHub b3d5833520 fix: unify ordering of dataset_items and dataset_run_items (#4063)
* fix: unify ordering of `dataset_items` and `dataset_run_items`

* fix: filter trace scores by observation id NULL
2024-11-05 19:03:30 +00:00
Steffen SchmitzandGitHub 20ed40510a chore: update cost calculation to handle null values from DB (#4057) 2024-11-05 18:22:10 +01:00
Steffen SchmitzandGitHub 9c5c57f134 chore: add test case for different input/output values (#4061) 2024-11-05 16:46:17 +00:00
Steffen SchmitzandGitHub 09b916aac9 chore: do not send auth errors to DLQ in bull (#4062) 2024-11-05 16:23:17 +00:00
marliessophieandGitHub 92843fef28 feat(datasets): view to compare different dataset runs (#3971) 2024-11-05 16:10:28 +01:00
Steffen SchmitzandGitHub 433951afda chore: retry legacy ingestion batches on errors (#4058) 2024-11-05 15:25:25 +01:00
Steffen SchmitzandGitHub aa1f861e4e chore: ignore all .env files aside from examples (#4059) 2024-11-05 13:25:27 +00:00
Hassieb PakzadandGitHub 66b6a09400 feat(models): add claude haiku 3.5 support (#4055) 2024-11-05 12:14:25 +01:00
Max DeichmannandGitHub edf0dc8399 feat: add metadata filter for all UI tables using clickhouse (#4022)
* feat: add metadata filter for all UI tables using clickhouse

* feat: add metadata filter for all UI tables using clickhouse

* push
2024-11-05 10:05:22 +00:00
Marc KlingenandGitHub 8d0c1219ca fix: env configuration for telemetry (#4052) 2024-11-04 23:04:07 +00:00
Max DeichmannandGitHub 33b3f69076 feat: query distinct models for charts (#4051) 2024-11-04 22:42:46 +01:00
Max DeichmannandGitHub f582941ad3 perf: fix timeseries clickouse (#4049) 2024-11-04 21:14:14 +00:00
Max DeichmannandGitHub 9fd57b4fbf feat: query second row of dashboards using clickhouse (#4048) 2024-11-04 20:18:37 +00:00
Max DeichmannandGitHub f61f69c238 perf: only join in dashboard if required (#4047) 2024-11-04 18:17:22 +00:00
Steffen SchmitzandGitHub 55c626d627 chore: reduce startTime missing logs to create events (#4044) 2024-11-04 14:52:09 +00:00
Max DeichmannandGitHub c6d3257916 feat: add first three dashboards in clickhouse (#4023) 2024-11-04 15:02:29 +01:00
Marc KlingenandGitHub 34f91e9872 chore(ui): slightly lighter bg of darkmode sidebar (#4042) 2024-11-04 15:01:58 +01:00
Steffen SchmitzandGitHub 5acd509fc0 chore: init migration scripts with state on restarts (#4039) 2024-11-04 13:37:28 +00:00
marliessophieandGitHub 765756b002 fix: scores tab in observation preview to only show scores liked to observation and trace id (#4030) 2024-11-04 12:54:21 +00:00
Steffen SchmitzandGitHub 64525744de chore: include usagedetails from postgres in ingestion pipeline (#4034) 2024-11-04 12:41:47 +00:00
Steffen SchmitzandGitHub 2b89f62add chore: remove temporary column from PG to CH migrations (#4031) 2024-11-04 13:26:42 +01:00
marliessophieandGitHub 0a67adf3e7 fix: show generations count on prompt metrics table (#4003) 2024-11-04 09:56:18 +00:00
marliessophieandGitHub 4ef7296e37 fix: multi-select header partially cut off on Firefox (#4029) 2024-11-04 09:45:18 +00:00
337 changed files with 26750 additions and 8627 deletions
+92
View File
@@ -0,0 +1,92 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
# Clickhouse
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_MIGRATION_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
# openssl rand -base64 32
# https://next-auth.js.org/configuration/options#secret
# NEXTAUTH_SECRET=""
NEXTAUTH_URL="http://localhost:3000"
NEXTAUTH_SECRET="secret"
# Langfuse Cloud Environment
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
# Langfuse experimental features
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
# Salt for API key hashing
SALT="salt"
# Email
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# DON'T PANIC: The Azurite Secrets are well-known and meant to be hard-coded
# S3 storage
S3_ENDPOINT=http://localhost:10000/devstoreaccount1
S3_ACCESS_KEY_ID=devstoreaccount1
S3_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
S3_BUCKET_NAME=langfuse
S3_REGION=auto
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
LANGFUSE_S3_MEDIA_UPLOAD_REGION=auto
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=devstoreaccount1
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==
LANGFUSE_S3_EVENT_UPLOAD_REGION=auto
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:10000/devstoreaccount1
## Necessary for minio compatibility
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
LANGFUSE_USE_AZURE_BLOB=true
# Set during docker build of application
# Used to disable environment verification at build time
# DOCKER_BUILD=1
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+22 -2
View File
@@ -11,6 +11,7 @@ CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_DISABLED="true"
# Next Auth
# You can generate a new secret on the command line with:
@@ -42,9 +43,20 @@ S3_REGION=us-east-1
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=false
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
@@ -66,4 +78,12 @@ REDIS_AUTH="myredissecret"
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
LANGFUSE_READ_FROM_POSTGRES_ONLY=false
LANGFUSE_RETURN_FROM_CLICKHOUSE=true
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE=true
LANGFUSE_READ_FROM_CLICKHOUSE_ONLY=true
LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING="true"
+81
View File
@@ -0,0 +1,81 @@
# When adding additional environment variables, the schema in "/src/env.mjs"
# should be updated accordingly.
# Prisma
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
# Clickhouse
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
CLICKHOUSE_URL="http://localhost:8123"
CLICKHOUSE_USER="clickhouse"
CLICKHOUSE_PASSWORD="clickhouse"
CLICKHOUSE_CLUSTER_ENABLED="false"
# Next Auth
# You can generate a new secret on the command line with:
# openssl rand -base64 32
# https://next-auth.js.org/configuration/options#secret
# NEXTAUTH_SECRET=""
NEXTAUTH_URL="http://localhost:3000"
NEXTAUTH_SECRET="secret"
# Langfuse Cloud Environment
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
# Langfuse experimental features
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
# Salt for API key hashing
SALT="salt"
# Email
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
# S3 storage
S3_ENDPOINT=http://localhost:9090
S3_ACCESS_KEY_ID=minio
S3_SECRET_ACCESS_KEY=miniosecret
S3_BUCKET_NAME=langfuse
S3_REGION=us-east-1
## Necessary for minio compatibility
S3_FORCE_PATH_STYLE=true
# # S3 Media Upload LOCAL
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED=true
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
# S3 Event Bucket Upload
## Set to true to test uploading all events to S3
LANGFUSE_S3_EVENT_UPLOAD_ENABLED=true
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:9090
## Necessary for minio compatibility
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
# Set during docker build of application
# Used to disable environment verification at build time
# DOCKER_BUILD=1
REDIS_HOST="127.0.0.1"
REDIS_PORT=6379
REDIS_AUTH="myredissecret"
# openssl rand -hex 32 used only here
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
# speeds up local development by not executing init scripts on server startup
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
+12 -2
View File
@@ -55,6 +55,7 @@ OTEL_SERVICE_NAME="langfuse"
# Auth, optional configuration
# AUTH_DOMAINS_WITH_SSO_ENFORCEMENT=domain1.com,domain2.com
# AUTH_IGNORE_ACCOUNT_FIELDS=foo,bar
# AUTH_DISABLE_USERNAME_PASSWORD=true
# AUTH_DISABLE_SIGNUP=true
# AUTH_SESSION_MAX_AGE=43200 # 30 days in minutes (default)
@@ -67,6 +68,10 @@ OTEL_SERVICE_NAME="langfuse"
# AUTH_GITHUB_CLIENT_ID=
# AUTH_GITHUB_CLIENT_SECRET=
# AUTH_GITHUB_ALLOW_ACCOUNT_LINKING=false
# AUTH_GITHUB_ENTERPRISE_CLIENT_ID=
# AUTH_GITHUB_ENTERPRISE_CLIENT_SECRET=
# AUTH_GITHUB_ENTERPRISE_BASE_URL=
# AUTH_GITHUB_ENTERPRISE_ALLOW_ACCOUNT_LINKING=false
# AUTH_GITLAB_CLIENT_ID=
# AUTH_GITLAB_CLIENT_SECRET=
# AUTH_GITLAB_ALLOW_ACCOUNT_LINKING=false
@@ -87,12 +92,17 @@ OTEL_SERVICE_NAME="langfuse"
# AUTH_COGNITO_CLIENT_SECRET=
# AUTH_COGNITO_ISSUER=
# AUTH_COGNITO_ALLOW_ACCOUNT_LINKING=false
# AUTH_KEYCLOAK_CLIENT_ID=
# AUTH_KEYCLOAK_CLIENT_SECRET=
# AUTH_KEYCLOAK_ISSUER=
# AUTH_KEYCLOAK_ALLOW_ACCOUNT_LINKING=false
# AUTH_CUSTOM_CLIENT_ID=
# AUTH_CUSTOM_CLIENT_SECRET=
# AUTH_CUSTOM_ISSUER=
# AUTH_CUSTOM_NAME=
# AUTH_CUSTOM_SCOPE="openid email profile" # optional
# AUTH_CUSTOM_ALLOW_ACCOUNT_LINKING=false
# AUTH_CUSTOM_ID_TOKEN=false # optional, default is true
# Transactional email, optional
# Defines the email address to use as the from address.
@@ -237,7 +247,7 @@ OTEL_SERVICE_NAME="langfuse"
# CLICKHOUSE_PASSWORD=
# Ingestion
# LANGFUSE_INGESTION_QUEUE_DELAY_SECONDS=
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_BATCH_SIZE=
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=
# LANGFUSE_INGESTION_CLICKHOUSE_MAX_ATTEMPTS=
@@ -245,4 +255,4 @@ OTEL_SERVICE_NAME="langfuse"
# LANGFUSE_ASYNC_INGESTION_PROCESSING="true"
# QUEUE_CONSUMER_LEGACY_INGESTION_QUEUE_IS_ENABLED="true"
## END Langfuse V3 Ingestion
## END Langfuse V3 Ingestion
+7 -2
View File
@@ -3,9 +3,14 @@ name: Codespell
on:
push:
branches: [main]
branches:
- "main"
tags:
- "v*"
pull_request:
branches: [main]
branches:
- "**"
merge_group:
permissions:
contents: read
+106 -11
View File
@@ -66,10 +66,10 @@ jobs:
run: |
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
tests-web:
tests-web-sync:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-web (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
name: tests-web-sync (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
strategy:
matrix:
node-version: [20]
@@ -82,7 +82,7 @@ jobs:
- uses: actions/checkout@v4
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- uses: pnpm/action-setup@v3
@@ -105,7 +105,7 @@ jobs:
- name: Load default env
run: |
cp .env.dev.example .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.example > .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.legacy.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev.yml up -d
@@ -131,17 +131,87 @@ jobs:
LANGFUSE_INIT_USER_EMAIL: "demo@langfuse.com"
LANGFUSE_INIT_USER_NAME: "Demo User"
LANGFUSE_INIT_USER_PASSWORD: "password"
- name: run test-sync
run: pnpm --filter=web run test-sync
tests-web-async:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
node-version: [20]
postgres-version: [12, 15]
blob-provider: ["", "-azure"]
steps:
- name: Set Swap Space
uses: pierotofy/set-swap-space@master
with:
swap-size-gb: 10
- uses: actions/checkout@v4
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- uses: pnpm/action-setup@v3
with:
version: 9.5.0
- name: Login to Docker Hub
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME_READ }}
password: ${{ secrets.DOCKERHUB_TOKEN_READ }}
- name: Use Node.js ${{ matrix.node-version }}
uses: actions/setup-node@v4
with:
node-version: ${{ matrix.node-version }}
cache: "pnpm"
cache-dependency-path: "pnpm-lock.yaml"
- name: install dependencies
run: |
pnpm install
- name: Load default env
run: |
cp .env.dev${{ matrix.blob-provider }}.example .env
grep -v -e '^S3_BUCKET_NAME=' -e '^REDIS_HOST=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
sleep 5 # Wait for PostgreSQL to accept connections
docker compose ps
env:
POSTGRES_VERSION: ${{ matrix.postgres-version }}
- name: Seed DB
run: |
pnpm run db:migrate
pnpm --filter=shared ch:up
- name: Build
run: pnpm run build
- name: Start Langfuse
run: (pnpm run start&)
env:
LANGFUSE_INIT_ORG_ID: "seed-org-id"
LANGFUSE_INIT_ORG_NAME: "Seed Org"
LANGFUSE_INIT_PROJECT_ID: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
LANGFUSE_INIT_PROJECT_NAME: "Seed Project"
LANGFUSE_INIT_PROJECT_PUBLIC_KEY: "pk-lf-1234567890"
LANGFUSE_INIT_PROJECT_SECRET_KEY: "sk-lf-1234567890"
LANGFUSE_INIT_USER_EMAIL: "demo@langfuse.com"
LANGFUSE_INIT_USER_NAME: "Demo User"
LANGFUSE_INIT_USER_PASSWORD: "password"
- name: run tests
run: pnpm --filter=web run test
tests-worker:
timeout-minutes: 20
runs-on: ubuntu-latest
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }})
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
strategy:
matrix:
node-version: [20]
postgres-version: [12, 15]
blob-provider: ["", "-azure"]
steps:
- name: Set Swap Space
uses: pierotofy/set-swap-space@master
@@ -167,17 +237,17 @@ jobs:
pnpm install
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.16.2/migrate.linux-amd64.tar.gz | tar xvz
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- name: Load default env
run: |
cp .env.dev.example .env
cp .env.dev.example web/.env
cp .env.dev.example worker/.env
cp .env.dev${{ matrix.blob-provider }}.example .env
cp .env.dev${{ matrix.blob-provider }}.example web/.env
cp .env.dev${{ matrix.blob-provider }}.example worker/.env
- name: Run + migrate
run: |
docker compose -f docker-compose.dev.yml up -d
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
sleep 5 # Wait for PostgreSQL to accept connections
docker compose ps
- name: Ensure no unhealthy status
@@ -258,9 +328,15 @@ jobs:
- name: install dependencies
run: |
pnpm install
- name: Install golang-migrate for Clickhouse migrations
run: |
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
sudo mv migrate /usr/bin/migrate
which migrate
- name: Load default env
run: |
cp .env.dev.example .env
echo "LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_ASYNC_INGESTION_PROCESSING=true" >> .env
echo "LANGFUSE_CACHE_API_KEY_ENABLED=true" >> .env
echo "LANGFUSE_CACHE_PROMPT_ENABLED=true" >> .env
@@ -269,14 +345,29 @@ jobs:
docker compose -f docker-compose.dev.yml up -d
docker compose ps
sleep 5 # Wait for PostgreSQL to accept connections
- name: Ensure Docker dependencies are healthy
run: |
if docker-compose ps | grep "(unhealthy)"; then
echo "One or more services are unhealthy"
exit 1
else
echo "All services are healthy"
fi
- name: Seed DB
run: |
pnpm run db:migrate
pnpm --filter=shared run ch:up
pnpm run db:seed:examples
- name: Build
run: pnpm run build
- name: Run server
run: (pnpm run start&)
- name: Check worker health
run: |
timeout 10 bash -c 'until curl -f http://localhost:3030/api/health; do sleep 2; done'
- name: Check server health
run: |
timeout 10 bash -c 'until curl -f http://localhost:3000/api/public/health; do sleep 2; done'
- name: Run e2e tests
run: pnpm --filter=web run test:e2e:server
@@ -286,11 +377,12 @@ jobs:
needs:
[
lint,
tests-web,
tests-web-sync,
tests-worker,
e2e-tests,
test-docker-build,
e2e-server-tests,
tests-web-async,
]
if: always()
steps:
@@ -349,6 +441,8 @@ jobs:
images: |
ghcr.io/langfuse/langfuse # GitHub
langfuse/langfuse # Docker Hub
flavor: |
latest=false
tags: |
type=ref,event=branch
type=ref,event=pr
@@ -356,6 +450,7 @@ jobs:
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=latest,enable=${{ startsWith(github.ref, 'refs/tags/v3') }}
- name: Build and push Docker image (web)
uses: docker/build-push-action@v4
with:
+1 -1
View File
@@ -3,7 +3,7 @@ on:
push:
# Pattern matched against refs/tags
tags:
- "v[0-9]+.[0-9]+.[0-9]+" # Semantic version tags
- "v3.[0-9]+.[0-9]+" # Semantic version tags
jobs:
release:
+6 -3
View File
@@ -35,9 +35,12 @@ yarn-error.log*
# local env files
# do not commit any .env files to git, except for the .env.example file. https://create.t3.gg/en/usage/env-variables#using-environment-variables
.env
.env.local
.env*.local
.env*
!.env.dev.example
!.env.dev-azure.example
!.env.local.example
!.env.prod.example
!.env.dev.legacy.example
# vercel
.vercel
+63
View File
@@ -0,0 +1,63 @@
services:
clickhouse:
image: clickhouse/clickhouse-server
user: "101:101"
container_name: clickhouse
hostname: clickhouse
environment:
CLICKHOUSE_DB: default
CLICKHOUSE_USER: clickhouse
CLICKHOUSE_PASSWORD: clickhouse
volumes:
- langfuse_clickhouse_data:/var/lib/clickhouse
- langfuse_clickhouse_logs:/var/log/clickhouse-server
ports:
- "8123:8123"
- "9000:9000"
depends_on:
- postgres
azurite:
image: mcr.microsoft.com/azure-storage/azurite
container_name: azurite
command: azurite-blob --blobHost 0.0.0.0
ports:
- "10000:10000"
volumes:
- langfuse_azurite_data:/data
redis:
image: redis:7.2.4
restart: always
command: >
--requirepass ${REDIS_AUTH:-myredissecret}
ports:
- 6379:6379
postgres:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
command: ["postgres", "-c", "log_statement=all"]
environment:
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=postgres
- POSTGRES_DB=postgres
ports:
- 5432:5432
volumes:
- langfuse_postgres_data:/var/lib/postgresql/data
volumes:
langfuse_postgres_data:
driver: local
langfuse_clickhouse_data:
driver: local
langfuse_clickhouse_logs:
driver: local
langfuse_azurite_data:
driver: local
@@ -18,9 +18,14 @@ services:
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES: ${LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES:-true}
LANGFUSE_ASYNC_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_INGESTION_PROCESSING:-true}
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: ${LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING:-true}
LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE: ${LANGFUSE_READ_DASHBOARDS_FROM_CLICKHOUSE:-true}
LANGFUSE_READ_FROM_POSTGRES_ONLY: ${LANGFUSE_READ_FROM_POSTGRES_ONLY:-false}
LANGFUSE_RETURN_FROM_CLICKHOUSE: ${LANGFUSE_RETURN_FROM_CLICKHOUSE:-true}
CLICKHOUSE_MIGRATION_URL: ${CLICKHOUSE_MIGRATION_URL:-clickhouse://clickhouse:9000}
CLICKHOUSE_URL: ${CLICKHOUSE_URL:-http://clickhouse:8123}
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse}
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: ${LANGFUSE_S3_EVENT_UPLOAD_ENABLED:-true}
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-us-east-1}
@@ -28,6 +33,15 @@ services:
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: ${LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: ${LANGFUSE_S3_EVENT_UPLOAD_PREFIX:-events/}
LANGFUSE_S3_MEDIA_UPLOAD_ENABLED: ${LANGFUSE_S3_MEDIA_UPLOAD_ENABLED:-true}
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET: ${LANGFUSE_S3_MEDIA_UPLOAD_BUCKET:-langfuse}
LANGFUSE_S3_MEDIA_UPLOAD_REGION: ${LANGFUSE_S3_MEDIA_UPLOAD_REGION:-us-east-1}
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID:-minio}
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY: ${LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY:-miniosecret}
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT: ${LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT:-http://minio:9000}
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE: ${LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE:-true}
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX: ${LANGFUSE_S3_MEDIA_UPLOAD_PREFIX:-media/}
REDIS_HOST: ${REDIS_HOST:-redis}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_AUTH: ${REDIS_AUTH:-myredissecret}
@@ -98,7 +112,7 @@ services:
ports:
- 6379:6379
healthcheck:
test: [ 'CMD', 'redis-cli', 'ping' ]
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 10s
retries: 10
@@ -107,7 +121,7 @@ services:
image: postgres:${POSTGRES_VERSION:-latest}
restart: always
healthcheck:
test: [ "CMD-SHELL", "pg_isready -U postgres" ]
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 3s
timeout: 3s
retries: 10
+4 -3
View File
@@ -27,8 +27,9 @@
"@langfuse/shared": "workspace:*",
"@opentelemetry/api": ">=1.0.0 <1.10.0",
"axios": "^1.7.7",
"next": "^14.2.15",
"next-auth": "^4.24.7",
"https-proxy-agent": "^7.0.6",
"next": "^14.2.21",
"next-auth": "^4.24.11",
"zod": "^3.23.8"
},
"devDependencies": {
@@ -47,7 +48,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.2.0"
}
}
}
+103
View File
@@ -0,0 +1,103 @@
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
imports:
commons: ./commons.yml
service:
auth: true
base-path: /api/public
endpoints:
get:
docs: Get a media record
method: GET
path: /media/{mediaId}
path-parameters:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
response: GetMediaResponse
patch:
docs: Patch a media record
method: PATCH
path: /media/{mediaId}
path-parameters:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
request: PatchMediaBody
getUploadUrl:
docs: Get a presigned upload URL for a media record
method: POST
path: /media
request: GetMediaUploadUrlRequest
response: GetMediaUploadUrlResponse
types:
GetMediaResponse:
properties:
mediaId:
type: string
docs: The unique langfuse identifier of a media record
contentType:
type: string
docs: The MIME type of the media record
contentLength:
type: integer
docs: The size of the media record in bytes
uploadedAt:
type: datetime
docs: The date and time when the media record was uploaded
url:
type: string
docs: The download URL of the media record
urlExpiry:
type: string
docs: The expiry date and time of the media record download URL
PatchMediaBody:
properties:
uploadedAt:
type: datetime
docs: The date and time when the media record was uploaded
uploadHttpStatus:
type: integer
docs: The HTTP status code of the upload
uploadHttpError:
type: optional<string>
docs: The HTTP error message of the upload
uploadTimeMs:
type: optional<integer>
docs: The time in milliseconds it took to upload the media record
GetMediaUploadUrlRequest:
properties:
traceId:
type: string
docs: The trace ID associated with the media record
observationId:
type: optional<string>
docs: The observation ID associated with the media record. If the media record is associated directly with a trace, this will be null.
contentType: MediaContentType
contentLength:
type: integer
docs: The size of the media record in bytes
sha256Hash:
type: string
docs: The SHA-256 hash of the media record
field:
type: string
docs: The trace / observation field the media record is associated with. This can be one of `input`, `output`, `metadata`
GetMediaUploadUrlResponse:
properties:
uploadUrl:
type: optional<string>
docs: The presigned upload URL. If the asset is already uploaded, this will be null
mediaId:
type: string
docs: The unique langfuse identifier of a media record
MediaContentType:
type: literal<"image/png","image/jpeg","image/jpg","image/webp","audio/mpeg","audio/mp3","audio/wav","text/plain","application/pdf">
docs: The MIME type of the media record
+21 -2
View File
@@ -147,10 +147,29 @@ types:
tags:
type: optional<list<string>>
docs: A list of tags associated with the trace referenced by score
GetScoresResponseData:
GetScoresResponseDataNumeric:
extends: commons.NumericScore
properties:
<<: commons.Score
trace: GetScoresResponseTraceData
GetScoresResponseDataCategorical:
extends: commons.CategoricalScore
properties:
trace: GetScoresResponseTraceData
GetScoresResponseDataBoolean:
extends: commons.BooleanScore
properties:
trace: GetScoresResponseTraceData
GetScoresResponseData:
discriminant: dataType
union:
NUMERIC: GetScoresResponseDataNumeric
CATEGORICAL: GetScoresResponseDataCategorical
BOOLEAN: GetScoresResponseDataBoolean
GetScoresResponse:
properties:
data: list<GetScoresResponseData>
+7 -2
View File
@@ -1,6 +1,6 @@
{
"name": "langfuse",
"version": "2.87.0",
"version": "2.95.2",
"author": "engineering@langfuse.com",
"license": "MIT",
"private": true,
@@ -83,7 +83,12 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.2.0",
"nanoid": "^3.3.8",
"katex": "^0.16.21"
},
"patchedDependencies": {
"next-auth@4.24.11": "patches/next-auth@4.24.11.patch"
}
},
"packageManager": "pnpm@9.5.0"
@@ -0,0 +1 @@
DROP TABLE traces ON CLUSTER default;
@@ -0,0 +1,32 @@
CREATE TABLE traces ON CLUSTER default (
`id` String,
`timestamp` DateTime64(3),
`name` String,
`user_id` Nullable(String),
`metadata` Map(LowCardinality(String), String),
`release` Nullable(String),
`version` Nullable(String),
`project_id` String,
`public` Bool,
`bookmarked` Bool,
`tags` Array(String),
`input` Nullable(String) CODEC(ZSTD(3)),
`output` Nullable(String) CODEC(ZSTD(3)),
`session_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
updated_at DateTime64(3) DEFAULT now(),
`event_ts` DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_res_metadata_key mapKeys(metadata) TYPE bloom_filter(0.01) GRANULARITY 1,
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp)
)
ORDER BY (
project_id,
toDate(timestamp),
id
);
@@ -0,0 +1 @@
DROP TABLE observations ON CLUSTER default;
@@ -0,0 +1,47 @@
CREATE TABLE observations ON CLUSTER default (
`id` String,
`trace_id` String,
`project_id` String,
`type` LowCardinality(String),
`parent_observation_id` Nullable(String),
`start_time` DateTime64(3),
`end_time` Nullable(DateTime64(3)),
`name` String,
`metadata` Map(LowCardinality(String), String),
`level` LowCardinality(String),
`status_message` Nullable(String),
`version` Nullable(String),
`input` Nullable(String) CODEC(ZSTD(3)),
`output` Nullable(String) CODEC(ZSTD(3)),
`provided_model_name` Nullable(String),
`internal_model_id` Nullable(String),
`model_parameters` Nullable(String),
`provided_usage_details` Map(LowCardinality(String), UInt64),
`usage_details` Map(LowCardinality(String), UInt64),
`provided_cost_details` Map(LowCardinality(String), Decimal64(12)),
`cost_details` Map(LowCardinality(String), Decimal64(12)),
`total_cost` Nullable(Decimal64(12)),
`completion_start_time` Nullable(DateTime64(3)),
`prompt_id` Nullable(String),
`prompt_name` Nullable(String),
`prompt_version` Nullable(UInt16),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
is_deleted UInt8,
INDEX idx_id id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_trace_id trace_id TYPE bloom_filter() GRANULARITY 1,
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
PRIMARY KEY (
project_id,
`type`,
toDate(start_time)
)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
@@ -0,0 +1 @@
DROP TABLE scores ON CLUSTER default;
@@ -0,0 +1,33 @@
CREATE TABLE scores ON CLUSTER default (
`id` String,
`timestamp` DateTime64(3),
`project_id` String,
`trace_id` String,
`observation_id` Nullable(String),
`name` String,
`value` Float64,
`source` String,
`comment` Nullable(String) CODEC(ZSTD(1)),
`author_user_id` Nullable(String),
`config_id` Nullable(String),
`data_type` String,
`string_value` Nullable(String),
`queue_id` Nullable(String),
`created_at` DateTime64(3) DEFAULT now(),
`updated_at` DateTime64(3) DEFAULT now(),
event_ts DateTime64(3),
`is_deleted` UInt8,
INDEX idx_id id TYPE bloom_filter(0.001) GRANULARITY 1,
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
) ENGINE = ReplicatedReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp),
name
)
ORDER BY (
project_id,
toDate(timestamp),
name,
id
)
@@ -0,0 +1 @@
ALTER TABLE observations ON CLUSTER default ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
@@ -0,0 +1 @@
ALTER TABLE observations ON CLUSTER default DROP INDEX IF EXISTS idx_project_id;
@@ -0,0 +1 @@
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_session_id;
@@ -0,0 +1,2 @@
ALTER TABLE traces ON CLUSTER default ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces ON CLUSTER default MATERIALIZE INDEX IF EXISTS idx_session_id;
@@ -22,11 +22,11 @@ CREATE TABLE traces (
INDEX idx_res_metadata_value mapValues(metadata) TYPE bloom_filter(0.01) GRANULARITY 1
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp)
project_id,
toDate(timestamp)
)
ORDER BY (
project_id,
toDate(timestamp),
id
);
project_id,
toDate(timestamp),
id
);
@@ -34,14 +34,14 @@ CREATE TABLE observations (
INDEX idx_project_id project_id TYPE bloom_filter() GRANULARITY 1
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(start_time)
PRIMARY KEY (
project_id,
`type`,
toDate(start_time)
)
project_id,
`type`,
toDate(start_time)
)
ORDER BY (
project_id,
`type`,
toDate(start_time),
id
);
project_id,
`type`,
toDate(start_time),
id
);
@@ -21,13 +21,13 @@ CREATE TABLE scores (
INDEX idx_project_trace_observation (project_id, trace_id, observation_id) TYPE bloom_filter(0.001) GRANULARITY 1
) ENGINE = ReplacingMergeTree(event_ts, is_deleted) Partition by toYYYYMM(timestamp)
PRIMARY KEY (
project_id,
toDate(timestamp),
name
)
project_id,
toDate(timestamp),
name
)
ORDER BY (
project_id,
toDate(timestamp),
name,
id
)
project_id,
toDate(timestamp),
name,
id
)
@@ -0,0 +1 @@
ALTER TABLE observations ADD INDEX IF NOT EXISTS idx_project_id project_id TYPE bloom_filter() GRANULARITY 1;
@@ -0,0 +1 @@
ALTER TABLE observations DROP INDEX IF EXISTS idx_project_id;
@@ -0,0 +1 @@
ALTER TABLE traces DROP INDEX IF EXISTS idx_session_id;
@@ -0,0 +1,2 @@
ALTER TABLE traces ADD INDEX IF NOT EXISTS idx_session_id session_id TYPE bloom_filter() GRANULARITY 1;
ALTER TABLE traces MATERIALIZE INDEX IF EXISTS idx_session_id;
+25 -7
View File
@@ -1,7 +1,13 @@
#!/bin/bash
# Load environment variables
source ../../.env
[ -f ../../.env ] && source ../../.env
# Check if CLICKHOUSE_URL is configured
if [ -z "${CLICKHOUSE_URL}" ]; then
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
exit 0
fi
# Check if golang-migrate is installed
if ! command -v migrate &> /dev/null
@@ -13,10 +19,22 @@ then
fi
# Construct the database URL
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the down command
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" down
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
fi
+55 -8
View File
@@ -1,24 +1,71 @@
import { clickhouseClient } from "@langfuse/shared/src/server";
import { prisma } from "../../src/db";
import {
clickhouseClient,
ObservationRecordReadType,
} from "@langfuse/shared/src/server";
import { Prisma, prisma } from "../../src/db";
import { redis } from "@langfuse/shared/src/server";
import { prepareClickhouse } from "../../scripts/prepareClickhouse";
import { createDatasets } from "../../prisma/seed";
import { queryClickhouse } from "../../src/server/repositories/clickhouse";
import { convertObservation } from "../../src/server/repositories/observations_converters";
async function main() {
try {
const projectIds = [
"7a88fb47-b4e2-43b8-a06c-a5ce950dc53a",
"239ad00f-562f-411d-af14-831c75ddd875",
]; // Example project IDs
const projectIds = ["7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"]; // Example project IDs
if (
await prisma.project.findFirst({
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
})
) {
projectIds.push("239ad00f-562f-411d-af14-831c75ddd875");
}
await prepareClickhouse(projectIds, {
numberOfDays: 3,
totalObservations: 1000,
totalObservations: 10000,
});
const project1 = await prisma.project.findFirst({
where: { id: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a" },
});
const project2 =
projectIds.length > 1
? await prisma.project.findFirst({
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
})
: await prisma.project.findFirst();
const query = `
SELECT *
FROM observations o
WHERE o.project_id IN ({projectIds: Array(String)})
LIMIT 2000;
`;
const res = await queryClickhouse<ObservationRecordReadType>({
query,
params: {
projectIds,
},
});
await createDatasets(
project1!,
project2!,
(await Promise.all(res.map(convertObservation))).map((o) => ({
...o,
metadata: {},
modelParameters: {},
input: {},
output: {},
})),
);
console.log("Clickhouse preparation completed successfully.");
} catch (error) {
console.error("Error during Clickhouse preparation:", error);
} finally {
await clickhouseClient.close();
await clickhouseClient().close();
await prisma.$disconnect();
redis?.disconnect();
console.log("Disconnected from Clickhouse.");
+25 -8
View File
@@ -1,7 +1,13 @@
#!/bin/bash
# Load environment variables
source ../../.env
[ -f ../../.env ] && source ../../.env
# Check if CLICKHOUSE_URL is configured
if [ -z "${CLICKHOUSE_URL}" ]; then
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
exit 0
fi
# Check if golang-migrate is installed
if ! command -v migrate &> /dev/null
@@ -13,11 +19,22 @@ then
fi
# Construct the database URL
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "true" ] ; then
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-cluster-name=default&x-migrations-table-engine=ReplicatedMergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations -database "$DATABASE_URL" up
# Execute the up command
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" up
else
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&secure=true&skip_verify=true&x-migrations-table-engine=MergeTree"
else
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=default&x-multi-statement=true&x-migrations-table-engine=MergeTree"
fi
# Execute the up command
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" up
fi
+12 -10
View File
@@ -43,7 +43,6 @@
"db:generate": "dotenv -e ../../.env -- npx prisma generate",
"db:seed:examples": "dotenv -e ../../.env -- npx prisma db seed -- --environment examples",
"db:seed:load": "dotenv -e ../../.env -- npx prisma db seed -- --environment load",
"ch:status": "dotenv -e ../../.env -- goose -dir './clickhouse/migrations/' status",
"ch:up": "bash clickhouse/scripts/up.sh",
"ch:down": "bash clickhouse/scripts/down.sh",
"ch:drop": "bash clickhouse/scripts/drop.sh",
@@ -60,13 +59,14 @@
"@aws-sdk/client-s3": "^3.675.0",
"@aws-sdk/lib-storage": "^3.675.0",
"@aws-sdk/s3-request-presigner": "^3.679.0",
"@azure/storage-blob": "^12.26.0",
"@clickhouse/client": "^1.4.0",
"@langchain/anthropic": "^0.3.1",
"@langchain/aws": "^0.1.0",
"@langchain/core": "^0.3.9",
"@langchain/openai": "^0.3.0",
"@langchain/anthropic": "^0.3.8",
"@langchain/aws": "^0.1.2",
"@langchain/core": "^0.3.18",
"@langchain/openai": "^0.3.14",
"@opentelemetry/api": ">=1.0.0 <1.10.0",
"@prisma/client": "^5.20.0",
"@prisma/client": "^5.22.0",
"@react-email/components": "^0.0.19",
"@react-email/render": "^0.0.15",
"@types/bcryptjs": "^2.4.6",
@@ -76,11 +76,13 @@
"dd-trace": "^5.23.1",
"decimal.js": "^10.4.3",
"exponential-backoff": "^3.1.1",
"https-proxy-agent": "^7.0.6",
"ioredis": "^5.4.1",
"kysely": "^0.27.4",
"langchain": "^0.3.2",
"langchain": "^0.3.6",
"langfuse-langchain": "3.30.3",
"lodash": "^4.17.21",
"next-auth": "^4.24.7",
"next-auth": "^4.24.11",
"nodemailer": "^6.9.15",
"prisma-extension-kysely": "^2.1.0",
"uuid": "^9.0.1",
@@ -104,7 +106,7 @@
"kysely-codegen": "^0.16.8",
"nodemon": "^3.1.7",
"prettier": "^3.3.3",
"prisma": "^5.20.0",
"prisma": "^5.22.0",
"prisma-erd-generator": "^1.11.2",
"prisma-kysely": "^1.8.0",
"ts-node": "^10.9.2",
@@ -119,7 +121,7 @@
},
"pnpm": {
"overrides": {
"jsonpath-plus": "10.0.0"
"jsonpath-plus": "10.0.7"
}
}
}
+39
View File
@@ -148,6 +148,7 @@ export type BackgroundMigration = {
name: string;
script: string;
args: unknown;
state: Generated<unknown>;
finished_at: Timestamp | null;
failed_at: Timestamp | null;
failed_reason: string | null;
@@ -277,6 +278,8 @@ export type JobExecution = {
end_time: Timestamp | null;
error: string | null;
job_input_trace_id: string | null;
job_input_observation_id: string | null;
job_input_dataset_item_id: string | null;
job_output_score_id: string | null;
};
export type LlmApiKeys = {
@@ -293,6 +296,20 @@ export type LlmApiKeys = {
config: unknown | null;
project_id: string;
};
export type Media = {
id: string;
sha_256_hash: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
uploaded_at: Timestamp | null;
upload_http_status: number | null;
upload_http_error: string | null;
bucket_path: string;
bucket_name: string;
content_type: string;
content_length: string;
};
export type MembershipInvitation = {
id: string;
email: string;
@@ -353,6 +370,16 @@ export type Observation = {
completion_start_time: Timestamp | null;
prompt_id: string | null;
};
export type ObservationMedia = {
id: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
media_id: string;
trace_id: string;
observation_id: string;
field: string;
};
export type ObservationView = {
id: string;
trace_id: string | null;
@@ -514,6 +541,15 @@ export type Trace = {
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
};
export type TraceMedia = {
id: string;
project_id: string;
created_at: Generated<Timestamp>;
updated_at: Generated<Timestamp>;
media_id: string;
trace_id: string;
field: string;
};
export type TraceSession = {
id: string;
created_at: Generated<Timestamp>;
@@ -578,8 +614,10 @@ export type DB = {
job_configurations: JobConfiguration;
job_executions: JobExecution;
llm_api_keys: LlmApiKeys;
media: Media;
membership_invitations: MembershipInvitation;
models: Model;
observation_media: ObservationMedia;
observations: Observation;
observations_view: ObservationView;
organization_memberships: OrganizationMembership;
@@ -593,6 +631,7 @@ export type DB = {
scores: Score;
Session: Session;
sso_configs: SsoConfig;
trace_media: TraceMedia;
trace_sessions: TraceSession;
traces: Trace;
traces_view: TraceView;
@@ -0,0 +1,2 @@
-- AlterTable
ALTER TABLE "background_migrations" ADD COLUMN "state" jsonb NOT NULL DEFAULT '{}';
@@ -0,0 +1,29 @@
INSERT INTO models (
id,
project_id,
model_name,
match_pattern,
start_date,
input_price,
output_price,
total_price,
unit,
tokenizer_id,
tokenizer_config
)
VALUES
-- https://docs.anthropic.com/en/docs/about-claude/models#model-comparison-table
('cm34aq60d000207ml0j1h31ar', NULL, 'claude-3-5-haiku-20241022', '(?i)^(claude-3-5-haiku-20241022|anthropic\.claude-3-5-haiku-20241022-v1:0|claude-3-5-haiku-V1@20241022)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL),
('cm34aqb9h000307ml6nypd618', NULL, 'claude-3.5-haiku-latest', '(?i)^(claude-3-5-haiku-latest)$', NULL, 0.000001, 0.000005, NULL, 'TOKENS', 'claude', NULL);
INSERT INTO prices (
id,
model_id,
usage_type,
price
)
VALUES
('cm34ax6mc000008jkfqed92mb', 'cm34aq60d000207ml0j1h31ar', 'input', 0.000001),
('cm34axb2o000108jk09wn9b47', 'cm34aqb9h000307ml6nypd618', 'input', 0.000001),
('cm34axeie000208jk8b2ke2t8', 'cm34aq60d000207ml0j1h31ar', 'output', 0.000005),
('cm34axi67000308jk7x1a7qko', 'cm34aqb9h000307ml6nypd618', 'output', 0.000005);
@@ -0,0 +1,71 @@
-- CreateTable
CREATE TABLE "media" (
"id" TEXT NOT NULL,
"sha_256_hash" CHAR(44) NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"uploaded_at" TIMESTAMP(3),
"upload_http_status" INTEGER,
"upload_http_error" TEXT,
"bucket_path" TEXT NOT NULL,
"bucket_name" TEXT NOT NULL,
"content_type" TEXT NOT NULL,
"content_length" BIGINT NOT NULL,
CONSTRAINT "media_pkey" PRIMARY KEY ("id")
);
-- CreateTable
CREATE TABLE "trace_media" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"media_id" TEXT NOT NULL,
"trace_id" TEXT NOT NULL,
"field" TEXT NOT NULL,
CONSTRAINT "trace_media_pkey" PRIMARY KEY ("id")
);
-- CreateTable
CREATE TABLE "observation_media" (
"id" TEXT NOT NULL,
"project_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"media_id" TEXT NOT NULL,
"trace_id" TEXT NOT NULL,
"observation_id" TEXT NOT NULL,
"field" TEXT NOT NULL,
CONSTRAINT "observation_media_pkey" PRIMARY KEY ("id")
);
-- CreateIndex
CREATE UNIQUE INDEX "media_project_id_sha_256_hash_key" ON "media"("project_id", "sha_256_hash");
-- CreateIndex
CREATE UNIQUE INDEX "trace_media_project_id_trace_id_media_id_field_key" ON "trace_media"("project_id", "trace_id", "media_id", "field");
-- CreateIndex
CREATE INDEX "observation_media_project_id_observation_id_idx" ON "observation_media"("project_id", "observation_id");
-- CreateIndex
CREATE UNIQUE INDEX "observation_media_project_id_trace_id_observation_id_media__key" ON "observation_media"("project_id", "trace_id", "observation_id", "media_id", "field");
-- AddForeignKey
ALTER TABLE "media" ADD CONSTRAINT "media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "trace_media" ADD CONSTRAINT "trace_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
-- AddForeignKey
ALTER TABLE "observation_media" ADD CONSTRAINT "observation_media_media_id_fkey" FOREIGN KEY ("media_id") REFERENCES "media"("id") ON DELETE CASCADE ON UPDATE CASCADE;
@@ -0,0 +1,6 @@
-- DropForeignKey
ALTER TABLE "job_executions" DROP CONSTRAINT "job_executions_job_input_trace_id_fkey";
-- AlterTable
ALTER TABLE "job_executions" ADD COLUMN "job_input_dataset_item_id" TEXT,
ADD COLUMN "job_input_observation_id" TEXT;
@@ -0,0 +1,25 @@
INSERT INTO models (
id,
project_id,
model_name,
match_pattern,
start_date,
input_price,
output_price,
total_price,
unit,
tokenizer_id,
tokenizer_config
)
VALUES
('cm3x0p8ev000008kyd96800c8', NULL, 'chatgpt-4o-latest', '(?i)^(chatgpt-4o-latest)$', NULL, 0.000005, 0.000015, NULL, 'TOKENS', 'openai', '{ "tokensPerMessage": 3, "tokensPerName": 1, "tokenizerModel": "gpt-4o" }');
INSERT INTO prices (
id,
model_id,
usage_type,
price
)
VALUES
('cm3x0psrz000108kydpxg9o2k', 'cm3x0p8ev000008kyd96800c8', 'input', 0.000005),
('cm3x0pyt7000208ky8737gdla', 'cm3x0p8ev000008kyd96800c8', 'output', 0.000015);
+62 -4
View File
@@ -136,6 +136,9 @@ model Project {
comment Comment[]
annotationQueue AnnotationQueue[]
annotationQueueItem AnnotationQueueItem[]
TraceMedia TraceMedia[]
Media Media[]
ObservationMedia ObservationMedia[]
@@index([orgId])
@@map("projects")
@@ -166,6 +169,7 @@ model BackgroundMigration {
name String @unique
script String @map("script")
args Json @map("args")
state Json @default("{}") @map("state")
finishedAt DateTime? @map("finished_at")
failedAt DateTime? @map("failed_at")
failedReason String? @map("failed_reason")
@@ -299,8 +303,6 @@ model Trace {
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
JobExecution JobExecution[]
@@index([projectId, timestamp])
@@index([sessionId])
@@index([name])
@@ -894,8 +896,11 @@ model JobExecution {
endTime DateTime? @map("end_time")
error String?
jobInputTraceId String? @map("job_input_trace_id")
trace Trace? @relation(fields: [jobInputTraceId], references: [id], onDelete: SetNull) // job remains when traces are deleted
jobInputTraceId String? @map("job_input_trace_id") // no fk constraint - traces in ClickHouse, deletion handled via project cascade
jobInputObservationId String? @map("job_input_observation_id") // no fk constraint - observations in ClickHouse, deletion handled via project cascade
jobInputDatasetItemId String? @map("job_input_dataset_item_id") // no fk constraint - job execution sensible standalone
jobOutputScoreId String? @map("job_output_score_id")
score Score? @relation(fields: [jobOutputScoreId], references: [id], onDelete: SetNull) // job remains when scores are deleted
@@ -960,3 +965,56 @@ model BatchExport {
@@index([status])
@@map("batch_exports")
}
model Media {
id String @id @default(cuid())
sha256Hash String @map("sha_256_hash") @db.Char(44)
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
uploadedAt DateTime? @map("uploaded_at")
uploadHttpStatus Int? @map("upload_http_status")
uploadHttpError String? @map("upload_http_error")
bucketPath String @map("bucket_path")
bucketName String @map("bucket_name")
contentType String @map("content_type")
contentLength BigInt @map("content_length")
TraceMedia TraceMedia[]
ObservationMedia ObservationMedia[]
@@unique([projectId, sha256Hash])
@@map("media")
}
model TraceMedia {
id String @id @default(cuid())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
mediaId String @map("media_id")
media Media @relation(fields: [mediaId], references: [id], onDelete: Cascade)
traceId String @map("trace_id")
field String @map("field")
@@unique([projectId, traceId, mediaId, field])
@@map("trace_media")
}
model ObservationMedia {
id String @id @default(cuid())
projectId String @map("project_id")
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
createdAt DateTime @default(now()) @map("created_at")
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
mediaId String @map("media_id")
media Media @relation(fields: [mediaId], references: [id], onDelete: Cascade)
traceId String @map("trace_id")
observationId String @map("observation_id")
field String @map("field")
@@unique([projectId, traceId, observationId, mediaId, field])
@@index([projectId, observationId])
@@map("observation_media")
}
+157 -118
View File
@@ -256,7 +256,7 @@ async function main() {
const queueIds = await generateQueuesForProject(
[project1, project2],
configIdsAndNames
configIdsAndNames,
);
const promptIds = await generatePromptsForProject([project1, project2]);
@@ -282,11 +282,11 @@ async function main() {
project2,
promptIds,
queueIds,
configIdsAndNames
configIdsAndNames,
);
logger.info(
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`
`Seeding ${traces.length} traces, ${observations.length} observations, and ${scores.length} scores`,
);
await uploadObjects(
@@ -296,7 +296,7 @@ async function main() {
sessions,
events,
comments,
queueItems
queueItems,
);
// If openai key is in environment, add it to the projects LLM API keys
@@ -314,7 +314,7 @@ async function main() {
});
} else {
logger.warn(
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key."
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key.",
);
}
@@ -387,16 +387,62 @@ async function main() {
update: {},
});
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
const dataset = await prisma.dataset.create({
data: {
name: `demo-dataset-${datasetNumber}`,
description:
datasetNumber === 0 ? "Dataset test description" : undefined,
projectId: project2.id,
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
},
});
await createDatasets(project1, project2, observations);
}
}
main()
.then(async () => {
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
})
.catch(async (e) => {
logger.error(e);
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
process.exit(1);
});
export async function createDatasets(
project1: {
id: string;
orgId: string;
createdAt: Date;
updatedAt: Date;
name: string;
},
project2: {
id: string;
orgId: string;
createdAt: Date;
updatedAt: Date;
name: string;
},
observations: Prisma.ObservationCreateManyInput[],
) {
for (let datasetNumber = 0; datasetNumber < 2; datasetNumber++) {
for (const projectId of [project1.id, project2.id]) {
const datasetName = `demo-dataset-${datasetNumber}`;
// check if ds already exists
const dataset =
(await prisma.dataset.findFirst({
where: {
projectId,
name: datasetName,
},
})) ??
(await prisma.dataset.create({
data: {
name: datasetName,
description:
datasetNumber === 0 ? "Dataset test description" : undefined,
projectId,
metadata: datasetNumber === 0 ? { key: "value" } : undefined,
},
}));
const datasetItemIds = [];
for (let i = 0; i < 18; i++) {
@@ -406,7 +452,7 @@ async function main() {
: undefined;
const datasetItem = await prisma.datasetItem.create({
data: {
projectId: project2.id,
projectId,
datasetId: dataset.id,
sourceTraceId: sourceObservation?.traceId,
sourceObservationId:
@@ -431,9 +477,16 @@ async function main() {
}
for (let datasetRunNumber = 0; datasetRunNumber < 5; datasetRunNumber++) {
const datasetRun = await prisma.datasetRuns.create({
data: {
projectId: project2.id,
const datasetRun = await prisma.datasetRuns.upsert({
where: {
datasetId_projectId_name: {
datasetId: dataset.id,
projectId,
name: `demo-dataset-run-${datasetRunNumber}`,
},
},
create: {
projectId,
name: `demo-dataset-run-${datasetRunNumber}`,
description: Math.random() > 0.5 ? "Dataset run description" : "",
datasetId: dataset.id,
@@ -445,11 +498,12 @@ async function main() {
["tag1", "tag2"],
][datasetRunNumber % 5],
},
update: {},
});
for (const datasetItemId of datasetItemIds) {
const relevantObservations = observations.filter(
(o) => o.projectId === project2.id
(o) => o.projectId === projectId,
);
const observation =
relevantObservations[
@@ -458,7 +512,7 @@ async function main() {
await prisma.datasetRunItems.create({
data: {
projectId: project2.id,
projectId,
datasetItemId,
traceId: observation.traceId as string,
observationId: Math.random() > 0.5 ? observation.id : undefined,
@@ -471,20 +525,6 @@ async function main() {
}
}
main()
.then(async () => {
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
})
.catch(async (e) => {
logger.error(e);
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from postgres and redis");
process.exit(1);
});
async function uploadObjects(
traces: Prisma.TraceCreateManyInput[],
observations: Prisma.ObservationCreateManyInput[],
@@ -492,7 +532,7 @@ async function uploadObjects(
sessions: Prisma.TraceSessionCreateManyInput[],
events: Prisma.ObservationCreateManyInput[],
comments: Prisma.CommentCreateManyInput[],
queueItems: Prisma.AnnotationQueueItemCreateManyInput[]
queueItems: Prisma.AnnotationQueueItemCreateManyInput[],
) {
let promises: Prisma.PrismaPromise<unknown>[] = [];
@@ -506,14 +546,14 @@ async function uploadObjects(
},
create: chunk[0]!,
update: {},
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Sessions ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -524,13 +564,13 @@ async function uploadObjects(
promises.push(
prisma.trace.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Traces ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -540,14 +580,14 @@ async function uploadObjects(
promises.push(
prisma.observation.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Observations ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -557,14 +597,14 @@ async function uploadObjects(
promises.push(
prisma.observation.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Events ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -574,13 +614,13 @@ async function uploadObjects(
promises.push(
prisma.score.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Scores ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -590,13 +630,13 @@ async function uploadObjects(
promises.push(
prisma.comment.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Comments ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -606,13 +646,13 @@ async function uploadObjects(
promises.push(
prisma.annotationQueueItem.createMany({
data: chunk,
})
}),
);
});
for (let i = 0; i < promises.length; i++) {
if (i + 1 >= promises.length || i % Math.ceil(promises.length / 10) === 0)
logger.info(
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`
`Seeding of Annotation Queue Items ${((i + 1) / promises.length) * 100}% complete`,
);
await promises[i];
}
@@ -634,7 +674,7 @@ function createObjects(
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
>
>,
) {
const traces: Prisma.TraceCreateManyInput[] = [];
const observations: Prisma.ObservationCreateManyInput[] = [];
@@ -649,7 +689,7 @@ function createObjects(
// print progress to console with a progress bar that refreshes every 10 iterations
// random date within last 90 days, with a linear bias towards more recent dates
const traceTs = new Date(
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000)
Date.now() - Math.floor(Math.random() ** 1.5 * 90 * 24 * 60 * 60 * 1000),
);
const envTag = envTags[Math.floor(Math.random() * envTags.length)];
@@ -805,11 +845,11 @@ function createObjects(
for (let j = 0; j < Math.floor(Math.random() * 10) + 1; j++) {
// add between 1 and 30 ms to trace timestamp
const spanTsStart = new Date(
traceTs.getTime() + Math.floor(Math.random() * 30)
traceTs.getTime() + Math.floor(Math.random() * 30),
);
// random duration of upto 5000ms
const spanTsEnd = new Date(
spanTsStart.getTime() + Math.floor(Math.random() * 5000)
spanTsStart.getTime() + Math.floor(Math.random() * 5000),
);
const span = {
@@ -844,22 +884,22 @@ function createObjects(
const generationTsStart = new Date(
spanTsStart.getTime() +
Math.floor(
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
)
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
),
);
const generationTsEnd = new Date(
generationTsStart.getTime() +
Math.floor(
Math.random() *
(spanTsEnd.getTime() - generationTsStart.getTime())
)
(spanTsEnd.getTime() - generationTsStart.getTime()),
),
);
// somewhere in the middle
const generationTsCompletionStart = new Date(
generationTsStart.getTime() +
Math.floor(
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3
)
(generationTsEnd.getTime() - generationTsStart.getTime()) / 3,
),
);
const promptTokens = Math.floor(Math.random() * 1000) + 300;
@@ -880,7 +920,7 @@ function createObjects(
const promptId =
promptIds.get(projectId)![
Math.floor(
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2)
Math.random() * Math.floor(promptIds.get(projectId)!.length / 2),
)
];
@@ -962,8 +1002,8 @@ function createObjects(
const eventTs = new Date(
spanTsStart.getTime() +
Math.floor(
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime())
)
Math.random() * (spanTsEnd.getTime() - spanTsStart.getTime()),
),
);
events.push({
@@ -985,7 +1025,7 @@ function createObjects(
}
// find unique sessions by id and projectid
const uniqueSessions: Prisma.TraceSessionCreateManyInput[] = Array.from(
new Set(sessions.map((session) => JSON.stringify(session)))
new Set(sessions.map((session) => JSON.stringify(session))),
).map((session) => JSON.parse(session) as Prisma.TraceSessionCreateManyInput);
return {
@@ -1007,65 +1047,64 @@ async function generatePromptsForProject(projects: Project[]) {
projects.map(async (project) => {
const promptIdsForProject = await generatePrompts(project);
promptIds.set(project.id, promptIdsForProject);
})
}),
);
return promptIds;
}
async function generatePrompts(project: Project) {
const promptIds: string[] = [];
const prompts = [
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 1 content",
name: "Prompt 1",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 2 content",
name: "Prompt 2",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "API",
prompt: "Prompt 3 content",
name: "Prompt 3 by API",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-${v4()}`,
projectId: project.id,
createdBy: "user-1",
prompt: "Prompt 4 content",
name: "Prompt 4",
version: 1,
labels: ["production", "latest"],
tags: ["tag1", "tag2"],
},
];
export const SEED_PROMPTS = [
{
id: `prompt-123`,
createdBy: "user-1",
prompt: "Prompt 1 content",
name: "Prompt 1",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-456`,
createdBy: "user-1",
prompt: "Prompt 2 content",
name: "Prompt 2",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-789`,
createdBy: "API",
prompt: "Prompt 3 content",
name: "Prompt 3 by API",
version: 1,
labels: ["production", "latest"],
},
{
id: `prompt-abc`,
createdBy: "user-1",
prompt: "Prompt 4 content",
name: "Prompt 4",
version: 1,
labels: ["production", "latest"],
tags: ["tag1", "tag2"],
},
];
for (const prompt of prompts) {
export const PROMPT_IDS: string[] = [];
async function generatePrompts(project: Project) {
const promptIds = [];
for (const prompt of SEED_PROMPTS) {
await prisma.prompt.upsert({
where: {
projectId_name_version: {
projectId: prompt.projectId,
projectId: prompt.id + project.id,
name: prompt.name,
version: prompt.version,
},
id: prompt.id + project.id,
},
create: {
id: prompt.id,
projectId: prompt.projectId,
id: prompt.id + project.id,
projectId: project.id,
createdBy: prompt.createdBy,
prompt: prompt.prompt,
name: prompt.name,
@@ -1073,9 +1112,7 @@ async function generatePrompts(project: Project) {
labels: prompt.labels,
tags: prompt.tags,
},
update: {
id: prompt.id,
},
update: {},
});
promptIds.push(prompt.id);
}
@@ -1129,6 +1166,7 @@ async function generatePrompts(project: Project) {
name: version.name,
version: version.version,
},
id: version.id,
},
create: {
id: version.id,
@@ -1159,6 +1197,7 @@ async function generatePrompts(project: Project) {
name: promptName,
version: i,
},
id: promptId,
},
create: {
id: promptId,
@@ -1193,7 +1232,7 @@ async function generateConfigsForProject(projects: Project[]) {
projects.map(async (project) => {
const configNameAndId = await generateConfigs(project);
projectIdsToConfigs.set(project.id, configNameAndId);
})
}),
);
return projectIdsToConfigs;
}
@@ -1343,7 +1382,7 @@ async function generateQueuesForProject(
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
>
>,
) {
const projectIdsToQueues: Map<string, string[]> = new Map();
@@ -1351,10 +1390,10 @@ async function generateQueuesForProject(
projects.map(async (project) => {
const queueIds = await generateQueues(
project,
configIdsAndNames.get(project.id) ?? []
configIdsAndNames.get(project.id) ?? [],
);
projectIdsToQueues.set(project.id, queueIds);
})
}),
);
return projectIdsToQueues;
}
@@ -1366,7 +1405,7 @@ async function generateQueues(
id: string;
dataType: ScoreDataType;
categories: ConfigCategory[] | null;
}[]
}[],
) {
const queue = {
id: `queue-${v4()}`,
+11 -10
View File
@@ -13,7 +13,7 @@ const createRandomProjectId = () => randomUUID().toString();
const prepareProjectsAndApiKeys = async (
numOfProjects: number,
opts: { requiredProjectIds: string[] }
opts: { requiredProjectIds: string[] },
) => {
const { requiredProjectIds } = opts;
const projectsToCreate = numOfProjects - requiredProjectIds.length;
@@ -51,7 +51,7 @@ const prepareProjectsAndApiKeys = async (
});
if (!apiKeyExists) {
const sk = await hashSecretKey(
`sk-${Math.random().toString(36).substr(2, 9)}`
`sk-${Math.random().toString(36).substr(2, 9)}`,
);
await prisma.apiKey.create({
data: {
@@ -79,31 +79,32 @@ async function main() {
let numberOfDays = parseInt(process.argv[3], 10);
let totalObservations = parseInt(process.argv[4], 10);
logger.info(process.argv);
logger.info(
`Preparing Clickhouse for ${numOfProjects} projects and ${numberOfDays} days with max Observations ${totalObservations}.`,
);
if (isNaN(totalObservations)) {
logger.warn(
"Total observations not provided or invalid. Defaulting to 1000 observations."
"Total observations not provided or invalid. Defaulting to 1000 observations.",
);
totalObservations = 1000;
}
if (isNaN(numOfProjects)) {
logger.warn(
"Number of projects not provided or invalid. Defaulting to 10 projects."
"Number of projects not provided or invalid. Defaulting to 10 projects.",
);
numOfProjects = 10;
}
if (isNaN(numberOfDays)) {
logger.warn(
"Number of days not provided or invalid. Defaulting to 3 days."
"Number of days not provided or invalid. Defaulting to 3 days.",
);
numberOfDays = 3;
}
logger.info(
`Preparing Clickhouse for ${numOfProjects} projects and ${numberOfDays} days with max Observations ${totalObservations}.`
);
try {
const projectIds = [
"7a88fb47-b4e2-43b8-a06c-a5ce950dc53a",
@@ -123,7 +124,7 @@ async function main() {
} catch (error) {
logger.error("Error during Clickhouse preparation:", error);
} finally {
await clickhouseClient.close();
await clickhouseClient().close();
await prisma.$disconnect();
redis?.disconnect();
logger.info("Disconnected from Clickhouse.");
+50 -12
View File
@@ -1,3 +1,5 @@
import { SEED_PROMPTS } from "../prisma/seed";
import { prisma } from "../src/db";
import { clickhouseClient, logger } from "../src/server";
function randn_bm(min: number, max: number, skew: number) {
@@ -60,7 +62,7 @@ export const prepareClickhouse = async (
SELECT toString(number) AS id,
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
concat('name_', toString(rand() % 100)) AS name,
concat('user_id_', toString(randUniform(0, 100))) AS user_id,
concat('user_id_', toInt64(randExponential(1 / 100))) AS user_id,
map('key', 'value') AS metadata,
concat('release_', toString(randUniform(0, 100))) AS release,
concat('version_', toString(randUniform(0, 100))) AS version,
@@ -70,7 +72,7 @@ export const prepareClickhouse = async (
array('tag1', 'tag2') as tags,
repeat('input', toInt64(randExponential(1 / 100))) AS input,
repeat('output', toInt64(randExponential(1 / 100))) AS output,
concat('session_', toString(rand() % 100)) AS session_id,
if(randUniform(0, 1) < 0.2, NULL, concat('session_', toString(rand() % 1000))) AS session_id,
timestamp AS created_at,
timestamp AS updated_at,
timestamp AS event_ts,
@@ -83,13 +85,13 @@ export const prepareClickhouse = async (
SELECT toString(number) AS id,
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
'${projectId}' AS project_id,
if(rand() < 0.47, 'GENERATION', if(rand() < 0.94, 'SPAN', 'EVENT')) AS type,
if(randUniform(0, 1) < 0.47, 'GENERATION', if(randUniform(0, 1) < 0.94, 'SPAN', 'EVENT')) AS type,
toString(rand()) AS parent_observation_id,
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS start_time,
addSeconds(start_time, if(rand() < 0.6, floor(randUniform(0, 20)), floor(randUniform(0, 3600)))) AS end_time,
concat('name', toString(rand() % 100)) AS name,
map('key', 'value') AS metadata,
if(rand() < 0.9, 'DEFAULT', if(rand() < 0.5, 'ERROR', if(rand() < 0.5, 'DEBUG', 'WARNING'))) AS level,
if(randUniform(0, 1) < 0.9, 'DEFAULT', if(randUniform(0, 1) < 0.5, 'ERROR', if(randUniform(0, 1) < 0.5, 'DEBUG', 'WARNING'))) AS level,
'status_message' AS status_message,
'version' AS version,
repeat('input', toInt64(randExponential(1 / 100))) AS input,
@@ -102,16 +104,22 @@ export const prepareClickhouse = async (
when number % 2 = 0 then 'cltr0w45b000008k1407o9qv1'
else 'clrntkjgy000f08jx79v9g1xj'
end as internal_model_id,
'{"temperature": 0.7, "max_tokens": 15d0}' AS model_parameters,
'{"temperature": 0.7, "max_tokens": 150}' AS model_parameters,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS provided_usage_details,
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))) AS usage_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS provided_cost_details,
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)) AS cost_details,
toDecimal64(randUniform(0, 2000), 12) AS total_cost,
start_time AS completion_start_time,
toString(rand()) AS prompt_id,
toString(rand()) AS prompt_name,
1000 AS prompt_version,
array(${SEED_PROMPTS.map((p) => `concat('${p.id}',project_id)`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_id,
array(${SEED_PROMPTS.map((p) => `'${p.name}'`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_name,
array(${SEED_PROMPTS.map((p) => `'${p.version}'`).join(
",",
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_version,
start_time AS created_at,
start_time AS updated_at,
start_time AS event_ts,
@@ -119,6 +127,8 @@ export const prepareClickhouse = async (
FROM numbers(${observationsPerProject});
`;
console.log(observationsQuery);
const scoresQuery = `
INSERT INTO scores
SELECT toString(floor(randUniform(0, 100))) AS id,
@@ -130,7 +140,7 @@ export const prepareClickhouse = async (
toString(floor(randUniform(0, ${observationsPerProject}))),
NULL
) AS observation_id,
concat('name_', toString(rand() % 100)) AS name,
concat('name_', toString(rand() % 10)) AS name,
randUniform(0, 100) as value,
'API' as source,
'comment' as comment,
@@ -150,13 +160,41 @@ export const prepareClickhouse = async (
for (const query of queries) {
logger.info(`Executing query: ${query}`);
await clickhouseClient.command({
await clickhouseClient().command({
query,
clickhouse_settings: {
wait_end_of_query: 1,
},
});
}
// we also need to upsert trace sessions in postgres
const sessionQuery = `
SELECT session_id, project_id
FROM traces
WHERE session_id IS NOT NULL;
`;
const sessionResult = await clickhouseClient().query({
query: sessionQuery,
format: "JSONEachRow",
});
const sessionData = await sessionResult.json<{
session_id: string;
project_id: string;
}>();
const idProjectIdCombinations = sessionData.map((session) => ({
id: session.session_id,
projectId: session.project_id,
public: Math.random() < 0.1,
bookmarked: Math.random() < 0.1,
}));
await prisma.traceSession.createMany({
data: idProjectIdCombinations,
skipDuplicates: true,
});
}
const tables = ["traces", "scores", "observations"];
@@ -174,7 +212,7 @@ export const prepareClickhouse = async (
ORDER BY count() desc
`;
const result = await clickhouseClient.query({
const result = await clickhouseClient().query({
query,
format: "TabSeparated",
});
@@ -205,7 +243,7 @@ export const prepareClickhouse = async (
ORDER BY event_date desc
`;
const result = await clickhouseClient.query({
const result = await clickhouseClient().query({
query,
format: "TabSeparated",
});
+2 -2
View File
@@ -76,7 +76,7 @@ export class KyselySingleton {
createQueryCompiler: () => new PostgresQueryCompiler(),
},
}),
})
}),
);
return KyselySingleton.instance;
@@ -103,7 +103,7 @@ if (process.env.NODE_ENV === "development") {
createQueryCompiler: () => new PostgresQueryCompiler(),
},
}),
})
}),
);
}
+22
View File
@@ -30,6 +30,16 @@ const EnvSchema = z.object({
CLICKHOUSE_URL: z.string().url().optional(),
CLICKHOUSE_USER: z.string().optional(),
CLICKHOUSE_PASSWORD: z.string().optional(),
LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_ASYNC_INGESTION_PROCESSING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_INGESTION_QUEUE_DELAY_MS: z.coerce
.number()
.nonnegative()
.default(15_000),
SALT: z.string().optional(), // used by components imported by web package
LANGFUSE_LOG_LEVEL: z
.enum(["trace", "debug", "info", "warn", "error", "fatal"])
@@ -38,6 +48,18 @@ const EnvSchema = z.object({
ENABLE_AWS_CLOUDWATCH_METRIC_PUBLISHING: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_S3_EVENT_UPLOAD_ENABLED: z.enum(["true", "false"]).default("false"),
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_PREFIX: z.string().default(""),
LANGFUSE_S3_EVENT_UPLOAD_REGION: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY: z.string().optional(),
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE: z
.enum(["true", "false"])
.default("false"),
LANGFUSE_USE_AZURE_BLOB: z.enum(["true", "false"]).default("false"),
STRIPE_SECRET_KEY: z.string().optional(),
});
export const env = EnvSchema.parse(removeEmptyEnvVariables(process.env));
+26 -2
View File
@@ -5,6 +5,7 @@ export const langfuseObjects = [
"span",
"generation",
"event",
"dataset_item",
] as const;
// variable mapping stored in the db for eval templates
@@ -21,7 +22,7 @@ export const variableMapping = z
(value) => value.langfuseObject === "trace" || value.objectName !== null,
{
message: "objectName is required for langfuseObjects other than trace",
}
},
);
export const variableMappingList = z.array(variableMapping);
@@ -44,7 +45,7 @@ const observationCols = [
{ name: "Output", id: "output", internal: 'o."output"' },
];
export const availableEvalVariables = [
export const availableTraceEvalVariables = [
{
id: "trace",
display: "Trace",
@@ -76,6 +77,28 @@ export const availableEvalVariables = [
},
];
export const availableDatasetEvalVariables = [
{
id: "dataset_item",
display: "Dataset item",
availableColumns: [
{
name: "Metadata",
id: "metadata",
type: "stringObject",
internal: 'd."metadata"',
},
{ name: "Input", id: "input", internal: 'd."input"' },
{
name: "Expected output",
id: "expected_output",
internal: 'd."expected_output"',
},
],
},
...availableTraceEvalVariables,
];
export const OutputSchema = z.object({
reasoning: z.string(),
score: z.string(),
@@ -83,6 +106,7 @@ export const OutputSchema = z.object({
export enum EvalTargetObject {
Trace = "trace",
Dataset = "dataset",
}
export const DEFAULT_TRACE_JOB_DELAY = 10_000;
@@ -0,0 +1,3 @@
export * from "./scoreTypes";
export * from "./types";
export * from "./scoreConfigTypes";
@@ -0,0 +1,37 @@
import { ScoreDataType, ScoreSource } from "@prisma/client";
export type CategoricalAggregate = {
type: "CATEGORICAL";
values: string[];
valueCounts: { value: string; count: number }[];
comment?: string | null;
};
export type NumericAggregate = {
type: "NUMERIC";
values: number[];
average: number;
comment?: string | null;
};
export type ScoreAggregate = Record<
string,
CategoricalAggregate | NumericAggregate
>;
export type ScoreSimplified = {
name: string;
dataType: ScoreDataType;
source: ScoreSource;
value?: number | null;
comment?: string | null;
stringValue?: string | null;
};
export type LastUserScore = ScoreSimplified & {
timestamp: string;
traceId: string;
observationId?: string | null;
userId: string;
};
+2 -2
View File
@@ -11,6 +11,7 @@ export * from "./server/auth/apiKeys";
export * from "./observationsTable";
export * from "./utils/zod";
export * from "./utils/json";
export * from "./utils/stringChecks";
export * from "./utils/objects";
export * from "./utils/typeChecks";
export * from "./features/entitlements/plans";
@@ -28,8 +29,7 @@ export * from "./features/batchExport/types";
export * from "./features/annotation/types";
// scores
export * from "./features/scores/scoreConfigTypes";
export * from "./features/scores/scoreTypes";
export * from "./features/scores";
// comments
export * from "./features/comments/types";
@@ -15,6 +15,7 @@ export const filterOperators = {
],
numberObject: ["=", ">", "<", ">=", "<="],
boolean: ["=", "<>"],
null: ["is null", "is not null"],
} as const;
export const timeFilter = z.object({
@@ -68,6 +69,12 @@ export const booleanFilter = z.object({
operator: z.enum(filterOperators.boolean),
value: z.boolean(),
});
export const nullFilter = z.object({
type: z.literal("null"),
column: z.string(),
operator: z.enum(filterOperators.null),
value: z.literal(""),
});
export const singleFilter = z.discriminatedUnion("type", [
timeFilter,
stringFilter,
@@ -77,4 +84,5 @@ export const singleFilter = z.discriminatedUnion("type", [
stringObjectFilter,
numberObjectFilter,
booleanFilter,
nullFilter,
]);
@@ -17,7 +17,6 @@ export function CustomSSOProvider<P extends CustomSSOUser>(
wellKnown: `${options.issuer}/.well-known/openid-configuration`,
authorization: { params: { scope: "openid email profile" } }, // overridden by options.authorization to be able to set custom scopes, deep merged with this default
checks: ["pkce", "state"],
idToken: true,
profile(profile) {
return {
id: profile.sub,
@@ -0,0 +1,65 @@
import type { OAuthConfig, OAuthUserConfig } from "next-auth/providers/oauth";
import type { GithubProfile, GithubEmail } from "next-auth/providers/github";
export function GitHubEnterpriseProvider<P extends GithubProfile>(
options: OAuthUserConfig<P> & {
enterprise?: {
baseUrl?: string;
};
}
): OAuthConfig<P> {
const baseUrl = options?.enterprise?.baseUrl ?? "https://github.com"
const apiBaseUrl = options?.enterprise?.baseUrl
? `${options?.enterprise?.baseUrl}/api/v3`
: "https://api.github.com"
return {
id: "github-enterprise",
name: "GitHub Enterprise",
type: "oauth",
authorization: {
url: `${baseUrl}/login/oauth/authorize`,
params: { scope: "read:user user:email" },
},
token: `${baseUrl}/login/oauth/access_token`,
userinfo: {
url: `${apiBaseUrl}/user`,
async request({ client, tokens }) {
const profile = await client.userinfo(tokens.access_token!)
if (!profile.email) {
// If the user does not have a public email, get another via the GitHub API
// See https://docs.github.com/en/rest/users/emails#list-email-addresses-for-the-authenticated-user
const res = await fetch(`${apiBaseUrl}/user/emails`, {
headers: { Authorization: `token ${tokens.access_token}` },
})
if (res.ok) {
const emails: GithubEmail[] = await res.json()
profile.email = (emails.find((e) => e.primary) ?? emails[0]).email
}
}
return profile
},
},
profile(profile) {
return {
id: profile.id.toString(),
name: profile.name ?? profile.login,
email: profile.email,
image: profile.avatar_url,
}
},
style: {
logo: "https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github.svg",
logoDark:
"https://raw.githubusercontent.com/nextauthjs/next-auth/main/packages/next-auth/provider-logos/github-dark.svg",
bg: "#fff",
bgDark: "#000",
text: "#000",
textDark: "#fff",
},
options,
}
}
+23 -10
View File
@@ -1,15 +1,28 @@
import { createClient } from "@clickhouse/client";
import { env } from "../../env";
import { NodeClickHouseClientConfigOptions } from "@clickhouse/client/dist/config";
export type ClickhouseClientType = ReturnType<typeof createClient>;
export const clickhouseClient = createClient({
url: env.CLICKHOUSE_URL,
username: env.CLICKHOUSE_USER,
password: env.CLICKHOUSE_PASSWORD,
database: "default",
clickhouse_settings: {
async_insert: 1,
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
},
});
export const clickhouseClient = (opts?: NodeClickHouseClientConfigOptions) =>
createClient({
...opts,
url: env.CLICKHOUSE_URL,
username: env.CLICKHOUSE_USER,
password: env.CLICKHOUSE_PASSWORD,
database: "default",
clickhouse_settings: {
async_insert: 1,
wait_for_async_insert: 1, // if disabled, we won't get errors from clickhouse
},
});
export const defaultClickhouseClient = clickhouseClient();
/**
* Accepts a JavaScript date and returns the DateTime in format YYYY-MM-DD HH:MM:SS
*/
export const convertDateToClickhouseDateTime = (date: Date): string => {
// 2024-11-06T20:37:00.123Z -> 2024-11-06 21:37:00.123
return date.toISOString().replace("T", " ").replace("Z", "");
};
+3 -1
View File
@@ -106,10 +106,12 @@ export function tableColumnsToSqlFilter(
", ",
)}] `;
break;
case "boolean":
valuePrisma = Prisma.sql`${filter.value}`;
break;
case "null":
valuePrisma = Prisma.sql``;
break;
}
const jsonKeyPrisma =
filter.type === "stringObject" || filter.type === "numberObject"
+10 -2
View File
@@ -1,27 +1,34 @@
export * from "./services/S3StorageService";
export * from "./services/StorageService";
export * from "./services/email/organizationInvitation/sendMembershipInvitationEmail";
export * from "./services/email/batchExportSuccess/sendBatchExportSuccessEmail";
export * from "./services/email/passwordReset/sendResetPasswordVerificationRequest";
export * from "./services/PromptService";
export * from "./services/traces-ui-table-service";
export * from "./auth/apiKeys";
export * from "./auth/customSsoProvider";
export * from "./auth/gitHubEnterpriseProvider";
export * from "./llm/fetchLLMCompletion";
export * from "./llm/types";
export * from "./utils/DatabaseReadStream";
export * from "./utils/transforms";
export * from "./clickhouse/client";
export * from "./clickhouse/schema-utils";
export * from "./clickhouse/schemaUtils";
export * from "./clickhouse/schema";
export * from "./repositories/definitions";
export * from "../server/ingestion/types";
export * from "./ingestion/modelMatch";
export * from "./ingestion/processEventBatch";
export * from "../server/ingestion/types";
export * from "../server/ingestion/validateAndInflateScore";
export * from "./redis/redis";
export * from "./redis/traceUpsert";
export * from "./redis/CloudUsageMeteringQueue";
export * from "./redis/getQueue";
export * from "./redis/datasetRunItemUpsert";
export * from "./redis/batchExport";
export * from "./redis/legacyIngestion";
export * from "./redis/ingestionQueue";
export * from "./redis/experimentCreateQueue";
export * from "./auth/types";
export * from "./ingestion/legacy/index";
export * from "./queues";
@@ -32,3 +39,4 @@ export * from "./instrumentation";
export * from "./logger";
export * from "./queries";
export * from "./repositories";
export * from "./redis/evalExecutionQueue";
@@ -20,6 +20,9 @@ import { jsonSchema } from "../../../utils/zod";
import { prisma } from "../../../db";
import { LegacyIngestionAccessScope } from ".";
import { logger } from "../../logger";
import { env } from "../../../env";
import { upsertTrace } from "../../repositories";
import { convertDateToClickhouseDateTime } from "../../clickhouse/client";
export interface EventProcessor {
auth(apiScope: LegacyIngestionAccessScope): void;
@@ -131,19 +134,6 @@ export class ObservationProcessor implements EventProcessor {
})
: undefined;
const traceId =
!this.event.body.traceId && !existingObservation
? // Create trace if no traceid
(
await prisma.trace.create({
data: {
projectId: apiScope.projectId,
name: this.event.body.name,
},
})
).id
: this.event.body.traceId;
// Token counts
const [newInputCount, newOutputCount] =
"usage" in this.event.body
@@ -230,11 +220,46 @@ export class ObservationProcessor implements EventProcessor {
return newId;
})();
let traceId = this.event.body?.traceId;
if (!this.event.body.traceId && !existingObservation) {
// Create trace if no traceId
traceId = observationId;
// Insert trace into postgres
await prisma.trace.upsert({
where: {
id: observationId,
},
create: {
projectId: apiScope.projectId,
name: this.event.body.name,
id: observationId,
timestamp: this.event.body.startTime || new Date(),
},
update: {},
});
if (env.CLICKHOUSE_URL) {
// Insert trace into clickhouse if enabled
await upsertTrace({
id: observationId,
project_id: apiScope.projectId,
timestamp: convertDateToClickhouseDateTime(
this.event.body.startTime
? new Date(this.event.body.startTime)
: new Date(),
),
created_at: convertDateToClickhouseDateTime(new Date()),
updated_at: convertDateToClickhouseDateTime(new Date()),
});
}
}
return {
id: observationId,
create: {
id: observationId,
traceId: traceId,
traceId,
type: type,
name: this.event.body.name,
startTime: this.event.body.startTime
@@ -367,7 +392,7 @@ export class ObservationProcessor implements EventProcessor {
text: body.input,
});
} else {
logger.info(
logger.debug(
`No input provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationInput = await prisma.observation.findFirst({
@@ -393,7 +418,7 @@ export class ObservationProcessor implements EventProcessor {
text: body.output,
});
} else {
logger.info(
logger.debug(
`No output provided, trying to calculate for id: ${existingObservation?.id}`,
);
const observationOutput = await prisma.observation.findFirst({
@@ -600,19 +625,32 @@ export class TraceProcessor implements EventProcessor {
: undefined;
if (body.sessionId) {
await prisma.traceSession.upsert({
where: {
id_projectId: {
try {
await prisma.traceSession.upsert({
where: {
id_projectId: {
id: body.sessionId,
projectId: apiScope.projectId,
},
},
create: {
id: body.sessionId,
projectId: apiScope.projectId,
},
},
create: {
id: body.sessionId,
projectId: apiScope.projectId,
},
update: {},
});
update: {},
});
} catch (e) {
if (
e instanceof Prisma.PrismaClientKnownRequestError &&
e.code === "P2002"
) {
logger.warn(
`Failed to upsert session. Session ${body.sessionId} in project ${apiScope.projectId} already exists`,
);
} else {
throw e;
}
}
}
// Do not use nested upserts or multiple where conditions as this should be a single native database upsert
@@ -3,13 +3,7 @@ import z from "zod";
import { ForbiddenError, UnauthorizedError } from "../../../errors";
import { eventTypes, ingestionApiSchema, IngestionEventType } from "../types";
import { getProcessorForEvent } from "./EventProcessor";
import { TraceUpsertEventType } from "../../queues";
import {
convertTraceUpsertEventsToRedisEvents,
TraceUpsertQueue,
} from "../../redis/traceUpsert";
import { ApiAccessScope } from "../../auth/types";
import { redis } from "../../redis/redis";
import { backOff } from "exponential-backoff";
import { Model } from "../../..";
import { logger } from "../../logger";
@@ -77,7 +71,7 @@ export const handleBatch = async (
// Decide how to handle the error: rethrow, continue, or push an error object to results
// For example, push an error object:
errors.push({
error: error,
error,
id: singleEvent.id,
type: singleEvent.type,
});
@@ -122,7 +116,7 @@ const handleSingleEvent = async (
restEvent = rest;
}
logger.info(
logger.debug(
`handling single event ${event.id} of type ${event.type}: ${JSON.stringify({ body: restEvent })}`,
);
@@ -163,42 +157,5 @@ export function cleanEvent(obj: unknown): unknown {
}
}
export const isNotNullOrUndefined = <T>(
val?: T | null,
): val is Exclude<T, null | undefined> => !isUndefinedOrNull(val);
export const isUndefinedOrNull = <T>(val?: T | null): val is undefined | null =>
val === undefined || val === null;
export const addTracesToTraceUpsertQueue = async (
batchResults: BatchResult[],
projectId: string,
): Promise<void> => {
const traceEvents: TraceUpsertEventType[] = batchResults
.filter((result) => result.type === eventTypes.TRACE_CREATE) // we only have create, no update.
.map((result) =>
result.result &&
typeof result.result === "object" &&
"id" in result.result
? // ingestion API only gets traces for one projectId
{ traceId: result.result.id as string, projectId }
: null,
)
.filter(isNotNullOrUndefined);
try {
if (env.NEXT_PUBLIC_LANGFUSE_CLOUD_REGION && redis) {
logger.info(`Sending ${traceEvents.length} events to worker via Redis`);
const queue = TraceUpsertQueue.getInstance();
if (!queue) {
logger.error("TraceUpsertQueue not initialized");
return;
}
await queue.addBulk(convertTraceUpsertEventsToRedisEvents(traceEvents));
}
} catch (error) {
logger.error("Error sending events to worker", error);
}
};
@@ -0,0 +1,396 @@
import { randomUUID } from "crypto";
import { z } from "zod";
import { type Model } from "../../db";
import { env } from "../../env";
import {
InvalidRequestError,
LangfuseNotFoundError,
UnauthorizedError,
} from "../../errors";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
import {
getCurrentSpan,
instrumentAsync,
instrumentSync,
recordIncrement,
traceException,
} from "../instrumentation";
import { logger } from "../logger";
import { LegacyIngestionEventType, QueueJobs } from "../queues";
import { IngestionQueue } from "../redis/ingestionQueue";
import { LegacyIngestionQueue } from "../redis/legacyIngestion";
import { redis } from "../redis/redis";
import { handleBatch } from "./legacy";
import {
StorageService,
StorageServiceFactory,
} from "../services/StorageService";
import { getProcessorForEvent } from "./legacy/EventProcessor";
import { eventTypes, ingestionEvent, IngestionEventType } from "./types";
export type TokenCountDelegate = (p: {
model: Model;
text: unknown;
}) => number | undefined;
let s3StorageServiceClient: StorageService;
const getS3StorageServiceClient = (bucketName: string): StorageService => {
if (!s3StorageServiceClient) {
s3StorageServiceClient = StorageServiceFactory.getInstance({
bucketName,
accessKeyId: env.LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID,
secretAccessKey: env.LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY,
endpoint: env.LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT,
region: env.LANGFUSE_S3_EVENT_UPLOAD_REGION,
forcePathStyle: env.LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE === "true",
});
}
return s3StorageServiceClient;
};
export const processEventBatch = async (
input: unknown[],
authCheck: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): Promise<{
successes: { id: string; status: number }[];
errors: {
id: string;
status: number;
message?: string;
error?: string;
}[];
}> => {
// add context of api call to the span
const currentSpan = getCurrentSpan();
recordIncrement("langfuse.ingestion.event", input.length);
currentSpan?.setAttribute("event_count", input.length);
/**************
* VALIDATION *
**************/
const validationErrors: { id: string; error: unknown }[] = [];
const authenticationErrors: { id: string; error: unknown }[] = [];
const batch: z.infer<typeof ingestionEvent>[] = input
.flatMap((event) => {
const parsed = instrumentSync(
{ name: "ingestion-zod-parse-individual-event" },
(span) => {
const parsedBody = ingestionEvent.safeParse(event);
if (parsedBody.data?.id !== undefined) {
span.setAttribute("object.id", parsedBody.data.id);
}
return parsedBody;
},
);
if (!parsed.success) {
validationErrors.push({
id:
typeof event === "object" && event && "id" in event
? typeof event.id === "string"
? event.id
: "unknown"
: "unknown",
error: new InvalidRequestError(parsed.error.message),
});
return [];
}
if (!isAuthorized(parsed.data, authCheck, tokenCountDelegate)) {
authenticationErrors.push({
id: parsed.data.id,
error: new UnauthorizedError("Access Scope Denied"),
});
return [];
}
return [parsed.data];
})
.flatMap((event) => {
if (event.type === eventTypes.SDK_LOG) {
// Log SDK_LOG events, but remove them from further processing
logger.info("SDK Log Event", { event });
return [];
}
return [event];
});
const sortedBatch = sortBatch(batch);
// We group events by eventBodyId which allows us to store and process them
// as one which reduces infra interactions per event. Only used in the S3 case.
const sortedBatchByEventBodyId = sortedBatch.reduce(
(
acc: Record<
string,
{
data: IngestionEventType[];
key: string;
eventBodyId: string;
type: (typeof eventTypes)[keyof typeof eventTypes];
}
>,
event,
) => {
if (!event.body?.id) {
return acc;
}
const key = `${getClickhouseEntityType(event.type)}-${event.body.id}`;
if (!acc[key]) {
acc[key] = {
data: [],
key: event.id,
type: event.type,
eventBodyId: event.body.id,
};
}
acc[key].data.push(event);
return acc;
},
{},
);
/********************
* ASYNC PROCESSING *
********************/
let s3UploadErrored = false;
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
await instrumentAsync({ name: "s3-upload-events" }, async () => {
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
throw new Error("S3 event store is enabled but no bucket is set");
}
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
// S3 Event Upload is currently blocking, but non-failing.
// If a promise rejects, we log it below, but do not throw an error.
// In this case, we upload the full batch into the Redis queue.
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) => {
// We upload the event in an array to the S3 bucket grouped by the eventBodyId.
// That way we batch updates from the same invocation into a single file and reduce
// write operations on S3.
const { data, key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${authCheck.scope.projectId}/${getClickhouseEntityType(type)}/${eventBodyId}/${key}.json`,
data,
);
}),
);
results.forEach((result) => {
if (result.status === "rejected") {
s3UploadErrored = true;
logger.error("Failed to upload event to S3", {
error: result.reason,
});
}
});
});
}
// Send each event individually to IngestionQueue for new processing
if (
env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" &&
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" &&
env.LANGFUSE_ASYNC_CLICKHOUSE_INGESTION_PROCESSING === "true" &&
redis &&
!s3UploadErrored
) {
const queue = IngestionQueue.getInstance();
const results = await Promise.allSettled(
Object.keys(sortedBatchByEventBodyId).map(async (id) =>
queue
? queue.add(
QueueJobs.IngestionJob,
{
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.IngestionJob as const,
payload: {
data: {
type: sortedBatchByEventBodyId[id].type,
eventBodyId: sortedBatchByEventBodyId[id].eventBodyId,
},
authCheck,
},
},
{
delay: env.LANGFUSE_INGESTION_QUEUE_DELAY_MS,
},
)
: Promise.reject("Failed to instantiate queue"),
),
);
results.forEach((result) => {
if (result.status === "rejected") {
logger.error("Failed to add event to IngestionQueue", {
error: result.reason,
});
}
});
}
// As part of the legacy processing we sent the entire batch to the worker.
if (env.LANGFUSE_ASYNC_INGESTION_PROCESSING === "true" && redis) {
const queue = LegacyIngestionQueue.getInstance();
if (queue) {
let addToQueueFailed = false;
const queuePayload: LegacyIngestionEventType =
env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true" && !s3UploadErrored
? {
data: Object.keys(sortedBatchByEventBodyId).map((id) => {
const { key, type, eventBodyId } = sortedBatchByEventBodyId[id];
return {
type,
eventBodyId,
eventId: key,
};
}),
authCheck,
useS3EventStore: true,
}
: { data: sortedBatch, authCheck, useS3EventStore: false };
try {
await queue.add(QueueJobs.LegacyIngestionJob, {
payload: queuePayload,
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.LegacyIngestionJob as const,
});
} catch (e: unknown) {
logger.warn(
"Failed to add batch to queue, falling back to sync processing",
e,
);
addToQueueFailed = true;
}
if (!addToQueueFailed) {
return aggregateBatchResult(
// we are not sending additional server errors to the client in case of early return
[...validationErrors, ...authenticationErrors],
sortedBatch.map((event) => ({ id: event.id, result: event })),
);
}
} else {
logger.error(
"Ingestion queue not initialized, falling back to sync processing",
);
}
}
/*******************
* SYNC PROCESSING *
*******************/
const result = await handleBatch(sortedBatch, authCheck, tokenCountDelegate);
// in case we did not return early, we return the result here
return aggregateBatchResult(
[...validationErrors, ...authenticationErrors, ...result.errors],
result.results,
);
};
const isAuthorized = (
event: IngestionEventType,
authScope: AuthHeaderValidVerificationResult,
tokenCountDelegate: TokenCountDelegate,
): boolean => {
try {
getProcessorForEvent(event, tokenCountDelegate).auth(authScope.scope);
return true;
} catch (error) {
return false;
}
};
/**
* Sorts a batch of ingestion events. Orders by: updating events last, sorted by timestamp asc.
*/
const sortBatch = (batch: Array<z.infer<typeof ingestionEvent>>) => {
const updateEvents: (typeof eventTypes)[keyof typeof eventTypes][] = [
eventTypes.GENERATION_UPDATE,
eventTypes.SPAN_UPDATE,
eventTypes.OBSERVATION_UPDATE, // legacy event type
];
const updates = batch
.filter((event) => updateEvents.includes(event.type))
.sort((a, b) => {
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
});
const others = batch
.filter((event) => !updateEvents.includes(event.type))
.sort((a, b) => {
return new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime();
});
// Return the array with non-update events first, followed by update events
return [...others, ...updates];
};
export const aggregateBatchResult = (
errors: Array<{ id: string; error: unknown }>,
results: Array<{ id: string; result: unknown }>,
) => {
const returnedErrors: {
id: string;
status: number;
message?: string;
error?: string;
}[] = [];
const successes: {
id: string;
status: number;
}[] = [];
errors.forEach((error) => {
if (error.error instanceof InvalidRequestError) {
returnedErrors.push({
id: error.id,
status: 400,
message: "Invalid request data",
error: error.error.message,
});
} else if (error.error instanceof UnauthorizedError) {
returnedErrors.push({
id: error.id,
status: 401,
message: "Authentication error",
error: error.error.message,
});
} else if (error.error instanceof LangfuseNotFoundError) {
returnedErrors.push({
id: error.id,
status: 404,
message: "Resource not found",
error: error.error.message,
});
} else {
returnedErrors.push({
id: error.id,
status: 500,
error: "Internal Server Error",
});
}
});
if (returnedErrors.length > 0) {
traceException(errors);
logger.error("Error processing events", returnedErrors);
}
results.forEach((result) => {
successes.push({
id: result.id,
status: 201,
});
});
return { successes, errors: returnedErrors };
};
@@ -3,7 +3,7 @@ import { z } from "zod";
import { NonEmptyString, jsonSchema } from "../../utils/zod";
import { ModelUsageUnit } from "../../constants";
import { ObservationLevel } from "@prisma/client";
import { ObservationLevel, ScoreSource } from "@prisma/client";
export const Usage = z.object({
input: z.number().int().nullish(),
@@ -163,6 +163,7 @@ const BaseScoreBody = z.object({
traceId: z.string(),
observationId: z.string().nullish(),
comment: z.string().nullish(),
source: z.nativeEnum(ScoreSource).default(ScoreSource.API),
});
/**
@@ -76,7 +76,7 @@ function inflateScoreBody(
const { body, projectId, scoreId, config } = params;
const relevantDataType = config?.dataType ?? body.dataType;
const scoreProps = { ...body, id: scoreId, projectId, source: "API" };
const scoreProps = { source: "API", ...body, id: scoreId, projectId };
if (typeof body.value === "number") {
if (relevantDataType && relevantDataType === ScoreDataType.BOOLEAN) {
@@ -1,10 +1,10 @@
import * as opentelemetry from "@opentelemetry/api";
import * as dd from "dd-trace";
import { env } from "../../env";
import {
CloudWatchClient,
PutMetricDataCommand,
} from "@aws-sdk/client-cloudwatch";
import * as opentelemetry from "@opentelemetry/api";
import * as dd from "dd-trace";
import { env } from "../../env";
import { logger } from "../logger";
// type CallbackFn<T> = () => T;
@@ -38,7 +38,7 @@ export async function instrumentAsync<T>(
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
ctx.name,
{
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
root: !ctx.traceContext && ctx.rootSpan,
kind: ctx.spanKind,
},
activeContext,
@@ -72,7 +72,7 @@ export function instrumentSync<T>(
return getTracer(ctx.traceScope ?? callback.name).startActiveSpan(
ctx.name,
{
root: !Boolean(ctx.traceContext) && ctx.rootSpan,
root: !ctx.traceContext && ctx.rootSpan,
kind: ctx.spanKind,
},
activeContext,
@@ -1,5 +1,7 @@
import type { ZodSchema } from "zod";
import { CallbackHandler } from "langfuse-langchain";
import { ChatAnthropic } from "@langchain/anthropic";
import { ChatBedrockConverse } from "@langchain/aws";
import {
@@ -17,11 +19,27 @@ import {
BedrockConfigSchema,
BedrockCredentialSchema,
} from "../../interfaces/customLLMProviderConfigSchemas";
import { AuthHeaderValidVerificationResult } from "../auth/types";
import {
processEventBatch,
type TokenCountDelegate,
} from "../ingestion/processEventBatch";
import { logger } from "../logger";
import { ChatMessage, ChatMessageRole, LLMAdapter, ModelParams } from "./types";
import type { BaseCallbackHandler } from "@langchain/core/callbacks/base";
type ProcessTracedEvents = () => Promise<void>;
export type TraceParams = {
traceName: string;
traceId: string;
projectId: string;
tags: string[];
tokenCountDelegate: TokenCountDelegate;
authCheck: AuthHeaderValidVerificationResult;
};
type LLMCompletionParams = {
messages: ChatMessage[];
modelParams: ModelParams;
@@ -31,6 +49,7 @@ type LLMCompletionParams = {
apiKey: string;
maxRetries?: number;
config?: Record<string, string> | null;
traceParams?: TraceParams;
};
type FetchLLMCompletionParams = LLMCompletionParams & {
@@ -40,25 +59,34 @@ type FetchLLMCompletionParams = LLMCompletionParams & {
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: true;
}
): Promise<IterableReadableStream<Uint8Array>>;
},
): Promise<{
completion: IterableReadableStream<Uint8Array>;
processTracedEvents: ProcessTracedEvents;
}>;
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: false;
}
): Promise<string>;
},
): Promise<{ completion: string; processTracedEvents: ProcessTracedEvents }>;
export async function fetchLLMCompletion(
params: LLMCompletionParams & {
streaming: false;
structuredOutputSchema: ZodSchema;
}
): Promise<unknown>;
},
): Promise<{
completion: unknown;
processTracedEvents: ProcessTracedEvents;
}>;
export async function fetchLLMCompletion(
params: FetchLLMCompletionParams
): Promise<string | IterableReadableStream<Uint8Array> | unknown> {
params: FetchLLMCompletionParams,
): Promise<{
completion: string | IterableReadableStream<Uint8Array> | unknown;
processTracedEvents: ProcessTracedEvents;
}> {
// the apiKey must never be printed to the console
const {
messages,
@@ -69,8 +97,39 @@ export async function fetchLLMCompletion(
baseURL,
maxRetries,
config,
traceParams,
} = params;
let finalCallbacks: BaseCallbackHandler[] | undefined = callbacks ?? [];
let processTracedEvents: ProcessTracedEvents = () => Promise.resolve();
if (traceParams) {
const handler = new CallbackHandler({
_projectId: traceParams.projectId,
_isLocalEventExportEnabled: true,
tags: traceParams.tags,
});
finalCallbacks.push(handler);
processTracedEvents = async () => {
try {
const events = await handler.langfuse._exportLocalEvents(
traceParams.projectId,
);
await processEventBatch(
JSON.parse(JSON.stringify(events)), // stringify to emulate network event batch from network call
traceParams.authCheck,
traceParams.tokenCountDelegate,
);
} catch (e) {
logger.error("Failed to process traced events", { error: e });
}
};
}
finalCallbacks = finalCallbacks.length > 0 ? finalCallbacks : undefined;
const finalMessages = messages.map((message) => {
if (message.role === ChatMessageRole.User)
return new HumanMessage(message.content);
@@ -89,7 +148,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
clientOptions: { maxRetries },
});
} else if (modelParams.adapter === LLMAdapter.OpenAI) {
@@ -99,7 +158,8 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
streamUsage: false, // https://github.com/langchain-ai/langchainjs/issues/6533
callbacks: finalCallbacks,
maxRetries,
configuration: {
baseURL,
@@ -114,7 +174,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
maxRetries,
});
} else if (modelParams.adapter === LLMAdapter.Bedrock) {
@@ -128,7 +188,7 @@ export async function fetchLLMCompletion(
temperature: modelParams.temperature,
maxTokens: modelParams.max_tokens,
topP: modelParams.top_p,
callbacks,
callbacks: finalCallbacks,
maxRetries,
});
} else {
@@ -137,10 +197,19 @@ export async function fetchLLMCompletion(
throw new Error("This model provider is not supported.");
}
const runConfig = {
callbacks: finalCallbacks,
runId: traceParams?.traceId,
runName: traceParams?.traceName,
};
if (params.structuredOutputSchema) {
return await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages);
return {
completion: await (chatModel as ChatOpenAI) // Typecast necessary due to https://github.com/langchain-ai/langchainjs/issues/6795
.withStructuredOutput(params.structuredOutputSchema)
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
/*
@@ -156,27 +225,41 @@ export async function fetchLLMCompletion(
Reference: https://platform.openai.com/docs/guides/reasoning/beta-limitations
*/
if (modelParams.model.startsWith("o1-")) {
return await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
})
.pipe(new StringOutputParser())
.invoke(
finalMessages.filter((message) => message._getType() !== "system")
);
return {
completion: await new ChatOpenAI({
openAIApiKey: apiKey,
modelName: modelParams.model,
temperature: 1,
maxTokens: undefined,
topP: undefined,
callbacks,
maxRetries,
configuration: {
baseURL,
},
})
.pipe(new StringOutputParser())
.invoke(
finalMessages.filter((message) => message._getType() !== "system"),
runConfig,
),
processTracedEvents,
};
}
if (streaming) {
return chatModel.pipe(new BytesOutputParser()).stream(finalMessages);
return {
completion: await chatModel
.pipe(new BytesOutputParser())
.stream(finalMessages, runConfig),
processTracedEvents,
};
}
return await chatModel.pipe(new StringOutputParser()).invoke(finalMessages);
return {
completion: await chatModel
.pipe(new StringOutputParser())
.invoke(finalMessages, runConfig),
processTracedEvents,
};
}
+29 -2
View File
@@ -26,6 +26,20 @@ export enum ChatMessageRole {
export const ChatMessageDefaultRoleSchema = z.nativeEnum(ChatMessageRole);
const ChatMessageSchema = z.object({
role: z.union([ChatMessageDefaultRoleSchema, z.string()]), // Users may ingest any string as role via API/SDK
content: z.string(),
});
export const ChatMessageListSchema = z.array(ChatMessageSchema);
export const TextPromptSchema = z.string().min(1, "Enter a prompt");
export const PromptContentSchema = z.union([
ChatMessageListSchema,
TextPromptSchema,
]);
export type PromptContent = z.infer<typeof PromptContentSchema>;
export type ModelParams = {
provider: string;
adapter: LLMAdapter;
@@ -49,7 +63,18 @@ export const ZodModelConfig = z.object({
top_p: z.coerce.number().optional(),
});
// NOTE: Update docs page when changing this!
// Experiment config
export const ExperimentMetadataSchema = z
.object({
prompt_id: z.string(),
provider: z.string(),
model: z.string(),
model_params: ZodModelConfig,
})
.strict();
export type ExperimentMetadata = z.infer<typeof ExperimentMetadataSchema>;
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
export const openAIModels = [
"gpt-4o",
"gpt-4o-2024-08-06",
@@ -76,11 +101,13 @@ export const openAIModels = [
export type OpenAIModel = (typeof openAIModels)[number];
// NOTE: Update docs page when changing this!
// NOTE: Update docs page when changing this! https://langfuse.com/docs/playground#openai-playground--anthropic-playground
export const anthropicModels = [
"claude-3-5-sonnet-20241022",
"claude-3-5-sonnet-20240620",
"claude-3-opus-20240229",
"claude-3-sonnet-20240229",
"claude-3-5-haiku-20241022",
"claude-3-haiku-20240307",
"claude-2.1",
"claude-2.0",
@@ -1,19 +1,14 @@
import { filterOperators } from "../../../interfaces/filters";
import { clickhouseCompliantRandomCharacters } from "../../repositories";
function randomCharacters() {
const chars = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
let result = "";
const randomArray = new Uint8Array(5);
crypto.getRandomValues(randomArray);
randomArray.forEach((number) => {
result += chars[number % chars.length];
});
return result;
}
export type ClickhouseOperator =
| (typeof filterOperators)[keyof typeof filterOperators][number]
| "!=";
export interface Filter {
apply(): ClickhouseFilter;
clickhouseTable: string;
operator: ClickhouseOperator;
field: string;
}
type ClickhouseFilter = {
query: string;
@@ -22,9 +17,9 @@ type ClickhouseFilter = {
export class StringFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected value: string;
protected operator: (typeof filterOperators)["string"][number];
public field: string;
public value: string;
public operator: (typeof filterOperators)["string"][number];
protected tablePrefix?: string;
constructor(opts: {
@@ -42,7 +37,7 @@ export class StringFilter implements Filter {
}
apply(): ClickhouseFilter {
const varName = `stringFilter${this.field}`;
const varName = `stringFilter${clickhouseCompliantRandomCharacters()}`;
const fieldWithPrefix = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
let query: string;
@@ -75,40 +70,44 @@ export class StringFilter implements Filter {
export class NumberFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected value: number;
protected operator: (typeof filterOperators)["number"][number];
public field: string;
public value: number;
public operator: (typeof filterOperators)["number"][number] | "!=";
public clickhouseTypeOverwrite?: string;
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["number"][number];
operator: (typeof filterOperators)["number"][number] | "!=";
value: number;
tablePrefix?: string;
clickhouseTypeOverwrite?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
this.clickhouseTypeOverwrite = opts.clickhouseTypeOverwrite;
}
apply(): ClickhouseFilter {
const uid = randomCharacters();
const uid = clickhouseCompliantRandomCharacters();
const varName = `numberFilter${uid}`;
const type = this.clickhouseTypeOverwrite ?? "Decimal64(12)";
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: Decimal}`,
params: { [varName]: this.value },
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: ${type}}`,
params: { [varName]: this.value.toString() },
};
}
}
export class DateTimeFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected value: Date;
protected operator: (typeof filterOperators)["datetime"][number];
public field: string;
public value: Date;
public operator: (typeof filterOperators)["datetime"][number];
protected tablePrefix?: string;
constructor(opts: {
@@ -126,7 +125,7 @@ export class DateTimeFilter implements Filter {
}
apply(): ClickhouseFilter {
const uid = randomCharacters();
const uid = clickhouseCompliantRandomCharacters();
const varName = `dateTimeFilter${uid}`;
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: DateTime64(3)}`,
@@ -137,9 +136,9 @@ export class DateTimeFilter implements Filter {
export class StringOptionsFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected values: string[];
protected operator: (typeof filterOperators.stringOptions)[number];
public field: string;
public values: string[];
public operator: (typeof filterOperators.stringOptions)[number];
protected tablePrefix?: string;
constructor(opts: {
@@ -157,7 +156,7 @@ export class StringOptionsFilter implements Filter {
}
apply(): ClickhouseFilter {
const uid = randomCharacters();
const uid = clickhouseCompliantRandomCharacters();
const varName = `stringOptionsFilter${uid}`;
return {
query:
@@ -169,12 +168,72 @@ export class StringOptionsFilter implements Filter {
}
}
// stringObject filter is used when we want to filter on a key value pair in a clickhouse map.
// As we use the MAP form clickhouse, we can only filter efficiently on the first level of a json obj.
export class StringObjectFilter implements Filter {
public clickhouseTable: string;
public field: string;
public key: string;
public value: string;
public operator: (typeof filterOperators)["stringObject"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["stringObject"][number];
key: string;
value: string;
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.value = opts.value;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
this.key = opts.key;
}
apply(): ClickhouseFilter {
const varKeyName = `stringObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
const varValueName = `stringObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
// const query: `${column}['{varKeyName: String}'] ${this.operator} {${varValueName}: String}`,
let query: string;
switch (this.operator) {
case "=":
query = `${column}[{${varKeyName}: String}] = {${varValueName}: String}`;
break;
case "contains":
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) > 0`;
break;
case "does not contain":
query = `position(${column}[{${varKeyName}: String}], {${varValueName}: String}) = 0`;
break;
case "starts with":
query = `startsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
break;
case "ends with":
query = `endsWith(${column}[{${varKeyName}: String}], {${varValueName}: String})`;
break;
default:
throw new Error(`Unsupported operator: ${this.operator}`);
}
return {
query,
params: { [varKeyName]: this.key, [varValueName]: this.value },
};
}
}
// this is used when we want to filter multiple values on a clickhouse column which is also an array
export class ArrayOptionsFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected values: string[];
protected operator: (typeof filterOperators.arrayOptions)[number];
public field: string;
public values: string[];
public operator: (typeof filterOperators.arrayOptions)[number];
protected tablePrefix?: string;
constructor(opts: {
@@ -192,7 +251,7 @@ export class ArrayOptionsFilter implements Filter {
}
apply(): ClickhouseFilter {
const uid = randomCharacters();
const uid = clickhouseCompliantRandomCharacters();
const varName = `arrayOptionsFilter${uid}`;
let query: string;
@@ -204,7 +263,7 @@ export class ArrayOptionsFilter implements Filter {
query = `hasAny({${varName}: Array(String)}, ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = False`;
break;
case "all of":
query = `arrayAll(x -> has({${varName}: Array(String)}, x), ${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}) = True`;
query = `hasAll(${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}, {${varName}: Array(String)}) = True`;
break;
default:
throw new Error(`Unsupported operator: ${this.operator}`);
@@ -217,18 +276,44 @@ export class ArrayOptionsFilter implements Filter {
}
}
export class NumberObjectFilter implements Filter {
export class NullFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected key: string;
protected value: number;
protected operator: (typeof filterOperators)["numberObject"][number];
public field: string;
public operator: (typeof filterOperators)["null"][number];
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["numberObject"][number];
operator: (typeof filterOperators)["null"][number];
tablePrefix?: string;
}) {
this.clickhouseTable = opts.clickhouseTable;
this.field = opts.field;
this.operator = opts.operator;
this.tablePrefix = opts.tablePrefix;
}
apply(): ClickhouseFilter {
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator}`,
params: {},
};
}
}
export class NumberObjectFilter implements Filter {
public clickhouseTable: string;
public field: string;
public key: string;
public value: number;
public operator: (typeof filterOperators)["numberObject"][number] | "!=";
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["numberObject"][number] | "!=";
key: string;
value: number;
tablePrefix?: string;
@@ -242,11 +327,11 @@ export class NumberObjectFilter implements Filter {
}
apply(): ClickhouseFilter {
const varKeyName = `numberObjectKeyFilter${randomCharacters()}`;
const varValueName = `numberObjectValueFilter${randomCharacters()}`;
const varKeyName = `numberObjectKeyFilter${clickhouseCompliantRandomCharacters()}`;
const varValueName = `numberObjectValueFilter${clickhouseCompliantRandomCharacters()}`;
const column = `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field}`;
return {
query: `empty(arrayFilter(x -> (((x.1) = {${varKeyName}: String}) AND ((x.2) ${this.operator} {${varValueName}: Decimal})), ${column})) = 0`,
query: `empty(arrayFilter(x -> (((x.1) = {${varKeyName}: String}) AND ((x.2) ${this.operator} {${varValueName}: Decimal64(12)})), ${column})) = 0`,
params: { [varKeyName]: this.key, [varValueName]: this.value },
};
}
@@ -254,13 +339,15 @@ export class NumberObjectFilter implements Filter {
export class BooleanFilter implements Filter {
public clickhouseTable: string;
protected field: string;
protected value: boolean;
public field: string;
public operator: (typeof filterOperators)["boolean"][number];
public value: boolean;
protected tablePrefix?: string;
constructor(opts: {
clickhouseTable: string;
field: string;
operator: (typeof filterOperators)["boolean"][number];
value: boolean;
tablePrefix?: string;
}) {
@@ -268,13 +355,14 @@ export class BooleanFilter implements Filter {
this.field = opts.field;
this.value = opts.value;
this.tablePrefix = opts.tablePrefix;
this.operator = opts.operator;
}
apply(): ClickhouseFilter {
const uid = randomCharacters();
const uid = clickhouseCompliantRandomCharacters();
const varName = `booleanFilter${uid}`;
return {
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} = {${varName}: Boolean}`,
query: `${this.tablePrefix ? this.tablePrefix + "." : ""}${this.field} ${this.operator} {${varName}: Boolean}`,
params: { [varName]: this.value },
};
}
@@ -283,7 +371,7 @@ export class BooleanFilter implements Filter {
export class FilterList {
private filters: Filter[];
constructor(filters: Filter[]) {
constructor(filters: Filter[] = []) {
this.filters = filters;
}
@@ -291,6 +379,14 @@ export class FilterList {
this.filters.push(...filter);
}
find(predicate: (filter: Filter) => boolean) {
return this.filters.find(predicate);
}
length() {
return this.filters.length;
}
public apply(): ClickhouseFilter {
if (this.filters.length === 0) {
return {
@@ -1,7 +1,7 @@
import z from "zod";
import { singleFilter } from "../../../interfaces/filters";
import { FilterCondition } from "../../../types";
import { isValidTableName } from "../../clickhouse/schema-utils";
import { isValidTableName } from "../../clickhouse/schemaUtils";
import { logger } from "../../logger";
import { UiColumnMapping } from "../../../tableDefinitions";
import {
@@ -13,6 +13,8 @@ import {
ArrayOptionsFilter,
BooleanFilter,
NumberObjectFilter,
StringObjectFilter,
NullFilter,
} from "./clickhouse-filter";
export class QueryBuilderError extends Error {
@@ -65,6 +67,7 @@ export const createFilterFromFilterState = (
operator: frontEndFilter.operator,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
clickhouseTypeOverwrite: column.clickhouseTypeOverwrite,
});
case "arrayOptions":
return new ArrayOptionsFilter({
@@ -79,6 +82,7 @@ export const createFilterFromFilterState = (
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
value: frontEndFilter.value,
operator: frontEndFilter.operator,
tablePrefix: column.queryPrefix,
});
case "numberObject":
@@ -90,11 +94,26 @@ export const createFilterFromFilterState = (
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "stringObject":
return new StringObjectFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
key: frontEndFilter.key,
value: frontEndFilter.value,
tablePrefix: column.queryPrefix,
});
case "null":
return new NullFilter({
clickhouseTable: column.clickhouseTableName,
field: column.clickhouseSelect,
operator: frontEndFilter.operator,
tablePrefix: column.queryPrefix,
});
default:
throw new QueryBuilderError(
`Invalid filter type: ${frontEndFilter.type}`,
);
const exhaustiveCheck: never = frontEndFilter;
logger.error(`Invalid filter type: ${JSON.stringify(exhaustiveCheck)}`);
throw new QueryBuilderError(`Invalid filter type`);
}
});
};
@@ -105,9 +124,9 @@ const matchAndVerifyTracesUiColumn = (
) => {
// tries to match the column name to the clickhouse table name
logger.debug(`Filter to match: ${JSON.stringify(filter)}`);
const uiTable = uiTableDefinitions.find(
(col) => col.uiTableName === filter.column, // matches on the NAME of the column in the UI.
(col) =>
col.uiTableName === filter.column || col.uiTableId === filter.column, // matches on the NAME of the column in the UI.
);
if (!uiTable) {
@@ -0,0 +1,20 @@
const regexIndefiniteCharacters = "%";
export const clickhouseSearchCondition = (query?: string) => {
return {
query: query
? `
AND (
id ILIKE {searchString: String} OR
user_id ILIKE {searchString: String} OR
name ILIKE {searchString: String}
)
`
: "",
params: query
? {
searchString: `${regexIndefiniteCharacters}${query}${regexIndefiniteCharacters}`,
}
: {},
};
};
@@ -11,13 +11,13 @@ export function parseTraceAllFilters(input: TableFilters) {
const filterCondition = tableColumnsToSqlFilterAndPrefix(
input.filter ?? [],
tracesTableCols,
"traces"
"traces",
);
const orderByCondition = orderByToPrismaSql(input.orderBy, tracesTableCols);
// to improve query performance, add timeseries filter to observation queries as well
const timeseriesFilter = input.filter?.find(
(f) => f.column === "Timestamp" && f.type === "datetime"
(f) => f.column === "Timestamp" && f.type === "datetime",
);
const observationTimeseriesFilter =
@@ -25,7 +25,7 @@ export function parseTraceAllFilters(input: TableFilters) {
? datetimeFilterToPrismaSql(
"start_time",
timeseriesFilter.operator,
timeseriesFilter.value
timeseriesFilter.value,
)
: Prisma.empty;
@@ -130,6 +130,6 @@ export function createTracesQuery({
${filterCondition}
${orderByCondition}
${limit ? Prisma.sql`LIMIT ${limit}` : Prisma.empty}
${page && limit ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
${page !== undefined && limit !== undefined ? Prisma.sql`OFFSET ${page * limit}` : Prisma.empty}
`;
}
@@ -7,3 +7,16 @@ export {
type FullObservationsWithScores,
type IOAndMetadataOmittedObservations,
} from "./createGenerationsQuery";
export {
FilterList,
StringFilter,
DateTimeFilter,
StringOptionsFilter,
NumberFilter,
ArrayOptionsFilter,
BooleanFilter,
NumberObjectFilter,
StringObjectFilter,
NullFilter,
type ClickhouseOperator,
} from "./clickhouse-sql/clickhouse-filter";
+40
View File
@@ -7,6 +7,7 @@ export enum EventName {
EvaluationExecution = "EvaluationExecution",
LegacyIngestion = "LegacyIngestion",
CloudUsageMetering = "CloudUsageMetering",
ExperimentCreate = "ExperimentCreate",
}
export const LegacyIngestionEventFull = z.object({
@@ -66,17 +67,36 @@ export const TraceUpsertEventSchema = z.object({
projectId: z.string(),
traceId: z.string(),
});
export const DatasetRunItemUpsertEventSchema = z.object({
projectId: z.string(),
datasetItemId: z.string(),
traceId: z.string(),
observationId: z.string().optional(),
});
export const EvalExecutionEvent = z.object({
projectId: z.string(),
jobExecutionId: z.string(),
delay: z.number().nullish(),
});
export const ExperimentCreateEventSchema = z.object({
projectId: z.string(),
datasetId: z.string(),
runId: z.string(),
description: z.string().optional(),
});
export type BatchExportJobType = z.infer<typeof BatchExportJobSchema>;
export type TraceUpsertEventType = z.infer<typeof TraceUpsertEventSchema>;
export type DatasetRunItemUpsertEventType = z.infer<
typeof DatasetRunItemUpsertEventSchema
>;
export type EvalExecutionEventType = z.infer<typeof EvalExecutionEvent>;
export type LegacyIngestionEventType = z.infer<typeof LegacyIngestionEvent>;
export type IngestionEventQueueType = z.infer<typeof IngestionEvent>;
export type ExperimentCreateEventType = z.infer<
typeof ExperimentCreateEventSchema
>;
export const EventBodySchema = z.union([
z.object({
@@ -91,26 +111,34 @@ export const EventBodySchema = z.union([
name: z.literal(EventName.BatchExport),
payload: BatchExportJobSchema,
}),
z.object({
name: z.literal(EventName.ExperimentCreate),
payload: ExperimentCreateEventSchema,
}),
]);
export type EventBodyType = z.infer<typeof EventBodySchema>;
export enum QueueName {
TraceUpsert = "trace-upsert", // Ingestion pipeline adds events on each Trace upsert
EvaluationExecution = "evaluation-execution-queue", // Worker executes Evals
DatasetRunItemUpsert = "dataset-run-item-upsert-queue",
BatchExport = "batch-export-queue",
IngestionQueue = "ingestion-queue", // Process single events with S3-merge
LegacyIngestionQueue = "legacy-ingestion-queue", // Used for batch processing of Ingestion
CloudUsageMeteringQueue = "cloud-usage-metering-queue",
ExperimentCreate = "experiment-create-queue",
}
export enum QueueJobs {
TraceUpsert = "trace-upsert",
DatasetRunItemUpsert = "dataset-run-item-upsert",
EvaluationExecution = "evaluation-execution-job",
BatchExportJob = "batch-export-job",
EnqueueBatchExportJobs = "enqueue-batch-export-jobs",
LegacyIngestionJob = "legacy-ingestion-job",
CloudUsageMeteringJob = "cloud-usage-metering-job",
IngestionJob = "ingestion-job",
ExperimentCreateJob = "experiment-create-job",
}
export type TQueueJobTypes = {
@@ -120,6 +148,12 @@ export type TQueueJobTypes = {
payload: TraceUpsertEventType;
name: QueueJobs.TraceUpsert;
};
[QueueName.DatasetRunItemUpsert]: {
timestamp: Date;
id: string;
payload: DatasetRunItemUpsertEventType;
name: QueueJobs.DatasetRunItemUpsert;
};
[QueueName.EvaluationExecution]: {
timestamp: Date;
id: string;
@@ -144,4 +178,10 @@ export type TQueueJobTypes = {
payload: IngestionEventQueueType;
name: QueueJobs.IngestionJob;
};
[QueueName.ExperimentCreate]: {
timestamp: Date;
id: string;
payload: ExperimentCreateEventType;
name: QueueJobs.ExperimentCreateJob;
};
};
@@ -0,0 +1,61 @@
import { Queue } from "bullmq";
import { env } from "../..";
import { logger } from "@azure/storage-blob";
import { QueueName, QueueJobs } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class CloudUsageMeteringQueue {
private static instance: Queue | null = null;
public static getInstance(): Queue | null {
if (!env.STRIPE_SECRET_KEY) {
return null;
}
if (CloudUsageMeteringQueue.instance) {
return CloudUsageMeteringQueue.instance;
}
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
CloudUsageMeteringQueue.instance = newRedis
? new Queue(QueueName.CloudUsageMeteringQueue, {
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 100,
attempts: 5,
backoff: {
type: "exponential",
delay: 5000,
},
},
})
: null;
CloudUsageMeteringQueue.instance?.on("error", (err) => {
logger.error("CloudUsageMeteringQueue error", err);
});
if (CloudUsageMeteringQueue.instance) {
CloudUsageMeteringQueue.instance.add(
QueueJobs.CloudUsageMeteringJob,
{},
{
repeat: { pattern: "5 * * * *" },
},
);
CloudUsageMeteringQueue.instance.add(
QueueJobs.CloudUsageMeteringJob,
{},
{},
);
}
return CloudUsageMeteringQueue.instance;
}
}
@@ -0,0 +1,47 @@
import { QueueName, TQueueJobTypes } from "../queues";
import { Queue } from "bullmq";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
import { logger } from "../logger";
export class DatasetRunItemUpsertQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.DatasetRunItemUpsert]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.DatasetRunItemUpsert]
> | null {
if (DatasetRunItemUpsertQueue.instance)
return DatasetRunItemUpsertQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
DatasetRunItemUpsertQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.DatasetRunItemUpsert]>(
QueueName.DatasetRunItemUpsert,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 5,
delay: 30_000, // 30 seconds
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
DatasetRunItemUpsertQueue.instance?.on("error", (err) => {
logger.error("DatasetRunItemUpsertQueue error", err);
});
return DatasetRunItemUpsertQueue.instance;
}
}
@@ -0,0 +1,45 @@
import { Queue } from "bullmq";
import { logger } from "../logger";
import { TQueueJobTypes, QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class EvalExecutionQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.EvaluationExecution]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.EvaluationExecution]
> | null {
if (EvalExecutionQueue.instance) return EvalExecutionQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
EvalExecutionQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.EvaluationExecution]>(
QueueName.EvaluationExecution,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 10,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
EvalExecutionQueue.instance?.on("error", (err) => {
logger.error("EvalExecutionQueue error", err);
});
return EvalExecutionQueue.instance;
}
}
@@ -0,0 +1,45 @@
import { Queue } from "bullmq";
import { logger } from "../logger";
import { TQueueJobTypes, QueueName } from "../queues";
import { createNewRedisInstance, redisQueueRetryOptions } from "./redis";
export class ExperimentCreateQueue {
private static instance: Queue<
TQueueJobTypes[QueueName.ExperimentCreate]
> | null = null;
public static getInstance(): Queue<
TQueueJobTypes[QueueName.ExperimentCreate]
> | null {
if (ExperimentCreateQueue.instance) return ExperimentCreateQueue.instance;
const newRedis = createNewRedisInstance({
enableOfflineQueue: false,
...redisQueueRetryOptions,
});
ExperimentCreateQueue.instance = newRedis
? new Queue<TQueueJobTypes[QueueName.ExperimentCreate]>(
QueueName.ExperimentCreate,
{
connection: newRedis,
defaultJobOptions: {
removeOnComplete: true,
removeOnFail: 10_000,
attempts: 2,
backoff: {
type: "exponential",
delay: 5000,
},
},
},
)
: null;
ExperimentCreateQueue.instance?.on("error", (err) => {
logger.error("ExperimentCreateQueue error", err);
});
return ExperimentCreateQueue.instance;
}
}
@@ -0,0 +1,34 @@
import { Queue } from "bullmq";
import { QueueName } from "../queues";
import { BatchExportQueue } from "./batchExport";
import { CloudUsageMeteringQueue } from "./CloudUsageMeteringQueue";
import { DatasetRunItemUpsertQueue } from "./datasetRunItemUpsert";
import { EvalExecutionQueue } from "./evalExecutionQueue";
import { ExperimentCreateQueue } from "./experimentCreateQueue";
import { IngestionQueue } from "./ingestionQueue";
import { LegacyIngestionQueue } from "./legacyIngestion";
import { TraceUpsertQueue } from "./traceUpsert";
export function getQueue(queueName: QueueName): Queue | null {
switch (queueName) {
case QueueName.LegacyIngestionQueue:
return LegacyIngestionQueue.getInstance();
case QueueName.BatchExport:
return BatchExportQueue.getInstance();
case QueueName.CloudUsageMeteringQueue:
return CloudUsageMeteringQueue.getInstance();
case QueueName.DatasetRunItemUpsert:
return DatasetRunItemUpsertQueue.getInstance();
case QueueName.EvaluationExecution:
return EvalExecutionQueue.getInstance();
case QueueName.ExperimentCreate:
return ExperimentCreateQueue.getInstance();
case QueueName.TraceUpsert:
return TraceUpsertQueue.getInstance();
case QueueName.IngestionQueue:
return IngestionQueue.getInstance();
default:
const exhaustiveCheckDefault: never = queueName;
throw new Error(`Queue ${queueName} not found`);
}
}
@@ -32,6 +32,7 @@ export class TraceUpsertQueue {
removeOnComplete: 100, // Important: If not true, new jobs for that ID would be ignored as jobs in the complete set are still considered as part of the queue
removeOnFail: 100_000,
attempts: 5,
delay: 10_000, // 10 seconds
backoff: {
type: "exponential",
delay: 5000,
@@ -48,43 +49,3 @@ export class TraceUpsertQueue {
return TraceUpsertQueue.instance;
}
}
export function convertTraceUpsertEventsToRedisEvents(
events: TraceUpsertEventType[],
) {
const uniqueTracesPerProject = events.reduce((acc, event) => {
if (!acc.get(event.projectId)) {
acc.set(event.projectId, new Set());
}
acc.get(event.projectId)?.add(event.traceId);
return acc;
}, new Map<string, Set<string>>());
return [...uniqueTracesPerProject.entries()]
.map((tracesPerProject) => {
const [projectId, traceIds] = tracesPerProject;
return [...traceIds].map((traceId) => ({
name: QueueJobs.TraceUpsert,
data: {
payload: {
projectId,
traceId,
},
id: randomUUID(),
timestamp: new Date(),
name: QueueJobs.TraceUpsert as const,
},
opts: {
removeOnFail: 1_000,
removeOnComplete: true,
attempts: 5,
backoff: {
type: "exponential",
delay: 1000,
},
},
}));
})
.flat();
}
@@ -0,0 +1,15 @@
## Repository docs
### Guarantees for relating data within Langfuse
- Finding a Trace based on an observation [Linear](https://linear.app/langfuse/issue/LFE-2745/improve-generations-table-query-performance)
- Traces can occur earlier than an observation.
- 96% of observations.start_time occur 2 mins later than the trace.timestamp
- There is a very large long-tail. Hence we will use a 2-day (2880 min) look back for now.
- Finding an Observation based on a Trace [Linear](https://linear.app/langfuse/issue/LFE-2409/table-queries)
- Observations have a very high likelihood of happening after the trace.
- 97% of traces.timestamp occur 2 mins earlier than the observation.start_time
- We will maintain a 1 hour cutoff for now.
- Finding traces/observations based on a Score timestamp
- For scores we have a very high likelihood of happening after the trace / observation.
- We maintain a 1 hour cutoff for now.
@@ -1,48 +1,130 @@
import { JsonNested } from "../../utils/zod";
import { env } from "../../env";
import { clickhouseClient } from "../clickhouse/client";
import {
clickhouseClient,
convertDateToClickhouseDateTime,
} from "../clickhouse/client";
import { logger } from "../logger";
import { getCurrentSpan, instrumentAsync } from "../instrumentation";
import { instrumentAsync } from "../instrumentation";
import {
StorageService,
StorageServiceFactory,
} from "../services/StorageService";
import { randomUUID } from "crypto";
import { getClickhouseEntityType } from "../clickhouse/schemaUtils";
import { NodeClickHouseClientConfigOptions } from "@clickhouse/client/dist/config";
export const convertRecordToJsonSchema = (
record: Record<string, string>,
): JsonNested | undefined => {
const jsonSchema: JsonNested = {};
let s3StorageServiceClient: StorageService;
// if record is empty, return undefined
if (Object.keys(record).length === 0) {
return undefined;
const getS3StorageServiceClient = (bucketName: string): StorageService => {
if (!s3StorageServiceClient) {
s3StorageServiceClient = StorageServiceFactory.getInstance({
bucketName,
accessKeyId: env.LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID,
secretAccessKey: env.LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY,
endpoint: env.LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT,
region: env.LANGFUSE_S3_EVENT_UPLOAD_REGION,
forcePathStyle: env.LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE === "true",
});
}
for (const key in record) {
try {
jsonSchema[key] = JSON.parse(record[key]);
} catch (e) {
jsonSchema[key] = record[key];
}
}
return jsonSchema;
return s3StorageServiceClient;
};
export async function upsertClickhouse<
T extends Record<string, unknown>,
>(opts: {
table: "scores" | "traces" | "observations";
records: T[];
eventBodyMapper: (body: T) => Record<string, unknown>;
}): Promise<void> {
return await instrumentAsync({ name: "clickhouse-upsert" }, async (span) => {
// https://opentelemetry.io/docs/specs/semconv/database/database-spans/
span.setAttribute("ch.query.table", opts.table);
// If event upload is enabled, we store all rows in S3 to have a backup
if (env.LANGFUSE_S3_EVENT_UPLOAD_ENABLED === "true") {
if (env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET === undefined) {
throw new Error("S3 event store is enabled but no bucket is set");
}
const s3Client = getS3StorageServiceClient(
env.LANGFUSE_S3_EVENT_UPLOAD_BUCKET,
);
await Promise.all(
opts.records.map((record) => {
// drop trailing s and pretend it's always a create.
// Only applicable to scores and traces.
let eventType = `${opts.table.slice(0, -1)}-create`;
if (opts.table === "observations") {
// @ts-ignore - If it's an observation we now that `type` is a string
eventType = `${record["type"].toLowerCase()}-create`;
}
s3Client.uploadJson(
`${env.LANGFUSE_S3_EVENT_UPLOAD_PREFIX}${record.project_id}/${getClickhouseEntityType(eventType)}/${record.id}/${randomUUID()}.json`,
[
{
id: randomUUID(),
timestamp: new Date().toISOString(),
type: eventType,
body: opts.eventBodyMapper(record),
},
],
);
}),
);
}
const res = await clickhouseClient().insert({
table: opts.table,
values: opts.records.map((record) => ({
...record,
event_ts: convertDateToClickhouseDateTime(new Date()),
})),
format: "JSONEachRow",
});
// same logic as for prisma. we want to see queries in development
if (env.NODE_ENV === "development") {
logger.info(`clickhouse:insert ${res.query_id} ${opts.table}`);
}
span.setAttribute("ch.queryId", res.query_id);
// add summary headers to the span. Helps to tune performance
const summaryHeader = res.response_headers["x-clickhouse-summary"];
if (summaryHeader) {
try {
const summary = Array.isArray(summaryHeader)
? JSON.parse(summaryHeader[0])
: JSON.parse(summaryHeader);
for (const key in summary) {
span.setAttribute(`ch.${key}`, summary[key]);
}
} catch (error) {
logger.debug(
`Failed to parse clickhouse summary header ${summaryHeader}`,
error,
);
}
}
});
}
export async function queryClickhouse<T>(opts: {
query: string;
params?: Record<string, unknown> | undefined;
}) {
clickhouseConfigs?: NodeClickHouseClientConfigOptions;
}): Promise<T[]> {
return await instrumentAsync({ name: "clickhouse-query" }, async (span) => {
// https://opentelemetry.io/docs/specs/semconv/database/database-spans/
span.setAttribute("ch.query.text", opts.query);
// same logic as for prisma. we want to see queries in development
if (env.NODE_ENV === "development") {
logger.info(`clickhouse:query ${opts.query}`);
}
const res = await clickhouseClient.query({
const res = await clickhouseClient(opts.clickhouseConfigs).query({
query: opts.query,
format: "JSONEachRow",
query_params: opts.params,
});
// same logic as for prisma. we want to see queries in development
if (env.NODE_ENV === "development") {
logger.info(`clickhouse:query ${res.query_id} ${opts.query}`);
}
span.setAttribute("ch.queryId", res.query_id);
@@ -68,6 +150,56 @@ export async function queryClickhouse<T>(opts: {
});
}
export async function commandClickhouse<T>(opts: {
query: string;
params?: Record<string, unknown> | undefined;
clickhouseConfigs?: NodeClickHouseClientConfigOptions;
}): Promise<void> {
return await instrumentAsync({ name: "clickhouse-command" }, async (span) => {
// https://opentelemetry.io/docs/specs/semconv/database/database-spans/
span.setAttribute("ch.query.text", opts.query);
const res = await clickhouseClient(opts.clickhouseConfigs).command({
query: opts.query,
query_params: opts.params,
});
// same logic as for prisma. we want to see queries in development
if (env.NODE_ENV === "development") {
logger.info(`clickhouse:query ${res.query_id} ${opts.query}`);
}
span.setAttribute("ch.queryId", res.query_id);
// add summary headers to the span. Helps to tune performance
const summaryHeader = res.response_headers["x-clickhouse-summary"];
if (summaryHeader) {
try {
const summary = Array.isArray(summaryHeader)
? JSON.parse(summaryHeader[0])
: JSON.parse(summaryHeader);
for (const key in summary) {
span.setAttribute(`ch.${key}`, summary[key]);
}
} catch (error) {
logger.debug(
`Failed to parse clickhouse summary header ${summaryHeader}`,
error,
);
}
}
});
}
export function parseClickhouseUTCDateTimeFormat(dateStr: string): Date {
return new Date(`${dateStr.replace(" ", "T")}Z`);
}
export function clickhouseCompliantRandomCharacters() {
const chars = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
let result = "";
const randomArray = new Uint8Array(5);
crypto.getRandomValues(randomArray);
randomArray.forEach((number) => {
result += chars[number % chars.length];
});
return result;
}
@@ -0,0 +1,9 @@
// Rule of thumb: If you join observations from left, use observations to trace and vice versa
// t.timestamp > observation.start_time - 2 days
export const OBSERVATIONS_TO_TRACE_INTERVAL = "INTERVAL 2 DAY";
// observation.start_time > t.timestamp - 1 hour
export const TRACE_TO_OBSERVATIONS_INTERVAL = "INTERVAL 1 HOUR";
// observation.start_time > s.timestamp - 1 hour
// t.timestamp > s.timestamp - 1 hour
export const SCORE_TO_TRACE_OBSERVATIONS_INTERVAL = "INTERVAL 1 HOUR";
@@ -0,0 +1,676 @@
import { queryClickhouse } from "./clickhouse";
import { createFilterFromFilterState } from "../queries/clickhouse-sql/factory";
import { FilterState } from "../../types";
import {
DateTimeFilter,
FilterList,
} from "../queries/clickhouse-sql/clickhouse-filter";
import { dashboardColumnDefinitions } from "../../tableDefinitions/mapDashboards";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import {
OBSERVATIONS_TO_TRACE_INTERVAL,
SCORE_TO_TRACE_OBSERVATIONS_INTERVAL,
TRACE_TO_OBSERVATIONS_INTERVAL,
} from "./constants";
export type DateTrunc = "year" | "month" | "week" | "day" | "hour" | "minute";
export const getTotalTraces = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
).apply();
const query = `
SELECT
count(id) as count
FROM traces t FINAL
WHERE project_id = {projectId: String}
AND ${chFilter.query}`;
const result = await queryClickhouse<{ count: number }>({
query,
params: {
projectId,
...chFilter.params,
},
});
if (result.length === 0) {
return undefined;
}
return [{ countTraceId: result[0].count }];
};
export const getObservationsCostGroupedByName = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const hasTraceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
// TODO: Validate whether we can filter traces on timestamp here.
const query = `
SELECT
provided_model_name as name,
sumMap(cost_details)['total'] as sum_cost_details,
sumMap(usage_details)['total'] as sum_usage_details
FROM observations o FINAL ${hasTraceFilter ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
GROUP BY provided_model_name
ORDER BY sumMap(cost_details)['total'] DESC
`;
const result = await queryClickhouse<{
name: string;
sum_cost_details: number;
sum_usage_details: number;
}>({
query,
params: {
projectId,
...appliedFilter.params,
},
});
return result;
};
export const getScoreAggregate = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const timeFilter = chFilter.find(
(f) =>
f.field === "timestamp" && (f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined;
const chFilterApplied = chFilter.apply();
const hasTraceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
// TODO: Validate whether we can filter traces on timestamp here.
const query = `
SELECT
s.name,
count(*) as count,
avg(s.value) as avg_value,
s.source,
s.data_type
FROM scores s FINAL
${hasTraceFilter ? "JOIN traces t FINAL ON t.id = s.trace_id AND t.project_id = s.project_id" : ""}
WHERE s.project_id = {projectId: String}
AND ${chFilterApplied.query}
${timeFilter && hasTraceFilter ? `AND t.timestamp >= {tracesTimestamp: DateTime64(3)} - ${SCORE_TO_TRACE_OBSERVATIONS_INTERVAL}` : ""}
GROUP BY s.name, s.source, s.data_type
ORDER BY count(*) DESC
`;
const result = await queryClickhouse<{
name: string;
count: string;
avg_value: string;
source: string;
data_type: string;
}>({
query,
params: {
projectId,
...chFilterApplied.params,
...(timeFilter
? { tracesTimestamp: convertDateToClickhouseDateTime(timeFilter.value) }
: {}),
},
});
return result;
};
export const groupTracesByTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
).apply();
const query = `
SELECT
${selectTimeseriesColumn(groupBy, "timestamp", "timestamp")},
count(*) as count
FROM traces t FINAL
WHERE project_id = {projectId: String}
AND ${chFilter.query}
GROUP BY timestamp
${orderByTimeSeries(groupBy, "timestamp")}
`;
const result = await queryClickhouse<{
timestamp: string;
count: string;
}>({
query,
params: {
projectId,
...chFilter.params,
},
});
return result.map((row) => ({
timestamp: new Date(row.timestamp),
countTraceId: Number(row.count),
}));
};
export const getObservationUsageByTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const tracesFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const timeFilter = tracesFilter
? (chFilter.find(
(f) =>
f.clickhouseTable === "observations" &&
f.field.includes("start_time") &&
(f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined)
: undefined;
const query = `
SELECT
${selectTimeseriesColumn(groupBy, "start_time", "start_time")},
sumMap(usage_details)['total'] as sum_usage_details,
sumMap(cost_details)['total'] as sum_cost_details,
provided_model_name
FROM observations o FINAL
${tracesFilter ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
${timeFilter ? `AND t.timestamp >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
GROUP BY start_time, provided_model_name
${orderByTimeSeries(groupBy, "start_time")}
`;
const result = await queryClickhouse<{
start_time: string;
sum_usage_details: string;
sum_cost_details: number;
provided_model_name: string;
}>({
query,
params: {
projectId,
...appliedFilter.params,
...(timeFilter
? { traceTimestamp: convertDateToClickhouseDateTime(timeFilter.value) }
: {}),
},
});
return result.map((row) => ({
start_time: new Date(row.start_time),
sum_usage_details: Number(row.sum_usage_details),
sum_cost_details: row.sum_cost_details,
provided_model_name: row.provided_model_name,
}));
};
export const getDistinctModels = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const tracesFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const timeFilter = tracesFilter
? (chFilter.find(
(f) =>
f.clickhouseTable === "observations" &&
f.field.includes("start_time") &&
(f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined)
: undefined;
// No need for final as duplicates are caught by distinct anyway.
const query = `
SELECT distinct(provided_model_name) as model
FROM observations o
${tracesFilter ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
${timeFilter ? `AND t.timestamp >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
`;
const result = await queryClickhouse<{ model: string }>({
query,
params: {
projectId,
...appliedFilter.params,
...(timeFilter
? { traceTimestamp: convertDateToClickhouseDateTime(timeFilter.value) }
: {}),
},
});
return result;
};
export const getScoresAggregateOverTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
// TODO: Validate whether we can filter traces on timestamp here.
const query = `
SELECT
${selectTimeseriesColumn(groupBy, "timestamp", "timestamp")},
name,
data_type,
source,
AVG(value) as avg_value
FROM scores FINAL
${traceFilter ? "JOIN traces t ON scores.trace_id = t.id AND scores.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
AND data_type IN ('NUMERIC', 'BOOLEAN')
GROUP BY
timestamp,
name,
data_type,
source
${orderByTimeSeries(groupBy, "timestamp")};
`;
const result = await queryClickhouse<{
timestamp: string;
name: string;
data_type: string;
source: string;
avg_value: number;
}>({
query,
params: {
projectId,
...appliedFilter.params,
},
});
return result.map((row) => ({
scoreTimestamp: new Date(row.timestamp),
scoreName: row.name,
scoreDataType: row.data_type,
scoreSource: row.source,
avgValue: Number(row.avg_value),
}));
};
export const getModelUsageByUser = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const timeFilter = chFilter.find(
(f) =>
f.clickhouseTable === "observations" &&
f.field.includes("start_time") &&
(f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined;
const query = `
SELECT
sumMap(usage_details)['total'] as sum_usage_details,
sumMap(cost_details)['total'] as sum_cost_details,
user_id
FROM observations o FINAL
JOIN traces t FINAL
ON o.trace_id = t.id AND o.project_id = t.project_id
WHERE project_id = {projectId: String}
AND t.user_id IS NOT NULL
AND ${appliedFilter.query}
${timeFilter ? `AND t.timestamp >= {traceTimestamp: DateTime64(3)} - ${OBSERVATIONS_TO_TRACE_INTERVAL}` : ""}
GROUP BY user_id
ORDER BY sum_cost_details DESC
`;
const result = await queryClickhouse<{
sum_usage_details: string;
sum_cost_details: number;
user_id: string;
}>({
query,
params: {
projectId,
...appliedFilter.params,
...(timeFilter
? { traceTimestamp: convertDateToClickhouseDateTime(timeFilter.value) }
: {}),
},
});
return result.map((row) => ({
sumUsageDetails: Number(row.sum_usage_details),
sumCostDetails: Number(row.sum_cost_details),
userId: row.user_id,
}));
};
export const getObservationLatencies = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
// Skipping FINAL here, as the quantiles are approximate to begin with.
const query = `
SELECT
quantiles(0.5, 0.9, 0.95, 0.99)(date_diff('milliseconds', o.start_time, o.end_time)) as quantiles,
name
FROM observations o
${chFilter.find((f) => f.clickhouseTable === "traces") ? "LEFT JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
GROUP BY name
ORDER BY quantiles[2] DESC
`;
const result = await queryClickhouse<{ quantiles: string[]; name: string }>({
query,
params: { projectId, ...appliedFilter.params },
});
return result.map((row) => ({
p50: Number(row.quantiles[0]) / 1000,
p90: Number(row.quantiles[1]) / 1000,
p95: Number(row.quantiles[2]) / 1000,
p99: Number(row.quantiles[3]) / 1000,
name: row.name,
}));
};
export const getTracesLatencies = async (
projectId: string,
filter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const timestampFilter = chFilter.find(
(f) =>
f.clickhouseTable === "traces" &&
f.field === 't."timestamp"' &&
(f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined;
// Skipping FINAL here, as the quantiles are approximate to begin with.
const query = `
WITH trace_latencies as (
select o.trace_id,
t.name,
o.project_id,
date_diff('milliseconds', min(o.start_time), coalesce(max(o.end_time), max(o.start_time))) as duration
FROM traces t
JOIN observations o
ON o.trace_id = t.id AND o.project_id = t.project_id
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
${timestampFilter ? `AND o.start_time > {dateTimeFilterObservations: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
GROUP BY o.project_id, o.trace_id, t.name
)
SELECT
quantiles(0.5, 0.9, 0.95, 0.99)(duration) as quantiles,
name
FROM trace_latencies
GROUP BY name
ORDER BY quantiles[2] DESC
`;
const result = await queryClickhouse<{ quantiles: string[]; name: string }>({
query,
params: {
projectId,
...appliedFilter.params,
...(timestampFilter
? { dateTimeFilterObservations: timestampFilter.value }
: {}),
},
});
return result.map((row) => ({
p50: Number(row.quantiles[0]) / 1000,
p90: Number(row.quantiles[1]) / 1000,
p95: Number(row.quantiles[2]) / 1000,
p99: Number(row.quantiles[3]) / 1000,
name: row.name,
}));
};
export const getModelLatenciesOverTime = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const appliedFilter = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
// Skipping FINAL here, as the quantiles are approximate to begin with.
const query = `
SELECT
${selectTimeseriesColumn(groupBy, "o.start_time", "start_time_bucket")},
provided_model_name,
quantiles(0.5, 0.75, 0.9, 0.95, 0.99)(date_diff('milliseconds', o.start_time, o.end_time)) as quantiles
FROM observations o
${traceFilter ? "JOIN traces t ON o.trace_id = t.id AND o.project_id = t.project_id" : ""}
WHERE project_id = {projectId: String}
AND ${appliedFilter.query}
GROUP BY provided_model_name, start_time_bucket
${orderByTimeSeries(groupBy, "start_time_bucket")};
`;
const result = await queryClickhouse<{
start_time_bucket: string;
provided_model_name: string;
quantiles: string[];
}>({ query, params: { projectId, ...appliedFilter.params } });
return result.map((row) => ({
p50: Number(row.quantiles[0]) / 1000,
p75: Number(row.quantiles[1]) / 1000,
p90: Number(row.quantiles[2]) / 1000,
p95: Number(row.quantiles[3]) / 1000,
p99: Number(row.quantiles[4]) / 1000,
model: row.provided_model_name,
start_time: new Date(row.start_time_bucket),
}));
};
export const getNumericScoreTimeSeries = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const chFilterRes = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const query = `
SELECT
${selectTimeseriesColumn(groupBy, "s.timestamp", "score_timestamp")},
s.name as score_name,
AVG(s.value) as avg_value
FROM scores s final
${traceFilter ? "JOIN traces t ON s.trace_id = t.id AND s.project_id = t.project_id" : ""}
WHERE s.project_id = {projectId: String}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
GROUP BY score_name, score_timestamp
${orderByTimeSeries(groupBy, "score_timestamp")}
`;
return queryClickhouse<{
score_timestamp: Date;
score_name: string;
avg_value: number;
}>({
query,
params: {
projectId,
...(chFilterRes ? chFilterRes.params : {}),
},
});
};
export const getCategoricalScoreTimeSeries = async (
projectId: string,
filter: FilterState,
groupBy: DateTrunc | undefined,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const chFilterRes = chFilter.apply();
const traceFilter = chFilter.find((f) => f.clickhouseTable === "traces");
const query = `
SELECT
${groupBy ? selectTimeseriesColumn(groupBy, "s.timestamp", "score_timestamp") + ", " : ""}
s.name as score_name,
s.data_type as score_data_type,
s.source as score_source,
s.string_value as score_value,
count(s.string_value) as count
FROM scores s final
${traceFilter ? "JOIN traces t ON s.trace_id = t.id AND s.project_id = t.project_id" : ""}
WHERE s.project_id = {projectId: String}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
GROUP BY score_name, score_data_type, score_source, score_value ${groupBy ? ", score_timestamp" : ""}
${groupBy ? orderByTimeSeries(groupBy, "score_timestamp") : ""}
`;
return queryClickhouse<{
score_timestamp?: Date;
score_name: string;
score_data_type: string;
score_source: string;
score_value: string;
count: number;
}>({
query,
params: {
projectId,
...(chFilterRes ? chFilterRes.params : {}),
},
});
};
const orderByTimeSeries = (dateTrunc: DateTrunc, col: string) => {
let interval;
switch (dateTrunc) {
case "year":
interval = "toIntervalYear(1)";
break;
case "month":
interval = "toIntervalMonth(1)";
break;
case "week":
interval = "toIntervalWeek(1)";
break;
case "day":
interval = "toIntervalDay(1)";
break;
case "hour":
interval = "toIntervalHour(1)";
break;
case "minute":
interval = "toIntervalMinute(1)";
break;
default:
return undefined;
}
return `ORDER BY ${col} ASC WITH FILL STEP ${interval}`;
};
const selectTimeseriesColumn = (
dateTrunc: DateTrunc,
col: string,
as: String,
) => {
let interval;
switch (dateTrunc) {
case "year":
interval = "toStartOfYear";
break;
case "month":
interval = "toStartOfMonth";
break;
case "week":
interval = "toStartOfWeek";
break;
case "day":
interval = "toStartOfDay";
break;
case "hour":
interval = "toStartOfHour";
break;
case "minute":
interval = "toStartOfMinute";
break;
default:
return undefined;
}
return `${interval}(${col}) as ${as}`;
};
@@ -9,10 +9,26 @@ export const clickhouseStringDateSchema = z
//https://clickhouse.com/docs/en/integrations/javascript#integral-types-int64-int128-int256-uint64-uint128-uint256
// clickhouse returns int64 as string
export const UsageCostStringSchema = z.record(z.string(), z.string());
export const UsageCostNumberSchema = z.record(z.string(), z.coerce.number());
export type UsageCostStringType = z.infer<typeof UsageCostStringSchema>;
export type UsageCostNumberType = z.infer<typeof UsageCostNumberSchema>;
export const UsageCostSchema = z
.record(z.string(), z.coerce.string().nullable())
.transform((val, ctx) => {
const result: Record<string, number> = {};
for (const key in val) {
if (val[key] !== null && val[key] !== undefined) {
const parsed = Number(val[key]);
if (isNaN(parsed)) {
ctx.addIssue({
code: z.ZodIssueCode.custom,
message: `Key ${key} is not a number`,
});
} else {
result[key] = parsed;
}
}
}
return result;
});
export type UsageCostType = z.infer<typeof UsageCostSchema>;
export const observationRecordBaseSchema = z.object({
id: z.string(),
@@ -47,10 +63,10 @@ export const observationRecordReadSchema = observationRecordBaseSchema.extend({
end_time: clickhouseStringDateSchema.nullish(),
completion_start_time: clickhouseStringDateSchema.nullish(),
event_ts: clickhouseStringDateSchema,
provided_usage_details: UsageCostStringSchema,
provided_cost_details: UsageCostNumberSchema,
usage_details: UsageCostStringSchema,
cost_details: UsageCostNumberSchema,
provided_usage_details: UsageCostSchema,
provided_cost_details: UsageCostSchema,
usage_details: UsageCostSchema,
cost_details: UsageCostSchema,
});
export type ObservationRecordReadType = z.infer<
typeof observationRecordReadSchema
@@ -64,10 +80,10 @@ export const observationRecordInsertSchema = observationRecordBaseSchema.extend(
end_time: z.number().nullish(),
completion_start_time: z.number().nullish(),
event_ts: z.number(),
provided_usage_details: UsageCostNumberSchema,
provided_cost_details: UsageCostNumberSchema,
usage_details: UsageCostNumberSchema,
cost_details: UsageCostNumberSchema,
provided_usage_details: UsageCostSchema,
provided_cost_details: UsageCostSchema,
usage_details: UsageCostSchema,
cost_details: UsageCostSchema,
},
);
export type ObservationRecordInsertType = z.infer<
@@ -113,8 +129,8 @@ export const scoreRecordBaseSchema = z.object({
project_id: z.string(),
trace_id: z.string(),
observation_id: z.string().nullish(),
name: z.string().nullish(),
value: z.union([z.number(), z.string()]).nullish(),
name: z.string(),
value: z.number().nullish(),
source: z.string(),
comment: z.string().nullish(),
author_user_id: z.string().nullish(),
@@ -159,11 +175,6 @@ export const convertObservationReadToInsert = (
): ObservationRecordInsertType => {
const convertDate = (date: string) => new Date(date).getTime();
const convertDetails = (details: Record<string, string>) =>
Object.fromEntries(
Object.entries(details).map(([key, value]) => [key, Number(value)]),
);
return {
...record,
created_at: convertDate(record.created_at),
@@ -174,9 +185,9 @@ export const convertObservationReadToInsert = (
? convertDate(record.completion_start_time)
: undefined,
event_ts: convertDate(record.event_ts),
provided_usage_details: convertDetails(record.provided_usage_details),
provided_usage_details: record.provided_usage_details,
provided_cost_details: record.provided_cost_details,
usage_details: convertDetails(record.usage_details),
usage_details: record.usage_details,
cost_details: record.cost_details,
};
};
@@ -268,9 +279,12 @@ export const convertPostgresObservationToInsert = (
: null,
provided_usage_details: {},
usage_details: {
input: observation.prompt_tokens,
output: observation.completion_tokens,
total: observation.total_tokens,
input: observation.prompt_tokens >= 0 ? observation.prompt_tokens : null,
output:
observation.completion_tokens >= 0
? observation.completion_tokens
: null,
total: observation.total_tokens >= 0 ? observation.total_tokens : null,
},
provided_cost_details: {
input: observation.input_cost?.toNumber() ?? null,
@@ -2,3 +2,8 @@ export * from "./scores";
export * from "./traces";
export * from "./observations";
export * from "./types";
export * from "./dashboards";
export * from "./traces_converters";
export * from "./scores_converters";
export * from "./observations_converters";
export * from "./clickhouse";
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,115 @@
import {
Observation,
ObservationView,
ObservationType,
ObservationLevel,
Prisma,
} from "@prisma/client";
import Decimal from "decimal.js";
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { ObservationRecordReadType } from "./definitions";
import { parseJsonPrioritised } from "../../utils/json";
import { jsonSchema } from "../../utils/zod";
export const convertObservationToView = (
record: ObservationRecordReadType,
): Omit<ObservationView, "inputPrice" | "outputPrice" | "totalPrice"> => {
// these cost are not used from the view. They are in the select statement but not in the
// Prisma file. We will not clean this up but keep it as it is for now.
// eslint-disable-next-line no-unused-vars
const { inputCost, outputCost, totalCost, internalModelId, ...rest } =
convertObservation(record ?? undefined);
return {
...rest,
latency: record.end_time
? parseClickhouseUTCDateTimeFormat(record.end_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.start_time).getTime()
: null,
timeToFirstToken: record.completion_start_time
? parseClickhouseUTCDateTimeFormat(record.start_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.completion_start_time).getTime()
: null,
promptName: record.prompt_name ?? null,
promptVersion: record.prompt_version ?? null,
modelId: record.internal_model_id ?? null,
};
};
export const convertObservation = (
record: ObservationRecordReadType,
): Omit<Observation, "internalModel"> & {
promptName: string | null;
promptVersion: number | null;
latency: number | null;
timeToFirstToken: number | null;
} => {
return {
id: record.id,
traceId: record.trace_id ?? null,
projectId: record.project_id,
type: record.type as ObservationType,
parentObservationId: record.parent_observation_id ?? null,
startTime: parseClickhouseUTCDateTimeFormat(record.start_time),
endTime: record.end_time
? parseClickhouseUTCDateTimeFormat(record.end_time)
: null,
name: record.name ?? null,
metadata: record.metadata,
level: record.level as ObservationLevel,
statusMessage: record.status_message ?? null,
version: record.version ?? null,
input: (record.input
? jsonSchema.parse(parseJsonPrioritised(record.input))
: null) as Prisma.JsonValue | null,
output: (record.output
? jsonSchema.parse(parseJsonPrioritised(record.output))
: null) as Prisma.JsonValue | null,
modelParameters: record.model_parameters
? JSON.parse(record.model_parameters)
: null,
completionStartTime: record.completion_start_time
? parseClickhouseUTCDateTimeFormat(record.completion_start_time)
: null,
promptId: record.prompt_id ?? null,
createdAt: parseClickhouseUTCDateTimeFormat(record.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(record.updated_at),
promptTokens: record.usage_details?.input
? Number(record.usage_details?.input)
: 0,
completionTokens: record.usage_details?.output
? Number(record.usage_details?.output)
: 0,
totalTokens: record.usage_details?.total
? Number(record.usage_details?.total)
: 0,
calculatedInputCost: record.cost_details?.input
? new Decimal(record.cost_details.input)
: null,
calculatedOutputCost: record.cost_details?.output
? new Decimal(record.cost_details.output)
: null,
calculatedTotalCost: record.cost_details?.total
? new Decimal(record.cost_details.total)
: null,
inputCost: record.cost_details?.input
? new Decimal(record.cost_details?.input)
: null,
outputCost: record.cost_details?.output
? new Decimal(record.cost_details?.output)
: null,
totalCost: record.total_cost ? new Decimal(record.total_cost) : null,
model: record.provided_model_name ?? null,
internalModelId: record.internal_model_id ?? null,
unit: "TOKENS", // to be removed.
promptName: record.prompt_name ?? null,
promptVersion: record.prompt_version ?? null,
latency: record.end_time
? parseClickhouseUTCDateTimeFormat(record.end_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.start_time).getTime()
: null,
timeToFirstToken: record.completion_start_time
? parseClickhouseUTCDateTimeFormat(record.start_time).getTime() -
parseClickhouseUTCDateTimeFormat(record.completion_start_time).getTime()
: null,
};
};
+426 -54
View File
@@ -1,74 +1,142 @@
import { ScoreDataType, ScoreSource } from "@prisma/client";
import { queryClickhouse } from "./clickhouse";
import { FilterList } from "../queries/clickhouse-filter/clickhouse-filter";
import { Score, ScoreDataType, ScoreSource } from "@prisma/client";
import {
commandClickhouse,
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
upsertClickhouse,
} from "./clickhouse";
import { FilterList } from "../queries/clickhouse-sql/clickhouse-filter";
import { FilterState } from "../../types";
import { createFilterFromFilterState } from "../queries/clickhouse-filter/factory";
import {
createFilterFromFilterState,
getProjectIdDefaultFilter,
} from "../queries/clickhouse-sql/factory";
import { OrderByState } from "../../interfaces/orderBy";
import {
dashboardColumnDefinitions,
scoresTableUiColumnDefinitions,
} from "../../tableDefinitions";
import { orderByToClickhouseSql } from "../queries/clickhouse-sql/orderby-factory";
import {
convertScoreAggregation,
convertToScore,
ScoreAggregation,
} from "./scores_converters";
import { SCORE_TO_TRACE_OBSERVATIONS_INTERVAL } from "./constants";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { ScoreRecordReadType } from "./definitions";
export type FetchScoresReturnType = {
id: string;
timestamp: string;
project_id: string;
trace_id: string;
observation_id: string | null;
name: string;
value: number;
source: string;
comment: string | null;
author_user_id: string | null;
config_id: string | null;
data_type: string;
string_value: string | null;
queue_id: string | null;
created_at: string;
updated_at: string;
event_ts: string;
is_deleted: number;
projectId: string;
export const searchExistingAnnotationScore = async (
projectId: string,
traceId: string,
observationId: string | null,
name: string | undefined,
configId: string | undefined,
) => {
if (!name && !configId) {
throw new Error("Either name or configId (or both) must be provided.");
}
const query = `
SELECT *
FROM scores s
WHERE s.project_id = {projectId: String}
AND s.source = 'ANNOTATION'
AND s.trace_id = {traceId: String}
${observationId ? `AND s.observation_id = {observationId: String}` : "AND isNull(s.observation_id)"}
AND (
FALSE
${name ? `OR s.name = {name: String}` : ""}
${configId ? `OR s.config_id = {configId: String}` : ""}
)
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
LIMIT 1
`;
const rows = await queryClickhouse<ScoreRecordReadType>({
query,
params: {
projectId,
name,
configId,
traceId,
observationId,
},
});
return rows.map(convertToScore).shift();
};
const convertToScore = (row: FetchScoresReturnType) => {
return {
id: row.id,
timestamp: new Date(row.timestamp),
projectId: row.project_id,
traceId: row.trace_id,
observationId: row.observation_id,
name: row.name,
value: row.value,
source: row.source as ScoreSource,
comment: row.comment,
authorUserId: row.author_user_id,
configId: row.config_id,
dataType: row.data_type as ScoreDataType,
stringValue: row.string_value,
queueId: row.queue_id,
createdAt: new Date(row.created_at),
updatedAt: new Date(row.updated_at),
};
export const getScoreById = async (
projectId: string,
scoreId: string,
source?: ScoreSource,
) => {
const query = `
SELECT *
FROM scores s
WHERE s.project_id = {projectId: String}
AND s.id = {scoreId: String}
${source ? `AND s.source = {source: String}` : ""}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
LIMIT 1
`;
const rows = await queryClickhouse<ScoreRecordReadType>({
query,
params: {
projectId,
scoreId,
...(source !== undefined ? { source } : {}),
},
});
return rows.map(convertToScore).shift();
};
/**
* Accepts a score in a Clickhouse-ready format.
* id, project_id, name, and timestamp must always be provided.
*/
export const upsertScore = async (score: Partial<ScoreRecordReadType>) => {
if (!["id", "project_id", "name", "timestamp"].every((key) => key in score)) {
throw new Error("Identifier fields must be provided to upsert Score.");
}
await upsertClickhouse({
table: "scores",
records: [score as ScoreRecordReadType],
eventBodyMapper: convertToScore,
});
};
export const getScoresForTraces = async (
projectId: string,
traceIds: string[],
timestamp?: Date,
limit?: number,
offset?: number,
) => {
const query = `
select
*
from scores s final
from scores s
WHERE s.project_id = {projectId: String}
AND s.trace_id IN ({traceIds: Array(String)})
AND s.trace_id IN ({traceIds: Array(String)})
${timestamp ? `AND s.timestamp >= {traceTimestamp: DateTime64(3)} - ${SCORE_TO_TRACE_OBSERVATIONS_INTERVAL}` : ""}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
${limit && offset ? `limit {limit: Int32} offset {offset: Int32}` : ""}
`;
const rows = await queryClickhouse<FetchScoresReturnType>({
const rows = await queryClickhouse<ScoreRecordReadType>({
query: query,
params: {
projectId: projectId,
traceIds: traceIds,
limit: limit,
offset: offset,
projectId,
traceIds,
limit,
offset,
...(timestamp
? { traceTimestamp: convertDateToClickhouseDateTime(timestamp) }
: {}),
},
});
@@ -84,13 +152,15 @@ export const getScoresForObservations = async (
const query = `
select
*
from scores s final
from scores s
WHERE s.project_id = {projectId: String}
AND s.observation_id IN ({observationIds: Array(String)})
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
${limit !== undefined && offset !== undefined ? `limit {limit: Int32} offset {offset: Int32}` : ""}
`;
const rows = await queryClickhouse<FetchScoresReturnType>({
const rows = await queryClickhouse<ScoreRecordReadType>({
query: query,
params: {
projectId: projectId,
@@ -104,12 +174,14 @@ export const getScoresForObservations = async (
};
export const getScoresGroupedByNameSourceType = async (projectId: string) => {
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
name,
source,
data_type
from scores s final
from scores s
WHERE s.project_id = {projectId: String}
GROUP BY name, source, data_type
ORDER BY count() desc
@@ -153,10 +225,12 @@ export const getScoresGroupedByName = async (
? new FilterList(chFilter).apply()
: undefined;
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
name as name
from scores s final
from scores s
WHERE s.project_id = {projectId: String}
AND has(['NUMERIC', 'BOOLEAN'], s.data_type)
${timestampFilterRes?.query ? `AND ${timestampFilterRes.query}` : ""}
@@ -177,3 +251,301 @@ export const getScoresGroupedByName = async (
return rows;
};
export const getScoresUiCount = async (props: {
projectId: string;
filter: FilterState;
orderBy: OrderByState;
limit?: number;
offset?: number;
}) => {
const rows = await getScoresUiGeneric<{ count: string }>({
select: `
count(*) as count
`,
...props,
});
return Number(rows[0].count);
};
export type ScoreUiTableRow = Score & {
traceName: string | null;
traceUserId: string | null;
traceTags: Array<string> | null;
};
export const getScoresUiTable = async (props: {
projectId: string;
filter: FilterState;
orderBy: OrderByState;
limit?: number;
offset?: number;
}): Promise<ScoreUiTableRow[]> => {
const rows = await getScoresUiGeneric<{
id: string;
project_id: string;
name: string;
value: number;
string_value: string | null;
timestamp: string;
source: string;
data_type: string;
comment: string | null;
trace_id: string;
observation_id: string | null;
author_user_id: string | null;
user_id: string | null;
trace_name: string | null;
trace_tags: Array<string> | null;
job_configuration_id: string | null;
author_user_image: string | null;
author_user_name: string | null;
config_id: string | null;
queue_id: string | null;
created_at: string;
updated_at: string;
}>({
select: `
s.id,
s.project_id,
s.name,
s.value,
s.string_value,
s.timestamp,
s.source,
s.data_type,
s.comment,
s.trace_id,
s.observation_id,
s.author_user_id,
t.user_id,
t.name,
t.tags,
s.created_at,
s.updated_at,
s.source,
s.config_id,
s.queue_id,
t.user_id,
t.name as trace_name,
t.tags as trace_tags
`,
...props,
});
return rows.map((row) => ({
projectId: row.project_id,
authorUserId: row.author_user_id,
traceId: row.trace_id,
observationId: row.observation_id,
traceUserId: row.user_id,
traceName: row.trace_name,
traceTags: row.trace_tags,
configId: row.config_id,
queueId: row.queue_id,
createdAt: parseClickhouseUTCDateTimeFormat(row.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(row.updated_at),
stringValue: row.string_value,
comment: row.comment,
dataType: row.data_type as ScoreDataType,
source: row.source as ScoreSource,
name: row.name,
value: row.value,
timestamp: parseClickhouseUTCDateTimeFormat(row.timestamp),
id: row.id,
}));
};
export const getScoresUiGeneric = async <T>(props: {
select: string;
projectId: string;
filter: FilterState;
orderBy: OrderByState;
limit?: number;
offset?: number;
}): Promise<T[]> => {
const { select, projectId, filter, orderBy, limit, offset } = props;
const { tracesFilter, scoresFilter, observationsFilter } =
getProjectIdDefaultFilter(projectId, { tracesPrefix: "t" });
scoresFilter.push(
...createFilterFromFilterState(filter, scoresTableUiColumnDefinitions),
);
const scoresFilterRes = scoresFilter.apply();
// TODO: Can we realistically apply a traces time filter here? Is an order by event_ts a risk?
const query = `
SELECT
${select}
FROM scores s final
LEFT JOIN traces t
ON s.trace_id = t.id
AND t.project_id = s.project_id
WHERE s.project_id = {projectId: String}
${scoresFilterRes?.query ? `AND ${scoresFilterRes.query}` : ""}
${orderByToClickhouseSql(orderBy ?? null, scoresTableUiColumnDefinitions)}
${limit !== undefined && offset !== undefined ? `limit {limit: Int32} offset {offset: Int32}` : ""}
`;
const rows = await queryClickhouse<T>({
query: query,
params: {
projectId: projectId,
...(scoresFilterRes ? scoresFilterRes.params : {}),
limit: limit,
offset: offset,
},
});
return rows;
};
export const getScoreNames = async (
projectId: string,
timestampFilter: FilterState,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(
timestampFilter,
scoresTableUiColumnDefinitions,
),
);
const timestampFilterRes = chFilter.apply();
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
name,
count(*) as count
from scores s
WHERE s.project_id = {projectId: String}
${timestampFilterRes?.query ? `AND ${timestampFilterRes.query}` : ""}
GROUP BY name
ORDER BY count() desc
LIMIT 1000;
`;
const rows = await queryClickhouse<{
name: string;
count: string;
}>({
query: query,
params: {
projectId: projectId,
...(timestampFilterRes ? timestampFilterRes.params : {}),
},
});
return rows.map((row) => ({
name: row.name,
count: Number(row.count),
}));
};
export const deleteScore = async (projectId: string, scoreId: string) => {
const query = `
DELETE FROM scores
WHERE project_id = {projectId: String}
AND id = {scoreId: String};
`;
await commandClickhouse({
query: query,
params: {
projectId,
scoreId,
},
});
};
export const deleteScoresByTraceIds = async (
projectId: string,
traceIds: string[],
) => {
const query = `
DELETE FROM scores
WHERE project_id = {projectId: String}
AND trace_id IN ({traceIds: Array(String)});
`;
await commandClickhouse({
query: query,
params: {
projectId,
traceIds,
},
});
};
export const getNumericScoreHistogram = async (
projectId: string,
filter: FilterState,
limit: number,
) => {
const chFilter = new FilterList(
createFilterFromFilterState(filter, dashboardColumnDefinitions),
);
const chFilterRes = chFilter.apply();
const query = `
select s.value
from scores s
WHERE s.project_id = {projectId: String}
${chFilterRes?.query ? `AND ${chFilterRes.query}` : ""}
ORDER BY s.event_ts DESC
LIMIT 1 BY s.id, s.project_id
${limit !== undefined ? `limit {limit: Int32}` : ""}
`;
return queryClickhouse<{ value: number }>({
query,
params: {
projectId,
limit,
...(chFilterRes ? chFilterRes.params : {}),
},
});
};
export const getAggregatedScoresForPrompts = async (
projectId: string,
promptIds: string[],
fetchScoreRelation: "observation" | "trace",
) => {
const query = `
SELECT
prompt_id,
s.id,
s.name,
s.string_value,
s.value,
s.source,
s.data_type,
s.comment
FROM scores s FINAL LEFT JOIN observations o FINAL
ON o.trace_id = s.trace_id
AND o.project_id = s.project_id
${fetchScoreRelation === "observation" ? "AND o.id = s.observation_id" : ""}
WHERE o.project_id = {projectId: String}
AND s.project_id = {projectId: String}
AND o.prompt_id IN ({promptIds: Array(String)})
AND o.type = 'GENERATION'
AND s.name IS NOT NULL
${fetchScoreRelation === "trace" ? "AND s.observation_id IS NULL" : ""}
`;
const rows = await queryClickhouse<ScoreAggregation & { prompt_id: string }>({
query,
params: {
projectId,
promptIds,
},
});
return rows.map((row) => ({
...convertScoreAggregation(row),
promptId: row.prompt_id,
}));
};
@@ -0,0 +1,45 @@
import { ScoreSource, ScoreDataType } from "@prisma/client";
import { ScoreRecordReadType } from "./definitions";
export type ScoreAggregation = {
id: string;
name: string;
string_value: string | null;
value: string;
source: string;
data_type: string;
comment: string | null;
};
export const convertToScore = (row: ScoreRecordReadType) => {
return {
id: row.id,
timestamp: new Date(row.timestamp),
projectId: row.project_id,
traceId: row.trace_id,
observationId: row.observation_id ?? null,
name: row.name,
value: row.value ?? null,
source: row.source as ScoreSource,
comment: row.comment ?? null,
authorUserId: row.author_user_id ?? null,
configId: row.config_id ?? null,
dataType: row.data_type as ScoreDataType,
stringValue: row.string_value ?? null,
queueId: row.queue_id ?? null,
createdAt: new Date(row.created_at),
updatedAt: new Date(row.updated_at),
};
};
export const convertScoreAggregation = (row: ScoreAggregation) => {
return {
id: row.id,
name: row.name,
stringValue: row.string_value,
value: Number(row.value),
source: row.source as ScoreSource,
dataType: row.data_type as ScoreDataType,
comment: row.comment,
};
};
+602 -246
View File
@@ -1,296 +1,652 @@
import {
commandClickhouse,
parseClickhouseUTCDateTimeFormat,
queryClickhouse,
upsertClickhouse,
} from "./clickhouse";
import {
createFilterFromFilterState,
getProjectIdDefaultFilter,
} from "../queries/clickhouse-filter/factory";
import { ObservationLevel, Trace } from "@prisma/client";
} from "../queries/clickhouse-sql/factory";
import { FilterState } from "../../types";
import { logger } from "../logger";
import { FilterList } from "../queries/clickhouse-filter/clickhouse-filter";
import {
DateTimeFilter,
FilterList,
StringFilter,
} from "../queries/clickhouse-sql/clickhouse-filter";
import { TraceRecordReadType } from "./definitions";
import { tracesTableUiColumnDefinitions } from "../../tableDefinitions/mapTracesTable";
import { TableCount } from "./types";
import { OrderByState } from "../../interfaces/orderBy";
import { orderByToClickhouseSql } from "../queries/clickhouse-filter/orderby-factory";
import { orderByToClickhouseSql } from "../queries/clickhouse-sql/orderby-factory";
import { UiColumnMapping } from "../../tableDefinitions";
import { sessionCols } from "../../tableDefinitions/mapSessionTable";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { convertClickhouseToDomain } from "./traces_converters";
import { clickhouseSearchCondition } from "../queries/clickhouse-sql/search";
import { TRACE_TO_OBSERVATIONS_INTERVAL } from "./constants";
import { FetchTracesTableProps } from "../services/traces-ui-table-service";
const convertClickhouseToDomain = (record: TraceRecordReadType): Trace => {
return {
id: record.id,
projectId: record.project_id,
name: record.name ?? null,
timestamp: parseClickhouseUTCDateTimeFormat(record.timestamp),
tags: record.tags,
bookmarked: record.bookmarked,
release: record.release ?? null,
version: record.version ?? null,
userId: record.user_id ?? null,
sessionId: record.session_id ?? null,
public: record.public,
input: record.input ?? null,
output: record.output ?? null,
metadata: record.metadata,
createdAt: parseClickhouseUTCDateTimeFormat(record.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(record.updated_at),
externalId: null,
};
};
export type TracesTableReturnType = Pick<
TraceRecordReadType,
| "project_id"
| "id"
| "name"
| "timestamp"
| "bookmarked"
| "release"
| "version"
| "user_id"
| "session_id"
| "tags"
| "metadata"
| "public"
> & {
level: ObservationLevel;
observation_count: number | null;
latency: string | null;
usage_details: Record<string, number>;
cost_details: Record<string, number>;
scores_avg: Array<{ name: string; avg_value: number }>;
};
export const getTracesTableCount = async (
export const checkTraceExists = async (
projectId: string,
traceId: string,
timestamp: Date | undefined,
filter: FilterState,
orderBy?: OrderByState,
limit?: number,
offset?: number,
) =>
getTracesTableGeneric<TableCount>({
select: "count(*) as count",
projectId,
filter,
orderBy,
limit,
offset,
): Promise<boolean> => {
const { tracesFilter } = getProjectIdDefaultFilter(projectId, {
tracesPrefix: "t",
});
export const getTracesTable = async (
tracesFilter.push(
...createFilterFromFilterState(filter, tracesTableUiColumnDefinitions),
new StringFilter({
clickhouseTable: "t",
field: "id",
operator: "=",
value: traceId,
}),
);
const tracesFilterRes = tracesFilter.apply();
const query = `
SELECT id, project_id
FROM traces t FINAL
WHERE ${tracesFilterRes.query}
${timestamp ? `AND timestamp >= {timestamp: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
`;
const rows = await queryClickhouse<{ id: string; project_id: string }>({
query,
params: {
...tracesFilterRes.params,
...(timestamp
? { timestamp: convertDateToClickhouseDateTime(timestamp) }
: {}),
},
});
return rows.length > 0;
};
/**
* Accepts a trace in a Clickhouse-ready format.
* id, project_id, and timestamp must always be provided.
*/
export const upsertTrace = async (trace: Partial<TraceRecordReadType>) => {
if (!["id", "project_id", "timestamp"].every((key) => key in trace)) {
throw new Error("Identifier fields must be provided to upsert Trace.");
}
await upsertClickhouse({
table: "traces",
records: [trace as TraceRecordReadType],
eventBodyMapper: convertClickhouseToDomain,
});
};
export const getTracesByIds = async (
traceIds: string[],
projectId: string,
filter: FilterState,
orderBy?: OrderByState,
limit?: number,
offset?: number,
timestamp?: Date,
) => {
const rows = await getTracesTableGeneric<TracesTableReturnType>({
select: `
t.id,
t.project_id,
t.timestamp,
t.tags,
t.bookmarked,
t.name,
t.release,
t.version,
t.user_id,
t.session_id,
os.latencyMs as latency,
os.cost_details as cost_details,
os.usage_details as usage_details,
os.level as level,
os.observation_count as observation_count,
s.scores_avg as scores_avg,
t.metadata,
t.public`,
projectId,
filter,
orderBy,
limit,
offset,
const query = `
SELECT *
FROM traces
WHERE id IN ({traceIds: Array(String)})
AND project_id = {projectId: String}
${timestamp ? `AND timestamp >= {timestamp: DateTime64(3)}` : ""}
ORDER BY event_ts DESC
LIMIT 1 by id, project_id;`;
const records = await queryClickhouse<TraceRecordReadType>({
query,
params: {
traceIds,
projectId,
timestamp: timestamp ? convertDateToClickhouseDateTime(timestamp) : null,
},
});
return records.map(convertClickhouseToDomain);
};
export const getTracesBySessionId = async (
projectId: string,
sessionIds: string[],
timestamp?: Date,
) => {
const query = `
SELECT *
FROM traces
WHERE session_id IN ({sessionIds: Array(String)})
AND project_id = {projectId: String}
${timestamp ? `AND timestamp >= {timestamp: DateTime64(3)}` : ""}
ORDER BY event_ts DESC
LIMIT 1 by id, project_id;`;
const records = await queryClickhouse<TraceRecordReadType>({
query,
params: {
sessionIds,
projectId,
timestamp: timestamp ? convertDateToClickhouseDateTime(timestamp) : null,
},
});
return records.map(convertClickhouseToDomain);
};
export const hasAnyTrace = async (projectId: string) => {
const query = `
SELECT count(*) as count
FROM traces
WHERE project_id = {projectId: String}
LIMIT 1
`;
const rows = await queryClickhouse<{ count: string }>({
query,
params: {
projectId,
},
});
return rows.length > 0 && Number(rows[0].count) > 0;
};
export const getTraceById = async (
traceId: string,
projectId: string,
timestamp?: Date,
) => {
const query = `
SELECT *
FROM traces
WHERE id = {traceId: String}
AND project_id = {projectId: String}
${timestamp ? `AND toDate(timestamp) = toDate({timestamp: DateTime64(3)})` : ""}
ORDER BY event_ts DESC
LIMIT 1
`;
const records = await queryClickhouse<TraceRecordReadType>({
query,
params: {
traceId,
projectId,
...(timestamp
? { timestamp: convertDateToClickhouseDateTime(timestamp) }
: {}),
},
});
const res = records.map(convertClickhouseToDomain);
return res.shift();
};
export const getTracesGroupedByName = async (
projectId: string,
tableDefinitions: UiColumnMapping[] = tracesTableUiColumnDefinitions,
timestampFilter?: FilterState,
) => {
const chFilter = timestampFilter
? createFilterFromFilterState(timestampFilter, tableDefinitions)
: undefined;
const timestampFilterRes = chFilter
? new FilterList(chFilter).apply()
: undefined;
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
name as name,
count(*) as count
from traces t
WHERE t.project_id = {projectId: String}
AND t.name IS NOT NULL
${timestampFilterRes?.query ? `AND ${timestampFilterRes.query}` : ""}
GROUP BY name
ORDER BY count(*) desc
LIMIT 1000;
`;
const rows = await queryClickhouse<{
name: string;
count: string;
}>({
query: query,
params: {
projectId: projectId,
...(timestampFilterRes ? timestampFilterRes.params : {}),
},
});
return rows;
};
type FetchTracesTableProps = {
select: string;
export const getTracesGroupedByUsers = async (
projectId: string,
filter: FilterState,
searchQuery?: string,
limit?: number,
offset?: number,
columns?: UiColumnMapping[],
) => {
const { tracesFilter } = getProjectIdDefaultFilter(projectId, {
tracesPrefix: "t",
});
tracesFilter.push(
...createFilterFromFilterState(
filter,
columns ?? tracesTableUiColumnDefinitions,
),
);
const tracesFilterRes = tracesFilter.apply();
const search = clickhouseSearchCondition(searchQuery);
// We mainly use queries like this to retrieve filter options.
// Therefore, we can skip final as some inaccuracy in count is acceptable.
const query = `
select
user_id as user,
count(*) as count
from traces t
WHERE t.project_id = {projectId: String}
AND t.user_id IS NOT NULL
AND t.user_id != ''
${tracesFilterRes?.query ? `AND ${tracesFilterRes.query}` : ""}
${search.query}
GROUP BY user
ORDER BY count desc
${limit !== undefined && offset !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const rows = await queryClickhouse<{
user: string;
count: string;
}>({
query: query,
params: {
limit,
offset,
projectId,
...(tracesFilterRes ? tracesFilterRes.params : {}),
...(searchQuery ? search.params : {}),
},
});
return rows;
};
export type GroupedTracesQueryProp = {
projectId: string;
filter: FilterState;
columns?: UiColumnMapping[];
};
export const getTracesGroupedByTags = async (props: GroupedTracesQueryProp) => {
const { projectId, filter, columns } = props;
const chFilter = createFilterFromFilterState(
filter,
columns ?? tracesTableUiColumnDefinitions,
);
const filterRes = new FilterList(chFilter).apply();
const query = `
select distinct(arrayJoin(tags)) as value
from traces t
WHERE t.project_id = {projectId: String}
${filterRes?.query ? `AND ${filterRes.query}` : ""}
LIMIT 1000;
`;
const rows = await queryClickhouse<{
value: string;
}>({
query: query,
params: {
projectId: projectId,
...(filterRes ? filterRes.params : {}),
},
});
return rows;
};
export type SessionDataReturnType = {
session_id: string;
max_timestamp: string;
min_timestamp: string;
trace_ids: string[];
user_ids: string[];
trace_count: number;
trace_tags: string[];
total_observations: number;
duration: number;
session_usage_details: Record<string, number>;
session_cost_details: Record<string, number>;
session_input_cost: string;
session_output_cost: string;
session_total_cost: string;
session_input_usage: string;
session_output_usage: string;
session_total_usage: string;
};
export const getSessionsTableCount = async (props: {
projectId: string;
filter: FilterState;
orderBy?: OrderByState;
limit?: number;
offset?: number;
page?: number;
}) => {
const rows = await getSessionsTableGeneric<{ count: string }>({
select: `
count(session_id) as count
`,
projectId: props.projectId,
filter: props.filter,
orderBy: props.orderBy,
limit: props.limit,
page: props.page,
});
return rows.length > 0 ? Number(rows[0].count) : 0;
};
const getTracesTableGeneric = async <T>(props: FetchTracesTableProps) => {
const { select, projectId, filter, orderBy, limit, offset } = props;
logger.info(`input filter ${JSON.stringify(filter)}`);
export const getSessionsTable = async (props: {
projectId: string;
filter: FilterState;
orderBy?: OrderByState;
limit?: number;
page?: number;
}) => {
const rows = await getSessionsTableGeneric<SessionDataReturnType>({
select: `
session_id,
max_timestamp,
min_timestamp,
trace_ids,
user_ids,
trace_count,
trace_tags,
total_observations,
duration,
session_usage_details,
session_cost_details,
session_input_cost,
session_output_cost,
session_total_cost,
session_input_usage,
session_output_usage,
session_total_usage
`,
projectId: props.projectId,
filter: props.filter,
orderBy: props.orderBy,
limit: props.limit,
page: props.page,
});
return rows;
};
const getSessionsTableGeneric = async <T>(props: FetchTracesTableProps) => {
const { select, projectId, filter, orderBy, limit, page } = props;
const { tracesFilter, scoresFilter, observationsFilter } =
getProjectIdDefaultFilter(projectId, { tracesPrefix: "t" });
getProjectIdDefaultFilter(projectId, { tracesPrefix: "s" });
tracesFilter.push(...createFilterFromFilterState(filter, sessionCols));
const tracesFilterRes = tracesFilter.apply();
const scoresAvgFilterRes = scoresFilter.apply();
const observationsStatsRes = observationsFilter.apply();
const traceTimestampFilter: DateTimeFilter | undefined = tracesFilter.find(
(f) =>
f.field === "min_timestamp" &&
(f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined;
const singleTraceFilter = traceTimestampFilter
? new FilterList([
new DateTimeFilter({
clickhouseTable: "traces",
field: "timestamp",
operator: traceTimestampFilter.operator,
value: traceTimestampFilter.value,
}),
]).apply()
: undefined;
const query = `
WITH observations_agg AS (
SELECT o.trace_id,
count(*) as obs_count,
min(o.start_time) as min_start_time,
max(o.end_time) as max_end_time,
sumMap(usage_details) as sum_usage_details,
sumMap(cost_details) as sum_cost_details,
anyLast(project_id) as project_id
FROM observations o FINAL
WHERE o.project_id = {projectId: String}
${traceTimestampFilter ? `AND o.start_time >= {observationsStartTime: DateTime64(3)} - ${TRACE_TO_OBSERVATIONS_INTERVAL}` : ""}
GROUP BY o.trace_id
),
session_data AS (
SELECT
t.session_id,
anyLast(t.project_id) as project_id,
max(t.timestamp) as max_timestamp,
min(t.timestamp) as min_timestamp,
groupArray(t.id) AS trace_ids,
groupUniqArray(t.user_id) AS user_ids,
count(*) as trace_count,
groupUniqArrayArray(t.tags) as trace_tags,
-- Aggregate observations data at session level
sum(o.obs_count) as total_observations,
date_diff('milliseconds', min(min_start_time), max(max_end_time)) as duration,
sumMap(o.sum_usage_details) as session_usage_details,
sumMap(o.sum_cost_details) as session_cost_details,
sumMap(o.sum_cost_details)['input'] as session_input_cost,
sumMap(o.sum_cost_details)['output'] as session_output_cost,
sumMap(o.sum_cost_details)['total'] as session_total_cost,
sumMap(o.sum_usage_details)['input'] as session_input_usage,
sumMap(o.sum_usage_details)['output'] as session_output_usage,
sumMap(o.sum_usage_details)['total'] as session_total_usage
FROM traces t FINAL
LEFT JOIN observations_agg o
ON t.id = o.trace_id AND t.project_id = o.project_id
WHERE t.session_id IS NOT NULL
AND t.project_id = {projectId: String}
${singleTraceFilter?.query ? ` AND ${singleTraceFilter.query}` : ""}
GROUP BY t.session_id
)
SELECT ${select}
FROM session_data s
WHERE ${tracesFilterRes.query ? tracesFilterRes.query : ""}
${orderByToClickhouseSql(orderBy ?? null, sessionCols)}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const obsStartTimeValue = traceTimestampFilter
? convertDateToClickhouseDateTime(traceTimestampFilter.value)
: null;
const res = await queryClickhouse<T>({
query: query,
params: {
projectId,
limit: limit,
offset: limit && page ? limit * page : 0,
...tracesFilterRes.params,
...observationsStatsRes.params,
...scoresAvgFilterRes.params,
...singleTraceFilter?.params,
...(obsStartTimeValue
? { observationsStartTime: obsStartTimeValue }
: {}),
},
});
return res;
};
export const getTracesIdentifierForSession = async (
projectId: string,
sessionId: string,
) => {
const query = `
SELECT
id,
user_id,
name,
timestamp,
project_id
FROM traces
WHERE (project_id = {projectId: String})
AND (session_id = {sessionId: String})
ORDER BY timestamp ASC
LIMIT 1 BY id, project_id;
`;
const rows = await queryClickhouse<{
id: string;
user_id: string;
name: string;
timestamp: string;
}>({
query: query,
params: {
projectId,
sessionId,
},
});
return rows.map((row) => ({
id: row.id,
userId: row.user_id,
name: row.name,
timestamp: parseClickhouseUTCDateTimeFormat(row.timestamp),
}));
};
export const deleteTraces = async (projectId: string, traceIds: string[]) => {
const query = `
DELETE FROM traces
WHERE project_id = {projectId: String}
AND id IN ({traceIds: Array(String)});
`;
await commandClickhouse({
query: query,
params: {
projectId,
traceIds,
},
});
};
export const getTotalUserCount = async (
projectId: string,
filter: FilterState,
searchQuery?: string,
): Promise<{ totalCount: bigint }[]> => {
const { tracesFilter } = getProjectIdDefaultFilter(projectId, {
tracesPrefix: "t",
});
tracesFilter.push(
...createFilterFromFilterState(filter, tracesTableUiColumnDefinitions),
);
const tracesFilterRes = tracesFilter.apply();
const scoresAvgFilterRes = scoresFilter.apply();
const observationsStatsRes = observationsFilter.apply();
const search = clickhouseSearchCondition(searchQuery);
const query = `
WITH observations_stats AS (
SELECT
COUNT(*) AS observation_count,
sumMap(usage_details) as usage_details,
SUM(total_cost) AS total_cost,
date_diff('seconds', least(min(start_time), min(end_time)), greatest(max(start_time), max(end_time))) as latencyMs,
multiIf(
arrayExists(x -> x = 'ERROR', groupArray(level)), 'ERROR',
arrayExists(x -> x = 'WARNING', groupArray(level)), 'WARNING',
arrayExists(x -> x = 'DEFAULT', groupArray(level)), 'DEFAULT',
'DEBUG'
) AS level,
sumMap(cost_details) as cost_details,
trace_id,
project_id
FROM
observations final
WHERE ${observationsStatsRes.query}
group by trace_id, project_id
),
SELECT COUNT(DISTINCT t.user_id) AS totalCount
FROM traces t
WHERE ${tracesFilterRes.query}
${search.query}
AND t.user_id IS NOT NULL
AND t.user_id != ''
`;
scores_avg AS (SELECT project_id,
trace_id,
groupArray(tuple(name, avg_value)) AS "scores_avg"
FROM (
SELECT project_id,
trace_id,
name,
avg(value) avg_value
FROM scores final
WHERE ${scoresAvgFilterRes.query}
GROUP BY project_id,
trace_id,
name
) tmp
GROUP BY project_id,
trace_id)
select
${select}
from traces t final
left join observations_stats os on os.project_id = t.project_id and os.trace_id = t.id
left join scores_avg s on s.project_id = t.project_id and s.trace_id = t.id
WHERE ${tracesFilterRes.query}
${orderByToClickhouseSql(orderBy ?? null, tracesTableUiColumnDefinitions)}
${limit !== undefined && offset !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
return await queryClickhouse<T>({
query: query,
params: {
limit: limit,
offset: offset,
...tracesFilterRes.params,
...observationsStatsRes.params,
...scoresAvgFilterRes.params,
},
});
};
export const getTraceByIdOrThrow = async (
traceId: string,
projectId: string,
) => {
const query = `SELECT * FROM traces where id = {traceId: String} and project_id = {projectId: String} order by event_ts desc LIMIT 1 by id, project_id`;
const records = await queryClickhouse<TraceRecordReadType>({
return queryClickhouse({
query,
params: { traceId, projectId },
params: {
...tracesFilterRes.params,
...search.params,
},
});
};
const res = records.map(convertClickhouseToDomain);
if (res.length !== 1) {
const errorMessage = `Trace not found or multiple traces found for traceId: ${traceId}, projectId: ${projectId}`;
logger.error(errorMessage);
throw new Error(errorMessage);
export const getUserMetrics = async (projectId: string, userIds: string[]) => {
if (userIds.length === 0) {
return [];
}
return res[0] as Trace;
};
export const getTracesGroupedByName = async (
projectId: string,
timestampFilter?: FilterState,
) => {
const chFilter = timestampFilter
? createFilterFromFilterState(
timestampFilter,
tracesTableUiColumnDefinitions,
)
: undefined;
const timestampFilterRes = chFilter
? new FilterList(chFilter).apply()
: undefined;
const query = `
select
name as value
from traces t final
WHERE t.project_id = {projectId: String}
AND t.name IS NOT NULL
${timestampFilterRes?.query ? `AND ${timestampFilterRes.query}` : ""}
GROUP BY name
ORDER BY name desc
LIMIT 1000;
`;
WITH observations_agg AS (
SELECT o.trace_id,
count(*) as obs_count,
sumMap(usage_details) as sum_usage_details,
sum(total_cost) as sum_total_cost,
anyLast(project_id) as project_id
FROM observations o FINAL
WHERE o.project_id = {projectId: String}
GROUP BY o.trace_id
),
user_metric_data AS (
SELECT t.user_id,
max(t.timestamp) as max_timestamp,
min(t.timestamp) as min_timestamp,
count(*) as trace_count,
sum(o.obs_count) as total_observations,
sum(o.sum_total_cost) as session_total_cost,
sumMap(o.sum_usage_details)['input'] as session_input_usage,
sumMap(o.sum_usage_details)['output'] as session_output_usage,
sumMap(o.sum_usage_details)['total'] as session_total_usage
FROM traces t FINAL
LEFT JOIN observations_agg o
ON t.id = o.trace_id
AND t.project_id = o.project_id
WHERE t.user_id IS NOT NULL
AND t.user_id != ''
AND t.user_id IN ({userIds: Array(String)})
AND t.project_id = {projectId: String}
GROUP BY t.user_id
)
SELECT user_id AS userId,
min_timestamp as firstTrace,
max_timestamp as lastTrace,
trace_count as totalTraces,
total_observations as totalObservations,
session_input_usage as totalPromptTokens,
session_output_usage as totalCompletionTokens,
session_total_usage as totalTokens,
session_total_cost as sumCalculatedTotalCost
FROM user_metric_data umd
`;
const rows = await queryClickhouse<{
value: string;
return queryClickhouse<{
userId: string;
firstTrace: Date | null;
lastTrace: Date | null;
totalPromptTokens: bigint;
totalCompletionTokens: bigint;
totalTokens: bigint;
totalObservations: bigint;
totalTraces: bigint;
sumCalculatedTotalCost: number;
}>({
query: query,
query,
params: {
projectId: projectId,
...(timestampFilterRes ? timestampFilterRes.params : {}),
projectId,
userIds,
},
});
return rows;
};
export const getTracesGroupedByTags = async (
projectId: string,
timestampFilter?: FilterState,
) => {
const chFilter = timestampFilter
? createFilterFromFilterState(
timestampFilter,
tracesTableUiColumnDefinitions,
)
: undefined;
const timestampFilterRes = chFilter
? new FilterList(chFilter).apply()
: undefined;
const query = `
select
distinct(arrayJoin(tags)) as value
from traces t final
WHERE t.project_id = {projectId: String}
${timestampFilterRes?.query ? `AND ${timestampFilterRes.query}` : ""}
LIMIT 1000;
`;
const rows = await queryClickhouse<{
value: string;
}>({
query: query,
params: {
projectId: projectId,
...(timestampFilterRes ? timestampFilterRes.params : {}),
},
});
return rows;
};
@@ -0,0 +1,106 @@
import { ObservationLevel, Trace } from "@prisma/client";
import { parseClickhouseUTCDateTimeFormat } from "./clickhouse";
import { TraceRecordReadType } from "./definitions";
import Decimal from "decimal.js";
import { ScoreAggregate } from "../../features/scores";
import { convertDateToClickhouseDateTime } from "../clickhouse/client";
import { TracesTableReturnType } from "../services/traces-ui-table-service";
export const convertTraceDomainToClickhouse = (
trace: Trace,
): TraceRecordReadType => {
return {
id: trace.id,
timestamp: convertDateToClickhouseDateTime(trace.timestamp),
name: trace.name,
user_id: trace.userId,
metadata: trace.metadata as Record<string, string>,
release: trace.release,
version: trace.version,
project_id: trace.projectId,
public: trace.public,
bookmarked: trace.bookmarked,
tags: trace.tags,
input: trace.input as string,
output: trace.output as string,
session_id: trace.sessionId,
created_at: convertDateToClickhouseDateTime(trace.createdAt),
updated_at: convertDateToClickhouseDateTime(trace.updatedAt),
event_ts: convertDateToClickhouseDateTime(new Date()),
is_deleted: 0,
};
};
export const convertClickhouseToDomain = (
record: TraceRecordReadType,
): Trace => {
return {
id: record.id,
projectId: record.project_id,
name: record.name ?? null,
timestamp: parseClickhouseUTCDateTimeFormat(record.timestamp),
tags: record.tags,
bookmarked: record.bookmarked,
release: record.release ?? null,
version: record.version ?? null,
userId: record.user_id ?? null,
sessionId: record.session_id ?? null,
public: record.public,
input: record.input ?? null,
output: record.output ?? null,
metadata: record.metadata,
createdAt: parseClickhouseUTCDateTimeFormat(record.created_at),
updatedAt: parseClickhouseUTCDateTimeFormat(record.updated_at),
externalId: null,
};
};
export type TracesAllReturnType = {
id: string;
timestamp: Date;
name: string | null;
projectId: string;
userId: string | null;
release: string | null;
version: string | null;
public: boolean;
bookmarked: boolean;
sessionId: string | null;
tags: string[];
};
export const convertToDomain = (row: TracesTableReturnType) => {
return {
id: row.id,
projectId: row.project_id,
timestamp: parseClickhouseUTCDateTimeFormat(row.timestamp),
tags: row.tags,
bookmarked: row.bookmarked,
name: row.name ?? null,
release: row.release ?? null,
version: row.version ?? null,
userId: row.user_id ?? null,
sessionId: row.session_id ?? null,
latency: Number(row.latency),
usageDetails: row.usage_details,
costDetails: row.cost_details,
level: row.level,
observationCount: Number(row.observation_count),
scoresAvg: row.scores_avg,
public: row.public,
};
};
export type TracesMetricsReturnType = {
id: string;
promptTokens: bigint;
completionTokens: bigint;
totalTokens: bigint;
latency: number | null;
level: ObservationLevel;
observationCount: bigint;
calculatedTotalCost: Decimal | null;
calculatedInputCost: Decimal | null;
calculatedOutputCost: Decimal | null;
scores: ScoreAggregate;
};
@@ -30,7 +30,7 @@ export class PromptService {
);
if (cachedPrompt) {
this.logInfo("Returning cached prompt for params", params);
this.logDebug("Returning cached prompt for params", params);
return cachedPrompt;
}
@@ -223,6 +223,10 @@ export class PromptService {
logger.info(`[PromptService] ${message}`, ...args);
}
private logDebug(message: string, ...args: any[]) {
logger.debug(`[PromptService] ${message}`, ...args);
}
private incrementMetric(name: Metrics, value: number = 1) {
try {
this.metricIncrementer?.(name, value);
@@ -1,143 +0,0 @@
import type { Readable } from "stream";
import {
GetObjectCommand,
ListObjectsV2Command,
PutObjectCommand,
S3Client,
} from "@aws-sdk/client-s3";
import { Upload } from "@aws-sdk/lib-storage";
import { getSignedUrl } from "@aws-sdk/s3-request-presigner";
import { logger } from "../logger";
type UploadFile = {
fileName: string;
fileType: string;
data: Readable | string;
expiresInSeconds: number;
};
export class S3StorageService {
private client: S3Client;
private bucketName: string;
constructor(params: {
accessKeyId: string | undefined;
secretAccessKey: string | undefined;
bucketName: string;
endpoint: string | undefined;
region: string | undefined;
forcePathStyle: boolean;
}) {
// Use accessKeyId and secretAccessKey if provided or fallback to default credentials
const { accessKeyId, secretAccessKey } = params;
const credentials =
accessKeyId !== undefined && secretAccessKey !== undefined
? {
accessKeyId,
secretAccessKey,
}
: undefined;
this.client = new S3Client({
credentials,
endpoint: params.endpoint,
region: params.region,
forcePathStyle: params.forcePathStyle,
});
this.bucketName = params.bucketName;
}
public async uploadFile({
fileName,
fileType,
data,
expiresInSeconds,
}: UploadFile): Promise<{ signedUrl: string }> {
try {
await new Upload({
client: this.client,
params: {
Bucket: this.bucketName,
Key: fileName,
Body: data,
ContentType: fileType,
},
}).done();
const signedUrl = await this.getSignedUrl(fileName, expiresInSeconds);
return { signedUrl };
} catch (err) {
logger.error(`Failed to upload file to ${fileName}`, err);
throw new Error("Failed to upload to S3 or generate signed URL");
}
}
public async uploadJson(path: string, body: Record<string, unknown>[]) {
const putCommand = new PutObjectCommand({
Bucket: this.bucketName,
Key: path,
Body: JSON.stringify(body),
ContentType: "application/json",
});
try {
await this.client.send(putCommand);
} catch (err) {
logger.error(`Failed to upload JSON to S3 ${path}`, err);
throw Error("Failed to upload JSON to S3");
}
}
public async download(path: string): Promise<string> {
const getCommand = new GetObjectCommand({
Bucket: this.bucketName,
Key: path,
});
try {
const response = await this.client.send(getCommand);
return (await response.Body?.transformToString()) ?? "";
} catch (err) {
logger.error(`Failed to download file from S3 ${path}`, err);
throw Error("Failed to download file from S3");
}
}
public async listFiles(prefix: string): Promise<string[]> {
const listCommand = new ListObjectsV2Command({
Bucket: this.bucketName,
Prefix: prefix,
});
try {
const response = await this.client.send(listCommand);
return (
response.Contents?.flatMap((file) => (file.Key ? [file.Key] : [])) ?? []
);
} catch (err) {
logger.error(`Failed to list files from S3 ${prefix}`, err);
throw Error("Failed to list files from S3");
}
}
private async getSignedUrl(
fileName: string,
ttlSeconds: number,
): Promise<string> {
try {
return await getSignedUrl(
this.client,
new GetObjectCommand({
Bucket: this.bucketName,
Key: fileName,
ResponseContentDisposition: `attachment; filename="${fileName}"`,
}),
{ expiresIn: ttlSeconds },
);
} catch (err) {
logger.error(`Failed to generate presigned URL for ${fileName}`, err);
throw Error("Failed to generate signed URL");
}
}
}
@@ -0,0 +1,432 @@
import { Readable } from "stream";
import {
GetObjectCommand,
ListObjectsV2Command,
PutObjectCommand,
S3Client,
} from "@aws-sdk/client-s3";
import { Upload } from "@aws-sdk/lib-storage";
import { getSignedUrl } from "@aws-sdk/s3-request-presigner";
import {
BlobSASPermissions,
BlobServiceClient,
ContainerClient,
StorageSharedKeyCredential,
} from "@azure/storage-blob";
import { logger } from "../logger";
import { env } from "../../env";
type UploadFile = {
fileName: string;
fileType: string;
data: Readable | string;
expiresInSeconds: number;
};
export interface StorageService {
uploadFile(params: UploadFile): Promise<{ signedUrl: string }>;
uploadJson(path: string, body: Record<string, unknown>[]): Promise<void>;
download(path: string): Promise<string>;
listFiles(prefix: string): Promise<string[]>;
getSignedUrl(
fileName: string,
ttlSeconds: number,
asAttachment?: boolean,
): Promise<string>;
getSignedUploadUrl(params: {
path: string;
ttlSeconds: number;
sha256Hash: string;
contentType: string;
contentLength: number;
}): Promise<string>;
}
export class StorageServiceFactory {
public static getInstance(params: {
accessKeyId: string | undefined;
secretAccessKey: string | undefined;
bucketName: string;
endpoint: string | undefined;
region: string | undefined;
forcePathStyle: boolean;
}): StorageService {
if (env.LANGFUSE_USE_AZURE_BLOB === "true") {
return new AzureBlobStorageService(params);
}
return new S3StorageService(params);
}
}
class AzureBlobStorageService implements StorageService {
private client: ContainerClient;
private container: string;
constructor(params: {
accessKeyId: string | undefined;
secretAccessKey: string | undefined;
bucketName: string;
endpoint: string | undefined;
region: string | undefined;
forcePathStyle: boolean;
}) {
const { accessKeyId, secretAccessKey, endpoint } = params;
if (!accessKeyId || !secretAccessKey || !endpoint) {
throw new Error(
`Endpoint, account and account key must be configured to use Azure Blob Storage`,
);
}
const sharedKeyCredential = new StorageSharedKeyCredential(
accessKeyId,
secretAccessKey,
);
const blobServiceClient = new BlobServiceClient(
endpoint,
sharedKeyCredential,
);
this.container = params.bucketName;
this.client = blobServiceClient.getContainerClient(this.container);
}
private async createContainerIfNotExists(): Promise<void> {
try {
await this.client.createIfNotExists();
} catch (err) {
logger.error(
`Failed to create Azure Blob Storage container ${this.container}`,
err,
);
throw Error("Failed to create Azure Blob Storage container ");
}
}
public async uploadFile(params: UploadFile): Promise<{ signedUrl: string }> {
const { fileName, data, expiresInSeconds } = params;
try {
await this.createContainerIfNotExists();
const blockBlobClient = this.client.getBlockBlobClient(fileName);
if (typeof data === "string") {
await blockBlobClient.upload(data, data.length);
} else if (data instanceof Readable) {
let offset = 0;
const blockIds = [];
for await (const chunk of data) {
const blockId = Buffer.from(`block-${offset}`).toString("base64");
const bufferChunk = Buffer.isBuffer(chunk)
? chunk
: Buffer.from(chunk);
await blockBlobClient.stageBlock(
blockId,
bufferChunk,
bufferChunk.length,
);
blockIds.push(blockId);
offset += bufferChunk.length;
}
await blockBlobClient.commitBlockList(blockIds);
} else {
throw new Error("Unsupported data type. Must be Readable or string.");
}
return {
signedUrl: await this.getSignedUrl(fileName, expiresInSeconds, false),
};
} catch (err) {
logger.error(
`Failed to upload file to Azure Blob Storage ${fileName}`,
err,
);
throw Error("Failed to upload file to Azure Blob Storage");
}
}
public async uploadJson(
path: string,
body: Record<string, unknown>[],
): Promise<void> {
await this.createContainerIfNotExists();
const blockBlobClient = this.client.getBlockBlobClient(path);
const content = JSON.stringify(body);
try {
await blockBlobClient.upload(content, content.length);
} catch (err) {
logger.error(`Failed to upload JSON to Azure Blob Storage ${path}`, err);
throw Error("Failed to upload JSON to Azure Blob Storage");
}
}
private async streamToString(
readableStream: NodeJS.ReadableStream,
): Promise<string> {
return new Promise((resolve, reject) => {
const chunks: string[] = [];
readableStream.on("data", (data) => {
chunks.push(data.toString());
});
readableStream.on("end", () => {
resolve(chunks.join(""));
});
readableStream.on("error", reject);
});
}
public async download(path: string): Promise<string> {
try {
await this.createContainerIfNotExists();
const blobClient = this.client.getBlobClient(path);
const downloadResponse = await blobClient.download();
if (!downloadResponse.readableStreamBody) {
throw Error("No stream body available");
}
return this.streamToString(downloadResponse.readableStreamBody);
} catch (err) {
logger.error(
`Failed to download file from Azure Blob Storage ${path}`,
err,
);
throw Error("Failed to download file from Azure Blob Storage");
}
}
public async listFiles(prefix: string): Promise<string[]> {
try {
await this.createContainerIfNotExists();
const result = await this.client.listBlobsFlat({ prefix });
const files = [];
for await (const blob of result) {
if (blob.name.startsWith(prefix)) {
files.push(blob.name);
}
}
return files;
} catch (err) {
logger.error(
`Failed to list files from Azure Blob Storage ${prefix}`,
err,
);
throw Error("Failed to list files from Azure Blob Storage");
}
}
public async getSignedUrl(
fileName: string,
ttlSeconds: number,
asAttachment?: boolean,
): Promise<string> {
try {
await this.createContainerIfNotExists();
const blockBlobClient = this.client.getBlockBlobClient(fileName);
return blockBlobClient.generateSasUrl({
permissions: BlobSASPermissions.parse("r"),
expiresOn: new Date(Date.now() + ttlSeconds * 1000),
contentDisposition: asAttachment
? `attachment; filename="${fileName}"`
: undefined,
});
} catch (err) {
logger.error(
`Failed to generate presigned URL for Azure Blob Storage ${fileName}`,
err,
);
throw Error("Failed to generate presigned URL for Azure Blob Storage");
}
}
public async getSignedUploadUrl(params: {
path: string;
ttlSeconds: number;
sha256Hash: string;
contentType: string;
contentLength: number;
}): Promise<string> {
const { path, ttlSeconds, contentType } = params;
try {
await this.createContainerIfNotExists();
const blockBlobClient = this.client.getBlockBlobClient(path);
return blockBlobClient.generateSasUrl({
permissions: BlobSASPermissions.parse("w"),
expiresOn: new Date(Date.now() + ttlSeconds * 1000),
contentType: contentType,
});
} catch (err) {
logger.error(
`Failed to generate presigned upload URL for Azure Blob Storage ${path}`,
err,
);
throw Error(
"Failed to generate presigned upload URL for Azure Blob Storage",
);
}
}
}
class S3StorageService implements StorageService {
private client: S3Client;
private bucketName: string;
constructor(params: {
accessKeyId: string | undefined;
secretAccessKey: string | undefined;
bucketName: string;
endpoint: string | undefined;
region: string | undefined;
forcePathStyle: boolean;
}) {
// Use accessKeyId and secretAccessKey if provided or fallback to default credentials
const { accessKeyId, secretAccessKey } = params;
const credentials =
accessKeyId !== undefined && secretAccessKey !== undefined
? {
accessKeyId,
secretAccessKey,
}
: undefined;
this.client = new S3Client({
credentials,
endpoint: params.endpoint,
region: params.region,
forcePathStyle: params.forcePathStyle,
});
this.bucketName = params.bucketName;
}
public async uploadFile({
fileName,
fileType,
data,
expiresInSeconds,
}: UploadFile): Promise<{ signedUrl: string }> {
try {
await new Upload({
client: this.client,
params: {
Bucket: this.bucketName,
Key: fileName,
Body: data,
ContentType: fileType,
},
}).done();
const signedUrl = await this.getSignedUrl(fileName, expiresInSeconds);
return { signedUrl };
} catch (err) {
logger.error(`Failed to upload file to ${fileName}`, err);
throw new Error("Failed to upload to S3 or generate signed URL");
}
}
public async uploadJson(path: string, body: Record<string, unknown>[]) {
const putCommand = new PutObjectCommand({
Bucket: this.bucketName,
Key: path,
Body: JSON.stringify(body),
ContentType: "application/json",
});
try {
await this.client.send(putCommand);
} catch (err) {
logger.error(`Failed to upload JSON to S3 ${path}`, err);
throw Error("Failed to upload JSON to S3");
}
}
public async download(path: string): Promise<string> {
const getCommand = new GetObjectCommand({
Bucket: this.bucketName,
Key: path,
});
try {
const response = await this.client.send(getCommand);
return (await response.Body?.transformToString()) ?? "";
} catch (err) {
logger.error(`Failed to download file from S3 ${path}`, err);
throw Error("Failed to download file from S3");
}
}
public async listFiles(prefix: string): Promise<string[]> {
const listCommand = new ListObjectsV2Command({
Bucket: this.bucketName,
Prefix: prefix,
});
try {
const response = await this.client.send(listCommand);
return (
response.Contents?.flatMap((file) => (file.Key ? [file.Key] : [])) ?? []
);
} catch (err) {
logger.error(`Failed to list files from S3 ${prefix}`, err);
throw Error("Failed to list files from S3");
}
}
public async getSignedUrl(
fileName: string,
ttlSeconds: number,
asAttachment: boolean = true,
): Promise<string> {
try {
return await getSignedUrl(
this.client,
new GetObjectCommand({
Bucket: this.bucketName,
Key: fileName,
ResponseContentDisposition: asAttachment
? `attachment; filename="${fileName}"`
: undefined,
}),
{ expiresIn: ttlSeconds },
);
} catch (err) {
logger.error(`Failed to generate presigned URL for ${fileName}`, err);
throw Error("Failed to generate signed URL");
}
}
public async getSignedUploadUrl(params: {
path: string;
ttlSeconds: number;
sha256Hash: string;
contentType: string;
contentLength: number;
}): Promise<string> {
const { path, ttlSeconds, contentType, contentLength, sha256Hash } = params;
return await getSignedUrl(
this.client,
new PutObjectCommand({
Bucket: this.bucketName,
Key: path,
ContentType: contentType,
ChecksumSHA256: sha256Hash,
ContentLength: contentLength,
}),
{
expiresIn: ttlSeconds,
signableHeaders: new Set(["content-type", "content-length"]),
unhoistableHeaders: new Set(["x-amz-checksum-sha256"]),
},
);
}
}
@@ -0,0 +1,248 @@
import { ObservationLevel } from "@prisma/client";
import { OrderByState } from "../../interfaces/orderBy";
import { tracesTableUiColumnDefinitions } from "../../tableDefinitions";
import { FilterState } from "../../types";
import {
StringFilter,
StringOptionsFilter,
DateTimeFilter,
} from "../queries/clickhouse-sql/clickhouse-filter";
import {
getProjectIdDefaultFilter,
createFilterFromFilterState,
} from "../queries/clickhouse-sql/factory";
import { orderByToClickhouseSql } from "../queries/clickhouse-sql/orderby-factory";
import { clickhouseSearchCondition } from "../queries/clickhouse-sql/search";
import { convertToDomain } from "../repositories";
import { queryClickhouse } from "../repositories/clickhouse";
import { TraceRecordReadType } from "../repositories/definitions";
export type TracesTableReturnType = Pick<
TraceRecordReadType,
| "project_id"
| "id"
| "name"
| "timestamp"
| "bookmarked"
| "release"
| "version"
| "user_id"
| "session_id"
| "tags"
| "public"
> & {
level: ObservationLevel;
observation_count: number | null;
latency: string | null;
usage_details: Record<string, number>;
cost_details: Record<string, number>;
scores_avg: Array<{ name: string; avg_value: number }>;
};
export type FetchTracesTableProps = {
select: string;
projectId: string;
filter: FilterState;
searchQuery?: string;
orderBy?: OrderByState;
limit?: number;
page?: number;
};
export const getTracesTableCount = async (props: {
projectId: string;
filter: FilterState;
searchQuery?: string;
orderBy?: OrderByState;
limit?: number;
page?: number;
}) => {
const countRows = await getTracesTableGeneric<{ count: string }>({
select: "count(*) as count",
...props,
});
const converted = countRows.map((row) => ({
count: Number(row.count),
}));
return converted.length > 0 ? converted[0].count : 0;
};
export const getTracesTable = async (
projectId: string,
filter: FilterState,
searchQuery?: string,
orderBy?: OrderByState,
limit?: number,
page?: number,
) => {
const rows = await getTracesTableGeneric<TracesTableReturnType>({
select: `
t.id,
t.project_id as project_id,
t.timestamp,
t.tags,
t.bookmarked,
t.name,
t.release,
t.version,
t.user_id,
t.session_id,
os.latency_milliseconds / 1000 as latency,
os.cost_details as cost_details,
os.usage_details as usage_details,
os.level as level,
os.observation_count as observation_count,
s.scores_avg as scores_avg,
t.public`,
projectId,
filter,
searchQuery,
orderBy,
limit,
page,
});
return rows.map(convertToDomain);
};
const getTracesTableGeneric = async <T>(props: FetchTracesTableProps) => {
const { select, projectId, filter, orderBy, limit, page, searchQuery } =
props;
const { tracesFilter, scoresFilter, observationsFilter } =
getProjectIdDefaultFilter(projectId, { tracesPrefix: "t" });
tracesFilter.push(
...createFilterFromFilterState(filter, tracesTableUiColumnDefinitions),
);
const traceIdFilter = tracesFilter.find(
(f) => f.clickhouseTable === "traces" && f.field === "id",
) as StringFilter | StringOptionsFilter | undefined;
traceIdFilter
? scoresFilter.push(
new StringOptionsFilter({
clickhouseTable: "scores",
field: "trace_id",
operator: "any of",
values:
traceIdFilter instanceof StringFilter
? [traceIdFilter.value]
: traceIdFilter.values,
}),
)
: null;
traceIdFilter
? observationsFilter.push(
new StringOptionsFilter({
clickhouseTable: "observations",
field: "trace_id",
operator: "any of",
values:
traceIdFilter instanceof StringFilter
? [traceIdFilter.value]
: traceIdFilter.values,
}),
)
: null;
// for query optimisation, we have to add the timeseries filter to observations + scores as well
// stats show, that 98% of all observations have their start_time larger than trace.timestamp - 5 min
const timeStampFilter = tracesFilter.find(
(f) =>
f.field === "timestamp" && (f.operator === ">=" || f.operator === ">"),
) as DateTimeFilter | undefined;
timeStampFilter
? scoresFilter.push(
new DateTimeFilter({
clickhouseTable: "scores",
field: "timestamp",
operator: ">=",
value: timeStampFilter.value,
}),
)
: null;
timeStampFilter
? observationsFilter.push(
new DateTimeFilter({
clickhouseTable: "observations",
field: "start_time",
operator: ">=",
value: timeStampFilter.value,
}),
)
: null;
const tracesFilterRes = tracesFilter.apply();
const scoresFilterRes = scoresFilter.apply();
const observationFilterRes = observationsFilter.apply();
const search = clickhouseSearchCondition(searchQuery);
const query = `
WITH observations_stats AS (
SELECT
COUNT(*) AS observation_count,
sumMap(usage_details) as usage_details,
SUM(total_cost) AS total_cost,
date_diff('milliseconds', least(min(start_time), min(end_time)), greatest(max(start_time), max(end_time))) as latency_milliseconds,
multiIf(
arrayExists(x -> x = 'ERROR', groupArray(level)), 'ERROR',
arrayExists(x -> x = 'WARNING', groupArray(level)), 'WARNING',
arrayExists(x -> x = 'DEFAULT', groupArray(level)), 'DEFAULT',
'DEBUG'
) AS level,
sumMap(cost_details) as cost_details,
trace_id,
project_id
FROM observations FINAL
WHERE ${observationFilterRes.query}
GROUP BY trace_id, project_id
),
scores_avg AS (
SELECT
project_id,
trace_id,
groupArray(tuple(name, avg_value)) AS "scores_avg"
FROM (
SELECT project_id,
trace_id,
name,
avg(value) avg_value
FROM scores final
WHERE ${scoresFilterRes.query}
GROUP BY project_id,
trace_id,
name
) tmp
GROUP BY project_id, trace_id
)
SELECT ${select}
FROM traces t final
LEFT JOIN observations_stats os on os.project_id = t.project_id and os.trace_id = t.id
LEFT JOIN scores_avg s on s.project_id = t.project_id and s.trace_id = t.id
WHERE ${tracesFilterRes.query}
${search.query}
${orderByToClickhouseSql(orderBy ?? null, tracesTableUiColumnDefinitions)}
${limit !== undefined && page !== undefined ? `LIMIT {limit: Int32} OFFSET {offset: Int32}` : ""}
`;
const res = await queryClickhouse<T>({
query: query,
params: {
limit: limit,
offset: limit && page ? limit * page : 0,
...tracesFilterRes.params,
...observationFilterRes.params,
...scoresFilterRes.params,
...search.params,
},
});
return res;
};
@@ -2,3 +2,5 @@ export * from "./sessionsView";
export * from "./types";
export * from "./mapObservationsTable";
export * from "./mapTracesTable";
export * from "./mapDashboards";
export * from "./mapScoresTable";
@@ -0,0 +1,88 @@
import { UiColumnMapping } from "./types";
export const dashboardColumnDefinitions: UiColumnMapping[] = [
{
uiTableName: "Trace Name",
uiTableId: "traceName",
clickhouseTableName: "traces",
clickhouseSelect: 't."name"',
},
{
uiTableName: "Tags",
uiTableId: "traceTags",
clickhouseTableName: "traces",
clickhouseSelect: 't."tags"',
},
{
uiTableName: "Timestamp",
uiTableId: "timestamp",
clickhouseTableName: "traces",
clickhouseSelect: 't."timestamp"',
},
{
clickhouseTableName: "scores",
clickhouseSelect: "name",
uiTableId: "scoreName",
uiTableName: "Score Name",
},
{
clickhouseTableName: "scores",
clickhouseSelect: "timestamp",
uiTableId: "scoreTimestamp",
uiTableName: "Score Timestamp",
},
{
clickhouseTableName: "scores",
clickhouseSelect: "source",
uiTableId: "scoreSource",
uiTableName: "Score Source",
},
{
clickhouseTableName: "scores",
clickhouseSelect: "data_type",
uiTableId: "scoreDataType",
uiTableName: "Scores Data Type",
},
{
clickhouseTableName: "scores",
clickhouseSelect: "value",
uiTableId: "value",
uiTableName: "value",
},
{
clickhouseTableName: "observations",
clickhouseSelect: "o.start_time",
uiTableId: "startTime",
uiTableName: "Start Time",
},
{
clickhouseTableName: "observations",
clickhouseSelect: "o.end_time",
uiTableId: "endTime",
uiTableName: "End Time",
},
{
clickhouseTableName: "observations",
clickhouseSelect: "o.type",
uiTableId: "type",
uiTableName: "Type",
},
{
clickhouseTableName: "traces",
clickhouseSelect: "t.user_id",
uiTableId: "userId",
uiTableName: "User",
},
{
clickhouseTableName: "traces",
clickhouseSelect: "t.release",
uiTableId: "release",
uiTableName: "Release",
},
{
clickhouseTableName: "traces",
clickhouseSelect: "t.version",
uiTableId: "version",
uiTableName: "Version",
},
];

Some files were not shown because too many files have changed in this diff Show More