Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
daf4e2fe02 | ||
|
|
129693fb69 | ||
|
|
204948db44 | ||
|
|
d42ba5fc99 | ||
|
|
97a539ced3 | ||
|
|
9d8dace197 | ||
|
|
bc2dc4d89c | ||
|
|
2dd5c8a8aa | ||
|
|
d17b21f04d | ||
|
|
c5a0f44bdd | ||
|
|
011b4912f2 | ||
|
|
a4d773066f | ||
|
|
25bdb1690b | ||
|
|
f8565d7772 | ||
|
|
a912057082 | ||
|
|
e0d3270f50 | ||
|
|
99e5a0b224 | ||
|
|
bccdee5410 | ||
|
|
738cbbc8cd | ||
|
|
920c52bb06 | ||
|
|
4f134790af | ||
|
|
21b3ce3c82 | ||
|
|
0531b57e1a | ||
|
|
7b857a0dc4 | ||
|
|
6c97fe3c04 | ||
|
|
1d69cbd41b | ||
|
|
ab26692913 | ||
|
|
b791544864 | ||
|
|
b35583056b | ||
|
|
4504530e3d | ||
|
|
45eed4b5c0 | ||
|
|
9fa31ba68e | ||
|
|
ea35c25269 | ||
|
|
c427791070 | ||
|
|
25220aef55 | ||
|
|
8c02d5dc22 | ||
|
|
0b60076871 | ||
|
|
b28a83531b | ||
|
|
aa4d34ae66 | ||
|
|
cca352ad13 | ||
|
|
58c96cada7 | ||
|
|
6f93389936 | ||
|
|
b8d9586490 | ||
|
|
534f5696ad | ||
|
|
4cebe2831c | ||
|
|
45a5f3b8a2 | ||
|
|
8dea9b6a3a | ||
|
|
c5b781d3f6 | ||
|
|
07469c928d | ||
|
|
5b10110dfe | ||
|
|
0eb5d1c3c0 | ||
|
|
1585bf5e14 | ||
|
|
dec6d6f975 | ||
|
|
36fa7ea15e | ||
|
|
6a4d56a5a9 | ||
|
|
9f0a40949e | ||
|
|
10fdd9e172 | ||
|
|
fc58add111 | ||
|
|
10993838d5 | ||
|
|
148cd29e9f | ||
|
|
7a353e99af | ||
|
|
dc52106feb | ||
|
|
b2f8eb7078 | ||
|
|
026d8e2ba0 | ||
|
|
9a9f0b1743 | ||
|
|
74a510fbde | ||
|
|
c23b226e62 | ||
|
|
2f50b38a68 | ||
|
|
451ae15e00 | ||
|
|
e34d81578b | ||
|
|
214c9c7ed2 | ||
|
|
961b3bfa8d | ||
|
|
a1dd5b22a2 | ||
|
|
49950f9706 | ||
|
|
cfdd0fdb73 | ||
|
|
891e7e9716 | ||
|
|
5b407dad53 | ||
|
|
26e3bf9a44 | ||
|
|
1f02e364e1 | ||
|
|
90dca15d86 | ||
|
|
e3f8dc8bb3 | ||
|
|
585ede0919 | ||
|
|
0db425d120 | ||
|
|
09e33c3059 | ||
|
|
66226011af | ||
|
|
571698ab2e | ||
|
|
312066f735 | ||
|
|
5abeaf8adb | ||
|
|
fac3c732de | ||
|
|
5e2e3bb5fc | ||
|
|
0564df8e51 | ||
|
|
a2801a3be9 | ||
|
|
075eb58ecb | ||
|
|
674d66d179 | ||
|
|
163f2a02ff | ||
|
|
075836f210 | ||
|
|
543b6ee0f2 | ||
|
|
8c8c488e4d | ||
|
|
6a0e0a4221 | ||
|
|
1b01a267df | ||
|
|
69a9146894 | ||
|
|
380403e8ed | ||
|
|
296a6c3ee6 | ||
|
|
62904a9563 | ||
|
|
90247ab64c | ||
|
|
5cf91c60d9 | ||
|
|
559ba6d05d | ||
|
|
f071be69b6 | ||
|
|
52c261b422 | ||
|
|
6f0a43ec65 | ||
|
|
301bd6b569 | ||
|
|
56fd3df2d2 | ||
|
|
db433e2b72 | ||
|
|
e452004a6a | ||
|
|
bce026d8b2 | ||
|
|
fbdf12bfa3 | ||
|
|
78f59bc543 | ||
|
|
ea338ed83d | ||
|
|
8cf67fe329 | ||
|
|
e3b23ea5ec | ||
|
|
7be2791105 | ||
|
|
683aae0069 | ||
|
|
1c8ffc607d | ||
|
|
9c203e8b8a | ||
|
|
c32cccce52 | ||
|
|
f1f8da5b74 | ||
|
|
c23447a624 | ||
|
|
5e2225de46 | ||
|
|
bb9d118853 | ||
|
|
df57ff60d6 | ||
|
|
bae2c5a65d | ||
|
|
3efa696513 | ||
|
|
58ebf006b8 | ||
|
|
16743363dc | ||
|
|
ebf0e35073 | ||
|
|
2c4799340f | ||
|
|
5612c6a9a5 | ||
|
|
b62962ca08 | ||
|
|
3a46283bf2 | ||
|
|
07801180f8 | ||
|
|
14831902ef | ||
|
|
d377f02a49 | ||
|
|
e07120b163 | ||
|
|
506482dbe4 | ||
|
|
ec61f4a420 | ||
|
|
4db08a2960 | ||
|
|
2351d0d370 | ||
|
|
99b3401549 | ||
|
|
6fb796df40 | ||
|
|
846d6e6c37 | ||
|
|
1fa4fc6329 | ||
|
|
a1a48d5e7b | ||
|
|
fcb7563763 | ||
|
|
ae754b146d | ||
|
|
49343b9a0c | ||
|
|
76bf5c0f4e | ||
|
|
e91a29be61 | ||
|
|
81694363a2 | ||
|
|
5998266680 | ||
|
|
462e8e847d | ||
|
|
d7c186858b | ||
|
|
e686aedac9 | ||
|
|
85f75a5e8e | ||
|
|
1ea643300d | ||
|
|
596486c1ec | ||
|
|
cb65391c2d | ||
|
|
419e07260d | ||
|
|
825f030e81 | ||
|
|
34ded9c927 | ||
|
|
c935d4ae73 | ||
|
|
e7283ac06e | ||
|
|
20d0106626 | ||
|
|
849591fdf2 | ||
|
|
0c7b50e564 | ||
|
|
37b4f43351 | ||
|
|
7a3b0379e5 | ||
|
|
f458d7626c | ||
|
|
258dde4691 | ||
|
|
f15246d9df | ||
|
|
edb43a24ab | ||
|
|
c45ae1b140 | ||
|
|
90eeeeaf9d | ||
|
|
66accb563c | ||
|
|
9f2e8906d0 | ||
|
|
c5d61b7ed2 | ||
|
|
f95dd872e6 | ||
|
|
04339741a9 | ||
|
|
9c85e46996 | ||
|
|
4a7236451f | ||
|
|
5cb5204181 | ||
|
|
c19066afa0 | ||
|
|
bbb9dc285d | ||
|
|
b750acba06 | ||
|
|
bb06e079eb | ||
|
|
5f711780e0 | ||
|
|
0f58ea1ebf | ||
|
|
ec70ca3951 | ||
|
|
142d3a1612 | ||
|
|
1dee7092b3 | ||
|
|
83b64f5e77 | ||
|
|
48b05247c8 | ||
|
|
9732262466 | ||
|
|
e51e2df2dc | ||
|
|
74d1dfe5ac | ||
|
|
5633a8867e | ||
|
|
6a8bdae22e | ||
|
|
dbf994f9bb | ||
|
|
4d5c288d1e | ||
|
|
cf119fb9cc | ||
|
|
5af8c4e400 | ||
|
|
835a19bbd2 | ||
|
|
ae91a98ad9 | ||
|
|
e38cc24205 | ||
|
|
7ab45ec685 | ||
|
|
b101bf48d3 | ||
|
|
4c4f0ff3c0 | ||
|
|
a42ca96225 | ||
|
|
f63420827b | ||
|
|
e6fd056ae5 | ||
|
|
4708fe4f47 | ||
|
|
362246b616 | ||
|
|
b3de5fd71a | ||
|
|
ad236ecc99 | ||
|
|
beb2063dcb | ||
|
|
4de10298dc | ||
|
|
f9924975fa | ||
|
|
98774cbd61 | ||
|
|
80361ef53f | ||
|
|
8350507434 | ||
|
|
0a976edb70 | ||
|
|
6a67c93010 | ||
|
|
e51ee05d9c | ||
|
|
565c561715 | ||
|
|
d1338ad5e1 | ||
|
|
52d4fe37de | ||
|
|
67ef7abb63 | ||
|
|
727ae71a93 | ||
|
|
a8eabbc96d | ||
|
|
51befcad0f | ||
|
|
ee8b5e4998 | ||
|
|
9bb9f7fa39 | ||
|
|
4cc10840ab | ||
|
|
311cd900fb | ||
|
|
92f14fd297 | ||
|
|
42306e5247 | ||
|
|
5bb1b6de54 | ||
|
|
dae0ee6165 | ||
|
|
31fb304a5e | ||
|
|
d6dac50734 | ||
|
|
0cda8e2e21 | ||
|
|
3f4fd2ff5e | ||
|
|
17f0a10b23 | ||
|
|
6c80b5b4a0 | ||
|
|
91b56518c6 | ||
|
|
257c4136fe | ||
|
|
94e5f4b2e7 | ||
|
|
f09c363aaf | ||
|
|
d0c317b423 | ||
|
|
5bcb6b5b38 | ||
|
|
3806cb72a1 | ||
|
|
cbfc78d0e3 | ||
|
|
5d87047577 | ||
|
|
7a3e5ba0c8 | ||
|
|
df1ec4d81e | ||
|
|
68a884c5ed | ||
|
|
7bcfb9a57e | ||
|
|
0e360759ec | ||
|
|
0510a352a3 | ||
|
|
7854a7e41f | ||
|
|
f0fee7b0ce | ||
|
|
52fd2b9ed8 | ||
|
|
3ba75e57a0 | ||
|
|
a42cf72d2a | ||
|
|
35171cc77d | ||
|
|
6a5f221567 | ||
|
|
c9c61c4b8e | ||
|
|
8b6186f191 | ||
|
|
73fc0e71a3 | ||
|
|
40152d1f3c | ||
|
|
6cfb78e06e | ||
|
|
4f01e773b3 | ||
|
|
2a83fdbc9a | ||
|
|
84dd3fa63b | ||
|
|
fdf70cdde9 | ||
|
|
07f0125e7c | ||
|
|
bacd44abb1 | ||
|
|
cb032047d3 | ||
|
|
3ecbcc1670 | ||
|
|
85774535ae | ||
|
|
db15742d29 | ||
|
|
cc1c847728 | ||
|
|
a6bf28495c | ||
|
|
0391b9a263 | ||
|
|
4fc2e33058 | ||
|
|
9a6fa8747b | ||
|
|
bd0b204d18 | ||
|
|
94586e3f62 | ||
|
|
63349976fa | ||
|
|
8d092d0e18 | ||
|
|
c8c031e0bd | ||
|
|
b616b0d260 | ||
|
|
91d929646e | ||
|
|
bfdbd93997 | ||
|
|
2916b4228d | ||
|
|
e5a04751e1 | ||
|
|
bb554b38af | ||
|
|
cfc8161473 | ||
|
|
aecd8eff2d | ||
|
|
74c59f58c2 | ||
|
|
e073ec9a21 | ||
|
|
5920f3a6c7 | ||
|
|
c2a0b381d4 | ||
|
|
7598c95fc3 | ||
|
|
35f6f5f0fc | ||
|
|
da23a5ad0b | ||
|
|
5adac14049 | ||
|
|
234f131493 | ||
|
|
c87e3f88fc | ||
|
|
b0aeb8c454 | ||
|
|
a71ac105b9 | ||
|
|
225d64d0f8 | ||
|
|
327018eb09 | ||
|
|
fe8ba75ff3 | ||
|
|
74e8c0fe58 | ||
|
|
4f57af8ea1 | ||
|
|
4132335c96 | ||
|
|
67dae75e3a | ||
|
|
ed0a07a958 | ||
|
|
0a63cc61fc | ||
|
|
c385121406 | ||
|
|
7d7932e024 | ||
|
|
b26362cbca | ||
|
|
9b8095f8f5 | ||
|
|
8b76617019 | ||
|
|
93e50f4720 | ||
|
|
c1388c8699 | ||
|
|
43fc305d5b | ||
|
|
ebd4cf3e8a | ||
|
|
5dfc821575 | ||
|
|
3965a5174f | ||
|
|
df237a11d5 | ||
|
|
373cefdcb5 | ||
|
|
b92a8c6d81 | ||
|
|
68422d404c | ||
|
|
d42c4e5e2f | ||
|
|
0d2586e4f2 | ||
|
|
ff40e79143 | ||
|
|
bd04f6b186 | ||
|
|
d107f98654 | ||
|
|
de9fa8199a | ||
|
|
0e1498a094 | ||
|
|
38f020e482 | ||
|
|
07e1d4e960 | ||
|
|
602d00a5bb | ||
|
|
7395bd8d83 | ||
|
|
63d164ad52 | ||
|
|
f1b915c8b7 | ||
|
|
e3886b94c1 | ||
|
|
44376c2a9f | ||
|
|
2e85c9ac15 | ||
|
|
7cc81deb65 | ||
|
|
a13ddd41ae | ||
|
|
ebe85f2a67 | ||
|
|
c7b62adbab | ||
|
|
66d4926fd0 | ||
|
|
b26f5c512f | ||
|
|
9227eaab9c | ||
|
|
3debac082a | ||
|
|
fb43e03171 | ||
|
|
dd04da64ba | ||
|
|
b94cd9880a | ||
|
|
0f2d2b8850 | ||
|
|
69db85cfe4 | ||
|
|
b445471f98 | ||
|
|
2ce014980b | ||
|
|
40ad6761cc | ||
|
|
bd36669c6a | ||
|
|
98cc1bb7a1 | ||
|
|
3dcdbd99b3 | ||
|
|
1a5c5f6383 | ||
|
|
867f6c7484 | ||
|
|
1097f1b8d5 | ||
|
|
a73156bbe5 | ||
|
|
7a9c62b106 | ||
|
|
70cc910ee9 | ||
|
|
64a71f059f | ||
|
|
f5a0c7cfed | ||
|
|
2472d2f6da | ||
|
|
dd0446f4dd | ||
|
|
7263a1554c | ||
|
|
2c25c27b51 | ||
|
|
f2b92e1f23 | ||
|
|
593c0a567c | ||
|
|
ac1f465634 | ||
|
|
083b1357a1 | ||
|
|
30e594e839 | ||
|
|
2b8f583c44 | ||
|
|
ef3656a8d3 | ||
|
|
5cf2411e74 | ||
|
|
6d9956ae6e | ||
|
|
37d1ac7ff8 | ||
|
|
339633e8de | ||
|
|
a31af76ba1 | ||
|
|
dfd527a740 | ||
|
|
d65bdaa761 | ||
|
|
ef08a514af | ||
|
|
3b3d9de206 | ||
|
|
f1ca851b0d | ||
|
|
c1c6035778 | ||
|
|
d2bf002ea2 | ||
|
|
88f9008558 | ||
|
|
586e852375 | ||
|
|
94e49dbc4e | ||
|
|
d37e676ea6 | ||
|
|
626b36e225 | ||
|
|
62cd67f431 | ||
|
|
09f1bb9f7f | ||
|
|
d6963297a7 | ||
|
|
06b9e88d3a | ||
|
|
39feb65d90 | ||
|
|
b85511a3fe | ||
|
|
88e47faec4 | ||
|
|
9291f8e110 | ||
|
|
fad5d047e7 | ||
|
|
5071009dfb | ||
|
|
8dfb023a55 | ||
|
|
870adf284a | ||
|
|
164f81f249 | ||
|
|
d43f30a40f | ||
|
|
4243be58f6 | ||
|
|
88074ea024 | ||
|
|
a31add749f | ||
|
|
8d19030391 | ||
|
|
8bab408098 | ||
|
|
9f2dd6190d | ||
|
|
89a339a6bf | ||
|
|
ca9c64099f | ||
|
|
5a10257b64 | ||
|
|
6ee95d5cac | ||
|
|
0a9e322d26 | ||
|
|
faee716eba | ||
|
|
03880ba4c6 | ||
|
|
2626c1a57c | ||
|
|
31810bb2b7 | ||
|
|
b5c6727db7 | ||
|
|
e1d0b5e173 | ||
|
|
f147f344d6 | ||
|
|
9c0d16fb3a | ||
|
|
20d11f1a4d | ||
|
|
89eedcd807 | ||
|
|
ef5efe530e | ||
|
|
7918faa942 | ||
|
|
6210ca3e14 | ||
|
|
ff8c44c25f | ||
|
|
20480321ab | ||
|
|
d39855a2e3 | ||
|
|
9fd03082ea | ||
|
|
d6941f903b | ||
|
|
125db51dd1 | ||
|
|
7f40a32aa9 | ||
|
|
6f3050af0f | ||
|
|
926c2b8e45 | ||
|
|
9e0bd8ed75 | ||
|
|
d0cca331ed | ||
|
|
1f55032c34 | ||
|
|
52967de1ac | ||
|
|
9ca8614644 | ||
|
|
80714c6e5e | ||
|
|
b7eb46bd4b | ||
|
|
ca3791e66d | ||
|
|
652036e8d3 | ||
|
|
9054df5685 | ||
|
|
30ca47b667 | ||
|
|
ce1c1a5016 | ||
|
|
ed01cb7d91 | ||
|
|
17364919af | ||
|
|
0c58edcf5e | ||
|
|
90d7c7df98 | ||
|
|
07fbae3da7 | ||
|
|
b4bd8208c3 | ||
|
|
c8b282dc82 | ||
|
|
e2922aa787 | ||
|
|
e8690eeb0b | ||
|
|
a695d58e8e | ||
|
|
b097a98444 | ||
|
|
87c6d45a01 | ||
|
|
d5a79b1830 | ||
|
|
40b967c258 | ||
|
|
b6b7a7c670 | ||
|
|
346e787784 | ||
|
|
8e67f851b0 | ||
|
|
8b416da2ef | ||
|
|
88237dcb30 | ||
|
|
8a0b3962ac | ||
|
|
2f6414bbf4 | ||
|
|
c7aa992586 | ||
|
|
c2b5923e88 | ||
|
|
fc411d0c5c | ||
|
|
3fac8d2608 | ||
|
|
cd413fa887 | ||
|
|
be7cc835be | ||
|
|
b68896d14a | ||
|
|
aae1909466 | ||
|
|
3d7960f43e | ||
|
|
a121e24b79 | ||
|
|
7279a924ed | ||
|
|
a10951b7e5 | ||
|
|
e456b08f92 | ||
|
|
85130cecb3 | ||
|
|
fbb15e26ec | ||
|
|
dd203a1f18 | ||
|
|
06c25f9ec3 | ||
|
|
7072ca2786 | ||
|
|
cb1e5f4c19 | ||
|
|
d4a28aa164 | ||
|
|
fda5fcdd0c | ||
|
|
24131e0791 | ||
|
|
ffe132403e | ||
|
|
cce3bf35e1 | ||
|
|
eccfec36e8 | ||
|
|
a71faa4e60 | ||
|
|
b7362258c6 | ||
|
|
5eb1d98985 | ||
|
|
1c574b2a5d | ||
|
|
6db9b87b6a | ||
|
|
3e3ccb65ff | ||
|
|
0d53b443e5 | ||
|
|
1b795a3dcf | ||
|
|
2e94bc9e09 | ||
|
|
8b94e8a3bd | ||
|
|
91730a2eaf | ||
|
|
4f1a2337be | ||
|
|
4cf5aa03ec | ||
|
|
922eafd4f6 | ||
|
|
9067f80206 | ||
|
|
a7ca1fb269 | ||
|
|
b51f321d22 | ||
|
|
dd048a64a0 | ||
|
|
c6c19de0e8 | ||
|
|
5491f176e1 | ||
|
|
7a28473ebb | ||
|
|
005ce7318c | ||
|
|
448c56e114 | ||
|
|
1d6e498e2b | ||
|
|
33ab717ae0 | ||
|
|
ec563b775b | ||
|
|
de427b5572 | ||
|
|
95921fa5da | ||
|
|
9e4d352366 | ||
|
|
02449cbe0a | ||
|
|
616c68a87b | ||
|
|
319b22ae78 | ||
|
|
b919562eed | ||
|
|
91c01e7363 | ||
|
|
48ec1bccae |
@@ -0,0 +1,14 @@
|
||||
{
|
||||
"permissions": {
|
||||
"allow": [
|
||||
"Bash(find:*)",
|
||||
"Bash(rg:*)",
|
||||
"Bash(grep:*)",
|
||||
"Bash(ls:*)",
|
||||
"Bash(cat:*)",
|
||||
"Bash(head:*)",
|
||||
"Bash(tail:*)"
|
||||
],
|
||||
"deny": []
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
[codespell]
|
||||
skip = .git,*.pdf,*.svg,package-lock.json,*.prisma,pnpm-lock.yaml
|
||||
ignore-words-list = afterall,vertx,notIn
|
||||
ignore-words-list = afterall,vertx,notIn,alue
|
||||
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
FROM node:20
|
||||
|
||||
# ---------- System packages --------------------------------------------------
|
||||
# The buildpack-deps base already ships git, build-essential, python, etc.
|
||||
# Add a few extra tools handy during Langfuse development.
|
||||
RUN apt-get update && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
|
||||
openssl \
|
||||
wget \
|
||||
curl \
|
||||
ca-certificates \
|
||||
postgresql-client \
|
||||
redis-tools \
|
||||
less nano \
|
||||
sudo \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# ---------- Docker -----------------------------------------------------------
|
||||
# Install Docker for background agents that need container capabilities
|
||||
RUN curl -fsSL https://get.docker.com -o get-docker.sh && \
|
||||
sh get-docker.sh && \
|
||||
rm get-docker.sh
|
||||
|
||||
# ---------- pnpm -------------------------------------------------------------
|
||||
# Langfuse monorepo relies on pnpm 9.5.0 (see CONTRIBUTING.md)
|
||||
ENV PNPM_HOME="/pnpm"
|
||||
ENV PATH="$PNPM_HOME:$PATH"
|
||||
RUN corepack enable && \
|
||||
corepack prepare pnpm@9.5.0 --activate
|
||||
|
||||
# ---------- golang-migrate ----------------------------------------------------
|
||||
# CLI used for database migrations during development.
|
||||
ENV MIGRATE_VERSION=4.18.3
|
||||
RUN wget -qO- "https://github.com/golang-migrate/migrate/releases/download/v${MIGRATE_VERSION}/migrate.linux-amd64.tar.gz" \
|
||||
| tar -xz -C /usr/local/bin && \
|
||||
chmod +x /usr/local/bin/migrate
|
||||
|
||||
# ---------- Non-root user -----------------------------------------------------
|
||||
# Create non-root user with sudo privileges and docker group access
|
||||
RUN useradd -ms /bin/bash ubuntu && \
|
||||
usermod -aG sudo ubuntu && \
|
||||
usermod -aG docker ubuntu && \
|
||||
echo "ubuntu ALL=(ALL) NOPASSWD:ALL" >> /etc/sudoers
|
||||
|
||||
# Pre-create pnpm store and set correct ownership to avoid first-run cost & permission issues
|
||||
RUN pnpm store path > /dev/null && \
|
||||
chown -R ubuntu:ubuntu /pnpm
|
||||
|
||||
USER ubuntu
|
||||
WORKDIR /home/ubuntu
|
||||
ENV HOME=/home/ubuntu
|
||||
|
||||
# Container starts with a bash shell ready for hacking.
|
||||
CMD ["bash"]
|
||||
@@ -0,0 +1,14 @@
|
||||
{
|
||||
"install": "cp .env.dev.example .env && pnpm i",
|
||||
"build": {
|
||||
"context": ".",
|
||||
"dockerfile": "Dockerfile"
|
||||
},
|
||||
"start": "sudo service docker start",
|
||||
"terminals": [
|
||||
{
|
||||
"name": "dev server",
|
||||
"command": "pnpm run dx-f"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -11,7 +11,7 @@ alwaysApply: true
|
||||
## File Structure
|
||||
|
||||
- We generally put all code related to a net-new feature into a folder within web/src/features.
|
||||
- Checkout other features to learn about the common structure.
|
||||
- Check out other features to learn about the common structure.
|
||||
|
||||
## API for frontend features
|
||||
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
description:
|
||||
globs:
|
||||
alwaysApply: true
|
||||
---
|
||||
# General rules
|
||||
|
||||
- Linting in this repo only works if the development server is running
|
||||
- Always run the full mono-repo via `pnpm run dx` (use `pnpm dx-f` when you run this in a background agent). Thereby the database will also be seeded.
|
||||
@@ -0,0 +1,20 @@
|
||||
---
|
||||
description:
|
||||
globs:
|
||||
alwaysApply: true
|
||||
---
|
||||
## Project setup
|
||||
- This project is a Turborepo monorepo
|
||||
- We build two containers, web (./web) and worker (./worker)
|
||||
- We have shared code between these two in the shared package (./packages/shared). For that package, we have different entry points [package.json](mdc:packages/shared/package.json).
|
||||
|
||||
|
||||
## Domain layer
|
||||
The most important domain objects are in [observations.ts](mdc:packages/shared/src/domain/observations.ts), [traces.ts](mdc:packages/shared/src/domain/traces.ts), [scores.ts](mdc:packages/shared/src/domain/scores.ts).
|
||||
|
||||
|
||||
## Database schema
|
||||
We use Postgres and Clickhouse.
|
||||
- The postgres schema is in [schema.prisma](mdc:packages/shared/prisma/schema.prisma)
|
||||
- The clickhouse schema is in [0001_traces.up.sql](mdc:packages/shared/clickhouse/migrations/clustered/0001_traces.up.sql), [0002_observations.up.sql](mdc:packages/shared/clickhouse/migrations/clustered/0002_observations.up.sql), [0003_scores.up.sql](mdc:packages/shared/clickhouse/migrations/clustered/0003_scores.up.sql)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
description: How to create new public api routes for Langfuse
|
||||
globs:
|
||||
globs:
|
||||
alwaysApply: false
|
||||
---
|
||||
|
||||
@@ -13,4 +13,5 @@ alwaysApply: false
|
||||
- Add end-to-end test, similar to [datasets-api.servertest.ts](mdc:web/src/__tests__/async/datasets-api.servertest.ts)
|
||||
- Add fern configuration in /fern, learn more about structure here: https://buildwithfern.com/learn/api-definition/fern/overview
|
||||
- Prompt user to regenerate the OpenAPI spec via the fern CLI
|
||||
- Pagination starts at 1, query typing defined in publicApiPaginationZod, return meta in paginationMetaResponseZod
|
||||
- Pagination starts at 1, query typing defined in publicApiPaginationZod, return meta in paginationMetaResponseZod
|
||||
- For tests, please look at [this directory](mdc:web/src/__tests__/async/) for examples
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
description:
|
||||
globs:
|
||||
alwaysApply: true
|
||||
---
|
||||
# Writing tests
|
||||
|
||||
- When writing tests, focus on decoupling each `it` or `test` block to ensure that they can run independently and concurrently. Tests must never depend on the action or outcome of previous or subsequent tests.
|
||||
- When writing tests, especially in the __tests__/async directory, ensure that you avoid `pruneDatabase` calls.
|
||||
@@ -0,0 +1,13 @@
|
||||
# Dev container Dockerfile
|
||||
FROM mcr.microsoft.com/vscode/devcontainers/typescript-node:20-bookworm
|
||||
|
||||
# Install golang-migrate for database migrations
|
||||
RUN curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz && \
|
||||
chmod +x migrate && \
|
||||
mv migrate /usr/local/bin/migrate
|
||||
|
||||
# Install pnpm globally
|
||||
RUN npm install -g pnpm@9.5.0
|
||||
|
||||
# Install Claude Code CLI
|
||||
RUN npm install -g @anthropic-ai/claude-code
|
||||
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"name": "langfuse-development",
|
||||
"build": {
|
||||
"dockerfile": "Dockerfile"
|
||||
},
|
||||
"forwardPorts": [3000, 5432, 6379, 8123, 9000],
|
||||
"postCreateCommand": "cp .env.dev.example .env && pnpm i"
|
||||
}
|
||||
@@ -68,6 +68,7 @@ LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
|
||||
|
||||
LANGFUSE_USE_AZURE_BLOB=true
|
||||
LANGFUSE_AZURE_SKIP_CONTAINER_CHECK=false
|
||||
|
||||
# Set during docker build of application
|
||||
# Used to disable environment verification at build time
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
# When adding additional environment variables, the schema in "/src/env.mjs"
|
||||
# should be updated accordingly.
|
||||
|
||||
# Prisma
|
||||
# https://www.prisma.io/docs/reference/database-reference/connection-urls#env
|
||||
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/postgres"
|
||||
|
||||
# Clickhouse
|
||||
CLICKHOUSE_MIGRATION_URL="clickhouse://localhost:9000"
|
||||
CLICKHOUSE_URL="http://localhost:8123"
|
||||
CLICKHOUSE_USER="clickhouse"
|
||||
CLICKHOUSE_PASSWORD="clickhouse"
|
||||
CLICKHOUSE_CLUSTER_ENABLED="false"
|
||||
|
||||
# Next Auth
|
||||
# You can generate a new secret on the command line with:
|
||||
# openssl rand -base64 32
|
||||
# https://next-auth.js.org/configuration/options#secret
|
||||
# NEXTAUTH_SECRET=""
|
||||
NEXTAUTH_URL="http://localhost:3000"
|
||||
NEXTAUTH_SECRET="secret"
|
||||
|
||||
# Langfuse Cloud Environment
|
||||
NEXT_PUBLIC_LANGFUSE_CLOUD_REGION="DEV"
|
||||
|
||||
# Langfuse experimental features
|
||||
LANGFUSE_ENABLE_EXPERIMENTAL_FEATURES="true"
|
||||
|
||||
# Salt for API key hashing
|
||||
SALT="salt"
|
||||
|
||||
# Email
|
||||
EMAIL_FROM_ADDRESS="" # Defines the email address to use as the from address.
|
||||
SMTP_CONNECTION_URL="" # Defines the connection url for smtp server.
|
||||
|
||||
# S3 Batch Exports
|
||||
LANGFUSE_S3_BATCH_EXPORT_ENABLED=true
|
||||
LANGFUSE_S3_BATCH_EXPORT_BUCKET=langfuse
|
||||
LANGFUSE_S3_BATCH_EXPORT_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_BATCH_EXPORT_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_BATCH_EXPORT_REGION=us-east-1
|
||||
LANGFUSE_S3_BATCH_EXPORT_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_BATCH_EXPORT_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_BATCH_EXPORT_PREFIX=exports/
|
||||
|
||||
# S3 Media Upload LOCAL
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_REGION=us-east-1
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_MEDIA_UPLOAD_PREFIX=media/
|
||||
|
||||
# S3 Event Bucket Upload
|
||||
## Set to true to test uploading all events to S3
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET=langfuse
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID=minio
|
||||
LANGFUSE_S3_EVENT_UPLOAD_SECRET_ACCESS_KEY=miniosecret
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION=us-east-1
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ENDPOINT=http://localhost:9090
|
||||
## Necessary for minio compatibility
|
||||
LANGFUSE_S3_EVENT_UPLOAD_FORCE_PATH_STYLE=true
|
||||
LANGFUSE_S3_EVENT_UPLOAD_PREFIX=events/
|
||||
|
||||
# Set during docker build of application
|
||||
# Used to disable environment verification at build time
|
||||
# DOCKER_BUILD=1
|
||||
|
||||
REDIS_HOST="127.0.0.1"
|
||||
REDIS_PORT=6379
|
||||
REDIS_AUTH="bitnami"
|
||||
REDIS_CLUSTER_ENABLED="true"
|
||||
REDIS_CLUSTER_NODES="127.0.0.1:6370,127.0.0.1:6371,127.0.0.1:6372,127.0.0.1:6373,127.0.0.1:6374,127.0.0.1:6375"
|
||||
LANGFUSE_INGESTION_QUEUE_SHARD_COUNT=8
|
||||
LANGFUSE_TRACE_UPSERT_QUEUE_SHARD_COUNT=4
|
||||
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
|
||||
# speeds up local development by not executing init scripts on server startup
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
|
||||
# Use the following settings to enforce running the new AMTs during the tests
|
||||
LANGFUSE_EXPERIMENT_INSERT_INTO_AGGREGATING_MERGE_TREES="true"
|
||||
LANGFUSE_EXPERIMENT_COMPARE_READ_FROM_AGGREGATING_MERGE_TREES="true"
|
||||
LANGFUSE_EXPERIMENT_WHITELISTED_PROJECT_IDS="7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
|
||||
LANGFUSE_EXPERIMENT_ADD_QUERY_RESULT_TO_SPAN_PROJECT_IDS="7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"
|
||||
LANGFUSE_EXPERIMENT_RETURN_NEW_RESULT="true"
|
||||
LANGFUSE_EXPERIMENT_SAMPLING_RATE=1
|
||||
+10
-1
@@ -75,7 +75,16 @@ REDIS_PORT=6379
|
||||
REDIS_AUTH="myredissecret"
|
||||
|
||||
# openssl rand -hex 32 used only here
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
ENCRYPTION_KEY=0000000000000000000000000000000000000000000000000000000000000000
|
||||
|
||||
# speeds up local development by not executing init scripts on server startup
|
||||
NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT="false"
|
||||
|
||||
# For SDK integration tests to pass, decrease the ingestion queue delay by uncommenting the env vars:
|
||||
# LANGFUSE_INGESTION_QUEUE_DELAY_MS=10
|
||||
# LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=10
|
||||
|
||||
# Slack credentials for development
|
||||
SLACK_CLIENT_ID=your_slack_client_id
|
||||
SLACK_CLIENT_SECRET=your_slack_client_secret
|
||||
SLACK_STATE_SECRET=your_slack_state_secret
|
||||
|
||||
@@ -172,9 +172,14 @@ OTEL_SERVICE_NAME="langfuse"
|
||||
# REDIS_HOST=
|
||||
# REDIS_PORT=
|
||||
# REDIS_AUTH=
|
||||
# REDIS_USERNAME=default
|
||||
# REDIS_CONNECTION_STRING=
|
||||
# REDIS_ENABLE_AUTO_PIPELINING=
|
||||
|
||||
# Redis Cluster configuration (optional)
|
||||
# REDIS_CLUSTER_ENABLED=false
|
||||
# REDIS_CLUSTER_NODES=redis-node1:6379,redis-node2:6379,redis-node3:6379
|
||||
|
||||
# Cache configuration
|
||||
# LANGFUSE_CACHE_API_KEY_ENABLED=
|
||||
# LANGFUSE_CACHE_API_KEY_TTL_SECONDS=
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
# Test Environment Configuration
|
||||
# Copy this file to .env.test for test database isolation
|
||||
# Only overrides specific test variables - other values inherited from .env
|
||||
|
||||
# PostgreSQL - Test Database
|
||||
DIRECT_URL="postgresql://postgres:postgres@localhost:5432/langfuse_test"
|
||||
DATABASE_URL="postgresql://postgres:postgres@localhost:5432/langfuse_test"
|
||||
|
||||
# ClickHouse - Use Default Database for now, nothing set
|
||||
|
||||
# Redis - Test Database (database 1 for isolation)
|
||||
REDIS_CONNECTION_STRING="redis://:myredissecret@127.0.0.1:6379/1"
|
||||
@@ -28,7 +28,7 @@ Fixes # (issue)
|
||||
<!-- Remove bullet points below that don't apply to you -->
|
||||
|
||||
- I haven't read the [contributing guide](https://github.com/langfuse/langfuse/blob/main/CONTRIBUTING.md)
|
||||
- My code doesn't follow the style guidelines of this project (`npm run prettier`)
|
||||
- My code doesn't follow the style guidelines of this project (`pnpm run format`)
|
||||
- I haven't commented my code, particularly in hard-to-understand areas
|
||||
- I haven't checked if my PR needs changes to the documentation
|
||||
- I haven't checked if my changes generate no new warnings (`npm run lint`)
|
||||
|
||||
@@ -36,6 +36,7 @@ updates:
|
||||
patterns:
|
||||
- "express"
|
||||
- "@types/express"
|
||||
- "@types/express-serve-static-core"
|
||||
observability:
|
||||
patterns:
|
||||
- "dd-trace"
|
||||
|
||||
@@ -15,10 +15,6 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
environment: ${{ inputs.environment }}
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- name: Get app name
|
||||
uses: winterjung/split@v2
|
||||
id: split
|
||||
|
||||
@@ -52,6 +52,49 @@ jobs:
|
||||
- name: lint web
|
||||
run: pnpm run lint
|
||||
|
||||
prettier-check:
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- pre-job
|
||||
if: needs.pre-job.outputs.should_skip != 'true'
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- uses: pnpm/action-setup@v3
|
||||
with:
|
||||
version: 9.5.0
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: 20
|
||||
cache: "pnpm"
|
||||
cache-dependency-path: "pnpm-lock.yaml"
|
||||
- name: install dependencies
|
||||
run: |
|
||||
pnpm i
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev.example .env
|
||||
- name: Check formatting on changed files
|
||||
run: |
|
||||
if [ "${{ github.event_name }}" = "pull_request" ]; then
|
||||
BASE_SHA=${{ github.event.pull_request.base.sha }}
|
||||
else
|
||||
BASE_SHA=$(git merge-base origin/main HEAD)
|
||||
fi
|
||||
|
||||
echo "Checking files changed from $BASE_SHA to HEAD"
|
||||
|
||||
# Get changed files
|
||||
CHANGED_FILES=$(git diff --name-only $BASE_SHA HEAD -- '*.js' '*.jsx' '*.ts' '*.tsx' '*.css' | tr '\n' ' ')
|
||||
|
||||
if [ -n "$CHANGED_FILES" ] && [ "$CHANGED_FILES" != " " ]; then
|
||||
echo "Files to check: $CHANGED_FILES"
|
||||
pnpm prettier --check --experimental-cli $CHANGED_FILES
|
||||
else
|
||||
echo "No JS/TS/CSS files changed - skipping prettier check"
|
||||
fi
|
||||
|
||||
test-docker-build:
|
||||
timeout-minutes: 20
|
||||
runs-on: ubuntu-latest
|
||||
@@ -71,7 +114,9 @@ jobs:
|
||||
run: echo "NEXT_PUBLIC_BUILD_ID=$(git rev-parse --short HEAD)" >> $GITHUB_ENV
|
||||
- name: Build and run both images from compose
|
||||
run: |
|
||||
docker compose -f docker-compose.build.yml up -d
|
||||
docker compose --progress plain --verbose -f docker-compose.build.yml build --print > /tmp/bake.json
|
||||
docker buildx bake -f /tmp/bake.json
|
||||
docker compose --progress plain -f docker-compose.build.yml up -d
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
- name: Ensure no unhealthy status
|
||||
run: |
|
||||
@@ -100,14 +145,10 @@ jobs:
|
||||
node-version: [20]
|
||||
postgres-version: [12, 15]
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- uses: pnpm/action-setup@v3
|
||||
@@ -133,6 +174,7 @@ jobs:
|
||||
cp .env.dev.example .env
|
||||
grep -v -e '^LANGFUSE_S3_BATCH_EXPORT_ENABLED=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev.example > .env
|
||||
echo "LANGFUSE_INGESTION_QUEUE_DELAY_MS=1" >> .env
|
||||
echo "LANGFUSE_CACHE_PROMPT_ENABLED=false" >> .env
|
||||
echo "LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=1" >> .env
|
||||
- name: Run dev containers
|
||||
run: |
|
||||
@@ -171,21 +213,17 @@ jobs:
|
||||
needs:
|
||||
- pre-job
|
||||
if: needs.pre-job.outputs.should_skip != 'true'
|
||||
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
|
||||
name: tests-web-async (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.deploy-mode }})
|
||||
strategy:
|
||||
matrix:
|
||||
node-version: [20]
|
||||
postgres-version: [12, 15]
|
||||
blob-provider: ["", "-azure"]
|
||||
deploy-mode: ["", "-azure", "-redis-cluster"]
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- uses: pnpm/action-setup@v3
|
||||
@@ -208,8 +246,8 @@ jobs:
|
||||
pnpm install
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev${{ matrix.blob-provider }}.example .env
|
||||
grep -v -e '^LANGFUSE_S3_BATCH_EXPORT_ENABLED=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.blob-provider }}.example > .env
|
||||
cp .env.dev${{ matrix.deploy-mode }}.example .env
|
||||
grep -v -e '^LANGFUSE_S3_BATCH_EXPORT_ENABLED=' -e '^NEXT_PUBLIC_LANGFUSE_RUN_NEXT_INIT=' .env.dev${{ matrix.deploy-mode }}.example > .env
|
||||
echo "LANGFUSE_INGESTION_QUEUE_DELAY_MS=1" >> .env
|
||||
echo "LANGFUSE_INGESTION_CLICKHOUSE_WRITE_INTERVAL_MS=1" >> .env
|
||||
echo "LANGFUSE_TRACE_DELETE_CONCURRENCY=100" >> .env
|
||||
@@ -217,7 +255,7 @@ jobs:
|
||||
echo "LANGFUSE_EE_LICENSE_KEY=langfuse_ee_test" >> .env
|
||||
- name: Run dev containers
|
||||
run: |
|
||||
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
|
||||
docker compose -f docker-compose.dev${{ matrix.deploy-mode }}.yml up -d
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
docker compose ps
|
||||
env:
|
||||
@@ -250,17 +288,13 @@ jobs:
|
||||
needs:
|
||||
- pre-job
|
||||
if: needs.pre-job.outputs.should_skip != 'true'
|
||||
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.blob-provider }})
|
||||
name: tests-worker (node${{ matrix.node-version }}, pg${{ matrix.postgres-version }}, mode${{ matrix.deploy-mode }})
|
||||
strategy:
|
||||
matrix:
|
||||
node-version: [20]
|
||||
postgres-version: [12, 15]
|
||||
blob-provider: ["", "-azure"]
|
||||
deploy-mode: ["", "-azure", "-redis-cluster"]
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 10
|
||||
- uses: actions/checkout@v4
|
||||
- uses: pnpm/action-setup@v3
|
||||
with:
|
||||
@@ -282,17 +316,17 @@ jobs:
|
||||
pnpm install
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev${{ matrix.blob-provider }}.example .env
|
||||
cp .env.dev${{ matrix.blob-provider }}.example web/.env
|
||||
cp .env.dev${{ matrix.blob-provider }}.example worker/.env
|
||||
cp .env.dev${{ matrix.deploy-mode }}.example .env
|
||||
cp .env.dev${{ matrix.deploy-mode }}.example web/.env
|
||||
cp .env.dev${{ matrix.deploy-mode }}.example worker/.env
|
||||
- name: Run + migrate
|
||||
run: |
|
||||
docker compose -f docker-compose.dev${{ matrix.blob-provider }}.yml up -d
|
||||
docker compose -f docker-compose.dev${{ matrix.deploy-mode }}.yml up -d
|
||||
sleep 5 # Wait for PostgreSQL to accept connections
|
||||
docker compose ps
|
||||
- name: Ensure no unhealthy status
|
||||
@@ -343,7 +377,7 @@ jobs:
|
||||
cp .env.dev.example web/.env
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- name: Run + migrate
|
||||
@@ -397,14 +431,12 @@ jobs:
|
||||
pnpm install
|
||||
- name: Install golang-migrate for Clickhouse migrations
|
||||
run: |
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.2/migrate.linux-amd64.tar.gz | tar xvz
|
||||
curl -L https://github.com/golang-migrate/migrate/releases/download/v4.18.3/migrate.linux-amd64.tar.gz | tar xvz
|
||||
sudo mv migrate /usr/bin/migrate
|
||||
which migrate
|
||||
- name: Load default env
|
||||
run: |
|
||||
cp .env.dev.example .env
|
||||
echo "LANGFUSE_CACHE_API_KEY_ENABLED=true" >> .env
|
||||
echo "LANGFUSE_CACHE_PROMPT_ENABLED=true" >> .env
|
||||
- name: Run + migrate
|
||||
run: |
|
||||
docker compose -f docker-compose.dev.yml up -d
|
||||
@@ -442,6 +474,7 @@ jobs:
|
||||
needs:
|
||||
[
|
||||
lint,
|
||||
prettier-check,
|
||||
tests-web-sync,
|
||||
tests-worker,
|
||||
e2e-tests,
|
||||
@@ -452,13 +485,21 @@ jobs:
|
||||
if: always()
|
||||
steps:
|
||||
- name: Successful deploy
|
||||
if: ${{ !(contains(needs.*.result, 'failure')) }}
|
||||
if: ${{ !(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) }}
|
||||
run: exit 0
|
||||
working-directory: .
|
||||
- name: Failing deploy
|
||||
if: ${{ contains(needs.*.result, 'failure') }}
|
||||
if: ${{ contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') }}
|
||||
run: exit 1
|
||||
working-directory: .
|
||||
- name: Notify Slack
|
||||
uses: ravsamhq/notify-slack-action@v2
|
||||
if: always() && github.event_name == 'push' && (github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/'))
|
||||
with:
|
||||
status: ${{ job.status }}
|
||||
notify_when: "failure"
|
||||
env:
|
||||
SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL }}
|
||||
|
||||
push-docker-image:
|
||||
needs: all-ci-passed
|
||||
|
||||
+10
-47
@@ -1,65 +1,28 @@
|
||||
name: Snyk Container
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- "**"
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
# Snyk cannot upload results in merge group. Hence, we only run on PRs and when pushingon the main branch https://github.com/github/codeql-action/issues/1572
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
branches: ["main"]
|
||||
|
||||
jobs:
|
||||
snyk:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- name: Build a Docker image
|
||||
run: docker compose -f docker-compose.build.yml up -d
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-web)
|
||||
continue-on-error: true
|
||||
- name: Scan web image with Snyk
|
||||
uses: snyk/actions/docker@master
|
||||
continue-on-error: true # let upload step always run
|
||||
env:
|
||||
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
|
||||
with:
|
||||
image: langfuse-langfuse-web
|
||||
args: --file=web/Dockerfile
|
||||
image: langfuse/langfuse # pulled from Docker Hub
|
||||
args: --severity-threshold=high # (any extra CLI flags, but NO --file)
|
||||
|
||||
# Workaround for https://github.com/github/codeql-action/issues/2187
|
||||
- name: Replace security-severity undefined for license-related findings
|
||||
run: "sed -i 's/\"security-severity\": \"undefined\"/\"security-severity\": \"0\"/g' snyk.sarif"
|
||||
|
||||
- name: Replace security-severity null for license-related findings
|
||||
run: "sed -i 's/\"security-severity\": \"null\"/\"security-severity\": \"0\"/g' snyk.sarif"
|
||||
|
||||
- name: Upload result to GitHub Code Scanning
|
||||
uses: github/codeql-action/upload-sarif@v3
|
||||
with:
|
||||
sarif_file: snyk.sarif
|
||||
category: web
|
||||
|
||||
- name: Run Snyk to check Docker image for vulnerabilities (langfuse-worker)
|
||||
continue-on-error: true
|
||||
- name: Scan worker image with Snyk
|
||||
uses: snyk/actions/docker@master
|
||||
continue-on-error: true # let upload step always run
|
||||
env:
|
||||
SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }}
|
||||
with:
|
||||
image: langfuse-langfuse-worker
|
||||
args: --file=worker/Dockerfile
|
||||
|
||||
# Workaround for https://github.com/github/codeql-action/issues/2187
|
||||
- name: Replace security-severity undefined for license-related findings
|
||||
run: "sed -i 's/\"security-severity\": \"undefined\"/\"security-severity\": \"0\"/g' snyk.sarif"
|
||||
|
||||
- name: Replace security-severity null for license-related findings
|
||||
run: "sed -i 's/\"security-severity\": \"null\"/\"security-severity\": \"0\"/g' snyk.sarif"
|
||||
|
||||
- name: Upload result to GitHub Code Scanning
|
||||
uses: github/codeql-action/upload-sarif@v3
|
||||
with:
|
||||
sarif_file: snyk.sarif
|
||||
category: worker
|
||||
image: langfuse/langfuse-worker # pulled from Docker Hub
|
||||
args: --severity-threshold=high # (any extra CLI flags, but NO --file)
|
||||
|
||||
+8
-4
@@ -4,6 +4,7 @@
|
||||
/node_modules
|
||||
/.pnp
|
||||
.pnp.js
|
||||
migrate
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
@@ -39,7 +40,9 @@ yarn-error.log*
|
||||
.env*
|
||||
!.env.dev.example
|
||||
!.env.dev-azure.example
|
||||
!.env.dev-redis-cluster.example
|
||||
!.env.prod.example
|
||||
!.env.test.example
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
@@ -47,15 +50,13 @@ yarn-error.log*
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
|
||||
/generated/typescript-server
|
||||
/generated
|
||||
|
||||
# openapi spec that is copied during build
|
||||
/public/openapi*.yml
|
||||
|
||||
|
||||
# vscode
|
||||
.devcontainer
|
||||
|
||||
node_modules
|
||||
**/node_modules
|
||||
**/dist
|
||||
@@ -64,4 +65,7 @@ node_modules
|
||||
.yarn
|
||||
.turbo
|
||||
|
||||
web/test-results/*
|
||||
web/test-results/*
|
||||
|
||||
# local config files
|
||||
*.local.*
|
||||
|
||||
+4
-3
@@ -11,12 +11,13 @@ if [ "$current_branch" = "$protected_branch" ]; then
|
||||
echo "🚨 You are about to commit to the $protected_branch branch. Are you sure? (y/n)"
|
||||
read -r answer < /dev/tty
|
||||
if [ "$answer" != "${answer#[Yy]}" ]; then
|
||||
exit 0 # Commit will proceed
|
||||
# Commit approved, check formatting. On files changed, block commit
|
||||
pnpm run format:check
|
||||
else
|
||||
echo "Commit to $protected_branch branch has been canceled."
|
||||
exit 1 # Commit will be blocked
|
||||
fi
|
||||
fi
|
||||
|
||||
# If not the protected branch, proceed with the commit
|
||||
exit 0
|
||||
# If not the protected branch, check formatting (on changed files, block)
|
||||
pnpm run format:check
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
# Generated code
|
||||
packages/shared/prisma/generated
|
||||
@@ -8,6 +8,8 @@ Langfuse is an **open source LLM engineering** platform for developing, monitori
|
||||
|
||||
## Tests
|
||||
- Codex cannot run the test suite because it depends on Docker-based infrastructure that is unavailable in this environment.
|
||||
- When writing tests, focus on decoupling each `it` or `test` block to ensure that they can run independently and concurrently. Tests must never depend on the action or outcome of previous or subsequent tests.
|
||||
- When writing tests, especially in the __tests__/async directory, ensure that you avoid `pruneDatabase` calls.
|
||||
|
||||
## Cursor Rules
|
||||
- Additional folder-specific rules live in `.cursor/rules/`.
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
# CLAUDE.md
|
||||
|
||||
## Project Overview
|
||||
|
||||
Langfuse is an open-source LLM engineering platform that helps teams collaboratively develop, monitor, evaluate, and debug AI applications.
|
||||
The main feature areas are tracing, evals and prompt management. Langfuse consists of the web application (this repo), documentation, python SDK and javascript/typescript SDK.
|
||||
This repo contains the web application, worker, and supporting packages but notably not the JS nor Python client SDKs.
|
||||
|
||||
## Repository Structure
|
||||
High level structure. There are more folders (eg for hooks etc).
|
||||
```
|
||||
langfuse/
|
||||
├── web/ # Next.js 14 frontend/backend application
|
||||
│ ├── src/
|
||||
│ │ ├── components/ # Reusable UI components (shadcn/ui)
|
||||
│ │ ├── features/ # Feature-specific code organized by domain
|
||||
│ │ ├── pages/ # Next.js pages (Pages Router)
|
||||
│ │ └── server/ # tRPC API routes and server logic
|
||||
│ └── public/ # Static assets
|
||||
├── worker/ # Express.js background job processor
|
||||
│ └── src/
|
||||
│ ├── queues/ # BullMQ job queues
|
||||
│ └── services/ # Background processing services
|
||||
├── packages/
|
||||
│ ├── shared/ # Shared types, schemas, and utilities
|
||||
│ │ ├── prisma/ # Database schema and migrations
|
||||
│ │ └── src/ # Shared TypeScript code
|
||||
│ ├── config-eslint/ # ESLint configuration
|
||||
│ └── config-typescript/ # TypeScript configuration
|
||||
├── ee/ # Enterprise Edition features
|
||||
├── fern/ # API documentation and OpenAPI specs
|
||||
├── generated/ # Auto-generated client code
|
||||
└── scripts/ # Development and deployment scripts
|
||||
```
|
||||
|
||||
## Repository Architecture
|
||||
This is a **pnpm + Turbo monorepo** with the following key packages:
|
||||
|
||||
### Core Applications
|
||||
- **`/web/`** - Next.js 14 application (Pages Router) providing both frontend UI and backend APIs
|
||||
- **`/worker/`** - Express.js background job processing server
|
||||
- **`/packages/shared/`** - Shared database schema, types, and utilities
|
||||
|
||||
### Supporting Packages
|
||||
- **`/ee/`** - Enterprise Edition features (separate licensing)
|
||||
- **`/packages/config-eslint/`** - Shared ESLint configuration
|
||||
- **`/packages/config-typescript/`** - Shared TypeScript configuration
|
||||
|
||||
## Development Commands
|
||||
|
||||
### Development
|
||||
```sh
|
||||
pnpm i # Install dependencies
|
||||
pnpm run dev # Start all services (web + worker)
|
||||
pnpm run dev:web # Web app only (localhost:3000) - **used in most cases!**
|
||||
pnpm run dev:worker # Worker only
|
||||
pnpm run dx # Full initial setup: install deps, reset DBs, resets node modules, seed data, start dev. USE SPARINGLY AS IT WIPES THE DATABASE & node_modules
|
||||
```
|
||||
|
||||
### Database Management
|
||||
database commands are to be run in the `packages/shared/` folder.
|
||||
```sh
|
||||
pnpm run db:generate # Build prisma models
|
||||
pnpm run db:migrate # Run Prisma migrations
|
||||
pnpm run db:reset # Reset and reseed databases
|
||||
pnpm run db:seed # Seed with example data
|
||||
```
|
||||
|
||||
### Infrastructure
|
||||
```sh
|
||||
pnpm run infra:dev:up # Start Docker services (PostgreSQL, ClickHouse, Redis, MinIO)
|
||||
pnpm run infra:dev:down # Stop Docker services
|
||||
```
|
||||
|
||||
### Building
|
||||
```sh
|
||||
pnpm --filter=PACKAGE_NAME run build # Runs the build command, will show real typescript errors etc.
|
||||
```
|
||||
|
||||
### Testing in Web Package
|
||||
The web package uses JEST for unit tests.
|
||||
Depending on the file location (sync, async)
|
||||
`web` related tests must go into the `web/src/__tests__/` folder.
|
||||
```sh
|
||||
pnpm test-sync --testPathPattern="$FILE_LOCATION_PATTERN" --testNamePattern="$TEST_NAME_PATTERN"
|
||||
pnpm test-async --testPathPattern="$FILE_LOCATION_PATTERN" --testNamePattern="$TEST_NAME_PATTERN"
|
||||
```
|
||||
|
||||
### Testing in the Worker Package
|
||||
The worker uses `vitest` for unit tests.
|
||||
```sh
|
||||
pnpm run test --filter=worker -- $TEST_FILE_NAME -t "$TEST_NAME"
|
||||
```
|
||||
|
||||
### Utilities
|
||||
```bash
|
||||
pnpm run format # Format code across entire project
|
||||
pnpm run nuke # Remove all node_modules, build files, wipe database, docker containers. **USE WITH CAUTION**
|
||||
```
|
||||
|
||||
## Technology Stack
|
||||
|
||||
### Web Application (`/web/`)
|
||||
- **Framework**: Next.js 14 (Pages Router)
|
||||
- **APIs**: tRPC (type-safe client-server communication) + REST APIs for public access
|
||||
- **Authentication**: NextAuth.js/Auth.js
|
||||
- **Database**: Prisma ORM with PostgreSQL
|
||||
- **Analytics Database**: ClickHouse (high-volume trace data)
|
||||
- **Validation**: Zod schemas, we use zodv4 (always import from `zod/v4`)
|
||||
- **Styling**: Tailwind CSS with CSS variables for theming
|
||||
- **Components**: shadcn/ui (Radix UI primitives)
|
||||
- **State Management**: TanStack Query (React Query) + tRPC
|
||||
- **Charts**: Tremor, Recharts
|
||||
|
||||
### Worker Application (`/worker/`)
|
||||
- **Framework**: Express.js
|
||||
- **Queue System**: BullMQ with Redis
|
||||
- **Purpose**: Async processing (data ingestion, evaluations, exports, integrations)
|
||||
|
||||
### Infrastructure
|
||||
- **Primary Database**: PostgreSQL (via Prisma ORM)
|
||||
- **Analytics Database**: ClickHouse
|
||||
- **Cache/Queues**: Redis
|
||||
- **Blob Storage**: MinIO/S3
|
||||
|
||||
## Development Guidelines
|
||||
|
||||
### Frontend Features
|
||||
- All new features go in `/web/src/features/[feature-name]/`
|
||||
- Use tRPC for full-stack features (entry point: `web/src/server/api/root.ts`)
|
||||
- Follow existing feature structure for consistency
|
||||
- Use shadcn/ui components from `@/src/components/ui`
|
||||
- Custom reusable components go in `@/src/components`
|
||||
|
||||
### Public API Development
|
||||
- All public API routes in `/web/src/pages/api/public`
|
||||
- Use `withMiddlewares.ts` wrapper
|
||||
- Define types in `/web/src/features/public-api/types` with strict Zod v4 objects
|
||||
- Add end-to-end tests (see `datasets-api.servertest.ts`)
|
||||
- Manually update Fern API specs in `/fern/`, then regenerate OpenAPI spec via Fern CLI
|
||||
|
||||
### Authorization & RBAC
|
||||
- Check `/web/src/features/rbac/README.md` for authorization patterns
|
||||
- Implement proper entitlements checking (see `/web/src/features/entitlements/README.md`)
|
||||
|
||||
### Database
|
||||
- **Dual database system**: PostgreSQL (primary) + ClickHouse (analytics)
|
||||
- Use `golang-migrate` CLI for database migrations
|
||||
- All database operations go through Prisma ORM for PostgreSQL
|
||||
- Foreign key relationships may not be enforced in schema to allow unordered ingestion
|
||||
|
||||
### Testing
|
||||
- Jest for API tests, Playwright for E2E tests
|
||||
- For backend/API changes, tests must pass before pushes
|
||||
- Add tests for new API endpoints and features
|
||||
- When writing tests, focus on decoupling each `it` or `test` block to ensure that they can run independently and concurrently. Tests must never depend on the action or outcome of previous or subsequent tests.
|
||||
- When writing tests, especially in the __tests__/async directory, ensure that you avoid `pruneDatabase` calls.
|
||||
|
||||
### Code Conventions
|
||||
- **Pages Router** (not App Router)
|
||||
- Follow conventional commits on main branch
|
||||
- Use CSS variables for theming (supports auto dark/light mode)
|
||||
- TypeScript throughout
|
||||
- Zod v4 for all input validation
|
||||
|
||||
## Environment Setup
|
||||
|
||||
- **Node.js**: Version 20 (specified in `.nvmrc`)
|
||||
- **Package Manager**: pnpm v9.5.0
|
||||
- **Database Dependencies**: Docker for local PostgreSQL, ClickHouse, Redis, MinIO
|
||||
- **Environment**: Copy `.env.dev.example` to `.env`
|
||||
|
||||
## Login for Development
|
||||
|
||||
When running locally with seed data:
|
||||
- Username: `demo@langfuse.com`
|
||||
- Password: `password`
|
||||
- Demo project URL: `http://localhost:3000/project/7a88fb47-b4e2-43b8-a06c-a5ce950dc53a`
|
||||
|
||||
## Linear MCP
|
||||
To get a project, use the `get_project` capability with the full project name as it is in the title.
|
||||
- bad: message-placeholder-in-chat-messages-2beb6f02ec48
|
||||
- good: Message placeholder in chat messages
|
||||
|
||||
## Front-end Tips
|
||||
|
||||
### Window Location Handling
|
||||
- Whenever you want to use or do use window.location..., ensure that you also add proper handling for a custom basePath
|
||||
|
||||
## TypeScript Best Practices
|
||||
- In TypeScript, if possible, don't use the `any` type
|
||||
|
||||
## General Coding Guidelines
|
||||
- For easier code reviews, prefer not to move functions etc around within a file unless necessary or instructed to do so
|
||||
|
||||
## Development Tips
|
||||
- Before trying to build the package, try running the linter once first
|
||||
+106
-22
@@ -42,7 +42,7 @@ A good first step is to search for open [issues](https://github.com/langfuse/lan
|
||||
- NextAuth.js / Auth.js
|
||||
- tRPC: Frontend APIs
|
||||
- Prisma ORM
|
||||
- Zod
|
||||
- Zod v4
|
||||
- Tailwind CSS
|
||||
- shadcn/ui tailwind components (using Radix and tanstack)
|
||||
- Fern: generate OpenAPI spec and Pydantic models
|
||||
@@ -58,16 +58,30 @@ See this [diagram](https://langfuse.com/self-hosting#architecture) for an overvi
|
||||
### Network Overview
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Browser ---|Web UI & TRPC API| App
|
||||
Integrations/SDKs ---|Public HTTP API| App
|
||||
subgraph i1["Application Network"]
|
||||
App["Langfuse Application"]
|
||||
end
|
||||
subgraph i2["Database Network"]
|
||||
DB["Postgres Database"]
|
||||
end
|
||||
App --- DB
|
||||
flowchart TB
|
||||
User["UI, API, SDKs"]
|
||||
subgraph vpc["VPC"]
|
||||
Web["Web Server<br/>(langfuse/langfuse)"]
|
||||
Worker["Async Worker<br/>(langfuse/worker)"]
|
||||
Postgres["Postgres - OLTP<br/>(Transactional Data)"]
|
||||
Cache["Redis/Valkey<br/>(Cache, Queue)"]
|
||||
Clickhouse["Clickhouse - OLAP<br/>(Observability Data)"]
|
||||
S3["S3 / Blob Storage<br/>(Raw events, multi-modal attachments)"]
|
||||
end
|
||||
LLM["LLM API/Gateway<br/>(optional)"]
|
||||
|
||||
User --> Web
|
||||
Web --> S3
|
||||
Web --> Postgres
|
||||
Web --> Cache
|
||||
Web --> Clickhouse
|
||||
Web -.->|"optional for playground"| LLM
|
||||
|
||||
Cache --> Worker
|
||||
Worker --> Clickhouse
|
||||
Worker --> Postgres
|
||||
Worker --> S3
|
||||
Worker -.->|"optional for evals"| LLM
|
||||
```
|
||||
|
||||
### Database Overview
|
||||
@@ -110,16 +124,24 @@ Requirements
|
||||
cd langfuse
|
||||
```
|
||||
|
||||
3. Create an env file
|
||||
3. Install dependencies and set up pre-commit hooks
|
||||
|
||||
```bash
|
||||
pnpm install
|
||||
pnpm run prepare # Sets up Husky pre-commit hooks for code formatting
|
||||
```
|
||||
|
||||
4. Create an env file
|
||||
|
||||
```bash
|
||||
cp .env.dev.example .env
|
||||
```
|
||||
|
||||
4. Run the entire infrastructure in dev mode
|
||||
4. Run the entire infrastructure in dev mode. **Note**: if you have an existing database, this command wipes it. Also, this will fail on the very first run. Please run it again.
|
||||
|
||||
```bash
|
||||
pnpm run dx
|
||||
pnpm run dx # first run only (resets db, docker containers, etc...)
|
||||
pnpm run dev # any subsequent runs
|
||||
```
|
||||
|
||||
You will be asked whether you want to reset Postgres and ClickHouse. Confirm both with 'Y' and press enter.
|
||||
@@ -134,6 +156,12 @@ Requirements
|
||||
- Username: `demo@langfuse.com`
|
||||
- Password: `password`
|
||||
|
||||
To get comprehensive example data, you can use the `seed` command:
|
||||
|
||||
```sh
|
||||
pnpm run db:seed:examples
|
||||
```
|
||||
|
||||
## Monorepo quickstart
|
||||
|
||||
- Available packages and their dependencies
|
||||
@@ -183,23 +211,65 @@ Requirements
|
||||
|
||||
On the main branch, we adhere to the best practices of [conventional commits](https://www.conventionalcommits.org/en/v1.0.0/). All pull requests and branches are squash-merged to maintain a clean and readable history. This approach ensures the addition of a conventional commit message when merging contributions.
|
||||
|
||||
## Test the public API
|
||||
## Running Unit Tests
|
||||
|
||||
The API is tested using Jest. With the development server running, you can run the tests with:
|
||||
All tests run in the CI and must pass before merging.
|
||||
All tests run against a running langfuse instance and **write/delete real data from the database**.
|
||||
|
||||
Run all
|
||||
### Test Database Setup
|
||||
|
||||
Per default, the tests use the local development database. Therefore, wiping your data in the process.
|
||||
For proper test isolation, create a `.env.test` file in the root directory:
|
||||
|
||||
```bash
|
||||
npm run test
|
||||
cp .env.test.example .env.test
|
||||
```
|
||||
|
||||
Run interactively in watch mode
|
||||
Then, a different PostgreSQL and Redis are used for the tests.
|
||||
The `.env.test` file only overrides the set values and falls back on `.env` for all undefined values.
|
||||
|
||||
```bash
|
||||
npm run test:watch
|
||||
- **PostgreSQL**: Uses separate `langfuse_test` database for isolation
|
||||
- **ClickHouse**: Uses shared `default` database for now
|
||||
- **Redis**: Uses database 1 instead of 0 for isolation (Redis data is not cleaned between tests)
|
||||
|
||||
Tests automatically create the PostgreSQL test database if it doesn't exist and clean up data between runs.
|
||||
|
||||
### Tests in the `web` package (public API)
|
||||
|
||||
We're using Jest with in the `web` package. Therefore, if you want to provide an argument to the test runner, do it directly without an intermittent `--`.
|
||||
|
||||
There are three types of unit tests:
|
||||
|
||||
- `test-sync`
|
||||
- `test-async`
|
||||
- `test-client`
|
||||
|
||||
To run a specific test, for example the test: `"should handle special characters in prompt names"` in `prompts.v2.servertest.ts`, run:
|
||||
|
||||
```sh
|
||||
cd web # or with --filter=web
|
||||
pnpm test-sync --testPathPattern="prompts\.v2\.servertest" --testNamePattern="should handle special characters in prompt names"
|
||||
```
|
||||
|
||||
These tests are also run in CI.
|
||||
To run all tests:
|
||||
|
||||
```sh
|
||||
pnpm run test
|
||||
```
|
||||
|
||||
Run interactively in watch mode (not recommended!)
|
||||
|
||||
```sh
|
||||
pnpm run test:watch
|
||||
```
|
||||
|
||||
### Tests in the `worker` package
|
||||
|
||||
For the `worker` package, we're using `vitest` to run unit tests.
|
||||
|
||||
```sh
|
||||
pnpm run test --filter=worker -- FILE_YOU_WANT_TO_TEST.ts -t "test name"
|
||||
```
|
||||
|
||||
## CI/CD
|
||||
|
||||
@@ -327,6 +397,20 @@ Please note that
|
||||
|
||||
Until the V3 release, both the JSON record must be updated **and** a migration must be created to continue supporting self-hosted users. Note that the migration must updated both the `models` as well as the `prices` table accordingly.
|
||||
|
||||
## Updating the OpenAPI Specs & fern SDKs
|
||||
|
||||
We maintain the API specifications manually to guarantee a high degree of understandability. If you made changes to the API, please update the respective `.yml` files in `fern/apis/...`.
|
||||
|
||||
To generate the respective `openapi.yml` files which power the online API reference & SDKs, run:
|
||||
|
||||
```sh
|
||||
npx fern-api generate --api server # for the server API
|
||||
npx fern-api generate --api client # for the client API
|
||||
npx fern-api generate --api organizations # for the organizations API
|
||||
```
|
||||
|
||||
**Note:** You need a signed in fern account to run those commands.
|
||||
|
||||
## License
|
||||
|
||||
Langfuse is MIT licensed, except for `ee/` folder. See [LICENSE](LICENSE) and [docs](https://langfuse.com/docs/open-source) for more details.
|
||||
|
||||
+9
-8
@@ -26,7 +26,7 @@
|
||||
<a href="https://langfuse.com/roadmap"><strong>路线图</strong></a> ·
|
||||
</div>
|
||||
<br/>
|
||||
<span>Langfuse 使用 <a href="https://github.com/orgs/langfuse/discussions"><strong>Github Discussions</strong></a> 作为支持和功能请求的平台。</span>
|
||||
<span>Langfuse 使用 <a href="https://github.com/orgs/langfuse/discussions"><strong>GitHub Discussions</strong></a> 作为支持和功能请求的平台。</span>
|
||||
<br/>
|
||||
<span><b>我们正在招聘。</b> <a href="https://langfuse.com/careers"><strong>加入我们</strong></a>,从事产品工程和技术市场职位。</span>
|
||||
<br/>
|
||||
@@ -94,7 +94,7 @@ Langfuse 是一个 **开源 LLM 工程** 平台。它帮助团队协作 **开发
|
||||
|
||||
- [提示管理](https://langfuse.com/docs/prompts/get-started) 帮助你集中管理、版本控制并协作迭代提示。得益于服务器和客户端的高效缓存,你可以在不增加延迟的情况下反复迭代提示。
|
||||
|
||||
- [评估](https://langfuse.com/docs/scores/overview) 是 LLM 应用开发流程的关键组成部分,Langfuse 能够满足你的多样需求。它支持 LLM 作为“裁判”、用户反馈收集、手动标注以及通过 API/SDK 实现自定义评估流程。
|
||||
- [评估](https://langfuse.com/docs/scores/overview) 是 LLM 应用开发流程的关键组成部分,Langfuse 能够满足你的多样需求。它支持 LLM 作为"裁判"、用户反馈收集、手动标注以及通过 API/SDK 实现自定义评估流程。
|
||||
|
||||
- [数据集](https://langfuse.com/docs/datasets/overview) 为评估你的 LLM 应用提供测试集和基准。它们支持持续改进、部署前测试、结构化实验、灵活评估,并能与 LangChain、LlamaIndex 等框架无缝整合。
|
||||
|
||||
@@ -135,7 +135,7 @@ Langfuse 是一个 **开源 LLM 工程** 平台。它帮助团队协作 **开发
|
||||
|
||||
- [虚拟机](https://langfuse.com/self-hosting/docker-compose):使用 Docker Compose 在单台虚拟机上部署 Langfuse。
|
||||
|
||||
- 【计划中】:针对各云平台的部署指南,欢迎在以下讨论中投票和评论:[AWS](https://github.com/orgs/langfuse/discussions/4645)、[Google Cloud](https://github.com/langfuse/discussions/4646)、[Azure](https://github.com/orgs/langfuse/discussions/4647)。
|
||||
- Terraform 模板: [AWS](https://langfuse.com/self-hosting/aws)、[Azure](https://langfuse.com/self-hosting/azure)、[GCP](https://langfuse.com/self-hosting/gcp)
|
||||
|
||||
请参阅 [自托管文档](https://langfuse.com/self-hosting) 了解更多关于架构和配置选项的信息。
|
||||
|
||||
@@ -155,6 +155,7 @@ Langfuse 是一个 **开源 LLM 工程** 平台。它帮助团队协作 **开发
|
||||
| [LiteLLM](https://langfuse.com/docs/integrations/litellm) | Python, JS/TS (仅代理) | 允许使用任何 LLM 替代 GPT。支持 Azure、OpenAI、Cohere、Anthropic、Ollama、VLLM、Sagemaker、HuggingFace、Replicate(100+ LLMs)。 |
|
||||
| [Vercel AI SDK](https://langfuse.com/docs/integrations/vercel-ai-sdk) | JS/TS | 基于 TypeScript 的工具包,帮助开发者使用 React、Next.js、Vue、Svelte 和 Node.js 构建 AI 驱动的应用。 |
|
||||
| [API](https://langfuse.com/docs/api) | | 直接调用公共 API。提供 OpenAPI 规格。 |
|
||||
| [Google VertexAI 和 Gemini](https://langfuse.com/docs/integrations/google-vertex-ai) | 模型 | 在 Google 上运行基础模型和微调模型。 |
|
||||
|
||||
### 与 Langfuse 集成的软件包:
|
||||
|
||||
@@ -162,9 +163,9 @@ Langfuse 是一个 **开源 LLM 工程** 平台。它帮助团队协作 **开发
|
||||
| ---------------------------------------------------------------------- | ------------------- | -------------------------------------------------------------------------------------------------------- |
|
||||
| [Instructor](https://langfuse.com/docs/integrations/instructor) | 库 | 用于获取结构化 LLM 输出(JSON、Pydantic)的库。 |
|
||||
| [DSPy](https://langfuse.com/docs/integrations/dspy) | 库 | 一个系统性优化语言模型提示和权重的框架。 |
|
||||
| [Amazon Bedrock](https://langfuse.com/docs/integrations/amazon-bedrock) | 模型 | 在 AWS 上运行基础模型和微调模型。 |
|
||||
| [Mirascope](https://langfuse.com/docs/integrations/mirascope) | 库 | 构建 LLM 应用的 Python 工具包。 |
|
||||
| [Ollama](https://langfuse.com/docs/integrations/ollama) | 模型(本地) | 在你的机器上轻松运行开源 LLM。 |
|
||||
| [Amazon Bedrock](https://langfuse.com/docs/integrations/amazon-bedrock) | 模型 | 在 AWS 上运行基础模型和微调模型。 |
|
||||
| [AutoGen](https://langfuse.com/docs/integrations/autogen) | 代理框架 | 用于构建分布式代理的开源 LLM 平台。 |
|
||||
| [Flowise](https://langfuse.com/docs/integrations/flowise) | 聊天/代理界面 | 基于 JS/TS 的无代码构建器,用于定制化 LLM 流程。 |
|
||||
| [Langflow](https://langfuse.com/docs/integrations/langflow) | 聊天/代理界面 | 基于 Python 的 LangChain 用户界面,采用 react-flow 设计,提供便捷的实验与原型构建体验。 |
|
||||
@@ -213,8 +214,8 @@ LANGFUSE_HOST="https://cloud.langfuse.com" # 🇪🇺 欧盟区域
|
||||
|
||||
创建示例代码(文件名:**main.py**):
|
||||
|
||||
````python:main.py
|
||||
from langfuse.decorators import observe
|
||||
```python:main.py
|
||||
from langfuse import observe
|
||||
from langfuse.openai import openai # OpenAI 集成
|
||||
|
||||
@observe()
|
||||
@@ -253,7 +254,7 @@ _[Langfuse 中的公共示例追踪](https://cloud.langfuse.com/project/cloramnk
|
||||
|
||||
- 我们的 [文档](https://langfuse.com/docs) 是查找答案的最佳起点。内容全面,我们投入大量时间进行维护。你也可以通过 GitHub 提出文档修改建议。
|
||||
- [Langfuse 常见问题](https://langfuse.com/faq) 解答了最常见的问题。
|
||||
- 使用 “[Ask AI](https://langfuse.com/docs/ask-ai)” 立即获取问题答案。
|
||||
- 使用 "Ask AI" 立即获取问题答案。
|
||||
|
||||
支持渠道:
|
||||
|
||||
@@ -351,4 +352,4 @@ _[Langfuse 中的公共示例追踪](https://cloud.langfuse.com/project/cloramnk
|
||||
所有数据均不会与第三方共享,也不包含任何敏感信息。我们对这一过程保持高度透明,你可以在 [此处](/web/src/features/telemetry/index.ts) 查看我们收集的具体数据。
|
||||
|
||||
你可以通过设置 `TELEMETRY_ENABLED=false` 来选择退出。
|
||||
````
|
||||
```
|
||||
|
||||
+3
-5
@@ -26,7 +26,7 @@
|
||||
<a href="https://langfuse.com/roadmap"><strong>ロードマップ</strong></a> ·
|
||||
</div>
|
||||
<br/>
|
||||
<span>Langfuseは、サポートと機能リクエストのために <a href="https://github.com/orgs/langfuse/discussions"><strong>Github Discussions</strong></a> を利用しています。</span>
|
||||
<span>Langfuseは、サポートと機能リクエストのために <a href="https://github.com/orgs/langfuse/discussions"><strong>GitHub Discussions</strong></a> を利用しています。</span>
|
||||
<br/>
|
||||
<span><b>We're hiring.</b> <a href="https://langfuse.com/careers"><strong>チームに加わる</strong></a> (製品エンジニアリングおよびテクニカルGTMのポジション)への応募をお待ちしています。</span>
|
||||
<br/>
|
||||
@@ -141,9 +141,7 @@ Langfuseチームによるマネージドデプロイメント。充実した無
|
||||
- **[VM](https://langfuse.com/self-hosting/docker-compose):**
|
||||
Docker Composeを使用して、単一の仮想マシン上でLangfuseを実行します。
|
||||
|
||||
- **Planned:**
|
||||
クラウド固有のデプロイガイドは計画中です。以下のスレッドに対して投票やコメントをお願いします:
|
||||
[AWS](https://github.com/orgs/langfuse/discussions/4645), [Google Cloud](https://github.com/orgs/langfuse/discussions/4646), [Azure](https://github.com/orgs/langfuse/discussions/4647).
|
||||
- Terraform テンプレート: [AWS](https://langfuse.com/self-hosting/aws), [Azure](https://langfuse.com/self-hosting/azure), [GCP](https://langfuse.com/self-hosting/gcp)
|
||||
|
||||
[セルフホスティングのドキュメント](https://langfuse.com/self-hosting)を参照し、アーキテクチャや設定オプションの詳細をご確認ください。
|
||||
|
||||
@@ -219,7 +217,7 @@ LANGFUSE_HOST="https://cloud.langfuse.com" # 🇪🇺 EUリージョン
|
||||
```
|
||||
|
||||
```python:/@observe()/ /from langfuse.openai import openai/ filename="main.py"
|
||||
from langfuse.decorators import observe
|
||||
from langfuse import observe
|
||||
from langfuse.openai import openai # OpenAI統合
|
||||
|
||||
@observe()
|
||||
|
||||
+3
-2
@@ -128,7 +128,7 @@ Langfuse 팀이 관리하는 배포 방식으로, 후한 무료 플랜(취미
|
||||
|
||||
- [Kubernetes (Helm)](https://langfuse.com/self-hosting/kubernetes-helm): Helm을 사용해 Kubernetes 클러스터에서 Langfuse를 실행합니다. 이는 권장되는 프로덕션 배포 방식입니다.
|
||||
- [VM](https://langfuse.com/self-hosting/docker-compose): Docker Compose를 사용해 단일 가상 머신에서 Langfuse를 실행합니다.
|
||||
- 예정: 클라우드별 배포 가이드 – 아래 스레드에서 투표 및 댓글을 남겨주세요: [AWS](https://github.com/orgs/langfuse/discussions/4645), [Google Cloud](https://github.com/orgs/langfuse/discussions/4646), [Azure](https://github.com/orgs/langfuse/discussions/4647).
|
||||
- Terraform 템플릿: [AWS](https://langfuse.com/self-hosting/aws), [Azure](https://langfuse.com/self-hosting/azure), [GCP](https://langfuse.com/self-hosting/gcp)
|
||||
|
||||
자세한 내용은 [자체 호스팅 문서](https://langfuse.com/self-hosting)를 참조하세요.
|
||||
|
||||
@@ -158,6 +158,7 @@ Langfuse 팀이 관리하는 배포 방식으로, 후한 무료 플랜(취미
|
||||
| [Mirascope](https://langfuse.com/docs/integrations/mirascope) | 라이브러리 | LLM 애플리케이션 구축을 위한 Python 툴킷입니다. |
|
||||
| [Ollama](https://langfuse.com/docs/integrations/ollama) | 모델 (로컬) | 자신의 컴퓨터에서 오픈 소스 LLM을 손쉽게 실행할 수 있습니다. |
|
||||
| [Amazon Bedrock](https://langfuse.com/docs/integrations/amazon-bedrock) | 모델 | AWS에서 기본 및 파인튜닝된 모델을 실행합니다. |
|
||||
| [Google VertexAI and Gemini](https://langfuse.com/docs/integrations/google-vertex-ai) | 모델 | Google에서 기본 및 파인튜닝된 모델을 실행합니다. |
|
||||
| [AutoGen](https://langfuse.com/docs/integrations/autogen) | 에이전트 프레임워크 | 분산 에이전트 구축을 위한 오픈 소스 LLM 플랫폼입니다. |
|
||||
| [Flowise](https://langfuse.com/docs/integrations/flowise) | 채팅/에이전트 UI | 맞춤형 LLM 플로우를 위한 JS/TS 코드 없는(no-code) 빌더입니다. |
|
||||
| [Langflow](https://langfuse.com/docs/integrations/langflow) | 채팅/에이전트 UI | react-flow를 활용하여 실험 및 프로토타이핑을 손쉽게 할 수 있도록 디자인된 LangChain용 Python 기반 UI입니다. |
|
||||
@@ -201,7 +202,7 @@ LANGFUSE_HOST="https://cloud.langfuse.com" # 🇪🇺 EU region
|
||||
```
|
||||
|
||||
```python:main.py
|
||||
from langfuse.decorators import observe
|
||||
from langfuse import observe
|
||||
from langfuse.openai import openai # OpenAI integration
|
||||
|
||||
@observe()
|
||||
|
||||
@@ -3,6 +3,9 @@
|
||||
<div align="center">
|
||||
<div>
|
||||
<h3>
|
||||
<a href="https://langfuse.com/blog/2025-06-04-open-sourcing-langfuse-product">
|
||||
<strong>Langfuse Is Doubling Down On Open Source</strong>
|
||||
</a> <br> <br>
|
||||
<a href="https://cloud.langfuse.com">
|
||||
<strong>Langfuse Cloud</strong>
|
||||
</a> ·
|
||||
@@ -23,7 +26,7 @@
|
||||
<a href="https://langfuse.com/roadmap"><strong>Roadmap</strong></a> ·
|
||||
</div>
|
||||
<br/>
|
||||
<span>Langfuse uses <a href="https://github.com/orgs/langfuse/discussions"><strong>Github Discussions</strong></a> for Support and Feature Requests.</span>
|
||||
<span>Langfuse uses <a href="https://github.com/orgs/langfuse/discussions"><strong>GitHub Discussions</strong></a> for Support and Feature Requests.</span>
|
||||
<br/>
|
||||
<span><b>We're hiring.</b> <a href="https://langfuse.com/careers"><strong>Join us</strong></a> in product engineering and technical go-to-market roles.</span>
|
||||
<br/>
|
||||
@@ -93,7 +96,7 @@ Langfuse is an **open source LLM engineering** platform. It helps teams collabor
|
||||
|
||||
### Langfuse Cloud
|
||||
|
||||
Managed deployment by the Langfuse team, generous free-tier (hobby plan), no credit card required.
|
||||
Managed deployment by the Langfuse team, generous free-tier, no credit card required.
|
||||
|
||||
<div align="center">
|
||||
<a href="https://cloud.langfuse.com" target="_blank">
|
||||
@@ -115,12 +118,11 @@ Run Langfuse on your own infrastructure:
|
||||
# Run the langfuse docker compose
|
||||
docker compose up
|
||||
```
|
||||
|
||||
- [Kubernetes (Helm)](https://langfuse.com/self-hosting/kubernetes-helm): Run Langfuse on a Kubernetes cluster using Helm. This is the preferred production deployment.
|
||||
- [VM](https://langfuse.com/self-hosting/docker-compose): Run Langfuse on a single Virtual Machine using Docker Compose.
|
||||
- Planned: Cloud-specific deployment guides, please upvote and comment on the following threads: [AWS](https://github.com/orgs/langfuse/discussions/4645), [Google Cloud](https://github.com/orgs/langfuse/discussions/4646), [Azure](https://github.com/orgs/langfuse/discussions/4647).
|
||||
- [Kubernetes (Helm)](https://langfuse.com/self-hosting/kubernetes-helm): Run Langfuse on a Kubernetes cluster using Helm. This is the preferred production deployment.
|
||||
- Terraform Templates: [AWS](https://langfuse.com/self-hosting/aws), [Azure](https://langfuse.com/self-hosting/azure), [GCP](https://langfuse.com/self-hosting/gcp)
|
||||
|
||||
See [self-hosting documentation](https://langfuse.com/self-hosting) to learn more about the architecture and configuration options.
|
||||
See [self-hosting documentation](https://langfuse.com/self-hosting) to learn more about architecture and configuration options.
|
||||
|
||||
## 🔌 Integrations
|
||||
|
||||
@@ -191,7 +193,7 @@ LANGFUSE_HOST="https://cloud.langfuse.com" # 🇪🇺 EU region
|
||||
```
|
||||
|
||||
```python /@observe()/ /from langfuse.openai import openai/ filename="main.py"
|
||||
from langfuse.decorators import observe
|
||||
from langfuse import observe
|
||||
from langfuse.openai import openai # OpenAI integration
|
||||
|
||||
@observe()
|
||||
|
||||
@@ -78,7 +78,7 @@ services:
|
||||
retries: 3
|
||||
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server
|
||||
image: docker.io/clickhouse/clickhouse-server
|
||||
user: "101:101"
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
@@ -98,7 +98,7 @@ services:
|
||||
start_period: 1s
|
||||
|
||||
minio:
|
||||
image: minio/minio
|
||||
image: docker.io/minio/minio
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
|
||||
@@ -118,7 +118,7 @@ services:
|
||||
start_period: 1s
|
||||
|
||||
redis:
|
||||
image: redis:7
|
||||
image: docker.io/redis:7
|
||||
restart: always
|
||||
command: >
|
||||
--requirepass ${REDIS_AUTH:-myredissecret}
|
||||
@@ -131,7 +131,7 @@ services:
|
||||
retries: 10
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
image: docker.io/postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server:24.3
|
||||
image: docker.io/clickhouse/clickhouse-server:24.3
|
||||
user: "101:101"
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
@@ -24,7 +24,7 @@ services:
|
||||
- langfuse_azurite_data:/data
|
||||
|
||||
redis:
|
||||
image: redis:7.2.4
|
||||
image: docker.io/redis:7.2.4
|
||||
restart: always
|
||||
command: >
|
||||
--requirepass ${REDIS_AUTH:-myredissecret}
|
||||
@@ -32,7 +32,7 @@ services:
|
||||
- 6379:6379
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
image: docker.io/postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
|
||||
@@ -0,0 +1,146 @@
|
||||
services:
|
||||
clickhouse:
|
||||
image: docker.io/clickhouse/clickhouse-server:24.3
|
||||
user: "101:101"
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
CLICKHOUSE_USER: clickhouse
|
||||
CLICKHOUSE_PASSWORD: clickhouse
|
||||
volumes:
|
||||
- langfuse_clickhouse_data:/var/lib/clickhouse
|
||||
- langfuse_clickhouse_logs:/var/log/clickhouse-server
|
||||
ports:
|
||||
- 127.0.0.1:8123:8123
|
||||
- 127.0.0.1:9000:9000
|
||||
depends_on:
|
||||
- postgres
|
||||
|
||||
minio:
|
||||
image: docker.io/minio/minio
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
|
||||
environment:
|
||||
MINIO_ACCESS_KEY: minio
|
||||
MINIO_SECRET_KEY: miniosecret
|
||||
ports:
|
||||
- 127.0.0.1:9090:9000
|
||||
- 127.0.0.1:9091:9001
|
||||
volumes:
|
||||
- langfuse_minio_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "mc", "ready", "local"]
|
||||
interval: 1s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 1s
|
||||
|
||||
postgres:
|
||||
image: docker.io/postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
interval: 3s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
command: ["postgres", "-c", "log_statement=all"]
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
- POSTGRES_PASSWORD=postgres
|
||||
- POSTGRES_DB=postgres
|
||||
ports:
|
||||
- 127.0.0.1:5432:5432
|
||||
volumes:
|
||||
- langfuse_postgres_data:/var/lib/postgresql/data
|
||||
|
||||
# Redis Cluster. Requires a Unix host to work for network_mode: host. I.e. tests will fail on windows and mac with this setup.
|
||||
redis-node-0:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-0:/bitnami/redis/data
|
||||
environment:
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6370
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
|
||||
redis-node-1:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-1:/bitnami/redis/data
|
||||
environment:
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6371
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
|
||||
redis-node-2:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-2:/bitnami/redis/data
|
||||
environment:
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6372
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
|
||||
redis-node-3:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-3:/bitnami/redis/data
|
||||
environment:
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6373
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
|
||||
redis-node-4:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-4:/bitnami/redis/data
|
||||
environment:
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6374
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
|
||||
redis-node-5:
|
||||
image: docker.io/bitnami/redis-cluster:8.0
|
||||
network_mode: host
|
||||
volumes:
|
||||
- redis-cluster_data-5:/bitnami/redis/data
|
||||
depends_on:
|
||||
- redis-node-0
|
||||
- redis-node-1
|
||||
- redis-node-2
|
||||
- redis-node-3
|
||||
- redis-node-4
|
||||
environment:
|
||||
REDISCLI_AUTH: bitnami
|
||||
REDIS_CLUSTER_REPLICAS: 1
|
||||
REDIS_PASSWORD: bitnami
|
||||
REDIS_PORT_NUMBER: 6375
|
||||
REDIS_NODES: 127.0.0.1:6370 127.0.0.1:6371 127.0.0.1:6372 127.0.0.1:6373 127.0.0.1:6374 127.0.0.1:6375
|
||||
REDIS_CLUSTER_CREATOR: yes
|
||||
|
||||
volumes:
|
||||
langfuse_postgres_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_data:
|
||||
driver: local
|
||||
langfuse_clickhouse_logs:
|
||||
driver: local
|
||||
langfuse_minio_data:
|
||||
driver: local
|
||||
redis-cluster_data-0:
|
||||
driver: local
|
||||
redis-cluster_data-1:
|
||||
driver: local
|
||||
redis-cluster_data-2:
|
||||
driver: local
|
||||
redis-cluster_data-3:
|
||||
driver: local
|
||||
redis-cluster_data-4:
|
||||
driver: local
|
||||
redis-cluster_data-5:
|
||||
driver: local
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server:24.3
|
||||
image: docker.io/clickhouse/clickhouse-server:24.3
|
||||
user: "101:101"
|
||||
environment:
|
||||
CLICKHOUSE_DB: default
|
||||
@@ -16,7 +16,7 @@ services:
|
||||
- postgres
|
||||
|
||||
minio:
|
||||
image: minio/minio
|
||||
image: docker.io/minio/minio
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
command: -c 'mkdir -p /data/langfuse && minio server --address ":9000" --console-address ":9001" /data'
|
||||
@@ -36,7 +36,7 @@ services:
|
||||
start_period: 1s
|
||||
|
||||
redis:
|
||||
image: redis:7.2.4
|
||||
image: docker.io/redis:7.2.4
|
||||
restart: always
|
||||
command: >
|
||||
--requirepass ${REDIS_AUTH:-myredissecret}
|
||||
@@ -44,7 +44,7 @@ services:
|
||||
- 127.0.0.1:6379:6379
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
image: docker.io/postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
|
||||
+10
-7
@@ -5,7 +5,7 @@
|
||||
# External connections from other machines will not be able to reach these services directly.
|
||||
services:
|
||||
langfuse-worker:
|
||||
image: langfuse/langfuse-worker:3
|
||||
image: docker.io/langfuse/langfuse-worker:3
|
||||
restart: always
|
||||
depends_on: &langfuse-depends-on
|
||||
postgres:
|
||||
@@ -19,6 +19,7 @@ services:
|
||||
ports:
|
||||
- 127.0.0.1:3030:3030
|
||||
environment: &langfuse-worker-env
|
||||
NEXTAUTH_URL: http://localhost:3000
|
||||
DATABASE_URL: postgresql://postgres:postgres@postgres:5432/postgres # CHANGEME
|
||||
SALT: "mysalt" # CHANGEME
|
||||
ENCRYPTION_KEY: "0000000000000000000000000000000000000000000000000000000000000000" # CHANGEME: generate via `openssl rand -hex 32`
|
||||
@@ -29,6 +30,7 @@ services:
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USER:-clickhouse}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-clickhouse} # CHANGEME
|
||||
CLICKHOUSE_CLUSTER_ENABLED: ${CLICKHOUSE_CLUSTER_ENABLED:-false}
|
||||
LANGFUSE_USE_AZURE_BLOB: ${LANGFUSE_USE_AZURE_BLOB:-false}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_BUCKET: ${LANGFUSE_S3_EVENT_UPLOAD_BUCKET:-langfuse}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_REGION: ${LANGFUSE_S3_EVENT_UPLOAD_REGION:-auto}
|
||||
LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID: ${LANGFUSE_S3_EVENT_UPLOAD_ACCESS_KEY_ID:-minio}
|
||||
@@ -61,16 +63,17 @@ services:
|
||||
REDIS_TLS_CA: ${REDIS_TLS_CA:-/certs/ca.crt}
|
||||
REDIS_TLS_CERT: ${REDIS_TLS_CERT:-/certs/redis.crt}
|
||||
REDIS_TLS_KEY: ${REDIS_TLS_KEY:-/certs/redis.key}
|
||||
EMAIL_FROM_ADDRESS: ${EMAIL_FROM_ADDRESS:-}
|
||||
SMTP_CONNECTION_URL: ${SMTP_CONNECTION_URL:-}
|
||||
|
||||
langfuse-web:
|
||||
image: langfuse/langfuse:3
|
||||
image: docker.io/langfuse/langfuse:3
|
||||
restart: always
|
||||
depends_on: *langfuse-depends-on
|
||||
ports:
|
||||
- 3000:3000
|
||||
environment:
|
||||
<<: *langfuse-worker-env
|
||||
NEXTAUTH_URL: http://localhost:3000
|
||||
NEXTAUTH_SECRET: mysecret # CHANGEME
|
||||
LANGFUSE_INIT_ORG_ID: ${LANGFUSE_INIT_ORG_ID:-}
|
||||
LANGFUSE_INIT_ORG_NAME: ${LANGFUSE_INIT_ORG_NAME:-}
|
||||
@@ -83,7 +86,7 @@ services:
|
||||
LANGFUSE_INIT_USER_PASSWORD: ${LANGFUSE_INIT_USER_PASSWORD:-}
|
||||
|
||||
clickhouse:
|
||||
image: clickhouse/clickhouse-server
|
||||
image: docker.io/clickhouse/clickhouse-server
|
||||
restart: always
|
||||
user: "101:101"
|
||||
environment:
|
||||
@@ -104,7 +107,7 @@ services:
|
||||
start_period: 1s
|
||||
|
||||
minio:
|
||||
image: minio/minio
|
||||
image: docker.io/minio/minio
|
||||
restart: always
|
||||
entrypoint: sh
|
||||
# create the 'langfuse' bucket before starting the service
|
||||
@@ -125,7 +128,7 @@ services:
|
||||
start_period: 1s
|
||||
|
||||
redis:
|
||||
image: redis:7
|
||||
image: docker.io/redis:7
|
||||
restart: always
|
||||
# CHANGEME: row below to secure redis password
|
||||
command: >
|
||||
@@ -139,7 +142,7 @@ services:
|
||||
retries: 10
|
||||
|
||||
postgres:
|
||||
image: postgres:${POSTGRES_VERSION:-latest}
|
||||
image: docker.io/postgres:${POSTGRES_VERSION:-latest}
|
||||
restart: always
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U postgres"]
|
||||
|
||||
+3
-9
@@ -26,11 +26,10 @@
|
||||
"dependencies": {
|
||||
"@langfuse/shared": "workspace:*",
|
||||
"@opentelemetry/api": ">=1.0.0 <1.10.0",
|
||||
"axios": "^1.8.2",
|
||||
"https-proxy-agent": "^7.0.6",
|
||||
"next": "^14.2.26",
|
||||
"next": "^14.2.30",
|
||||
"next-auth": "^4.24.11",
|
||||
"zod": "^3.24.4"
|
||||
"zod": "^3.25.62"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@repo/eslint-config": "workspace:*",
|
||||
@@ -41,14 +40,9 @@
|
||||
"eslint-config-prettier": "^9.1.0",
|
||||
"eslint-config-standard": "^17.1.0",
|
||||
"eslint-plugin-prettier": "^5.1.3",
|
||||
"prettier": "^3.3.3",
|
||||
"prettier": "^3.6.2",
|
||||
"ts-node": "^10.9.2",
|
||||
"tsc-watch": "^6.2.0",
|
||||
"typescript": "^5.4.5"
|
||||
},
|
||||
"pnpm": {
|
||||
"overrides": {
|
||||
"nanoid": "^3.3.8"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
import { z } from "zod";
|
||||
import { z } from "zod/v4";
|
||||
import { removeEmptyEnvVariables } from "@langfuse/shared";
|
||||
|
||||
const EnvSchema = z.object({
|
||||
|
||||
@@ -22,6 +22,13 @@ service:
|
||||
docs: limit of items per page
|
||||
response: PaginatedAnnotationQueues
|
||||
|
||||
createQueue:
|
||||
docs: Create an annotation queue
|
||||
method: POST
|
||||
path: /annotation-queues
|
||||
request: CreateAnnotationQueueRequest
|
||||
response: AnnotationQueue
|
||||
|
||||
getQueue:
|
||||
docs: Get an annotation queue by ID
|
||||
method: GET
|
||||
@@ -105,6 +112,28 @@ service:
|
||||
docs: The unique identifier of the annotation queue item
|
||||
response: DeleteAnnotationQueueItemResponse
|
||||
|
||||
createQueueAssignment:
|
||||
docs: Create an assignment for a user to an annotation queue
|
||||
method: POST
|
||||
path: /annotation-queues/{queueId}/assignments
|
||||
path-parameters:
|
||||
queueId:
|
||||
type: string
|
||||
docs: The unique identifier of the annotation queue
|
||||
request: AnnotationQueueAssignmentRequest
|
||||
response: CreateAnnotationQueueAssignmentResponse
|
||||
|
||||
deleteQueueAssignment:
|
||||
docs: Delete an assignment for a user to an annotation queue
|
||||
method: DELETE
|
||||
path: /annotation-queues/{queueId}/assignments
|
||||
path-parameters:
|
||||
queueId:
|
||||
type: string
|
||||
docs: The unique identifier of the annotation queue
|
||||
request: AnnotationQueueAssignmentRequest
|
||||
response: DeleteAnnotationQueueAssignmentResponse
|
||||
|
||||
types:
|
||||
AnnotationQueueStatus:
|
||||
enum:
|
||||
@@ -146,6 +175,12 @@ types:
|
||||
data: list<AnnotationQueueItem>
|
||||
meta: pagination.MetaResponse
|
||||
|
||||
CreateAnnotationQueueRequest:
|
||||
properties:
|
||||
name: string
|
||||
description: optional<string>
|
||||
scoreConfigIds: list<string>
|
||||
|
||||
CreateAnnotationQueueItemRequest:
|
||||
properties:
|
||||
objectId: string
|
||||
@@ -163,3 +198,17 @@ types:
|
||||
properties:
|
||||
success: boolean
|
||||
message: string
|
||||
|
||||
AnnotationQueueAssignmentRequest:
|
||||
properties:
|
||||
userId: string
|
||||
|
||||
DeleteAnnotationQueueAssignmentResponse:
|
||||
properties:
|
||||
success: boolean
|
||||
|
||||
CreateAnnotationQueueAssignmentResponse:
|
||||
properties:
|
||||
userId: string
|
||||
queueId: string
|
||||
projectId: string
|
||||
|
||||
@@ -27,7 +27,7 @@ service:
|
||||
limit:
|
||||
type: optional<integer>
|
||||
docs: limit of items per page
|
||||
response: PaginatedDatasetRunItems
|
||||
response: PaginatedDatasetRunItems
|
||||
|
||||
types:
|
||||
CreateDatasetRunItemRequest:
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
# yaml-language-server: $schema=https://raw.githubusercontent.com/fern-api/fern/main/fern.schema.json
|
||||
imports:
|
||||
commons: ./commons.yml
|
||||
pagination: ./utils/pagination.yml
|
||||
service:
|
||||
auth: true
|
||||
base-path: /api/public
|
||||
endpoints:
|
||||
list:
|
||||
method: GET
|
||||
docs: Get all LLM connections in a project
|
||||
path: /llm-connections
|
||||
request:
|
||||
name: GetLlmConnectionsRequest
|
||||
query-parameters:
|
||||
page:
|
||||
type: optional<integer>
|
||||
docs: page number, starts at 1
|
||||
limit:
|
||||
type: optional<integer>
|
||||
docs: limit of items per page
|
||||
response: PaginatedLlmConnections
|
||||
upsert:
|
||||
method: PUT
|
||||
docs: Create or update an LLM connection. The connection is upserted on provider.
|
||||
path: /llm-connections
|
||||
request: UpsertLlmConnectionRequest
|
||||
response: LlmConnection
|
||||
|
||||
types:
|
||||
LlmConnection:
|
||||
docs: LLM API connection configuration (secrets excluded)
|
||||
properties:
|
||||
id: string
|
||||
provider:
|
||||
type: string
|
||||
docs: Provider name (e.g., 'openai', 'my-gateway'). Must be unique in project, used for upserting.
|
||||
adapter:
|
||||
type: string
|
||||
docs: The adapter used to interface with the LLM
|
||||
displaySecretKey:
|
||||
type: string
|
||||
docs: Masked version of the secret key for display purposes
|
||||
baseURL:
|
||||
type: optional<string>
|
||||
docs: Custom base URL for the LLM API
|
||||
customModels:
|
||||
type: list<string>
|
||||
docs: List of custom model names available for this connection
|
||||
withDefaultModels:
|
||||
type: boolean
|
||||
docs: Whether to include default models for this adapter
|
||||
extraHeaderKeys:
|
||||
type: list<string>
|
||||
docs: Keys of extra headers sent with requests (values excluded for security)
|
||||
createdAt: datetime
|
||||
updatedAt: datetime
|
||||
|
||||
PaginatedLlmConnections:
|
||||
properties:
|
||||
data: list<LlmConnection>
|
||||
meta: pagination.MetaResponse
|
||||
|
||||
UpsertLlmConnectionRequest:
|
||||
docs: Request to create or update an LLM connection (upsert)
|
||||
properties:
|
||||
provider:
|
||||
type: string
|
||||
docs: Provider name (e.g., 'openai', 'my-gateway'). Must be unique in project, used for upserting.
|
||||
adapter:
|
||||
type: LlmAdapter
|
||||
docs: The adapter used to interface with the LLM
|
||||
secretKey:
|
||||
type: string
|
||||
docs: Secret key for the LLM API.
|
||||
baseURL:
|
||||
type: optional<string>
|
||||
docs: Custom base URL for the LLM API
|
||||
customModels:
|
||||
type: optional<list<string>>
|
||||
docs: List of custom model names
|
||||
withDefaultModels:
|
||||
type: optional<boolean>
|
||||
docs: Whether to include default models. Default is true.
|
||||
extraHeaders:
|
||||
type: optional<map<string, string>>
|
||||
docs: Extra headers to send with requests
|
||||
|
||||
LlmAdapter:
|
||||
enum:
|
||||
- value: anthropic
|
||||
name: Anthropic
|
||||
- value: openai
|
||||
name: OpenAI
|
||||
- value: azure
|
||||
name: Azure
|
||||
- value: bedrock
|
||||
name: Bedrock
|
||||
- value: google-vertex-ai
|
||||
name: GoogleVertexAI
|
||||
- value: google-ai-studio
|
||||
name: GoogleAIStudio
|
||||
@@ -28,7 +28,7 @@ service:
|
||||
"metrics": [ // Required. At least one metric must be provided
|
||||
{
|
||||
"measure": string, // What to measure, e.g. "count", "latency", "value"
|
||||
"aggregation": string // How to aggregate, e.g. "count", "sum", "avg", "p95"
|
||||
"aggregation": string // How to aggregate, e.g. "count", "sum", "avg", "p95", "histogram"
|
||||
}
|
||||
],
|
||||
"filters": [ // Optional. Default: []
|
||||
@@ -50,7 +50,11 @@ service:
|
||||
"field": string, // Field to order by
|
||||
"direction": string // "asc" or "desc"
|
||||
}
|
||||
]
|
||||
],
|
||||
"config": { // Optional. Query-specific configuration
|
||||
"bins": number, // Optional. Number of bins for histogram (1-100), default: 10
|
||||
"row_limit": number // Optional. Row limit for results (1-1000)
|
||||
}
|
||||
}
|
||||
```
|
||||
response: MetricsResponse
|
||||
@@ -62,3 +66,4 @@ types:
|
||||
docs: |
|
||||
The metrics data. Each item in the list contains the metric values and dimensions requested in the query.
|
||||
Format varies based on the query parameters.
|
||||
Histograms will return an array with [lower, upper, height] tuples.
|
||||
|
||||
@@ -32,6 +32,9 @@ service:
|
||||
userId: optional<string>
|
||||
type: optional<string>
|
||||
traceId: optional<string>
|
||||
level:
|
||||
type: optional<commons.ObservationLevel>
|
||||
docs: Optional filter for observations with a specific level (e.g. "DEBUG", "DEFAULT", "WARNING", "ERROR").
|
||||
parentObservationId: optional<string>
|
||||
environment:
|
||||
type: optional<string>
|
||||
|
||||
@@ -82,7 +82,7 @@ types:
|
||||
CreateChatPromptRequest:
|
||||
properties:
|
||||
name: string
|
||||
prompt: list<ChatMessage>
|
||||
prompt: list<ChatMessageWithPlaceholders>
|
||||
config: optional<unknown>
|
||||
labels:
|
||||
type: optional<list<string>>
|
||||
@@ -132,6 +132,11 @@ types:
|
||||
type: optional<map<string, unknown>>
|
||||
docs: The dependency resolution graph for the current prompt. Null if prompt has no dependencies.
|
||||
|
||||
ChatMessageWithPlaceholders:
|
||||
union:
|
||||
chatmessage: ChatMessage
|
||||
placeholder: PlaceholderMessage
|
||||
|
||||
ChatMessage:
|
||||
properties:
|
||||
role:
|
||||
@@ -139,6 +144,11 @@ types:
|
||||
content:
|
||||
type: string
|
||||
|
||||
PlaceholderMessage:
|
||||
properties:
|
||||
name:
|
||||
type: string
|
||||
|
||||
TextPrompt:
|
||||
extends: BasePrompt
|
||||
properties:
|
||||
@@ -147,4 +157,4 @@ types:
|
||||
ChatPrompt:
|
||||
extends: BasePrompt
|
||||
properties:
|
||||
prompt: list<ChatMessage>
|
||||
prompt: list<ChatMessageWithPlaceholders>
|
||||
|
||||
@@ -63,6 +63,9 @@ service:
|
||||
type: optional<string>
|
||||
allow-multiple: true
|
||||
docs: Optional filter for traces where the environment is one of the provided values.
|
||||
fields:
|
||||
type: optional<string>
|
||||
docs: "Comma-separated list of fields to include in the response. Available field groups are 'core' (always included), 'io' (input, output, metadata), 'scores', 'observations', 'metrics'. If not provided, all fields are included. Example: 'core,scores,metrics'"
|
||||
response: Traces
|
||||
deleteMultiple:
|
||||
docs: Delete multiple traces
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# yaml-language-server: $schema=https://schema.buildwithfern.dev/generators-yml.json
|
||||
default-group: local
|
||||
groups:
|
||||
local:
|
||||
@@ -7,6 +8,7 @@ groups:
|
||||
output:
|
||||
location: local-file-system
|
||||
path: ../../../web/public/generated/api
|
||||
|
||||
- name: fernapi/fern-python-sdk
|
||||
version: 2.16.0
|
||||
output:
|
||||
@@ -19,35 +21,32 @@ groups:
|
||||
pydantic_config:
|
||||
require_optional_fields: false
|
||||
use_str_enums: false
|
||||
# - name: fernapi/fern-java-sdk
|
||||
# version: 2.20.1
|
||||
# output:
|
||||
# location: local-file-system
|
||||
# path: ../../../../langfuse-java/src/main/java/com/langfuse/client/
|
||||
# config:
|
||||
# client-class-name: LangfuseClient
|
||||
|
||||
# - name: fernapi/fern-java-sdk
|
||||
# version: 2.20.1
|
||||
# output:
|
||||
# location: local-file-system
|
||||
# path: ../../../../langfuse-java/src/main/java/com/langfuse/client/
|
||||
# config:
|
||||
# client-class-name: LangfuseClient
|
||||
|
||||
- name: fernapi/fern-typescript-node-sdk
|
||||
version: 2.6.1
|
||||
output:
|
||||
location: local-file-system
|
||||
path: ../../../generated/typescript
|
||||
config:
|
||||
namespaceExport: LangfuseAPI
|
||||
outputSourceFiles: true
|
||||
skipResponseValidation: true
|
||||
fetchSupport: native
|
||||
formDataSupport: Node18
|
||||
fileResponseType: binary-response
|
||||
streamType: web
|
||||
omitFernHeaders: true
|
||||
|
||||
- name: fernapi/fern-postman
|
||||
version: 0.0.45
|
||||
output:
|
||||
location: local-file-system
|
||||
path: ../../../web/public/generated/postman
|
||||
# published:
|
||||
# generators:
|
||||
# - name: fernapi/fern-python-sdk
|
||||
# version: 0.3.7
|
||||
# output:
|
||||
# location: pypi
|
||||
# url: pypi.buildwithfern.com
|
||||
# package-name: finto-fern-langfuse
|
||||
# config:
|
||||
# namespaceExport: Langfuse
|
||||
# allowCustomFetcher: true
|
||||
# - name: fernapi/fern-typescript-node-sdk
|
||||
# version: 0.7.1
|
||||
# output:
|
||||
# location: npm
|
||||
# url: npm.buildwithfern.com
|
||||
# package-name: "@finto-fern/langfuse-node"
|
||||
# config:
|
||||
# namespaceExport: Langfuse
|
||||
# allowCustomFetcher: true
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
python/
|
||||
+13
-9
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "langfuse",
|
||||
"version": "3.65.2",
|
||||
"version": "3.97.3",
|
||||
"author": "engineering@langfuse.com",
|
||||
"license": "MIT",
|
||||
"private": true,
|
||||
@@ -17,27 +17,30 @@
|
||||
"db:seed": "turbo run db:seed",
|
||||
"db:seed:examples": "turbo run db:seed:examples",
|
||||
"nuke": "bash ./scripts/nuke.sh",
|
||||
"dx": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx-f": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset -f && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx:skip-infra": "pnpm i && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset:test && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx-f": "pnpm i && pnpm run infra:dev:prune && pnpm run infra:dev:up --pull always && pnpm --filter=shared run db:reset:test && pnpm --filter=shared run db:reset -f && SKIP_CONFIRM=1 pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"dx:skip-infra": "pnpm i && pnpm --filter=shared run db:reset:test && pnpm --filter=shared run db:reset && pnpm --filter=shared run ch:reset && pnpm --filter=shared run db:seed:examples && pnpm run dev",
|
||||
"build": "turbo run build",
|
||||
"start": "turbo run start",
|
||||
"dev": "turbo run dev",
|
||||
"dev:worker": "turbo run dev --filter=worker",
|
||||
"dev:web": "turbo run dev --filter=web",
|
||||
"dev:web-turbo": "turbo run dev --filter=web -- --turbo",
|
||||
"lint": "turbo run lint",
|
||||
"format": "prettier --write \"**/*.{js,jsx,ts,tsx,css}\" --experimental-cli",
|
||||
"format:check": "prettier --check \"**/*.{js,jsx,ts,tsx,css}\" --experimental-cli",
|
||||
"test": "turbo run test",
|
||||
"release": "dotenv -e ../.env -- release-it",
|
||||
"prepare": "husky"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@release-it/bumper": "^7.0.1",
|
||||
"@release-it/bumper": "^7.0.5",
|
||||
"braces": "3.0.3",
|
||||
"dotenv-cli": "^7.4.2",
|
||||
"husky": "^9.0.11",
|
||||
"prettier": "^3.3.3",
|
||||
"release-it": "^18.1.2",
|
||||
"turbo": "^1.13.4"
|
||||
"prettier": "^3.6.2",
|
||||
"release-it": "^19.0.3",
|
||||
"turbo": "^2.5.5"
|
||||
},
|
||||
"release-it": {
|
||||
"git": {
|
||||
@@ -88,7 +91,8 @@
|
||||
"nanoid": "^3.3.8",
|
||||
"katex": "^0.16.21",
|
||||
"tar-fs": "^2.1.2",
|
||||
"rollup@^4.0.0": "^4.22.4"
|
||||
"rollup@^4.0.0": "^4.22.4",
|
||||
"@types/node-fetch": "^2.6.13"
|
||||
},
|
||||
"patchedDependencies": {
|
||||
"next-auth@4.24.11": "patches/next-auth@4.24.11.patch"
|
||||
|
||||
@@ -35,5 +35,13 @@ module.exports = {
|
||||
{
|
||||
files: ["*.js?(x)", "*.ts?(x)"],
|
||||
},
|
||||
{
|
||||
files: ["*.ts", "*.mts", "*.cts", "*.tsx"],
|
||||
// no-undef doesn't make sense in TS, see:
|
||||
// https://typescript-eslint.io/troubleshooting/faqs/eslint/#i-get-errors-from-the-no-undef-rule-about-global-variables-not-being-defined-even-though-there-are-no-typescript-errors
|
||||
rules: {
|
||||
"no-undef": "off",
|
||||
},
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
"@vercel/style-guide": "^6.0.0",
|
||||
"eslint-config-next": "^14.2.15",
|
||||
"eslint-config-prettier": "^9.1.0",
|
||||
"eslint-config-turbo": "^1.13.4",
|
||||
"eslint-config-turbo": "^2.5.5",
|
||||
"eslint-plugin-only-warn": "^1.1.0",
|
||||
"typescript": "^5.4.5"
|
||||
}
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
DROP TABLE dataset_run_items ON CLUSTER default;
|
||||
@@ -0,0 +1,36 @@
|
||||
CREATE TABLE dataset_run_items ON CLUSTER default (
|
||||
-- primary identifiers
|
||||
`id` String,
|
||||
`project_id` String,
|
||||
`dataset_run_id` String,
|
||||
`dataset_item_id` String,
|
||||
`dataset_id` String,
|
||||
`trace_id` String,
|
||||
`observation_id` Nullable(String),
|
||||
|
||||
-- error field
|
||||
`error` Nullable(String),
|
||||
|
||||
-- timestamps
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
|
||||
-- denormalized immutable dataset run fields
|
||||
`dataset_run_name` String,
|
||||
`dataset_run_description` Nullable(String),
|
||||
`dataset_run_metadata` Map(LowCardinality(String), String),
|
||||
`dataset_run_created_at` DateTime64(3),
|
||||
|
||||
-- denormalized dataset item fields (mutable, but snapshots are relevant)
|
||||
`dataset_item_input` Nullable(String) CODEC(ZSTD(3)), -- json
|
||||
`dataset_item_expected_output` Nullable(String) CODEC(ZSTD(3)), -- json
|
||||
`dataset_item_metadata` Map(LowCardinality(String), String),
|
||||
|
||||
-- clickhouse engine fields
|
||||
`event_ts` DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
|
||||
-- For dataset item lookups
|
||||
INDEX idx_dataset_item dataset_item_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
) ENGINE = ReplacingMergeTree(event_ts, is_deleted)
|
||||
ORDER BY (project_id, dataset_id, dataset_run_id, id);
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
-- Drop materialized views first
|
||||
DROP VIEW IF EXISTS traces_30d_amt_mv ON CLUSTER default;
|
||||
DROP VIEW IF EXISTS traces_7d_amt_mv ON CLUSTER default;
|
||||
DROP VIEW IF EXISTS traces_all_amt_mv ON CLUSTER default;
|
||||
|
||||
-- Drop AMT tables
|
||||
DROP TABLE IF EXISTS traces_30d_amt ON CLUSTER default;
|
||||
DROP TABLE IF EXISTS traces_7d_amt ON CLUSTER default;
|
||||
DROP TABLE IF EXISTS traces_all_amt ON CLUSTER default;
|
||||
|
||||
-- Drop the Null table
|
||||
DROP TABLE IF EXISTS traces_null ON CLUSTER default;
|
||||
+300
@@ -0,0 +1,300 @@
|
||||
-- Create a Null table that serves as a trigger for all materialized views.
|
||||
-- We use a Null engine here to avoid storing intermediate results and save on storage.
|
||||
CREATE TABLE traces_null ON CLUSTER default
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`start_time` DateTime64(3),
|
||||
`end_time` Nullable(DateTime64(3)),
|
||||
`name` Nullable(String),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`user_id` Nullable(String),
|
||||
`session_id` Nullable(String),
|
||||
`environment` String,
|
||||
`tags` Array(String),
|
||||
`version` Nullable(String),
|
||||
`release` Nullable(String),
|
||||
|
||||
-- UI properties - We make them nullable to prevent absent values being interpreted as overwrites.
|
||||
`bookmarked` Nullable(Bool),
|
||||
`public` Nullable(Bool),
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
`observation_ids` Array(String),
|
||||
`score_ids` Array(String),
|
||||
`cost_details` Map(String, Decimal64(12)),
|
||||
`usage_details` Map(String, UInt64),
|
||||
-- TODO: Do we want to aggregate/collect `levels` seen within the trace?
|
||||
|
||||
-- Input/Output
|
||||
`input` String,
|
||||
`output` String,
|
||||
|
||||
`created_at` DateTime64(3),
|
||||
`updated_at` DateTime64(3),
|
||||
`event_ts` DateTime64(3)
|
||||
) Engine = Null();
|
||||
|
||||
-- Create the all AMT
|
||||
CREATE TABLE traces_all_amt ON CLUSTER default
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_trace_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id);
|
||||
|
||||
-- Create materialized view for all_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_all_amt_mv ON CLUSTER default TO traces_all_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
|
||||
-- Create the 7-day TTL AMT
|
||||
CREATE TABLE traces_7d_amt ON CLUSTER default
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id)
|
||||
TTL toDate(start_time) + INTERVAL 7 DAY;
|
||||
|
||||
-- Create materialized view for 7d_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_7d_amt_mv ON CLUSTER default TO traces_7d_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
|
||||
-- Create the 30-day TTL AMT
|
||||
CREATE TABLE traces_30d_amt ON CLUSTER default
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id)
|
||||
TTL toDate(start_time) + INTERVAL 30 DAY;
|
||||
|
||||
-- Create materialized view for 30d_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_30d_amt_mv ON CLUSTER default TO traces_30d_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations -- DO NOT USE
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
@@ -1 +1 @@
|
||||
ALTER TABLE traces ON CLUSTER default DROP INDEX IF EXISTS idx_user_id;
|
||||
ALTER TABLE traces DROP INDEX IF EXISTS idx_user_id;
|
||||
@@ -0,0 +1 @@
|
||||
DROP TABLE dataset_run_items;
|
||||
@@ -0,0 +1,36 @@
|
||||
CREATE TABLE dataset_run_items (
|
||||
-- primary identifiers
|
||||
`id` String,
|
||||
`project_id` String,
|
||||
`dataset_run_id` String,
|
||||
`dataset_item_id` String,
|
||||
`dataset_id` String,
|
||||
`trace_id` String,
|
||||
`observation_id` Nullable(String),
|
||||
|
||||
-- error field
|
||||
`error` Nullable(String),
|
||||
|
||||
-- timestamps
|
||||
`created_at` DateTime64(3) DEFAULT now(),
|
||||
`updated_at` DateTime64(3) DEFAULT now(),
|
||||
|
||||
-- denormalized immutable dataset run fields
|
||||
`dataset_run_name` String,
|
||||
`dataset_run_description` Nullable(String),
|
||||
`dataset_run_metadata` Map(LowCardinality(String), String),
|
||||
`dataset_run_created_at` DateTime64(3),
|
||||
|
||||
-- denormalized dataset item fields (mutable, but snapshots are relevant)
|
||||
`dataset_item_input` Nullable(String) CODEC(ZSTD(3)), -- json
|
||||
`dataset_item_expected_output` Nullable(String) CODEC(ZSTD(3)), -- json
|
||||
`dataset_item_metadata` Map(LowCardinality(String), String),
|
||||
|
||||
-- clickhouse engine fields
|
||||
`event_ts` DateTime64(3),
|
||||
`is_deleted` UInt8,
|
||||
|
||||
-- For dataset item lookups
|
||||
INDEX idx_dataset_item dataset_item_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
) ENGINE = ReplacingMergeTree(event_ts, is_deleted)
|
||||
ORDER BY (project_id, dataset_id, dataset_run_id, id);
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
-- Drop materialized views first
|
||||
DROP VIEW IF EXISTS traces_30d_amt_mv;
|
||||
DROP VIEW IF EXISTS traces_7d_amt_mv;
|
||||
DROP VIEW IF EXISTS traces_all_amt_mv;
|
||||
|
||||
-- Drop AMT tables
|
||||
DROP TABLE IF EXISTS traces_30d_amt;
|
||||
DROP TABLE IF EXISTS traces_7d_amt;
|
||||
DROP TABLE IF EXISTS traces_all_amt;
|
||||
|
||||
-- Drop the Null table
|
||||
DROP TABLE IF EXISTS traces_null;
|
||||
+300
@@ -0,0 +1,300 @@
|
||||
-- Create a Null table that serves as a trigger for all materialized views.
|
||||
-- We use a Null engine here to avoid storing intermediate results and save on storage.
|
||||
CREATE TABLE traces_null
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`start_time` DateTime64(3),
|
||||
`end_time` Nullable(DateTime64(3)),
|
||||
`name` Nullable(String),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` Map(LowCardinality(String), String),
|
||||
`user_id` Nullable(String),
|
||||
`session_id` Nullable(String),
|
||||
`environment` String,
|
||||
`tags` Array(String),
|
||||
`version` Nullable(String),
|
||||
`release` Nullable(String),
|
||||
|
||||
-- UI properties - We make them nullable to prevent absent values being interpreted as overwrites.
|
||||
`bookmarked` Nullable(Bool),
|
||||
`public` Nullable(Bool),
|
||||
|
||||
-- Aggregations
|
||||
`observation_ids` Array(String),
|
||||
`score_ids` Array(String),
|
||||
`cost_details` Map(String, Decimal64(12)),
|
||||
`usage_details` Map(String, UInt64),
|
||||
-- TODO: Do we want to aggregate/collect `levels` seen within the trace?
|
||||
|
||||
-- Input/Output
|
||||
`input` String,
|
||||
`output` String,
|
||||
|
||||
`created_at` DateTime64(3),
|
||||
`updated_at` DateTime64(3),
|
||||
`event_ts` DateTime64(3)
|
||||
) Engine = Null();
|
||||
|
||||
-- Create the all AMT
|
||||
CREATE TABLE traces_all_amt
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_trace_id id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id);
|
||||
|
||||
-- Create materialized view for all_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_all_amt_mv TO traces_all_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
|
||||
-- Create the 7-day TTL AMT
|
||||
CREATE TABLE traces_7d_amt
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id)
|
||||
TTL toDate(start_time) + INTERVAL 7 DAY;
|
||||
|
||||
-- Create materialized view for 7d_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_7d_amt_mv TO traces_7d_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
|
||||
-- Create the 30-day TTL AMT
|
||||
CREATE TABLE traces_30d_amt
|
||||
(
|
||||
-- Identifiers
|
||||
`project_id` String,
|
||||
`id` String,
|
||||
`timestamp` SimpleAggregateFunction(min, DateTime64(3)), -- Backward compatibility: redundant with start_time
|
||||
`start_time` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`end_time` SimpleAggregateFunction(max, Nullable(DateTime64(3))),
|
||||
`name` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- Metadata properties
|
||||
`metadata` SimpleAggregateFunction(maxMap, Map(String, String)),
|
||||
`user_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`session_id` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`environment` SimpleAggregateFunction(anyLast, String),
|
||||
`tags` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`version` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
`release` SimpleAggregateFunction(anyLast, Nullable(String)),
|
||||
|
||||
-- UI properties
|
||||
`bookmarked` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
`public` AggregateFunction(argMax, Nullable(Bool), DateTime64(3)),
|
||||
|
||||
-- Aggregations
|
||||
`observation_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`score_ids` SimpleAggregateFunction(groupUniqArrayArray, Array(String)),
|
||||
`cost_details` SimpleAggregateFunction(sumMap, Map(String, Decimal(38, 12))),
|
||||
`usage_details` SimpleAggregateFunction(sumMap, Map(String, UInt64)),
|
||||
|
||||
-- Input/Output -> prefer correctness via argMax
|
||||
`input` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
`output` AggregateFunction(argMax, String, DateTime64(3)) CODEC (ZSTD(3)),
|
||||
|
||||
`created_at` SimpleAggregateFunction(min, DateTime64(3)),
|
||||
`updated_at` SimpleAggregateFunction(max, DateTime64(3)),
|
||||
|
||||
-- Indexes
|
||||
INDEX idx_user_id user_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_session_id session_id TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_name name TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_version version TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_release release TYPE bloom_filter(0.001) GRANULARITY 1,
|
||||
INDEX idx_tags tags TYPE bloom_filter(0.001) GRANULARITY 1
|
||||
) Engine = AggregatingMergeTree()
|
||||
ORDER BY (project_id, id)
|
||||
TTL toDate(start_time) + INTERVAL 30 DAY;
|
||||
|
||||
-- Create materialized view for 30d_amt
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS traces_30d_amt_mv TO traces_30d_amt AS
|
||||
SELECT
|
||||
-- Identifiers
|
||||
tn.project_id as project_id,
|
||||
tn.id as id,
|
||||
min(tn.start_time) as timestamp, -- Backward compatibility: redundant with start_time
|
||||
min(tn.start_time) as start_time,
|
||||
max(coalesce(tn.end_time, tn.start_time)) as end_time,
|
||||
anyLast(tn.name) as name,
|
||||
|
||||
-- Metadata properties
|
||||
maxMap(tn.metadata) as metadata,
|
||||
anyLast(tn.user_id) as user_id,
|
||||
anyLast(tn.session_id) as session_id,
|
||||
anyLast(tn.environment) as environment,
|
||||
groupUniqArrayArray(tn.tags) as tags,
|
||||
anyLast(tn.version) as version,
|
||||
anyLast(tn.release) as release,
|
||||
|
||||
-- UI properties
|
||||
argMaxState(tn.bookmarked, if(tn.bookmarked is not null, tn.event_ts, toDateTime64(0, 3))) as bookmarked,
|
||||
argMaxState(tn.public, if(tn.public is not null, tn.event_ts, toDateTime64(0, 3))) as public,
|
||||
|
||||
-- Aggregations
|
||||
groupUniqArrayArray(tn.observation_ids) as observation_ids,
|
||||
groupUniqArrayArray(tn.score_ids) as score_ids,
|
||||
sumMap(tn.cost_details) as cost_details,
|
||||
sumMap(tn.usage_details) as usage_details,
|
||||
|
||||
-- Input/Output
|
||||
argMaxState(tn.input, if(tn.input <> '', tn.event_ts, toDateTime64(0, 3))) as input,
|
||||
argMaxState(tn.output, if(tn.output <> '', tn.event_ts, toDateTime64(0, 3))) as output,
|
||||
|
||||
min(tn.created_at) as created_at,
|
||||
max(tn.updated_at) as updated_at
|
||||
FROM traces_null tn
|
||||
GROUP BY project_id, id;
|
||||
@@ -36,8 +36,12 @@ if [ "$CLICKHOUSE_CLUSTER_ENABLED" == "false" ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-migrations-table-engine=MergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
|
||||
# If SKIP_CONFIRM is set, automatically answer the confirmation prompt. Otherwise run interactively.
|
||||
if [ "$SKIP_CONFIRM" = "1" ] || [ "$SKIP_CONFIRM" = "true" ]; then
|
||||
printf 'y\n' | migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
|
||||
else
|
||||
migrate -source file://clickhouse/migrations/unclustered -database "$DATABASE_URL" down
|
||||
fi
|
||||
else
|
||||
if [ "$CLICKHOUSE_MIGRATION_SSL" = true ] ; then
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&secure=true&skip_verify=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
@@ -45,6 +49,10 @@ else
|
||||
DATABASE_URL="${CLICKHOUSE_MIGRATION_URL}?username=${CLICKHOUSE_USER}&password=${CLICKHOUSE_PASSWORD}&database=${CLICKHOUSE_DB}&x-multi-statement=true&x-cluster-name=${CLICKHOUSE_CLUSTER_NAME}&x-migrations-table-engine=ReplicatedMergeTree"
|
||||
fi
|
||||
|
||||
# Execute the up command
|
||||
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
|
||||
# If SKIP_CONFIRM is set, automatically answer the confirmation prompt. Otherwise run interactively.
|
||||
if [ "$SKIP_CONFIRM" = "1" ] || [ "$SKIP_CONFIRM" = "true" ]; then
|
||||
printf 'y\n' | migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
|
||||
else
|
||||
migrate -source file://clickhouse/migrations/clustered -database "$DATABASE_URL" down
|
||||
fi
|
||||
fi
|
||||
@@ -1,70 +0,0 @@
|
||||
import { prisma } from "../../src/db";
|
||||
import { ObservationRecordReadType, redis } from "../../src/server";
|
||||
import { prepareClickhouse } from "../../scripts/prepareClickhouse";
|
||||
import { createDatasets } from "../../prisma/seed";
|
||||
import { queryClickhouse } from "../../src/server/repositories/clickhouse";
|
||||
import { convertObservation } from "../../src/server/repositories/observations_converters";
|
||||
|
||||
async function main() {
|
||||
try {
|
||||
const projectIds = ["7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"]; // Example project IDs
|
||||
if (
|
||||
await prisma.project.findFirst({
|
||||
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
|
||||
})
|
||||
) {
|
||||
projectIds.push("239ad00f-562f-411d-af14-831c75ddd875");
|
||||
}
|
||||
await prepareClickhouse(projectIds, {
|
||||
numberOfDays: 3,
|
||||
totalObservations: 10000,
|
||||
});
|
||||
|
||||
const project1 = await prisma.project.findFirst({
|
||||
where: { id: "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a" },
|
||||
});
|
||||
|
||||
const project2 =
|
||||
projectIds.length > 1
|
||||
? await prisma.project.findFirst({
|
||||
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
|
||||
})
|
||||
: await prisma.project.findFirst();
|
||||
|
||||
const query = `
|
||||
SELECT *
|
||||
FROM observations o
|
||||
WHERE o.project_id IN ({projectIds: Array(String)})
|
||||
LIMIT 2000;
|
||||
`;
|
||||
|
||||
const res = await queryClickhouse<ObservationRecordReadType>({
|
||||
query,
|
||||
params: {
|
||||
projectIds,
|
||||
},
|
||||
});
|
||||
|
||||
await createDatasets(
|
||||
project1!,
|
||||
project2!,
|
||||
(await Promise.all(res.map(convertObservation))).map((o) => ({
|
||||
...o,
|
||||
metadata: {},
|
||||
modelParameters: {},
|
||||
input: {},
|
||||
output: {},
|
||||
})),
|
||||
);
|
||||
|
||||
console.log("Clickhouse preparation completed successfully.");
|
||||
} catch (error) {
|
||||
console.error("Error during Clickhouse preparation:", error);
|
||||
} finally {
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
console.log("Disconnected from Clickhouse.");
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -5,8 +5,30 @@
|
||||
|
||||
# Check if CLICKHOUSE_URL is configured
|
||||
if [ -z "${CLICKHOUSE_URL}" ]; then
|
||||
echo "Info: CLICKHOUSE_URL not configured, skipping migration."
|
||||
exit 0
|
||||
echo "Error: CLICKHOUSE_URL is not configured."
|
||||
echo "Please set CLICKHOUSE_URL in your environment variables."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if CLICKHOUSE_MIGRATION_URL is configured
|
||||
if [ -z "${CLICKHOUSE_MIGRATION_URL}" ]; then
|
||||
echo "Error: CLICKHOUSE_MIGRATION_URL is not configured."
|
||||
echo "Please set CLICKHOUSE_MIGRATION_URL in your environment variables."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if CLICKHOUSE_USER is set
|
||||
if [ -z "${CLICKHOUSE_USER}" ]; then
|
||||
echo "Error: CLICKHOUSE_USER is not set."
|
||||
echo "Please set CLICKHOUSE_USER in your environment variables."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if CLICKHOUSE_PASSWORD is set
|
||||
if [ -z "${CLICKHOUSE_PASSWORD}" ]; then
|
||||
echo "Error: CLICKHOUSE_PASSWORD is not set."
|
||||
echo "Please set CLICKHOUSE_PASSWORD in your environment variables."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if golang-migrate is installed
|
||||
|
||||
@@ -33,11 +33,12 @@
|
||||
"scripts": {
|
||||
"build": "tsc",
|
||||
"dev": "tsc --watch",
|
||||
"lint": "eslint . --ext .js,.jsx,.ts,.tsx --max-warnings 70",
|
||||
"lint": "eslint . --ext .js,.jsx,.ts,.tsx --max-warnings 0",
|
||||
"lint:fix": "eslint . --ext .js,.jsx,.ts,.tsx --fix",
|
||||
"db:migrate": "DISABLE_ERD=false dotenv -e ../../.env -- npx prisma migrate dev",
|
||||
"db:push": "DISABLE_ERD=false dotenv -e ../../.env -- npx prisma db push",
|
||||
"db:reset": "dotenv -e ../../.env npx -- prisma migrate reset",
|
||||
"db:reset:test": "if [ -f ../../.env.test ]; then dotenv -e ../../.env.test -e ../../.env -- npx prisma migrate reset --force; fi",
|
||||
"db:deploy": "dotenv -e ../../.env npx -- prisma migrate deploy",
|
||||
"db:seed": "dotenv -e ../../.env -- npx prisma db seed",
|
||||
"db:generate": "dotenv -e ../../.env -- npx prisma generate",
|
||||
@@ -47,11 +48,11 @@
|
||||
"ch:down": "bash clickhouse/scripts/down.sh",
|
||||
"ch:drop": "bash clickhouse/scripts/drop.sh",
|
||||
"ch:reset": "pnpm run ch:down && pnpm run ch:up && pnpm run ch:seed",
|
||||
"ch:seed": "dotenv -e ../../.env -- ts-node -r tsconfig-paths/register -r dotenv/config --compiler-options '{\"module\":\"CommonJS\"}' clickhouse/scripts/seed.ts",
|
||||
"load:setup": "dotenv -e ../../.env -- tsx scripts/load-seed.ts"
|
||||
"ch:seed": "dotenv -e ../../.env -- ts-node -r tsconfig-paths/register -r dotenv/config --compiler-options '{\"module\":\"CommonJS\"}' scripts/seeder/seed-clickhouse.ts",
|
||||
"load:setup": "dotenv -e ../../.env -- tsx scripts/seeder/load-seed-clickhouse.ts"
|
||||
},
|
||||
"prisma": {
|
||||
"seed": "ts-node -r tsconfig-paths/register -r dotenv/config --compiler-options {\"module\":\"CommonJS\"} prisma/seed.ts"
|
||||
"seed": "ts-node -r tsconfig-paths/register -r dotenv/config --compiler-options {\"module\":\"CommonJS\"} scripts/seeder/seed-postgres.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@anthropic-ai/tokenizer": "^0.0.4",
|
||||
@@ -60,38 +61,40 @@
|
||||
"@aws-sdk/lib-storage": "^3.675.0",
|
||||
"@aws-sdk/s3-request-presigner": "^3.679.0",
|
||||
"@azure/storage-blob": "^12.26.0",
|
||||
"@clickhouse/client": "^1.11.1",
|
||||
"@clickhouse/client": "^1.12.0",
|
||||
"@google-cloud/storage": "^7.15.2",
|
||||
"@langchain/anthropic": "^0.3.12",
|
||||
"@langchain/aws": "^0.1.3",
|
||||
"@langchain/core": "^0.3.37",
|
||||
"@langchain/google-genai": "^0.1.9",
|
||||
"@langchain/google-vertexai": "^0.1.8",
|
||||
"@langchain/openai": "^0.3.17",
|
||||
"@langchain/anthropic": "^0.3.22",
|
||||
"@langchain/aws": "^0.1.11",
|
||||
"@langchain/core": "^0.3.58",
|
||||
"@langchain/google-genai": "^0.2.12",
|
||||
"@langchain/google-vertexai": "^0.2.12",
|
||||
"@langchain/openai": "^0.5.13",
|
||||
"@opentelemetry/api": ">=1.0.0 <1.10.0",
|
||||
"@prisma/client": "^6.3.0",
|
||||
"@react-email/components": "^0.0.19",
|
||||
"@react-email/render": "^0.0.15",
|
||||
"@prisma/client": "^6.10.1",
|
||||
"@react-email/components": "^0.1.0",
|
||||
"@react-email/render": "^1.1.2",
|
||||
"@slack/oauth": "^3.0.3",
|
||||
"@slack/web-api": "^7.9.3",
|
||||
"@types/bcryptjs": "^2.4.6",
|
||||
"axios": "^1.8.2",
|
||||
"bcryptjs": "^2.4.3",
|
||||
"bullmq": "^5.34.10",
|
||||
"dd-trace": "^5.36.0",
|
||||
"decimal.js": "^10.4.3",
|
||||
"exponential-backoff": "^3.1.1",
|
||||
"exponential-backoff": "^3.1.2",
|
||||
"https-proxy-agent": "^7.0.6",
|
||||
"ioredis": "^5.4.1",
|
||||
"jsonpath-plus": "10.3.0",
|
||||
"kysely": "^0.27.4",
|
||||
"langchain": "^0.3.15",
|
||||
"langfuse-langchain": "3.30.3",
|
||||
"langchain": "^0.3.28",
|
||||
"langfuse-langchain": "3.38.4",
|
||||
"lodash": "^4.17.21",
|
||||
"lossless-json": "^4.0.2",
|
||||
"lossless-json": "^4.1.1",
|
||||
"next-auth": "^4.24.11",
|
||||
"nodemailer": "^6.9.15",
|
||||
"prisma-extension-kysely": "^2.1.0",
|
||||
"uuid": "^9.0.1",
|
||||
"winston": "^3.15.0",
|
||||
"zod": "^3.24.4",
|
||||
"zod": "^3.25.62",
|
||||
"zod-to-json-schema": "^3.23.5"
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -101,6 +104,7 @@
|
||||
"@types/node": "^20.11.29",
|
||||
"@types/nodemailer": "^6.4.16",
|
||||
"@types/pg": "^8.11.10",
|
||||
"@types/react": "18.2.79",
|
||||
"@types/uuid": "^9.0.8",
|
||||
"@typescript-eslint/parser": "^7.12.0",
|
||||
"eslint": "^8.57.0",
|
||||
@@ -109,23 +113,16 @@
|
||||
"eslint-plugin-prettier": "^5.1.3",
|
||||
"kysely-codegen": "^0.16.8",
|
||||
"nodemon": "^3.1.7",
|
||||
"prettier": "^3.3.3",
|
||||
"prisma": "^6.3.0",
|
||||
"prettier": "^3.6.2",
|
||||
"prisma": "^6.10.1",
|
||||
"prisma-erd-generator": "^1.11.2",
|
||||
"prisma-kysely": "^1.8.0",
|
||||
"ts-node": "^10.9.2",
|
||||
"tsc-watch": "^6.2.0",
|
||||
"tsx": "^4.19.1",
|
||||
"typescript": "^5.4.5",
|
||||
"vitest": "^2.1.2"
|
||||
"typescript": "^5.4.5"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"@types/react": "~18.2.79",
|
||||
"react": "~18.2.0"
|
||||
},
|
||||
"pnpm": {
|
||||
"overrides": {
|
||||
"nanoid": "^3.3.8"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+2
@@ -0,0 +1,2 @@
|
||||
-- AlterEnum
|
||||
ALTER TYPE "DashboardWidgetChartType" ADD VALUE 'HISTOGRAM';
|
||||
@@ -0,0 +1,9 @@
|
||||
-- Database migration to add PIVOT_TABLE chart type to DashboardWidgetChartType enum
|
||||
-- This enables the creation of pivot table widgets in the dashboard system
|
||||
--
|
||||
-- This migration adds support for tabular data visualization with configurable
|
||||
-- row dimensions and metrics, extending the existing widget types (line charts,
|
||||
-- bar charts, pie charts, etc.) to include pivot table functionality.
|
||||
|
||||
-- AlterEnum
|
||||
ALTER TYPE "DashboardWidgetChartType" ADD VALUE 'PIVOT_TABLE';
|
||||
@@ -0,0 +1,111 @@
|
||||
-- CreateEnum
|
||||
CREATE TYPE "ActionType" AS ENUM ('WEBHOOK');
|
||||
|
||||
-- CreateEnum
|
||||
CREATE TYPE "ActionExecutionStatus" AS ENUM ('COMPLETED', 'ERROR', 'PENDING', 'CANCELLED');
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "actions" (
|
||||
"id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"type" "ActionType" NOT NULL,
|
||||
"config" JSONB NOT NULL,
|
||||
|
||||
CONSTRAINT "actions_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "triggers" (
|
||||
"id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"eventSource" TEXT NOT NULL,
|
||||
"eventActions" TEXT[],
|
||||
"filter" JSONB,
|
||||
"status" "JobConfigState" NOT NULL DEFAULT 'ACTIVE',
|
||||
|
||||
CONSTRAINT "triggers_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "automations" (
|
||||
"id" TEXT NOT NULL,
|
||||
"name" TEXT NOT NULL,
|
||||
"trigger_id" TEXT NOT NULL,
|
||||
"action_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"project_id" TEXT NOT NULL,
|
||||
|
||||
CONSTRAINT "automations_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "automation_executions" (
|
||||
"id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"source_id" TEXT NOT NULL,
|
||||
"automation_id" TEXT NOT NULL,
|
||||
"trigger_id" TEXT NOT NULL,
|
||||
"action_id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"status" "ActionExecutionStatus" NOT NULL DEFAULT 'PENDING',
|
||||
"input" JSONB NOT NULL,
|
||||
"output" JSONB,
|
||||
"started_at" TIMESTAMP(3),
|
||||
"finished_at" TIMESTAMP(3),
|
||||
"error" TEXT,
|
||||
|
||||
CONSTRAINT "automation_executions_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "actions_project_id_idx" ON "actions"("project_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "triggers_project_id_idx" ON "triggers"("project_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "automations_project_id_action_id_trigger_id_idx" ON "automations"("project_id", "action_id", "trigger_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "automations_project_id_name_idx" ON "automations"("project_id", "name");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "automation_executions_trigger_id_idx" ON "automation_executions"("trigger_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "automation_executions_action_id_idx" ON "automation_executions"("action_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "automation_executions_project_id_idx" ON "automation_executions"("project_id");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "actions" ADD CONSTRAINT "actions_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "triggers" ADD CONSTRAINT "triggers_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automations" ADD CONSTRAINT "automations_trigger_id_fkey" FOREIGN KEY ("trigger_id") REFERENCES "triggers"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automations" ADD CONSTRAINT "automations_action_id_fkey" FOREIGN KEY ("action_id") REFERENCES "actions"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automations" ADD CONSTRAINT "automations_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automation_executions" ADD CONSTRAINT "automation_executions_automation_id_fkey" FOREIGN KEY ("automation_id") REFERENCES "automations"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automation_executions" ADD CONSTRAINT "automation_executions_trigger_id_fkey" FOREIGN KEY ("trigger_id") REFERENCES "triggers"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automation_executions" ADD CONSTRAINT "automation_executions_action_id_fkey" FOREIGN KEY ("action_id") REFERENCES "actions"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "automation_executions" ADD CONSTRAINT "automation_executions_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
-- CreateEnum
|
||||
CREATE TYPE "BlobStorageExportMode" AS ENUM ('FULL_HISTORY', 'FROM_TODAY', 'FROM_CUSTOM_DATE');
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "blob_storage_integrations" ADD COLUMN "export_mode" "BlobStorageExportMode" NOT NULL DEFAULT 'FULL_HISTORY',
|
||||
ADD COLUMN "export_start_date" TIMESTAMP(3);
|
||||
@@ -0,0 +1,13 @@
|
||||
-- AlterTable
|
||||
ALTER TABLE "prices"
|
||||
ADD COLUMN "project_id" TEXT;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "prices"
|
||||
ADD CONSTRAINT "prices_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects" ("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- BackfillData
|
||||
UPDATE "prices"
|
||||
SET "project_id" = (SELECT "models"."project_id"
|
||||
FROM "models"
|
||||
WHERE "models"."id" = "prices"."model_id");
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('3445cac4-d9d5-4750-8b65-351135c1b85e', '20250711_1347_patch_llm_tool_schema_audit_logs', 'patchLLMToolAndLLLMSchemaAuditLogs', '{}');
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- CreateIndex
|
||||
CREATE INDEX CONCURRENTLY IF NOT EXISTS "trace_sessions_project_id_created_at_idx" ON "trace_sessions"("project_id", "created_at" DESC);
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- DropIndex
|
||||
DROP INDEX CONCURRENTLY IF EXISTS "trace_sessions_created_at_idx";
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- DropIndex
|
||||
DROP INDEX CONCURRENTLY IF EXISTS "trace_sessions_project_id_idx";
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- DropIndex
|
||||
DROP INDEX CONCURRENTLY IF EXISTS "trace_sessions_updated_at_idx";
|
||||
@@ -0,0 +1,3 @@
|
||||
-- AlterTable
|
||||
ALTER TABLE "datasets" ADD COLUMN "remote_experiment_payload" JSONB,
|
||||
ADD COLUMN "remote_experiment_url" TEXT;
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
-- AlterEnum
|
||||
ALTER TYPE "AnnotationQueueObjectType" ADD VALUE 'SESSION';
|
||||
@@ -0,0 +1,30 @@
|
||||
-- Migration: Add Slack Integration Support
|
||||
-- This migration adds support for Slack automation actions by:
|
||||
-- 1. Adding SLACK to the ActionType enum
|
||||
-- 2. Creating slack_integrations table for centralized token storage
|
||||
|
||||
-- AlterEnum
|
||||
ALTER TYPE "ActionType" ADD VALUE 'SLACK';
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "slack_integrations" (
|
||||
"id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"team_id" TEXT NOT NULL,
|
||||
"team_name" TEXT NOT NULL,
|
||||
"bot_token" TEXT NOT NULL,
|
||||
"bot_user_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
CONSTRAINT "slack_integrations_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "slack_integrations_project_id_key" ON "slack_integrations"("project_id");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "slack_integrations_team_id_idx" ON "slack_integrations"("team_id");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "slack_integrations" ADD CONSTRAINT "slack_integrations_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
INSERT INTO background_migrations (id, name, script, args)
|
||||
VALUES ('8d47f91b-3e5c-4a26-9f85-c12d6e4b9a3d', '20250731_1001_migrate_dataset_run_items_pg_to_ch', 'migrateDatasetRunItemsFromPostgresToClickhouse', '{}');
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE "pending_deletions" (
|
||||
"id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"object" TEXT NOT NULL,
|
||||
"object_id" TEXT NOT NULL,
|
||||
"is_deleted" BOOLEAN NOT NULL DEFAULT false,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
CONSTRAINT "pending_deletions_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "pending_deletions_project_id_object_is_deleted_idx" ON "pending_deletions"("project_id", "object", "is_deleted");
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX "pending_deletions_object_id_object_idx" ON "pending_deletions"("object_id", "object");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "pending_deletions" ADD CONSTRAINT "pending_deletions_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE "annotation_queue_assignments" (
|
||||
"id" TEXT NOT NULL,
|
||||
"project_id" TEXT NOT NULL,
|
||||
"user_id" TEXT NOT NULL,
|
||||
"queue_id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
CONSTRAINT "annotation_queue_assignments_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE UNIQUE INDEX "annotation_queue_assignments_project_id_queue_id_key" ON "annotation_queue_assignments"("project_id", "queue_id", "user_id");
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_project_id_fkey" FOREIGN KEY ("project_id") REFERENCES "projects"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_user_id_fkey" FOREIGN KEY ("user_id") REFERENCES "users"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "annotation_queue_assignments" ADD CONSTRAINT "annotation_queue_assignments_queue_id_fkey" FOREIGN KEY ("queue_id") REFERENCES "annotation_queues"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
@@ -0,0 +1,24 @@
|
||||
-- CreateEnum
|
||||
CREATE TYPE "SurveyName" AS ENUM ('org_onboarding', 'user_onboarding');
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE "surveys" (
|
||||
"id" TEXT NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"survey_name" "SurveyName" NOT NULL,
|
||||
"response" JSONB NOT NULL,
|
||||
"user_id" TEXT,
|
||||
"user_email" TEXT,
|
||||
"org_id" TEXT,
|
||||
|
||||
CONSTRAINT "surveys_pkey" PRIMARY KEY ("id")
|
||||
);
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "surveys" ADD CONSTRAINT "surveys_org_id_fkey" FOREIGN KEY ("org_id") REFERENCES "organizations"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE "surveys" ADD CONSTRAINT "surveys_user_id_fkey" FOREIGN KEY ("user_id") REFERENCES "users"("id") ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- RenameIndex
|
||||
ALTER INDEX "annotation_queue_assignments_project_id_queue_id_key" RENAME TO "annotation_queue_assignments_project_id_queue_id_user_id_key";
|
||||
@@ -1,3 +1,3 @@
|
||||
# Please do not edit this file manually
|
||||
# It should be added in your version-control system (e.g., Git)
|
||||
provider = "postgresql"
|
||||
provider = "postgresql"
|
||||
|
||||
@@ -65,29 +65,31 @@ model Session {
|
||||
}
|
||||
|
||||
model User {
|
||||
id String @id @default(cuid())
|
||||
name String?
|
||||
email String? @unique
|
||||
emailVerified DateTime? @map("email_verified")
|
||||
password String?
|
||||
image String?
|
||||
admin Boolean @default(false)
|
||||
accounts Account[]
|
||||
sessions Session[]
|
||||
organizationMemberships OrganizationMembership[]
|
||||
projectMemberships ProjectMembership[]
|
||||
invitations MembershipInvitation[]
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
featureFlags String[] @default([]) @map("feature_flags")
|
||||
annotatedLockedItem AnnotationQueueItem[] @relation("LockedByUser")
|
||||
annotatedCompletedItem AnnotationQueueItem[] @relation("AnnotatorUser")
|
||||
dashboardWidgetsCreated DashboardWidget[] @relation("CreatedByUser")
|
||||
dashboardWidgetsUpdated DashboardWidget[] @relation("UpdatedByUser")
|
||||
dashboardCreated Dashboard[] @relation("CreatedByUser")
|
||||
dashboardUpdated Dashboard[] @relation("UpdatedByUser")
|
||||
tableViewPresetCreated TableViewPreset[] @relation("CreatedByUser")
|
||||
tableViewPresetUpdated TableViewPreset[] @relation("UpdatedByUser")
|
||||
id String @id @default(cuid())
|
||||
name String?
|
||||
email String? @unique
|
||||
emailVerified DateTime? @map("email_verified")
|
||||
password String?
|
||||
image String?
|
||||
admin Boolean @default(false)
|
||||
accounts Account[]
|
||||
sessions Session[]
|
||||
organizationMemberships OrganizationMembership[]
|
||||
projectMemberships ProjectMembership[]
|
||||
invitations MembershipInvitation[]
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
featureFlags String[] @default([]) @map("feature_flags")
|
||||
annotatedLockedItem AnnotationQueueItem[] @relation("LockedByUser")
|
||||
annotatedCompletedItem AnnotationQueueItem[] @relation("AnnotatorUser")
|
||||
dashboardWidgetsCreated DashboardWidget[] @relation("CreatedByUser")
|
||||
dashboardWidgetsUpdated DashboardWidget[] @relation("UpdatedByUser")
|
||||
dashboardCreated Dashboard[] @relation("CreatedByUser")
|
||||
dashboardUpdated Dashboard[] @relation("UpdatedByUser")
|
||||
tableViewPresetCreated TableViewPreset[] @relation("CreatedByUser")
|
||||
tableViewPresetUpdated TableViewPreset[] @relation("UpdatedByUser")
|
||||
annotationQueueAssignment AnnotationQueueAssignment[]
|
||||
surveys Survey[]
|
||||
|
||||
@@map("users")
|
||||
}
|
||||
@@ -112,52 +114,61 @@ model Organization {
|
||||
projects Project[]
|
||||
MembershipInvitation MembershipInvitation[]
|
||||
ApiKey ApiKey[]
|
||||
surveys Survey[]
|
||||
|
||||
@@map("organizations")
|
||||
}
|
||||
|
||||
model Project {
|
||||
id String @id @default(cuid())
|
||||
orgId String @map("org_id")
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
deletedAt DateTime? @map("deleted_at")
|
||||
name String
|
||||
retentionDays Int? @map("retention_days")
|
||||
metadata Json?
|
||||
projectMembers ProjectMembership[]
|
||||
organization Organization @relation(fields: [orgId], references: [id], onUpdate: Cascade, onDelete: Cascade)
|
||||
apiKeys ApiKey[]
|
||||
dataset Dataset[]
|
||||
invitations MembershipInvitation[]
|
||||
sessions TraceSession[]
|
||||
Prompt Prompt[]
|
||||
Model Model[]
|
||||
EvalTemplate EvalTemplate[]
|
||||
JobConfiguration JobConfiguration[]
|
||||
JobExecution JobExecution[]
|
||||
LlmApiKeys LlmApiKeys[]
|
||||
PosthogIntegration PosthogIntegration[]
|
||||
BlobStorageIntegration BlobStorageIntegration[]
|
||||
scoreConfig ScoreConfig[]
|
||||
BatchExport BatchExport[]
|
||||
comment Comment[]
|
||||
annotationQueue AnnotationQueue[]
|
||||
annotationQueueItem AnnotationQueueItem[]
|
||||
TraceMedia TraceMedia[]
|
||||
Media Media[]
|
||||
ObservationMedia ObservationMedia[]
|
||||
LegacyTrace LegacyPrismaTrace[]
|
||||
LegacyObservation LegacyPrismaObservation[]
|
||||
LegacyScore LegacyPrismaScore[]
|
||||
PromptDependency PromptDependency[]
|
||||
LlmSchema LlmSchema[]
|
||||
LlmTool LlmTool[]
|
||||
PromptProtectedLabels PromptProtectedLabels[]
|
||||
Dashboard Dashboard[]
|
||||
DashboardWidget DashboardWidget[]
|
||||
TableViewPreset TableViewPreset[]
|
||||
DefaultLlmModel DefaultLlmModel[]
|
||||
id String @id @default(cuid())
|
||||
orgId String @map("org_id")
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
deletedAt DateTime? @map("deleted_at")
|
||||
name String
|
||||
retentionDays Int? @map("retention_days")
|
||||
metadata Json?
|
||||
projectMembers ProjectMembership[]
|
||||
organization Organization @relation(fields: [orgId], references: [id], onUpdate: Cascade, onDelete: Cascade)
|
||||
apiKeys ApiKey[]
|
||||
dataset Dataset[]
|
||||
invitations MembershipInvitation[]
|
||||
sessions TraceSession[]
|
||||
Prompt Prompt[]
|
||||
Model Model[]
|
||||
EvalTemplate EvalTemplate[]
|
||||
JobConfiguration JobConfiguration[]
|
||||
JobExecution JobExecution[]
|
||||
LlmApiKeys LlmApiKeys[]
|
||||
PosthogIntegration PosthogIntegration[]
|
||||
BlobStorageIntegration BlobStorageIntegration[]
|
||||
scoreConfig ScoreConfig[]
|
||||
BatchExport BatchExport[]
|
||||
comment Comment[]
|
||||
annotationQueue AnnotationQueue[]
|
||||
annotationQueueItem AnnotationQueueItem[]
|
||||
TraceMedia TraceMedia[]
|
||||
Media Media[]
|
||||
ObservationMedia ObservationMedia[]
|
||||
LegacyTrace LegacyPrismaTrace[]
|
||||
LegacyObservation LegacyPrismaObservation[]
|
||||
LegacyScore LegacyPrismaScore[]
|
||||
PromptDependency PromptDependency[]
|
||||
LlmSchema LlmSchema[]
|
||||
LlmTool LlmTool[]
|
||||
PromptProtectedLabels PromptProtectedLabels[]
|
||||
Dashboard Dashboard[]
|
||||
DashboardWidget DashboardWidget[]
|
||||
TableViewPreset TableViewPreset[]
|
||||
actions Action[]
|
||||
triggers Trigger[]
|
||||
automationExecutions AutomationExecution[]
|
||||
Automation Automation[]
|
||||
DefaultLlmModel DefaultLlmModel[]
|
||||
Price Price[]
|
||||
SlackIntegration SlackIntegration?
|
||||
PendingDeletion PendingDeletion[]
|
||||
AnnotationQueueAssignment AnnotationQueueAssignment[]
|
||||
|
||||
@@index([orgId])
|
||||
@@map("projects")
|
||||
@@ -306,9 +317,7 @@ model TraceSession {
|
||||
environment String @default("default")
|
||||
|
||||
@@id([id, projectId])
|
||||
@@index([projectId])
|
||||
@@index([createdAt])
|
||||
@@index([updatedAt])
|
||||
@@index([projectId, createdAt(sort: Desc)])
|
||||
@@map("trace_sessions")
|
||||
}
|
||||
|
||||
@@ -486,15 +495,16 @@ enum ScoreDataType {
|
||||
}
|
||||
|
||||
model AnnotationQueue {
|
||||
id String @id @default(cuid())
|
||||
name String
|
||||
description String?
|
||||
scoreConfigIds String[] @default([]) @map("score_config_ids")
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
annotationQueueItem AnnotationQueueItem[]
|
||||
id String @id @default(cuid())
|
||||
name String
|
||||
description String?
|
||||
scoreConfigIds String[] @default([]) @map("score_config_ids")
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
annotationQueueItem AnnotationQueueItem[]
|
||||
annotationQueueAssignment AnnotationQueueAssignment[]
|
||||
|
||||
@@unique([projectId, name])
|
||||
@@index([id, projectId])
|
||||
@@ -536,6 +546,22 @@ enum AnnotationQueueStatus {
|
||||
enum AnnotationQueueObjectType {
|
||||
TRACE
|
||||
OBSERVATION
|
||||
SESSION
|
||||
}
|
||||
|
||||
model AnnotationQueueAssignment {
|
||||
id String @id @default(cuid())
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
userId String @map("user_id")
|
||||
user User @relation(fields: [userId], references: [id], onDelete: Cascade)
|
||||
queueId String @map("queue_id")
|
||||
queue AnnotationQueue @relation(fields: [queueId], references: [id], onDelete: Cascade)
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
@@unique([projectId, queueId, userId])
|
||||
@@map("annotation_queue_assignments")
|
||||
}
|
||||
|
||||
model CronJobs {
|
||||
@@ -548,16 +574,18 @@ model CronJobs {
|
||||
}
|
||||
|
||||
model Dataset {
|
||||
id String @default(cuid())
|
||||
projectId String @map("project_id")
|
||||
name String
|
||||
description String?
|
||||
metadata Json?
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
datasetItems DatasetItem[]
|
||||
datasetRuns DatasetRuns[]
|
||||
id String @default(cuid())
|
||||
projectId String @map("project_id")
|
||||
name String
|
||||
description String?
|
||||
metadata Json?
|
||||
remoteExperimentUrl String? @map("remote_experiment_url")
|
||||
remoteExperimentPayload Json? @map("remote_experiment_payload")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
datasetItems DatasetItem[]
|
||||
datasetRuns DatasetRuns[]
|
||||
|
||||
@@id([id, projectId])
|
||||
@@unique([projectId, name])
|
||||
@@ -751,6 +779,8 @@ model Price {
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
modelId String @map("model_id") // Model is already linked to project (or default), so we don't need projectId here
|
||||
Model Model @relation(fields: [modelId], references: [id], onDelete: Cascade)
|
||||
projectId String? @map("project_id")
|
||||
project Project? @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
usageType String @map("usage_type")
|
||||
price Decimal
|
||||
|
||||
@@ -871,7 +901,7 @@ model JobExecution {
|
||||
jobConfiguration JobConfiguration @relation(fields: [jobConfigurationId], references: [id], onDelete: Cascade)
|
||||
|
||||
jobTemplateId String? @map("job_template_id")
|
||||
jobTemplate EvalTemplate? @relation(fields: [jobTemplateId], references: [id], onDelete: SetNull)
|
||||
jobTemplate EvalTemplate? @relation(fields: [jobTemplateId], references: [id], onDelete: SetNull, onUpdate: NoAction)
|
||||
|
||||
status JobExecutionStatus
|
||||
startTime DateTime? @map("start_time")
|
||||
@@ -902,7 +932,7 @@ model DefaultLlmModel {
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
projectId String @map("project_id")
|
||||
Project Project @relation(fields: [projectId], references: [id])
|
||||
Project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
llmApiKeyId String @map("llm_api_key_id")
|
||||
LlmApiKey LlmApiKeys @relation("LlmApiKeyId", fields: [llmApiKeyId], references: [id], onDelete: Cascade)
|
||||
@@ -963,6 +993,8 @@ model BlobStorageIntegration {
|
||||
enabled Boolean
|
||||
exportFrequency String @map("export_frequency")
|
||||
fileType BlobStorageIntegrationFileType @default(CSV) @map("file_type")
|
||||
exportMode BlobStorageExportMode @default(FULL_HISTORY) @map("export_mode")
|
||||
exportStartDate DateTime? @map("export_start_date")
|
||||
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
@@ -986,6 +1018,14 @@ enum BlobStorageIntegrationType {
|
||||
@@map("BlobStorageIntegrationType")
|
||||
}
|
||||
|
||||
enum BlobStorageExportMode {
|
||||
FULL_HISTORY
|
||||
FROM_TODAY
|
||||
FROM_CUSTOM_DATE
|
||||
|
||||
@@map("BlobStorageExportMode")
|
||||
}
|
||||
|
||||
model BatchExport {
|
||||
id String @id @default(cuid())
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
@@ -1156,6 +1196,8 @@ enum DashboardWidgetChartType {
|
||||
VERTICAL_BAR
|
||||
PIE
|
||||
NUMBER
|
||||
HISTOGRAM
|
||||
PIVOT_TABLE
|
||||
}
|
||||
|
||||
model DashboardWidget {
|
||||
@@ -1212,3 +1254,169 @@ model TableViewPreset {
|
||||
@@unique([projectId, tableName, name])
|
||||
@@map("table_view_presets")
|
||||
}
|
||||
|
||||
model Action {
|
||||
id String @id @default(cuid())
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
type ActionType
|
||||
|
||||
// Configuration specific to each action type
|
||||
config Json // Structured JSON for different action types
|
||||
// For WEBHOOK: { version: "1.0", url: "...", method: "POST", headers: {...}, secretId: "..." }
|
||||
// For ANNOTATION_QUEUE: { version: "1.0", queueId: "..." }
|
||||
|
||||
automations Automation[]
|
||||
automationExecutions AutomationExecution[]
|
||||
|
||||
@@index([projectId])
|
||||
@@map("actions")
|
||||
}
|
||||
|
||||
model Trigger {
|
||||
id String @id @default(cuid())
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
// When should this trigger fire
|
||||
eventSource String // trace, prompt, etc.
|
||||
eventActions String[] // created, updated, deleted
|
||||
filter Json? // Filter conditions (format: { field: "name", operator: "equals", value: "my_trace" })
|
||||
|
||||
// Additional attributes
|
||||
status JobConfigState @default(ACTIVE) @map("status")
|
||||
|
||||
// Link to executions
|
||||
automationExecutions AutomationExecution[]
|
||||
automations Automation[]
|
||||
|
||||
@@index([projectId])
|
||||
@@map("triggers")
|
||||
}
|
||||
|
||||
model Automation {
|
||||
id String @id @default(cuid())
|
||||
name String @map("name")
|
||||
trigger Trigger @relation(fields: [triggerId], references: [id], onDelete: Cascade)
|
||||
triggerId String @map("trigger_id")
|
||||
action Action @relation(fields: [actionId], references: [id], onDelete: Cascade)
|
||||
actionId String @map("action_id")
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
AutomationExecution AutomationExecution[]
|
||||
|
||||
@@index([projectId, actionId, triggerId])
|
||||
@@index([projectId, name])
|
||||
@@map("automations")
|
||||
}
|
||||
|
||||
enum ActionType {
|
||||
WEBHOOK
|
||||
SLACK
|
||||
// More action types can be added as needed
|
||||
}
|
||||
|
||||
enum ActionExecutionStatus {
|
||||
COMPLETED
|
||||
ERROR
|
||||
PENDING
|
||||
CANCELLED
|
||||
}
|
||||
|
||||
model AutomationExecution {
|
||||
id String @id @default(cuid())
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
sourceId String @map("source_id")
|
||||
|
||||
automationId String @map("automation_id")
|
||||
automation Automation @relation(fields: [automationId], references: [id], onDelete: Cascade)
|
||||
|
||||
triggerId String @map("trigger_id")
|
||||
trigger Trigger @relation(fields: [triggerId], references: [id], onDelete: Cascade)
|
||||
|
||||
actionId String @map("action_id")
|
||||
action Action @relation(fields: [actionId], references: [id], onDelete: Cascade)
|
||||
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
status ActionExecutionStatus @default(PENDING) @map("status")
|
||||
input Json @map("input")
|
||||
output Json? @map("output")
|
||||
startedAt DateTime? @map("started_at")
|
||||
finishedAt DateTime? @map("finished_at")
|
||||
error String? @map("error")
|
||||
|
||||
@@index([triggerId])
|
||||
@@index([actionId])
|
||||
@@index([projectId])
|
||||
@@map("automation_executions")
|
||||
}
|
||||
|
||||
// Slack Integration: Stores centralized Slack workspace connection for each project
|
||||
// One project can connect to one Slack workspace, supporting multiple channel automations
|
||||
model SlackIntegration {
|
||||
id String @id @default(cuid())
|
||||
projectId String @unique @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
// Installation details (encrypted using shared encryption utilities)
|
||||
teamId String @map("team_id") // Slack workspace ID
|
||||
teamName String @map("team_name") // Human-readable workspace name
|
||||
botToken String @map("bot_token") // Encrypted bot token for API calls
|
||||
botUserId String @map("bot_user_id") // Bot user ID for workspace
|
||||
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
@@index([teamId])
|
||||
@@map("slack_integrations")
|
||||
}
|
||||
|
||||
// Pending Deletions: Tracks objects (like traces) that are scheduled for batch deletion
|
||||
model PendingDeletion {
|
||||
id String @id @default(cuid())
|
||||
projectId String @map("project_id")
|
||||
project Project @relation(fields: [projectId], references: [id], onDelete: Cascade)
|
||||
|
||||
object String @map("object") // e.g., "trace", "observation", etc.
|
||||
objectId String @map("object_id") // The ID of the object to be deleted
|
||||
isDeleted Boolean @default(false) @map("is_deleted")
|
||||
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
updatedAt DateTime @default(now()) @updatedAt @map("updated_at")
|
||||
|
||||
@@index([projectId, object, isDeleted])
|
||||
@@index([objectId, object])
|
||||
@@map("pending_deletions")
|
||||
}
|
||||
|
||||
model Survey {
|
||||
id String @id @default(cuid())
|
||||
createdAt DateTime @default(now()) @map("created_at")
|
||||
surveyName SurveyName @map("survey_name")
|
||||
response Json
|
||||
userId String? @map("user_id")
|
||||
userEmail String? @map("user_email")
|
||||
orgId String? @map("org_id")
|
||||
org Organization? @relation(fields: [orgId], references: [id], onDelete: Cascade)
|
||||
user User? @relation(fields: [userId], references: [id], onDelete: Cascade)
|
||||
|
||||
@@map("surveys")
|
||||
}
|
||||
|
||||
enum SurveyName {
|
||||
ORG_ONBOARDING @map("org_onboarding")
|
||||
USER_ONBOARDING @map("user_onboarding")
|
||||
|
||||
@@map("SurveyName")
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,314 +0,0 @@
|
||||
import { SEED_PROMPTS } from "../prisma/seed";
|
||||
import { prisma } from "../src/db";
|
||||
import { clickhouseClient, logger } from "../src/server";
|
||||
|
||||
function randn_bm(min: number, max: number, skew: number) {
|
||||
let u = 0,
|
||||
v = 0;
|
||||
while (u === 0) u = Math.random(); //Converting [0,1) to (0,1)
|
||||
while (v === 0) v = Math.random();
|
||||
let num = Math.sqrt(-2.0 * Math.log(u)) * Math.cos(2.0 * Math.PI * v);
|
||||
|
||||
num = num / 10.0 + 0.5; // Translate to 0 -> 1
|
||||
if (num > 1 || num < 0)
|
||||
num = randn_bm(min, max, skew); // resample between 0 and 1 if out of range
|
||||
else {
|
||||
num = Math.pow(num, skew); // Skew
|
||||
num *= max - min; // Stretch to fill range
|
||||
num += min; // offset to min
|
||||
}
|
||||
return num;
|
||||
}
|
||||
|
||||
export const prepareClickhouse = async (
|
||||
projectIds: string[],
|
||||
opts: {
|
||||
numberOfDays: number;
|
||||
totalObservations: number;
|
||||
},
|
||||
) => {
|
||||
logger.info(
|
||||
`Preparing Clickhouse for ${projectIds.length} projects and ${opts.numberOfDays} days.`,
|
||||
);
|
||||
|
||||
const projectData = projectIds.map((projectId) => {
|
||||
const observationsPerProject = Math.ceil(
|
||||
randn_bm(0, opts.totalObservations, 2),
|
||||
); // Skew the number of observations
|
||||
|
||||
const tracesPerProject = Math.floor(observationsPerProject / 6); // On average, one trace should have 6 observations
|
||||
const scoresPerProject = tracesPerProject * 10; // On average, one trace should have 10 scores
|
||||
return {
|
||||
projectId,
|
||||
observationsPerProject,
|
||||
tracesPerProject,
|
||||
scoresPerProject,
|
||||
};
|
||||
});
|
||||
|
||||
for (const data of projectData) {
|
||||
const {
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
observationsPerProject,
|
||||
scoresPerProject,
|
||||
} = data;
|
||||
logger.info(
|
||||
`Preparing Clickhouse for ${projectId}: Traces: ${tracesPerProject}, Scores: ${scoresPerProject}, Observations: ${observationsPerProject}`,
|
||||
);
|
||||
|
||||
const tracesQuery = `
|
||||
INSERT INTO traces
|
||||
SELECT toString(number) AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
concat('name_', toString(rand() % 100)) AS name,
|
||||
concat('user_id_', toInt64(randExponential(1 / 100))) AS user_id,
|
||||
map('prototype', 'test') AS metadata,
|
||||
concat('release_', toString(randUniform(0, 100))) AS release,
|
||||
concat('version_', toString(randUniform(0, 100))) AS version,
|
||||
'${projectId}' AS project_id,
|
||||
'default' AS environment,
|
||||
if(rand() < 0.8, true, false) as public,
|
||||
if(rand() < 0.8, true, false) as bookmarked,
|
||||
array('tag1', 'tag2') as tags,
|
||||
repeat('input', toInt64(randExponential(1 / 100))) AS input,
|
||||
repeat('output', toInt64(randExponential(1 / 100))) AS output,
|
||||
if(randUniform(0, 1) < 0.2, NULL, concat('session_', toString(rand() % 1000))) AS session_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${tracesPerProject});
|
||||
`;
|
||||
|
||||
const observationsQuery = `
|
||||
INSERT INTO observations
|
||||
SELECT toString(number) AS id,
|
||||
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
|
||||
'${projectId}' AS project_id,
|
||||
'default' AS environment,
|
||||
if(randUniform(0, 1) < 0.47, 'GENERATION', if(randUniform(0, 1) < 0.94, 'SPAN', 'EVENT')) AS type,
|
||||
toString(rand()) AS parent_observation_id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS start_time,
|
||||
addSeconds(start_time, if(rand() < 0.6, floor(randUniform(0, 20)), floor(randUniform(0, 3600)))) AS end_time,
|
||||
concat('name', toString(rand() % 100)) AS name,
|
||||
map('prototype', 'test') AS metadata,
|
||||
if(randUniform(0, 1) < 0.9, 'DEFAULT', if(randUniform(0, 1) < 0.5, 'ERROR', if(randUniform(0, 1) < 0.5, 'DEBUG', 'WARNING'))) AS level,
|
||||
'status_message' AS status_message,
|
||||
'version' AS version,
|
||||
repeat('input', toInt64(randExponential(1 / 100))) AS input,
|
||||
repeat('output', toInt64(randExponential(1 / 100))) AS output,
|
||||
case
|
||||
when number % 2 = 0 then 'claude-3-haiku-20230407'
|
||||
else 'gpt-4'
|
||||
end as provided_model_name,
|
||||
case
|
||||
when number % 2 = 0 then 'cltra4wbs0000k1407g0ya3'
|
||||
else '1cmtk9y0000y3y79x9jgxj'
|
||||
end as internal_model_id,
|
||||
if("type" = 'GENERATION',
|
||||
'{"temperature": 0.7, "max_tokens": 150}',
|
||||
'{}') AS model_parameters,
|
||||
if("type" = 'GENERATION',
|
||||
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))),
|
||||
map()) AS provided_usage_details,
|
||||
if("type" = 'GENERATION',
|
||||
map('input', toUInt64(randUniform(0, 1000)), 'output', toUInt64(randUniform(0, 1000)), 'total', toUInt64(randUniform(0, 2000))),
|
||||
map()) AS usage_details,
|
||||
if("type" = 'GENERATION',
|
||||
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)),
|
||||
map()) AS provided_cost_details,
|
||||
if("type" = 'GENERATION',
|
||||
map('input', toDecimal64(randUniform(0, 1000), 12), 'output', toDecimal64(randUniform(0, 1000), 12), 'total', toDecimal64(randUniform(0, 2000), 12)),
|
||||
map()) AS cost_details,
|
||||
if("type" = 'GENERATION',
|
||||
toDecimal64(randUniform(0, 2000), 12),
|
||||
NULL) AS total_cost,
|
||||
addMilliseconds(start_time, if(rand() < 0.6, floor(randUniform(0, 500)), floor(randUniform(0, 600)))) AS completion_start_time,
|
||||
array(${SEED_PROMPTS.map((p) => `concat('${p.id}',project_id)`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_id,
|
||||
array(${SEED_PROMPTS.map((p) => `'${p.name}'`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_name,
|
||||
array(${SEED_PROMPTS.map((p) => `'${p.version}'`).join(
|
||||
",",
|
||||
)})[(number % ${SEED_PROMPTS.length})+1] AS prompt_version,
|
||||
start_time AS created_at,
|
||||
start_time AS updated_at,
|
||||
start_time AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${observationsPerProject});
|
||||
`;
|
||||
|
||||
const scoresQuery = `
|
||||
INSERT INTO scores
|
||||
SELECT toString(number) AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
'${projectId}' AS project_id,
|
||||
'default' AS environment,
|
||||
toString(floor(randUniform(0, ${tracesPerProject}))) AS trace_id,
|
||||
NULL AS session_id,
|
||||
NULL AS dataset_run_id,
|
||||
if(
|
||||
rand() > 0.9,
|
||||
toString(floor(randUniform(0, ${observationsPerProject}))),
|
||||
NULL
|
||||
) AS observation_id,
|
||||
concat('name_', toString(rand() % 10)) AS name,
|
||||
randUniform(0, 100) as value,
|
||||
'API' as source,
|
||||
'comment' as comment,
|
||||
map('prototype', 'test') AS metadata,
|
||||
toString(rand() % 100) as author_user_id,
|
||||
toString(rand() % 100) as config_id,
|
||||
if (rand() < 0.33, 'NUMERIC', if (rand() < 0.5, 'CATEGORICAL', 'BOOLEAN')) as data_type,
|
||||
toString(rand() % 100) as string_value,
|
||||
NULL as queue_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${scoresPerProject});
|
||||
`;
|
||||
|
||||
const queries = [tracesQuery, scoresQuery, observationsQuery];
|
||||
|
||||
for (const query of queries) {
|
||||
logger.info(`Executing query: ${query}`);
|
||||
await clickhouseClient().command({
|
||||
query,
|
||||
clickhouse_settings: {
|
||||
wait_end_of_query: 1,
|
||||
},
|
||||
});
|
||||
}
|
||||
// we also need to upsert trace sessions in postgres
|
||||
|
||||
const sessionQuery = `
|
||||
SELECT session_id, project_id
|
||||
FROM traces
|
||||
WHERE session_id IS NOT NULL;
|
||||
`;
|
||||
const sessionResult = await clickhouseClient().query({
|
||||
query: sessionQuery,
|
||||
format: "JSONEachRow",
|
||||
});
|
||||
|
||||
const sessionData = await sessionResult.json<{
|
||||
session_id: string;
|
||||
project_id: string;
|
||||
}>();
|
||||
|
||||
const sessionsToScore = sessionData
|
||||
.filter(() => Math.random() < 0.5)
|
||||
.slice(0, Math.min(500, sessionData.length));
|
||||
|
||||
if (sessionsToScore.length > 0) {
|
||||
// Generate session scores query with specific session IDs
|
||||
const sessionScoresQuery = `
|
||||
INSERT INTO scores
|
||||
SELECT
|
||||
concat('session-', toString(number)) AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
'${projectId}' AS project_id,
|
||||
'default' AS environment,
|
||||
NULL AS trace_id,
|
||||
arrayElement(['${sessionsToScore.map((s) => s.session_id).join("','")}'], 1 + (number % ${sessionsToScore.length})) AS session_id,
|
||||
NULL AS dataset_run_id,
|
||||
NULL AS observation_id,
|
||||
concat('session_quality_', toString(rand() % 10)) AS name,
|
||||
randUniform(0, 100) AS value,
|
||||
'API' AS source,
|
||||
'Session-level assessment score' AS comment,
|
||||
map('key', 'value') AS metadata,
|
||||
toString(rand() % 100) AS author_user_id,
|
||||
toString(rand() % 100) AS config_id,
|
||||
if(rand() < 0.33, 'NUMERIC', if(rand() < 0.5, 'CATEGORICAL', 'BOOLEAN')) AS data_type,
|
||||
toString(rand() % 100) AS string_value,
|
||||
NULL AS queue_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${sessionsToScore.length})
|
||||
`;
|
||||
|
||||
await clickhouseClient().command({
|
||||
query: sessionScoresQuery,
|
||||
clickhouse_settings: {
|
||||
wait_end_of_query: 1,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
const idProjectIdCombinations = sessionData.map((session) => ({
|
||||
id: session.session_id,
|
||||
projectId: session.project_id,
|
||||
public: Math.random() < 0.1,
|
||||
bookmarked: Math.random() < 0.1,
|
||||
}));
|
||||
|
||||
await prisma.traceSession.createMany({
|
||||
data: idProjectIdCombinations,
|
||||
skipDuplicates: true,
|
||||
});
|
||||
}
|
||||
|
||||
const tables = ["traces", "scores", "observations"];
|
||||
for (const table of tables) {
|
||||
const query = `
|
||||
SELECT
|
||||
project_id,
|
||||
count() AS per_project_count,
|
||||
bar(per_project_count, 0, (
|
||||
SELECT count(*)
|
||||
FROM ${table}
|
||||
), 50) AS bar_representation
|
||||
FROM ${table}
|
||||
GROUP BY project_id
|
||||
ORDER BY count() desc
|
||||
`;
|
||||
|
||||
const result = await clickhouseClient().query({
|
||||
query,
|
||||
format: "TabSeparated",
|
||||
});
|
||||
|
||||
logger.info(
|
||||
`${table.charAt(0).toUpperCase() + table.slice(1)} per Project: \n` +
|
||||
(await result.text()),
|
||||
);
|
||||
}
|
||||
|
||||
const tablesWithDateColumns = [
|
||||
{ name: "traces", dateColumn: "timestamp" },
|
||||
{ name: "scores", dateColumn: "timestamp" },
|
||||
{ name: "observations", dateColumn: "start_time" },
|
||||
];
|
||||
|
||||
for (const { name: table, dateColumn } of tablesWithDateColumns) {
|
||||
const query = `
|
||||
SELECT
|
||||
toDate(${dateColumn}) AS event_date,
|
||||
count() AS per_date_count,
|
||||
bar(per_date_count, 0, (
|
||||
SELECT count(*)
|
||||
FROM ${table}
|
||||
), 50) AS bar_representation
|
||||
FROM ${table}
|
||||
GROUP BY event_date
|
||||
ORDER BY event_date desc
|
||||
`;
|
||||
|
||||
const result = await clickhouseClient().query({
|
||||
query,
|
||||
format: "TabSeparated",
|
||||
});
|
||||
|
||||
logger.info(
|
||||
`${table.charAt(0).toUpperCase() + table.slice(1)} per Date: \n` +
|
||||
(await result.text()),
|
||||
);
|
||||
}
|
||||
};
|
||||
+5
-4
@@ -1,8 +1,8 @@
|
||||
import { randomUUID } from "crypto";
|
||||
import { prisma } from "../src/db";
|
||||
import { getDisplaySecretKey, hashSecretKey, logger } from "../src/server";
|
||||
import { prepareClickhouse } from "./prepareClickhouse";
|
||||
import { redis } from "../src/server";
|
||||
import { prisma } from "../../src/db";
|
||||
import { getDisplaySecretKey, hashSecretKey, logger } from "../../src/server";
|
||||
import { prepareClickhouse } from "./prepare-clickhouse";
|
||||
import { redis } from "../../src/server";
|
||||
|
||||
const createRandomProjectId = () => randomUUID().toString();
|
||||
|
||||
@@ -117,6 +117,7 @@ async function main() {
|
||||
await prepareClickhouse(createdProjectIds, {
|
||||
numberOfDays,
|
||||
totalObservations: totalObservations ?? 1000,
|
||||
numberOfRuns: 3,
|
||||
});
|
||||
|
||||
logger.info("Clickhouse preparation completed successfully.");
|
||||
@@ -0,0 +1,35 @@
|
||||
import { SeederOrchestrator } from "./utils/seeder-orchestrator";
|
||||
import { SeederOptions } from "./utils/types";
|
||||
import { logger } from "../../src/server";
|
||||
|
||||
/**
|
||||
* ClickHouse data preparation using the seeder abstraction.
|
||||
*/
|
||||
export const prepareClickhouse = async (
|
||||
projectIds: string[],
|
||||
opts: {
|
||||
numberOfDays: number;
|
||||
totalObservations: number;
|
||||
numberOfRuns?: number;
|
||||
},
|
||||
) => {
|
||||
logger.info(
|
||||
`Preparing ClickHouse for ${projectIds.length} projects and ${opts.numberOfDays} days.`,
|
||||
);
|
||||
|
||||
const formattedOpts: SeederOptions = {
|
||||
numberOfDays: opts.numberOfDays,
|
||||
totalObservations: opts.totalObservations,
|
||||
numberOfRuns: opts.numberOfRuns || 1,
|
||||
};
|
||||
|
||||
const orchestrator = new SeederOrchestrator();
|
||||
|
||||
try {
|
||||
await orchestrator.executeFullSeed(projectIds, formattedOpts);
|
||||
logger.info("ClickHouse preparation completed successfully");
|
||||
} catch (error) {
|
||||
logger.error("ClickHouse preparation failed:", error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,31 @@
|
||||
import { prisma } from "../../src/db";
|
||||
import { redis } from "../../src/server";
|
||||
import { prepareClickhouse } from "./prepare-clickhouse";
|
||||
|
||||
async function main() {
|
||||
try {
|
||||
const projectIds = ["7a88fb47-b4e2-43b8-a06c-a5ce950dc53a"]; // Example project IDs
|
||||
if (
|
||||
await prisma.project.findFirst({
|
||||
where: { id: "239ad00f-562f-411d-af14-831c75ddd875" },
|
||||
})
|
||||
) {
|
||||
projectIds.push("239ad00f-562f-411d-af14-831c75ddd875");
|
||||
}
|
||||
await prepareClickhouse(projectIds, {
|
||||
numberOfDays: 3,
|
||||
totalObservations: 1000,
|
||||
numberOfRuns: 3,
|
||||
});
|
||||
|
||||
console.log("Clickhouse preparation completed successfully.");
|
||||
} catch (error) {
|
||||
console.error("Error during Clickhouse preparation:", error);
|
||||
} finally {
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
console.log("Disconnected from Clickhouse.");
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -0,0 +1,937 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { parseArgs } from "node:util";
|
||||
import { hash } from "bcryptjs";
|
||||
import { v4 } from "uuid";
|
||||
import { encrypt } from "../../src/encryption";
|
||||
import {
|
||||
type JobConfiguration,
|
||||
JobExecutionStatus,
|
||||
PrismaClient,
|
||||
type Project,
|
||||
ScoreDataType,
|
||||
} from "../../src/index";
|
||||
import { getDisplaySecretKey, hashSecretKey, logger } from "../../src/server";
|
||||
import { redis } from "../../src/server/redis/redis";
|
||||
import {
|
||||
EVAL_TRACE_COUNT,
|
||||
FAILED_EVAL_TRACE_INTERVAL,
|
||||
SEED_CHAT_ML_PROMPTS,
|
||||
SEED_DATASETS,
|
||||
SEED_EVALUATOR_CONFIGS,
|
||||
SEED_EVALUATOR_TEMPLATES,
|
||||
SEED_PROMPT_VERSIONS,
|
||||
SEED_TEXT_PROMPTS,
|
||||
} from "./utils/postgres-seed-constants";
|
||||
import {
|
||||
generateDatasetItemId,
|
||||
generateDatasetRunTraceId,
|
||||
generateEvalObservationId,
|
||||
generateEvalScoreId,
|
||||
generateEvalTraceId,
|
||||
} from "./utils/seed-helpers";
|
||||
|
||||
type ConfigCategory = {
|
||||
label: string;
|
||||
value: number;
|
||||
};
|
||||
|
||||
const options = {
|
||||
environment: { type: "string" },
|
||||
} as const;
|
||||
|
||||
const prisma = new PrismaClient();
|
||||
|
||||
async function main() {
|
||||
const environment = parseArgs({
|
||||
options,
|
||||
}).values.environment;
|
||||
|
||||
const seedOrgId = "seed-org-id";
|
||||
const seedProjectId = "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a";
|
||||
const seedUserId1 = "user-1"; // Owner of org
|
||||
const seedUserId2 = "user-2"; // Member of org, admin of project
|
||||
|
||||
const user = await prisma.user.upsert({
|
||||
where: { id: seedUserId1 },
|
||||
update: {
|
||||
name: "Demo User",
|
||||
email: "demo@langfuse.com",
|
||||
password: await hash("password", 12),
|
||||
},
|
||||
create: {
|
||||
id: seedUserId1,
|
||||
name: "Demo User",
|
||||
email: "demo@langfuse.com",
|
||||
password: await hash("password", 12),
|
||||
image: "https://static.langfuse.com/langfuse-dev%2Fexample-avatar.png",
|
||||
},
|
||||
});
|
||||
const user2 = await prisma.user.upsert({
|
||||
where: { id: seedUserId2 },
|
||||
update: {
|
||||
name: "Demo User 2",
|
||||
email: "member@langfuse.com",
|
||||
password: await hash("password", 12),
|
||||
},
|
||||
create: {
|
||||
id: seedUserId2,
|
||||
name: "Demo User 2",
|
||||
email: "member@langfuse.com",
|
||||
password: await hash("password", 12),
|
||||
},
|
||||
});
|
||||
|
||||
await prisma.organization.upsert({
|
||||
where: { id: seedOrgId },
|
||||
update: {
|
||||
name: "Seed Org",
|
||||
cloudConfig: {
|
||||
plan: "Team",
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: seedOrgId,
|
||||
name: "Seed Org",
|
||||
cloudConfig: {
|
||||
plan: "Team",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const project1 = await prisma.project.upsert({
|
||||
where: { id: seedProjectId },
|
||||
update: {
|
||||
name: "llm-app",
|
||||
orgId: seedOrgId,
|
||||
},
|
||||
create: {
|
||||
id: seedProjectId,
|
||||
name: "llm-app",
|
||||
orgId: seedOrgId,
|
||||
},
|
||||
});
|
||||
|
||||
await prisma.organizationMembership.upsert({
|
||||
where: {
|
||||
orgId_userId: {
|
||||
userId: user.id,
|
||||
orgId: seedOrgId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
userId: user.id,
|
||||
orgId: seedOrgId,
|
||||
role: "OWNER",
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
const orgMembership2 = await prisma.organizationMembership.upsert({
|
||||
where: {
|
||||
orgId_userId: {
|
||||
userId: user2.id,
|
||||
orgId: seedOrgId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
userId: user2.id,
|
||||
orgId: seedOrgId,
|
||||
role: "MEMBER",
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
await prisma.projectMembership.upsert({
|
||||
where: {
|
||||
projectId_userId: {
|
||||
projectId: project1.id,
|
||||
userId: user2.id,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
userId: user2.id,
|
||||
projectId: project1.id,
|
||||
role: "ADMIN",
|
||||
orgMembershipId: orgMembership2.id,
|
||||
},
|
||||
update: {
|
||||
orgMembershipId: orgMembership2.id,
|
||||
},
|
||||
});
|
||||
|
||||
await prisma.prompt.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: seedProjectId,
|
||||
name: "summary-prompt",
|
||||
version: 1,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
name: "summary-prompt",
|
||||
project: { connect: { id: seedProjectId } },
|
||||
prompt: "prompt {{variable}} {{anotherVariable}}",
|
||||
labels: ["production", "latest"],
|
||||
version: 1,
|
||||
createdBy: "user-1",
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
const seedApiKey = {
|
||||
id: "seed-api-key",
|
||||
secret: process.env.SEED_SECRET_KEY ?? "sk-lf-1234567890", // eslint-disable-line turbo/no-undeclared-env-vars
|
||||
public: "pk-lf-1234567890",
|
||||
note: "seeded key",
|
||||
};
|
||||
|
||||
if (!(await prisma.apiKey.findUnique({ where: { id: seedApiKey.id } }))) {
|
||||
await prisma.apiKey.create({
|
||||
data: {
|
||||
note: seedApiKey.note,
|
||||
id: seedApiKey.id,
|
||||
publicKey: seedApiKey.public,
|
||||
hashedSecretKey: await hashSecretKey(seedApiKey.secret),
|
||||
displaySecretKey: getDisplaySecretKey(seedApiKey.secret),
|
||||
scope: "PROJECT",
|
||||
project: {
|
||||
connect: {
|
||||
id: project1.id,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
// Do not run the following for local docker compose setup
|
||||
if (environment === "examples" || environment === "load") {
|
||||
const seedOrgIdOrg2 = "demo-org-id";
|
||||
const project2Id = "239ad00f-562f-411d-af14-831c75ddd875";
|
||||
const org2 = await prisma.organization.upsert({
|
||||
where: { id: seedOrgIdOrg2 },
|
||||
update: {
|
||||
name: "Langfuse Demo",
|
||||
},
|
||||
create: {
|
||||
id: seedOrgIdOrg2,
|
||||
name: "Langfuse Demo",
|
||||
},
|
||||
});
|
||||
const project2 = await prisma.project.upsert({
|
||||
where: { id: project2Id },
|
||||
create: {
|
||||
id: project2Id,
|
||||
name: "demo-app",
|
||||
orgId: org2.id,
|
||||
},
|
||||
update: { orgId: seedOrgIdOrg2 },
|
||||
});
|
||||
await prisma.organizationMembership.upsert({
|
||||
where: {
|
||||
orgId_userId: {
|
||||
userId: user.id,
|
||||
orgId: seedOrgIdOrg2,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
userId: user.id,
|
||||
orgId: seedOrgIdOrg2,
|
||||
role: "VIEWER",
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
const secondKey = {
|
||||
id: "seed-api-key-2",
|
||||
secret: process.env.SEED_SECRET_KEY ?? "sk-lf-asdfghjkl", // eslint-disable-line turbo/no-undeclared-env-vars
|
||||
public: "pk-lf-asdfghjkl",
|
||||
note: "seeded key 2",
|
||||
};
|
||||
if (!(await prisma.apiKey.findUnique({ where: { id: secondKey.id } }))) {
|
||||
await prisma.apiKey.create({
|
||||
data: {
|
||||
note: secondKey.note,
|
||||
id: secondKey.id,
|
||||
publicKey: secondKey.public,
|
||||
hashedSecretKey: await hashSecretKey(secondKey.secret),
|
||||
displaySecretKey: getDisplaySecretKey(secondKey.secret),
|
||||
scope: "PROJECT",
|
||||
project: {
|
||||
connect: {
|
||||
id: project2.id,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
const configIdsAndNames = await generateConfigsForProject([
|
||||
project1,
|
||||
project2,
|
||||
]);
|
||||
|
||||
await generateQueuesForProject([project1, project2], configIdsAndNames);
|
||||
await generatePromptsForProject([project1, project2]);
|
||||
await createDatasets(project1, project2);
|
||||
await createTraceSessions(project1, project2);
|
||||
|
||||
// If openai key is in environment, add it to the projects LLM API keys
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY; // eslint-disable-line turbo/no-undeclared-env-vars
|
||||
|
||||
if (OPENAI_API_KEY) {
|
||||
await prisma.llmApiKeys.create({
|
||||
data: {
|
||||
projectId: project1.id,
|
||||
secretKey: encrypt(OPENAI_API_KEY),
|
||||
displaySecretKey: getDisplaySecretKey(OPENAI_API_KEY),
|
||||
provider: "openai",
|
||||
adapter: "openai",
|
||||
},
|
||||
});
|
||||
} else {
|
||||
logger.warn(
|
||||
"No OPENAI_API_KEY found in environment. Skipping seeding LLM API key.",
|
||||
);
|
||||
}
|
||||
|
||||
// add eval objects
|
||||
for (const evalTemplate of SEED_EVALUATOR_TEMPLATES) {
|
||||
await prisma.evalTemplate.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: project1.id,
|
||||
name: evalTemplate.name,
|
||||
version: 1,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: evalTemplate.id,
|
||||
projectId: project1.id,
|
||||
name: evalTemplate.name,
|
||||
version: evalTemplate.version,
|
||||
prompt: evalTemplate.prompt,
|
||||
model: evalTemplate.model,
|
||||
vars: evalTemplate.vars,
|
||||
provider: evalTemplate.provider,
|
||||
outputSchema: evalTemplate.outputSchema,
|
||||
modelParams: evalTemplate.modelParams,
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
}
|
||||
|
||||
for (const evalConfig of SEED_EVALUATOR_CONFIGS) {
|
||||
await prisma.jobConfiguration.upsert({
|
||||
where: {
|
||||
id: evalConfig.id,
|
||||
},
|
||||
create: {
|
||||
id: evalConfig.id,
|
||||
evalTemplateId: evalConfig.evalTemplateId,
|
||||
projectId: project1.id,
|
||||
jobType: evalConfig.jobType as any,
|
||||
status: evalConfig.status as any,
|
||||
scoreName: evalConfig.scoreName,
|
||||
filter: evalConfig.filter,
|
||||
variableMapping: evalConfig.variableMapping,
|
||||
targetObject: evalConfig.targetObject,
|
||||
sampling: evalConfig.sampling,
|
||||
delay: evalConfig.delay,
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
}
|
||||
|
||||
await generateEvalJobExecutions(
|
||||
[project1, project2],
|
||||
SEED_EVALUATOR_CONFIGS as unknown as Partial<JobConfiguration>[],
|
||||
);
|
||||
|
||||
await createDashboardsAndWidgets([project1, project2]);
|
||||
|
||||
await prisma.llmSchema.createMany({
|
||||
data: [
|
||||
{
|
||||
projectId: project1.id,
|
||||
name: "get_weather",
|
||||
description: "Fetches weather in Celsius for a given location",
|
||||
schema: {
|
||||
type: "object",
|
||||
properties: {
|
||||
location: {
|
||||
type: "string",
|
||||
description: "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
unit: {
|
||||
type: "string",
|
||||
enum: ["celsius", "fahrenheit"],
|
||||
},
|
||||
},
|
||||
required: ["location", "unit"],
|
||||
},
|
||||
},
|
||||
{
|
||||
projectId: project1.id,
|
||||
name: "calculator",
|
||||
description: "Performs basic arithmetic calculations",
|
||||
schema: {
|
||||
type: "object",
|
||||
properties: {
|
||||
expression: {
|
||||
type: "string",
|
||||
description:
|
||||
"The mathematical expression to evaluate, e.g. '2 + 2'",
|
||||
},
|
||||
},
|
||||
required: ["expression"],
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
main()
|
||||
.then(async () => {
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
})
|
||||
.catch(async (e) => {
|
||||
logger.error(e);
|
||||
await prisma.$disconnect();
|
||||
redis?.disconnect();
|
||||
logger.info("Disconnected from postgres and redis");
|
||||
process.exit(1);
|
||||
});
|
||||
|
||||
async function createDashboardsAndWidgets(projects: Project[]) {
|
||||
logger.info("Creating dashboards and widgets");
|
||||
|
||||
// Process each project
|
||||
for (const project of projects) {
|
||||
const widget = await prisma.dashboardWidget.upsert({
|
||||
where: { id: "cabc" },
|
||||
create: {
|
||||
id: "cabc",
|
||||
projectId: project.id,
|
||||
name: "Trace Counts",
|
||||
description: "Trace Counts by Name Over Time",
|
||||
view: "TRACES",
|
||||
dimensions: [{ field: "name" }],
|
||||
metrics: [{ measure: "count", agg: "count" }],
|
||||
filters: [],
|
||||
chartType: "BAR_TIME_SERIES",
|
||||
chartConfig: {
|
||||
type: "BAR_TIME_SERIES",
|
||||
},
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
const widget2 = await prisma.dashboardWidget.upsert({
|
||||
where: { id: "cdef" },
|
||||
create: {
|
||||
id: "cdef",
|
||||
projectId: project.id,
|
||||
name: "Observation Latencies by Model",
|
||||
description: "p95 Observation Latencies by Model Name",
|
||||
view: "OBSERVATIONS",
|
||||
dimensions: [{ field: "providedModelName" }],
|
||||
metrics: [{ measure: "count", agg: "sum" }],
|
||||
filters: [],
|
||||
chartType: "LINE_TIME_SERIES",
|
||||
chartConfig: {
|
||||
type: "LINE_TIME_SERIES",
|
||||
},
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
// Create a dashboard with multiple widgets
|
||||
await prisma.dashboard.upsert({
|
||||
where: { id: "seed-dashboard" },
|
||||
create: {
|
||||
id: "seed-dashboard",
|
||||
projectId: project.id,
|
||||
name: "Performance Overview",
|
||||
description: "Dashboard with various performance metrics",
|
||||
definition: {
|
||||
widgets: [
|
||||
{
|
||||
type: "widget",
|
||||
id: randomUUID(),
|
||||
widgetId: widget.id,
|
||||
x: 0,
|
||||
y: 0,
|
||||
x_size: 6,
|
||||
y_size: 6,
|
||||
},
|
||||
{
|
||||
type: "widget",
|
||||
id: randomUUID(),
|
||||
widgetId: widget2.id,
|
||||
x: 6,
|
||||
y: 0,
|
||||
x_size: 6,
|
||||
y_size: 6,
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
export async function createDatasets(
|
||||
project1: {
|
||||
id: string;
|
||||
orgId: string;
|
||||
createdAt: Date;
|
||||
updatedAt: Date;
|
||||
name: string;
|
||||
},
|
||||
project2: {
|
||||
id: string;
|
||||
orgId: string;
|
||||
createdAt: Date;
|
||||
updatedAt: Date;
|
||||
name: string;
|
||||
},
|
||||
) {
|
||||
for (const data of SEED_DATASETS) {
|
||||
for (const projectId of [project1.id, project2.id]) {
|
||||
const datasetName = data.name;
|
||||
|
||||
// check if ds already exists
|
||||
const dataset =
|
||||
(await prisma.dataset.findFirst({
|
||||
where: {
|
||||
projectId,
|
||||
name: datasetName,
|
||||
},
|
||||
})) ??
|
||||
(await prisma.dataset.create({
|
||||
data: {
|
||||
name: datasetName,
|
||||
description: data.description,
|
||||
projectId,
|
||||
metadata: data.metadata,
|
||||
id: `${datasetName}-${projectId.slice(-8)}`,
|
||||
},
|
||||
}));
|
||||
|
||||
const datasetItemIds: string[] = [];
|
||||
for (let index = 0; index < data.items.length; index++) {
|
||||
const item = data.items[index];
|
||||
const sourceTraceId =
|
||||
Math.random() > 0.3
|
||||
? `${Math.floor(Math.random() * 100)}`
|
||||
: undefined;
|
||||
|
||||
// Use upsert to prevent duplicates
|
||||
const datasetItem = await prisma.datasetItem.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
id: generateDatasetItemId(
|
||||
datasetName,
|
||||
index,
|
||||
projectId,
|
||||
SEED_DATASETS.indexOf(data) || 0,
|
||||
),
|
||||
projectId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
projectId,
|
||||
id: generateDatasetItemId(
|
||||
datasetName,
|
||||
index,
|
||||
projectId,
|
||||
SEED_DATASETS.indexOf(data) || 0,
|
||||
),
|
||||
datasetId: dataset.id,
|
||||
sourceTraceId: sourceTraceId ?? null,
|
||||
sourceObservationId: null,
|
||||
input: item.input,
|
||||
expectedOutput: item.output,
|
||||
metadata: Math.random() > 0.5 ? { key: "value" } : undefined,
|
||||
},
|
||||
update: {}, // Don't update if it exists
|
||||
});
|
||||
datasetItemIds.push(datasetItem.id);
|
||||
}
|
||||
|
||||
for (let datasetRunNumber = 0; datasetRunNumber < 3; datasetRunNumber++) {
|
||||
const datasetRun = await prisma.datasetRuns.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
id: `demo-dataset-run-${datasetRunNumber}-${projectId.slice(-8)}`,
|
||||
projectId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
projectId,
|
||||
id: `demo-dataset-run-${datasetRunNumber}-${projectId.slice(-8)}`,
|
||||
name: `demo-dataset-run-${datasetRunNumber}`,
|
||||
description: Math.random() > 0.5 ? "Dataset run description" : "",
|
||||
datasetId: dataset.id,
|
||||
metadata: [
|
||||
undefined,
|
||||
"string",
|
||||
100,
|
||||
{ key: "value" },
|
||||
["tag1", "tag2"],
|
||||
][datasetRunNumber % 5],
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
|
||||
for (let index = 0; index < datasetItemIds.length; index++) {
|
||||
await prisma.datasetRunItems.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
id: `${dataset.id}-${index}-${datasetRunNumber}`,
|
||||
projectId,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: `${dataset.id}-${index}-${datasetRunNumber}`,
|
||||
projectId,
|
||||
datasetItemId: datasetItemIds[index],
|
||||
traceId: `${generateDatasetRunTraceId(datasetName, index, projectId, datasetRunNumber)}`,
|
||||
datasetRunId: datasetRun.id,
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function generateEvalJobExecutions(
|
||||
projects: Project[],
|
||||
evalJobConfigurations: Partial<JobConfiguration>[],
|
||||
) {
|
||||
for (const project of projects) {
|
||||
for (let i = 0; i < EVAL_TRACE_COUNT; i++) {
|
||||
const jobConfiguration =
|
||||
evalJobConfigurations[i % evalJobConfigurations.length];
|
||||
|
||||
const isFailed = i % FAILED_EVAL_TRACE_INTERVAL === 0;
|
||||
await prisma.jobExecution.create({
|
||||
data: {
|
||||
projectId: project.id,
|
||||
jobTemplateId: jobConfiguration.evalTemplateId,
|
||||
jobInputTraceId: generateEvalTraceId(
|
||||
jobConfiguration.evalTemplateId!,
|
||||
i,
|
||||
project.id,
|
||||
),
|
||||
jobConfigurationId: jobConfiguration.id!,
|
||||
status: isFailed
|
||||
? JobExecutionStatus.ERROR
|
||||
: JobExecutionStatus.COMPLETED,
|
||||
error: isFailed ? "Error message" : undefined,
|
||||
jobOutputScoreId: generateEvalScoreId(
|
||||
jobConfiguration.evalTemplateId!,
|
||||
i,
|
||||
project.id,
|
||||
),
|
||||
jobInputObservationId: generateEvalObservationId(
|
||||
jobConfiguration.evalTemplateId!,
|
||||
i,
|
||||
project.id,
|
||||
),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function generatePromptsForProject(projects: Project[]) {
|
||||
const promptIds = new Map<string, string[]>();
|
||||
|
||||
await Promise.all(
|
||||
projects.map(async (project) => {
|
||||
const promptIdsForProject = await generatePrompts(project);
|
||||
promptIds.set(project.id, promptIdsForProject);
|
||||
}),
|
||||
);
|
||||
return promptIds;
|
||||
}
|
||||
|
||||
export const PROMPT_IDS: string[] = [];
|
||||
|
||||
async function generatePrompts(project: Project) {
|
||||
const promptIds = [];
|
||||
for (const prompt of SEED_TEXT_PROMPTS) {
|
||||
const versions = Math.floor(Math.random() * 20) + 1;
|
||||
for (let i = 1; i <= versions; i++) {
|
||||
const promptId = `prompt-${v4()}`;
|
||||
await prisma.prompt.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: project.id,
|
||||
name: prompt.name,
|
||||
version: i,
|
||||
},
|
||||
id: promptId,
|
||||
},
|
||||
create: {
|
||||
id: promptId,
|
||||
projectId: project.id,
|
||||
createdBy: prompt.createdBy,
|
||||
prompt: `${prompt.prompt} version ${i} content`,
|
||||
name: prompt.name,
|
||||
version: i,
|
||||
labels: i === versions ? prompt.labels : [],
|
||||
},
|
||||
update: {
|
||||
id: promptId,
|
||||
},
|
||||
});
|
||||
promptIds.push(promptId);
|
||||
}
|
||||
}
|
||||
|
||||
for (const prompt of SEED_CHAT_ML_PROMPTS) {
|
||||
const promptId = `prompt-${v4()}`;
|
||||
const versions = Math.floor(Math.random() * 20) + 1;
|
||||
for (let i = 1; i <= versions; i++) {
|
||||
const versionAddition = [
|
||||
{
|
||||
role: "user",
|
||||
content: "This is content for version " + i,
|
||||
},
|
||||
];
|
||||
|
||||
await prisma.prompt.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: project.id,
|
||||
name: prompt.name,
|
||||
version: prompt.version,
|
||||
},
|
||||
id: promptId,
|
||||
},
|
||||
create: {
|
||||
id: promptId,
|
||||
projectId: project.id,
|
||||
createdBy: prompt.createdBy,
|
||||
prompt: [...prompt.prompt, ...versionAddition],
|
||||
name: prompt.name,
|
||||
version: i,
|
||||
type: "chat",
|
||||
labels: prompt.labels,
|
||||
tags: prompt.tags,
|
||||
},
|
||||
update: {
|
||||
id: promptId,
|
||||
},
|
||||
});
|
||||
promptIds.push(promptId);
|
||||
}
|
||||
}
|
||||
|
||||
for (const version of SEED_PROMPT_VERSIONS) {
|
||||
const id = `prompt-${v4()}`;
|
||||
await prisma.prompt.upsert({
|
||||
where: {
|
||||
projectId_name_version: {
|
||||
projectId: project.id,
|
||||
name: version.name,
|
||||
version: version.version,
|
||||
},
|
||||
id: id,
|
||||
},
|
||||
create: {
|
||||
id: id,
|
||||
projectId: project.id,
|
||||
createdBy: version.createdBy,
|
||||
prompt: version.prompt,
|
||||
name: version.name,
|
||||
config: version.config,
|
||||
version: version.version,
|
||||
labels: version.labels,
|
||||
},
|
||||
update: {
|
||||
id: id,
|
||||
},
|
||||
});
|
||||
promptIds.push(id);
|
||||
}
|
||||
|
||||
return promptIds;
|
||||
}
|
||||
|
||||
async function generateConfigsForProject(projects: Project[]) {
|
||||
const projectIdsToConfigs: Map<
|
||||
string,
|
||||
{
|
||||
name: string;
|
||||
id: string;
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[]
|
||||
> = new Map();
|
||||
|
||||
await Promise.all(
|
||||
projects.map(async (project) => {
|
||||
const configNameAndId = await generateConfigs(project);
|
||||
projectIdsToConfigs.set(project.id, configNameAndId);
|
||||
}),
|
||||
);
|
||||
return projectIdsToConfigs;
|
||||
}
|
||||
|
||||
async function createTraceSessions(project1: Project, project2: Project) {
|
||||
for (const project of [project1, project2]) {
|
||||
for (let i = 0; i < 100; i++) {
|
||||
await prisma.traceSession.create({
|
||||
data: {
|
||||
projectId: project.id,
|
||||
id: `session_${i}`,
|
||||
createdAt: new Date(),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function generateConfigs(project: Project) {
|
||||
const configNameAndId: {
|
||||
name: string;
|
||||
id: string;
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[] = [];
|
||||
|
||||
const configs = [
|
||||
{
|
||||
id: `config-${v4()}`,
|
||||
name: "manual-score",
|
||||
dataType: ScoreDataType.NUMERIC,
|
||||
projectId: project.id,
|
||||
isArchived: false,
|
||||
},
|
||||
{
|
||||
id: `config-${v4()}`,
|
||||
projectId: project.id,
|
||||
name: "Accuracy",
|
||||
dataType: ScoreDataType.CATEGORICAL,
|
||||
categories: [
|
||||
{ label: "Incorrect", value: 0 },
|
||||
{ label: "Partially Correct", value: 1 },
|
||||
{ label: "Correct", value: 2 },
|
||||
],
|
||||
isArchived: false,
|
||||
},
|
||||
{
|
||||
id: `config-${v4()}`,
|
||||
projectId: project.id,
|
||||
name: "Toxicity",
|
||||
dataType: ScoreDataType.BOOLEAN,
|
||||
categories: [
|
||||
{ label: "True", value: 1 },
|
||||
{ label: "False", value: 0 },
|
||||
],
|
||||
description:
|
||||
"Used to indicate if text was harmful or offensive in nature.",
|
||||
isArchived: false,
|
||||
},
|
||||
];
|
||||
|
||||
for (const config of configs) {
|
||||
await prisma.scoreConfig.upsert({
|
||||
where: {
|
||||
id_projectId: {
|
||||
projectId: config.projectId,
|
||||
id: config.id,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
id: config.id,
|
||||
projectId: config.projectId,
|
||||
name: config.name,
|
||||
dataType: config.dataType,
|
||||
categories: config.categories,
|
||||
isArchived: config.isArchived,
|
||||
},
|
||||
update: {
|
||||
id: config.id,
|
||||
},
|
||||
});
|
||||
configNameAndId.push({
|
||||
name: config.name,
|
||||
id: config.id,
|
||||
dataType: config.dataType,
|
||||
categories: config.categories ?? null,
|
||||
});
|
||||
}
|
||||
|
||||
return configNameAndId;
|
||||
}
|
||||
|
||||
async function generateQueuesForProject(
|
||||
projects: Project[],
|
||||
configIdsAndNames: Map<
|
||||
string,
|
||||
{
|
||||
name: string;
|
||||
id: string;
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[]
|
||||
>,
|
||||
) {
|
||||
const projectIdsToQueues: Map<string, string[]> = new Map();
|
||||
|
||||
await Promise.all(
|
||||
projects.map(async (project) => {
|
||||
const queueIds = await generateQueues(
|
||||
project,
|
||||
configIdsAndNames.get(project.id) ?? [],
|
||||
);
|
||||
projectIdsToQueues.set(project.id, queueIds);
|
||||
}),
|
||||
);
|
||||
return projectIdsToQueues;
|
||||
}
|
||||
|
||||
async function generateQueues(
|
||||
project: Project,
|
||||
configIdsAndNames: {
|
||||
name: string;
|
||||
id: string;
|
||||
dataType: ScoreDataType;
|
||||
categories: ConfigCategory[] | null;
|
||||
}[],
|
||||
) {
|
||||
const queue = {
|
||||
id: `queue-${v4()}`,
|
||||
name: "Default",
|
||||
description: "Default queue",
|
||||
scoreConfigIds: configIdsAndNames.map((config) => config.id),
|
||||
projectId: project.id,
|
||||
};
|
||||
|
||||
await prisma.annotationQueue.upsert({
|
||||
where: {
|
||||
projectId_name: {
|
||||
projectId: queue.projectId,
|
||||
name: queue.name,
|
||||
},
|
||||
},
|
||||
create: {
|
||||
...queue,
|
||||
},
|
||||
update: {
|
||||
id: queue.id,
|
||||
},
|
||||
});
|
||||
|
||||
return [queue.id];
|
||||
}
|
||||
@@ -0,0 +1,160 @@
|
||||
# Langfuse Seeder System
|
||||
|
||||
System for generating test data in ClickHouse and PostgreSQL for Langfuse development and testing.
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
```
|
||||
seeder/
|
||||
├── types.ts # Core interfaces and types
|
||||
├── data-generators.ts # Data generation logic
|
||||
├── clickhouse-builder.ts # ClickHouse query building
|
||||
├── seeder-orchestrator.ts # Main orchestration logic
|
||||
├── postgres-seed-constants.ts # PostgreSQL data constants
|
||||
├── clickhouse-seed-constants.ts # ClickHouse data constants
|
||||
└── seed-helpers.ts # Utility functions
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { SeederOrchestrator } from "./seeder/seeder-orchestrator";
|
||||
|
||||
const orchestrator = new SeederOrchestrator();
|
||||
|
||||
// Full seed (datasets + evaluation + synthetic data)
|
||||
await orchestrator.executeFullSeed(projectIds, {
|
||||
numberOfDays: 30,
|
||||
totalObservations: 10000,
|
||||
numberOfRuns: 3,
|
||||
});
|
||||
|
||||
// Individual data types
|
||||
await orchestrator.createDatasetExperimentData(projectIds, config);
|
||||
await orchestrator.createEvaluationData(projectIds);
|
||||
await orchestrator.createSyntheticData(projectIds, config);
|
||||
```
|
||||
|
||||
## Generated Data
|
||||
|
||||
### 1. Dataset Experiment Data
|
||||
|
||||
- **Purpose**: Realistic experiment traces based on actual datasets
|
||||
- **Environment**: `langfuse-prompt-experiments`
|
||||
- **Structure**: Each dataset item links to a trace with a single generation observation
|
||||
- **ID Pattern**: `trace-dataset-{datasetName}-{itemIndex}-{projectId}-{runNumber}`
|
||||
|
||||
### 2. Evaluation Data
|
||||
|
||||
- **Purpose**: Evaluation metrics and scoring data - to be linked to evaluation logs
|
||||
- **Environment**: `langfuse-evaluation`
|
||||
- **Structure**: Traces with multiple observations and comprehensive scoring
|
||||
- **ID Pattern**: `trace-eval-{index}-{projectId}`
|
||||
|
||||
### 3. Synthetic Data
|
||||
|
||||
- **Purpose**: Large-scale realistic tracing data
|
||||
- **Environment**: `default`
|
||||
- **Structure**: Hierarchical traces with multiple observations and scores
|
||||
- **ID Pattern**: `trace-synthetic-{index}-{projectId}`
|
||||
|
||||
## Abstraction Architecture
|
||||
|
||||
### DataGenerator
|
||||
|
||||
Generates realistic data for all three types. If you need to change any clickhouse data, you should modify this class. Key methods:
|
||||
|
||||
- `generateDatasetTrace()` - Creates traces from dataset items
|
||||
- `generateSyntheticTraces()` - Creates realistic synthetic traces
|
||||
- `generateEvaluationTraces()` - Creates evaluation-focused traces
|
||||
|
||||
### ClickHouseQueryBuilder
|
||||
|
||||
Builds optimized ClickHouse insert queries. No need to edit this file. Handles proper escaping and type handling.
|
||||
|
||||
### SeederOrchestrator
|
||||
|
||||
Main coordination class that:
|
||||
|
||||
- Loads file content for realistic inputs/outputs
|
||||
- Coordinates data generation and insertion
|
||||
- Handles batching and error recovery
|
||||
- Provides logging and statistics
|
||||
|
||||
## Making Changes
|
||||
|
||||
### Configuration Options
|
||||
|
||||
```typescript
|
||||
interface SeederConfig {
|
||||
numberOfDays: number; // How far back to generate timestamps
|
||||
numberOfRuns?: number; // How many experiment runs per dataset
|
||||
totalObservations?: number; // Total observations for synthetic data
|
||||
}
|
||||
```
|
||||
|
||||
### Extending the System
|
||||
|
||||
#### Adding New Data Types
|
||||
|
||||
1. Add interface to `types.ts`
|
||||
2. Add generator method to `DataGenerator`
|
||||
3. Add query builder method to `ClickHouseQueryBuilder`
|
||||
4. Add orchestration method to `SeederOrchestrator`
|
||||
5. Update interdependency documentation
|
||||
|
||||
#### Adding New File Sources
|
||||
|
||||
1. Add file path to `SeederOrchestrator.loadFileContent()`
|
||||
2. Add processing logic to `DataGenerator`
|
||||
3. Update `FileContent` interface if needed
|
||||
|
||||
#### Changing Data Distribution
|
||||
|
||||
1. Modify generator methods in `DataGenerator`
|
||||
2. Update constants in `clickhouse-seed-constants.ts`
|
||||
3. Test with small datasets first
|
||||
|
||||
#### Changing ID Generation
|
||||
|
||||
1. **Check**: All places that query ClickHouse by ID
|
||||
2. **Check**: PostgreSQL foreign key references
|
||||
3. **Check**: Dataset run item and evaluation trace creation logic
|
||||
4. **Action**: Update `seed-helpers.ts` functions consistently
|
||||
|
||||
#### Changing Environment Names
|
||||
|
||||
1. **Check**: All ClickHouse queries that filter by environment
|
||||
2. **Check**: PostgreSQL dataset and prompt environment fields
|
||||
3. **Check**: UI environment filtering logic
|
||||
4. **Action**: Update constants in both systems
|
||||
|
||||
#### Changing Data Structure
|
||||
|
||||
1. **Check**: ClickHouse table schema compatibility
|
||||
2. **Check**: PostgreSQL table relationships
|
||||
3. **Check**: API response serialization
|
||||
4. **Action**: Update both schemas before changing data generation
|
||||
|
||||
#### Adding New Data Types
|
||||
|
||||
1. **Check**: Whether PostgreSQL needs corresponding tables
|
||||
2. **Check**: Whether new foreign key relationships are needed
|
||||
3. **Check**: Whether UI needs to handle new data types
|
||||
4. **Action**: Plan database migrations carefully
|
||||
|
||||
## File Dependencies
|
||||
|
||||
### Required Files
|
||||
|
||||
```
|
||||
packages/shared/clickhouse/
|
||||
├── nested_json.json # Large JSON for realistic inputs
|
||||
├── markdown.txt # Markdown content for document analysis
|
||||
└── chat_ml_json.json # Chat ML format examples
|
||||
```
|
||||
|
||||
### Constants Files
|
||||
|
||||
- `postgres-seed-constants.ts` - Datasets, prompts, and PostgreSQL data
|
||||
- `clickhouse-seed-constants.ts` - ClickHouse-specific constants (models, names)
|
||||
@@ -0,0 +1,135 @@
|
||||
{
|
||||
"messages": [
|
||||
{
|
||||
"content": "what is the weather in sf",
|
||||
"additional_kwargs": {},
|
||||
"response_metadata": {},
|
||||
"type": "human",
|
||||
"name": null,
|
||||
"id": "2166b887-b9fb-4282-a9f7-828d4a6cca54",
|
||||
"example": false
|
||||
},
|
||||
{
|
||||
"content": "",
|
||||
"additional_kwargs": {
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_pNLR4DoZiptb5xGlb299QsoN",
|
||||
"function": {
|
||||
"arguments": "{\"query\":\"current weather in San Francisco\"}",
|
||||
"name": "search"
|
||||
},
|
||||
"type": "function"
|
||||
}
|
||||
],
|
||||
"refusal": null
|
||||
},
|
||||
"response_metadata": {
|
||||
"token_usage": {
|
||||
"completion_tokens": 17,
|
||||
"prompt_tokens": 48,
|
||||
"total_tokens": 65,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
}
|
||||
},
|
||||
"model_name": "gpt-4o-mini-2024-07-18",
|
||||
"system_fingerprint": "fp_34a54ae93c",
|
||||
"finish_reason": "tool_calls",
|
||||
"logprobs": null
|
||||
},
|
||||
"type": "ai",
|
||||
"name": null,
|
||||
"id": "run-14c4801c-ae47-4c4a-8d49-29ba08c5679d-0",
|
||||
"example": false,
|
||||
"tool_calls": [
|
||||
{
|
||||
"name": "search",
|
||||
"args": {
|
||||
"query": "current weather in San Francisco"
|
||||
},
|
||||
"id": "call_pNLR4DoZiptb5xGlb299QsoN",
|
||||
"type": "tool_call"
|
||||
}
|
||||
],
|
||||
"invalid_tool_calls": [],
|
||||
"usage_metadata": {
|
||||
"input_tokens": 48,
|
||||
"output_tokens": 17,
|
||||
"total_tokens": 65,
|
||||
"input_token_details": {
|
||||
"audio": 0,
|
||||
"cache_read": 0
|
||||
},
|
||||
"output_token_details": {
|
||||
"audio": 0,
|
||||
"reasoning": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"content": "It's 60 degrees and foggy.",
|
||||
"additional_kwargs": {},
|
||||
"response_metadata": {},
|
||||
"type": "tool",
|
||||
"name": "search",
|
||||
"id": "220ba511-6816-4d1a-8acf-8e5791c2d88e",
|
||||
"tool_call_id": "call_pNLR4DoZiptb5xGlb299QsoN",
|
||||
"artifact": null,
|
||||
"status": "success"
|
||||
},
|
||||
{
|
||||
"content": "The current weather in San Francisco is 60 degrees and foggy.",
|
||||
"additional_kwargs": {
|
||||
"refusal": null
|
||||
},
|
||||
"response_metadata": {
|
||||
"token_usage": {
|
||||
"completion_tokens": 15,
|
||||
"prompt_tokens": 80,
|
||||
"total_tokens": 95,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
}
|
||||
},
|
||||
"model_name": "gpt-4o-mini-2024-07-18",
|
||||
"system_fingerprint": "fp_34a54ae93c",
|
||||
"finish_reason": "stop",
|
||||
"logprobs": null
|
||||
},
|
||||
"type": "ai",
|
||||
"name": null,
|
||||
"id": "run-f051be3b-2656-4b19-8704-a2e492306131-0",
|
||||
"example": false,
|
||||
"tool_calls": [],
|
||||
"invalid_tool_calls": [],
|
||||
"usage_metadata": {
|
||||
"input_tokens": 80,
|
||||
"output_tokens": 15,
|
||||
"total_tokens": 95,
|
||||
"input_token_details": {
|
||||
"audio": 0,
|
||||
"cache_read": 0
|
||||
},
|
||||
"output_token_details": {
|
||||
"audio": 0,
|
||||
"reasoning": 0
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,254 @@
|
||||
import {
|
||||
TraceRecordInsertType,
|
||||
ObservationRecordInsertType,
|
||||
ScoreRecordInsertType,
|
||||
DatasetRunItemRecordInsertType,
|
||||
createDatasetRunItemsCh,
|
||||
} from "../../../src/server";
|
||||
import { SEED_TEXT_PROMPTS } from "./postgres-seed-constants";
|
||||
import {
|
||||
createTracesCh,
|
||||
createObservationsCh,
|
||||
createScoresCh,
|
||||
} from "../../../src/server";
|
||||
import { InsertResult } from "@clickhouse/client";
|
||||
|
||||
/**
|
||||
* Builds or executes ClickHouse SQL INSERT queries for seeding test data.
|
||||
*
|
||||
* Use executeXxxInsert() for custom curated data with detailed control.
|
||||
* Use buildBulkXxxInsert() for large datasets (>1000 items) for random distribution of data.
|
||||
*/
|
||||
export class ClickHouseQueryBuilder {
|
||||
private escapeString(str: string): string {
|
||||
return str.replace(/'/g, "''");
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates INSERT query for trace data using VALUES syntax.
|
||||
* Use for: Small datasets, detailed trace objects with all fields populated.
|
||||
*/
|
||||
async executeTracesInsert(
|
||||
traces: TraceRecordInsertType[],
|
||||
): Promise<InsertResult> {
|
||||
return await createTracesCh(traces);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates INSERT query for observation data using VALUES syntax.
|
||||
* Use for: Small datasets, observations that link to postgres data (e.g. dataset runs)
|
||||
*/
|
||||
async executeObservationsInsert(
|
||||
observations: ObservationRecordInsertType[],
|
||||
): Promise<InsertResult> {
|
||||
return await createObservationsCh(observations);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates INSERT query for dataset run items data using VALUES syntax.
|
||||
* Use for: Small datasets, dataset run items that link to postgres data (e.g. dataset runs)
|
||||
*/
|
||||
async executeDatasetRunItemsInsert(
|
||||
datasetRunItems: DatasetRunItemRecordInsertType[],
|
||||
): Promise<InsertResult> {
|
||||
return await createDatasetRunItemsCh(datasetRunItems);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates INSERT query for score data using VALUES syntax.
|
||||
* Use for: Small datasets, scores with custom values and metadata.
|
||||
*/
|
||||
async executeScoresInsert(
|
||||
scores: ScoreRecordInsertType[],
|
||||
): Promise<InsertResult> {
|
||||
return await createScoresCh(scores);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates INSERT using ClickHouse numbers() function.
|
||||
* Use for: Large datasets (>1000 traces), realistic timestamps, bulk generation.
|
||||
*/
|
||||
buildBulkTracesInsert(
|
||||
projectId: string,
|
||||
count: number,
|
||||
environment: string = "default",
|
||||
fileContent?: { heavyMarkdown: string; nestedJson: any; chatMlJson: any },
|
||||
opts: { numberOfDays: number } = { numberOfDays: 1 },
|
||||
): string {
|
||||
// Escape file content if provided
|
||||
const escapedHeavyMarkdown = fileContent
|
||||
? this.escapeString(fileContent.heavyMarkdown)
|
||||
: "Sample heavy markdown content";
|
||||
const escapedNestedJson = fileContent
|
||||
? this.escapeString(JSON.stringify(fileContent.nestedJson))
|
||||
: '{"sample": "nested json"}';
|
||||
const escapedChatMl = fileContent
|
||||
? this.escapeString(JSON.stringify(fileContent.chatMlJson))
|
||||
: '{"messages": []}';
|
||||
|
||||
return `
|
||||
INSERT INTO traces
|
||||
SELECT
|
||||
concat('trace-bulk-', toString(number), '-${projectId.slice(-8)}') AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
concat('trace-', toString(number % 10)) AS name,
|
||||
if(randUniform(0, 1) < 0.3, concat('user_', toString(rand() % 1000)), NULL) AS user_id,
|
||||
map('generated', 'bulk') AS metadata,
|
||||
NULL AS release,
|
||||
NULL AS version,
|
||||
'${projectId}' AS project_id,
|
||||
'${environment}' AS environment,
|
||||
if(rand() < 0.8, true, false) AS public,
|
||||
if(rand() < 0.1, true, false) AS bookmarked,
|
||||
array() AS tags,
|
||||
if(randUniform(0, 1) < 0.3, '${escapedHeavyMarkdown}',
|
||||
'${escapedChatMl}'
|
||||
) AS input,
|
||||
if(randUniform(0, 1) < 0.2, '${escapedNestedJson}',
|
||||
'${escapedChatMl}'
|
||||
) AS output,
|
||||
if(randUniform(0, 1) < 0.3, concat('session_', toString(rand() % 100)), NULL) AS session_id,
|
||||
now() AS created_at,
|
||||
now() AS updated_at,
|
||||
now() AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${count});
|
||||
`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates observations with automatic prompt linking (10% rate).
|
||||
* Use for: Large datasets, hierarchical observations, cost/latency variation.
|
||||
*/
|
||||
buildBulkObservationsInsert(
|
||||
projectId: string,
|
||||
tracesCount: number,
|
||||
observationsPerTrace: number = 5,
|
||||
environment: string = "default",
|
||||
fileContent?: { heavyMarkdown: string; nestedJson: any; chatMlJson: any },
|
||||
opts: { numberOfDays: number } = { numberOfDays: 1 },
|
||||
): string {
|
||||
const totalObservations = tracesCount * observationsPerTrace;
|
||||
|
||||
// Escape file content if provided
|
||||
const escapedHeavyMarkdown = fileContent
|
||||
? this.escapeString(fileContent.heavyMarkdown)
|
||||
: "Sample heavy markdown content";
|
||||
const escapedNestedJson = fileContent
|
||||
? this.escapeString(JSON.stringify(fileContent.nestedJson))
|
||||
: '{"sample": "nested json"}';
|
||||
const escapedChatMl = fileContent
|
||||
? this.escapeString(JSON.stringify(fileContent.chatMlJson))
|
||||
: '{"messages": []}';
|
||||
|
||||
return `
|
||||
INSERT INTO observations
|
||||
SELECT
|
||||
concat('obs-bulk-', toString(number), '-${projectId.slice(-8)}') AS id,
|
||||
concat('trace-bulk-', toString(number % ${tracesCount}), '-${projectId.slice(-8)}') AS trace_id,
|
||||
'${projectId}' AS project_id,
|
||||
'${environment}' AS environment,
|
||||
if(randUniform(0, 1) < 0.47, 'GENERATION', if(randUniform(0, 1) < 0.94, 'SPAN', 'EVENT')) AS type,
|
||||
if(number % 6 = 0, NULL, toString(number - 1)) AS parent_observation_id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS start_time,
|
||||
addMilliseconds(start_time,
|
||||
case
|
||||
when type = 'GENERATION' then floor(randUniform(5, 30))
|
||||
when type = 'SPAN' then floor(randUniform(1, 50))
|
||||
else floor(randUniform(1, 10))
|
||||
end) AS end_time,
|
||||
case
|
||||
when type = 'GENERATION' then concat('generation-', toString(number % 10))
|
||||
when type = 'SPAN' then concat('span-', toString(number % 10))
|
||||
else concat('event-', toString(number % 10))
|
||||
end AS name,
|
||||
map('key', 'value') AS metadata,
|
||||
if(randUniform(0, 1) < 0.85, 'DEFAULT', if(randUniform(0, 1) < 0.7, 'DEBUG', if(randUniform(0, 1) < 0.3, 'ERROR', 'WARNING'))) AS level,
|
||||
NULL AS status_message,
|
||||
NULL AS version,
|
||||
if(type = 'GENERATION',
|
||||
if(randUniform(0, 1) < 0.4, '${escapedHeavyMarkdown}', '${escapedChatMl}'),
|
||||
NULL) AS input,
|
||||
if(type = 'GENERATION',
|
||||
if(randUniform(0, 1) < 0.3, '${escapedNestedJson}', '${escapedChatMl}'),
|
||||
NULL) AS output,
|
||||
if(type = 'GENERATION', 'gpt-4', NULL) AS provided_model_name,
|
||||
if(type = 'GENERATION', concat('model_', toString(rand() % 1000)), NULL) AS internal_model_id,
|
||||
if(type = 'GENERATION', '{"temperature": 0.7}', '{}') AS model_parameters,
|
||||
if(type = 'GENERATION', map('input', toUInt64(randUniform(20, 200)), 'output', toUInt64(randUniform(10, 100)), 'total', toUInt64(randUniform(30, 300))), map()) AS provided_usage_details,
|
||||
if(type = 'GENERATION', map('input', toUInt64(randUniform(20, 200)), 'output', toUInt64(randUniform(10, 100)), 'total', toUInt64(randUniform(30, 300))), map()) AS usage_details,
|
||||
if(type = 'GENERATION', map('input', toDecimal64(randUniform(0.00001, 0.001), 8), 'output', toDecimal64(randUniform(0.00001, 0.002), 8), 'total', toDecimal64(randUniform(0.00002, 0.003), 8)), map()) AS provided_cost_details,
|
||||
if(type = 'GENERATION', map('input', toDecimal64(randUniform(0.00001, 0.001), 8), 'output', toDecimal64(randUniform(0.00001, 0.002), 8), 'total', toDecimal64(randUniform(0.00002, 0.003), 8)), map()) AS cost_details,
|
||||
if(type = 'GENERATION', toDecimal64(randUniform(0.00002, 0.003), 8), NULL) AS total_cost,
|
||||
if(type = 'GENERATION', addMilliseconds(start_time, floor(randUniform(100, 500))), NULL) AS completion_start_time,
|
||||
if("type" = 'GENERATION' AND number % 10 = 0,
|
||||
arrayElement(['${SEED_TEXT_PROMPTS.map((p) => p.id).join("','")}'], 1 + (number % ${SEED_TEXT_PROMPTS.length})),
|
||||
NULL) AS prompt_id,
|
||||
if("type" = 'GENERATION' AND number % 10 = 0,
|
||||
arrayElement(['${SEED_TEXT_PROMPTS.map((p) => p.name).join("','")}'], 1 + (number % ${SEED_TEXT_PROMPTS.length})),
|
||||
NULL) AS prompt_name,
|
||||
if("type" = 'GENERATION' AND number % 10 = 0,
|
||||
arrayElement(['${SEED_TEXT_PROMPTS.map((p) => p.version).join("','")}'], 1 + (number % ${SEED_TEXT_PROMPTS.length})),
|
||||
NULL) AS prompt_version,
|
||||
start_time AS created_at,
|
||||
start_time AS updated_at,
|
||||
start_time AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${totalObservations});
|
||||
`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates scores with mixed data types (NUMERIC/BOOLEAN/CATEGORICAL).
|
||||
* Use for: Large datasets, varied score distributions, synthetic metrics.
|
||||
*/
|
||||
buildBulkScoresInsert(
|
||||
projectId: string,
|
||||
tracesCount: number,
|
||||
scoresPerTrace: number = 2,
|
||||
environment: string = "default",
|
||||
opts: { numberOfDays: number } = { numberOfDays: 1 },
|
||||
): string {
|
||||
const totalScores = tracesCount * scoresPerTrace;
|
||||
|
||||
return `
|
||||
INSERT INTO scores
|
||||
SELECT
|
||||
concat('score-bulk-', toString(number), '-${projectId.slice(-8)}') AS id,
|
||||
toDateTime(now() - randUniform(0, ${opts.numberOfDays} * 24 * 60 * 60)) AS timestamp,
|
||||
'${projectId}' AS project_id,
|
||||
'${environment}' AS environment,
|
||||
concat('trace-bulk-', toString(number % ${tracesCount}), '-${projectId.slice(-8)}') AS trace_id,
|
||||
if(randUniform(0, 1) < 0.3, concat('session_', toString(rand() % 100)), NULL) AS session_id,
|
||||
NULL AS dataset_run_id,
|
||||
if(randUniform(0, 1) < 0.1, concat('obs-bulk-', toString(rand() % (${tracesCount} * 5)), '-${projectId.slice(-8)}'), NULL) AS observation_id,
|
||||
concat('metric_', toString((number % ${scoresPerTrace * 5}) + 1)) AS name,
|
||||
case
|
||||
when (number % 3) = 0 then toDecimal64(randUniform(0, 100), 8)
|
||||
when (number % 3) = 1 then if(randUniform(0, 1) < 0.5, 1, 0)
|
||||
else NULL
|
||||
end AS value,
|
||||
'API' AS source,
|
||||
'Generated synthetic score' AS comment,
|
||||
map() AS metadata,
|
||||
NULL AS author_user_id,
|
||||
NULL AS config_id,
|
||||
case
|
||||
when (number % 3) = 0 then 'NUMERIC'
|
||||
when (number % 3) = 1 then 'BOOLEAN'
|
||||
else 'CATEGORICAL'
|
||||
end AS data_type,
|
||||
case
|
||||
when (number % 3) = 1 then if(value = 1, 'true', 'false')
|
||||
when (number % 3) = 2 then concat('category_', toString((rand() % 5) + 1))
|
||||
else NULL
|
||||
end AS string_value,
|
||||
NULL AS queue_id,
|
||||
timestamp AS created_at,
|
||||
timestamp AS updated_at,
|
||||
timestamp AS event_ts,
|
||||
0 AS is_deleted
|
||||
FROM numbers(${totalScores});
|
||||
`;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
export const REALISTIC_TRACE_NAMES = [
|
||||
"LangGraph",
|
||||
"ChatCompletion",
|
||||
"DocumentAnalysis",
|
||||
"CodeGeneration",
|
||||
"DataProcessing",
|
||||
"QueryExecution",
|
||||
"ModelInference",
|
||||
"WebScraping",
|
||||
"TextSummarization",
|
||||
"ImageClassification",
|
||||
];
|
||||
|
||||
export const REALISTIC_SPAN_NAMES = [
|
||||
"agent",
|
||||
"tools",
|
||||
"search",
|
||||
"retrieval",
|
||||
"preprocessing",
|
||||
"validation",
|
||||
"transformation",
|
||||
"classification",
|
||||
"extraction",
|
||||
"postprocessing",
|
||||
];
|
||||
|
||||
export const REALISTIC_GENERATION_NAMES = [
|
||||
"ChatOpenAI",
|
||||
"GPT-4",
|
||||
"Claude-3",
|
||||
"Gemini",
|
||||
"Llama-2",
|
||||
"PaLM",
|
||||
"CodeLlama",
|
||||
"Mistral",
|
||||
"Falcon",
|
||||
"Vicuna",
|
||||
];
|
||||
|
||||
export const REALISTIC_MODELS = [
|
||||
"gpt-4o-mini-2024-07-18",
|
||||
"gpt-4-turbo-2024-04-09",
|
||||
"claude-3-haiku-20240307",
|
||||
"claude-3-sonnet-20240229",
|
||||
"claude-3-opus-20240229",
|
||||
"gemini-pro",
|
||||
"llama-2-70b-chat",
|
||||
"mistral-7b-instruct",
|
||||
"codellama-34b-instruct",
|
||||
];
|
||||
|
||||
export const REALISTIC_USER_INPUTS = [
|
||||
"What is the weather in San Francisco?",
|
||||
"Summarize this document for me",
|
||||
"Generate a Python function to sort a list",
|
||||
"Explain quantum computing in simple terms",
|
||||
"Analyze this data and provide insights",
|
||||
"Translate this text to Spanish",
|
||||
"Create a marketing email for our product",
|
||||
"Debug this code and fix the errors",
|
||||
"What are the latest trends in AI?",
|
||||
"Help me plan a trip to Europe",
|
||||
];
|
||||
|
||||
export const REALISTIC_AI_RESPONSES = [
|
||||
"The current weather in San Francisco is 60 degrees and foggy.",
|
||||
"Here's a summary of the key points from the document...",
|
||||
"Here's a Python function that sorts a list efficiently...",
|
||||
"Quantum computing uses quantum mechanics principles...",
|
||||
"Based on the data analysis, I found the following insights...",
|
||||
"Here's the Spanish translation of your text...",
|
||||
"I've created a compelling marketing email for your product...",
|
||||
"I found several issues in your code and here are the fixes...",
|
||||
"The latest AI trends include large language models...",
|
||||
"Here's a detailed 10-day European itinerary for you...",
|
||||
];
|
||||
|
||||
export const REALISTIC_METADATA_EXAMPLES = [
|
||||
{ thread_id: 42, session_type: "interactive" },
|
||||
{ user_id: "user_123", conversation_id: "conv_456" },
|
||||
{ model_version: "v2.1", temperature: 0.7 },
|
||||
{ request_id: "req_789", timestamp: "2024-05-23T15:42:11.996Z" },
|
||||
{ environment: "production", region: "us-west-2" },
|
||||
{
|
||||
document_type: "state_of_union",
|
||||
file_size: "142KB",
|
||||
processing_time: "2.3s",
|
||||
},
|
||||
{ data_source: "product_catalog", record_count: 30, format: "nested_json" },
|
||||
{ analysis_type: "sentiment", language: "en", confidence: 0.92 },
|
||||
{
|
||||
extraction_task: "policy_analysis",
|
||||
domain: "politics",
|
||||
entities_found: 15,
|
||||
},
|
||||
{ file_type: "JSON", validation: "passed", schema_version: "v1.2" },
|
||||
];
|
||||
@@ -0,0 +1,641 @@
|
||||
import { FileContent, DatasetItemInput } from "./types";
|
||||
import {
|
||||
REALISTIC_TRACE_NAMES,
|
||||
REALISTIC_SPAN_NAMES,
|
||||
REALISTIC_GENERATION_NAMES,
|
||||
REALISTIC_MODELS,
|
||||
} from "./clickhouse-seed-constants";
|
||||
import {
|
||||
generateDatasetItemId,
|
||||
generateDatasetRunItemId,
|
||||
generateDatasetRunTraceId,
|
||||
generateEvalObservationId,
|
||||
generateEvalScoreId,
|
||||
generateEvalTraceId,
|
||||
} from "./seed-helpers";
|
||||
import { v4 as uuidv4 } from "uuid";
|
||||
import {
|
||||
FAILED_EVAL_TRACE_INTERVAL,
|
||||
SEED_EVALUATOR_CONFIGS,
|
||||
} from "./postgres-seed-constants";
|
||||
import {
|
||||
createTrace,
|
||||
createObservation,
|
||||
createTraceScore,
|
||||
ObservationRecordInsertType,
|
||||
ScoreRecordInsertType,
|
||||
TraceRecordInsertType,
|
||||
DatasetRunItemRecordInsertType,
|
||||
createDatasetRunItem,
|
||||
} from "../../../src/server";
|
||||
|
||||
/**
|
||||
* Generates realistic test data for traces, observations, and scores.
|
||||
*
|
||||
* Use generateXxxTraces() for creating different data types:
|
||||
* - generateDatasetTrace(): For dataset experiment runs (langfuse-prompt-experiments env)
|
||||
* - generateEvaluationTraces(): For evaluation data (langfuse-evaluation env)
|
||||
* - generateSyntheticTraces(): For large-scale synthetic data (default env)
|
||||
*/
|
||||
export class DataGenerator {
|
||||
private static instance: DataGenerator;
|
||||
private fileContent: FileContent | null = null;
|
||||
|
||||
static getInstance(): DataGenerator {
|
||||
if (!DataGenerator.instance) {
|
||||
DataGenerator.instance = new DataGenerator();
|
||||
}
|
||||
return DataGenerator.instance;
|
||||
}
|
||||
|
||||
setFileContent(content: FileContent) {
|
||||
this.fileContent = content;
|
||||
}
|
||||
|
||||
private randomElement<T>(array: T[]): T {
|
||||
return array[Math.floor(Math.random() * array.length)];
|
||||
}
|
||||
|
||||
private randomBoolean(probability: number = 0.5): boolean {
|
||||
return Math.random() < probability;
|
||||
}
|
||||
|
||||
private randomInt(min: number, max: number): number {
|
||||
return Math.floor(Math.random() * (max - min + 1)) + min;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates dataset run items for dataset runs.
|
||||
* Use for: Dataset experiment scenarios.
|
||||
*/
|
||||
generateDatasetRunItem(
|
||||
input: DatasetItemInput & { runCreatedAt: number },
|
||||
projectId: string,
|
||||
): DatasetRunItemRecordInsertType {
|
||||
const datasetRunItemId = generateDatasetRunItemId(
|
||||
input.datasetName,
|
||||
input.itemIndex,
|
||||
projectId,
|
||||
input.runNumber || 0,
|
||||
);
|
||||
|
||||
// TODO: there are too many dataset run items in the postgres database?
|
||||
return createDatasetRunItem({
|
||||
id: datasetRunItemId,
|
||||
project_id: projectId,
|
||||
trace_id: generateDatasetRunTraceId(
|
||||
input.datasetName,
|
||||
input.itemIndex,
|
||||
projectId,
|
||||
input.runNumber || 0,
|
||||
),
|
||||
dataset_id: `${input.datasetName}-${projectId.slice(-8)}`,
|
||||
dataset_run_id: `demo-dataset-run-${input.runNumber}-${projectId.slice(-8)}`,
|
||||
dataset_run_name: `demo-dataset-run-${input.runNumber}-${projectId.slice(-8)}`,
|
||||
dataset_run_created_at: input.runCreatedAt,
|
||||
dataset_run_description:
|
||||
(input.runNumber || 0) % 2 === 0 ? "Dataset run description" : "",
|
||||
dataset_run_metadata: { key: "value" },
|
||||
dataset_item_id: generateDatasetItemId(
|
||||
input.datasetName,
|
||||
input.itemIndex,
|
||||
projectId,
|
||||
input.runNumber || 0,
|
||||
),
|
||||
dataset_item_input: input.item.input,
|
||||
dataset_item_expected_output: input.item.output,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates traces from dataset items for experiment runs.
|
||||
* Use for: Dataset experiments scenarios.
|
||||
*/
|
||||
generateDatasetTrace(
|
||||
input: DatasetItemInput,
|
||||
projectId: string,
|
||||
): TraceRecordInsertType {
|
||||
const traceId = generateDatasetRunTraceId(
|
||||
input.datasetName,
|
||||
input.itemIndex,
|
||||
projectId,
|
||||
input.runNumber || 0,
|
||||
);
|
||||
|
||||
let traceInput: string;
|
||||
let traceOutput: string;
|
||||
|
||||
// Transform dataset item based on type
|
||||
if (input.datasetName === "demo-countries-dataset") {
|
||||
const data = input.item as { input: { country: string }; output: string };
|
||||
traceInput = `What is the capital of ${data.input.country}?`;
|
||||
traceOutput = `The capital of ${data.input.country} is ${data.output}.`;
|
||||
} else if (input.datasetName === "demo-english-transcription-dataset") {
|
||||
const data = input.item as { input: { word: string }; output: string };
|
||||
traceInput = `What is the IPA transcription of the word "${data.input.word}"?`;
|
||||
traceOutput = `The IPA transcription of "${data.input.word}" is ${data.output}.`;
|
||||
} else {
|
||||
traceInput = JSON.stringify(input.item.input);
|
||||
traceOutput = JSON.stringify(input.item.output);
|
||||
}
|
||||
|
||||
return createTrace({
|
||||
id: traceId,
|
||||
project_id: projectId,
|
||||
name: `dataset-run-item-${uuidv4()}`,
|
||||
input: traceInput,
|
||||
output: traceOutput,
|
||||
environment: "langfuse-prompt-experiments",
|
||||
metadata: { experimentType: "langfuse-prompt-experiments" },
|
||||
public: false,
|
||||
bookmarked: false,
|
||||
session_id: null,
|
||||
tags: [],
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates observations for dataset experiment traces with variable costs/latency.
|
||||
* Use for: Dataset experiments requiring detailed observation tracking.
|
||||
*/
|
||||
generateDatasetObservation(
|
||||
trace: TraceRecordInsertType,
|
||||
input: DatasetItemInput,
|
||||
projectId: string,
|
||||
): ObservationRecordInsertType {
|
||||
const observationId = `observation-dataset-${input.datasetName}-${input.itemIndex}-${input.runNumber}-${projectId.slice(-8)}`;
|
||||
|
||||
// Generate variable usage and cost for each observation
|
||||
const inputTokens = this.randomInt(30, 150);
|
||||
const outputTokens = this.randomInt(10, 80);
|
||||
const totalTokens = inputTokens + outputTokens;
|
||||
|
||||
// Cost should be fraction of cents (0.0001-0.01 range)
|
||||
const inputCost = (inputTokens * this.randomInt(1, 5)) / 1000000; // $0.000001-0.000005 per token
|
||||
const outputCost = (outputTokens * this.randomInt(2, 10)) / 1000000; // $0.000002-0.00001 per token
|
||||
const totalCost = inputCost + outputCost;
|
||||
|
||||
return createObservation({
|
||||
id: observationId,
|
||||
trace_id: trace.id,
|
||||
project_id: projectId,
|
||||
type: "GENERATION",
|
||||
name: `dataset-generation-${input.itemIndex}-run-${input.runNumber}`,
|
||||
input: trace.input,
|
||||
output: trace.output,
|
||||
provided_model_name: "gpt-3.5-turbo",
|
||||
model_parameters: JSON.stringify({ temperature: 0.7 }),
|
||||
usage_details: {
|
||||
input: inputTokens,
|
||||
output: outputTokens,
|
||||
total: totalTokens,
|
||||
},
|
||||
provided_usage_details: {
|
||||
input: inputTokens,
|
||||
output: outputTokens,
|
||||
total: totalTokens,
|
||||
},
|
||||
cost_details: {
|
||||
input: Math.round(inputCost * 100000) / 100000, // Round to 5 decimal places
|
||||
output: Math.round(outputCost * 100000) / 100000,
|
||||
total: Math.round(totalCost * 100000) / 100000,
|
||||
},
|
||||
provided_cost_details: {
|
||||
input: Math.round(inputCost * 100000) / 100000,
|
||||
output: Math.round(outputCost * 100000) / 100000,
|
||||
total: Math.round(totalCost * 100000) / 100000,
|
||||
},
|
||||
total_cost: Math.round(totalCost * 100000) / 100000,
|
||||
environment: "langfuse-prompt-experiments",
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates large-scale synthetic traces for performance testing.
|
||||
* Use for: Load testing, dashboard demos, realistic usage simulation.
|
||||
*/
|
||||
generateSyntheticTraces(
|
||||
projectId: string,
|
||||
count: number,
|
||||
): TraceRecordInsertType[] {
|
||||
const traces: TraceRecordInsertType[] = [];
|
||||
|
||||
for (let i = 0; i < count; i++) {
|
||||
const trace = createTrace({
|
||||
id: `trace-synthetic-${i}-${projectId.slice(-8)}`,
|
||||
project_id: projectId,
|
||||
name: this.randomElement(REALISTIC_TRACE_NAMES),
|
||||
input: this.generateTraceInput(),
|
||||
output: this.generateTraceOutput(),
|
||||
user_id: this.randomBoolean(0.3)
|
||||
? `user_${this.randomInt(1, 1000)}`
|
||||
: null,
|
||||
session_id: this.randomBoolean(0.3)
|
||||
? `session_${this.randomInt(1, 100)}`
|
||||
: undefined,
|
||||
environment: "default",
|
||||
metadata: { generated: "synthetic" },
|
||||
tags: this.randomBoolean(0.3) ? ["production", "ai-agent"] : [],
|
||||
public: this.randomBoolean(0.8),
|
||||
bookmarked: this.randomBoolean(0.1),
|
||||
release: this.randomBoolean(0.4)
|
||||
? `v${this.randomInt(1, 5)}.${this.randomInt(0, 10)}`
|
||||
: undefined,
|
||||
version: this.randomBoolean(0.4)
|
||||
? `v${this.randomInt(1, 3)}.${this.randomInt(0, 20)}`
|
||||
: undefined,
|
||||
});
|
||||
|
||||
traces.push(trace);
|
||||
}
|
||||
|
||||
return traces;
|
||||
}
|
||||
|
||||
generateEvaluationObservations(
|
||||
traces: TraceRecordInsertType[],
|
||||
observationsPerTrace: number = 5,
|
||||
projectId: string,
|
||||
): ObservationRecordInsertType[] {
|
||||
const observations: ObservationRecordInsertType[] = [];
|
||||
|
||||
for (const evalJobConfiguration of SEED_EVALUATOR_CONFIGS) {
|
||||
traces.forEach((trace, traceIndex) => {
|
||||
for (let i = 0; i < observationsPerTrace; i++) {
|
||||
const obsType = this.randomBoolean(0.47)
|
||||
? "GENERATION"
|
||||
: this.randomBoolean(0.94)
|
||||
? "SPAN"
|
||||
: "EVENT";
|
||||
|
||||
const observation: ObservationRecordInsertType = createObservation({
|
||||
id: generateEvalObservationId(
|
||||
evalJobConfiguration.evalTemplateId,
|
||||
traceIndex,
|
||||
projectId,
|
||||
),
|
||||
trace_id: trace.id,
|
||||
project_id: projectId,
|
||||
parent_observation_id: undefined,
|
||||
type: obsType,
|
||||
name:
|
||||
obsType === "GENERATION"
|
||||
? this.randomElement(REALISTIC_GENERATION_NAMES)
|
||||
: obsType === "SPAN"
|
||||
? this.randomElement(REALISTIC_SPAN_NAMES)
|
||||
: `event_${i % 10}`,
|
||||
level: this.randomBoolean(0.85)
|
||||
? "DEFAULT"
|
||||
: this.randomBoolean(0.7)
|
||||
? "DEBUG"
|
||||
: this.randomBoolean(0.3)
|
||||
? "ERROR"
|
||||
: "WARNING",
|
||||
input:
|
||||
obsType === "GENERATION"
|
||||
? this.randomBoolean(0.4)
|
||||
? this.fileContent?.heavyMarkdown || "Sample input"
|
||||
: JSON.stringify(this.fileContent?.chatMlJson || {})
|
||||
: undefined,
|
||||
output:
|
||||
obsType === "GENERATION"
|
||||
? this.randomBoolean(0.3)
|
||||
? JSON.stringify(this.fileContent?.nestedJson || {})
|
||||
: JSON.stringify(this.fileContent?.chatMlJson || {})
|
||||
: undefined,
|
||||
provided_model_name:
|
||||
obsType === "GENERATION"
|
||||
? this.randomElement(REALISTIC_MODELS)
|
||||
: undefined,
|
||||
model_parameters:
|
||||
obsType === "GENERATION"
|
||||
? JSON.stringify({ temperature: 0.7 })
|
||||
: undefined,
|
||||
usage_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(20, 200),
|
||||
output: this.randomInt(10, 100),
|
||||
total: this.randomInt(30, 300),
|
||||
}
|
||||
: undefined,
|
||||
provided_usage_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(20, 200),
|
||||
output: this.randomInt(10, 100),
|
||||
total: this.randomInt(30, 300),
|
||||
}
|
||||
: undefined,
|
||||
cost_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(1, 10) / 100000,
|
||||
output: this.randomInt(1, 20) / 100000,
|
||||
total: this.randomInt(2, 30) / 100000,
|
||||
}
|
||||
: undefined,
|
||||
provided_cost_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(1, 10) / 100000,
|
||||
output: this.randomInt(1, 20) / 100000,
|
||||
total: this.randomInt(2, 30) / 100000,
|
||||
}
|
||||
: undefined,
|
||||
environment: "langfuse-evaluation",
|
||||
});
|
||||
|
||||
observations.push(observation);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
return observations;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates synthetic observations with automatic prompt linking (5% rate).
|
||||
* Use for: Large datasets, hierarchical observation structures, cost variation.
|
||||
*/
|
||||
generateSyntheticObservations(
|
||||
traces: TraceRecordInsertType[],
|
||||
observationsPerTrace: number = 5,
|
||||
): ObservationRecordInsertType[] {
|
||||
const observations: ObservationRecordInsertType[] = [];
|
||||
|
||||
traces.forEach((trace, traceIndex) => {
|
||||
for (let i = 0; i < observationsPerTrace; i++) {
|
||||
const obsType = this.randomElement(["GENERATION", "SPAN", "EVENT"]);
|
||||
|
||||
const observation: ObservationRecordInsertType = createObservation({
|
||||
id: `obs-synthetic-${traceIndex}-${i}`,
|
||||
trace_id: trace.id,
|
||||
project_id: trace.project_id,
|
||||
parent_observation_id:
|
||||
i > 0 ? `obs-synthetic-${traceIndex}-${i - 1}` : undefined,
|
||||
type: obsType as any,
|
||||
name:
|
||||
obsType === "GENERATION"
|
||||
? this.randomElement(REALISTIC_GENERATION_NAMES)
|
||||
: this.randomElement(REALISTIC_SPAN_NAMES),
|
||||
input:
|
||||
obsType === "GENERATION"
|
||||
? this.generateObservationInput()
|
||||
: undefined,
|
||||
output:
|
||||
obsType === "GENERATION"
|
||||
? this.generateObservationOutput()
|
||||
: undefined,
|
||||
provided_model_name:
|
||||
obsType === "GENERATION"
|
||||
? this.randomElement(REALISTIC_MODELS)
|
||||
: undefined,
|
||||
model_parameters:
|
||||
obsType === "GENERATION"
|
||||
? JSON.stringify({ temperature: 0.7 })
|
||||
: undefined,
|
||||
usage_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(20, 200),
|
||||
output: this.randomInt(10, 100),
|
||||
total: this.randomInt(30, 300),
|
||||
}
|
||||
: undefined,
|
||||
provided_usage_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(20, 200),
|
||||
output: this.randomInt(10, 100),
|
||||
total: this.randomInt(30, 300),
|
||||
}
|
||||
: undefined,
|
||||
cost_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(1, 10) / 100000,
|
||||
output: this.randomInt(1, 20) / 100000,
|
||||
total: this.randomInt(2, 30) / 100000,
|
||||
}
|
||||
: undefined,
|
||||
provided_cost_details:
|
||||
obsType === "GENERATION"
|
||||
? {
|
||||
input: this.randomInt(1, 10) / 100000,
|
||||
output: this.randomInt(1, 20) / 100000,
|
||||
total: this.randomInt(2, 30) / 100000,
|
||||
}
|
||||
: undefined,
|
||||
level: this.randomBoolean(0.85)
|
||||
? "DEFAULT"
|
||||
: this.randomBoolean(0.7)
|
||||
? "DEBUG"
|
||||
: this.randomBoolean(0.5)
|
||||
? "WARNING"
|
||||
: "ERROR",
|
||||
environment: trace.environment,
|
||||
});
|
||||
|
||||
observations.push(observation);
|
||||
}
|
||||
});
|
||||
|
||||
return observations;
|
||||
}
|
||||
|
||||
generateSyntheticScores(
|
||||
traces: TraceRecordInsertType[],
|
||||
observations: ObservationRecordInsertType[],
|
||||
scoresPerTrace: number = 2,
|
||||
): ScoreRecordInsertType[] {
|
||||
const scores: ScoreRecordInsertType[] = [];
|
||||
|
||||
traces.forEach((trace, traceIndex) => {
|
||||
for (let i = 0; i < scoresPerTrace; i++) {
|
||||
const scoreType = this.randomElement([
|
||||
"NUMERIC",
|
||||
"CATEGORICAL",
|
||||
"BOOLEAN",
|
||||
]);
|
||||
|
||||
let value: number | undefined;
|
||||
let stringValue: string | undefined;
|
||||
|
||||
switch (scoreType) {
|
||||
case "NUMERIC":
|
||||
value = Math.random() * 100;
|
||||
break;
|
||||
case "CATEGORICAL":
|
||||
stringValue = `category_${this.randomInt(1, 5)}`;
|
||||
break;
|
||||
case "BOOLEAN":
|
||||
value = this.randomBoolean() ? 1 : 0;
|
||||
stringValue = value === 1 ? "true" : "false";
|
||||
break;
|
||||
}
|
||||
|
||||
const score: ScoreRecordInsertType = createTraceScore({
|
||||
id: `score-synthetic-${traceIndex}-${i}`,
|
||||
project_id: trace.project_id,
|
||||
trace_id: trace.id,
|
||||
observation_id: this.randomBoolean(0.1)
|
||||
? this.randomElement(
|
||||
observations.filter((o) => o.trace_id === trace.id),
|
||||
)?.id
|
||||
: undefined,
|
||||
name: `metric_${this.randomInt(1, 10)}`,
|
||||
value,
|
||||
string_value: stringValue,
|
||||
data_type: scoreType as any,
|
||||
source: "API",
|
||||
comment: "Generated score\ntest",
|
||||
environment: trace.environment,
|
||||
});
|
||||
|
||||
scores.push(score);
|
||||
}
|
||||
});
|
||||
|
||||
return scores;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates evaluation traces for testing evaluator configurations.
|
||||
* Use for: Evaluation testing, score validation, evaluator development.
|
||||
*/
|
||||
generateEvaluationTraces(
|
||||
projectId: string,
|
||||
count: number,
|
||||
): TraceRecordInsertType[] {
|
||||
const traces: TraceRecordInsertType[] = [];
|
||||
|
||||
for (const evalJobConfiguration of SEED_EVALUATOR_CONFIGS) {
|
||||
for (let i = 0; i < count; i++) {
|
||||
const traceId = generateEvalTraceId(
|
||||
evalJobConfiguration.evalTemplateId,
|
||||
i,
|
||||
projectId,
|
||||
);
|
||||
const trace = createTrace({
|
||||
id: traceId,
|
||||
session_id: null,
|
||||
project_id: projectId,
|
||||
name: this.randomElement(REALISTIC_TRACE_NAMES),
|
||||
input: this.generateEvaluationInput(),
|
||||
output: this.generateEvaluationOutput(),
|
||||
user_id: this.randomBoolean(0.3)
|
||||
? `user_${this.randomInt(1, 1000)}`
|
||||
: null,
|
||||
environment: "langfuse-evaluation",
|
||||
metadata: { purpose: "evaluation" },
|
||||
tags: this.randomBoolean(0.3) ? ["production", "ai-agent"] : [],
|
||||
public: this.randomBoolean(0.8),
|
||||
bookmarked: this.randomBoolean(0.1),
|
||||
release: this.randomBoolean(0.4)
|
||||
? `v${this.randomInt(1, 5)}.${this.randomInt(0, 10)}`
|
||||
: null,
|
||||
version: this.randomBoolean(0.4)
|
||||
? `v${this.randomInt(1, 3)}.${this.randomInt(0, 20)}`
|
||||
: null,
|
||||
});
|
||||
|
||||
traces.push(trace);
|
||||
}
|
||||
}
|
||||
|
||||
return traces;
|
||||
}
|
||||
|
||||
private generateTraceInput(): string {
|
||||
if (!this.fileContent) return "Sample input";
|
||||
|
||||
// Match original logic: 30% chance of heavy markdown, otherwise chatML
|
||||
return this.randomBoolean(0.3)
|
||||
? this.fileContent.heavyMarkdown
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
private generateTraceOutput(): string {
|
||||
if (!this.fileContent) return "Sample output";
|
||||
|
||||
// Match original logic: 20% chance of nested JSON, otherwise chatML
|
||||
return this.randomBoolean(0.2)
|
||||
? JSON.stringify(this.fileContent.nestedJson)
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
private generateObservationInput(): string {
|
||||
if (!this.fileContent) return "Sample observation input";
|
||||
|
||||
// Match original logic: 40% chance of heavy markdown, otherwise chatML
|
||||
return this.randomBoolean(0.4)
|
||||
? this.fileContent.heavyMarkdown
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
private generateObservationOutput(): string {
|
||||
if (!this.fileContent) return "Sample observation output";
|
||||
|
||||
// Match original logic: 30% chance of nested JSON, otherwise chatML
|
||||
return this.randomBoolean(0.3)
|
||||
? JSON.stringify(this.fileContent.nestedJson)
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
private generateEvaluationInput(): string {
|
||||
if (!this.fileContent) return "Evaluation input";
|
||||
|
||||
return this.randomBoolean(0.3)
|
||||
? this.fileContent.heavyMarkdown
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
private generateEvaluationOutput(): string {
|
||||
if (!this.fileContent) return "Evaluation output";
|
||||
|
||||
return this.randomBoolean(0.2)
|
||||
? JSON.stringify(this.fileContent.nestedJson)
|
||||
: JSON.stringify(this.fileContent.chatMlJson);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates exactly one score per evaluation trace with prefixed IDs.
|
||||
* Use for: Evaluation traces that need score validation, evaluator testing.
|
||||
*/
|
||||
generateEvaluationScores(
|
||||
traces: TraceRecordInsertType[],
|
||||
_observations: ObservationRecordInsertType[],
|
||||
projectId: string,
|
||||
): ScoreRecordInsertType[] {
|
||||
const scores: ScoreRecordInsertType[] = [];
|
||||
|
||||
for (const evalJobConfiguration of SEED_EVALUATOR_CONFIGS) {
|
||||
traces.forEach((trace, traceIndex) => {
|
||||
if (traceIndex % FAILED_EVAL_TRACE_INTERVAL === 0) return;
|
||||
// Create exactly one score per evaluation trace with prefixed ID
|
||||
const score: ScoreRecordInsertType = createTraceScore({
|
||||
id: generateEvalScoreId(
|
||||
evalJobConfiguration.evalTemplateId,
|
||||
traceIndex,
|
||||
projectId,
|
||||
), // Use prefixed ID pattern
|
||||
project_id: projectId,
|
||||
trace_id: trace.id,
|
||||
observation_id: undefined, // Score is for the entire trace, not a specific observation
|
||||
name: `evaluation_score-${evalJobConfiguration.evalTemplateId}`,
|
||||
value: Math.random() * 100, // Random evaluation score 0-100
|
||||
string_value: undefined,
|
||||
data_type: "NUMERIC",
|
||||
source: "EVAL",
|
||||
comment: "Evaluation trace score",
|
||||
environment: trace.environment,
|
||||
});
|
||||
|
||||
scores.push(score);
|
||||
});
|
||||
}
|
||||
|
||||
return scores;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,266 @@
|
||||
# 🚀 Advanced AI Development Guide
|
||||
|
||||
## 📋 Table of Contents
|
||||
- [Getting Started](#getting-started)
|
||||
- [Architecture Overview](#architecture)
|
||||
- [Implementation Details](#implementation)
|
||||
- [Best Practices](#best-practices)
|
||||
|
||||
---
|
||||
|
||||
## 🎯 Getting Started
|
||||
|
||||
### Prerequisites
|
||||
\`\`\`bash
|
||||
npm install @langfuse/core
|
||||
pip install langfuse
|
||||
\`\`\`
|
||||
|
||||
### Quick Setup
|
||||
1. **Initialize your project**
|
||||
\`\`\`typescript
|
||||
import { Langfuse } from 'langfuse'
|
||||
|
||||
const langfuse = new Langfuse({
|
||||
secretKey: process.env.LANGFUSE_SECRET_KEY,
|
||||
publicKey: process.env.LANGFUSE_PUBLIC_KEY,
|
||||
baseUrl: 'https://cloud.langfuse.com'
|
||||
})
|
||||
\`\`\`
|
||||
|
||||
2. **Create your first trace**
|
||||
\`\`\`python
|
||||
from langfuse import Langfuse
|
||||
|
||||
langfuse = Langfuse()
|
||||
trace = langfuse.trace(name="chat-application")
|
||||
\`\`\`
|
||||
|
||||
---
|
||||
|
||||
## 🏗️ Architecture Overview
|
||||
|
||||
### System Components
|
||||
|
||||
| Component | Description | Status |
|
||||
|-----------|-------------|--------|
|
||||
| **Core Engine** | Main processing unit | ✅ Active |
|
||||
| **API Gateway** | Request routing | ✅ Active |
|
||||
| **Data Store** | Persistence layer | ⚠️ Maintenance |
|
||||
| **Analytics** | Metrics & insights | 🚧 Development |
|
||||
|
||||
### Data Flow
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
A[User Request] --> B[API Gateway]
|
||||
B --> C{Route Decision}
|
||||
C -->|Trace| D[Trace Handler]
|
||||
C -->|Generation| E[Generation Handler]
|
||||
C -->|Score| F[Score Handler]
|
||||
D --> G[Database]
|
||||
E --> G
|
||||
F --> G
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## ⚙️ Implementation Details
|
||||
|
||||
### Trace Management
|
||||
|
||||
> **Note:** Traces are the foundation of observability in LLM applications.
|
||||
|
||||
#### Creating Traces
|
||||
\`\`\`typescript
|
||||
// Basic trace creation
|
||||
const trace = langfuse.trace({
|
||||
name: "user-query-processing",
|
||||
userId: "user-123",
|
||||
sessionId: "session-456",
|
||||
metadata: {
|
||||
environment: "production",
|
||||
version: "2.1.0"
|
||||
}
|
||||
})
|
||||
|
||||
// Nested observations
|
||||
const span = trace.span({
|
||||
name: "document-retrieval",
|
||||
input: { query: "What is machine learning?" },
|
||||
metadata: { vectorStore: "pinecone" }
|
||||
})
|
||||
|
||||
const generation = span.generation({
|
||||
name: "answer-generation",
|
||||
model: "gpt-4",
|
||||
input: retrievedDocs,
|
||||
output: generatedAnswer,
|
||||
usage: {
|
||||
promptTokens: 1250,
|
||||
completionTokens: 420,
|
||||
totalTokens: 1670
|
||||
}
|
||||
})
|
||||
\`\`\`
|
||||
|
||||
### Advanced Features
|
||||
|
||||
#### 🔄 Async Processing
|
||||
\`\`\`python
|
||||
import asyncio
|
||||
from langfuse import Langfuse
|
||||
|
||||
async def process_batch():
|
||||
langfuse = Langfuse()
|
||||
|
||||
tasks = []
|
||||
for item in batch_items:
|
||||
task = asyncio.create_task(
|
||||
process_item_with_tracing(langfuse, item)
|
||||
)
|
||||
tasks.append(task)
|
||||
|
||||
results = await asyncio.gather(*tasks)
|
||||
return results
|
||||
\`\`\`
|
||||
|
||||
#### 🎯 Custom Scoring
|
||||
\`\`\`typescript
|
||||
// Automated scoring
|
||||
trace.score({
|
||||
name: "relevance",
|
||||
value: 0.95,
|
||||
comment: "Highly relevant response"
|
||||
})
|
||||
|
||||
// Human feedback scoring
|
||||
trace.score({
|
||||
name: "user-satisfaction",
|
||||
value: 1,
|
||||
source: "user-feedback",
|
||||
comment: "User rated 5/5 stars"
|
||||
})
|
||||
\`\`\`
|
||||
|
||||
---
|
||||
|
||||
## 🎨 Best Practices
|
||||
|
||||
### 📊 Monitoring & Observability
|
||||
|
||||
#### Key Metrics to Track
|
||||
- **Latency**: P50, P95, P99 response times
|
||||
- **Token Usage**: Cost optimization
|
||||
- **Error Rates**: System reliability
|
||||
- **User Satisfaction**: Quality metrics
|
||||
|
||||
#### Dashboard Setup
|
||||
\`\`\`yaml
|
||||
# monitoring-config.yml
|
||||
dashboards:
|
||||
- name: "LLM Performance"
|
||||
panels:
|
||||
- type: "time-series"
|
||||
title: "Response Latency"
|
||||
query: "avg(response_time) by (model)"
|
||||
- type: "stat"
|
||||
title: "Daily Token Usage"
|
||||
query: "sum(tokens_used)"
|
||||
- type: "table"
|
||||
title: "Top Errors"
|
||||
query: "topk(10, count by (error_type))"
|
||||
\`\`\`
|
||||
|
||||
### 🔐 Security Considerations
|
||||
|
||||
> ⚠️ **Important**: Never log sensitive user data in traces
|
||||
|
||||
#### Data Sanitization
|
||||
\`\`\`python
|
||||
def sanitize_input(data):
|
||||
"""Remove PII from trace data"""
|
||||
sanitized = data.copy()
|
||||
|
||||
# Remove email addresses
|
||||
sanitized = re.sub(r'\\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\\.[A-Z|a-z]{2,}\\b',
|
||||
'[EMAIL_REDACTED]', sanitized)
|
||||
|
||||
# Remove phone numbers
|
||||
sanitized = re.sub(r'\\b\\d{3}-\\d{3}-\\d{4}\\b',
|
||||
'[PHONE_REDACTED]', sanitized)
|
||||
|
||||
return sanitized
|
||||
\`\`\`
|
||||
|
||||
### 🚀 Performance Optimization
|
||||
|
||||
#### Batch Processing
|
||||
\`\`\`typescript
|
||||
// Efficient batch uploads
|
||||
const batchSize = 100
|
||||
const traces = []
|
||||
|
||||
for (let i = 0; i < data.length; i += batchSize) {
|
||||
const batch = data.slice(i, i + batchSize)
|
||||
const processedBatch = await Promise.all(
|
||||
batch.map(item => processWithLangfuse(item))
|
||||
)
|
||||
traces.push(...processedBatch)
|
||||
}
|
||||
|
||||
// Flush all traces at once
|
||||
await langfuse.flushAsync()
|
||||
\`\`\`
|
||||
|
||||
---
|
||||
|
||||
## 📚 Advanced Examples
|
||||
|
||||
### Multi-Agent System Tracing
|
||||
\`\`\`python
|
||||
class MultiAgentTracer:
|
||||
def __init__(self):
|
||||
self.langfuse = Langfuse()
|
||||
|
||||
async def orchestrate_agents(self, task):
|
||||
# Main orchestration trace
|
||||
main_trace = self.langfuse.trace(
|
||||
name="multi-agent-orchestration",
|
||||
input={"task": task}
|
||||
)
|
||||
|
||||
# Agent 1: Research
|
||||
research_span = main_trace.span(name="research-agent")
|
||||
research_result = await self.research_agent.process(task)
|
||||
research_span.end(output=research_result)
|
||||
|
||||
# Agent 2: Analysis
|
||||
analysis_span = main_trace.span(name="analysis-agent")
|
||||
analysis_result = await self.analysis_agent.process(research_result)
|
||||
analysis_span.end(output=analysis_result)
|
||||
|
||||
# Agent 3: Synthesis
|
||||
synthesis_span = main_trace.span(name="synthesis-agent")
|
||||
final_result = await self.synthesis_agent.process(analysis_result)
|
||||
synthesis_span.end(output=final_result)
|
||||
|
||||
main_trace.end(output=final_result)
|
||||
return final_result
|
||||
\`\`\`
|
||||
|
||||
---
|
||||
|
||||
## 🎉 Conclusion
|
||||
|
||||
With proper implementation of Langfuse tracing, you can:
|
||||
|
||||
- ✅ **Monitor** your LLM applications in real-time
|
||||
- ✅ **Debug** issues with detailed trace information
|
||||
- ✅ **Optimize** performance and costs
|
||||
- ✅ **Scale** your applications with confidence
|
||||
|
||||
### Next Steps
|
||||
1. Review the [official documentation](https://langfuse.com/docs)
|
||||
2. Join our [Discord community](https://discord.gg/langfuse)
|
||||
3. Check out [example projects](https://github.com/langfuse/langfuse)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,494 @@
|
||||
// Datasets
|
||||
const SEED_DATASET_ITEMS_COUNTRIES = [
|
||||
{ input: { country: "France" }, output: "Paris" },
|
||||
{ input: { country: "Germany" }, output: "Berlin" },
|
||||
{ input: { country: "Italy" }, output: "Rome" },
|
||||
{ input: { country: "Spain" }, output: "Madrid" },
|
||||
{ input: { country: "United Kingdom" }, output: "London" },
|
||||
{ input: { country: "Japan" }, output: "Tokyo" },
|
||||
{ input: { country: "China" }, output: "Beijing" },
|
||||
{ input: { country: "India" }, output: "New Delhi" },
|
||||
{ input: { country: "Brazil" }, output: "Brasília" },
|
||||
{ input: { country: "Canada" }, output: "Ottawa" },
|
||||
{ input: { country: "Australia" }, output: "Canberra" },
|
||||
{ input: { country: "South Africa" }, output: "Pretoria" },
|
||||
{ input: { country: "Mexico" }, output: "Mexico City" },
|
||||
{ input: { country: "Russia" }, output: "Moscow" },
|
||||
{ input: { country: "Egypt" }, output: "Cairo" },
|
||||
{ input: { country: "Turkey" }, output: "Ankara" },
|
||||
{ input: { country: "Indonesia" }, output: "Jakarta" },
|
||||
{ input: { country: "South Korea" }, output: "Seoul" },
|
||||
{ input: { country: "Saudi Arabia" }, output: "Riyadh" },
|
||||
{ input: { country: "Argentina" }, output: "Buenos Aires" },
|
||||
{ input: { country: "Nigeria" }, output: "Abuja" },
|
||||
{ input: { country: "Pakistan" }, output: "Islamabad" },
|
||||
{ input: { country: "Thailand" }, output: "Bangkok" },
|
||||
{ input: { country: "Vietnam" }, output: "Hanoi" },
|
||||
{ input: { country: "Malaysia" }, output: "Kuala Lumpur" },
|
||||
{ input: { country: "Philippines" }, output: "Manila" },
|
||||
{ input: { country: "Singapore" }, output: "Singapore" },
|
||||
{ input: { country: "New Zealand" }, output: "Wellington" },
|
||||
{ input: { country: "Sweden" }, output: "Stockholm" },
|
||||
{ input: { country: "Norway" }, output: "Oslo" },
|
||||
{ input: { country: "Denmark" }, output: "Copenhagen" },
|
||||
{ input: { country: "Finland" }, output: "Helsinki" },
|
||||
{ input: { country: "Netherlands" }, output: "Amsterdam" },
|
||||
{ input: { country: "Belgium" }, output: "Brussels" },
|
||||
{ input: { country: "Switzerland" }, output: "Bern" },
|
||||
{ input: { country: "Austria" }, output: "Vienna" },
|
||||
{ input: { country: "Portugal" }, output: "Lisbon" },
|
||||
{ input: { country: "Greece" }, output: "Athens" },
|
||||
{ input: { country: "Poland" }, output: "Warsaw" },
|
||||
{ input: { country: "Ukraine" }, output: "Kyiv" },
|
||||
{ input: { country: "Romania" }, output: "Bucharest" },
|
||||
{ input: { country: "Hungary" }, output: "Budapest" },
|
||||
{ input: { country: "Czech Republic" }, output: "Prague" },
|
||||
{ input: { country: "Slovakia" }, output: "Bratislava" },
|
||||
{ input: { country: "Croatia" }, output: "Zagreb" },
|
||||
{ input: { country: "Serbia" }, output: "Belgrade" },
|
||||
{ input: { country: "Bulgaria" }, output: "Sofia" },
|
||||
{ input: { country: "Ireland" }, output: "Dublin" },
|
||||
{ input: { country: "Iceland" }, output: "Reykjavik" },
|
||||
{ input: { country: "Estonia" }, output: "Tallinn" },
|
||||
{ input: { country: "Latvia" }, output: "Riga" },
|
||||
{ input: { country: "Lithuania" }, output: "Vilnius" },
|
||||
];
|
||||
|
||||
const SEED_DATASET_ITEMS_IPA = [
|
||||
{ input: { word: "the" }, output: "/ðə/" },
|
||||
{ input: { word: "be" }, output: "/bi/" },
|
||||
{ input: { word: "to" }, output: "/tu/" },
|
||||
{ input: { word: "of" }, output: "/əv/" },
|
||||
{ input: { word: "and" }, output: "/ænd/" },
|
||||
{ input: { word: "a" }, output: "/ə/" },
|
||||
{ input: { word: "in" }, output: "/ɪn/" },
|
||||
{ input: { word: "that" }, output: "/ðæt/" },
|
||||
{ input: { word: "have" }, output: "/hæv/" },
|
||||
{ input: { word: "I" }, output: "/aɪ/" },
|
||||
{ input: { word: "it" }, output: "/ɪt/" },
|
||||
{ input: { word: "for" }, output: "/fɔr/" },
|
||||
{ input: { word: "not" }, output: "/nɑt/" },
|
||||
{ input: { word: "on" }, output: "/ɑn/" },
|
||||
{ input: { word: "with" }, output: "/wɪð/" },
|
||||
{ input: { word: "he" }, output: "/hi/" },
|
||||
{ input: { word: "as" }, output: "/æz/" },
|
||||
{ input: { word: "you" }, output: "/ju/" },
|
||||
{ input: { word: "do" }, output: "/du/" },
|
||||
{ input: { word: "at" }, output: "/æt/" },
|
||||
{ input: { word: "this" }, output: "/ðɪs/" },
|
||||
{ input: { word: "but" }, output: "/bʌt/" },
|
||||
{ input: { word: "his" }, output: "/hɪz/" },
|
||||
{ input: { word: "by" }, output: "/baɪ/" },
|
||||
{ input: { word: "from" }, output: "/frʌm/" },
|
||||
{ input: { word: "they" }, output: "/ðeɪ/" },
|
||||
{ input: { word: "we" }, output: "/wi/" },
|
||||
{ input: { word: "say" }, output: "/seɪ/" },
|
||||
{ input: { word: "her" }, output: "/hər/" },
|
||||
{ input: { word: "she" }, output: "/ʃi/" },
|
||||
{ input: { word: "or" }, output: "/ɔr/" },
|
||||
{ input: { word: "an" }, output: "/æn/" },
|
||||
{ input: { word: "will" }, output: "/wɪl/" },
|
||||
{ input: { word: "my" }, output: "/maɪ/" },
|
||||
{ input: { word: "one" }, output: "/wʌn/" },
|
||||
{ input: { word: "all" }, output: "/ɔl/" },
|
||||
{ input: { word: "would" }, output: "/wʊd/" },
|
||||
{ input: { word: "there" }, output: "/ðɛr/" },
|
||||
{ input: { word: "their" }, output: "/ðɛr/" },
|
||||
{ input: { word: "what" }, output: "/wʌt/" },
|
||||
{ input: { word: "so" }, output: "/soʊ/" },
|
||||
{ input: { word: "up" }, output: "/ʌp/" },
|
||||
{ input: { word: "out" }, output: "/aʊt/" },
|
||||
{ input: { word: "if" }, output: "/ɪf/" },
|
||||
{ input: { word: "about" }, output: "/əˈbaʊt/" },
|
||||
{ input: { word: "who" }, output: "/hu/" },
|
||||
{ input: { word: "get" }, output: "/gɛt/" },
|
||||
{ input: { word: "which" }, output: "/wɪtʃ/" },
|
||||
{ input: { word: "go" }, output: "/goʊ/" },
|
||||
{ input: { word: "me" }, output: "/mi/" },
|
||||
{ input: { word: "when" }, output: "/wɛn/" },
|
||||
{ input: { word: "make" }, output: "/meɪk/" },
|
||||
{ input: { word: "can" }, output: "/kæn/" },
|
||||
{ input: { word: "like" }, output: "/laɪk/" },
|
||||
{ input: { word: "time" }, output: "/taɪm/" },
|
||||
{ input: { word: "no" }, output: "/noʊ/" },
|
||||
{ input: { word: "just" }, output: "/dʒʌst/" },
|
||||
{ input: { word: "him" }, output: "/hɪm/" },
|
||||
{ input: { word: "know" }, output: "/noʊ/" },
|
||||
{ input: { word: "take" }, output: "/teɪk/" },
|
||||
{ input: { word: "person" }, output: "/ˈpərsən/" },
|
||||
{ input: { word: "into" }, output: "/ˈɪntu/" },
|
||||
{ input: { word: "year" }, output: "/jɪr/" },
|
||||
{ input: { word: "your" }, output: "/jʊər/" },
|
||||
{ input: { word: "good" }, output: "/gʊd/" },
|
||||
{ input: { word: "some" }, output: "/sʌm/" },
|
||||
{ input: { word: "could" }, output: "/kʊd/" },
|
||||
{ input: { word: "them" }, output: "/ðɛm/" },
|
||||
{ input: { word: "see" }, output: "/si/" },
|
||||
{ input: { word: "other" }, output: "/ˈʌðər/" },
|
||||
{ input: { word: "than" }, output: "/ðæn/" },
|
||||
{ input: { word: "then" }, output: "/ðɛn/" },
|
||||
{ input: { word: "now" }, output: "/naʊ/" },
|
||||
{ input: { word: "look" }, output: "/lʊk/" },
|
||||
{ input: { word: "only" }, output: "/ˈoʊnli/" },
|
||||
{ input: { word: "come" }, output: "/kʌm/" },
|
||||
{ input: { word: "its" }, output: "/ɪts/" },
|
||||
{ input: { word: "over" }, output: "/ˈoʊvər/" },
|
||||
{ input: { word: "think" }, output: "/θɪŋk/" },
|
||||
{ input: { word: "also" }, output: "/ˈɔlsoʊ/" },
|
||||
{ input: { word: "back" }, output: "/bæk/" },
|
||||
{ input: { word: "after" }, output: "/ˈæftər/" },
|
||||
{ input: { word: "use" }, output: "/juz/" },
|
||||
{ input: { word: "two" }, output: "/tu/" },
|
||||
{ input: { word: "how" }, output: "/haʊ/" },
|
||||
{ input: { word: "our" }, output: "/aʊər/" },
|
||||
{ input: { word: "work" }, output: "/wɜrk/" },
|
||||
{ input: { word: "first" }, output: "/fɜrst/" },
|
||||
{ input: { word: "well" }, output: "/wɛl/" },
|
||||
{ input: { word: "way" }, output: "/weɪ/" },
|
||||
{ input: { word: "even" }, output: "/ˈivɪn/" },
|
||||
{ input: { word: "new" }, output: "/nu/" },
|
||||
{ input: { word: "want" }, output: "/wɑnt/" },
|
||||
{ input: { word: "because" }, output: "/bɪˈkɔz/" },
|
||||
{ input: { word: "any" }, output: "/ˈɛni/" },
|
||||
{ input: { word: "these" }, output: "/ðiz/" },
|
||||
{ input: { word: "give" }, output: "/gɪv/" },
|
||||
{ input: { word: "day" }, output: "/deɪ/" },
|
||||
{ input: { word: "most" }, output: "/moʊst/" },
|
||||
{ input: { word: "us" }, output: "/ʌs/" },
|
||||
];
|
||||
|
||||
export const SEED_DATASETS = [
|
||||
{
|
||||
name: "demo-countries-dataset",
|
||||
description: "Dataset for countries",
|
||||
metadata: {
|
||||
key: "value",
|
||||
},
|
||||
items: SEED_DATASET_ITEMS_COUNTRIES,
|
||||
},
|
||||
{
|
||||
name: "demo-english-transcription-dataset",
|
||||
description:
|
||||
"Dataset for english transcription, where words are represented in their international phonetic alphabet (IPA)",
|
||||
metadata: {
|
||||
key: "value",
|
||||
},
|
||||
items: SEED_DATASET_ITEMS_IPA,
|
||||
},
|
||||
];
|
||||
|
||||
// Prompts
|
||||
export const SEED_TEXT_PROMPTS = [
|
||||
{
|
||||
id: `prompt-parent`,
|
||||
createdBy: "user-1",
|
||||
prompt:
|
||||
'You are a very enthusiastic Langfuse representative who loves to help people! Langfuse is an open-source observability tool for developers of applications that use Large Language Models (LLMs). Given the following sections from the Langfuse documentation, answer the question using only that information, outputted in markdown format. Refer to the respective links of the documentation.\n \nSTART of Langfuse Documentation\n"""\n{{context}} {{context}}\n"""\nEND of Langfuse Documentation\n \nAnswer as markdown (including related code snippets if available), use highlights and paragraphs to structure the text. Use emojis in your answers. Do not mention that you are "enthusiastic", the user does not need to know, will feel it from the style of your answers. Only use information that is available in the context, do not make up any code that is not in the context. If you are unsure and the answer is not explicitly written in the documentation, say "Sorry, I don\'t know how to help with that." If the user is having problems using Langfuse, tell her to reach out to the founders directly via the chat widget. Make it crisp.\n\n@@@langfusePrompt:name=child-prompt|label=production@@@',
|
||||
name: "parent-prompt",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-child`,
|
||||
createdBy: "user-1",
|
||||
prompt: `Please follow these guidelines:
|
||||
- Refer to the respective links of the documentation
|
||||
- Be kind.
|
||||
- Include emojis where it makes sense.
|
||||
- If the users have problems using Langfuse, tell them to reach out to the founders directly via the chat widget or GitHub at the end of your answer.
|
||||
- Answer as markdown, use highlights and paragraphs to structure the text.
|
||||
- Do not mention that you are "enthusiastic", the user does not need to know, will feel it from the style of your answers.`,
|
||||
name: "child-prompt",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-123`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 1 content",
|
||||
name: "prompt-1",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-456`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 2 content",
|
||||
name: "prompt-2",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-789`,
|
||||
createdBy: "API",
|
||||
prompt: "Prompt 3 content",
|
||||
name: "prompt-3-by-api",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
{
|
||||
id: `prompt-abc`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 content",
|
||||
name: "prompt-4",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `countries-experiment-prompt`,
|
||||
createdBy: "user-1",
|
||||
prompt: "What is the capital of {{country}}?",
|
||||
name: "countries-experiment-prompt",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `folder-customer-prompt-1`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Folder prompt 1 content",
|
||||
name: "folder/customer/prompt-1",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `folder-customer-prompt-2`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Folder prompt 2 content",
|
||||
name: "folder/customer/prompt-2",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `folder-prompt-1`,
|
||||
createdBy: "user-1",
|
||||
prompt: "Folder prompt 1 content",
|
||||
name: "folder/prompt-1",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `prompt-with-many-labels`,
|
||||
createdBy: "user-1",
|
||||
prompt:
|
||||
"This is a comprehensive prompt for testing multiple label scenarios. It demonstrates how prompts can be tagged with numerous labels for organization, categorization, and filtering purposes. Use this prompt to understand how label management works at scale. Variables: {{input}}",
|
||||
name: "prompt-with-many-labels",
|
||||
version: 1,
|
||||
labels: [
|
||||
"production",
|
||||
"latest",
|
||||
"v1",
|
||||
"v2",
|
||||
"stable",
|
||||
"beta",
|
||||
"alpha",
|
||||
"test",
|
||||
"development",
|
||||
"staging",
|
||||
"experimental",
|
||||
"feature",
|
||||
"bugfix",
|
||||
"hotfix",
|
||||
"critical",
|
||||
"high-priority",
|
||||
"medium-priority",
|
||||
"low-priority",
|
||||
"urgent",
|
||||
"customer-facing",
|
||||
"internal",
|
||||
"public",
|
||||
"private",
|
||||
"confidential",
|
||||
"ai",
|
||||
"nlp",
|
||||
"chatbot",
|
||||
"assistant",
|
||||
"automation",
|
||||
"ml",
|
||||
"data",
|
||||
"analytics",
|
||||
"monitoring",
|
||||
"logging",
|
||||
"debug",
|
||||
"performance",
|
||||
"security",
|
||||
"compliance",
|
||||
"audit",
|
||||
"review",
|
||||
"approved",
|
||||
"rejected",
|
||||
"pending",
|
||||
"archived",
|
||||
"deprecated",
|
||||
"legacy",
|
||||
"migration",
|
||||
"upgrade",
|
||||
"downgrade",
|
||||
"template",
|
||||
"example",
|
||||
],
|
||||
tags: [],
|
||||
},
|
||||
];
|
||||
|
||||
export const SEED_CHAT_ML_PROMPTS = [
|
||||
{
|
||||
id: `prompt-abc`,
|
||||
createdBy: "user-1",
|
||||
prompt: [
|
||||
{
|
||||
role: "system",
|
||||
content:
|
||||
'You are a very enthusiastic Langfuse representative who loves to help people! Langfuse is an open-source observability tool for developers of applications that use Large Language Models (LLMs). Given the following sections from the Langfuse documentation, answer the question using only that information, outputted in markdown format.\n\nPlease follow these guidelines:\n- Refer to the respective links of the documentation and select quality examples\n- Be kind.\n- Include emojis where it makes sense.\n- If the users have problems using Langfuse, tell them to reach out to the founders directly via the chat widget or GitHub at the end of your answer.\n- Answer as markdown, use highlights and paragraphs to structure the text.\n- Do not mention that you are "enthusiastic", the user does not need to know, will feel it from the style of your answers.\n- Only use information that is available in the context, do not make up any code that is not in the context.\n- Always put an empji at the end of the message.',
|
||||
},
|
||||
{
|
||||
role: "assistant",
|
||||
content:
|
||||
"All right, what is the documentation that I am meant to exclusively use to answer the question?",
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: "<documentation>\n```\n{{context}}\n```\n</documentation>",
|
||||
},
|
||||
{
|
||||
role: "assistant",
|
||||
content:
|
||||
"Answering in next message based on your instructions only. What is the question?",
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: "{{question}}",
|
||||
},
|
||||
],
|
||||
name: "prompt-chat-ml",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
{
|
||||
id: `prompt-chat-placeholder`,
|
||||
createdBy: "user-1",
|
||||
prompt: [
|
||||
{
|
||||
role: "system",
|
||||
content:
|
||||
'You are a very enthusiastic Langfuse representative who loves to help people! Langfuse is an open-source observability tool for developers of applications that use Large Language Models (LLMs). Given the following sections from the Langfuse documentation, answer the question using only that information, outputted in markdown format.\n\nPlease follow these guidelines:\n- Refer to the respective links of the documentation and select quality examples\n- Be kind.\n- Include emojis where it makes sense.\n- If the users have problems using Langfuse, tell them to reach out to the founders directly via the chat widget or GitHub at the end of your answer.\n- Answer as markdown, use highlights and paragraphs to structure the text.\n- Do not mention that you are "enthusiastic", the user does not need to know, will feel it from the style of your answers.\n- Only use information that is available in the context, do not make up any code that is not in the context.\n- Always put an empji at the end of the message.',
|
||||
},
|
||||
{
|
||||
type: "placeholder",
|
||||
name: "message_history",
|
||||
},
|
||||
{
|
||||
role: "assistant",
|
||||
content:
|
||||
"Answering in next message based on your instructions only. What is the question?",
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: "{{question}}",
|
||||
},
|
||||
],
|
||||
name: "prompt-chat-ml-with-placeholder",
|
||||
version: 1,
|
||||
labels: ["production", "latest"],
|
||||
tags: ["tag1", "tag2"],
|
||||
},
|
||||
];
|
||||
|
||||
export const SEED_PROMPT_VERSIONS = [
|
||||
{
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 version 1 content with {{variable}}",
|
||||
name: "prompt-4-with-variable-and-config",
|
||||
config: {
|
||||
temperature: 0.7,
|
||||
},
|
||||
version: 1,
|
||||
},
|
||||
{
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 version 2 content with {{variable}}",
|
||||
name: "prompt-4-with-variable-and-config",
|
||||
config: {
|
||||
temperature: 0.7,
|
||||
topP: 0.9,
|
||||
},
|
||||
version: 2,
|
||||
labels: ["production"],
|
||||
},
|
||||
{
|
||||
createdBy: "user-1",
|
||||
prompt: "Prompt 4 version 3 content with {{variable}}",
|
||||
name: "prompt-4-with-variable-and-config",
|
||||
config: {
|
||||
temperature: 0.7,
|
||||
topP: 0.9,
|
||||
frequencyPenalty: 0.5,
|
||||
},
|
||||
version: 3,
|
||||
labels: ["production", "latest"],
|
||||
},
|
||||
];
|
||||
|
||||
// evaluators
|
||||
export const SEED_EVALUATOR_TEMPLATES = [
|
||||
{
|
||||
id: "toxicity-template",
|
||||
name: "toxicity-template",
|
||||
version: 1,
|
||||
prompt:
|
||||
"Please evaluate the toxicity of the following text {{input}} {{output}}",
|
||||
model: "gpt-3.5-turbo",
|
||||
vars: ["input", "output"],
|
||||
provider: "openai",
|
||||
outputSchema: {
|
||||
score: "provide a score between 0 and 1",
|
||||
reasoning: "one sentence reasoning for the score",
|
||||
},
|
||||
modelParams: {
|
||||
temperature: 0.7,
|
||||
outputTokenLimit: 100,
|
||||
topP: 0.9,
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
export const SEED_EVALUATOR_CONFIGS = [
|
||||
{
|
||||
id: "toxicity-job",
|
||||
evalTemplateId: "toxicity-template",
|
||||
jobType: "EVAL",
|
||||
status: "ACTIVE",
|
||||
scoreName: "toxicity",
|
||||
filter: [
|
||||
{
|
||||
type: "string",
|
||||
value: "user",
|
||||
column: "User ID",
|
||||
operator: "contains",
|
||||
},
|
||||
],
|
||||
variableMapping: [
|
||||
{
|
||||
langfuseObject: "trace",
|
||||
selectedColumnId: "input",
|
||||
templateVariable: "input",
|
||||
},
|
||||
{
|
||||
langfuseObject: "trace",
|
||||
selectedColumnId: "metadata",
|
||||
templateVariable: "output",
|
||||
},
|
||||
],
|
||||
targetObject: "trace",
|
||||
sampling: 1,
|
||||
delay: 5_000,
|
||||
},
|
||||
];
|
||||
|
||||
export const EVAL_TRACE_COUNT = 100;
|
||||
export const FAILED_EVAL_TRACE_INTERVAL = 10;
|
||||
@@ -0,0 +1,50 @@
|
||||
export const generateDatasetRunItemId = (
|
||||
datasetName: string,
|
||||
itemIndex: number,
|
||||
projectId: string,
|
||||
runNumber: number,
|
||||
) => {
|
||||
return `dataset-run-item-${datasetName}-${itemIndex}-${projectId.slice(-8)}-${runNumber}`;
|
||||
};
|
||||
|
||||
export const generateDatasetItemId = (
|
||||
datasetName: string,
|
||||
itemIndex: number,
|
||||
projectId: string,
|
||||
runNumber: number,
|
||||
) => {
|
||||
return `dataset-item-${datasetName}-${itemIndex}-${projectId.slice(-8)}-${runNumber}`;
|
||||
};
|
||||
|
||||
export const generateDatasetRunTraceId = (
|
||||
datasetName: string,
|
||||
itemIndex: number,
|
||||
projectId: string,
|
||||
runNumber: number,
|
||||
) => {
|
||||
return `trace-dataset-${datasetName}-${itemIndex}-${projectId.slice(-8)}-${runNumber}`;
|
||||
};
|
||||
|
||||
export const generateEvalTraceId = (
|
||||
evalTemplateId: string,
|
||||
index: number,
|
||||
projectId: string,
|
||||
) => {
|
||||
return `trace-eval-${evalTemplateId}-${projectId.slice(-8)}-${index}`;
|
||||
};
|
||||
|
||||
export const generateEvalObservationId = (
|
||||
evalTemplateId: string,
|
||||
index: number,
|
||||
projectId: string,
|
||||
) => {
|
||||
return `observation-eval-${evalTemplateId}-${projectId.slice(-8)}-${index}`;
|
||||
};
|
||||
|
||||
export const generateEvalScoreId = (
|
||||
evalTemplateId: string,
|
||||
index: number,
|
||||
projectId: string,
|
||||
) => {
|
||||
return `score-eval-${evalTemplateId}-${projectId.slice(-8)}-${index}`;
|
||||
};
|
||||
@@ -0,0 +1,344 @@
|
||||
import { FileContent, SeederOptions } from "./types";
|
||||
import { DataGenerator } from "./data-generators";
|
||||
import { ClickHouseQueryBuilder } from "./clickhouse-builder";
|
||||
import { EVAL_TRACE_COUNT, SEED_DATASETS } from "./postgres-seed-constants";
|
||||
import {
|
||||
clickhouseClient,
|
||||
logger,
|
||||
ObservationRecordInsertType,
|
||||
TraceRecordInsertType,
|
||||
} from "../../../src/server";
|
||||
import path from "path";
|
||||
import { readFileSync } from "fs";
|
||||
|
||||
/**
|
||||
* Orchestrates seeding operations across ClickHouse and PostgreSQL.
|
||||
*
|
||||
* Use createXxxData() for specific data types:
|
||||
* - createDatasetExperimentData(): Dataset runs in langfuse-prompt-experiments env
|
||||
* - createEvaluationData(): Evaluation data in langfuse-evaluation env
|
||||
* - createSyntheticData(): Large synthetic data in default env
|
||||
* - executeFullSeed(): All data types together
|
||||
*/
|
||||
export class SeederOrchestrator {
|
||||
private dataGenerator: DataGenerator;
|
||||
private queryBuilder: ClickHouseQueryBuilder;
|
||||
private fileContent: FileContent | null = null;
|
||||
|
||||
constructor() {
|
||||
this.dataGenerator = DataGenerator.getInstance();
|
||||
this.queryBuilder = new ClickHouseQueryBuilder();
|
||||
this.loadFileContent();
|
||||
}
|
||||
|
||||
private loadFileContent() {
|
||||
try {
|
||||
const nestedJsonPath = path.join(__dirname, "./nested_json.json");
|
||||
const heavyMarkdownPath = path.join(__dirname, "./markdown.txt");
|
||||
const chatMlJsonPath = path.join(__dirname, "./chat_ml_json.json");
|
||||
|
||||
const nestedJsonContent = JSON.parse(
|
||||
readFileSync(nestedJsonPath, "utf-8"),
|
||||
);
|
||||
const heavyMarkdownContent = readFileSync(heavyMarkdownPath, "utf-8");
|
||||
const chatMlJsonContent = JSON.parse(
|
||||
readFileSync(chatMlJsonPath, "utf-8"),
|
||||
);
|
||||
|
||||
// Truncate large content for reasonable test data size
|
||||
const truncatedNestedJson = {
|
||||
...nestedJsonContent,
|
||||
products: nestedJsonContent.products?.slice(0, 3) || [],
|
||||
};
|
||||
|
||||
const truncatedChatMlJson = {
|
||||
...chatMlJsonContent,
|
||||
messages: chatMlJsonContent.messages?.slice(0, 4) || [],
|
||||
};
|
||||
|
||||
this.fileContent = {
|
||||
nestedJson: truncatedNestedJson,
|
||||
heavyMarkdown: heavyMarkdownContent,
|
||||
chatMlJson: truncatedChatMlJson,
|
||||
};
|
||||
|
||||
this.dataGenerator.setFileContent(this.fileContent);
|
||||
} catch (error) {
|
||||
logger.warn(
|
||||
"Could not load file content for seeding, using fallback data",
|
||||
error,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates dataset experiment data for A/B testing and prompt comparisons.
|
||||
* Use for: Experiment tracking, dataset-based evaluations, prompt testing.
|
||||
*/
|
||||
async createDatasetExperimentData(
|
||||
projectIds: string[],
|
||||
opts: SeederOptions,
|
||||
): Promise<void> {
|
||||
logger.info(
|
||||
`Creating dataset experiment data for ${projectIds.length} projects.`,
|
||||
);
|
||||
|
||||
for (const projectId of projectIds) {
|
||||
logger.info(`Processing project ${projectId}`);
|
||||
|
||||
const numberOfRuns = opts.numberOfRuns || 1;
|
||||
|
||||
for (let runNumber = 0; runNumber < numberOfRuns; runNumber++) {
|
||||
logger.info(
|
||||
`Processing run ${runNumber + 1}/${numberOfRuns} for project ${projectId}`,
|
||||
);
|
||||
// const now = Date.now();
|
||||
|
||||
const traces: TraceRecordInsertType[] = [];
|
||||
const observations: ObservationRecordInsertType[] = [];
|
||||
// const datasetRunItems: DatasetRunItemRecordInsertType[] = [];
|
||||
|
||||
for (const seedDataset of SEED_DATASETS) {
|
||||
for (const [itemIndex, datasetItem] of seedDataset.items.entries()) {
|
||||
// // Generate dataset run item data
|
||||
// const datasetRunItem = this.dataGenerator.generateDatasetRunItem(
|
||||
// {
|
||||
// datasetName: seedDataset.name,
|
||||
// itemIndex,
|
||||
// item: datasetItem,
|
||||
// runNumber,
|
||||
// runCreatedAt: now,
|
||||
// },
|
||||
// projectId,
|
||||
// );
|
||||
|
||||
// Generate trace data
|
||||
const trace = this.dataGenerator.generateDatasetTrace(
|
||||
{
|
||||
datasetName: seedDataset.name,
|
||||
itemIndex,
|
||||
item: datasetItem,
|
||||
runNumber,
|
||||
},
|
||||
projectId,
|
||||
);
|
||||
|
||||
// Generate observation data
|
||||
const observation = this.dataGenerator.generateDatasetObservation(
|
||||
trace,
|
||||
{
|
||||
datasetName: seedDataset.name,
|
||||
itemIndex,
|
||||
item: datasetItem,
|
||||
runNumber,
|
||||
},
|
||||
projectId,
|
||||
);
|
||||
|
||||
traces.push(trace);
|
||||
observations.push(observation);
|
||||
// datasetRunItems.push(datasetRunItem);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
await this.queryBuilder.executeTracesInsert(traces);
|
||||
await this.queryBuilder.executeObservationsInsert(observations);
|
||||
// await this.queryBuilder.executeDatasetRunItemsInsert(datasetRunItems);
|
||||
} catch (error) {
|
||||
logger.error(`✗ Insert failed:`, error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates evaluation data for testing evaluator configurations.
|
||||
* Use for: Evaluator development, score validation, evaluation testing.
|
||||
*/
|
||||
async createEvaluationData(projectIds: string[]): Promise<void> {
|
||||
logger.info(`Creating evaluation data for ${projectIds.length} projects.`);
|
||||
|
||||
for (const projectId of projectIds) {
|
||||
logger.info(`Processing evaluation data for project ${projectId}`);
|
||||
|
||||
const evalTracesPerProject = EVAL_TRACE_COUNT;
|
||||
const evalObservationsPerTrace = 10;
|
||||
|
||||
// Generate evaluation traces
|
||||
const traces = this.dataGenerator.generateEvaluationTraces(
|
||||
projectId,
|
||||
evalTracesPerProject,
|
||||
);
|
||||
|
||||
// Generate evaluation observations
|
||||
const observations = this.dataGenerator.generateEvaluationObservations(
|
||||
traces,
|
||||
evalObservationsPerTrace,
|
||||
projectId,
|
||||
);
|
||||
|
||||
// Generate scores - exactly one score per evaluation trace
|
||||
const scores = this.dataGenerator.generateEvaluationScores(
|
||||
traces,
|
||||
observations,
|
||||
projectId,
|
||||
);
|
||||
|
||||
await this.queryBuilder.executeTracesInsert(traces);
|
||||
await this.queryBuilder.executeObservationsInsert(observations);
|
||||
await this.queryBuilder.executeScoresInsert(scores);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates large-scale synthetic data for performance testing and demos.
|
||||
* Use for: Load testing, dashboard demos, realistic usage simulation.
|
||||
*/
|
||||
async createSyntheticData(
|
||||
projectIds: string[],
|
||||
opts: SeederOptions,
|
||||
): Promise<void> {
|
||||
logger.info(`Creating synthetic data for ${projectIds.length} projects.`);
|
||||
|
||||
for (const projectId of projectIds) {
|
||||
logger.info(`Processing synthetic data for project ${projectId}`);
|
||||
|
||||
const observationsPerTrace = 15;
|
||||
const tracesPerProject = Math.floor(
|
||||
(opts.totalObservations || 1000) / observationsPerTrace,
|
||||
);
|
||||
const scoresPerTrace = 10;
|
||||
|
||||
// For large datasets, use bulk generation for better performance
|
||||
if (tracesPerProject > 100) {
|
||||
logger.info(`Using bulk generation for ${tracesPerProject} traces`);
|
||||
|
||||
const traceQuery = this.queryBuilder.buildBulkTracesInsert(
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
"default",
|
||||
this.fileContent || undefined,
|
||||
{ numberOfDays: opts.numberOfDays },
|
||||
);
|
||||
const observationQuery = this.queryBuilder.buildBulkObservationsInsert(
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
observationsPerTrace,
|
||||
"default",
|
||||
this.fileContent || undefined,
|
||||
{ numberOfDays: opts.numberOfDays },
|
||||
);
|
||||
const scoreQuery = this.queryBuilder.buildBulkScoresInsert(
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
scoresPerTrace,
|
||||
"default",
|
||||
{ numberOfDays: opts.numberOfDays },
|
||||
);
|
||||
|
||||
await this.executeQuery(traceQuery);
|
||||
await this.executeQuery(observationQuery);
|
||||
await this.executeQuery(scoreQuery);
|
||||
} else {
|
||||
// Use detailed generation for smaller datasets
|
||||
const traces = this.dataGenerator.generateSyntheticTraces(
|
||||
projectId,
|
||||
tracesPerProject,
|
||||
);
|
||||
const observations = this.dataGenerator.generateSyntheticObservations(
|
||||
traces,
|
||||
observationsPerTrace,
|
||||
);
|
||||
const scores = this.dataGenerator.generateSyntheticScores(
|
||||
traces,
|
||||
observations,
|
||||
scoresPerTrace,
|
||||
);
|
||||
|
||||
await this.queryBuilder.executeTracesInsert(traces);
|
||||
await this.queryBuilder.executeObservationsInsert(observations);
|
||||
await this.queryBuilder.executeScoresInsert(scores);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Executes complete seeding: datasets + evaluation + synthetic data.
|
||||
* Use for: Full system setup, comprehensive testing, complete data reset.
|
||||
*/
|
||||
async executeFullSeed(
|
||||
projectIds: string[],
|
||||
opts: SeederOptions,
|
||||
): Promise<void> {
|
||||
logger.info("Starting full seed process");
|
||||
|
||||
try {
|
||||
// Create dataset experiment data
|
||||
await this.createDatasetExperimentData(projectIds, opts);
|
||||
|
||||
// Create evaluation data
|
||||
await this.createEvaluationData(projectIds);
|
||||
|
||||
// Create synthetic data
|
||||
await this.createSyntheticData(projectIds, opts);
|
||||
|
||||
// Log completion statistics (commented out to reduce terminal noise)
|
||||
await this.logStatistics();
|
||||
|
||||
logger.info("Full seed process completed successfully");
|
||||
} catch (error) {
|
||||
logger.error("Seed process failed:", error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
private async executeQuery(query: string): Promise<void> {
|
||||
try {
|
||||
await clickhouseClient().command({
|
||||
query,
|
||||
clickhouse_settings: {
|
||||
wait_end_of_query: 1,
|
||||
},
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error("Query execution failed:", error);
|
||||
logger.error("Failed query:", query);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
private async logStatistics(): Promise<void> {
|
||||
const tables = ["traces", "scores", "observations"];
|
||||
|
||||
for (const table of tables) {
|
||||
try {
|
||||
const query = `
|
||||
SELECT
|
||||
project_id,
|
||||
count() AS per_project_count,
|
||||
bar(per_project_count, 0, (
|
||||
SELECT count(*)
|
||||
FROM ${table}
|
||||
), 50) AS bar_representation
|
||||
FROM ${table}
|
||||
GROUP BY project_id
|
||||
ORDER BY count() desc
|
||||
`;
|
||||
|
||||
const result = await clickhouseClient().query({
|
||||
query,
|
||||
format: "TabSeparated",
|
||||
});
|
||||
|
||||
logger.info(
|
||||
`${table.charAt(0).toUpperCase() + table.slice(1)} per Project: \n` +
|
||||
(await result.text()),
|
||||
);
|
||||
} catch (error) {
|
||||
logger.warn(`Could not log statistics for ${table}:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user