diff --git a/.envrc b/.envrc index 2483236..5a8fc5e 100644 --- a/.envrc +++ b/.envrc @@ -1,3 +1,5 @@ # shellcheck shell=bash use flake -dotenv_if_exists + +# shellcheck disable=SC2154 +eval "$("$direnv" dotenv bash <(sops -d --output-type dotenv secrets.yaml))" diff --git a/.gitignore b/.gitignore index eaeb69d..5255b32 100644 --- a/.gitignore +++ b/.gitignore @@ -8,4 +8,11 @@ result-* # cache __pycache__ -cocoindex.db +.pytest_cache/ + +# runtime state: engine db, vector store +var/ + +# secrets: the encrypted secrets.yaml is committed, plaintext never is +.env +secrets.yaml.dec diff --git a/.sops.yaml b/.sops.yaml new file mode 100644 index 0000000..dd1f9b9 --- /dev/null +++ b/.sops.yaml @@ -0,0 +1,9 @@ +creation_rules: + - key_groups: + - age: + - age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + path_regex: ^secrets\.yaml$ + - key_groups: + - age: + - age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + path_regex: ^evals/questions\.enc\.yaml$ diff --git a/evals/.gitignore b/evals/.gitignore new file mode 100644 index 0000000..8866ef9 --- /dev/null +++ b/evals/.gitignore @@ -0,0 +1,5 @@ +# The labels quote channel content, so only the encrypted copy is committed. +* +!.gitignore +!questions.example.yaml +!questions.enc.yaml diff --git a/evals/questions.enc.yaml b/evals/questions.enc.yaml new file mode 100644 index 0000000..6eac09e --- /dev/null +++ b/evals/questions.enc.yaml @@ -0,0 +1,15 @@ +data: ENC[AES256_GCM,data:jQOm2pPLH5Cbmlvd21T3KBFasl/baPgLE+wja75PaGAfhIvUSRV1PQEo10fBljZQUqUeekQbQr2A5lhzWBVr7wdCMj2euOrJJF3OhJSWZ/mkjhgQP9IHB88Az4eD0FH5O4h1Zbpmj7nOrsLGeZ4mswriJlN5GcjNrSSWTZ2ds3NB7dOMKBbxALJpMMqYRxmU5wkkdzNkEncb00S2FOurMbNRVwXuLkl9UyNWsGgWJSaodaQ4MfubRtLRzzDk/Um/JrtPijc04zLmljBVWC4xoO/rmej3hQByD67dSc/uF45Th07OSj+ZXU/dzhiUoc1mVDSZ7XrilbL+fNzyKJls0dXFg0+lATLDNdb204n1SuRtT7cvAaHWlAxcRB6Q5j853/0GrulTxg4kiq0SyGu+MSOCHfAePRlTxLADUoTC9tznceM5hpaf7I+u08+axmDs5c95URgQmT4orxEVC9kMM+BAwKgQN+Jq8BHhqfYlOFLJ+H66BBX3ftyErJ8Ij0oMFiM2ZuDvsdyE/VaISQtq4d4ekzk8Sl9BRNnVuuPJSPxpSacf8G/REVwwRLi5vpAXfiyQ0mUDdVOaLLP0zv5JyJV5TttDD86zCg3JJO/Dtvld5rm1vIT3Up4nNwhfW6OouMpt2xPWBm8TyLyFjOboNQavbpEl0iHJUteRDheWZM621mK7eK1dDwxgZyFP9trIhpyWGjvDdSROZ+ny7i34+0kYcwwhyiQfb1n/OVp+a5hp9x6MHu32zDjLXges93y2y0Fhflo9FOlUFB9qyuY9fIPNeCX1aSXjMHcVpg1kOvJRW7J31NmOS8tAQrb2Zs72CVPwMA58H3hfxM6oCfQ1TNjXlJ5RU6F/yCP0rCxnxOmkS4l/jnjArD/Lg6BDjRw8/2MHWD8/zIv97oc6/078XQiFJDUOIb16U1ALFayjwN+/SJ5rrgrCl/rRiXmRkMkqgHF5n68XVwm4Kou1Jcb9IjC7eNjl/sB/iDGAUCO5WW9gX0cihLLJtyRq05nXRM3caIOgUI2vYCGbqwwdCyjYuW+RjB8934ACpR6D+9EMQkfdi0xc2tpw+EGctPAaLLdAma7I9DdVzCyhPuGSbA5UA0w4U30iA4xZctHWZz2eex9LAuPndwDPzFaCkKeZaFesZlkz6ubxOXHswFpyUiDsNQ0f7cAOhr0uoRQS/MMsskDEdfS591Q/BuXYZaA2sKyFyEiAGh6FLU9QuI0VJu++8rDcO1SpEJFeTlEU1xNPDId0HvUitURWhGyLXeZ4pLpF13BzRtCcNBvVfldw0EVg38DUa4r7dyyWmoBIjXB2hatqoXfI7BqqMKdqUNF7P0VmpMe1rITE0ZstMt3rkCcSH3Q2zPO6lwgrSo5KMQPVHszGJvDY3EeTnSqo6gUKH3FPxid4ECEGzhHJSbPO1VpWawFF2/lbCP3j37m6IymYs+m4HPJtQzbxQOtO6I7lyGSPYOw+caUtCcEFYxHbx1KFMyNMyJ0QIotzUdRIVcw84c7oIqurb4lf/G4Urxk/BWpPx6NnQ0us3PKq+1n6YAjZIjzPoGZEmLiNoU9pKDASMbRJ+t2soT/CE95z8WcXQ1qPPpfAwFFggnQM81HnqdCiDlEpwdxv1QsQ9bRUl3Njtad+M/CdUl0s+C4k/8vD73RP2aoRwLo+M6rq7RG60Tec1zPrIBXgyEGIYrjZlhHN4lWFgKgOsE6JdQxGOy7zLS+TQAa86C1+ZqqMKqFdXrM2W4DSsBvwwYLb2agN5/s50tAuI/FhPcxNg70BZjfgYe4Xg1iIF//lgrjzwKYEjhXIbURb1xzXS8DhfO9J/4A62KfvrX1rlQuJIm+QO9qOAlxFZWnl988GWIZ7eo6r5XYoBQR8sWgVQtZEm8kTHjL8Qj2d0IceGW2DxC1wxTGAQ1VRJr78KjqICPUtq3WwOIJkRt2yGbpdXss5a7L2hGNdkKi1O+vtfEq7UzE70leEV9t0UBThTC3zhj3MKkS6uvI7Nd5zG/2PePO5uXzHSZyXIy1SPFloFx9QBgMuL7qvsFniklhQKW2mQgB1zuTu9DiTGHU0mup98YM/YLqm3yWgu0pWV5aalWgMUbaufluonNil24Nzww3CwvAEA6tjR2c6aumT6n9CARN4bWlRsmh9DhWEzTwAn8FZGS5MFjp+r1Jkfar79mVJ1njm1dhfYgos0jsM4CxRoQto92G80o2f235w0Gp1flnBn353Z50FJiC3/V+wFF1Y7XUfjGtcXQCg6vZBjVk6jg4J9M3brkJgHX+VERF8aewLvrn+NyhjvoSIeiUbTABTwXs6fbjyIO6uT+9MQrC8JNo/uyXGZselpRj6EhAaVFTbEIaXbkyg6x8mPnsVgwjqVGIughoEROGFNLYG6JeHzmvn3XS10sErvKrfBEgHYSJuRdqGLFgJLGuK+x2EqnyfMUQpvnTh5mnPT7+NZ+iD3/VfKBUohs1Hp493Gbgw+fRPJ9o2kaaGnVTy8N2SN7vo2Eta2QQ74WzHsgEVhVFIIZhLLrsAiPom317+uOuBlSO66rZZwjybHwd9bra6qjNIiVxTkthMys7A67+lwJEWkricSdu3SzEIgCRSlMkRmAkla0BYyCzfbCrp3quQTcz9kZ0mmFy5+SdO07pAWAvolnK249zYL2YPVJObBkVO70x8omco099dba8xKWVP2Xprdnw4zP7sU/lzlwfmuJZprqnJUMwCKGuqwhsClkJP527P1G0OewKw8EUxpr/a3uU5HXZ0U9SC5zFxEoFtlcKnVYfBM/xFsyFmPzZXhowrX9I8Cg7eKklDjhcT3b/gWYEiIR/SnzhGelbUBvCopSOj/uEPoOVn0ReYVSpGwX8T95QBQCUo7DeI2WhG9guuy2hRSbCjfq9XxDzRgD9iNe1IpHVYPvEhplxv1x+IuiaZZmAH9DcJrVNJqxFhd9C65xXOfdY2ZlLLE2khILUpcvsoWPC7mSHNZUwexSQwdMjowCBb5qgCAKi/yDj4H7C60epYtqL4RsvbGDNmw267wEmsOrGrw34DMl1C5PFawia0vnqJdFeloZrdOqlJ/ANZQdoZaf9GgPdyGn5iPoUzU+Sey/XIx+AS69F5X8qBcq0y8e9qtm/QLEgKSdpSopJkygiQiovG8uPgTu5crV86AgFPKKFAp5cSudO4eizFo61q26DbjqdPmunvOdZcZ8XNgzO/igCQk8KLFiQvtD9fwIeQyHfLV8BbzGLndEvzCzuVNmDeuj6jIUBO5wuROH99iO0kudtzYDMcJF2IXLuQETwOV9pMllMAvLujuIU2qVDIvc3osGaM+QssFL2DxQKvRH4DUoAMZ+S92SKRCArL/dXi6AMCZ058MQwXKF5W51DPRa/HWa1ehWkkaGaBbgKWdBHhp6mRvwmWFo2M3c5izMUiKzfnXbcuZQadzlElHTm/gxvcTiil5vUyqO6ctnoRTIIYYITuCSDNgtbPiNzEi8LWZ9dxPA1lFwHkELOVSrNWbHvXKIoWBCW+ocQD++GaDTBIhrnnbVYa0Eosx40oMsd6NRX2iB8jDh/HsfF7KcBwE/FJRq35QHUsbAq9p8I0LCH+emEULdaIFHK8yL6miSoM2QB5I27R2q3cRrnRKIAcrGtDfiQFTtigz1OVGCcGsIISexi+A7Thg9JSRoPdXFv2Y3xFH4N0rURUzlLRo1vmcfNRwVzZtB+Z+bH7odZyKHO3Q6+Z6U0BdUiwiN4SnR7JnLzhCxzYPizcHM6m02bUFsbRae0tca+hs5PIxvyhcLLVPwXIAavNcOj2Pqfx+rlILfJh5SMVD3cci3yYmKRCFoZ13Z+6LRaiFHd2krAxpJpmBTY9cr0LBn5svN3wFSC5wUQxiFhIlrsHaF+y+wXJ/tcWq6qWJEPJz0Po1eQhHDVBvpVJO8J6bg/low4HRB6+PeJQ7IgtkXT/7DPJnlhU5UG3SBm+7nip3sNjgD/kRydo2jNJkNABBnrqrzk9TvXX/wdJoRY9shNhSqqctpaXegglB2jTz44pNCxmS+ip0N/+sqnxaRedEcLgXgwpFwPD6Z7At/eqklm0jrnlAKW01gPzi6rFzvHuPOQ0w1iQhGA48dMbRLK/adqy6TcZBcKKjz0jtw+K9HNcSWXMKNru+AdI2hbK/GkLyZa53awmOYv3sTU7sFbl47bDjOkmVIC0ROKEBXS8oMBeK2AIlfNgSrqrtZXS9J7dSjAWjkN8sQVakf7s3ANuNGrFhQO1OOJX3EyVVhyFwGIFhNFFl/gjyN9Eut3xbCNClLMQPLVRwMSLCUvT1eFFCVZkJ2d9XGl96Ccf8rD0AE/tPJibM5vJYje3q+fhYuKN3t/AYqVr5eaBSrIuYyKiZFGcDcMQY2agzc6wM+VnKZ1hosnJeV+e07uBhFdCAuCcO50c7zsw2u7r/8R/xXQNhSDfUaPw42rMemx3nUgAgg2OL57zkH7a0yR8zII6SufY2EHhMnw8LMK3kYKPLhasAwF73V7wu6hkR5W6sCjy4Mr4fpaf5CHzR2ioRm5/w+Faz/vK8QrI1ii0eRmAN/UirXijsrTdsva68DetR2rG/EhUzdWIwKfkXcjI8yIsA5mdkh71pS/YPg4m2tYz5jwFvkC+/MFhyQLtsD2ANwTtGptdaHVf8zMslHKsFm8znY69pUDovqGjVAtAcEnzHmzIHiTVkyhi7Q6+r5C2GQc3F8VukaqUeaoTmDFKXvJ2CGKpFtQ2vpfPic+HeF46upSjhFBFabi1Nm0GkFWhsEIitY842cFCirMhG3ST2RYxgZMX+7aAl8yxeaMFZC5D1x7Km7EbypWGlIrYAB6hX1nTR7klHvCUDdPNO/w8+VqgpbwurnV0/O3rcoGlS8BrDFBCHZYP3Ym5KlLNthLwWR3+KzDW1ITCRks4hY5s9WeYFe/Zz/591gIuTXDxraEic6yObekZfNW6QNC5hLiha6KBLZrQXSCq1zHjwGvspr8SZN6PomXKRXQ1p5Z9saPRd5e92F/AY+2sT9uGQVsFwZ6Eh8rm0BLgVlNjgiWx7dWZoI6FFyWZBpzZUEBovf2a5iuIxu56h5ULoNtyYo7jlin3VQsjxkk9qxWKCPwrsWLurOtb+9skV2w2k1FbrwKDHpZCbmZpPoZe+geIhK8dALLExFji2Euux6ES59T/3fO36M7AxRd5IZ+ODofPH9QygclRarQam/Nzk/8cX5rFvNkqu1JVsekUzeW8sno7kfdPWeLHEF1/vjADOI5NqYWWEdNWloWArO1iO6IetRjvbrgwDropIBy1c+z1/DKFhjIcBi3K7Vz/2tKTYgfYS9YSP3W/r86nyHIUKOdlLC18M9xa+6hQC05V3NXmG+8ol+TTC/0jjJvnQOTuQluH8uhM5b29MIjjbUdPGXorx+res7xjwxtd0sYskIXuJoDWfCa4s3ene8lncEa2IqSe/MbCfKqYPcmBmgoGaPWyn6jwaUG7al7Wv2H78pSVIcks+t737ZiOebn5hls4v+fiWaREXQuwfetvju5DA6hywJmLzPf5df5IutMv8uZujhzd+FYKEsLLqJ35We/kpBgMAxPijRO5JZdYvP7zNzN98cMmjfUMSx59p3O5io9fg5dBp47QxBeSo9whWucCRcS7vKel6gn8ZEqzusqdjkSXXITt4kvYppSRaG0IW5ZiIoUh8s/o6Ir9unYB44hRMZxhIJsFtwPv3qhzZbBQuuqjmPcjh/9IOvHpsUCAJu6O5+BIOYoS9g6VJfGR2GVgTs2690862XWPkclwGSUxUmi2aiQgEGFtY7zoMjNCnKHkENVQMm+/FDrGFTD3tda498f5BmzgsM1pYghrmP4LLsBrVUc/ng0FfwSOVy/5UEyRQAnhlLkoMUYor+puJUzcLXNRFfh6w6oaV6zMr1MnCoTP87Dwo8VHzSmt/wZhUkhhFQvoWgRBBYPbJZiGsZnwQXGKn3m/t9Fqt0HxKZb5ucteJwqXX6UeOetlEFLJiCvSJBUWw6D0JbsPNvSQPJGgCmxaUtp6x0D5/IvKjjM8lo8nZBw++cF+ycFo+5dPq7UiYYO2X2gS/RYTLWcdWRmIbFk+VnBVvMkJztjhMvvUoXE+0jiet84yReASDqFRH1s6lHsPfmPz+PJE2dudkjEKZzSBr84WcWf4DsWqq7yc8EERTPWfMVQPQmDzL6Kzf5dlcfeTZanhZaM165jJcADpI4aeLShhJlTPiSYxEg6+cvk+T/zVgQmefYQDhSABDKjzRk0nEwjVjOEV9IgJb+YybxuHX6Szs9Mguzbr3n+Lj63UVYmSjBJfuW0lh6acnOLV+7mPmZvgTEhA+hplsNqqufBMa+hURBXJ118XYGJ9v9JoaUuXgRJFmAnddtf4TcYo+7vOeBfizeCELVBmSL5s5KpPVx340KujHwAe5ZQsnDKHdHamHGRaZ7ZpzNJtBkcDlR+EMTbPBzM6tHuGkVZDzbSpID1+rTYhXGZPWpURHvac7C5oPylfJ3Yu9LplpDThUmqWegIfQqdPVGyPBLEdNct9UsCPxqyI0IM31ZueWJtK1iSx340OHrs0ybDV78VbWlB5bxooZWaN8O9JlOrplpdsheQ/AWQUtH2s87YymlY/uaTfw5SJmSAsbyMMOG2pecu4OC65KBp/DEKOgg9L6xtz1ucbPf2ltH9bNE1pZTRh5zVko184+T4WDe27q4/Yor/5iKbo9JUS7MUSDYHKZ/03Yi3kWVPbE1eyuWXt5B3Sf921MJ/OX52wD7wl38cQ76WLJ/96M/a6OZTFblHwj4xqKUumGEiNxg4qQQppHRwkrTd0C9DFHMn2hDffD3wPRclssUR6UdYkFNAoONqf1urrdvJ1HNRMAb+MKcZVyVz2WiiNjIeyNGzwr8GDQdn9zyeAWn0sKgEGVbX4YyxjHGVxiBk45wBZP3NZkXMK8PBgpjBsDdg4OuoBrvds4wDR89Tb52PqcDQyiGP07wHuH8WTDl33jDFiLraFzDNiWMmVJnveabwUewEb0wN2yFg1+Y3c/NxkQ8xvxY+1NsRKDoE0e+2tOiHp21yvECalpsn2fUlVou5OglKrmxGwnqSSXHgY4+9V82iGFKkIozhx4N5e01inBg1598OIHidpc8LoDSH2ZXeK6UXLrv2UxiQ0gPmiSfvBXFXuLLZa5nHvLea3+GWltenPszWu2ZwF69yIlYkBbU4hay536ZZ1mUV/VLsCD599r60zmGzuXQTLn8//1CMIffHQRyJdwrCXbVmJtLG4BjeyfJhXzubXF5EIAOExJZDE9lYSBVLdVXmgj10w6N8c5SaXAR1KKDztWfLTq0GuwC6iKTBAjGQeNsTv/oomdtYDI5f9BvohYf+wNUZDlEhjtOUyYCasVJCxDSCnNzr3+DksIeMtalfhalPGwBUcXfnBTuAOamCsOmb/eeSvR0KbT0H1oDVFjT7XEZClJebHy/0nARH5mmTYtVuGnZrX7cbUtSckWLuu+OGIpY9fB9HWSkKNS5IFbGR8g5zQ0djCj7IvTEOybleJd6C9IHrVWo/7yBDjYMq3KtU2PipnglSBjImBkYYRM6VPheDD0FqC4Mccf4vSPAS4OSHvYODwpty8htpVqgQs3k550uSH9xl3qxrBmT7Ef8YiM+Hu3ONj/gr1SQ/X2BHpXpSfUonO9IErL8mjBPPvxYEGFPApzZyHkJ9VEfX9V2cohFQ2yL7NgC8js3MK5mOCpoLAC4dldo+ZsVhUWFmSVjDyoy1mBdM81UOIrKmxLnjiXWcjh5TsEYqxgU0fibSd1r3qp/XMXwGrNq/wsEcbJn61ualUzUDMy9jkOJgP3J5mneBGS7MoYDYK+ykBFmMXs+yfv5lSoD53myQfO2WW19ECBJNQmlMh3Bgf0U71sEeZwCVv27p+cYCTFHZXB23+vK9+YutcX7UNFL+gGKDl0ZC2+HHdpns7+fp3B0xFs4wC/DnF7l1U2+6kVZC2xU4Rg4BGVpv2huvq8xsIdIYbH4m40Vkjbl8uCErOSIGEstyzDVsTwukRqHvQmYMLA9WQFNigXezgSQCx02EcT7K19LOWBzVR1F/mXEK9IrZyVzwcrbdVvqds/MCkSoujF7WSqDjKIyEav0haw3uhYcD3YcflRvlgt1ZoNKwzj1nvJGQwqKBLwP6bd8aDLy15LzIQnT8JpT65upyum12fcEVBdWdHQM9N15M5nuJdHVuPs6ZYCUpJrsY70yjuPo5sKr9CwnnDh0zKkXE5NH7/OY9apwih6QNI0cSZPz0jpX0QSWO1wBNtdJ49u55yxZG3hyhLgCTQsHSnqhlrKoGRl9HlPuZ0Vtv0LLPolsCN7yNzQoT5ooEUIdVG6e6JDL6Cob+6DeyaF50Q6nHWMzznOg1RY0qXpT4aU9dBF0wqT79MIVA+gNKthM2H7PLUsdzx4r9s8XGTdS26bVH3vnIJsiWWZdAWTEFKjOmuRPsfY7fod7AWClk9Vu53VGpsW0R7+OSYZmLGd4hQ1y9sN/ovQBxnxukVFDXL1wobQI/A5dd4pOGK50zMbHOpcXfSao/HCMOQ98DmVRAwJ5zh5InJQ1FjjKGQdzqPsG9L7AjTKPjtFZvB4CqBkc4S6p0XG3bnI/P5pC89sJ6xYOMJUZpEwDvUWyMC0FjFlkvktJZyr/4TsNY4tWqZSwLimqmzzvFlycdKfpSHQWey78regtQUTmdumignwW/PJqvXB80BlZSaCSln7hRV/REAf8w7AkQUfIDbxGuCrJjn7Ioom/2a1gCNeBCoTjGSM6npMjcwBhXlQHHozUAQ4noVk1zUrTQm0uc/oKSHOM7BGDvT8eHQ5nCQNPbnc73782kbAwpqvBqSN+d3CjBmQyao4ZGf+lj+JESC6z+LwlL1FH6bYwXGxFKJQIqzDLKBpAtL1hX7LmhKgUCkjiQ4qRrrH1SpIyEyok7R/WvujwGsDuVfUrezDrWlQ18mmgQuq31Wc0NtV0zObLAuTgG3tNPungHehXIKC6OZAwDTz8tR0zACBmlfzyrHe0E1N9ILrvpZ9/H+2SRSjmN1tev9/AYNZCgCO25C24RiBN1iFrYLKvoXywvkOvys24jnJ6lRB4Z3fFyPoKKdkYhz7zhvzZPPwUJzsRWU7BBPG22FLAuWXFZtEzeUzREzVezEl53ebsNxC5A9+/2XTTUBYfDDIwFabbCxlIxc8husKszSPGFjwBlD/6ixRqKFx7jxW1lW92qVZD1E6b8jOhrHpyovL+ou5FZ1hp2MLOp1XEnGlZrfwBmQyzUY8upowwQV6ze6okAde7pQLPgNYk1loCXIf2GoETkcHyVX5rtO3g1QlZbaEoTQFqjnVu4wprMJXM9I10l4FWq/ypVSihwU59ZdGI6mG5RdqD0VAPbuuwnrl3N25Nw+NjL8iFu1FM7Mg4qGLT280zCF+XPD+M5RGkgm3n5JHb2WbsCgPCIJCj8th6BbAAmmkuy7oa5cJinslkM+Hbi7yu0/mgaDro0CK9u30G7YMMNCXbWnOaRRHOwRe/6rYYjyOb3Ev0aYAK7eywlE86u0OGjRDYZvtnG9p6lOy7soOABNONFS2wb6rc5ihzY0fUGeNpSAIFLF4CZWXRI7kfCaVaoJAGK+WhW7lipRXwka0Zx/Gvb5YcI8Goq4Q8bFJ9BEjEH+i+ezX4DQUC/lTL+IqGMpvDLIa9WeeTx14vWVslJnOfgxQz8hzASn3y0TIYMmcn6MlipLakqWjUx3LL6OTi8SzND6PE5demhCf5MRllziczVU6HY+CYcQ22X3Z6LmkIa7iKRYqyEUVHvDBKHv8yg79y51mjlQph6GtGigIZ8BpJ2KaxbQNEXuZKdcvzQ3w/rPn2mgJxmOBs44CCeIGmwL6shNdm7ZcpMqV5HMOD82n+irWbFQE0AQz7UtHLPuUfLNx0Yp2fBgIr01sQFcxpn2oxEZAeJIBh76DvtivkE66NUy4oODXsIVbldyvnQJtuZCnW3zpj+c06amP33mRjqhslFLziXc8UNo/dAIZ98bTIupucY7Jq80jJ9rxLdVxpAxCwY3QwaqxWZxqgnAk+mCdutHygpZJbGU1jX2Krr1e9UZ1gjD7jkgw3aEgbWrtRC5VnJkDMTvhk+k9vZOFJgVL+mNliMG3/Kwk4QWKNe3XZzYNxlPo8r5j/zaHZlpYS3aRYdFqGVvfp9CCPGyTXiOxNJRGXb6Jtne8Gv6u7oLMTWV5spXZvEfT2+Nll3++7kho2ros9hOan2KnsIQkqEdShaocNQtHDQG+NbFX9RUF5kCf3qwrwpo3xOZGdgBM1h49v2TD4W/EB3816ALAfU4Y6mFbBaPsyGPOljFoRRuj4ALI+d+ix9IKn3ay+mzF7VEpBBGyY9PerleUqQfRrs7M5STjaAljiOTNcicfgIMQ9InwKaT9d1guxvZaH1snJ9rIFqVz4E8EOowvwEI9iqoN7Kp8rZG4kCDj8Dbox2HfgV1TkKRMzvsELgf0dUWBwPl2yOgnX5mNld8GSIRZr5QCUlJpVfQ2lyID8StrgwaWj/AMtBv3wpHKRT95DSmIf4QgiMxQHL8DrS/5drBPqgwtlRJs4m/QkxHHBEB2k8LbXIpCrSZl8fBQCJyJOekBibIklbHLA3a7fezT2ElFmUN26UnOCH55BJa/5476eLBVfqP1YQ4y3rPfWOl2UIeNuDr1Le37Q+yHrkv26mrZWWOPqQxs+KTagD/G+x9dxDWFMCGnt7Qv95UEDTwGyjO9YfnR01COigdigFyWUDvec8kAYHh1yt997wjweZEUHalWC7sIEa2wiykO6l2MqHHUTMMFrArM81SbE/b1bNxJP4elB1vToQdFMygYvERMqRKjINqCdNhFGERQmIfvoYwiIm5ikO032VXT1EqwG6wkUoQ1AtjAUHkOY3ARptz2Z9XvZLnlN/8dA8sXCwdHjXd6yzi+NpbRvrUzvd9xfzTzC540kGn2Ksnb+ECldEEITn/Yen/baTIrmlK3Iwgf5VYO5jO34GmXUTr0fMaNiJDRIi17Yhkr+Gol9xfoGTGy5a2+277ocWKGFHSB2cLJOV9SjI/TGcTduCZXEsQHsRDiZSgjwnSy9aDN0pz9CQFeDmCP6BS9EheCN7z1do1vwEoBL/z82wvZcMtf9GRS761hie/yrS6fveUbEsIwJqarlK6FZhVrk3czTXA0lgTWcmB1TG5rDkaGzZYSYWwD+K9fzpRveHsvBjt5CHSYDfVnWWkB6nZyaVoc0l1Z0Z945AuNNClDgIYIjD58h5QcVOe6jOrDLzeekfTyej/Mws5+PWcjstXVQYpQHtHbup0FVZM7SQkUSZvz5oZRicS7FbcvzNEbIBq3yOm8Bgdik8LMTGA4quUgXmSOmSJHeBN/mX2FP+44a2L0vhclodk3Ec74QnR0DsCyr8LGjStgwB64vdMBmt0hE1mXDo+NeJdPTG0qB/0GnAy/eZW5B3tkJ+1snZOUgIJf55m3KK41jnWAT9bGb9Vp8fJNC9F314o0O8OzQneIj0+5yJ4kZPZvrITsIigJzEB88FQPhOV1c5x6ssJZrskpYntpzsUReCeoUgmRSeIm6Jl1RWVp1DSkjSSky9rCU6yYpjXoQFLL1GI5jpx4aAa5Te4quXs3Q1jYqP+PzhrjuZMr5VdMkZccDaJezHR8kRIbc2Q34rOkqfWRj3sqOL1EgvW6yFlJQt6nj6ISwyvqxdwY5F8Dpg/KIH5Nd5RA3s8F5CdlzAPGQ4xjKkaLd91ZLg+b63UxPKaOOInzyE7BIwIOIjjORURvxMCFFKL2DY12irPnxEl900/8UntEf5XTZZYKJsgQC/qeNCYbPxaNBldFWn633hLrAz20QKvzLTIoPCFBl+ppl6t+oa4XyDKxqtMCnzW1dgew2n8/ZEx/Av57YSOvd+q+Was0oU9679042JR/n3XqV/LPgegjEunI+fUs8ZmTdsJ7YH+Ptjs2242d4R9gvddlQdytamVGKxOvsN1en4Eldz0GJ88Wt6RgPkzeZHz3zwj4R6b2GQy25ymSlHh7I0cEEoN3EFBTeUqPotLE7oG1cDUzZ52I5WlGrUyLlwOfZTNIBDWrXa56oK9qh1Vy553sehevhYPmRgBDi1ep57icypIy9ZvIVh51q/XnPXLB7KDVXGHM0O/heQq6GbO3E89D34s+86eyM7dy65gb0TQowKPo2hfhe9EUHQ/JQBawZjSujfu6OcrXJLldN4sRQ9XvniKA26zaVAVqoaN/mun6XRngx5Y+QaS7k42VDJsSnvNhRcwWa70cpzVML7H04iczmO4Es62tjv8iTbOlQQgoNKNkPGj/ILHiA92u8us7c+NSM0b1tXt/jGTrUVVCUXsggNYrLo2/U79OkIjrc/bp4/XTSChz7FDSQzmDHrDojq9ojjkXzK0VI6vxzZYgSrsU/39NuVcMmpf68R8xeWJDolU9R4t5W3qYy1TGfGO9NP2E3ld+VRBt4gNPXO4c3U7dAf9FEK/MPR9G4a+m40VrGDMY4V2RPkk7g8ryCe608ATzd9yJCne9nKjO6v7mplCJLVnwmxWBovMVzPPMSCtaNoAcP4bNANswBpc3Vvg2Fa+i7dUMrif3KG06z1sK+sKzXv0Y0k4Nw/I2FTPEJ3L/8C7P1A0Hsgb0OY2vbdUu74HLiUeVhczBv/VHTuw6xhnsVV/AMfgFUSEWglRHXO7oAQ5EJl03EQiOo5P4IG0/LwYCoo/JThcpN1f8C1T1G47EPtNzardmTdokZK9uFrRN+eOX8PBlIm+tebIKbwzSsiFvmzg1CtvXJlqw7PIbkmlGQGiVuKBI8z8dsZTRsZvMRQ+rmWpiDe0YULA0UcMns8iNbv8HOxUHWli37Ij3imsOv5Gndzz0DqdSI1m1KrLDanVHSo8QrDrdQApSnKvP+86q65qRU0uxRGwSQISwT5zP1+ST3IgmN2lEmEKbyFuhW41az0qY9j7gl3CCLNPpLWW3X7+GMrITt/1nifOQjqbUpH2BrZoQ336IAMe9uRXtqPjHsjWPXrQ9eRrEhgIlB576jd9aKcgsEgElL80gRE4AWxozo+hgITS69OeUI3udnQ8wCNoBYskplfJJupo0PdrGvVo4HRz6efbBfnZIF6gHHH8+fb/JP+ruNfNSnMPNumjUE93q7nrfjX0yusTE3jny51um1rUtvflCCxFsxFcJsgbe6HTltake6CMaxhw+Ux3G/GFct/K46TAFHj8c9T8K+bwu7pCKBNvXZIoY6UBTAAA2MouXJw2+F2AtMAyM9KmyxVUQXLc/aYS3c3/x1W7BWSepBF+nk/hWN1kiofIlf79fSjME8tpFxUckuZu8cxO/UNQC6qEYByZigwoBR7Mbdii25kK9/ZhKL0bRJ/FmUNuo5vNKNekuEKclJtXR1Wi3V8Ts/WF1+unrjVSUfjachR25F5CDvvKZzW3ZyREQH+mBnrq82oB9pBaG2VyrW4B73fSKt+odOOrzKo9oBkX7oWeh04NMWTQmeHINbOAQeBhEYQ43gCg3Wwh2fy6BMrqNnHADLjHrNIjs69Vep7E7VK220xIZFRM+OCVyRH5H3HsPM5S21ui9Vi5/1bFtFlQ4dGzTkJ4TB3cCqx2qH2PTJGQxQ14Sl1ox3egSQdakPMyvegm/QHhZkJY3kml8I23trTlvfW/I/xWWvpkKpkAAM7o208XzCZW7Y/17nc2jlAFHKXiWT7um7JQVfE3iDhjE7xGYBWcRVocdzdio/NW/qJB9ov5BcViLt/AfzPlajWQcqmifcUNCOvm3bNiZRIKq/TPR2ziiiXtiQ4Kxn7tUkTEgoX1C3VIGCpMapEwkkCt2Y9CC5CpFgeK72IVeWJYzgSJCtPC/pC+TzmWUC8a7e2QH1BBYM3oivRKDROv+IRB/ku2edwEN2HjGHdKklbE568tk8OpXOIarxcR/fOwcCYzHgf1vyleBfaDN4pbM0WBJxv3c+nk3/tQS744VFYtW2Wybiv4LyZBb2nj7htqnY/mw7FM32abqtCTZdYMlHqM+97Z5pM2Tpq5c6jCqZ2vfRX3N6LSMWmudWsEJxPFSxio7O6bUvkgcq6V0p/r8o+eOhTUzw9VRpQzRaisPYYIZllEmlrIgKcPifCuW3v0vLbZxkBZx70uGxK6F0ZW6a2peLaT0ARAMhq7HmI/obP3fS90kpXtlkqRwM7U7rN03Dl9SYLTCgkTQQoBdSXas3c63oENMA1L9C5o4/5HTrfUFFzwJBstSOdefVRxALOYcrmBRKnkyqCm4MbRS7Kqe0xOg3VISfIRyMMRaV2CKiMmAJE94IgnLIacWGPcJ/2XfKaoJuhoXRf7GkG5GD1UbH7Tu91pxIBzW1oThb+B+XNjMyMmzza+/WHa1UN1po/XVKIGdMUcG3BelyXCiihh2HIrrdFb3RnFAhzlIQ92d8fW1RN82Q4kPCBW7YvYFkLlA7nqr0GML4PXXWJRWbJkdxNcMbHzY2IW1H5mBwYXLCksIHwCCX8ZS0ekEejqoWylJte++V+pAfn1BO+BsoMWam9he2t2qLa36qkI5d3v8Qfh5MS67emmmXaGQk94TAJvg0BA5+IXEehztQ5flAygIsIyCCT47UdCNC9JM8/1a++cV2F/OBXgTS4hVhGqO+OlSKUL3HhZxfe3XJAlYGtkWxcTGAN5p2XBli35eW+IVGQYT+qMI6sDCiw5scaZKZlzZf1huHzCO1s55/LEx8TINHixEA4rC5VHnzIh2eJY05hZkSpf1g1rBC+H6YlzN2BcisfePciAuQ7+/9EA5zRsmw6VBJboRziNGi6gAA/FotrXCbosP/OqZJox+zAGnOfAre/2HfYiOsI+GnG4gkoQfp5LTplq+0dRXw6hYNbUR8qNdTwC2M7FS5210tU2GF44ZctJpzjRsno5+U8cKwFTz+lN3XcGxbciy+6pizY+mY2yY7/TDQDmrJ6p6kmQcMRU44UpvV7WHpoMGa/Fd5Pk/sbgUXaOPGeYYf+adMu2J40IYbTNhmPW1iw8e5o=,iv:BhzkNDko9jS64JhDMjlyAe2o4M7xRZvVRK9w5Kb7UxA=,tag:HNniTwaqQZKJ2k22lLxhGw==,type:str] +sops: + age: + - enc: | + -----BEGIN AGE ENCRYPTED FILE----- + YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA5VEhjYjc5SWNobm9BV0VR + bFU5Y2FhMS9GL2NFRWFBbFVKUFhDdnVVYlg4CjIwTEF2T1U4SHpjd3pjMmxHSzZY + SE80NXA1M0R2MFVPNTZiMi9ZcXZ5YU0KLS0tIHc3MlFQTGZiZVBLTnM1eklUamly + WmFiVFMrMVBRcGprVmlHQURQbDRGa1UKCrO5Fjn4EmOF2+b5kPGlJPaB93oFTJxY + nXPniqdN+jsDc5Z8uoCpGDybOtJn67wiwKZ1l7fniIVu3gAp7t4g1g== + -----END AGE ENCRYPTED FILE----- + recipient: age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + lastmodified: "2026-09-23T07:12:51Z" + mac: ENC[AES256_GCM,data:biR2CcelZsxkSaIIoDg1fnTMgaVfT3V8xE7SlbP/EabudvCPBOLXmuXbJngTLtUINV8WTDaTjqcWFOllCM3sywYXPfLkPg9auKos1t3A1a+825s/dnQMERN6UbCHUhPs9d/w1Uxnw4dbzCTVQY6i4hAqTZip47eVDJw/GZzJOW8=,iv:5qhp9bssLL2VTTBqelwsvBNgFuyLNOp/bfg/8PMpYIo=,tag:BZ6gGQjgWhRdLeYUSkuRlg==,type:str] + version: 3.13.3 diff --git a/evals/questions.example.yaml b/evals/questions.example.yaml new file mode 100644 index 0000000..2eea3ea --- /dev/null +++ b/evals/questions.example.yaml @@ -0,0 +1,27 @@ +# Question set for `python -m slack_index.evals`. Copy to questions.yaml and fill +# it with questions whose answers you already know; that file stays untracked +# because its labels quote channel content. +# +# - question: what someone would actually type, NOT the wording of the message +# expected: [source_id, ...] # thread_ts or file id; any hit in top-k counts +# tags: [category, ...] # scored per tag, so weak spots stay visible +# +# Paraphrase away from the message text: a question copied from the message lets +# lexical retrieval win for free and stops measuring anything. +# +# Categories in use: +# concept — about content, worded differently from the message +# lexical — hinges on an exact identifier (part number, hostname, error string) +# decision — the answer is an agreement or conclusion, often a short ack +# temporal — scheduling, or what happened when +# file — the answer lives in a shared file, not a message +# people — who said or owns something +- question: "did we settle on bumping that library" + expected: ["1700000000.000100"] + tags: [decision] +- question: "what was the env var the deploy script reads" + expected: ["1700000100.000200"] + tags: [lexical] +- question: "the design doc shared last week" + expected: ["F00EXAMPLE1", "1700000200.000300"] + tags: [file] diff --git a/nix/devshell.nix b/nix/devshell.nix index 5c41f5d..5a43eef 100644 --- a/nix/devshell.nix +++ b/nix/devshell.nix @@ -13,6 +13,7 @@ let git lmdb ruff + sops uv ]; env = { diff --git a/nix/treefmt.nix b/nix/treefmt.nix index bd1ee13..a95986d 100644 --- a/nix/treefmt.nix +++ b/nix/treefmt.nix @@ -24,5 +24,9 @@ settings.formatter.ruff-check.priority = 1; settings.formatter.ruff-format.priority = 2; - settings.global.excludes = [ "**/.direnv/**" ]; + settings.global.excludes = [ + "**/.direnv/**" + "secrets.yaml" + "**/*.enc.yaml" + ]; } diff --git a/pyproject.toml b/pyproject.toml index 2b2336a..27a9d46 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,13 +5,20 @@ description = "CocoIndex playground" requires-python = ">=3.14" dependencies = [ "aiohttp>=3.14.3", + "anthropic>=1.8.0", "cocoindex[lancedb,sentence-transformers]>=1.0.24", "numpy>=2.5.3", + "pyyaml>=6.0.3", "slack-sdk>=3.44.1", ] [dependency-groups] -dev = ["mypy>=1.14", "ipykernel>=7.3.0", "pytest>=9.1.1"] +dev = [ + "mypy>=1.14", + "ipykernel>=7.3.0", + "pytest>=9.1.1", + "types-pyyaml>=6.0.12.20260906", +] [tool.uv] package = false diff --git a/secrets.yaml b/secrets.yaml new file mode 100644 index 0000000..3f3d836 --- /dev/null +++ b/secrets.yaml @@ -0,0 +1,31 @@ +#ENC[AES256_GCM,data:4nxLgDN+tVSdCkWVbnmeMI0Ro9eo/lSuKsxe6434HNEUnP/Yv6IDw8JWZbrvblc5sZWgnUqI4CStdJet2PXZa+HpfUCx6Sd5WA==,iv:d+1bGA71iMcYFtv5NDOT7Q/k9Vy1+LyOpDuUj6b2KjU=,tag:jCRYG0Wun7yPBaHBi+rlhw==,type:comment] +#ENC[AES256_GCM,data:VpYCLgmUnaBx5AaY7Tj2E9Lh0umhGdUZxb9yZq+9n3pQo8AidJdYUK5KIB+V2AtdsHjOe/gGwso=,iv:fMfx7xlU92GE1NDR2aPmy/Mdbr0ItgMwUxPlRzU6K3s=,tag:BBWoyREI+8kLPCR245iVnw==,type:comment] +SLACK_BOT_TOKEN: ENC[AES256_GCM,data:cVNAnDTK1hQamns/dPXY/zodL04dOvb+O7nW/eahdQ0Onf4TWvDmUe3rWVjguiK4KQ8o++v6VV6Vfw==,iv:uu/Ujpc7IObUJu/Xuu2tcl1P8EQNybKIgGEGTdH2Klw=,tag:/cc0FdN0Gp8EPcuR3XG7dQ==,type:str] +#ENC[AES256_GCM,data:6bfENRrTG/d/4bd6/9AolE3fz6J/1RXqsxHaHFFhgvfLWcm6i8V8Vle9VLdzzmsGUfSMJS0E,iv:NUwJZuwfNaE9CaLcpSTPSyXj2yp6lp6ssKa26y8KLjI=,tag:c0QWWKxvI4BBhkBV896YHA==,type:comment] +SLACK_CHANNEL_IDS: ENC[AES256_GCM,data:0vGH9saX+EGtCCs=,iv:gZ4bIMQr8OeANw8nMat79A4Uy245+TL8sqUf0jGfVuA=,tag:N654rfTrZJltvVs01C5dsw==,type:str] +#ENC[AES256_GCM,data:/87rtbLj3SQpP+kasbwH05zqo30SHHaPtaEcG4eSKQgl9hTW7w9288bnq0pnThP50BnjYORMajyzugxWW+5Ru0zjUQ==,iv:4SIzwgrLP0PzAveX/C5PMbFbLwGNvQ8eb8U6tC3JJZ8=,tag:b7zCQ9NgVUz88FvYQhH15g==,type:comment] +ANTHROPIC_API_KEY: ENC[AES256_GCM,data:LjEV90smqXsLU1HOGD3YShEEmGalXj5cS3A1DHGRingUZH0SKox7ceeOgXWME8ztrhwOOe7BUNmSS1wUiTOgQa9fD7ILhoADzCw6gMcGGAYunC8hrWU2JHhNnSYih1qbjOczssAkf38y6BhW,iv:Uf+MGNzy/PdTus4IloN5tH7ldl8P7VOmZoFU+fnnj6s=,tag:HB8csr+7I2x+9Uk3tK+xiA==,type:str] +#ENC[AES256_GCM,data:DZfAI37opf/HUr0f6dRn/chYXPZAFPyd0AcjXwEUEcsYxTFDIgJgvCFqCl1cfqYUmgUxNHCB+4dhbc1eazsvLneUFyJvemulhk0scg==,iv:P6+Ek/oAJN20IllyUoJ1WGJ6E7jA1Nq0NbmWs3DLlpI=,tag:8Y8VDnwWbEgXYRkTwA/PYw==,type:comment] +#ENC[AES256_GCM,data:HoSUNGy3R1zwmp2LvTHKM37P6U5aRe69qzqCtURXT6BAt1F/pCZrvIMyQXU=,iv:ZRziCRWcA+Hca5I7KY+ov0iBkai8UZXd8gjdPDZ1T/Q=,tag:rnnx7Chm2IRkIshKAWGd6g==,type:comment] +#ENC[AES256_GCM,data:cI58KLZYzOIdZk+eFv9m/cbU1JJg2GOfAYedtkw+jNKbDhT+XV/AuhTh,iv:8NsIc4v4GB9qEBDHycmhu3j1zNSNVpSmtfDx0YbgfIs=,tag:DrCwC/do5WHbijZoztVttw==,type:comment] +#ENC[AES256_GCM,data:0dOad7gf7q4AvBUIsVtvHR3Hx/9x8W9yQjXRKwO5J0bWVskKgiiDW971hA==,iv:mxMy+eo4DmqfoRN8I0ueMa82krSw5Kp+Fgkh3Evqf20=,tag:TlkZgkqY1BO3LQxTW701pA==,type:comment] +#ENC[AES256_GCM,data:aCZNjkgqSQkX3WqYfPFcupWK5O6y+EXMJIbrxUoE,iv:4z+NqPAMJuhfylnubb5VDBg4pY4qV17O/EPWJPKU91o=,tag:7FDJsU1HF8clOQ0KMy+c1g==,type:comment] +#ENC[AES256_GCM,data:J4krqZs26uZbYk9+cL/RbakKUjPMqAIQBsiQKhsY68lDXt9rM9k7znlNpQTxSzCrwWj15wv1GA==,iv:1J13sn/q0HZu4uptsyQ/jCLDKB9SsxyJPIKEtJmNdbg=,tag:24MiA+B0JW/4qq5AhO/btQ==,type:comment] +#ENC[AES256_GCM,data:FqccKI0ZTaiYx/KWfp+J1l3i0EY5bKa08bHOV1Ut,iv:h8eBY6IRPJ4QfwM+k4jYEHxgTAbnEhBUZ20aN043MQQ=,tag:am/VRBvo7YO/CmTVdbB7Ww==,type:comment] +#ENC[AES256_GCM,data:zLaA23xAd1sVwBsVXD+Dun1EX6bJbohsVXctjHHa,iv:7+p50PqD8qbz9RhuqcINFGZMHCqXeu4wC0ixY3SACE4=,tag:4YuSzP1jV1/DeDYHFFZiFw==,type:comment] +#ENC[AES256_GCM,data:5VxRntT1v/dxvxPx49mhTZqjeYOpP/yAuxLilRQDsA2BHk6sdw==,iv:z8jxiqsBwvJATEBNX5jttIHIsKr/3kDJ1AMhIrAl4/Q=,tag:9Ao52YQL1WQA3pJG9mw+/Q==,type:comment] +sops: + age: + - enc: | + -----BEGIN AGE ENCRYPTED FILE----- + YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBuTGRTaDRKR0NOeTU1eEQr + NjNXZlB5bUNqR29qR2w5SjMwTjRPNUlpZkNRCmJsZWtObzcrcmx1YkRzc0ZWcjdr + ZkNOcXNON2IxdmtQaUtGVFEyM2dCK1EKLS0tIHFvajY3dVBzNEJBYW9oTkF4TnU4 + di9rd0FJTGV1TUJPSHRpU0NHancwZFEK48gfvnkzQoPgD0kQGrwxzYFp+Agp4k7K + C1vBUNEgRXrv0zpnNL7CChXKAb2zzfgGN9x1NOkSQ8Yd7NGjz1cfJg== + -----END AGE ENCRYPTED FILE----- + recipient: age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + lastmodified: "2026-09-23T06:33:35Z" + mac: ENC[AES256_GCM,data:PoO2ZUovKek6lSkQhgkofnn4CEoyAL6q3knv1wjI461JaxJ9NYSmdKu9jYlyQy/pC3OzVKtsNLWG8n5Mk98H51gY9NtVFLLHLlrIqyEGMPvi25wfKql2ft4AJVlTd+U8qTcDBb9Dv6rLgMggFl9F4YuVN0lb4V25+AnC2EFiDdA=,iv:B3hOWQh1IKPx0xLUK5usF1Nqu5OJzraTklzukCa3zZs=,tag:idVox70tlxxGQVqcyNThDw==,type:str] + unencrypted_suffix: _unencrypted + version: 3.13.3 diff --git a/secrets.yaml.example b/secrets.yaml.example new file mode 100644 index 0000000..7fd9013 --- /dev/null +++ b/secrets.yaml.example @@ -0,0 +1,28 @@ +# Copy to secrets.yaml and encrypt it: `sops -e -i secrets.yaml`. +# Afterwards edit in place with `sops secrets.yaml`; the encrypted file is +# committed, the plaintext never is. +# +# Nothing reads this file at run time. The dev shell's .envrc decrypts it into +# environment variables; in production the unit supplies the same variables from +# its own credentials. The application only ever reads the environment. + +# Bot token with channels:history, groups:history, files:read, users:read. +# The bot must be a member of every channel listed below. +SLACK_BOT_TOKEN: xoxb-replace-me + +# Comma-separated channel ids, e.g. C0123ABCD,C0456EFGH +SLACK_CHANNEL_IDS: "" + +# Distillation. An org-scoped key also needs ANTHROPIC_WORKSPACE_ID. +ANTHROPIC_API_KEY: sk-ant-replace-me +ANTHROPIC_WORKSPACE_ID: "" + +# Optional overrides; the committed defaults in slack_index/config.py are the +# source of truth, these are for experiments. +#SLACK_INDEX_EMBED_MODEL: nlpai-lab/KURE-v1 +#SLACK_INDEX_DISTILL_MODEL: claude-haiku-4-5 +#SLACK_INDEX_RERANK_DEVICE: mps +# Unset or 0 indexes the channel from its first message. +#SLACK_INDEX_LOOKBACK_DAYS: "0" +#SLACK_INDEX_POLL_SECONDS: "60" +#SLACK_INDEX_MAX_FILE_BYTES: "5242880" diff --git a/slack_index/__init__.py b/slack_index/__init__.py new file mode 100644 index 0000000..065b126 --- /dev/null +++ b/slack_index/__init__.py @@ -0,0 +1 @@ +"""Index Slack channel conversations and shared files into LanceDB.""" diff --git a/slack_index/app.py b/slack_index/app.py new file mode 100644 index 0000000..ad39384 --- /dev/null +++ b/slack_index/app.py @@ -0,0 +1,90 @@ +"""Pipeline entry point. + +cocoindex update slack_index/app.py # one-shot catch-up +cocoindex update -L slack_index/app.py # live: re-scan every poll interval +""" + +from __future__ import annotations + +from collections.abc import AsyncIterator + +import cocoindex as coco +from anthropic import AsyncAnthropic +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +from slack_index import config +from slack_index.context import DISTILLER, EMBEDDER, LANCE_DB, SLACK, SLACK_LIMIT +from slack_index.distill import Distiller +from slack_index.files import process_file +from slack_index.models import SlackChunk +from slack_index.source import SlackChannelFiles, SlackChannelThreads +from slack_index.threads import process_thread + +_settings = config.Settings.from_env() + + +@coco.lifespan +async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None]: + config.VAR_DIR.mkdir(parents=True, exist_ok=True) + builder.settings.db_path = config.DB_PATH + builder.provide(SLACK, AsyncWebClient(token=config.bot_token())) + builder.provide( + DISTILLER, + Distiller( + AsyncAnthropic( + api_key=config.anthropic_api_key(), + default_headers=config.anthropic_headers(), + ), + config.DISTILL_MODEL, + ), + ) + builder.provide(SLACK_LIMIT, RateLimiter(config.SLACK_REQUESTS_PER_SECOND)) + builder.provide(EMBEDDER, SentenceTransformerEmbedder(_settings.embed_model)) + builder.provide(LANCE_DB, await lancedb.connect_async(str(config.LANCEDB_URI))) + yield + + +@coco.fn +async def app_main(settings: config.Settings) -> None: + table = await lancedb.mount_table_target( + LANCE_DB, + config.TABLE_NAME, + await lancedb.TableSchema.from_class(SlackChunk, primary_key=["id"]), + ) + table.declare_vector_index(column="embedding") + + client = coco.use_context(SLACK) + limiter = coco.use_context(SLACK_LIMIT) + for channel in settings.channel_ids: + threads = SlackChannelThreads( + client, + limiter, + channel, + lookback=settings.lookback, + poll_interval=settings.poll_interval, + ) + files = SlackChannelFiles( + client, + limiter, + channel, + lookback=settings.lookback, + poll_interval=settings.poll_interval, + ) + # The channel is part of the subpath so each channel keeps its own + # component subtree — and its own rows — across runs. + await coco.mount_each( + coco.ComponentSubpath("threads", channel), process_thread, threads, table + ) + await coco.mount_each( + coco.ComponentSubpath("files", channel), + process_file, + files, + table, + settings.max_file_bytes, + ) + + +app = coco.App(coco.AppConfig(name="SlackIndex"), app_main, settings=_settings) diff --git a/slack_index/chunking.py b/slack_index/chunking.py new file mode 100644 index 0000000..e0b882a --- /dev/null +++ b/slack_index/chunking.py @@ -0,0 +1,71 @@ +"""Text -> chunks -> embedded rows, shared by the thread and file pipelines.""" + +from __future__ import annotations + +import datetime +from dataclasses import dataclass + +import cocoindex as coco +from cocoindex.connectors import lancedb +from cocoindex.ops.text import RecursiveSplitter +from cocoindex.resources.chunk import Chunk +from cocoindex.resources.id import IdGenerator + +from slack_index.config import CHUNK_OVERLAP, CHUNK_SIZE +from slack_index.context import EMBEDDER +from slack_index.models import SlackChunk + +_splitter = RecursiveSplitter() + + +@dataclass(frozen=True, slots=True) +class ChunkMeta: + """Row fields every chunk of one source shares.""" + + kind: str + channel: str + source_id: str + # Every source-side id this document answers for — a window covers several + # message timestamps, so a lookup by any one of them must find it. + covered: str + permalink: str + author: str + posted_at: datetime.datetime + + +@coco.fn +async def _declare_chunk( + chunk: Chunk, + meta: ChunkMeta, + id_gen: IdGenerator, + table: lancedb.TableTarget[SlackChunk], +) -> None: + table.declare_row( + row=SlackChunk( + id=await id_gen.next_id(chunk.text), + kind=meta.kind, + channel=meta.channel, + source_id=meta.source_id, + covered=meta.covered, + permalink=meta.permalink, + author=meta.author, + posted_at=meta.posted_at, + text=chunk.text, + embedding=await coco.use_context(EMBEDDER).embed(chunk.text), + ), + ) + + +async def declare_chunks( + text: str, + meta: ChunkMeta, + table: lancedb.TableTarget[SlackChunk], +) -> None: + chunks = _splitter.split( + text, + chunk_size=CHUNK_SIZE, + chunk_overlap=CHUNK_OVERLAP, + language="markdown", + ) + id_gen = IdGenerator() + await coco.map(_declare_chunk, chunks, meta, id_gen, table) diff --git a/slack_index/config.py b/slack_index/config.py new file mode 100644 index 0000000..0828581 --- /dev/null +++ b/slack_index/config.py @@ -0,0 +1,122 @@ +"""Runtime configuration, resolved from the environment.""" + +from __future__ import annotations + +import datetime +import os +import pathlib +from dataclasses import dataclass + +# Everything mutable the pipeline produces (engine state, vector store, downloaded +# attachments) lives under one directory, so a reset is a single `rm -rf`. +_REPO_ROOT = pathlib.Path(__file__).resolve().parent.parent +VAR_DIR = pathlib.Path(os.environ.get("SLACK_INDEX_VAR_DIR", _REPO_ROOT / "var")) + +DB_PATH = VAR_DIR / "cocoindex.db" +LANCEDB_URI = VAR_DIR / "lancedb" + +TABLE_NAME = "slack_chunks" + +# Consecutive messages inside this gap belong to the same conversation. Caps stop a +# busy afternoon from collapsing into one undifferentiated document. +WINDOW_GAP = datetime.timedelta(minutes=15) +WINDOW_MAX_MESSAGES = 20 +WINDOW_MAX_CHARS = 2000 +CHUNK_SIZE = 1200 +CHUNK_OVERLAP = 200 + +# Korean-specialised retrieval model; an English-only model scored MRR 0.16 here +# against this one's 0.79. +EMBED_MODEL = "nlpai-lab/KURE-v1" + +# Cross-encoder over the shortlist. Korean-capable, same family as the embedder. +RERANK_MODEL = "BAAI/bge-reranker-v2-m3" +# How many distinct sources the retriever hands the reranker. Measured: 10 and 20 +# score identically while 10 is 2.4x faster, and 30-50 start costing accuracy — +# extra candidates are extra distractors. The pool never shrinks below the +# requested result count. +RERANK_CANDIDATES = 10 + +# Bulk extraction over short conversations: the cheapest current model is enough. +DISTILL_MODEL = "claude-haiku-4-5" +DISTILL_MAX_TOKENS = 1024 +DISTILL_TEMPERATURE = 0.0 + +# conversations.* and files.* are Slack tier 3 methods: 50+ requests per minute. +SLACK_REQUESTS_PER_SECOND = 50 / 60 + + +@dataclass(frozen=True, slots=True) +class Settings: + channel_ids: tuple[str, ...] + embed_model: str + # None indexes the channel from its first message. + lookback: datetime.timedelta | None + poll_interval: datetime.timedelta + max_file_bytes: int + + @classmethod + def from_env(cls) -> Settings: + channels = setting("SLACK_CHANNEL_IDS", "") or "" + channel_ids = tuple(c.strip() for c in channels.split(",") if c.strip()) + if not channel_ids: + raise RuntimeError( + "SLACK_CHANNEL_IDS is empty: set it to a comma-separated list of channel " + "ids, e.g. C0123ABCD,C0456EFGH" + ) + return cls( + channel_ids=channel_ids, + embed_model=setting("SLACK_INDEX_EMBED_MODEL", EMBED_MODEL) or EMBED_MODEL, + lookback=_lookback_from_env(), + poll_interval=datetime.timedelta( + seconds=float(setting("SLACK_INDEX_POLL_SECONDS", "60") or "60") + ), + max_file_bytes=int( + setting("SLACK_INDEX_MAX_FILE_BYTES", str(5 * 1024 * 1024)) + or str(5 * 1024 * 1024) + ), + ) + + +def setting(name: str, default: str | None = None) -> str | None: + """Configuration arrives as environment variables — from sops through direnv in + a dev shell, from the unit's credentials in production. + + An empty value counts as unset: a stray .env, which the cocoindex CLI auto-loads + from the first one it finds upwards, would otherwise blank out a credential that + the environment actually carries. + """ + return os.environ.get(name) or default + + +def _lookback_from_env() -> datetime.timedelta | None: + """Unset or 0 means the whole channel history.""" + days = float(setting("SLACK_INDEX_LOOKBACK_DAYS", "0") or "0") + return datetime.timedelta(days=days) if days > 0 else None + + +def anthropic_api_key() -> str: + """Passed explicitly: the SDK reads only the environment, and the key lives in + the secrets file.""" + key = setting("ANTHROPIC_API_KEY") + if not key: + raise RuntimeError( + "ANTHROPIC_API_KEY is not set; the dev shell loads it from secrets.yaml" + ) + return key + + +def anthropic_headers() -> dict[str, str]: + """An org-scoped API key must name the workspace on every request; a + workspace-scoped key needs nothing.""" + workspace = setting("ANTHROPIC_WORKSPACE_ID") + return {"anthropic-workspace-id": workspace} if workspace else {} + + +def bot_token() -> str: + token = setting("SLACK_BOT_TOKEN") + if not token: + raise RuntimeError( + "SLACK_BOT_TOKEN is not set in secrets.yaml or the environment" + ) + return token diff --git a/slack_index/context.py b/slack_index/context.py new file mode 100644 index 0000000..1aa9336 --- /dev/null +++ b/slack_index/context.py @@ -0,0 +1,24 @@ +"""Context keys shared by every component in the pipeline.""" + +from __future__ import annotations + +import typing as _typing + +import cocoindex as coco +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +if _typing.TYPE_CHECKING: + from slack_index import distill + +SLACK = coco.ContextKey[AsyncWebClient]("slack") +# detect_change: a different distillation model must re-distill everything. +DISTILLER: coco.ContextKey[distill.Distiller] = coco.ContextKey( + "distiller", detect_change=True +) +SLACK_LIMIT = coco.ContextKey[RateLimiter]("slack_rate_limit") +# detect_change: swapping the embedding model must re-embed everything. +EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("embedder", detect_change=True) +LANCE_DB = coco.ContextKey[lancedb.LanceAsyncConnection]("lancedb") diff --git a/slack_index/distill.py b/slack_index/distill.py new file mode 100644 index 0000000..e5df83b --- /dev/null +++ b/slack_index/distill.py @@ -0,0 +1,107 @@ +"""LLM distillation: a short header that says what a conversation was about. + +Cerebras' knowledge base replaces the raw transcript with an LLM-normalised +question/summary/resolution and keeps the raw text for full-text search only. We +have no full-text index yet, so the distilled header is prepended to the raw +transcript instead of replacing it: the summary supplies the wording a searcher +would use, the transcript keeps the exact strings (catalog numbers, strain names) +that a summary always drops. +""" + +from __future__ import annotations + +import json + +import cocoindex as coco +from anthropic import AsyncAnthropic +from pydantic import BaseModel, Field + +from slack_index.config import DISTILL_MAX_TOKENS, DISTILL_TEMPERATURE +from slack_index.context import DISTILLER + +_SYSTEM = """You summarise chat from a molecular biology lab's Slack channel. +The chat is Korean and informal; the science terms are not. Write in Korean. + +Copy identifiers exactly as they appear — protein and strain names, mutations, +catalog numbers, plasmids, dates, quantities. Never translate or normalise them. +Never add a fact that is not in the transcript.""" + +_INSTRUCTION = """Normalise this conversation for a search index. + +question: one line — the question this conversation answers, or what it records. +summary: one or two sentences on what was discussed. +resolution: what was decided. Leave it empty when nothing was. +keywords: 3-8 proper nouns and identifiers someone would search by. + +Conversation: +""" + + +class Distiller: + """Owns the client and the model id. + + The model id is the memo key: swapping models must re-distill everything, + while a new client object for the same model must not. + """ + + def __init__(self, client: AsyncAnthropic, model: str) -> None: + self._client = client + self._model = model + + def __coco_memo_key__(self) -> object: + return self._model + + async def run(self, transcript: str) -> Distilled: + response = await self._client.messages.parse( + model=self._model, + max_tokens=DISTILL_MAX_TOKENS, + system=_SYSTEM, + messages=[{"role": "user", "content": _INSTRUCTION + transcript}], + output_format=Distilled, + # `parse()` takes no sampling arguments; temperature rides along in the + # body so a re-distill does not reword an unchanged conversation. + extra_body={"temperature": DISTILL_TEMPERATURE}, + ) + parsed = response.parsed_output + if parsed is None: + raise RuntimeError( + f"distillation returned no structured output: {response.stop_reason}" + ) + return parsed + + +class Distilled(BaseModel): + question: str = Field( + description="One line: the question this conversation answers" + ) + summary: str = Field(description="One or two sentences on what was discussed") + resolution: str = Field(description="What was decided; empty if still open") + keywords: list[str] = Field(description="Identifiers and proper nouns to search by") + + +def render(distilled: Distilled) -> str: + lines = [f"[Q] {distilled.question}", f"[Summary] {distilled.summary}"] + # An open question should not be indexed as if it had an answer. + if distilled.resolution.strip(): + lines.append(f"[Resolution] {distilled.resolution}") + lines.append(f"[Keywords] {', '.join(distilled.keywords)}") + return "\n".join(lines) + + +# The prompt and the output schema shape every distillation but live outside the +# function body, where cocoindex's logic fingerprint cannot see them. Declaring them +# as deps is what makes editing a prompt re-distill the channel instead of silently +# serving summaries written by the old one. +_PROMPT_DEPS = ( + _SYSTEM, + _INSTRUCTION, + DISTILL_MAX_TOKENS, + DISTILL_TEMPERATURE, + json.dumps(Distilled.model_json_schema(), sort_keys=True), +) + + +@coco.fn(memo=True, deps=_PROMPT_DEPS) +async def distill(transcript: str) -> str: + """Memoized on the transcript, so an unchanged conversation is never re-billed.""" + return render(await coco.use_context(DISTILLER).run(transcript)) diff --git a/slack_index/evals.py b/slack_index/evals.py new file mode 100644 index 0000000..efbd25b --- /dev/null +++ b/slack_index/evals.py @@ -0,0 +1,203 @@ +"""Score the index against a hand-labelled question set. + + python -m slack_index.evals + python -m slack_index.evals --questions evals/questions.yaml --top-k 10 + +Every retrieval change — a different embedding model, grouping, a reranker — is +judged by re-running this against the same questions. +""" + +from __future__ import annotations + +import argparse +import asyncio +import pathlib +import statistics +from collections import defaultdict +from dataclasses import dataclass, field + +import yaml + +from slack_index.search import Searcher + +DEFAULT_QUESTIONS = pathlib.Path("evals/questions.yaml") +# The labels quote channel content, so the repo carries only this encrypted copy; +# `sops -d` writes the plaintext the scorer reads. +ENCRYPTED_QUESTIONS = pathlib.Path("evals/questions.enc.yaml") +DEFAULT_KS = (1, 3, 10) + + +@dataclass(frozen=True, slots=True) +class Question: + question: str + expected: frozenset[str] + tags: tuple[str, ...] + + @property + def answerable(self) -> bool: + """An empty `expected` means the channel has no answer: the question is + there to show what the index does when nothing is relevant.""" + return bool(self.expected) + + +@dataclass +class Report: + total: int = 0 + hits: dict[int, int] = field(default_factory=dict) + reciprocal_rank_sum: float = 0.0 + misses: list[str] = field(default_factory=list) + + def recall(self, k: int) -> float: + return self.hits.get(k, 0) / self.total if self.total else 0.0 + + @property + def mrr(self) -> float: + return self.reciprocal_rank_sum / self.total if self.total else 0.0 + + +def load_questions(path: pathlib.Path) -> list[Question]: + """Read the plaintext question set, or decrypt the committed one when a fresh + checkout has no plaintext yet.""" + if not path.exists() and path == DEFAULT_QUESTIONS: + raise FileNotFoundError( + f"{path} is not there; decrypt the committed copy first:\n" + f" sops -d {ENCRYPTED_QUESTIONS} > {path}" + ) + raw = yaml.safe_load(path.read_text()) or [] + return [ + Question( + question=item["question"], + expected=frozenset(item.get("expected") or ()), + tags=tuple(item.get("tags", ())), + ) + for item in raw + ] + + +def _record( + report: Report, ks: tuple[int, ...], rank: int | None, question: str +) -> None: + report.total += 1 + if rank is None: + report.misses.append(question) + return + report.reciprocal_rank_sum += 1.0 / rank + for k in ks: + if rank <= k: + report.hits[k] = report.hits.get(k, 0) + 1 + + +@dataclass +class Scores: + """Top-1 similarity, split by whether an answer exists at all. If the two + overlap, no score cutoff can stop the index from answering confidently about + something the channel never discussed.""" + + answerable: list[float] = field(default_factory=list) + unanswerable: list[float] = field(default_factory=list) + + +async def evaluate( + searcher: Searcher, + questions: list[Question], + *, + ks: tuple[int, ...] = DEFAULT_KS, + top_k: int | None = None, + candidates: int | None = None, +) -> tuple[Report, dict[str, Report], Scores]: + limit = top_k or max(ks) + overall = Report() + per_tag: dict[str, Report] = defaultdict(Report) + scores = Scores() + + for question in questions: + hits = await searcher.search(question.question, limit, candidates=candidates) + top_score = hits[0].score if hits else 0.0 + if not question.answerable: + scores.unanswerable.append(top_score) + continue + scores.answerable.append(top_score) + rank = next( + ( + i + for i, hit in enumerate(hits, start=1) + if question.expected & ({hit.source_id} | hit.covered) + ), + None, + ) + _record(overall, ks, rank, question.question) + for tag in question.tags: + _record(per_tag[tag], ks, rank, question.question) + return overall, dict(per_tag), scores + + +def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: + cells = " ".join( + f"recall@{k}: {report.recall(k):.2f} ({report.hits.get(k, 0)}/{report.total})" + for k in ks + ) + return f"{name:<10} {cells} MRR: {report.mrr:.2f}" + + +async def run( + questions_path: pathlib.Path, + top_k: int, + *, + rerank: bool, + candidates: int | None = None, +) -> None: + questions = load_questions(questions_path) + searcher = await Searcher.open(rerank=rerank) + overall, per_tag, scores = await evaluate( + searcher, questions, top_k=top_k, candidates=candidates + ) + + print(f"rerank: {'on' if rerank else 'off'}") + print(_format("overall", overall, DEFAULT_KS)) + for tag in sorted(per_tag): + print(_format(tag, per_tag[tag], DEFAULT_KS)) + + if scores.unanswerable: + answerable = sorted(scores.answerable) + p25 = answerable[len(answerable) // 4] + print( + f"\ntop-1 score answerable: min {min(answerable):.3f}," + f" p25 {p25:.3f}, median {statistics.median(answerable):.3f}" + f" | unanswerable ({len(scores.unanswerable)}):" + f" median {statistics.median(scores.unanswerable):.3f}," + f" max {max(scores.unanswerable):.3f}" + ) + if overall.misses: + print(f"\nmissed ({len(overall.misses)}):") + for question in overall.misses: + print(f" {question}") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--questions", type=pathlib.Path, default=DEFAULT_QUESTIONS) + parser.add_argument("--top-k", type=int, default=max(DEFAULT_KS)) + parser.add_argument( + "--no-rerank", + action="store_true", + help="score the retriever alone, to measure what reranking adds", + ) + parser.add_argument( + "--candidates", + type=int, + default=None, + help="how many sources the reranker sees (default: config.RERANK_CANDIDATES)", + ) + args = parser.parse_args() + asyncio.run( + run( + args.questions, + args.top_k, + rerank=not args.no_rerank, + candidates=args.candidates, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/slack_index/files.py b/slack_index/files.py new file mode 100644 index 0000000..8279aa5 --- /dev/null +++ b/slack_index/files.py @@ -0,0 +1,83 @@ +"""One component per shared file: download, extract text, chunk, embed.""" + +from __future__ import annotations + +import datetime +import logging + +import aiohttp +import cocoindex as coco +from cocoindex.connectors import lancedb + +from slack_index.chunking import ChunkMeta, declare_chunks +from slack_index.config import bot_token +from slack_index.context import SLACK_LIMIT +from slack_index.models import FileRef, SlackChunk +from slack_index.users import display_name + +_logger = logging.getLogger(__name__) + +# Formats readable as-is. Binary documents (pdf, docx, pptx) need a converter — +# add one in `extract_text` and they flow through the rest of the pipeline unchanged. +TEXT_MIMETYPE_PREFIXES = ("text/",) +TEXT_MIMETYPES = frozenset( + { + "application/json", + "application/xml", + "application/x-ndjson", + "application/x-sh", + "application/javascript", + } +) + + +def is_text(mimetype: str) -> bool: + return mimetype.startswith(TEXT_MIMETYPE_PREFIXES) or mimetype in TEXT_MIMETYPES + + +async def download(ref: FileRef) -> bytes: + """Fetch a file's bytes. `url_private_download` needs the bot token, not the API.""" + await coco.use_context(SLACK_LIMIT).acquire() + headers = {"Authorization": f"Bearer {bot_token()}"} + async with ( + aiohttp.ClientSession(headers=headers) as session, + session.get(ref.url) as response, + ): + response.raise_for_status() + return await response.read() + + +async def extract_text(ref: FileRef, max_bytes: int) -> str | None: + """Return the file's text, or None when there is nothing indexable in it.""" + if not is_text(ref.mimetype): + _logger.info("skipping %s (%s): no extractor", ref.name, ref.mimetype) + return None + if ref.size > max_bytes: + _logger.info("skipping %s: %d bytes exceeds the limit", ref.name, ref.size) + return None + return (await download(ref)).decode("utf-8", errors="replace") + + +@coco.fn(memo=True) +async def process_file( + ref: FileRef, + table: lancedb.TableTarget[SlackChunk], + max_bytes: int, +) -> None: + text = await extract_text(ref, max_bytes) + if text is None: + return + + await declare_chunks( + f"# {ref.name}\n\n{text}", + ChunkMeta( + kind="file", + channel=ref.channel, + source_id=ref.file_id, + covered=ref.file_id, + permalink=ref.permalink, + author=await display_name(ref.user), + posted_at=datetime.datetime.fromtimestamp(ref.created, tz=datetime.UTC), + ), + table, + ) diff --git a/slack_index/models.py b/slack_index/models.py new file mode 100644 index 0000000..060a7ed --- /dev/null +++ b/slack_index/models.py @@ -0,0 +1,75 @@ +"""Source-side identities and the indexed row schema.""" + +from __future__ import annotations + +import datetime +from dataclasses import dataclass +from typing import Annotated + +from numpy.typing import NDArray + +from slack_index.context import EMBEDDER + + +@dataclass(frozen=True, slots=True) +class Message: + """A top-level message as the channel scan saw it.""" + + ts: str + user: str | None + text: str + + +@dataclass(frozen=True, slots=True) +class ConversationRef: + """One indexable conversation: a thread, or a run of consecutive messages. + + Every field takes part in change detection, so a new reply, an edit or a + neighbour joining the run re-runs this conversation and nothing else. + """ + + channel: str + start_ts: str + revision: str + reply_count: int + messages: tuple[Message, ...] + + @property + def is_thread(self) -> bool: + """Threads carry their replies in Slack, not in the scan.""" + return self.reply_count > 0 + + @property + def covered_ts(self) -> tuple[str, ...]: + return tuple(message.ts for message in self.messages) + + +@dataclass(frozen=True, slots=True) +class FileRef: + """A file shared in the channel, as listed by ``files.list``.""" + + channel: str + file_id: str + name: str + mimetype: str + size: int + created: int + url: str + permalink: str + user: str | None + + +@dataclass +class SlackChunk: + """One embedded chunk — of a conversation or of a shared file.""" + + id: int + kind: str + channel: str + source_id: str + covered: str + permalink: str + author: str + posted_at: datetime.datetime + text: str + embedding: Annotated[NDArray, EMBEDDER] diff --git a/slack_index/query.py b/slack_index/query.py new file mode 100644 index 0000000..ba4c701 --- /dev/null +++ b/slack_index/query.py @@ -0,0 +1,28 @@ +"""Search the index.""" + +from __future__ import annotations + +import asyncio +import sys + +from slack_index.search import Searcher + +TOP_K = 5 + + +async def run(query: str, *, top_k: int = TOP_K) -> None: + searcher = await Searcher.open() + for hit in await searcher.search(query, top_k): + print(f"[{hit.score:.3f}] {hit.kind} by {hit.author} — {hit.permalink}") + print(f" {hit.text[:300]}") + print("---") + + +def main() -> None: + if len(sys.argv) < 2: + sys.exit("usage: python -m slack_index.query ") + asyncio.run(run(" ".join(sys.argv[1:]))) + + +if __name__ == "__main__": + main() diff --git a/slack_index/rerank.py b/slack_index/rerank.py new file mode 100644 index 0000000..1dc5672 --- /dev/null +++ b/slack_index/rerank.py @@ -0,0 +1,67 @@ +"""Cross-encoder reranking of the candidate pool. + +The retriever embeds query and document separately, so it can only compare them +through one vector each. A cross-encoder reads the pair together and scores the +match directly: slower, so it only ever sees the shortlist the retriever produced. +""" + +from __future__ import annotations + +import asyncio +from typing import TYPE_CHECKING + +import torch +from sentence_transformers import CrossEncoder + +from slack_index import config + +if TYPE_CHECKING: + from slack_index.search import Hit + + +def default_device() -> str: + """Scoring 20 pairs per query is the slowest step in a search; on this laptop + the GPU is an order of magnitude faster than the CPU fallback.""" + override = config.setting("SLACK_INDEX_RERANK_DEVICE") + if override: + return override + if torch.backends.mps.is_available(): + return "mps" + if torch.cuda.is_available(): + return "cuda" + return "cpu" + + +class Reranker: + """Loads the cross-encoder lazily: a process that never reranks never pays for it.""" + + def __init__(self, model_name: str, device: str | None = None) -> None: + self._model_name = model_name + self._device = device or default_device() + self._model: CrossEncoder | None = None + + def __coco_memo_key__(self) -> object: + return self._model_name + + def _load(self) -> CrossEncoder: + if self._model is None: + self._model = CrossEncoder(self._model_name, device=self._device) + return self._model + + async def rank(self, query: str, hits: list[Hit], top_k: int) -> list[Hit]: + """Reorder *hits* by cross-encoder score and keep the best *top_k*. + + Reranking can only reorder what it is given — a document the retriever + never returned cannot be rescued here. + """ + if not hits: + return [] + model = self._load() + pairs = [(query, hit.text) for hit in hits] + scores = await asyncio.to_thread(model.predict, pairs) + rescored = [ + hit.with_score(float(score)) + for hit, score in zip(hits, scores, strict=True) + ] + rescored.sort(key=lambda hit: hit.score, reverse=True) + return rescored[:top_k] diff --git a/slack_index/search.py b/slack_index/search.py new file mode 100644 index 0000000..15971f3 --- /dev/null +++ b/slack_index/search.py @@ -0,0 +1,94 @@ +"""Retrieval over the built index, shared by the query CLI and the evals.""" + +from __future__ import annotations + +from dataclasses import dataclass, replace + +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from lancedb.table import AsyncTable + +from slack_index import config +from slack_index.rerank import Reranker + +# One source can own many chunks; over-fetch so that collapsing them still +# leaves top_k distinct sources. +_CANDIDATE_FACTOR = 5 + + +@dataclass(frozen=True, slots=True) +class Hit: + source_id: str + covered: frozenset[str] + kind: str + author: str + permalink: str + text: str + score: float + + def with_score(self, score: float) -> Hit: + return replace(self, score=score) + + +class Searcher: + """Holds the embedder and the open table so a run of queries pays for them once.""" + + def __init__( + self, + table: AsyncTable, + embedder: SentenceTransformerEmbedder, + reranker: Reranker | None, + ) -> None: + self._table = table + self._embedder = embedder + self._reranker = reranker + + @classmethod + async def open(cls, *, rerank: bool = True) -> Searcher: + settings = config.Settings.from_env() + conn = await lancedb.connect_async(str(config.LANCEDB_URI)) + table = await conn.open_table(config.TABLE_NAME) + return cls( + table, + SentenceTransformerEmbedder(settings.embed_model), + Reranker(config.RERANK_MODEL) if rerank else None, + ) + + async def search( + self, query: str, top_k: int, *, candidates: int | None = None + ) -> list[Hit]: + """Best chunk per source, ranked — a thread that chunked into ten pieces + should occupy one result slot, not ten. + + *candidates* sizes the shortlist handed to the reranker; it is the knob that + trades reranking latency against the recall the reranker has to work with. + """ + pool = candidates if candidates is not None else config.RERANK_CANDIDATES + wanted = max(top_k, pool) if self._reranker else top_k + vector = await self._embedder.embed(query) + request = await self._table.search(vector, vector_column_name="embedding") + rows = await request.limit(wanted * _CANDIDATE_FACTOR).to_list() + + hits: list[Hit] = [] + seen: set[str] = set() + for row in rows: + source_id = row["source_id"] + if source_id in seen: + continue + seen.add(source_id) + hits.append( + Hit( + source_id=source_id, + covered=frozenset(row["covered"].split()), + kind=row["kind"], + author=row["author"], + permalink=row["permalink"], + text=row["text"], + score=1.0 - row["_distance"], + ) + ) + if len(hits) == wanted: + break + if self._reranker is None: + return hits + return await self._reranker.rank(query, hits, top_k) diff --git a/slack_index/source.py b/slack_index/source.py new file mode 100644 index 0000000..c3ab7e7 --- /dev/null +++ b/slack_index/source.py @@ -0,0 +1,236 @@ +"""Slack sources as live keyed maps. + +Both views scan the channel through the Web API and, in live mode, re-scan on a +fixed interval. A re-scan is enough to drive deletes too: whatever the scan stops +reporting loses its component, and the rows that component declared are dropped. +""" + +from __future__ import annotations + +import asyncio +import datetime +from collections.abc import AsyncIterator +from typing import Any, Protocol + +from cocoindex.connectorkits import SingleWatcherGuard +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +from slack_index.config import WINDOW_GAP, WINDOW_MAX_CHARS, WINDOW_MAX_MESSAGES +from slack_index.models import ConversationRef, FileRef, Message + +# Joins, leaves and topic changes carry no content worth searching. +SKIP_SUBTYPES = frozenset( + {"channel_join", "channel_leave", "channel_topic", "channel_purpose", "bot_add"} +) +# Only files whose bytes Slack actually hosts can be fetched and read. +INDEXABLE_FILE_MODES = frozenset({"hosted", "snippet", "post", "space"}) + + +class _Subscriber(Protocol): + async def update_all(self) -> None: ... + async def mark_ready(self) -> None: ... + + +async def _poll( + subscriber: _Subscriber, interval: datetime.timedelta, guard: SingleWatcherGuard +) -> None: + with guard: + await subscriber.update_all() + # In catch-up mode mark_ready() ends watch() here; in live mode it returns. + await subscriber.mark_ready() + while True: + await asyncio.sleep(interval.total_seconds()) + await subscriber.update_all() + + +def oldest_ts(lookback: datetime.timedelta | None) -> str | None: + """Slack treats a missing `oldest` as "from the beginning".""" + if lookback is None: + return None + return str((datetime.datetime.now(tz=datetime.UTC) - lookback).timestamp()) + + +def next_cursor(response: Any) -> str | None: + """Slack paginates by handing back a cursor; an empty one means the last page.""" + metadata: dict[str, Any] = response.get("response_metadata", {}) + return metadata.get("next_cursor") or None + + +def group_messages( + channel: str, messages: list[Message], meta: dict[str, tuple[str, int]] +) -> list[ConversationRef]: + """Cut a channel's messages into indexable conversations. + + A message with replies is its own conversation, as Slack already grouped it. + Everything else is grouped with its neighbours: a message that is only a date, + or only "ok", is an answer rather than a document — on its own it is + unretrievable, and it means nothing without the message above it. + + `meta` maps a ts to its (revision, reply_count) from the scan. + """ + conversations: list[ConversationRef] = [] + run: list[Message] = [] + run_chars = 0 + + def flush() -> None: + nonlocal run, run_chars + if not run: + return + conversations.append( + ConversationRef( + channel=channel, + start_ts=run[0].ts, + revision=max(meta[m.ts][0] for m in run), + reply_count=0, + messages=tuple(run), + ) + ) + run = [] + run_chars = 0 + + for message in sorted(messages, key=lambda m: float(m.ts)): + revision, reply_count = meta[message.ts] + if reply_count > 0: + flush() + conversations.append( + ConversationRef( + channel=channel, + start_ts=message.ts, + revision=revision, + reply_count=reply_count, + messages=(message,), + ) + ) + continue + gap = float(message.ts) - float(run[-1].ts) if run else 0.0 + if run and ( + gap > WINDOW_GAP.total_seconds() + or len(run) >= WINDOW_MAX_MESSAGES + or run_chars + len(message.text) > WINDOW_MAX_CHARS + ): + flush() + run.append(message) + run_chars += len(message.text) + flush() + return conversations + + +class SlackChannelThreads: + """LiveMapView over a channel's conversations: key = first ts, value = `ConversationRef`.""" + + def __init__( + self, + client: AsyncWebClient, + limiter: RateLimiter, + channel: str, + *, + lookback: datetime.timedelta | None, + poll_interval: datetime.timedelta, + ) -> None: + self._client = client + self._limiter = limiter + self._channel = channel + self._lookback = lookback + self._poll_interval = poll_interval + self._guard = SingleWatcherGuard(f"SlackChannelThreads({channel})") + + async def _scan(self) -> AsyncIterator[tuple[str, ConversationRef]]: + messages: list[Message] = [] + meta: dict[str, tuple[str, int]] = {} + cursor: str | None = None + oldest = oldest_ts(self._lookback) + while True: + await self._limiter.acquire() + response = await self._client.conversations_history( + channel=self._channel, oldest=oldest, limit=200, cursor=cursor + ) + for message in response["messages"]: + if message.get("subtype") in SKIP_SUBTYPES: + continue + ts = message["ts"] + # A reply bumps latest_reply, an edit bumps edited.ts — take the + # larger so either one re-runs the conversation. + revision = max( + ts, + message.get("latest_reply", ts), + message.get("edited", {}).get("ts", ts), + ) + meta[ts] = (revision, int(message.get("reply_count", 0))) + messages.append( + Message( + ts=ts, + user=message.get("user") or message.get("bot_id"), + text=message.get("text", ""), + ) + ) + cursor = next_cursor(response) + if cursor is None: + break + + for conversation in group_messages(self._channel, messages, meta): + yield conversation.start_ts, conversation + + def __aiter__(self) -> AsyncIterator[tuple[str, ConversationRef]]: + return self._scan() + + async def watch(self, subscriber: _Subscriber) -> None: + await _poll(subscriber, self._poll_interval, self._guard) + + +class SlackChannelFiles: + """LiveMapView over a channel's shared files: key = file id, value = `FileRef`.""" + + def __init__( + self, + client: AsyncWebClient, + limiter: RateLimiter, + channel: str, + *, + lookback: datetime.timedelta | None, + poll_interval: datetime.timedelta, + ) -> None: + self._client = client + self._limiter = limiter + self._channel = channel + self._lookback = lookback + self._poll_interval = poll_interval + self._guard = SingleWatcherGuard(f"SlackChannelFiles({channel})") + + async def _scan(self) -> AsyncIterator[tuple[str, FileRef]]: + cursor: str | None = None + ts_from = oldest_ts(self._lookback) + while True: + await self._limiter.acquire() + response = await self._client.files_list( + channel=self._channel, ts_from=ts_from, limit=200, cursor=cursor + ) + for file in response["files"]: + if file.get("mode") not in INDEXABLE_FILE_MODES: + continue + url = file.get("url_private_download") or file.get("url_private") + if not url: + continue + yield ( + file["id"], + FileRef( + channel=self._channel, + file_id=file["id"], + name=file.get("name") or file.get("title") or file["id"], + mimetype=file.get("mimetype", ""), + size=int(file.get("size", 0)), + created=int(file.get("created", 0)), + url=url, + permalink=file.get("permalink", ""), + user=file.get("user"), + ), + ) + cursor = next_cursor(response) + if cursor is None: + return + + def __aiter__(self) -> AsyncIterator[tuple[str, FileRef]]: + return self._scan() + + async def watch(self, subscriber: _Subscriber) -> None: + await _poll(subscriber, self._poll_interval, self._guard) diff --git a/slack_index/threads.py b/slack_index/threads.py new file mode 100644 index 0000000..b95b6de --- /dev/null +++ b/slack_index/threads.py @@ -0,0 +1,86 @@ +"""One component per conversation: gather its messages, render, chunk, embed.""" + +from __future__ import annotations + +import datetime +from typing import Any + +import cocoindex as coco +from cocoindex.connectors import lancedb + +from slack_index.chunking import ChunkMeta, declare_chunks +from slack_index.context import SLACK, SLACK_LIMIT +from slack_index.distill import distill +from slack_index.models import ConversationRef, Message, SlackChunk +from slack_index.source import next_cursor +from slack_index.users import display_name + + +def thread_permalink(channel: str, thread_ts: str) -> str: + return f"https://slack.com/archives/{channel}/p{thread_ts.replace('.', '')}" + + +def ts_to_datetime(ts: str) -> datetime.datetime: + return datetime.datetime.fromtimestamp(float(ts), tz=datetime.UTC) + + +def speaker_id(message: dict[str, Any]) -> str | None: + return message.get("user") or message.get("bot_id") + + +async def fetch_replies(ref: ConversationRef) -> list[Message]: + client = coco.use_context(SLACK) + limiter = coco.use_context(SLACK_LIMIT) + messages: list[Message] = [] + cursor: str | None = None + while True: + await limiter.acquire() + response = await client.conversations_replies( + channel=ref.channel, ts=ref.start_ts, limit=200, cursor=cursor + ) + messages.extend( + Message(ts=m["ts"], user=speaker_id(m), text=m.get("text", "")) + for m in response["messages"] + ) + cursor = next_cursor(response) + if cursor is None: + return messages + + +async def conversation_messages(ref: ConversationRef) -> list[Message]: + """A grouped run already arrived complete in the scan; only a thread needs a call.""" + if not ref.is_thread: + return list(ref.messages) + return await fetch_replies(ref) + + +@coco.fn(memo=True) +async def process_thread( + ref: ConversationRef, + table: lancedb.TableTarget[SlackChunk], +) -> None: + messages = await conversation_messages(ref) + if not messages: + return + + lines: list[str] = [] + for message in messages: + speaker = await display_name(message.user) + lines.append(f"**{speaker}**: {message.text}") + + transcript = "\n\n".join(lines) + # The header goes in front of the transcript, not instead of it: without a + # lexical index, dropping the raw text would drop every exact identifier. + await declare_chunks( + f"{await distill(transcript)}\n\n---\n\n{transcript}", + ChunkMeta( + kind="message", + channel=ref.channel, + source_id=ref.start_ts, + covered=" ".join(m.ts for m in messages), + permalink=thread_permalink(ref.channel, ref.start_ts), + author=await display_name(messages[0].user), + posted_at=ts_to_datetime(ref.start_ts), + ), + table, + ) diff --git a/slack_index/users.py b/slack_index/users.py new file mode 100644 index 0000000..0cc1cae --- /dev/null +++ b/slack_index/users.py @@ -0,0 +1,29 @@ +"""Slack user id -> display name, resolved once per user.""" + +from __future__ import annotations + +import cocoindex as coco +from slack_sdk.errors import SlackApiError + +from slack_index.context import SLACK, SLACK_LIMIT + +UNKNOWN_AUTHOR = "unknown" + + +@coco.fn(memo=True) +async def display_name(user_id: str | None) -> str: + """Memoized so a channel full of the same handful of people costs a few calls. + + Bot and app ids are not users, so Slack rejects them — their raw id is the + best label available. + """ + if not user_id: + return UNKNOWN_AUTHOR + await coco.use_context(SLACK_LIMIT).acquire() + try: + response = await coco.use_context(SLACK).users_info(user=user_id) + except SlackApiError: + return user_id + user = response["user"] + profile = user.get("profile", {}) + return profile.get("display_name") or user.get("real_name") or user_id diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..86a07e9 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,51 @@ +"""A Slack client stub that replays recorded API responses.""" + +from __future__ import annotations + +import json +import pathlib +from typing import Any + +import pytest + +FIXTURES = pathlib.Path(__file__).parent / "fixtures" + + +def load_fixture(name: str) -> dict[str, Any]: + return json.loads((FIXTURES / f"{name}.json").read_text()) + + +class FakeSlackClient: + """Returns the queued response per method call and records the call kwargs.""" + + def __init__(self, **responses: list[dict[str, Any]]) -> None: + self._responses = {name: list(pages) for name, pages in responses.items()} + self.calls: list[tuple[str, dict[str, Any]]] = [] + + def _next(self, method: str, kwargs: dict[str, Any]) -> dict[str, Any]: + self.calls.append((method, kwargs)) + pages = self._responses[method] + if not pages: + raise AssertionError(f"{method} called more times than there are pages") + return pages.pop(0) + + async def conversations_history(self, **kwargs: Any) -> dict[str, Any]: + return self._next("conversations_history", kwargs) + + async def files_list(self, **kwargs: Any) -> dict[str, Any]: + return self._next("files_list", kwargs) + + +@pytest.fixture +def history_client() -> FakeSlackClient: + return FakeSlackClient( + conversations_history=[ + load_fixture("history_page1"), + load_fixture("history_page2"), + ] + ) + + +@pytest.fixture +def files_client() -> FakeSlackClient: + return FakeSlackClient(files_list=[load_fixture("files_list")]) diff --git a/tests/fixtures/files_list.json b/tests/fixtures/files_list.json new file mode 100644 index 0000000..b2847b4 --- /dev/null +++ b/tests/fixtures/files_list.json @@ -0,0 +1,38 @@ +{ + "ok": true, + "files": [ + { + "id": "F0TEXT", + "mode": "hosted", + "name": "runbook.md", + "title": "runbook", + "mimetype": "text/markdown", + "size": 2048, + "created": 1758470500, + "user": "U0LEAD", + "url_private_download": "https://files.slack.com/files-pri/T0-F0TEXT/download/runbook.md", + "permalink": "https://example.slack.com/files/U0LEAD/F0TEXT/runbook.md" + }, + { + "id": "F0EXTERNAL", + "mode": "external", + "name": "design.fig", + "mimetype": "application/octet-stream", + "size": 100, + "created": 1758470600, + "user": "U0DEV", + "permalink": "https://example.slack.com/files/U0DEV/F0EXTERNAL/design.fig" + }, + { + "id": "F0NOURL", + "mode": "hosted", + "name": "gone.txt", + "mimetype": "text/plain", + "size": 10, + "created": 1758470700, + "user": "U0DEV", + "permalink": "https://example.slack.com/files/U0DEV/F0NOURL/gone.txt" + } + ], + "response_metadata": { "next_cursor": "" } +} diff --git a/tests/fixtures/history_page1.json b/tests/fixtures/history_page1.json new file mode 100644 index 0000000..248d03f --- /dev/null +++ b/tests/fixtures/history_page1.json @@ -0,0 +1,30 @@ +{ + "ok": true, + "messages": [ + { + "type": "message", + "user": "U0LEAD", + "ts": "1758470400.000100", + "text": "deploy rollback runbook?", + "thread_ts": "1758470400.000100", + "reply_count": 3, + "latest_reply": "1758470999.000500" + }, + { + "type": "message", + "subtype": "channel_join", + "user": "U0NEW", + "ts": "1758470300.000100", + "text": "has joined the channel" + }, + { + "type": "message", + "user": "U0DEV", + "ts": "1758470200.000100", + "text": "staging is green", + "edited": { "user": "U0DEV", "ts": "1758470260.000000" } + } + ], + "has_more": true, + "response_metadata": { "next_cursor": "cursor-page-2" } +} diff --git a/tests/fixtures/history_page2.json b/tests/fixtures/history_page2.json new file mode 100644 index 0000000..87b16d4 --- /dev/null +++ b/tests/fixtures/history_page2.json @@ -0,0 +1,13 @@ +{ + "ok": true, + "messages": [ + { + "type": "message", + "user": "U0OPS", + "ts": "1758460000.000100", + "text": "postmortem doc is up" + } + ], + "has_more": false, + "response_metadata": { "next_cursor": "" } +} diff --git a/tests/test_config.py b/tests/test_config.py new file mode 100644 index 0000000..a4d5914 --- /dev/null +++ b/tests/test_config.py @@ -0,0 +1,32 @@ +"""Configuration comes from the environment, and an empty variable is not a value.""" + +from __future__ import annotations + +import pytest + +from slack_index import config + + +def test_a_set_variable_is_used(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("SLACK_BOT_TOKEN", "xoxb-from-env") + + assert config.setting("SLACK_BOT_TOKEN") == "xoxb-from-env" + assert config.bot_token() == "xoxb-from-env" + + +def test_an_empty_variable_falls_back_to_the_default( + monkeypatch: pytest.MonkeyPatch, +) -> None: + # A leftover .env — which the cocoindex CLI auto-loads — must not blank a value out. + monkeypatch.setenv("SLACK_INDEX_POLL_SECONDS", "") + + assert config.setting("SLACK_INDEX_POLL_SECONDS", "60") == "60" + + +def test_a_missing_token_says_where_it_comes_from( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("SLACK_BOT_TOKEN", raising=False) + + with pytest.raises(RuntimeError, match="secrets.yaml"): + config.bot_token() diff --git a/tests/test_files.py b/tests/test_files.py new file mode 100644 index 0000000..dac2a28 --- /dev/null +++ b/tests/test_files.py @@ -0,0 +1,41 @@ +"""What gets read, and what gets skipped before any download happens.""" + +from __future__ import annotations + +import asyncio + +from slack_index.files import extract_text, is_text +from slack_index.models import FileRef + +MAX_BYTES = 1024 + + +def _ref(mimetype: str, size: int) -> FileRef: + return FileRef( + channel="C0TEST", + file_id="F0TEST", + name="sample", + mimetype=mimetype, + size=size, + created=1758470500, + url="https://files.slack.com/files-pri/T0-F0TEST/download/sample", + permalink="https://example.slack.com/files/U0LEAD/F0TEST/sample", + user="U0LEAD", + ) + + +def test_is_text_covers_text_and_structured_formats() -> None: + assert is_text("text/markdown") + assert is_text("application/json") + assert not is_text("application/pdf") + assert not is_text("image/png") + + +def test_binary_file_is_skipped_without_downloading() -> None: + assert asyncio.run(extract_text(_ref("application/pdf", 10), MAX_BYTES)) is None + + +def test_oversized_file_is_skipped_without_downloading() -> None: + assert ( + asyncio.run(extract_text(_ref("text/plain", MAX_BYTES + 1), MAX_BYTES)) is None + ) diff --git a/tests/test_grouping.py b/tests/test_grouping.py new file mode 100644 index 0000000..bc93065 --- /dev/null +++ b/tests/test_grouping.py @@ -0,0 +1,69 @@ +"""A message that only makes sense next to its neighbour must be indexed with it.""" + +from __future__ import annotations + +from slack_index.config import WINDOW_MAX_CHARS, WINDOW_MAX_MESSAGES +from slack_index.models import Message +from slack_index.source import group_messages + +CHANNEL = "C0TEST" +BASE = 1758470000.0 + + +def _run(*specs: tuple[float, str, int]) -> list: + """Each spec is (seconds after BASE, text, reply_count).""" + messages = [] + meta = {} + for offset, text, replies in specs: + ts = f"{BASE + offset:.6f}" + messages.append(Message(ts=ts, user="U0", text=text)) + meta[ts] = (ts, replies) + return group_messages(CHANNEL, messages, meta) + + +def test_question_and_its_answer_land_in_one_conversation() -> None: + conversations = _run((0, "when was the meeting?", 0), (20, "the 27th", 0)) + + assert len(conversations) == 1 + assert [m.text for m in conversations[0].messages] == [ + "when was the meeting?", + "the 27th", + ] + # The answer's own ts stays findable even though the key is the first message. + assert conversations[0].start_ts == f"{BASE:.6f}" + assert f"{BASE + 20:.6f}" in conversations[0].covered_ts + + +def test_a_long_silence_starts_a_new_conversation() -> None: + conversations = _run((0, "first", 0), (3600, "much later", 0)) + + assert [c.start_ts for c in conversations] == [f"{BASE:.6f}", f"{BASE + 3600:.6f}"] + + +def test_a_thread_stays_on_its_own_and_breaks_the_run() -> None: + conversations = _run((0, "before", 0), (10, "thread parent", 2), (20, "after", 0)) + + assert [(c.start_ts, c.is_thread, len(c.messages)) for c in conversations] == [ + (f"{BASE:.6f}", False, 1), + (f"{BASE + 10:.6f}", True, 1), + (f"{BASE + 20:.6f}", False, 1), + ] + + +def test_a_busy_stretch_is_capped() -> None: + conversations = _run(*[(i, "short", 0) for i in range(WINDOW_MAX_MESSAGES + 5)]) + + assert len(conversations) == 2 + assert len(conversations[0].messages) == WINDOW_MAX_MESSAGES + + +def test_a_long_message_does_not_swallow_the_next_one() -> None: + conversations = _run((0, "x" * WINDOW_MAX_CHARS, 0), (10, "next", 0)) + + assert [len(c.messages) for c in conversations] == [1, 1] + + +def test_revision_is_the_latest_within_the_conversation() -> None: + conversations = _run((0, "first", 0), (30, "later", 0)) + + assert conversations[0].revision == f"{BASE + 30:.6f}" diff --git a/tests/test_source.py b/tests/test_source.py new file mode 100644 index 0000000..4289ceb --- /dev/null +++ b/tests/test_source.py @@ -0,0 +1,103 @@ +"""The channel scan decides what gets indexed, dropped, and grouped together.""" + +from __future__ import annotations + +import asyncio +import datetime +from typing import Any, TypeVar + +from cocoindex.resources.rate_limit import RateLimiter + +from slack_index.models import FileRef +from slack_index.source import SlackChannelFiles, SlackChannelThreads, oldest_ts +from tests.conftest import FakeSlackClient + +CHANNEL = "C0TEST" +T = TypeVar("T") + + +def _limiter() -> RateLimiter: + # Fast enough that the tests never actually wait on it. + return RateLimiter(10_000) + + +def _collect(source: Any) -> list[tuple[str, Any]]: + async def run() -> list[tuple[str, Any]]: + return [item async for item in source] + + return asyncio.run(run()) + + +def _threads(client: FakeSlackClient) -> SlackChannelThreads: + return SlackChannelThreads( + client, # type: ignore[arg-type] + _limiter(), + CHANNEL, + lookback=datetime.timedelta(days=30), + poll_interval=datetime.timedelta(seconds=60), + ) + + +def _files(client: FakeSlackClient) -> SlackChannelFiles: + return SlackChannelFiles( + client, # type: ignore[arg-type] + _limiter(), + CHANNEL, + lookback=datetime.timedelta(days=30), + poll_interval=datetime.timedelta(seconds=60), + ) + + +def test_scan_paginates_and_skips_noise(history_client: FakeSlackClient) -> None: + items = _collect(_threads(history_client)) + + # Chronological, one conversation each: the three are minutes-to-hours apart. + assert [key for key, _ in items] == [ + "1758460000.000100", + "1758470200.000100", + "1758470400.000100", + ] + cursors = [kwargs.get("cursor") for _, kwargs in history_client.calls] + assert cursors == [None, "cursor-page-2"] + + +def test_revision_tracks_replies_and_edits(history_client: FakeSlackClient) -> None: + items = dict(_collect(_threads(history_client))) + + replied = items["1758470400.000100"] + assert replied.is_thread + assert replied.reply_count == 3 + assert replied.revision == "1758470999.000500" + + edited = items["1758470200.000100"] + assert not edited.is_thread + assert edited.revision == "1758470260.000000" + + untouched = items["1758460000.000100"] + assert untouched.revision == untouched.start_ts + + +def test_no_lookback_means_no_cutoff() -> None: + assert oldest_ts(None) is None + assert oldest_ts(datetime.timedelta(days=1)) is not None + + +def test_file_scan_keeps_only_fetchable_files(files_client: FakeSlackClient) -> None: + items = _collect(_files(files_client)) + + assert items == [ + ( + "F0TEXT", + FileRef( + channel=CHANNEL, + file_id="F0TEXT", + name="runbook.md", + mimetype="text/markdown", + size=2048, + created=1758470500, + url="https://files.slack.com/files-pri/T0-F0TEXT/download/runbook.md", + permalink="https://example.slack.com/files/U0LEAD/F0TEXT/runbook.md", + user="U0LEAD", + ), + ) + ] diff --git a/uv.lock b/uv.lock index 9688fe8..0fc5952 100644 --- a/uv.lock +++ b/uv.lock @@ -103,6 +103,24 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/99/91/8acff4f5e50511b911bbccb72b8628a49c68ce14148cd9f6431094859a90/annotated_types-0.8.0-py3-none-any.whl", hash = "sha256:f072f4d804ea359e4eaf198b1af7a8b0943881a87f31bb764f8bf219bb9419e0", size = 13427, upload-time = "2026-07-23T20:16:12.938Z" }, ] +[[package]] +name = "anthropic" +version = "1.8.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio" }, + { name = "docstring-parser" }, + { name = "httpx2" }, + { name = "jiter" }, + { name = "pydantic" }, + { name = "sniffio" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/65/b8/f4de0e90bbd641e86a1d6b20e017442033a2c1d2f5381799f82057115bd7/anthropic-1.8.0.tar.gz", hash = "sha256:9c1783ed90f409617749a61c5ab98e20624a572626f2e0a15cea03ed8e1401e5", size = 1221762, upload-time = "2026-09-22T16:25:38.167Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5b/18/5d25a703b66ba34e9277f875f692a3bea3b0ff4d47ee5d188521a01cdd2a/anthropic-1.8.0-py3-none-any.whl", hash = "sha256:79a4516a21e64fd7b15be1a49ebf544bd6376c96a971a77365358823a2717bc6", size = 1348276, upload-time = "2026-09-22T16:25:36.515Z" }, +] + [[package]] name = "anyio" version = "4.15.1" @@ -335,8 +353,10 @@ version = "0.1.0" source = { virtual = "." } dependencies = [ { name = "aiohttp" }, + { name = "anthropic" }, { name = "cocoindex", extra = ["lancedb", "sentence-transformers"] }, { name = "numpy" }, + { name = "pyyaml" }, { name = "slack-sdk" }, ] @@ -345,13 +365,16 @@ dev = [ { name = "ipykernel" }, { name = "mypy" }, { name = "pytest" }, + { name = "types-pyyaml" }, ] [package.metadata] requires-dist = [ { name = "aiohttp", specifier = ">=3.14.3" }, + { name = "anthropic", specifier = ">=1.8.0" }, { name = "cocoindex", extras = ["lancedb", "sentence-transformers"], specifier = ">=1.0.24" }, { name = "numpy", specifier = ">=2.5.3" }, + { name = "pyyaml", specifier = ">=6.0.3" }, { name = "slack-sdk", specifier = ">=3.44.1" }, ] @@ -360,6 +383,7 @@ dev = [ { name = "ipykernel", specifier = ">=7.3.0" }, { name = "mypy", specifier = ">=1.14" }, { name = "pytest", specifier = ">=9.1.1" }, + { name = "types-pyyaml", specifier = ">=6.0.12.20260906" }, ] [[package]] @@ -480,6 +504,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/02/c3/253a89ee03fc9b9682f1541728eb66db7db22148cd94f89ab22528cd1e1b/deprecation-2.1.0-py2.py3-none-any.whl", hash = "sha256:a10811591210e1fb0e768a8c25517cabeabcba6f0bf96564f8ff45189f90b14a", size = 11178, upload-time = "2020-04-20T14:23:36.581Z" }, ] +[[package]] +name = "docstring-parser" +version = "0.18.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e0/4d/f332313098c1de1b2d2ff91cf2674415cc7cddab2ca1b01ae29774bd5fdf/docstring_parser-0.18.0.tar.gz", hash = "sha256:292510982205c12b1248696f44959db3cdd1740237a968ea1e2e7a900eeb2015", size = 29341, upload-time = "2026-04-14T04:09:19.867Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a7/5f/ed01f9a3cdffbd5a008556fc7b2a08ddb1cc6ace7effa7340604b1d16699/docstring_parser-0.18.0-py3-none-any.whl", hash = "sha256:b3fcbed555c47d8479be0796ef7e19c2670d428d72e96da63f3a40122860374b", size = 22484, upload-time = "2026-04-14T04:09:18.638Z" }, +] + [[package]] name = "executing" version = "2.2.1" @@ -594,6 +627,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7e/f5/f66802a942d491edb555dd61e3a9961140fd64c90bce1eafd741609d334d/httpcore-1.0.9-py3-none-any.whl", hash = "sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55", size = 78784, upload-time = "2025-04-24T22:06:20.566Z" }, ] +[[package]] +name = "httpcore2" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "h11" }, + { name = "truststore" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/15/8c/e925b1c92018abb3a1863ce1549d76d2381e334d21d65d4ac8f65dabd78a/httpcore2-2.13.0.tar.gz", hash = "sha256:2adc8be4fb285fbcd6d894298db3b52c177e74b6674eda3a76bd36be3292a3db", size = 67740, upload-time = "2026-09-14T14:18:04.717Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7e/0d/117a771a2bb91df334b66bf4da14cd02f21aefbcfe53180f336ce55e8f90/httpcore2-2.13.0-py3-none-any.whl", hash = "sha256:35ae5be347aa40467b4a5dc032ac67ebb6d27189fc97e8cebcf99616f6a1bb9e", size = 83162, upload-time = "2026-09-14T14:18:02.529Z" }, +] + [[package]] name = "httpx" version = "0.28.1" @@ -609,6 +655,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/2a/39/e50c7c3a983047577ee07d2a9e53faf5a69493943ec3f6a384bdc792deb2/httpx-0.28.1-py3-none-any.whl", hash = "sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad", size = 73517, upload-time = "2024-12-06T15:37:21.509Z" }, ] +[[package]] +name = "httpx2" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio", marker = "sys_platform != 'emscripten'" }, + { name = "httpcore2", marker = "sys_platform != 'emscripten'" }, + { name = "httpx2-jsfetch", marker = "sys_platform == 'emscripten'" }, + { name = "idna" }, + { name = "truststore", marker = "sys_platform != 'emscripten'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b9/a0/e9deef4654132857b5a5dbe4eddd0ac59c2814500e11f2f5044cd81103ee/httpx2-2.13.0.tar.gz", hash = "sha256:81bd07dc67a3701729ef1f777a3c00c915d4539604fdb5afd327f8682f6b7b44", size = 100290, upload-time = "2026-09-14T14:18:05.486Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fe/d1/a0c72b0e006df654709fbc366cc5bcb53e5aee13e1e3395152c6dd293376/httpx2-2.13.0-py3-none-any.whl", hash = "sha256:fc12720cedf72faa26cca6b4ca394e05c894e7d7933fc45cafe767960804e49a", size = 95565, upload-time = "2026-09-14T14:18:03.553Z" }, +] + +[[package]] +name = "httpx2-jsfetch" +version = "1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/cd/c4/0e5636363151a2a1795e0a77617168b9ca438e1748ec05fc9b5687f93d64/httpx2_jsfetch-1.0.tar.gz", hash = "sha256:70a0e3eabfef7cce5ad9c629f7d01ca05e418f586646f4ddf14782e4c1454c60", size = 6872, upload-time = "2026-08-07T00:13:07.492Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9b/43/832f631d32e4f1211caa2ba368317739fe71f0b8530e4c9d15dc454bac2a/httpx2_jsfetch-1.0-py3-none-any.whl", hash = "sha256:cb916b707601e69a07721aabc8f3f6659be3a6893bc1ff5c6f9e02241df2da32", size = 6382, upload-time = "2026-08-07T00:13:06.567Z" }, +] + [[package]] name = "huggingface-hub" version = "1.32.0" @@ -728,6 +799,69 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" }, ] +[[package]] +name = "jiter" +version = "0.17.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9c/1f/8176d92e001f86505424b41664032ae26a882bc9ca41a32c803f373f9195/jiter-0.17.0.tar.gz", hash = "sha256:03e432f226a453851079fb84cd17c6da9991eab723e28d716f14ae3d906e0c12", size = 229037, upload-time = "2026-09-12T15:14:14.253Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/01/9e/23065f8e2c7a4c372c1b6f6622e4cfab4dc786cb5150052b1527e6a6a840/jiter-0.17.0-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:00d783a779c5664e16dbad5e3a3c3a75e128b07dd5f4765159658d9210a50ca5", size = 292210, upload-time = "2026-09-12T15:12:35.613Z" }, + { url = "https://files.pythonhosted.org/packages/ea/81/67b58647560bc82a4490d722caa8561d7a86a9f45d4fa620b7e5fe282c7a/jiter-0.17.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:0619d806e260ecf0c2a64521942c94af5d547c9ec99b55ae4f51b538b5576a76", size = 321512, upload-time = "2026-09-12T15:12:36.907Z" }, + { url = "https://files.pythonhosted.org/packages/c7/07/6658359a25f55927f7f8bf0e16465dee2ccd0b2a1a5208acc0df8972e074/jiter-0.17.0-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:dc0288ce39190ee33fe6e4ec73161eed34e7e2da509b525546ca061778d62b64", size = 343897, upload-time = "2026-09-12T15:12:38.189Z" }, + { url = "https://files.pythonhosted.org/packages/46/04/5d50a9f0319cbdc37fd53c27f8c313d46afc34f1b048219ae6d8ea068da4/jiter-0.17.0-cp314-cp314-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:5a52a430d04225ffde633e6840bf2381d34c019ff98526b5929755b9052fb199", size = 326519, upload-time = "2026-09-12T15:12:39.532Z" }, + { url = "https://files.pythonhosted.org/packages/bb/c7/d02517832b29eb8275fdd0f4ce0f17b80f58cc4c3ebecd4d9ace990d633d/jiter-0.17.0-cp314-cp314-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:37f33d327900bf2879613b3363fd48df97b4232d0c41f54bcf2e790c2fc40a71", size = 341369, upload-time = "2026-09-12T15:12:41.486Z" }, + { url = "https://files.pythonhosted.org/packages/3b/07/499b5f5603501cdd93a73a6a176dfad9c96555a3ae58ca9f8e3acba63dc9/jiter-0.17.0-cp314-cp314-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:6cf564d43c4388149ca58ee571d0f5ccf875e20d1fd4662fd94cc0d1ea3b10ef", size = 352160, upload-time = "2026-09-12T15:12:42.721Z" }, + { url = "https://files.pythonhosted.org/packages/f5/75/b04013c7743269d4533ef4e746fc0ed678a143968dd7448658e3f51daad2/jiter-0.17.0-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:523c499235fb65add25d4bb01b1c4709ce695efdc7deb6c0a7bc515b5c44e0fb", size = 345018, upload-time = "2026-09-12T15:12:44.192Z" }, + { url = "https://files.pythonhosted.org/packages/1d/96/cbb6fd1e42a77c8412ec4643db95059b30cdfc635e387cc9193e098ce268/jiter-0.17.0-cp314-cp314-manylinux_2_31_riscv64.whl", hash = "sha256:455e4ab35cb2a4a91a8404e08fd3c621bae433922e59bf1c494fe20a426b013b", size = 329244, upload-time = "2026-09-12T15:12:45.491Z" }, + { url = "https://files.pythonhosted.org/packages/15/67/d3be402f398566a379bf40ae65be5c3505b14d9e95e0802a597ddde7ddee/jiter-0.17.0-cp314-cp314-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6871973bfbd4408f7f1c632b30bbb5bbd9671c1bc8650af6823e24b7be13709b", size = 335693, upload-time = "2026-09-12T15:12:46.935Z" }, + { url = "https://files.pythonhosted.org/packages/7f/8d/98e2c4130b93d64f1d67c89060b928d04102549bf05e64451c9e6024f9ca/jiter-0.17.0-cp314-cp314-musllinux_1_1_aarch64.whl", hash = "sha256:77f6aac0137309b31448c1bdcda4c6c77077664a6d018ece8d94019c68a5a5b9", size = 484329, upload-time = "2026-09-12T15:12:48.361Z" }, + { url = "https://files.pythonhosted.org/packages/78/5e/8da91e49f0fbca37c3489fb4cf3ad6676d4965f00ae5468bca3a2513737a/jiter-0.17.0-cp314-cp314-musllinux_1_1_x86_64.whl", hash = "sha256:93946d89fa04d5ba64dd323a8dd8d901676cb8a3c81d99ae4f6c051a9b4c3f2f", size = 521358, upload-time = "2026-09-12T15:12:49.856Z" }, + { url = "https://files.pythonhosted.org/packages/be/21/5388684a5a38af3557cd9c2424b9827c71809cff24373c75ef9d0d3dfba9/jiter-0.17.0-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:70f19a2ca8429f91e82eeffb2f51cb87bc2d6e953b009b91a92d29c3a16ccb03", size = 110459, upload-time = "2026-09-12T15:12:51.747Z" }, + { url = "https://files.pythonhosted.org/packages/b1/ad/58b3a93525d2ffca7f54d9dee441381990082bd1172fbeb8d6a3f72a4dc3/jiter-0.17.0-cp314-cp314-win32.whl", hash = "sha256:71dbd74314c5df52a1bccf7b8bca46d14e943af7a2012e73b23f49977ef194c8", size = 185043, upload-time = "2026-09-12T15:12:54.477Z" }, + { url = "https://files.pythonhosted.org/packages/7a/4a/1aa520eb6c359b262c14ff995ca7283837208ddfb1202082ce9d73cf214d/jiter-0.17.0-cp314-cp314-win_amd64.whl", hash = "sha256:ac3c6ee3264d6f5c44c617f90bc7e8b9e1587e7d6708c9d8f811cb65582ee312", size = 227163, upload-time = "2026-09-12T15:12:55.931Z" }, + { url = "https://files.pythonhosted.org/packages/cf/e4/5997f648794bd9b499491d0ff480b096cc9a9c65bdba29f57568e6aa1705/jiter-0.17.0-cp314-cp314-win_arm64.whl", hash = "sha256:6219adaf59711ba7063a52496e8ec6d3fa3e209d7827d83eee3b2abc780a1744", size = 183505, upload-time = "2026-09-12T15:12:58.196Z" }, + { url = "https://files.pythonhosted.org/packages/ac/4a/84a5ec271d09f7590b6073af5ee4abb44eab4ccace453b7e2c5ce45234ca/jiter-0.17.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:59bddbe6f9ffecc68d641e1e2d619ce64cf8a9e9eeb74e5c518f74fc87abf1b0", size = 321527, upload-time = "2026-09-12T15:12:59.394Z" }, + { url = "https://files.pythonhosted.org/packages/39/71/9e1fd0045f5920b4c36be35c3f0f0dfd123668684f8ad352619d7aa44183/jiter-0.17.0-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6cb41cd1432f1dc19a231cf70b54d42b2c9f05085155859263fce06fa4d41388", size = 340865, upload-time = "2026-09-12T15:13:00.756Z" }, + { url = "https://files.pythonhosted.org/packages/b7/2b/14627fd2bc377f3dd09491bcace6b90e34b4d7fea2f1f3295031ff91f528/jiter-0.17.0-cp314-cp314t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:fd7790aa79c8b518e512ebcdfce9f11d8ef5f30efd43720c8a19a548b39fa489", size = 325412, upload-time = "2026-09-12T15:13:02.152Z" }, + { url = "https://files.pythonhosted.org/packages/4c/f3/8d5808f7bf0f456bde79e6393587183a0cee5f83d179fe1f7f1eff2ba067/jiter-0.17.0-cp314-cp314t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:dbbfe4e3c21c8166980cddc5bee1a315df082454f007947dfb6fb73800768165", size = 340473, upload-time = "2026-09-12T15:13:03.485Z" }, + { url = "https://files.pythonhosted.org/packages/4f/da/1d8c7c6c4ae6b2423b94a81b6b907d37b28f87664e077427b531bf1b5313/jiter-0.17.0-cp314-cp314t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8c286860abfe8b100cac1c02e225e5776eb9216edd71ba17cdb237da4af32bc9", size = 350757, upload-time = "2026-09-12T15:13:04.828Z" }, + { url = "https://files.pythonhosted.org/packages/eb/96/c1813dcca15c5a370145a448aaea7d1f83f6f0228a5f1130e79340ee385f/jiter-0.17.0-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f753eb70b1474a29e635e7542ff7312e6d6b951e0b25e8a2e8c34eeb1ddcd478", size = 345203, upload-time = "2026-09-12T15:13:06.131Z" }, + { url = "https://files.pythonhosted.org/packages/d7/f7/fc61cbcf2992d169ede13648fc3fd8e2d3171a3669dde43cd4db556549ac/jiter-0.17.0-cp314-cp314t-manylinux_2_31_riscv64.whl", hash = "sha256:eae86b1f027031e39db2e0e9c4842221edb7b8cd474d23f87a79b3bd4b651768", size = 328322, upload-time = "2026-09-12T15:13:07.392Z" }, + { url = "https://files.pythonhosted.org/packages/8f/88/46418a3abbdffb7dc41b314200360f24f75faaeb35573e81c92de322cce9/jiter-0.17.0-cp314-cp314t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:5bf350452a43173e69e1fc74847c57a60e3d7515807287f29849baa2a85d8718", size = 336570, upload-time = "2026-09-12T15:13:08.666Z" }, + { url = "https://files.pythonhosted.org/packages/f0/28/b8a55b949be6306df8888e365a8df05441de8a7b11289f6957004302e41e/jiter-0.17.0-cp314-cp314t-musllinux_1_1_aarch64.whl", hash = "sha256:da139721f4b7cafdbff580a4f511ea24cb91f4909330c6b926a1ca53836c0a59", size = 482879, upload-time = "2026-09-12T15:13:10.037Z" }, + { url = "https://files.pythonhosted.org/packages/75/3b/21d0afa53ba0680962c39f3eb95ed2946f8793369ed44b0c82b490723081/jiter-0.17.0-cp314-cp314t-musllinux_1_1_x86_64.whl", hash = "sha256:8079849db9a1371bfd90bad088458a8fb836261879df2233cc9632464ecf64e1", size = 520406, upload-time = "2026-09-12T15:13:11.456Z" }, + { url = "https://files.pythonhosted.org/packages/ef/03/bcbaf8b6b9ea23c2c074411f8ecfbb02d820abac5d0cb8f4e280209174a2/jiter-0.17.0-cp314-cp314t-win32.whl", hash = "sha256:8f770b0c77e5fac482e1ba03ca1a7e18286bfb213d749932a00a7e4cd5de5e06", size = 184434, upload-time = "2026-09-12T15:13:13.037Z" }, + { url = "https://files.pythonhosted.org/packages/7a/b5/5d6ce2c93ef6fe1241b37a9005547f9b6d58db1f07f39fe95807d4b98f51/jiter-0.17.0-cp314-cp314t-win_amd64.whl", hash = "sha256:c4289293e5278d9314b00f15c37f2120fa51d3d68565292e715524c750e775a9", size = 227392, upload-time = "2026-09-12T15:13:14.933Z" }, + { url = "https://files.pythonhosted.org/packages/f5/4b/1e52baf90187606e33a7b8cfa8f96f5829acd7f01870077eb01059ab76d0/jiter-0.17.0-cp314-cp314t-win_arm64.whl", hash = "sha256:4dfbfe5a6e1e80a7082af559f66386405025ec278833e0c649f69cbc6e1004cc", size = 182776, upload-time = "2026-09-12T15:13:16.239Z" }, + { url = "https://files.pythonhosted.org/packages/05/fc/efe3ac75564ab10f53517958f5ccdc231fc7334af66c76776cb554a88967/jiter-0.17.0-cp315-cp315-macosx_10_12_x86_64.whl", hash = "sha256:84963d3f395ef5e9a32ce47155e08a7962fa292c159a10cb98b931cef1416925", size = 292143, upload-time = "2026-09-12T15:13:17.502Z" }, + { url = "https://files.pythonhosted.org/packages/d1/4c/46982118d91f9ffe9714319d21ec4f98d9b7e0cfd9062826c524a54de24e/jiter-0.17.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:ffa0380ad091de7d3fc33e17a97ff479851ee18a0a2a3ee56ff3215cdc886656", size = 321341, upload-time = "2026-09-12T15:13:19.133Z" }, + { url = "https://files.pythonhosted.org/packages/e7/12/9b1ac6ecc6307049913db54839ddba1c11c1ef72c5a8bbb5514bc3b50d1b/jiter-0.17.0-cp315-cp315-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:755079792868ce5d4938e83b91a0939b34fb858a1ca65a104f2d771bea57faa1", size = 344383, upload-time = "2026-09-12T15:13:20.508Z" }, + { url = "https://files.pythonhosted.org/packages/a9/b6/527cc72af836d824e9d4d666e64f0a1ca7eafd662a8da9657b78592172ba/jiter-0.17.0-cp315-cp315-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:3bf4dc2b84a464117fb097d15a25c58d100d2692888e3b0d92df5b48ed16b7c0", size = 326841, upload-time = "2026-09-12T15:13:21.83Z" }, + { url = "https://files.pythonhosted.org/packages/d1/41/567f98617e88005b249503b933803f633ec6ba2d427cf4cc35e5c832125c/jiter-0.17.0-cp315-cp315-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:02a360707033d8cef53f7f3480817a1489177a259ec6ec01e98c37e0b922ddca", size = 341354, upload-time = "2026-09-12T15:13:23.323Z" }, + { url = "https://files.pythonhosted.org/packages/40/da/b29cda895b785f7d426e224638a885b6145a08ce853b381f34afe3e88c5d/jiter-0.17.0-cp315-cp315-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:300ce01ab0215e3dea4d00090143c909aedc65c0f809b3c07983e1d038f291b9", size = 351985, upload-time = "2026-09-12T15:13:26.526Z" }, + { url = "https://files.pythonhosted.org/packages/f7/5c/8a73829e7389e72ea298a450f2b3cb58e71a3e464b45f6d8753740f1c4f5/jiter-0.17.0-cp315-cp315-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:746243a080b4ca790b8499af3d7cf9825d5f5987933950cd818e767ee353d826", size = 346052, upload-time = "2026-09-12T15:13:27.887Z" }, + { url = "https://files.pythonhosted.org/packages/1d/2f/98d6001026932c095ba440925570123043bed29f5ff56158dfe729a9e81b/jiter-0.17.0-cp315-cp315-manylinux_2_31_riscv64.whl", hash = "sha256:b550585523339b71cb852b811aae49d08d7601ad8ffe9f5dc1562f4c3d22fd87", size = 329159, upload-time = "2026-09-12T15:13:31.569Z" }, + { url = "https://files.pythonhosted.org/packages/94/2e/708dc1d2678f092c31c12754e860cd8353e6a85ecbdb1010157edca0da9e/jiter-0.17.0-cp315-cp315-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:0239520085cac678e77a606fd7e3f1c60c371d719790c5e3807388d3da4354c2", size = 336001, upload-time = "2026-09-12T15:13:32.846Z" }, + { url = "https://files.pythonhosted.org/packages/f4/f0/75a5ae38862f4eaf0fe2f8a9fbf6484c4890df04c06dcdffc45e36bca61a/jiter-0.17.0-cp315-cp315-musllinux_1_1_aarch64.whl", hash = "sha256:eb2295da7c3769f6719b227a237aa6a5cfa6550e478bc838001b592c57e16575", size = 484281, upload-time = "2026-09-12T15:13:35.333Z" }, + { url = "https://files.pythonhosted.org/packages/a0/32/6636fae811c27c7f93e1b11fb5800de6a5c9e4269a27cf718e0b31218ad1/jiter-0.17.0-cp315-cp315-musllinux_1_1_x86_64.whl", hash = "sha256:e088612ff90ebc9247e1a43074b72835804261c47e6a6c01cb3ddcb55360d688", size = 521300, upload-time = "2026-09-12T15:13:37.101Z" }, + { url = "https://files.pythonhosted.org/packages/61/aa/12df7e0b0b1a2602e3d5a5a7104d7d9700f254b400f134a9b50955c4d231/jiter-0.17.0-cp315-cp315-win32.whl", hash = "sha256:0b52d52035b3907c5b1f6277857b29c1cbfc965e24e0f27330dbed83edb591ec", size = 185138, upload-time = "2026-09-12T15:13:38.901Z" }, + { url = "https://files.pythonhosted.org/packages/ba/ec/3dd2e495032cddde05723c1f4c743b67a23e55d2af244692a7f58f0cdae3/jiter-0.17.0-cp315-cp315-win_amd64.whl", hash = "sha256:10f5558eed511b830488003449d942bd75829ad6257dc58cb9a03e596a7777b1", size = 226950, upload-time = "2026-09-12T15:13:40.17Z" }, + { url = "https://files.pythonhosted.org/packages/c3/c7/ef85704e0a57e9cadb2babc05f6d7c5df4a1c75da1a6ee31e1986b0099a5/jiter-0.17.0-cp315-cp315-win_arm64.whl", hash = "sha256:fa13acf1046f95df808c64b1310705e143fab87aee73ae00cc42d640867fd2c1", size = 183618, upload-time = "2026-09-12T15:13:41.432Z" }, + { url = "https://files.pythonhosted.org/packages/0e/9a/a4b348349de68762b58d6713973d363ad80a1c741d0bf8def7975f0ecb26/jiter-0.17.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:af2f7501580f274b63c4b2283bc425f5df7edf06ae5b171e5f87d912ff359a20", size = 321155, upload-time = "2026-09-12T15:13:42.716Z" }, + { url = "https://files.pythonhosted.org/packages/c1/70/aebd6d0b5f0677de3a3d0bdc4a05fac949b97c4ede454c8809f180ac7b17/jiter-0.17.0-cp315-cp315t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:10c5349312e5cb02b7a21e123a57665afa895953f05bf252a9dd4c13a572b7ab", size = 340985, upload-time = "2026-09-12T15:13:44.115Z" }, + { url = "https://files.pythonhosted.org/packages/a7/82/4c3b49796b5eb62f3f5046f957683f4ba0135fe1a60957c11180512460df/jiter-0.17.0-cp315-cp315t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:86f3f9343a288eb85a81ef20a752b2f84564296636db54a9fff0b5c8deaf1df2", size = 325670, upload-time = "2026-09-12T15:13:45.901Z" }, + { url = "https://files.pythonhosted.org/packages/bc/43/f6341ecb4872202a4ef150486fcee0e1ace4aa3da39b71b82061452cdd3a/jiter-0.17.0-cp315-cp315t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:4607ec7d93355fbc25b8dc5189153cf21d66063b9f9cd04dd2774e6e783f9b6a", size = 340339, upload-time = "2026-09-12T15:13:47.442Z" }, + { url = "https://files.pythonhosted.org/packages/f9/c4/bc2c86e08fa065e03cb2fbc53b367c3640a7d257ef9d877b29118ea636b7/jiter-0.17.0-cp315-cp315t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:10cd64a5720ad7f809ac5466ff1705813f1b6b510f195a73acafba0ac0e1f675", size = 350705, upload-time = "2026-09-12T15:13:48.848Z" }, + { url = "https://files.pythonhosted.org/packages/9d/67/91f12aa111cca6e3a197c3e36bf60a034bf9f122f6d41112a639e44217d8/jiter-0.17.0-cp315-cp315t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:efe9f61bb30174d2f5c8396445c360c96c44e78164d0815dfe627ccf57849574", size = 345011, upload-time = "2026-09-12T15:13:50.215Z" }, + { url = "https://files.pythonhosted.org/packages/f5/cb/9f5556e8f6ec89755fb5a709d8eb8270c9a324e31079eda0dfbeca451b6e/jiter-0.17.0-cp315-cp315t-manylinux_2_31_riscv64.whl", hash = "sha256:370d8fe5bf201dc6925e8a84c81ac7291f74d9fd1778234fc79d517064a5c76b", size = 328268, upload-time = "2026-09-12T15:13:51.809Z" }, + { url = "https://files.pythonhosted.org/packages/22/98/153f20680fb75781a490fb849940e2b00f95035c7aa054df592f36ed33fc/jiter-0.17.0-cp315-cp315t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6b303d88e6a0bda789ec4b7801c7bad68e27230ba1fe4baffc756d1fbd32dc9d", size = 337024, upload-time = "2026-09-12T15:13:53.095Z" }, + { url = "https://files.pythonhosted.org/packages/af/59/b16c9be3a5035df4466cc72e888188c027562de90a723d290ab6814cb9d4/jiter-0.17.0-cp315-cp315t-musllinux_1_1_aarch64.whl", hash = "sha256:30793a24a31e968969757c9e08d830cbb15a2cd3c4959b4498b38f4b1c2258eb", size = 482766, upload-time = "2026-09-12T15:13:55.713Z" }, + { url = "https://files.pythonhosted.org/packages/d0/55/667dea313094024bef082175d6bfe8976f90d1c00c926af9df1d8e0eab48/jiter-0.17.0-cp315-cp315t-musllinux_1_1_x86_64.whl", hash = "sha256:686c93d86f2b426c803024b805bd161a6cd10e9627c23e901640eab646c0ad8a", size = 520367, upload-time = "2026-09-12T15:13:57.674Z" }, + { url = "https://files.pythonhosted.org/packages/21/e3/4b1a43501fb9ed17b01d137e380cb0e8fdcb39a254ce31aa2ab95bc861ac/jiter-0.17.0-cp315-cp315t-win32.whl", hash = "sha256:86d703d9faa1ffc8ae4e9de0fa007712ed2171b5c0d93811a8e2e105ac729b0d", size = 184603, upload-time = "2026-09-12T15:13:59.27Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f2/b8ee0372b6ebdf1bde5cc44495d5291d17f961065f5b48f8616cc67cac2e/jiter-0.17.0-cp315-cp315t-win_amd64.whl", hash = "sha256:42b0260445251b1bc520a63baa94a32d88e0f931fba234f1764db7feb7c72174", size = 227936, upload-time = "2026-09-12T15:14:00.472Z" }, + { url = "https://files.pythonhosted.org/packages/a4/b4/923a1215daba959aed8355973315cb3f81f53e0d01c5b211870a27b41f45/jiter-0.17.0-cp315-cp315t-win_arm64.whl", hash = "sha256:d47687806f9c54c84ea38733507081337922beca90ce819c7d852dd485bc0f23", size = 182977, upload-time = "2026-09-12T15:14:01.799Z" }, +] + [[package]] name = "joblib" version = "1.6.0" @@ -2005,6 +2139,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ab/fc/67352b742fc6fa520a550581b0284f5757e4e31a32327c2a00276747fd30/slack_sdk-3.44.1-py2.py3-none-any.whl", hash = "sha256:d6f20a0fbe3fecf9cac955c99d686301b48a645b7045472c3a0cdd186d7c42b2", size = 319865, upload-time = "2026-09-03T14:21:12.405Z" }, ] +[[package]] +name = "sniffio" +version = "1.3.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a2/87/a6771e1546d97e7e041b6ae58d80074f81b7d5121207425c964ddf5cfdbd/sniffio-1.3.1.tar.gz", hash = "sha256:f4324edc670a0f49750a81b895f35c3adb843cca46f0530f79fc1babb23789dc", size = 20372, upload-time = "2024-02-25T23:20:04.057Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" }, +] + [[package]] name = "stack-data" version = "0.6.3" @@ -2167,6 +2310,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fe/d1/aa8a3e935c37efee7945984fdb64d7e0851bf6d920afd97b2d21f9d23360/triton-3.8.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:74217bb56ed8692759227758e4c4b3bd2d608a209c1a7a081bf361fb4c2c1bf9", size = 248077577, upload-time = "2026-08-28T15:56:24.94Z" }, ] +[[package]] +name = "truststore" +version = "0.10.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/53/a3/1585216310e344e8102c22482f6060c7a6ea0322b63e026372e6dcefcfd6/truststore-0.10.4.tar.gz", hash = "sha256:9d91bd436463ad5e4ee4aba766628dd6cd7010cf3e2461756b3303710eebc301", size = 26169, upload-time = "2025-08-12T18:49:02.73Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/19/97/56608b2249fe206a67cd573bc93cd9896e1efb9e98bce9c163bcdc704b88/truststore-0.10.4-py3-none-any.whl", hash = "sha256:adaeaecf1cbb5f4de3b1959b42d41f6fab57b2b1666adb59e89cb0b53361d981", size = 18660, upload-time = "2025-08-12T18:49:01.46Z" }, +] + [[package]] name = "typer" version = "0.27.2" @@ -2182,6 +2334,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/dc/bf/205d0004930ede8f542fb58f601526fccf4ae7626075ca1e6c4de5d3d652/typer-0.27.2-py3-none-any.whl", hash = "sha256:b3a5fc4342d5fc8fda8fc3010b1cf117e9249aab7fae800c2eff62fd3842d97d", size = 123130, upload-time = "2026-08-28T10:26:53.752Z" }, ] +[[package]] +name = "types-pyyaml" +version = "6.0.12.20260906" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/90/6e/abec85b9013db5b934b0280a6dd104904d84f7bcbaab2e2f3def87ac7463/types_pyyaml-6.0.12.20260906.tar.gz", hash = "sha256:f59c1cc05010b833d2d72287bbaa72610106b28d42d89a907313117faba85212", size = 18649, upload-time = "2026-09-06T06:35:35.362Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/15/c0/fc0644b7ddcfb969e95845837143cb5173ddd6e06ee4ba5fc493cd9329b7/types_pyyaml-6.0.12.20260906-py3-none-any.whl", hash = "sha256:bca893ff0d51df5c9053137d5d0e6ccd36e939a196356f1d5c16372422f5137b", size = 21282, upload-time = "2026-09-06T06:35:34.372Z" }, +] + [[package]] name = "typing-extensions" version = "4.16.0"