From a7af7c4e8f2369d28fd50c754f3394184d2346ad Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:44:51 +0900 Subject: [PATCH 01/12] slack: index channel history and files Make Slack conversations and shared text files searchable locally so channel context can survive API pagination, edits, and live updates without manual exports. --- .env.example | 13 +++ .gitignore | 8 +- slack_index/__init__.py | 1 + slack_index/app.py | 78 ++++++++++++++ slack_index/chunking.py | 67 ++++++++++++ slack_index/config.py | 64 +++++++++++ slack_index/context.py | 15 +++ slack_index/files.py | 82 +++++++++++++++ slack_index/models.py | 55 ++++++++++ slack_index/query.py | 39 +++++++ slack_index/source.py | 169 ++++++++++++++++++++++++++++++ slack_index/threads.py | 73 +++++++++++++ slack_index/users.py | 29 +++++ tests/__init__.py | 0 tests/conftest.py | 51 +++++++++ tests/fixtures/files_list.json | 38 +++++++ tests/fixtures/history_page1.json | 30 ++++++ tests/fixtures/history_page2.json | 13 +++ tests/test_files.py | 41 ++++++++ tests/test_source.py | 102 ++++++++++++++++++ 20 files changed, 967 insertions(+), 1 deletion(-) create mode 100644 .env.example create mode 100644 slack_index/__init__.py create mode 100644 slack_index/app.py create mode 100644 slack_index/chunking.py create mode 100644 slack_index/config.py create mode 100644 slack_index/context.py create mode 100644 slack_index/files.py create mode 100644 slack_index/models.py create mode 100644 slack_index/query.py create mode 100644 slack_index/source.py create mode 100644 slack_index/threads.py create mode 100644 slack_index/users.py create mode 100644 tests/__init__.py create mode 100644 tests/conftest.py create mode 100644 tests/fixtures/files_list.json create mode 100644 tests/fixtures/history_page1.json create mode 100644 tests/fixtures/history_page2.json create mode 100644 tests/test_files.py create mode 100644 tests/test_source.py diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..3b9c195 --- /dev/null +++ b/.env.example @@ -0,0 +1,13 @@ +# Bot token (xoxb-...) with channels:history, groups:history, files:read, users:read. +# The bot must be a member of every channel listed below. +SLACK_BOT_TOKEN=xoxb-replace-me + +# Comma-separated channel ids, e.g. C0123ABCD,C0456EFGH +SLACK_CHANNEL_IDS= + +# Optional +#SLACK_INDEX_EMBED_MODEL=sentence-transformers/all-MiniLM-L6-v2 +#SLACK_INDEX_LOOKBACK_DAYS=90 +#SLACK_INDEX_POLL_SECONDS=60 +#SLACK_INDEX_MAX_FILE_BYTES=5242880 +#SLACK_INDEX_VAR_DIR=var diff --git a/.gitignore b/.gitignore index eaeb69d..5d609b8 100644 --- a/.gitignore +++ b/.gitignore @@ -8,4 +8,10 @@ result-* # cache __pycache__ -cocoindex.db +.pytest_cache/ + +# runtime state: engine db, vector store +var/ + +# secrets +.env diff --git a/slack_index/__init__.py b/slack_index/__init__.py new file mode 100644 index 0000000..065b126 --- /dev/null +++ b/slack_index/__init__.py @@ -0,0 +1 @@ +"""Index Slack channel conversations and shared files into LanceDB.""" diff --git a/slack_index/app.py b/slack_index/app.py new file mode 100644 index 0000000..4dac94b --- /dev/null +++ b/slack_index/app.py @@ -0,0 +1,78 @@ +"""Pipeline entry point. + +cocoindex update slack_index/app.py # one-shot catch-up +cocoindex update -L slack_index/app.py # live: re-scan every poll interval +""" + +from __future__ import annotations + +from collections.abc import AsyncIterator + +import cocoindex as coco +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +from slack_index import config +from slack_index.context import EMBEDDER, LANCE_DB, SLACK, SLACK_LIMIT +from slack_index.files import process_file +from slack_index.models import SlackChunk +from slack_index.source import SlackChannelFiles, SlackChannelThreads +from slack_index.threads import process_thread + +_settings = config.Settings.from_env() + + +@coco.lifespan +async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None]: + config.VAR_DIR.mkdir(parents=True, exist_ok=True) + builder.settings.db_path = config.DB_PATH + builder.provide(SLACK, AsyncWebClient(token=config.bot_token())) + builder.provide(SLACK_LIMIT, RateLimiter(config.SLACK_REQUESTS_PER_SECOND)) + builder.provide(EMBEDDER, SentenceTransformerEmbedder(_settings.embed_model)) + builder.provide(LANCE_DB, await lancedb.connect_async(str(config.LANCEDB_URI))) + yield + + +@coco.fn +async def app_main(settings: config.Settings) -> None: + table = await lancedb.mount_table_target( + LANCE_DB, + config.TABLE_NAME, + await lancedb.TableSchema.from_class(SlackChunk, primary_key=["id"]), + ) + table.declare_vector_index(column="embedding") + + client = coco.use_context(SLACK) + limiter = coco.use_context(SLACK_LIMIT) + for channel in settings.channel_ids: + threads = SlackChannelThreads( + client, + limiter, + channel, + lookback=settings.lookback, + poll_interval=settings.poll_interval, + ) + files = SlackChannelFiles( + client, + limiter, + channel, + lookback=settings.lookback, + poll_interval=settings.poll_interval, + ) + # The channel is part of the subpath so each channel keeps its own + # component subtree — and its own rows — across runs. + await coco.mount_each( + coco.ComponentSubpath("threads", channel), process_thread, threads, table + ) + await coco.mount_each( + coco.ComponentSubpath("files", channel), + process_file, + files, + table, + settings.max_file_bytes, + ) + + +app = coco.App(coco.AppConfig(name="SlackIndex"), app_main, settings=_settings) diff --git a/slack_index/chunking.py b/slack_index/chunking.py new file mode 100644 index 0000000..204f02f --- /dev/null +++ b/slack_index/chunking.py @@ -0,0 +1,67 @@ +"""Text -> chunks -> embedded rows, shared by the thread and file pipelines.""" + +from __future__ import annotations + +import datetime +from dataclasses import dataclass + +import cocoindex as coco +from cocoindex.connectors import lancedb +from cocoindex.ops.text import RecursiveSplitter +from cocoindex.resources.chunk import Chunk +from cocoindex.resources.id import IdGenerator + +from slack_index.config import CHUNK_OVERLAP, CHUNK_SIZE +from slack_index.context import EMBEDDER +from slack_index.models import SlackChunk + +_splitter = RecursiveSplitter() + + +@dataclass(frozen=True, slots=True) +class ChunkMeta: + """Row fields every chunk of one source shares.""" + + kind: str + channel: str + source_id: str + permalink: str + author: str + posted_at: datetime.datetime + + +@coco.fn +async def _declare_chunk( + chunk: Chunk, + meta: ChunkMeta, + id_gen: IdGenerator, + table: lancedb.TableTarget[SlackChunk], +) -> None: + table.declare_row( + row=SlackChunk( + id=await id_gen.next_id(chunk.text), + kind=meta.kind, + channel=meta.channel, + source_id=meta.source_id, + permalink=meta.permalink, + author=meta.author, + posted_at=meta.posted_at, + text=chunk.text, + embedding=await coco.use_context(EMBEDDER).embed(chunk.text), + ), + ) + + +async def declare_chunks( + text: str, + meta: ChunkMeta, + table: lancedb.TableTarget[SlackChunk], +) -> None: + chunks = _splitter.split( + text, + chunk_size=CHUNK_SIZE, + chunk_overlap=CHUNK_OVERLAP, + language="markdown", + ) + id_gen = IdGenerator() + await coco.map(_declare_chunk, chunks, meta, id_gen, table) diff --git a/slack_index/config.py b/slack_index/config.py new file mode 100644 index 0000000..65989ee --- /dev/null +++ b/slack_index/config.py @@ -0,0 +1,64 @@ +"""Runtime configuration, resolved from the environment.""" + +from __future__ import annotations + +import datetime +import os +import pathlib +from dataclasses import dataclass + +# Everything mutable the pipeline produces (engine state, vector store, downloaded +# attachments) lives under one directory, so a reset is a single `rm -rf`. +_REPO_ROOT = pathlib.Path(__file__).resolve().parent.parent +VAR_DIR = pathlib.Path(os.environ.get("SLACK_INDEX_VAR_DIR", _REPO_ROOT / "var")) + +DB_PATH = VAR_DIR / "cocoindex.db" +LANCEDB_URI = VAR_DIR / "lancedb" + +TABLE_NAME = "slack_chunks" +CHUNK_SIZE = 1200 +CHUNK_OVERLAP = 200 + +# conversations.* and files.* are Slack tier 3 methods: 50+ requests per minute. +SLACK_REQUESTS_PER_SECOND = 50 / 60 + + +@dataclass(frozen=True, slots=True) +class Settings: + channel_ids: tuple[str, ...] + embed_model: str + lookback: datetime.timedelta + poll_interval: datetime.timedelta + max_file_bytes: int + + @classmethod + def from_env(cls) -> Settings: + channels = os.environ.get("SLACK_CHANNEL_IDS", "") + channel_ids = tuple(c.strip() for c in channels.split(",") if c.strip()) + if not channel_ids: + raise RuntimeError( + "SLACK_CHANNEL_IDS is empty: set it to a comma-separated list of " + "channel ids (e.g. C0123ABCD,C0456EFGH)" + ) + return cls( + channel_ids=channel_ids, + embed_model=os.environ.get( + "SLACK_INDEX_EMBED_MODEL", "sentence-transformers/all-MiniLM-L6-v2" + ), + lookback=datetime.timedelta( + days=float(os.environ.get("SLACK_INDEX_LOOKBACK_DAYS", "90")) + ), + poll_interval=datetime.timedelta( + seconds=float(os.environ.get("SLACK_INDEX_POLL_SECONDS", "60")) + ), + max_file_bytes=int( + os.environ.get("SLACK_INDEX_MAX_FILE_BYTES", str(5 * 1024 * 1024)) + ), + ) + + +def bot_token() -> str: + token = os.environ.get("SLACK_BOT_TOKEN") + if not token: + raise RuntimeError("SLACK_BOT_TOKEN is not set") + return token diff --git a/slack_index/context.py b/slack_index/context.py new file mode 100644 index 0000000..8016159 --- /dev/null +++ b/slack_index/context.py @@ -0,0 +1,15 @@ +"""Context keys shared by every component in the pipeline.""" + +from __future__ import annotations + +import cocoindex as coco +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +SLACK = coco.ContextKey[AsyncWebClient]("slack") +SLACK_LIMIT = coco.ContextKey[RateLimiter]("slack_rate_limit") +# detect_change: swapping the embedding model must re-embed everything. +EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("embedder", detect_change=True) +LANCE_DB = coco.ContextKey[lancedb.LanceAsyncConnection]("lancedb") diff --git a/slack_index/files.py b/slack_index/files.py new file mode 100644 index 0000000..e3fd52b --- /dev/null +++ b/slack_index/files.py @@ -0,0 +1,82 @@ +"""One component per shared file: download, extract text, chunk, embed.""" + +from __future__ import annotations + +import datetime +import logging + +import aiohttp +import cocoindex as coco +from cocoindex.connectors import lancedb + +from slack_index.chunking import ChunkMeta, declare_chunks +from slack_index.config import bot_token +from slack_index.context import SLACK_LIMIT +from slack_index.models import FileRef, SlackChunk +from slack_index.users import display_name + +_logger = logging.getLogger(__name__) + +# Formats readable as-is. Binary documents (pdf, docx, pptx) need a converter — +# add one in `extract_text` and they flow through the rest of the pipeline unchanged. +TEXT_MIMETYPE_PREFIXES = ("text/",) +TEXT_MIMETYPES = frozenset( + { + "application/json", + "application/xml", + "application/x-ndjson", + "application/x-sh", + "application/javascript", + } +) + + +def is_text(mimetype: str) -> bool: + return mimetype.startswith(TEXT_MIMETYPE_PREFIXES) or mimetype in TEXT_MIMETYPES + + +async def download(ref: FileRef) -> bytes: + """Fetch a file's bytes. `url_private_download` needs the bot token, not the API.""" + await coco.use_context(SLACK_LIMIT).acquire() + headers = {"Authorization": f"Bearer {bot_token()}"} + async with ( + aiohttp.ClientSession(headers=headers) as session, + session.get(ref.url) as response, + ): + response.raise_for_status() + return await response.read() + + +async def extract_text(ref: FileRef, max_bytes: int) -> str | None: + """Return the file's text, or None when there is nothing indexable in it.""" + if not is_text(ref.mimetype): + _logger.info("skipping %s (%s): no extractor", ref.name, ref.mimetype) + return None + if ref.size > max_bytes: + _logger.info("skipping %s: %d bytes exceeds the limit", ref.name, ref.size) + return None + return (await download(ref)).decode("utf-8", errors="replace") + + +@coco.fn(memo=True) +async def process_file( + ref: FileRef, + table: lancedb.TableTarget[SlackChunk], + max_bytes: int, +) -> None: + text = await extract_text(ref, max_bytes) + if text is None: + return + + await declare_chunks( + f"# {ref.name}\n\n{text}", + ChunkMeta( + kind="file", + channel=ref.channel, + source_id=ref.file_id, + permalink=ref.permalink, + author=await display_name(ref.user), + posted_at=datetime.datetime.fromtimestamp(ref.created, tz=datetime.UTC), + ), + table, + ) diff --git a/slack_index/models.py b/slack_index/models.py new file mode 100644 index 0000000..2a28467 --- /dev/null +++ b/slack_index/models.py @@ -0,0 +1,55 @@ +"""Source-side identities and the indexed row schema.""" + +from __future__ import annotations + +import datetime +from dataclasses import dataclass +from typing import Annotated + +from numpy.typing import NDArray + +from slack_index.context import EMBEDDER + + +@dataclass(frozen=True, slots=True) +class ThreadRef: + """A thread as seen by the channel scan. + + Every field takes part in change detection: when the scan reports a different + value the thread's component re-runs, and nothing else does. + """ + + channel: str + thread_ts: str + revision: str + reply_count: int + + +@dataclass(frozen=True, slots=True) +class FileRef: + """A file shared in the channel, as listed by ``files.list``.""" + + channel: str + file_id: str + name: str + mimetype: str + size: int + created: int + url: str + permalink: str + user: str | None + + +@dataclass +class SlackChunk: + """One embedded chunk — of a thread transcript or of a shared file.""" + + id: int + kind: str + channel: str + source_id: str + permalink: str + author: str + posted_at: datetime.datetime + text: str + embedding: Annotated[NDArray, EMBEDDER] diff --git a/slack_index/query.py b/slack_index/query.py new file mode 100644 index 0000000..e8354ca --- /dev/null +++ b/slack_index/query.py @@ -0,0 +1,39 @@ +"""Search the index.""" + +from __future__ import annotations + +import asyncio +import sys + +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder + +from slack_index import config + +TOP_K = 5 + + +async def search(query: str, *, top_k: int = TOP_K) -> None: + settings = config.Settings.from_env() + embedder = SentenceTransformerEmbedder(settings.embed_model) + conn = await lancedb.connect_async(str(config.LANCEDB_URI)) + table = await conn.open_table(config.TABLE_NAME) + + request = await table.search( + await embedder.embed(query), vector_column_name="embedding" + ) + for row in await request.limit(top_k).to_list(): + score = 1.0 - row["_distance"] + print(f"[{score:.3f}] {row['kind']} by {row['author']} — {row['permalink']}") + print(f" {row['text'][:300]}") + print("---") + + +def main() -> None: + if len(sys.argv) < 2: + sys.exit("usage: python -m slack_index.query ") + asyncio.run(search(" ".join(sys.argv[1:]))) + + +if __name__ == "__main__": + main() diff --git a/slack_index/source.py b/slack_index/source.py new file mode 100644 index 0000000..004a7fc --- /dev/null +++ b/slack_index/source.py @@ -0,0 +1,169 @@ +"""Slack sources as live keyed maps. + +Both views scan the channel through the Web API and, in live mode, re-scan on a +fixed interval. A re-scan is enough to drive deletes too: whatever the scan stops +reporting loses its component, and the rows that component declared are dropped. +""" + +from __future__ import annotations + +import asyncio +import datetime +from collections.abc import AsyncIterator +from typing import Any, Protocol + +from cocoindex.connectorkits import SingleWatcherGuard +from cocoindex.resources.rate_limit import RateLimiter +from slack_sdk.web.async_client import AsyncWebClient + +from slack_index.models import FileRef, ThreadRef + +# Joins, leaves and topic changes carry no content worth searching. +SKIP_SUBTYPES = frozenset( + {"channel_join", "channel_leave", "channel_topic", "channel_purpose", "bot_add"} +) +# Only files whose bytes Slack actually hosts can be fetched and read. +INDEXABLE_FILE_MODES = frozenset({"hosted", "snippet", "post", "space"}) + + +class _Subscriber(Protocol): + async def update_all(self) -> None: ... + async def mark_ready(self) -> None: ... + + +async def _poll( + subscriber: _Subscriber, interval: datetime.timedelta, guard: SingleWatcherGuard +) -> None: + with guard: + await subscriber.update_all() + # In catch-up mode mark_ready() ends watch() here; in live mode it returns. + await subscriber.mark_ready() + while True: + await asyncio.sleep(interval.total_seconds()) + await subscriber.update_all() + + +def _oldest_ts(lookback: datetime.timedelta) -> str: + return str((datetime.datetime.now(tz=datetime.UTC) - lookback).timestamp()) + + +def next_cursor(response: Any) -> str | None: + """Slack paginates by handing back a cursor; an empty one means the last page.""" + metadata: dict[str, Any] = response.get("response_metadata", {}) + return metadata.get("next_cursor") or None + + +class SlackChannelThreads: + """LiveMapView over a channel's threads: key = ``thread_ts``, value = `ThreadRef`.""" + + def __init__( + self, + client: AsyncWebClient, + limiter: RateLimiter, + channel: str, + *, + lookback: datetime.timedelta, + poll_interval: datetime.timedelta, + ) -> None: + self._client = client + self._limiter = limiter + self._channel = channel + self._lookback = lookback + self._poll_interval = poll_interval + self._guard = SingleWatcherGuard(f"SlackChannelThreads({channel})") + + async def _scan(self) -> AsyncIterator[tuple[str, ThreadRef]]: + cursor: str | None = None + oldest = _oldest_ts(self._lookback) + while True: + await self._limiter.acquire() + response = await self._client.conversations_history( + channel=self._channel, oldest=oldest, limit=200, cursor=cursor + ) + for message in response["messages"]: + if message.get("subtype") in SKIP_SUBTYPES: + continue + ts = message["ts"] + # A reply bumps latest_reply, an edit bumps edited.ts — take the + # larger so either one re-runs the thread. + revision = max( + ts, + message.get("latest_reply", ts), + message.get("edited", {}).get("ts", ts), + ) + yield ( + ts, + ThreadRef( + channel=self._channel, + thread_ts=ts, + revision=revision, + reply_count=int(message.get("reply_count", 0)), + ), + ) + cursor = next_cursor(response) + if cursor is None: + return + + def __aiter__(self) -> AsyncIterator[tuple[str, ThreadRef]]: + return self._scan() + + async def watch(self, subscriber: _Subscriber) -> None: + await _poll(subscriber, self._poll_interval, self._guard) + + +class SlackChannelFiles: + """LiveMapView over a channel's shared files: key = file id, value = `FileRef`.""" + + def __init__( + self, + client: AsyncWebClient, + limiter: RateLimiter, + channel: str, + *, + lookback: datetime.timedelta, + poll_interval: datetime.timedelta, + ) -> None: + self._client = client + self._limiter = limiter + self._channel = channel + self._lookback = lookback + self._poll_interval = poll_interval + self._guard = SingleWatcherGuard(f"SlackChannelFiles({channel})") + + async def _scan(self) -> AsyncIterator[tuple[str, FileRef]]: + cursor: str | None = None + ts_from = _oldest_ts(self._lookback) + while True: + await self._limiter.acquire() + response = await self._client.files_list( + channel=self._channel, ts_from=ts_from, limit=200, cursor=cursor + ) + for file in response["files"]: + if file.get("mode") not in INDEXABLE_FILE_MODES: + continue + url = file.get("url_private_download") or file.get("url_private") + if not url: + continue + yield ( + file["id"], + FileRef( + channel=self._channel, + file_id=file["id"], + name=file.get("name") or file.get("title") or file["id"], + mimetype=file.get("mimetype", ""), + size=int(file.get("size", 0)), + created=int(file.get("created", 0)), + url=url, + permalink=file.get("permalink", ""), + user=file.get("user"), + ), + ) + cursor = next_cursor(response) + if cursor is None: + return + + def __aiter__(self) -> AsyncIterator[tuple[str, FileRef]]: + return self._scan() + + async def watch(self, subscriber: _Subscriber) -> None: + await _poll(subscriber, self._poll_interval, self._guard) diff --git a/slack_index/threads.py b/slack_index/threads.py new file mode 100644 index 0000000..c0db270 --- /dev/null +++ b/slack_index/threads.py @@ -0,0 +1,73 @@ +"""One component per thread: fetch replies, render, chunk, embed.""" + +from __future__ import annotations + +import datetime +from typing import Any + +import cocoindex as coco +from cocoindex.connectors import lancedb + +from slack_index.chunking import ChunkMeta, declare_chunks +from slack_index.context import SLACK, SLACK_LIMIT +from slack_index.models import SlackChunk, ThreadRef +from slack_index.source import next_cursor +from slack_index.users import display_name + + +def thread_permalink(channel: str, thread_ts: str) -> str: + return f"https://slack.com/archives/{channel}/p{thread_ts.replace('.', '')}" + + +def ts_to_datetime(ts: str) -> datetime.datetime: + return datetime.datetime.fromtimestamp(float(ts), tz=datetime.UTC) + + +def speaker_id(message: dict[str, Any]) -> str | None: + return message.get("user") or message.get("bot_id") + + +async def fetch_replies(ref: ThreadRef) -> list[dict[str, Any]]: + client = coco.use_context(SLACK) + limiter = coco.use_context(SLACK_LIMIT) + messages: list[dict[str, Any]] = [] + cursor: str | None = None + while True: + await limiter.acquire() + response = await client.conversations_replies( + channel=ref.channel, ts=ref.thread_ts, limit=200, cursor=cursor + ) + messages.extend(response["messages"]) + cursor = next_cursor(response) + if cursor is None: + return messages + + +@coco.fn(memo=True) +async def process_thread( + ref: ThreadRef, + table: lancedb.TableTarget[SlackChunk], +) -> None: + messages = await fetch_replies(ref) + if not messages: + return + + # The whole thread is embedded as one document: a reply only means something + # next to the message it answers, and a chunk lifted out of it loses that. + lines: list[str] = [] + for message in messages: + speaker = await display_name(speaker_id(message)) + lines.append(f"**{speaker}**: {message.get('text', '')}") + + await declare_chunks( + "\n\n".join(lines), + ChunkMeta( + kind="message", + channel=ref.channel, + source_id=ref.thread_ts, + permalink=thread_permalink(ref.channel, ref.thread_ts), + author=await display_name(speaker_id(messages[0])), + posted_at=ts_to_datetime(ref.thread_ts), + ), + table, + ) diff --git a/slack_index/users.py b/slack_index/users.py new file mode 100644 index 0000000..0cc1cae --- /dev/null +++ b/slack_index/users.py @@ -0,0 +1,29 @@ +"""Slack user id -> display name, resolved once per user.""" + +from __future__ import annotations + +import cocoindex as coco +from slack_sdk.errors import SlackApiError + +from slack_index.context import SLACK, SLACK_LIMIT + +UNKNOWN_AUTHOR = "unknown" + + +@coco.fn(memo=True) +async def display_name(user_id: str | None) -> str: + """Memoized so a channel full of the same handful of people costs a few calls. + + Bot and app ids are not users, so Slack rejects them — their raw id is the + best label available. + """ + if not user_id: + return UNKNOWN_AUTHOR + await coco.use_context(SLACK_LIMIT).acquire() + try: + response = await coco.use_context(SLACK).users_info(user=user_id) + except SlackApiError: + return user_id + user = response["user"] + profile = user.get("profile", {}) + return profile.get("display_name") or user.get("real_name") or user_id diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..86a07e9 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,51 @@ +"""A Slack client stub that replays recorded API responses.""" + +from __future__ import annotations + +import json +import pathlib +from typing import Any + +import pytest + +FIXTURES = pathlib.Path(__file__).parent / "fixtures" + + +def load_fixture(name: str) -> dict[str, Any]: + return json.loads((FIXTURES / f"{name}.json").read_text()) + + +class FakeSlackClient: + """Returns the queued response per method call and records the call kwargs.""" + + def __init__(self, **responses: list[dict[str, Any]]) -> None: + self._responses = {name: list(pages) for name, pages in responses.items()} + self.calls: list[tuple[str, dict[str, Any]]] = [] + + def _next(self, method: str, kwargs: dict[str, Any]) -> dict[str, Any]: + self.calls.append((method, kwargs)) + pages = self._responses[method] + if not pages: + raise AssertionError(f"{method} called more times than there are pages") + return pages.pop(0) + + async def conversations_history(self, **kwargs: Any) -> dict[str, Any]: + return self._next("conversations_history", kwargs) + + async def files_list(self, **kwargs: Any) -> dict[str, Any]: + return self._next("files_list", kwargs) + + +@pytest.fixture +def history_client() -> FakeSlackClient: + return FakeSlackClient( + conversations_history=[ + load_fixture("history_page1"), + load_fixture("history_page2"), + ] + ) + + +@pytest.fixture +def files_client() -> FakeSlackClient: + return FakeSlackClient(files_list=[load_fixture("files_list")]) diff --git a/tests/fixtures/files_list.json b/tests/fixtures/files_list.json new file mode 100644 index 0000000..b2847b4 --- /dev/null +++ b/tests/fixtures/files_list.json @@ -0,0 +1,38 @@ +{ + "ok": true, + "files": [ + { + "id": "F0TEXT", + "mode": "hosted", + "name": "runbook.md", + "title": "runbook", + "mimetype": "text/markdown", + "size": 2048, + "created": 1758470500, + "user": "U0LEAD", + "url_private_download": "https://files.slack.com/files-pri/T0-F0TEXT/download/runbook.md", + "permalink": "https://example.slack.com/files/U0LEAD/F0TEXT/runbook.md" + }, + { + "id": "F0EXTERNAL", + "mode": "external", + "name": "design.fig", + "mimetype": "application/octet-stream", + "size": 100, + "created": 1758470600, + "user": "U0DEV", + "permalink": "https://example.slack.com/files/U0DEV/F0EXTERNAL/design.fig" + }, + { + "id": "F0NOURL", + "mode": "hosted", + "name": "gone.txt", + "mimetype": "text/plain", + "size": 10, + "created": 1758470700, + "user": "U0DEV", + "permalink": "https://example.slack.com/files/U0DEV/F0NOURL/gone.txt" + } + ], + "response_metadata": { "next_cursor": "" } +} diff --git a/tests/fixtures/history_page1.json b/tests/fixtures/history_page1.json new file mode 100644 index 0000000..248d03f --- /dev/null +++ b/tests/fixtures/history_page1.json @@ -0,0 +1,30 @@ +{ + "ok": true, + "messages": [ + { + "type": "message", + "user": "U0LEAD", + "ts": "1758470400.000100", + "text": "deploy rollback runbook?", + "thread_ts": "1758470400.000100", + "reply_count": 3, + "latest_reply": "1758470999.000500" + }, + { + "type": "message", + "subtype": "channel_join", + "user": "U0NEW", + "ts": "1758470300.000100", + "text": "has joined the channel" + }, + { + "type": "message", + "user": "U0DEV", + "ts": "1758470200.000100", + "text": "staging is green", + "edited": { "user": "U0DEV", "ts": "1758470260.000000" } + } + ], + "has_more": true, + "response_metadata": { "next_cursor": "cursor-page-2" } +} diff --git a/tests/fixtures/history_page2.json b/tests/fixtures/history_page2.json new file mode 100644 index 0000000..87b16d4 --- /dev/null +++ b/tests/fixtures/history_page2.json @@ -0,0 +1,13 @@ +{ + "ok": true, + "messages": [ + { + "type": "message", + "user": "U0OPS", + "ts": "1758460000.000100", + "text": "postmortem doc is up" + } + ], + "has_more": false, + "response_metadata": { "next_cursor": "" } +} diff --git a/tests/test_files.py b/tests/test_files.py new file mode 100644 index 0000000..dac2a28 --- /dev/null +++ b/tests/test_files.py @@ -0,0 +1,41 @@ +"""What gets read, and what gets skipped before any download happens.""" + +from __future__ import annotations + +import asyncio + +from slack_index.files import extract_text, is_text +from slack_index.models import FileRef + +MAX_BYTES = 1024 + + +def _ref(mimetype: str, size: int) -> FileRef: + return FileRef( + channel="C0TEST", + file_id="F0TEST", + name="sample", + mimetype=mimetype, + size=size, + created=1758470500, + url="https://files.slack.com/files-pri/T0-F0TEST/download/sample", + permalink="https://example.slack.com/files/U0LEAD/F0TEST/sample", + user="U0LEAD", + ) + + +def test_is_text_covers_text_and_structured_formats() -> None: + assert is_text("text/markdown") + assert is_text("application/json") + assert not is_text("application/pdf") + assert not is_text("image/png") + + +def test_binary_file_is_skipped_without_downloading() -> None: + assert asyncio.run(extract_text(_ref("application/pdf", 10), MAX_BYTES)) is None + + +def test_oversized_file_is_skipped_without_downloading() -> None: + assert ( + asyncio.run(extract_text(_ref("text/plain", MAX_BYTES + 1), MAX_BYTES)) is None + ) diff --git a/tests/test_source.py b/tests/test_source.py new file mode 100644 index 0000000..8c6394e --- /dev/null +++ b/tests/test_source.py @@ -0,0 +1,102 @@ +"""The channel scans decide what gets indexed and what gets dropped.""" + +from __future__ import annotations + +import asyncio +import datetime +from typing import Any, TypeVar + +from cocoindex.resources.rate_limit import RateLimiter + +from slack_index.models import FileRef, ThreadRef +from slack_index.source import SlackChannelFiles, SlackChannelThreads +from tests.conftest import FakeSlackClient + +CHANNEL = "C0TEST" +T = TypeVar("T") + + +def _limiter() -> RateLimiter: + # Fast enough that the tests never actually wait on it. + return RateLimiter(10_000) + + +def _collect(source: Any) -> list[tuple[str, Any]]: + async def run() -> list[tuple[str, Any]]: + return [item async for item in source] + + return asyncio.run(run()) + + +def _threads(client: FakeSlackClient) -> SlackChannelThreads: + return SlackChannelThreads( + client, # type: ignore[arg-type] + _limiter(), + CHANNEL, + lookback=datetime.timedelta(days=30), + poll_interval=datetime.timedelta(seconds=60), + ) + + +def _files(client: FakeSlackClient) -> SlackChannelFiles: + return SlackChannelFiles( + client, # type: ignore[arg-type] + _limiter(), + CHANNEL, + lookback=datetime.timedelta(days=30), + poll_interval=datetime.timedelta(seconds=60), + ) + + +def test_thread_scan_paginates_and_skips_noise(history_client: FakeSlackClient) -> None: + items = _collect(_threads(history_client)) + + assert [key for key, _ in items] == [ + "1758470400.000100", + "1758470200.000100", + "1758460000.000100", + ] + cursors = [kwargs.get("cursor") for _, kwargs in history_client.calls] + assert cursors == [None, "cursor-page-2"] + + +def test_thread_revision_tracks_replies_and_edits( + history_client: FakeSlackClient, +) -> None: + items = dict(_collect(_threads(history_client))) + + replied = items["1758470400.000100"] + assert replied == ThreadRef( + channel=CHANNEL, + thread_ts="1758470400.000100", + revision="1758470999.000500", + reply_count=3, + ) + + edited = items["1758470200.000100"] + assert edited.revision == "1758470260.000000" + assert edited.reply_count == 0 + + untouched = items["1758460000.000100"] + assert untouched.revision == untouched.thread_ts + + +def test_file_scan_keeps_only_fetchable_files(files_client: FakeSlackClient) -> None: + items = _collect(_files(files_client)) + + assert items == [ + ( + "F0TEXT", + FileRef( + channel=CHANNEL, + file_id="F0TEXT", + name="runbook.md", + mimetype="text/markdown", + size=2048, + created=1758470500, + url="https://files.slack.com/files-pri/T0-F0TEXT/download/runbook.md", + permalink="https://example.slack.com/files/U0LEAD/F0TEXT/runbook.md", + user="U0LEAD", + ), + ) + ] From 6a1ddf57cf0b95ce43b391674d437466e3808792 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 00:44:45 +0900 Subject: [PATCH 02/12] slack: fetch replies only for threads that have them A message without replies already arrives complete in the channel scan, so fetching it again spent one of the 50 Slack calls per minute on nothing: on the indexed channel that is 236 calls down to 10. Carrying the root text in ThreadRef also re-runs a thread when its message is edited, which the previous latest_reply comparison missed. --- slack_index/models.py | 2 ++ slack_index/source.py | 3 +++ slack_index/threads.py | 14 +++++++++++++- tests/test_source.py | 2 ++ tests/test_threads.py | 39 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 59 insertions(+), 1 deletion(-) create mode 100644 tests/test_threads.py diff --git a/slack_index/models.py b/slack_index/models.py index 2a28467..118c395 100644 --- a/slack_index/models.py +++ b/slack_index/models.py @@ -23,6 +23,8 @@ class ThreadRef: thread_ts: str revision: str reply_count: int + user: str | None + text: str @dataclass(frozen=True, slots=True) diff --git a/slack_index/source.py b/slack_index/source.py index 004a7fc..8941e5b 100644 --- a/slack_index/source.py +++ b/slack_index/source.py @@ -98,6 +98,9 @@ async def _scan(self) -> AsyncIterator[tuple[str, ThreadRef]]: thread_ts=ts, revision=revision, reply_count=int(message.get("reply_count", 0)), + user=message.get("user") or message.get("bot_id"), + # Carried so a message without replies needs no further call. + text=message.get("text", ""), ), ) cursor = next_cursor(response) diff --git a/slack_index/threads.py b/slack_index/threads.py index c0db270..1b8bbd6 100644 --- a/slack_index/threads.py +++ b/slack_index/threads.py @@ -27,6 +27,18 @@ def speaker_id(message: dict[str, Any]) -> str | None: return message.get("user") or message.get("bot_id") +def needs_replies(ref: ThreadRef) -> bool: + """A reply-less message is already complete in the scan, so fetching it again + would spend one of the 50 Slack calls per minute on nothing.""" + return ref.reply_count > 0 + + +async def thread_messages(ref: ThreadRef) -> list[dict[str, Any]]: + if not needs_replies(ref): + return [{"user": ref.user, "text": ref.text, "ts": ref.thread_ts}] + return await fetch_replies(ref) + + async def fetch_replies(ref: ThreadRef) -> list[dict[str, Any]]: client = coco.use_context(SLACK) limiter = coco.use_context(SLACK_LIMIT) @@ -48,7 +60,7 @@ async def process_thread( ref: ThreadRef, table: lancedb.TableTarget[SlackChunk], ) -> None: - messages = await fetch_replies(ref) + messages = await thread_messages(ref) if not messages: return diff --git a/tests/test_source.py b/tests/test_source.py index 8c6394e..2909707 100644 --- a/tests/test_source.py +++ b/tests/test_source.py @@ -71,6 +71,8 @@ def test_thread_revision_tracks_replies_and_edits( thread_ts="1758470400.000100", revision="1758470999.000500", reply_count=3, + user="U0LEAD", + text="deploy rollback runbook?", ) edited = items["1758470200.000100"] diff --git a/tests/test_threads.py b/tests/test_threads.py new file mode 100644 index 0000000..d8196de --- /dev/null +++ b/tests/test_threads.py @@ -0,0 +1,39 @@ +"""A reply-less message must not cost a Slack call.""" + +from __future__ import annotations + +import asyncio + +from slack_index.models import ThreadRef +from slack_index.threads import needs_replies, thread_messages, thread_permalink + + +def _ref(reply_count: int) -> ThreadRef: + return ThreadRef( + channel="C0TEST", + thread_ts="1758470400.000100", + revision="1758470400.000100", + reply_count=reply_count, + user="U0LEAD", + text="staging is green", + ) + + +def test_lone_message_is_rendered_from_the_scan() -> None: + ref = _ref(0) + assert not needs_replies(ref) + # No Slack client in context: reaching the API here would raise. + assert asyncio.run(thread_messages(ref)) == [ + {"user": "U0LEAD", "text": "staging is green", "ts": ref.thread_ts} + ] + + +def test_thread_with_replies_still_needs_fetching() -> None: + assert needs_replies(_ref(3)) + + +def test_permalink_points_at_the_thread() -> None: + assert ( + thread_permalink("C0TEST", "1758470400.000100") + == "https://slack.com/archives/C0TEST/p1758470400000100" + ) From 90a73d524091dda94a3a6b33da6ab44c1fa5b8d6 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 01:19:54 +0900 Subject: [PATCH 03/12] slack: add a retrieval scorer and a question-set format MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every retrieval change from here — a different embedding model, grouping short messages, a reranker — is a guess until the same questions are re-scored against it, so the recall@k/MRR scorer lands before any of them. The labels quote an internal channel, so the question set itself stays untracked and only its format and tags are committed. Search moves into its own module so the CLI and the scorer rank identically, including collapsing a source's chunks into a single result. --- evals/.gitignore | 1 + pyproject.toml | 8 ++- slack_index/evals.py | 131 ++++++++++++++++++++++++++++++++++++++++++ slack_index/query.py | 25 +++----- slack_index/search.py | 70 ++++++++++++++++++++++ uv.lock | 13 +++++ 6 files changed, 229 insertions(+), 19 deletions(-) create mode 100644 evals/.gitignore create mode 100644 slack_index/evals.py create mode 100644 slack_index/search.py diff --git a/evals/.gitignore b/evals/.gitignore new file mode 100644 index 0000000..72e8ffc --- /dev/null +++ b/evals/.gitignore @@ -0,0 +1 @@ +* diff --git a/pyproject.toml b/pyproject.toml index 2b2336a..9396d83 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -7,11 +7,17 @@ dependencies = [ "aiohttp>=3.14.3", "cocoindex[lancedb,sentence-transformers]>=1.0.24", "numpy>=2.5.3", + "pyyaml>=6.0.3", "slack-sdk>=3.44.1", ] [dependency-groups] -dev = ["mypy>=1.14", "ipykernel>=7.3.0", "pytest>=9.1.1"] +dev = [ + "mypy>=1.14", + "ipykernel>=7.3.0", + "pytest>=9.1.1", + "types-pyyaml>=6.0.12.20260906", +] [tool.uv] package = false diff --git a/slack_index/evals.py b/slack_index/evals.py new file mode 100644 index 0000000..83c86a0 --- /dev/null +++ b/slack_index/evals.py @@ -0,0 +1,131 @@ +"""Score the index against a hand-labelled question set. + + python -m slack_index.evals + python -m slack_index.evals --questions evals/questions.yaml --top-k 10 + +Every retrieval change — a different embedding model, grouping, a reranker — is +judged by re-running this against the same questions. +""" + +from __future__ import annotations + +import argparse +import asyncio +import pathlib +from collections import defaultdict +from dataclasses import dataclass, field + +import yaml + +from slack_index.search import Searcher + +DEFAULT_QUESTIONS = pathlib.Path("evals/questions.yaml") +DEFAULT_KS = (1, 3, 10) + + +@dataclass(frozen=True, slots=True) +class Question: + question: str + expected: frozenset[str] + tags: tuple[str, ...] + + +@dataclass +class Report: + total: int = 0 + hits: dict[int, int] = field(default_factory=dict) + reciprocal_rank_sum: float = 0.0 + misses: list[str] = field(default_factory=list) + + def recall(self, k: int) -> float: + return self.hits.get(k, 0) / self.total if self.total else 0.0 + + @property + def mrr(self) -> float: + return self.reciprocal_rank_sum / self.total if self.total else 0.0 + + +def load_questions(path: pathlib.Path) -> list[Question]: + raw = yaml.safe_load(path.read_text()) or [] + return [ + Question( + question=item["question"], + expected=frozenset(item["expected"]), + tags=tuple(item.get("tags", ())), + ) + for item in raw + ] + + +def _record( + report: Report, ks: tuple[int, ...], rank: int | None, question: str +) -> None: + report.total += 1 + if rank is None: + report.misses.append(question) + return + report.reciprocal_rank_sum += 1.0 / rank + for k in ks: + if rank <= k: + report.hits[k] = report.hits.get(k, 0) + 1 + + +async def evaluate( + searcher: Searcher, + questions: list[Question], + *, + ks: tuple[int, ...] = DEFAULT_KS, + top_k: int | None = None, +) -> tuple[Report, dict[str, Report]]: + limit = top_k or max(ks) + overall = Report() + per_tag: dict[str, Report] = defaultdict(Report) + + for question in questions: + hits = await searcher.search(question.question, limit) + rank = next( + ( + i + for i, hit in enumerate(hits, start=1) + if hit.source_id in question.expected + ), + None, + ) + _record(overall, ks, rank, question.question) + for tag in question.tags: + _record(per_tag[tag], ks, rank, question.question) + return overall, dict(per_tag) + + +def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: + cells = " ".join( + f"recall@{k}: {report.recall(k):.2f} ({report.hits.get(k, 0)}/{report.total})" + for k in ks + ) + return f"{name:<10} {cells} MRR: {report.mrr:.2f}" + + +async def run(questions_path: pathlib.Path, top_k: int) -> None: + questions = load_questions(questions_path) + searcher = await Searcher.open() + overall, per_tag = await evaluate(searcher, questions, top_k=top_k) + + print(_format("overall", overall, DEFAULT_KS)) + for tag in sorted(per_tag): + print(_format(tag, per_tag[tag], DEFAULT_KS)) + if overall.misses: + print(f"\nmissed ({len(overall.misses)}):") + for question in overall.misses: + print(f" {question}") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--questions", type=pathlib.Path, default=DEFAULT_QUESTIONS) + parser.add_argument("--top-k", type=int, default=max(DEFAULT_KS)) + args = parser.parse_args() + asyncio.run(run(args.questions, args.top_k)) + + +if __name__ == "__main__": + main() diff --git a/slack_index/query.py b/slack_index/query.py index e8354ca..ba4c701 100644 --- a/slack_index/query.py +++ b/slack_index/query.py @@ -5,34 +5,23 @@ import asyncio import sys -from cocoindex.connectors import lancedb -from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder - -from slack_index import config +from slack_index.search import Searcher TOP_K = 5 -async def search(query: str, *, top_k: int = TOP_K) -> None: - settings = config.Settings.from_env() - embedder = SentenceTransformerEmbedder(settings.embed_model) - conn = await lancedb.connect_async(str(config.LANCEDB_URI)) - table = await conn.open_table(config.TABLE_NAME) - - request = await table.search( - await embedder.embed(query), vector_column_name="embedding" - ) - for row in await request.limit(top_k).to_list(): - score = 1.0 - row["_distance"] - print(f"[{score:.3f}] {row['kind']} by {row['author']} — {row['permalink']}") - print(f" {row['text'][:300]}") +async def run(query: str, *, top_k: int = TOP_K) -> None: + searcher = await Searcher.open() + for hit in await searcher.search(query, top_k): + print(f"[{hit.score:.3f}] {hit.kind} by {hit.author} — {hit.permalink}") + print(f" {hit.text[:300]}") print("---") def main() -> None: if len(sys.argv) < 2: sys.exit("usage: python -m slack_index.query ") - asyncio.run(search(" ".join(sys.argv[1:]))) + asyncio.run(run(" ".join(sys.argv[1:]))) if __name__ == "__main__": diff --git a/slack_index/search.py b/slack_index/search.py new file mode 100644 index 0000000..24101da --- /dev/null +++ b/slack_index/search.py @@ -0,0 +1,70 @@ +"""Retrieval over the built index, shared by the query CLI and the evals.""" + +from __future__ import annotations + +from dataclasses import dataclass + +from cocoindex.connectors import lancedb +from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder +from lancedb.table import AsyncTable + +from slack_index import config + +# One source can own many chunks; over-fetch so that collapsing them still +# leaves top_k distinct sources. +_CANDIDATE_FACTOR = 5 + + +@dataclass(frozen=True, slots=True) +class Hit: + source_id: str + kind: str + author: str + permalink: str + text: str + score: float + + +class Searcher: + """Holds the embedder and the open table so a run of queries pays for them once.""" + + def __init__( + self, table: AsyncTable, embedder: SentenceTransformerEmbedder + ) -> None: + self._table = table + self._embedder = embedder + + @classmethod + async def open(cls) -> Searcher: + settings = config.Settings.from_env() + conn = await lancedb.connect_async(str(config.LANCEDB_URI)) + table = await conn.open_table(config.TABLE_NAME) + return cls(table, SentenceTransformerEmbedder(settings.embed_model)) + + async def search(self, query: str, top_k: int) -> list[Hit]: + """Best chunk per source, ranked — a thread that chunked into ten pieces + should occupy one result slot, not ten.""" + vector = await self._embedder.embed(query) + request = await self._table.search(vector, vector_column_name="embedding") + rows = await request.limit(top_k * _CANDIDATE_FACTOR).to_list() + + hits: list[Hit] = [] + seen: set[str] = set() + for row in rows: + source_id = row["source_id"] + if source_id in seen: + continue + seen.add(source_id) + hits.append( + Hit( + source_id=source_id, + kind=row["kind"], + author=row["author"], + permalink=row["permalink"], + text=row["text"], + score=1.0 - row["_distance"], + ) + ) + if len(hits) == top_k: + break + return hits diff --git a/uv.lock b/uv.lock index 9688fe8..86e804f 100644 --- a/uv.lock +++ b/uv.lock @@ -337,6 +337,7 @@ dependencies = [ { name = "aiohttp" }, { name = "cocoindex", extra = ["lancedb", "sentence-transformers"] }, { name = "numpy" }, + { name = "pyyaml" }, { name = "slack-sdk" }, ] @@ -345,6 +346,7 @@ dev = [ { name = "ipykernel" }, { name = "mypy" }, { name = "pytest" }, + { name = "types-pyyaml" }, ] [package.metadata] @@ -352,6 +354,7 @@ requires-dist = [ { name = "aiohttp", specifier = ">=3.14.3" }, { name = "cocoindex", extras = ["lancedb", "sentence-transformers"], specifier = ">=1.0.24" }, { name = "numpy", specifier = ">=2.5.3" }, + { name = "pyyaml", specifier = ">=6.0.3" }, { name = "slack-sdk", specifier = ">=3.44.1" }, ] @@ -360,6 +363,7 @@ dev = [ { name = "ipykernel", specifier = ">=7.3.0" }, { name = "mypy", specifier = ">=1.14" }, { name = "pytest", specifier = ">=9.1.1" }, + { name = "types-pyyaml", specifier = ">=6.0.12.20260906" }, ] [[package]] @@ -2182,6 +2186,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/dc/bf/205d0004930ede8f542fb58f601526fccf4ae7626075ca1e6c4de5d3d652/typer-0.27.2-py3-none-any.whl", hash = "sha256:b3a5fc4342d5fc8fda8fc3010b1cf117e9249aab7fae800c2eff62fd3842d97d", size = 123130, upload-time = "2026-08-28T10:26:53.752Z" }, ] +[[package]] +name = "types-pyyaml" +version = "6.0.12.20260906" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/90/6e/abec85b9013db5b934b0280a6dd104904d84f7bcbaab2e2f3def87ac7463/types_pyyaml-6.0.12.20260906.tar.gz", hash = "sha256:f59c1cc05010b833d2d72287bbaa72610106b28d42d89a907313117faba85212", size = 18649, upload-time = "2026-09-06T06:35:35.362Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/15/c0/fc0644b7ddcfb969e95845837143cb5173ddd6e06ee4ba5fc493cd9329b7/types_pyyaml-6.0.12.20260906-py3-none-any.whl", hash = "sha256:bca893ff0d51df5c9053137d5d0e6ccd36e939a196356f1d5c16372422f5137b", size = 21282, upload-time = "2026-09-06T06:35:34.372Z" }, +] + [[package]] name = "typing-extensions" version = "4.16.0" From 0eadd4512a9ec01a84cce7c98241832393683f18 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:07:08 +0900 Subject: [PATCH 04/12] slack: index the channel from its first message The 90-day window was hiding more than half the channel: the full history is 443 messages back to 2025-08, and re-scoring the same questions over it drops MRR from 0.78 to 0.74, which is the honest number. An unset or zero SLACK_INDEX_LOOKBACK_DAYS now means no cutoff, and Slack reads a missing `oldest` the same way. The scorer also learns about questions the channel cannot answer: an empty `expected` skips recall and reports the top hit's score instead, because an index that answers confidently about something never discussed is its own kind of failure. --- .env.example | 1 + slack_index/config.py | 13 +++++++++---- slack_index/evals.py | 42 ++++++++++++++++++++++++++++++++++++++---- slack_index/source.py | 13 ++++++++----- tests/test_source.py | 7 ++++++- 5 files changed, 62 insertions(+), 14 deletions(-) diff --git a/.env.example b/.env.example index 3b9c195..63be9f3 100644 --- a/.env.example +++ b/.env.example @@ -7,6 +7,7 @@ SLACK_CHANNEL_IDS= # Optional #SLACK_INDEX_EMBED_MODEL=sentence-transformers/all-MiniLM-L6-v2 +# Unset or 0 indexes the channel from its first message. #SLACK_INDEX_LOOKBACK_DAYS=90 #SLACK_INDEX_POLL_SECONDS=60 #SLACK_INDEX_MAX_FILE_BYTES=5242880 diff --git a/slack_index/config.py b/slack_index/config.py index 65989ee..45b4edc 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -27,7 +27,8 @@ class Settings: channel_ids: tuple[str, ...] embed_model: str - lookback: datetime.timedelta + # None indexes the channel from its first message. + lookback: datetime.timedelta | None poll_interval: datetime.timedelta max_file_bytes: int @@ -45,9 +46,7 @@ def from_env(cls) -> Settings: embed_model=os.environ.get( "SLACK_INDEX_EMBED_MODEL", "sentence-transformers/all-MiniLM-L6-v2" ), - lookback=datetime.timedelta( - days=float(os.environ.get("SLACK_INDEX_LOOKBACK_DAYS", "90")) - ), + lookback=_lookback_from_env(), poll_interval=datetime.timedelta( seconds=float(os.environ.get("SLACK_INDEX_POLL_SECONDS", "60")) ), @@ -57,6 +56,12 @@ def from_env(cls) -> Settings: ) +def _lookback_from_env() -> datetime.timedelta | None: + """Unset or 0 means the whole channel history.""" + days = float(os.environ.get("SLACK_INDEX_LOOKBACK_DAYS", "0")) + return datetime.timedelta(days=days) if days > 0 else None + + def bot_token() -> str: token = os.environ.get("SLACK_BOT_TOKEN") if not token: diff --git a/slack_index/evals.py b/slack_index/evals.py index 83c86a0..a2e7390 100644 --- a/slack_index/evals.py +++ b/slack_index/evals.py @@ -12,6 +12,7 @@ import argparse import asyncio import pathlib +import statistics from collections import defaultdict from dataclasses import dataclass, field @@ -29,6 +30,12 @@ class Question: expected: frozenset[str] tags: tuple[str, ...] + @property + def answerable(self) -> bool: + """An empty `expected` means the channel has no answer: the question is + there to show what the index does when nothing is relevant.""" + return bool(self.expected) + @dataclass class Report: @@ -50,7 +57,7 @@ def load_questions(path: pathlib.Path) -> list[Question]: return [ Question( question=item["question"], - expected=frozenset(item["expected"]), + expected=frozenset(item.get("expected") or ()), tags=tuple(item.get("tags", ())), ) for item in raw @@ -70,19 +77,35 @@ def _record( report.hits[k] = report.hits.get(k, 0) + 1 +@dataclass +class Scores: + """Top-1 similarity, split by whether an answer exists at all. If the two + overlap, no score cutoff can stop the index from answering confidently about + something the channel never discussed.""" + + answerable: list[float] = field(default_factory=list) + unanswerable: list[float] = field(default_factory=list) + + async def evaluate( searcher: Searcher, questions: list[Question], *, ks: tuple[int, ...] = DEFAULT_KS, top_k: int | None = None, -) -> tuple[Report, dict[str, Report]]: +) -> tuple[Report, dict[str, Report], Scores]: limit = top_k or max(ks) overall = Report() per_tag: dict[str, Report] = defaultdict(Report) + scores = Scores() for question in questions: hits = await searcher.search(question.question, limit) + top_score = hits[0].score if hits else 0.0 + if not question.answerable: + scores.unanswerable.append(top_score) + continue + scores.answerable.append(top_score) rank = next( ( i @@ -94,7 +117,7 @@ async def evaluate( _record(overall, ks, rank, question.question) for tag in question.tags: _record(per_tag[tag], ks, rank, question.question) - return overall, dict(per_tag) + return overall, dict(per_tag), scores def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: @@ -108,11 +131,22 @@ def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: async def run(questions_path: pathlib.Path, top_k: int) -> None: questions = load_questions(questions_path) searcher = await Searcher.open() - overall, per_tag = await evaluate(searcher, questions, top_k=top_k) + overall, per_tag, scores = await evaluate(searcher, questions, top_k=top_k) print(_format("overall", overall, DEFAULT_KS)) for tag in sorted(per_tag): print(_format(tag, per_tag[tag], DEFAULT_KS)) + + if scores.unanswerable: + answerable = sorted(scores.answerable) + p25 = answerable[len(answerable) // 4] + print( + f"\ntop-1 score answerable: min {min(answerable):.3f}," + f" p25 {p25:.3f}, median {statistics.median(answerable):.3f}" + f" | unanswerable ({len(scores.unanswerable)}):" + f" median {statistics.median(scores.unanswerable):.3f}," + f" max {max(scores.unanswerable):.3f}" + ) if overall.misses: print(f"\nmissed ({len(overall.misses)}):") for question in overall.misses: diff --git a/slack_index/source.py b/slack_index/source.py index 8941e5b..e9f17d7 100644 --- a/slack_index/source.py +++ b/slack_index/source.py @@ -43,7 +43,10 @@ async def _poll( await subscriber.update_all() -def _oldest_ts(lookback: datetime.timedelta) -> str: +def oldest_ts(lookback: datetime.timedelta | None) -> str | None: + """Slack treats a missing `oldest` as "from the beginning".""" + if lookback is None: + return None return str((datetime.datetime.now(tz=datetime.UTC) - lookback).timestamp()) @@ -62,7 +65,7 @@ def __init__( limiter: RateLimiter, channel: str, *, - lookback: datetime.timedelta, + lookback: datetime.timedelta | None, poll_interval: datetime.timedelta, ) -> None: self._client = client @@ -74,7 +77,7 @@ def __init__( async def _scan(self) -> AsyncIterator[tuple[str, ThreadRef]]: cursor: str | None = None - oldest = _oldest_ts(self._lookback) + oldest = oldest_ts(self._lookback) while True: await self._limiter.acquire() response = await self._client.conversations_history( @@ -123,7 +126,7 @@ def __init__( limiter: RateLimiter, channel: str, *, - lookback: datetime.timedelta, + lookback: datetime.timedelta | None, poll_interval: datetime.timedelta, ) -> None: self._client = client @@ -135,7 +138,7 @@ def __init__( async def _scan(self) -> AsyncIterator[tuple[str, FileRef]]: cursor: str | None = None - ts_from = _oldest_ts(self._lookback) + ts_from = oldest_ts(self._lookback) while True: await self._limiter.acquire() response = await self._client.files_list( diff --git a/tests/test_source.py b/tests/test_source.py index 2909707..cdee079 100644 --- a/tests/test_source.py +++ b/tests/test_source.py @@ -9,7 +9,7 @@ from cocoindex.resources.rate_limit import RateLimiter from slack_index.models import FileRef, ThreadRef -from slack_index.source import SlackChannelFiles, SlackChannelThreads +from slack_index.source import SlackChannelFiles, SlackChannelThreads, oldest_ts from tests.conftest import FakeSlackClient CHANNEL = "C0TEST" @@ -102,3 +102,8 @@ def test_file_scan_keeps_only_fetchable_files(files_client: FakeSlackClient) -> ), ) ] + + +def test_no_lookback_means_no_cutoff() -> None: + assert oldest_ts(None) is None + assert oldest_ts(datetime.timedelta(days=1)) is not None From 85f0c9356c0ac4844aac4193a2c7330cf238bb95 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:15:04 +0900 Subject: [PATCH 05/12] slack: group consecutive messages into one conversation A message that is only a date is an answer, not a document: on its own it is unretrievable, and a five-character vector sits near every query. Messages within 15 minutes of each other now form one document, while a threaded message stays its own, which takes 443 documents down to 166 and the median document from 29 to 126 characters. Scored over the same questions, MRR goes 0.74 to 0.79 and the questions whose answer needs its neighbour go 0.48 to 0.73, with nothing regressing. Rows carry the timestamps they cover so a label written against a single message still matches the conversation that swallowed it, which keeps the question set independent of how the grouping is tuned. --- slack_index/chunking.py | 4 ++ slack_index/config.py | 6 +++ slack_index/evals.py | 2 +- slack_index/files.py | 1 + slack_index/models.py | 34 +++++++++++---- slack_index/search.py | 2 + slack_index/source.py | 91 ++++++++++++++++++++++++++++++++++------- slack_index/threads.py | 53 +++++++++++------------- tests/test_grouping.py | 69 +++++++++++++++++++++++++++++++ tests/test_source.py | 40 ++++++++---------- tests/test_threads.py | 39 ------------------ 11 files changed, 227 insertions(+), 114 deletions(-) create mode 100644 tests/test_grouping.py delete mode 100644 tests/test_threads.py diff --git a/slack_index/chunking.py b/slack_index/chunking.py index 204f02f..e0b882a 100644 --- a/slack_index/chunking.py +++ b/slack_index/chunking.py @@ -25,6 +25,9 @@ class ChunkMeta: kind: str channel: str source_id: str + # Every source-side id this document answers for — a window covers several + # message timestamps, so a lookup by any one of them must find it. + covered: str permalink: str author: str posted_at: datetime.datetime @@ -43,6 +46,7 @@ async def _declare_chunk( kind=meta.kind, channel=meta.channel, source_id=meta.source_id, + covered=meta.covered, permalink=meta.permalink, author=meta.author, posted_at=meta.posted_at, diff --git a/slack_index/config.py b/slack_index/config.py index 45b4edc..e6f3c69 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -16,6 +16,12 @@ LANCEDB_URI = VAR_DIR / "lancedb" TABLE_NAME = "slack_chunks" + +# Consecutive messages inside this gap belong to the same conversation. Caps stop a +# busy afternoon from collapsing into one undifferentiated document. +WINDOW_GAP = datetime.timedelta(minutes=15) +WINDOW_MAX_MESSAGES = 20 +WINDOW_MAX_CHARS = 2000 CHUNK_SIZE = 1200 CHUNK_OVERLAP = 200 diff --git a/slack_index/evals.py b/slack_index/evals.py index a2e7390..dc4338c 100644 --- a/slack_index/evals.py +++ b/slack_index/evals.py @@ -110,7 +110,7 @@ async def evaluate( ( i for i, hit in enumerate(hits, start=1) - if hit.source_id in question.expected + if question.expected & ({hit.source_id} | hit.covered) ), None, ) diff --git a/slack_index/files.py b/slack_index/files.py index e3fd52b..8279aa5 100644 --- a/slack_index/files.py +++ b/slack_index/files.py @@ -74,6 +74,7 @@ async def process_file( kind="file", channel=ref.channel, source_id=ref.file_id, + covered=ref.file_id, permalink=ref.permalink, author=await display_name(ref.user), posted_at=datetime.datetime.fromtimestamp(ref.created, tz=datetime.UTC), diff --git a/slack_index/models.py b/slack_index/models.py index 118c395..060a7ed 100644 --- a/slack_index/models.py +++ b/slack_index/models.py @@ -12,19 +12,36 @@ @dataclass(frozen=True, slots=True) -class ThreadRef: - """A thread as seen by the channel scan. +class Message: + """A top-level message as the channel scan saw it.""" - Every field takes part in change detection: when the scan reports a different - value the thread's component re-runs, and nothing else does. + ts: str + user: str | None + text: str + + +@dataclass(frozen=True, slots=True) +class ConversationRef: + """One indexable conversation: a thread, or a run of consecutive messages. + + Every field takes part in change detection, so a new reply, an edit or a + neighbour joining the run re-runs this conversation and nothing else. """ channel: str - thread_ts: str + start_ts: str revision: str reply_count: int - user: str | None - text: str + messages: tuple[Message, ...] + + @property + def is_thread(self) -> bool: + """Threads carry their replies in Slack, not in the scan.""" + return self.reply_count > 0 + + @property + def covered_ts(self) -> tuple[str, ...]: + return tuple(message.ts for message in self.messages) @dataclass(frozen=True, slots=True) @@ -44,12 +61,13 @@ class FileRef: @dataclass class SlackChunk: - """One embedded chunk — of a thread transcript or of a shared file.""" + """One embedded chunk — of a conversation or of a shared file.""" id: int kind: str channel: str source_id: str + covered: str permalink: str author: str posted_at: datetime.datetime diff --git a/slack_index/search.py b/slack_index/search.py index 24101da..73f53e7 100644 --- a/slack_index/search.py +++ b/slack_index/search.py @@ -18,6 +18,7 @@ @dataclass(frozen=True, slots=True) class Hit: source_id: str + covered: frozenset[str] kind: str author: str permalink: str @@ -58,6 +59,7 @@ async def search(self, query: str, top_k: int) -> list[Hit]: hits.append( Hit( source_id=source_id, + covered=frozenset(row["covered"].split()), kind=row["kind"], author=row["author"], permalink=row["permalink"], diff --git a/slack_index/source.py b/slack_index/source.py index e9f17d7..c3ab7e7 100644 --- a/slack_index/source.py +++ b/slack_index/source.py @@ -16,7 +16,8 @@ from cocoindex.resources.rate_limit import RateLimiter from slack_sdk.web.async_client import AsyncWebClient -from slack_index.models import FileRef, ThreadRef +from slack_index.config import WINDOW_GAP, WINDOW_MAX_CHARS, WINDOW_MAX_MESSAGES +from slack_index.models import ConversationRef, FileRef, Message # Joins, leaves and topic changes carry no content worth searching. SKIP_SUBTYPES = frozenset( @@ -56,8 +57,67 @@ def next_cursor(response: Any) -> str | None: return metadata.get("next_cursor") or None +def group_messages( + channel: str, messages: list[Message], meta: dict[str, tuple[str, int]] +) -> list[ConversationRef]: + """Cut a channel's messages into indexable conversations. + + A message with replies is its own conversation, as Slack already grouped it. + Everything else is grouped with its neighbours: a message that is only a date, + or only "ok", is an answer rather than a document — on its own it is + unretrievable, and it means nothing without the message above it. + + `meta` maps a ts to its (revision, reply_count) from the scan. + """ + conversations: list[ConversationRef] = [] + run: list[Message] = [] + run_chars = 0 + + def flush() -> None: + nonlocal run, run_chars + if not run: + return + conversations.append( + ConversationRef( + channel=channel, + start_ts=run[0].ts, + revision=max(meta[m.ts][0] for m in run), + reply_count=0, + messages=tuple(run), + ) + ) + run = [] + run_chars = 0 + + for message in sorted(messages, key=lambda m: float(m.ts)): + revision, reply_count = meta[message.ts] + if reply_count > 0: + flush() + conversations.append( + ConversationRef( + channel=channel, + start_ts=message.ts, + revision=revision, + reply_count=reply_count, + messages=(message,), + ) + ) + continue + gap = float(message.ts) - float(run[-1].ts) if run else 0.0 + if run and ( + gap > WINDOW_GAP.total_seconds() + or len(run) >= WINDOW_MAX_MESSAGES + or run_chars + len(message.text) > WINDOW_MAX_CHARS + ): + flush() + run.append(message) + run_chars += len(message.text) + flush() + return conversations + + class SlackChannelThreads: - """LiveMapView over a channel's threads: key = ``thread_ts``, value = `ThreadRef`.""" + """LiveMapView over a channel's conversations: key = first ts, value = `ConversationRef`.""" def __init__( self, @@ -75,7 +135,9 @@ def __init__( self._poll_interval = poll_interval self._guard = SingleWatcherGuard(f"SlackChannelThreads({channel})") - async def _scan(self) -> AsyncIterator[tuple[str, ThreadRef]]: + async def _scan(self) -> AsyncIterator[tuple[str, ConversationRef]]: + messages: list[Message] = [] + meta: dict[str, tuple[str, int]] = {} cursor: str | None = None oldest = oldest_ts(self._lookback) while True: @@ -88,29 +150,28 @@ async def _scan(self) -> AsyncIterator[tuple[str, ThreadRef]]: continue ts = message["ts"] # A reply bumps latest_reply, an edit bumps edited.ts — take the - # larger so either one re-runs the thread. + # larger so either one re-runs the conversation. revision = max( ts, message.get("latest_reply", ts), message.get("edited", {}).get("ts", ts), ) - yield ( - ts, - ThreadRef( - channel=self._channel, - thread_ts=ts, - revision=revision, - reply_count=int(message.get("reply_count", 0)), + meta[ts] = (revision, int(message.get("reply_count", 0))) + messages.append( + Message( + ts=ts, user=message.get("user") or message.get("bot_id"), - # Carried so a message without replies needs no further call. text=message.get("text", ""), - ), + ) ) cursor = next_cursor(response) if cursor is None: - return + break + + for conversation in group_messages(self._channel, messages, meta): + yield conversation.start_ts, conversation - def __aiter__(self) -> AsyncIterator[tuple[str, ThreadRef]]: + def __aiter__(self) -> AsyncIterator[tuple[str, ConversationRef]]: return self._scan() async def watch(self, subscriber: _Subscriber) -> None: diff --git a/slack_index/threads.py b/slack_index/threads.py index 1b8bbd6..3454771 100644 --- a/slack_index/threads.py +++ b/slack_index/threads.py @@ -1,4 +1,4 @@ -"""One component per thread: fetch replies, render, chunk, embed.""" +"""One component per conversation: gather its messages, render, chunk, embed.""" from __future__ import annotations @@ -10,7 +10,7 @@ from slack_index.chunking import ChunkMeta, declare_chunks from slack_index.context import SLACK, SLACK_LIMIT -from slack_index.models import SlackChunk, ThreadRef +from slack_index.models import ConversationRef, Message, SlackChunk from slack_index.source import next_cursor from slack_index.users import display_name @@ -27,59 +27,56 @@ def speaker_id(message: dict[str, Any]) -> str | None: return message.get("user") or message.get("bot_id") -def needs_replies(ref: ThreadRef) -> bool: - """A reply-less message is already complete in the scan, so fetching it again - would spend one of the 50 Slack calls per minute on nothing.""" - return ref.reply_count > 0 - - -async def thread_messages(ref: ThreadRef) -> list[dict[str, Any]]: - if not needs_replies(ref): - return [{"user": ref.user, "text": ref.text, "ts": ref.thread_ts}] - return await fetch_replies(ref) - - -async def fetch_replies(ref: ThreadRef) -> list[dict[str, Any]]: +async def fetch_replies(ref: ConversationRef) -> list[Message]: client = coco.use_context(SLACK) limiter = coco.use_context(SLACK_LIMIT) - messages: list[dict[str, Any]] = [] + messages: list[Message] = [] cursor: str | None = None while True: await limiter.acquire() response = await client.conversations_replies( - channel=ref.channel, ts=ref.thread_ts, limit=200, cursor=cursor + channel=ref.channel, ts=ref.start_ts, limit=200, cursor=cursor + ) + messages.extend( + Message(ts=m["ts"], user=speaker_id(m), text=m.get("text", "")) + for m in response["messages"] ) - messages.extend(response["messages"]) cursor = next_cursor(response) if cursor is None: return messages +async def conversation_messages(ref: ConversationRef) -> list[Message]: + """A grouped run already arrived complete in the scan; only a thread needs a call.""" + if not ref.is_thread: + return list(ref.messages) + return await fetch_replies(ref) + + @coco.fn(memo=True) async def process_thread( - ref: ThreadRef, + ref: ConversationRef, table: lancedb.TableTarget[SlackChunk], ) -> None: - messages = await thread_messages(ref) + messages = await conversation_messages(ref) if not messages: return - # The whole thread is embedded as one document: a reply only means something - # next to the message it answers, and a chunk lifted out of it loses that. lines: list[str] = [] for message in messages: - speaker = await display_name(speaker_id(message)) - lines.append(f"**{speaker}**: {message.get('text', '')}") + speaker = await display_name(message.user) + lines.append(f"**{speaker}**: {message.text}") await declare_chunks( "\n\n".join(lines), ChunkMeta( kind="message", channel=ref.channel, - source_id=ref.thread_ts, - permalink=thread_permalink(ref.channel, ref.thread_ts), - author=await display_name(speaker_id(messages[0])), - posted_at=ts_to_datetime(ref.thread_ts), + source_id=ref.start_ts, + covered=" ".join(m.ts for m in messages), + permalink=thread_permalink(ref.channel, ref.start_ts), + author=await display_name(messages[0].user), + posted_at=ts_to_datetime(ref.start_ts), ), table, ) diff --git a/tests/test_grouping.py b/tests/test_grouping.py new file mode 100644 index 0000000..bc93065 --- /dev/null +++ b/tests/test_grouping.py @@ -0,0 +1,69 @@ +"""A message that only makes sense next to its neighbour must be indexed with it.""" + +from __future__ import annotations + +from slack_index.config import WINDOW_MAX_CHARS, WINDOW_MAX_MESSAGES +from slack_index.models import Message +from slack_index.source import group_messages + +CHANNEL = "C0TEST" +BASE = 1758470000.0 + + +def _run(*specs: tuple[float, str, int]) -> list: + """Each spec is (seconds after BASE, text, reply_count).""" + messages = [] + meta = {} + for offset, text, replies in specs: + ts = f"{BASE + offset:.6f}" + messages.append(Message(ts=ts, user="U0", text=text)) + meta[ts] = (ts, replies) + return group_messages(CHANNEL, messages, meta) + + +def test_question_and_its_answer_land_in_one_conversation() -> None: + conversations = _run((0, "when was the meeting?", 0), (20, "the 27th", 0)) + + assert len(conversations) == 1 + assert [m.text for m in conversations[0].messages] == [ + "when was the meeting?", + "the 27th", + ] + # The answer's own ts stays findable even though the key is the first message. + assert conversations[0].start_ts == f"{BASE:.6f}" + assert f"{BASE + 20:.6f}" in conversations[0].covered_ts + + +def test_a_long_silence_starts_a_new_conversation() -> None: + conversations = _run((0, "first", 0), (3600, "much later", 0)) + + assert [c.start_ts for c in conversations] == [f"{BASE:.6f}", f"{BASE + 3600:.6f}"] + + +def test_a_thread_stays_on_its_own_and_breaks_the_run() -> None: + conversations = _run((0, "before", 0), (10, "thread parent", 2), (20, "after", 0)) + + assert [(c.start_ts, c.is_thread, len(c.messages)) for c in conversations] == [ + (f"{BASE:.6f}", False, 1), + (f"{BASE + 10:.6f}", True, 1), + (f"{BASE + 20:.6f}", False, 1), + ] + + +def test_a_busy_stretch_is_capped() -> None: + conversations = _run(*[(i, "short", 0) for i in range(WINDOW_MAX_MESSAGES + 5)]) + + assert len(conversations) == 2 + assert len(conversations[0].messages) == WINDOW_MAX_MESSAGES + + +def test_a_long_message_does_not_swallow_the_next_one() -> None: + conversations = _run((0, "x" * WINDOW_MAX_CHARS, 0), (10, "next", 0)) + + assert [len(c.messages) for c in conversations] == [1, 1] + + +def test_revision_is_the_latest_within_the_conversation() -> None: + conversations = _run((0, "first", 0), (30, "later", 0)) + + assert conversations[0].revision == f"{BASE + 30:.6f}" diff --git a/tests/test_source.py b/tests/test_source.py index cdee079..4289ceb 100644 --- a/tests/test_source.py +++ b/tests/test_source.py @@ -1,4 +1,4 @@ -"""The channel scans decide what gets indexed and what gets dropped.""" +"""The channel scan decides what gets indexed, dropped, and grouped together.""" from __future__ import annotations @@ -8,7 +8,7 @@ from cocoindex.resources.rate_limit import RateLimiter -from slack_index.models import FileRef, ThreadRef +from slack_index.models import FileRef from slack_index.source import SlackChannelFiles, SlackChannelThreads, oldest_ts from tests.conftest import FakeSlackClient @@ -48,39 +48,38 @@ def _files(client: FakeSlackClient) -> SlackChannelFiles: ) -def test_thread_scan_paginates_and_skips_noise(history_client: FakeSlackClient) -> None: +def test_scan_paginates_and_skips_noise(history_client: FakeSlackClient) -> None: items = _collect(_threads(history_client)) + # Chronological, one conversation each: the three are minutes-to-hours apart. assert [key for key, _ in items] == [ - "1758470400.000100", - "1758470200.000100", "1758460000.000100", + "1758470200.000100", + "1758470400.000100", ] cursors = [kwargs.get("cursor") for _, kwargs in history_client.calls] assert cursors == [None, "cursor-page-2"] -def test_thread_revision_tracks_replies_and_edits( - history_client: FakeSlackClient, -) -> None: +def test_revision_tracks_replies_and_edits(history_client: FakeSlackClient) -> None: items = dict(_collect(_threads(history_client))) replied = items["1758470400.000100"] - assert replied == ThreadRef( - channel=CHANNEL, - thread_ts="1758470400.000100", - revision="1758470999.000500", - reply_count=3, - user="U0LEAD", - text="deploy rollback runbook?", - ) + assert replied.is_thread + assert replied.reply_count == 3 + assert replied.revision == "1758470999.000500" edited = items["1758470200.000100"] + assert not edited.is_thread assert edited.revision == "1758470260.000000" - assert edited.reply_count == 0 untouched = items["1758460000.000100"] - assert untouched.revision == untouched.thread_ts + assert untouched.revision == untouched.start_ts + + +def test_no_lookback_means_no_cutoff() -> None: + assert oldest_ts(None) is None + assert oldest_ts(datetime.timedelta(days=1)) is not None def test_file_scan_keeps_only_fetchable_files(files_client: FakeSlackClient) -> None: @@ -102,8 +101,3 @@ def test_file_scan_keeps_only_fetchable_files(files_client: FakeSlackClient) -> ), ) ] - - -def test_no_lookback_means_no_cutoff() -> None: - assert oldest_ts(None) is None - assert oldest_ts(datetime.timedelta(days=1)) is not None diff --git a/tests/test_threads.py b/tests/test_threads.py deleted file mode 100644 index d8196de..0000000 --- a/tests/test_threads.py +++ /dev/null @@ -1,39 +0,0 @@ -"""A reply-less message must not cost a Slack call.""" - -from __future__ import annotations - -import asyncio - -from slack_index.models import ThreadRef -from slack_index.threads import needs_replies, thread_messages, thread_permalink - - -def _ref(reply_count: int) -> ThreadRef: - return ThreadRef( - channel="C0TEST", - thread_ts="1758470400.000100", - revision="1758470400.000100", - reply_count=reply_count, - user="U0LEAD", - text="staging is green", - ) - - -def test_lone_message_is_rendered_from_the_scan() -> None: - ref = _ref(0) - assert not needs_replies(ref) - # No Slack client in context: reaching the API here would raise. - assert asyncio.run(thread_messages(ref)) == [ - {"user": "U0LEAD", "text": "staging is green", "ts": ref.thread_ts} - ] - - -def test_thread_with_replies_still_needs_fetching() -> None: - assert needs_replies(_ref(3)) - - -def test_permalink_points_at_the_thread() -> None: - assert ( - thread_permalink("C0TEST", "1758470400.000100") - == "https://slack.com/archives/C0TEST/p1758470400000100" - ) From 87d723561ea90a2b869a2446c4bac016d0d752de Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:56:49 +0900 Subject: [PATCH 06/12] slack: prepend an LLM summary to each conversation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A searcher's wording rarely appears in a chat log: the question is asked once, in passing, and the answer is a fragment. claude-haiku-4-5 now writes a question/summary/resolution/keywords header for each conversation, which lifts MRR 0.79 to 0.81 and recall@1 0.69 to 0.73, with decisions at 0.96 and dates at 0.73. The header is prepended, not substituted as Cerebras does, because their raw transcripts stay searchable in a full-text index and ours have nowhere else to live: a summary drops the catalog numbers and strain names that lexical questions turn on, and those questions held exactly at 0.82 across the change. The cost lands on questions whose answer is two lines long, which now carry a four-line header and score 0.73 to 0.65 — the reranker is the next thing to try against that. The distiller is keyed by model id, so a different model re-distills everything while a new client for the same model does not. --- .env.example | 7 +- pyproject.toml | 1 + slack_index/app.py | 11 ++- slack_index/config.py | 19 +++++- slack_index/context.py | 9 +++ slack_index/distill.py | 89 +++++++++++++++++++++++++ slack_index/threads.py | 6 +- uv.lock | 148 +++++++++++++++++++++++++++++++++++++++++ 8 files changed, 284 insertions(+), 6 deletions(-) create mode 100644 slack_index/distill.py diff --git a/.env.example b/.env.example index 63be9f3..75d97d9 100644 --- a/.env.example +++ b/.env.example @@ -5,8 +5,13 @@ SLACK_BOT_TOKEN=xoxb-replace-me # Comma-separated channel ids, e.g. C0123ABCD,C0456EFGH SLACK_CHANNEL_IDS= +# Distillation. An org-scoped key also needs ANTHROPIC_WORKSPACE_ID. +ANTHROPIC_API_KEY=sk-ant-replace-me +#ANTHROPIC_WORKSPACE_ID= + # Optional -#SLACK_INDEX_EMBED_MODEL=sentence-transformers/all-MiniLM-L6-v2 +#SLACK_INDEX_DISTILL_MODEL=claude-haiku-4-5 +#SLACK_INDEX_EMBED_MODEL=nlpai-lab/KURE-v1 # Unset or 0 indexes the channel from its first message. #SLACK_INDEX_LOOKBACK_DAYS=90 #SLACK_INDEX_POLL_SECONDS=60 diff --git a/pyproject.toml b/pyproject.toml index 9396d83..26aa7fc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,6 +5,7 @@ description = "CocoIndex playground" requires-python = ">=3.14" dependencies = [ "aiohttp>=3.14.3", + "anthropic>=1.8.0", "cocoindex[lancedb,sentence-transformers]>=1.0.24", "numpy>=2.5.3", "pyyaml>=6.0.3", diff --git a/slack_index/app.py b/slack_index/app.py index 4dac94b..eecd7ef 100644 --- a/slack_index/app.py +++ b/slack_index/app.py @@ -9,13 +9,15 @@ from collections.abc import AsyncIterator import cocoindex as coco +from anthropic import AsyncAnthropic from cocoindex.connectors import lancedb from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder from cocoindex.resources.rate_limit import RateLimiter from slack_sdk.web.async_client import AsyncWebClient from slack_index import config -from slack_index.context import EMBEDDER, LANCE_DB, SLACK, SLACK_LIMIT +from slack_index.context import DISTILLER, EMBEDDER, LANCE_DB, SLACK, SLACK_LIMIT +from slack_index.distill import Distiller from slack_index.files import process_file from slack_index.models import SlackChunk from slack_index.source import SlackChannelFiles, SlackChannelThreads @@ -29,6 +31,13 @@ async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None] config.VAR_DIR.mkdir(parents=True, exist_ok=True) builder.settings.db_path = config.DB_PATH builder.provide(SLACK, AsyncWebClient(token=config.bot_token())) + builder.provide( + DISTILLER, + Distiller( + AsyncAnthropic(default_headers=config.anthropic_headers()), + config.DISTILL_MODEL, + ), + ) builder.provide(SLACK_LIMIT, RateLimiter(config.SLACK_REQUESTS_PER_SECOND)) builder.provide(EMBEDDER, SentenceTransformerEmbedder(_settings.embed_model)) builder.provide(LANCE_DB, await lancedb.connect_async(str(config.LANCEDB_URI))) diff --git a/slack_index/config.py b/slack_index/config.py index e6f3c69..366326b 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -25,6 +25,14 @@ CHUNK_SIZE = 1200 CHUNK_OVERLAP = 200 +# Korean-specialised retrieval model; an English-only model scored MRR 0.16 here +# against this one's 0.79. +EMBED_MODEL = "nlpai-lab/KURE-v1" + +# Bulk extraction over short conversations: the cheapest current model is enough. +DISTILL_MODEL = os.environ.get("SLACK_INDEX_DISTILL_MODEL", "claude-haiku-4-5") +DISTILL_MAX_TOKENS = 1024 + # conversations.* and files.* are Slack tier 3 methods: 50+ requests per minute. SLACK_REQUESTS_PER_SECOND = 50 / 60 @@ -49,9 +57,7 @@ def from_env(cls) -> Settings: ) return cls( channel_ids=channel_ids, - embed_model=os.environ.get( - "SLACK_INDEX_EMBED_MODEL", "sentence-transformers/all-MiniLM-L6-v2" - ), + embed_model=os.environ.get("SLACK_INDEX_EMBED_MODEL", EMBED_MODEL), lookback=_lookback_from_env(), poll_interval=datetime.timedelta( seconds=float(os.environ.get("SLACK_INDEX_POLL_SECONDS", "60")) @@ -68,6 +74,13 @@ def _lookback_from_env() -> datetime.timedelta | None: return datetime.timedelta(days=days) if days > 0 else None +def anthropic_headers() -> dict[str, str]: + """An org-scoped API key must name the workspace on every request; a + workspace-scoped key needs nothing.""" + workspace = os.environ.get("ANTHROPIC_WORKSPACE_ID") + return {"anthropic-workspace-id": workspace} if workspace else {} + + def bot_token() -> str: token = os.environ.get("SLACK_BOT_TOKEN") if not token: diff --git a/slack_index/context.py b/slack_index/context.py index 8016159..1aa9336 100644 --- a/slack_index/context.py +++ b/slack_index/context.py @@ -2,13 +2,22 @@ from __future__ import annotations +import typing as _typing + import cocoindex as coco from cocoindex.connectors import lancedb from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder from cocoindex.resources.rate_limit import RateLimiter from slack_sdk.web.async_client import AsyncWebClient +if _typing.TYPE_CHECKING: + from slack_index import distill + SLACK = coco.ContextKey[AsyncWebClient]("slack") +# detect_change: a different distillation model must re-distill everything. +DISTILLER: coco.ContextKey[distill.Distiller] = coco.ContextKey( + "distiller", detect_change=True +) SLACK_LIMIT = coco.ContextKey[RateLimiter]("slack_rate_limit") # detect_change: swapping the embedding model must re-embed everything. EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("embedder", detect_change=True) diff --git a/slack_index/distill.py b/slack_index/distill.py new file mode 100644 index 0000000..99290a1 --- /dev/null +++ b/slack_index/distill.py @@ -0,0 +1,89 @@ +"""LLM distillation: a short header that says what a conversation was about. + +Cerebras' knowledge base replaces the raw transcript with an LLM-normalised +question/summary/resolution and keeps the raw text for full-text search only. We +have no full-text index yet, so the distilled header is prepended to the raw +transcript instead of replacing it: the summary supplies the wording a searcher +would use, the transcript keeps the exact strings (catalog numbers, strain names) +that a summary always drops. +""" + +from __future__ import annotations + +import cocoindex as coco +from anthropic import AsyncAnthropic +from pydantic import BaseModel, Field + +from slack_index.config import DISTILL_MAX_TOKENS +from slack_index.context import DISTILLER + +_SYSTEM = """You summarise chat from a molecular biology lab's Slack channel. +The chat is Korean and informal; the science terms are not. Write in Korean. + +Copy identifiers exactly as they appear — protein and strain names, mutations, +catalog numbers, plasmids, dates, quantities. Never translate or normalise them. +Never add a fact that is not in the transcript.""" + +_INSTRUCTION = """Normalise this conversation for a search index. + +question: one line — the question this conversation answers, or what it records. +summary: one or two sentences on what was discussed. +resolution: what was decided. Leave it empty when nothing was. +keywords: 3-8 proper nouns and identifiers someone would search by. + +Conversation: +""" + + +class Distiller: + """Owns the client and the model id. + + The model id is the memo key: swapping models must re-distill everything, + while a new client object for the same model must not. + """ + + def __init__(self, client: AsyncAnthropic, model: str) -> None: + self._client = client + self._model = model + + def __coco_memo_key__(self) -> object: + return self._model + + async def run(self, transcript: str) -> Distilled: + response = await self._client.messages.parse( + model=self._model, + max_tokens=DISTILL_MAX_TOKENS, + system=_SYSTEM, + messages=[{"role": "user", "content": _INSTRUCTION + transcript}], + output_format=Distilled, + ) + parsed = response.parsed_output + if parsed is None: + raise RuntimeError( + f"distillation returned no structured output: {response.stop_reason}" + ) + return parsed + + +class Distilled(BaseModel): + question: str = Field( + description="One line: the question this conversation answers" + ) + summary: str = Field(description="One or two sentences on what was discussed") + resolution: str = Field(description="What was decided; empty if still open") + keywords: list[str] = Field(description="Identifiers and proper nouns to search by") + + +def render(distilled: Distilled) -> str: + lines = [f"[Q] {distilled.question}", f"[Summary] {distilled.summary}"] + # An open question should not be indexed as if it had an answer. + if distilled.resolution.strip(): + lines.append(f"[Resolution] {distilled.resolution}") + lines.append(f"[Keywords] {', '.join(distilled.keywords)}") + return "\n".join(lines) + + +@coco.fn(memo=True) +async def distill(transcript: str) -> str: + """Memoized on the transcript, so an unchanged conversation is never re-billed.""" + return render(await coco.use_context(DISTILLER).run(transcript)) diff --git a/slack_index/threads.py b/slack_index/threads.py index 3454771..b95b6de 100644 --- a/slack_index/threads.py +++ b/slack_index/threads.py @@ -10,6 +10,7 @@ from slack_index.chunking import ChunkMeta, declare_chunks from slack_index.context import SLACK, SLACK_LIMIT +from slack_index.distill import distill from slack_index.models import ConversationRef, Message, SlackChunk from slack_index.source import next_cursor from slack_index.users import display_name @@ -67,8 +68,11 @@ async def process_thread( speaker = await display_name(message.user) lines.append(f"**{speaker}**: {message.text}") + transcript = "\n\n".join(lines) + # The header goes in front of the transcript, not instead of it: without a + # lexical index, dropping the raw text would drop every exact identifier. await declare_chunks( - "\n\n".join(lines), + f"{await distill(transcript)}\n\n---\n\n{transcript}", ChunkMeta( kind="message", channel=ref.channel, diff --git a/uv.lock b/uv.lock index 86e804f..0fc5952 100644 --- a/uv.lock +++ b/uv.lock @@ -103,6 +103,24 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/99/91/8acff4f5e50511b911bbccb72b8628a49c68ce14148cd9f6431094859a90/annotated_types-0.8.0-py3-none-any.whl", hash = "sha256:f072f4d804ea359e4eaf198b1af7a8b0943881a87f31bb764f8bf219bb9419e0", size = 13427, upload-time = "2026-07-23T20:16:12.938Z" }, ] +[[package]] +name = "anthropic" +version = "1.8.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio" }, + { name = "docstring-parser" }, + { name = "httpx2" }, + { name = "jiter" }, + { name = "pydantic" }, + { name = "sniffio" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/65/b8/f4de0e90bbd641e86a1d6b20e017442033a2c1d2f5381799f82057115bd7/anthropic-1.8.0.tar.gz", hash = "sha256:9c1783ed90f409617749a61c5ab98e20624a572626f2e0a15cea03ed8e1401e5", size = 1221762, upload-time = "2026-09-22T16:25:38.167Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5b/18/5d25a703b66ba34e9277f875f692a3bea3b0ff4d47ee5d188521a01cdd2a/anthropic-1.8.0-py3-none-any.whl", hash = "sha256:79a4516a21e64fd7b15be1a49ebf544bd6376c96a971a77365358823a2717bc6", size = 1348276, upload-time = "2026-09-22T16:25:36.515Z" }, +] + [[package]] name = "anyio" version = "4.15.1" @@ -335,6 +353,7 @@ version = "0.1.0" source = { virtual = "." } dependencies = [ { name = "aiohttp" }, + { name = "anthropic" }, { name = "cocoindex", extra = ["lancedb", "sentence-transformers"] }, { name = "numpy" }, { name = "pyyaml" }, @@ -352,6 +371,7 @@ dev = [ [package.metadata] requires-dist = [ { name = "aiohttp", specifier = ">=3.14.3" }, + { name = "anthropic", specifier = ">=1.8.0" }, { name = "cocoindex", extras = ["lancedb", "sentence-transformers"], specifier = ">=1.0.24" }, { name = "numpy", specifier = ">=2.5.3" }, { name = "pyyaml", specifier = ">=6.0.3" }, @@ -484,6 +504,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/02/c3/253a89ee03fc9b9682f1541728eb66db7db22148cd94f89ab22528cd1e1b/deprecation-2.1.0-py2.py3-none-any.whl", hash = "sha256:a10811591210e1fb0e768a8c25517cabeabcba6f0bf96564f8ff45189f90b14a", size = 11178, upload-time = "2020-04-20T14:23:36.581Z" }, ] +[[package]] +name = "docstring-parser" +version = "0.18.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e0/4d/f332313098c1de1b2d2ff91cf2674415cc7cddab2ca1b01ae29774bd5fdf/docstring_parser-0.18.0.tar.gz", hash = "sha256:292510982205c12b1248696f44959db3cdd1740237a968ea1e2e7a900eeb2015", size = 29341, upload-time = "2026-04-14T04:09:19.867Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a7/5f/ed01f9a3cdffbd5a008556fc7b2a08ddb1cc6ace7effa7340604b1d16699/docstring_parser-0.18.0-py3-none-any.whl", hash = "sha256:b3fcbed555c47d8479be0796ef7e19c2670d428d72e96da63f3a40122860374b", size = 22484, upload-time = "2026-04-14T04:09:18.638Z" }, +] + [[package]] name = "executing" version = "2.2.1" @@ -598,6 +627,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7e/f5/f66802a942d491edb555dd61e3a9961140fd64c90bce1eafd741609d334d/httpcore-1.0.9-py3-none-any.whl", hash = "sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55", size = 78784, upload-time = "2025-04-24T22:06:20.566Z" }, ] +[[package]] +name = "httpcore2" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "h11" }, + { name = "truststore" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/15/8c/e925b1c92018abb3a1863ce1549d76d2381e334d21d65d4ac8f65dabd78a/httpcore2-2.13.0.tar.gz", hash = "sha256:2adc8be4fb285fbcd6d894298db3b52c177e74b6674eda3a76bd36be3292a3db", size = 67740, upload-time = "2026-09-14T14:18:04.717Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7e/0d/117a771a2bb91df334b66bf4da14cd02f21aefbcfe53180f336ce55e8f90/httpcore2-2.13.0-py3-none-any.whl", hash = "sha256:35ae5be347aa40467b4a5dc032ac67ebb6d27189fc97e8cebcf99616f6a1bb9e", size = 83162, upload-time = "2026-09-14T14:18:02.529Z" }, +] + [[package]] name = "httpx" version = "0.28.1" @@ -613,6 +655,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/2a/39/e50c7c3a983047577ee07d2a9e53faf5a69493943ec3f6a384bdc792deb2/httpx-0.28.1-py3-none-any.whl", hash = "sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad", size = 73517, upload-time = "2024-12-06T15:37:21.509Z" }, ] +[[package]] +name = "httpx2" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio", marker = "sys_platform != 'emscripten'" }, + { name = "httpcore2", marker = "sys_platform != 'emscripten'" }, + { name = "httpx2-jsfetch", marker = "sys_platform == 'emscripten'" }, + { name = "idna" }, + { name = "truststore", marker = "sys_platform != 'emscripten'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b9/a0/e9deef4654132857b5a5dbe4eddd0ac59c2814500e11f2f5044cd81103ee/httpx2-2.13.0.tar.gz", hash = "sha256:81bd07dc67a3701729ef1f777a3c00c915d4539604fdb5afd327f8682f6b7b44", size = 100290, upload-time = "2026-09-14T14:18:05.486Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fe/d1/a0c72b0e006df654709fbc366cc5bcb53e5aee13e1e3395152c6dd293376/httpx2-2.13.0-py3-none-any.whl", hash = "sha256:fc12720cedf72faa26cca6b4ca394e05c894e7d7933fc45cafe767960804e49a", size = 95565, upload-time = "2026-09-14T14:18:03.553Z" }, +] + +[[package]] +name = "httpx2-jsfetch" +version = "1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/cd/c4/0e5636363151a2a1795e0a77617168b9ca438e1748ec05fc9b5687f93d64/httpx2_jsfetch-1.0.tar.gz", hash = "sha256:70a0e3eabfef7cce5ad9c629f7d01ca05e418f586646f4ddf14782e4c1454c60", size = 6872, upload-time = "2026-08-07T00:13:07.492Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9b/43/832f631d32e4f1211caa2ba368317739fe71f0b8530e4c9d15dc454bac2a/httpx2_jsfetch-1.0-py3-none-any.whl", hash = "sha256:cb916b707601e69a07721aabc8f3f6659be3a6893bc1ff5c6f9e02241df2da32", size = 6382, upload-time = "2026-08-07T00:13:06.567Z" }, +] + [[package]] name = "huggingface-hub" version = "1.32.0" @@ -732,6 +799,69 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" }, ] +[[package]] +name = "jiter" +version = "0.17.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9c/1f/8176d92e001f86505424b41664032ae26a882bc9ca41a32c803f373f9195/jiter-0.17.0.tar.gz", hash = "sha256:03e432f226a453851079fb84cd17c6da9991eab723e28d716f14ae3d906e0c12", size = 229037, upload-time = "2026-09-12T15:14:14.253Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/01/9e/23065f8e2c7a4c372c1b6f6622e4cfab4dc786cb5150052b1527e6a6a840/jiter-0.17.0-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:00d783a779c5664e16dbad5e3a3c3a75e128b07dd5f4765159658d9210a50ca5", size = 292210, upload-time = "2026-09-12T15:12:35.613Z" }, + { url = "https://files.pythonhosted.org/packages/ea/81/67b58647560bc82a4490d722caa8561d7a86a9f45d4fa620b7e5fe282c7a/jiter-0.17.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:0619d806e260ecf0c2a64521942c94af5d547c9ec99b55ae4f51b538b5576a76", size = 321512, upload-time = "2026-09-12T15:12:36.907Z" }, + { url = "https://files.pythonhosted.org/packages/c7/07/6658359a25f55927f7f8bf0e16465dee2ccd0b2a1a5208acc0df8972e074/jiter-0.17.0-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:dc0288ce39190ee33fe6e4ec73161eed34e7e2da509b525546ca061778d62b64", size = 343897, upload-time = "2026-09-12T15:12:38.189Z" }, + { url = "https://files.pythonhosted.org/packages/46/04/5d50a9f0319cbdc37fd53c27f8c313d46afc34f1b048219ae6d8ea068da4/jiter-0.17.0-cp314-cp314-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:5a52a430d04225ffde633e6840bf2381d34c019ff98526b5929755b9052fb199", size = 326519, upload-time = "2026-09-12T15:12:39.532Z" }, + { url = "https://files.pythonhosted.org/packages/bb/c7/d02517832b29eb8275fdd0f4ce0f17b80f58cc4c3ebecd4d9ace990d633d/jiter-0.17.0-cp314-cp314-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:37f33d327900bf2879613b3363fd48df97b4232d0c41f54bcf2e790c2fc40a71", size = 341369, upload-time = "2026-09-12T15:12:41.486Z" }, + { url = "https://files.pythonhosted.org/packages/3b/07/499b5f5603501cdd93a73a6a176dfad9c96555a3ae58ca9f8e3acba63dc9/jiter-0.17.0-cp314-cp314-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:6cf564d43c4388149ca58ee571d0f5ccf875e20d1fd4662fd94cc0d1ea3b10ef", size = 352160, upload-time = "2026-09-12T15:12:42.721Z" }, + { url = "https://files.pythonhosted.org/packages/f5/75/b04013c7743269d4533ef4e746fc0ed678a143968dd7448658e3f51daad2/jiter-0.17.0-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:523c499235fb65add25d4bb01b1c4709ce695efdc7deb6c0a7bc515b5c44e0fb", size = 345018, upload-time = "2026-09-12T15:12:44.192Z" }, + { url = "https://files.pythonhosted.org/packages/1d/96/cbb6fd1e42a77c8412ec4643db95059b30cdfc635e387cc9193e098ce268/jiter-0.17.0-cp314-cp314-manylinux_2_31_riscv64.whl", hash = "sha256:455e4ab35cb2a4a91a8404e08fd3c621bae433922e59bf1c494fe20a426b013b", size = 329244, upload-time = "2026-09-12T15:12:45.491Z" }, + { url = "https://files.pythonhosted.org/packages/15/67/d3be402f398566a379bf40ae65be5c3505b14d9e95e0802a597ddde7ddee/jiter-0.17.0-cp314-cp314-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6871973bfbd4408f7f1c632b30bbb5bbd9671c1bc8650af6823e24b7be13709b", size = 335693, upload-time = "2026-09-12T15:12:46.935Z" }, + { url = "https://files.pythonhosted.org/packages/7f/8d/98e2c4130b93d64f1d67c89060b928d04102549bf05e64451c9e6024f9ca/jiter-0.17.0-cp314-cp314-musllinux_1_1_aarch64.whl", hash = "sha256:77f6aac0137309b31448c1bdcda4c6c77077664a6d018ece8d94019c68a5a5b9", size = 484329, upload-time = "2026-09-12T15:12:48.361Z" }, + { url = "https://files.pythonhosted.org/packages/78/5e/8da91e49f0fbca37c3489fb4cf3ad6676d4965f00ae5468bca3a2513737a/jiter-0.17.0-cp314-cp314-musllinux_1_1_x86_64.whl", hash = "sha256:93946d89fa04d5ba64dd323a8dd8d901676cb8a3c81d99ae4f6c051a9b4c3f2f", size = 521358, upload-time = "2026-09-12T15:12:49.856Z" }, + { url = "https://files.pythonhosted.org/packages/be/21/5388684a5a38af3557cd9c2424b9827c71809cff24373c75ef9d0d3dfba9/jiter-0.17.0-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:70f19a2ca8429f91e82eeffb2f51cb87bc2d6e953b009b91a92d29c3a16ccb03", size = 110459, upload-time = "2026-09-12T15:12:51.747Z" }, + { url = "https://files.pythonhosted.org/packages/b1/ad/58b3a93525d2ffca7f54d9dee441381990082bd1172fbeb8d6a3f72a4dc3/jiter-0.17.0-cp314-cp314-win32.whl", hash = "sha256:71dbd74314c5df52a1bccf7b8bca46d14e943af7a2012e73b23f49977ef194c8", size = 185043, upload-time = "2026-09-12T15:12:54.477Z" }, + { url = "https://files.pythonhosted.org/packages/7a/4a/1aa520eb6c359b262c14ff995ca7283837208ddfb1202082ce9d73cf214d/jiter-0.17.0-cp314-cp314-win_amd64.whl", hash = "sha256:ac3c6ee3264d6f5c44c617f90bc7e8b9e1587e7d6708c9d8f811cb65582ee312", size = 227163, upload-time = "2026-09-12T15:12:55.931Z" }, + { url = "https://files.pythonhosted.org/packages/cf/e4/5997f648794bd9b499491d0ff480b096cc9a9c65bdba29f57568e6aa1705/jiter-0.17.0-cp314-cp314-win_arm64.whl", hash = "sha256:6219adaf59711ba7063a52496e8ec6d3fa3e209d7827d83eee3b2abc780a1744", size = 183505, upload-time = "2026-09-12T15:12:58.196Z" }, + { url = "https://files.pythonhosted.org/packages/ac/4a/84a5ec271d09f7590b6073af5ee4abb44eab4ccace453b7e2c5ce45234ca/jiter-0.17.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:59bddbe6f9ffecc68d641e1e2d619ce64cf8a9e9eeb74e5c518f74fc87abf1b0", size = 321527, upload-time = "2026-09-12T15:12:59.394Z" }, + { url = "https://files.pythonhosted.org/packages/39/71/9e1fd0045f5920b4c36be35c3f0f0dfd123668684f8ad352619d7aa44183/jiter-0.17.0-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6cb41cd1432f1dc19a231cf70b54d42b2c9f05085155859263fce06fa4d41388", size = 340865, upload-time = "2026-09-12T15:13:00.756Z" }, + { url = "https://files.pythonhosted.org/packages/b7/2b/14627fd2bc377f3dd09491bcace6b90e34b4d7fea2f1f3295031ff91f528/jiter-0.17.0-cp314-cp314t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:fd7790aa79c8b518e512ebcdfce9f11d8ef5f30efd43720c8a19a548b39fa489", size = 325412, upload-time = "2026-09-12T15:13:02.152Z" }, + { url = "https://files.pythonhosted.org/packages/4c/f3/8d5808f7bf0f456bde79e6393587183a0cee5f83d179fe1f7f1eff2ba067/jiter-0.17.0-cp314-cp314t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:dbbfe4e3c21c8166980cddc5bee1a315df082454f007947dfb6fb73800768165", size = 340473, upload-time = "2026-09-12T15:13:03.485Z" }, + { url = "https://files.pythonhosted.org/packages/4f/da/1d8c7c6c4ae6b2423b94a81b6b907d37b28f87664e077427b531bf1b5313/jiter-0.17.0-cp314-cp314t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8c286860abfe8b100cac1c02e225e5776eb9216edd71ba17cdb237da4af32bc9", size = 350757, upload-time = "2026-09-12T15:13:04.828Z" }, + { url = "https://files.pythonhosted.org/packages/eb/96/c1813dcca15c5a370145a448aaea7d1f83f6f0228a5f1130e79340ee385f/jiter-0.17.0-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f753eb70b1474a29e635e7542ff7312e6d6b951e0b25e8a2e8c34eeb1ddcd478", size = 345203, upload-time = "2026-09-12T15:13:06.131Z" }, + { url = "https://files.pythonhosted.org/packages/d7/f7/fc61cbcf2992d169ede13648fc3fd8e2d3171a3669dde43cd4db556549ac/jiter-0.17.0-cp314-cp314t-manylinux_2_31_riscv64.whl", hash = "sha256:eae86b1f027031e39db2e0e9c4842221edb7b8cd474d23f87a79b3bd4b651768", size = 328322, upload-time = "2026-09-12T15:13:07.392Z" }, + { url = "https://files.pythonhosted.org/packages/8f/88/46418a3abbdffb7dc41b314200360f24f75faaeb35573e81c92de322cce9/jiter-0.17.0-cp314-cp314t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:5bf350452a43173e69e1fc74847c57a60e3d7515807287f29849baa2a85d8718", size = 336570, upload-time = "2026-09-12T15:13:08.666Z" }, + { url = "https://files.pythonhosted.org/packages/f0/28/b8a55b949be6306df8888e365a8df05441de8a7b11289f6957004302e41e/jiter-0.17.0-cp314-cp314t-musllinux_1_1_aarch64.whl", hash = "sha256:da139721f4b7cafdbff580a4f511ea24cb91f4909330c6b926a1ca53836c0a59", size = 482879, upload-time = "2026-09-12T15:13:10.037Z" }, + { url = "https://files.pythonhosted.org/packages/75/3b/21d0afa53ba0680962c39f3eb95ed2946f8793369ed44b0c82b490723081/jiter-0.17.0-cp314-cp314t-musllinux_1_1_x86_64.whl", hash = "sha256:8079849db9a1371bfd90bad088458a8fb836261879df2233cc9632464ecf64e1", size = 520406, upload-time = "2026-09-12T15:13:11.456Z" }, + { url = "https://files.pythonhosted.org/packages/ef/03/bcbaf8b6b9ea23c2c074411f8ecfbb02d820abac5d0cb8f4e280209174a2/jiter-0.17.0-cp314-cp314t-win32.whl", hash = "sha256:8f770b0c77e5fac482e1ba03ca1a7e18286bfb213d749932a00a7e4cd5de5e06", size = 184434, upload-time = "2026-09-12T15:13:13.037Z" }, + { url = "https://files.pythonhosted.org/packages/7a/b5/5d6ce2c93ef6fe1241b37a9005547f9b6d58db1f07f39fe95807d4b98f51/jiter-0.17.0-cp314-cp314t-win_amd64.whl", hash = "sha256:c4289293e5278d9314b00f15c37f2120fa51d3d68565292e715524c750e775a9", size = 227392, upload-time = "2026-09-12T15:13:14.933Z" }, + { url = "https://files.pythonhosted.org/packages/f5/4b/1e52baf90187606e33a7b8cfa8f96f5829acd7f01870077eb01059ab76d0/jiter-0.17.0-cp314-cp314t-win_arm64.whl", hash = "sha256:4dfbfe5a6e1e80a7082af559f66386405025ec278833e0c649f69cbc6e1004cc", size = 182776, upload-time = "2026-09-12T15:13:16.239Z" }, + { url = "https://files.pythonhosted.org/packages/05/fc/efe3ac75564ab10f53517958f5ccdc231fc7334af66c76776cb554a88967/jiter-0.17.0-cp315-cp315-macosx_10_12_x86_64.whl", hash = "sha256:84963d3f395ef5e9a32ce47155e08a7962fa292c159a10cb98b931cef1416925", size = 292143, upload-time = "2026-09-12T15:13:17.502Z" }, + { url = "https://files.pythonhosted.org/packages/d1/4c/46982118d91f9ffe9714319d21ec4f98d9b7e0cfd9062826c524a54de24e/jiter-0.17.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:ffa0380ad091de7d3fc33e17a97ff479851ee18a0a2a3ee56ff3215cdc886656", size = 321341, upload-time = "2026-09-12T15:13:19.133Z" }, + { url = "https://files.pythonhosted.org/packages/e7/12/9b1ac6ecc6307049913db54839ddba1c11c1ef72c5a8bbb5514bc3b50d1b/jiter-0.17.0-cp315-cp315-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:755079792868ce5d4938e83b91a0939b34fb858a1ca65a104f2d771bea57faa1", size = 344383, upload-time = "2026-09-12T15:13:20.508Z" }, + { url = "https://files.pythonhosted.org/packages/a9/b6/527cc72af836d824e9d4d666e64f0a1ca7eafd662a8da9657b78592172ba/jiter-0.17.0-cp315-cp315-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:3bf4dc2b84a464117fb097d15a25c58d100d2692888e3b0d92df5b48ed16b7c0", size = 326841, upload-time = "2026-09-12T15:13:21.83Z" }, + { url = "https://files.pythonhosted.org/packages/d1/41/567f98617e88005b249503b933803f633ec6ba2d427cf4cc35e5c832125c/jiter-0.17.0-cp315-cp315-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:02a360707033d8cef53f7f3480817a1489177a259ec6ec01e98c37e0b922ddca", size = 341354, upload-time = "2026-09-12T15:13:23.323Z" }, + { url = "https://files.pythonhosted.org/packages/40/da/b29cda895b785f7d426e224638a885b6145a08ce853b381f34afe3e88c5d/jiter-0.17.0-cp315-cp315-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:300ce01ab0215e3dea4d00090143c909aedc65c0f809b3c07983e1d038f291b9", size = 351985, upload-time = "2026-09-12T15:13:26.526Z" }, + { url = "https://files.pythonhosted.org/packages/f7/5c/8a73829e7389e72ea298a450f2b3cb58e71a3e464b45f6d8753740f1c4f5/jiter-0.17.0-cp315-cp315-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:746243a080b4ca790b8499af3d7cf9825d5f5987933950cd818e767ee353d826", size = 346052, upload-time = "2026-09-12T15:13:27.887Z" }, + { url = "https://files.pythonhosted.org/packages/1d/2f/98d6001026932c095ba440925570123043bed29f5ff56158dfe729a9e81b/jiter-0.17.0-cp315-cp315-manylinux_2_31_riscv64.whl", hash = "sha256:b550585523339b71cb852b811aae49d08d7601ad8ffe9f5dc1562f4c3d22fd87", size = 329159, upload-time = "2026-09-12T15:13:31.569Z" }, + { url = "https://files.pythonhosted.org/packages/94/2e/708dc1d2678f092c31c12754e860cd8353e6a85ecbdb1010157edca0da9e/jiter-0.17.0-cp315-cp315-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:0239520085cac678e77a606fd7e3f1c60c371d719790c5e3807388d3da4354c2", size = 336001, upload-time = "2026-09-12T15:13:32.846Z" }, + { url = "https://files.pythonhosted.org/packages/f4/f0/75a5ae38862f4eaf0fe2f8a9fbf6484c4890df04c06dcdffc45e36bca61a/jiter-0.17.0-cp315-cp315-musllinux_1_1_aarch64.whl", hash = "sha256:eb2295da7c3769f6719b227a237aa6a5cfa6550e478bc838001b592c57e16575", size = 484281, upload-time = "2026-09-12T15:13:35.333Z" }, + { url = "https://files.pythonhosted.org/packages/a0/32/6636fae811c27c7f93e1b11fb5800de6a5c9e4269a27cf718e0b31218ad1/jiter-0.17.0-cp315-cp315-musllinux_1_1_x86_64.whl", hash = "sha256:e088612ff90ebc9247e1a43074b72835804261c47e6a6c01cb3ddcb55360d688", size = 521300, upload-time = "2026-09-12T15:13:37.101Z" }, + { url = "https://files.pythonhosted.org/packages/61/aa/12df7e0b0b1a2602e3d5a5a7104d7d9700f254b400f134a9b50955c4d231/jiter-0.17.0-cp315-cp315-win32.whl", hash = "sha256:0b52d52035b3907c5b1f6277857b29c1cbfc965e24e0f27330dbed83edb591ec", size = 185138, upload-time = "2026-09-12T15:13:38.901Z" }, + { url = "https://files.pythonhosted.org/packages/ba/ec/3dd2e495032cddde05723c1f4c743b67a23e55d2af244692a7f58f0cdae3/jiter-0.17.0-cp315-cp315-win_amd64.whl", hash = "sha256:10f5558eed511b830488003449d942bd75829ad6257dc58cb9a03e596a7777b1", size = 226950, upload-time = "2026-09-12T15:13:40.17Z" }, + { url = "https://files.pythonhosted.org/packages/c3/c7/ef85704e0a57e9cadb2babc05f6d7c5df4a1c75da1a6ee31e1986b0099a5/jiter-0.17.0-cp315-cp315-win_arm64.whl", hash = "sha256:fa13acf1046f95df808c64b1310705e143fab87aee73ae00cc42d640867fd2c1", size = 183618, upload-time = "2026-09-12T15:13:41.432Z" }, + { url = "https://files.pythonhosted.org/packages/0e/9a/a4b348349de68762b58d6713973d363ad80a1c741d0bf8def7975f0ecb26/jiter-0.17.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:af2f7501580f274b63c4b2283bc425f5df7edf06ae5b171e5f87d912ff359a20", size = 321155, upload-time = "2026-09-12T15:13:42.716Z" }, + { url = "https://files.pythonhosted.org/packages/c1/70/aebd6d0b5f0677de3a3d0bdc4a05fac949b97c4ede454c8809f180ac7b17/jiter-0.17.0-cp315-cp315t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:10c5349312e5cb02b7a21e123a57665afa895953f05bf252a9dd4c13a572b7ab", size = 340985, upload-time = "2026-09-12T15:13:44.115Z" }, + { url = "https://files.pythonhosted.org/packages/a7/82/4c3b49796b5eb62f3f5046f957683f4ba0135fe1a60957c11180512460df/jiter-0.17.0-cp315-cp315t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:86f3f9343a288eb85a81ef20a752b2f84564296636db54a9fff0b5c8deaf1df2", size = 325670, upload-time = "2026-09-12T15:13:45.901Z" }, + { url = "https://files.pythonhosted.org/packages/bc/43/f6341ecb4872202a4ef150486fcee0e1ace4aa3da39b71b82061452cdd3a/jiter-0.17.0-cp315-cp315t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:4607ec7d93355fbc25b8dc5189153cf21d66063b9f9cd04dd2774e6e783f9b6a", size = 340339, upload-time = "2026-09-12T15:13:47.442Z" }, + { url = "https://files.pythonhosted.org/packages/f9/c4/bc2c86e08fa065e03cb2fbc53b367c3640a7d257ef9d877b29118ea636b7/jiter-0.17.0-cp315-cp315t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:10cd64a5720ad7f809ac5466ff1705813f1b6b510f195a73acafba0ac0e1f675", size = 350705, upload-time = "2026-09-12T15:13:48.848Z" }, + { url = "https://files.pythonhosted.org/packages/9d/67/91f12aa111cca6e3a197c3e36bf60a034bf9f122f6d41112a639e44217d8/jiter-0.17.0-cp315-cp315t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:efe9f61bb30174d2f5c8396445c360c96c44e78164d0815dfe627ccf57849574", size = 345011, upload-time = "2026-09-12T15:13:50.215Z" }, + { url = "https://files.pythonhosted.org/packages/f5/cb/9f5556e8f6ec89755fb5a709d8eb8270c9a324e31079eda0dfbeca451b6e/jiter-0.17.0-cp315-cp315t-manylinux_2_31_riscv64.whl", hash = "sha256:370d8fe5bf201dc6925e8a84c81ac7291f74d9fd1778234fc79d517064a5c76b", size = 328268, upload-time = "2026-09-12T15:13:51.809Z" }, + { url = "https://files.pythonhosted.org/packages/22/98/153f20680fb75781a490fb849940e2b00f95035c7aa054df592f36ed33fc/jiter-0.17.0-cp315-cp315t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6b303d88e6a0bda789ec4b7801c7bad68e27230ba1fe4baffc756d1fbd32dc9d", size = 337024, upload-time = "2026-09-12T15:13:53.095Z" }, + { url = "https://files.pythonhosted.org/packages/af/59/b16c9be3a5035df4466cc72e888188c027562de90a723d290ab6814cb9d4/jiter-0.17.0-cp315-cp315t-musllinux_1_1_aarch64.whl", hash = "sha256:30793a24a31e968969757c9e08d830cbb15a2cd3c4959b4498b38f4b1c2258eb", size = 482766, upload-time = "2026-09-12T15:13:55.713Z" }, + { url = "https://files.pythonhosted.org/packages/d0/55/667dea313094024bef082175d6bfe8976f90d1c00c926af9df1d8e0eab48/jiter-0.17.0-cp315-cp315t-musllinux_1_1_x86_64.whl", hash = "sha256:686c93d86f2b426c803024b805bd161a6cd10e9627c23e901640eab646c0ad8a", size = 520367, upload-time = "2026-09-12T15:13:57.674Z" }, + { url = "https://files.pythonhosted.org/packages/21/e3/4b1a43501fb9ed17b01d137e380cb0e8fdcb39a254ce31aa2ab95bc861ac/jiter-0.17.0-cp315-cp315t-win32.whl", hash = "sha256:86d703d9faa1ffc8ae4e9de0fa007712ed2171b5c0d93811a8e2e105ac729b0d", size = 184603, upload-time = "2026-09-12T15:13:59.27Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f2/b8ee0372b6ebdf1bde5cc44495d5291d17f961065f5b48f8616cc67cac2e/jiter-0.17.0-cp315-cp315t-win_amd64.whl", hash = "sha256:42b0260445251b1bc520a63baa94a32d88e0f931fba234f1764db7feb7c72174", size = 227936, upload-time = "2026-09-12T15:14:00.472Z" }, + { url = "https://files.pythonhosted.org/packages/a4/b4/923a1215daba959aed8355973315cb3f81f53e0d01c5b211870a27b41f45/jiter-0.17.0-cp315-cp315t-win_arm64.whl", hash = "sha256:d47687806f9c54c84ea38733507081337922beca90ce819c7d852dd485bc0f23", size = 182977, upload-time = "2026-09-12T15:14:01.799Z" }, +] + [[package]] name = "joblib" version = "1.6.0" @@ -2009,6 +2139,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ab/fc/67352b742fc6fa520a550581b0284f5757e4e31a32327c2a00276747fd30/slack_sdk-3.44.1-py2.py3-none-any.whl", hash = "sha256:d6f20a0fbe3fecf9cac955c99d686301b48a645b7045472c3a0cdd186d7c42b2", size = 319865, upload-time = "2026-09-03T14:21:12.405Z" }, ] +[[package]] +name = "sniffio" +version = "1.3.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a2/87/a6771e1546d97e7e041b6ae58d80074f81b7d5121207425c964ddf5cfdbd/sniffio-1.3.1.tar.gz", hash = "sha256:f4324edc670a0f49750a81b895f35c3adb843cca46f0530f79fc1babb23789dc", size = 20372, upload-time = "2024-02-25T23:20:04.057Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" }, +] + [[package]] name = "stack-data" version = "0.6.3" @@ -2171,6 +2310,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fe/d1/aa8a3e935c37efee7945984fdb64d7e0851bf6d920afd97b2d21f9d23360/triton-3.8.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:74217bb56ed8692759227758e4c4b3bd2d608a209c1a7a081bf361fb4c2c1bf9", size = 248077577, upload-time = "2026-08-28T15:56:24.94Z" }, ] +[[package]] +name = "truststore" +version = "0.10.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/53/a3/1585216310e344e8102c22482f6060c7a6ea0322b63e026372e6dcefcfd6/truststore-0.10.4.tar.gz", hash = "sha256:9d91bd436463ad5e4ee4aba766628dd6cd7010cf3e2461756b3303710eebc301", size = 26169, upload-time = "2025-08-12T18:49:02.73Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/19/97/56608b2249fe206a67cd573bc93cd9896e1efb9e98bce9c163bcdc704b88/truststore-0.10.4-py3-none-any.whl", hash = "sha256:adaeaecf1cbb5f4de3b1959b42d41f6fab57b2b1666adb59e89cb0b53361d981", size = 18660, upload-time = "2025-08-12T18:49:01.46Z" }, +] + [[package]] name = "typer" version = "0.27.2" From 236a22cc5f3e575afa301c33b3a1066d5026d7f1 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:03:29 +0900 Subject: [PATCH 07/12] slack: make distillation reproducible The prompt and the output schema decide what every summary says, but they are module constants that the logic fingerprint cannot see, so editing a prompt would have left the channel indexed by summaries the old prompt wrote. Declaring them as deps fixes that: a one-space edit to the system prompt now changes the fingerprint and re-distills. Sampling was also unpinned, so an unchanged conversation came back with a differently worded summary on every re-run; temperature 0 rides along in the request body because parse() takes no sampling arguments. Re-distilling the same conversations with only the sampling changed moves overall MRR 0.81 to 0.80 and the split slice 0.65 to 0.75, which puts a number on how much of a per-tag reading is noise: the distillation gain claimed in the previous commit is inside that band, while recall@10 rising to 0.99 is not. --- slack_index/config.py | 1 + slack_index/distill.py | 22 ++++++++++++++++++++-- 2 files changed, 21 insertions(+), 2 deletions(-) diff --git a/slack_index/config.py b/slack_index/config.py index 366326b..b98a174 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -32,6 +32,7 @@ # Bulk extraction over short conversations: the cheapest current model is enough. DISTILL_MODEL = os.environ.get("SLACK_INDEX_DISTILL_MODEL", "claude-haiku-4-5") DISTILL_MAX_TOKENS = 1024 +DISTILL_TEMPERATURE = 0.0 # conversations.* and files.* are Slack tier 3 methods: 50+ requests per minute. SLACK_REQUESTS_PER_SECOND = 50 / 60 diff --git a/slack_index/distill.py b/slack_index/distill.py index 99290a1..e5df83b 100644 --- a/slack_index/distill.py +++ b/slack_index/distill.py @@ -10,11 +10,13 @@ from __future__ import annotations +import json + import cocoindex as coco from anthropic import AsyncAnthropic from pydantic import BaseModel, Field -from slack_index.config import DISTILL_MAX_TOKENS +from slack_index.config import DISTILL_MAX_TOKENS, DISTILL_TEMPERATURE from slack_index.context import DISTILLER _SYSTEM = """You summarise chat from a molecular biology lab's Slack channel. @@ -56,6 +58,9 @@ async def run(self, transcript: str) -> Distilled: system=_SYSTEM, messages=[{"role": "user", "content": _INSTRUCTION + transcript}], output_format=Distilled, + # `parse()` takes no sampling arguments; temperature rides along in the + # body so a re-distill does not reword an unchanged conversation. + extra_body={"temperature": DISTILL_TEMPERATURE}, ) parsed = response.parsed_output if parsed is None: @@ -83,7 +88,20 @@ def render(distilled: Distilled) -> str: return "\n".join(lines) -@coco.fn(memo=True) +# The prompt and the output schema shape every distillation but live outside the +# function body, where cocoindex's logic fingerprint cannot see them. Declaring them +# as deps is what makes editing a prompt re-distill the channel instead of silently +# serving summaries written by the old one. +_PROMPT_DEPS = ( + _SYSTEM, + _INSTRUCTION, + DISTILL_MAX_TOKENS, + DISTILL_TEMPERATURE, + json.dumps(Distilled.model_json_schema(), sort_keys=True), +) + + +@coco.fn(memo=True, deps=_PROMPT_DEPS) async def distill(transcript: str) -> str: """Memoized on the transcript, so an unchanged conversation is never re-billed.""" return render(await coco.use_context(DISTILLER).run(transcript)) From 088c7ec3791293bd396eb47467bc0e0a4eefeefe Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:17:47 +0900 Subject: [PATCH 08/12] slack: rerank the shortlist with a cross-encoder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The retriever compares a query and a document through one vector each, which is why the answer kept landing in the pool but not at the top: at this point every remaining failure was an ordering failure. A cross-encoder reads the pair together and rescores the 20 sources the retriever shortlists, taking MRR 0.80 to 0.88, recall@1 0.70 to 0.79, and leaving no question whose answer is missing from the results. Dates gain the most — several meetings look alike to an embedding and not to a model reading the question beside them. It also gives the index a way to say nothing fits. Cosine put the five questions the channel cannot answer inside the range of the real ones; the cross-encoder scores them at most 0.002 against a median of 0.690, so a threshold finally means something. Reranking is query-time, so the eval can run with and without it against one index, and the search picks the GPU when there is one: 2.2s per query on mps against 5s on the CPU. --- evals/questions.example.yaml | 30 ++++++++++++++++ pyproject.toml | 8 ++--- slack_index/config.py | 6 ++++ slack_index/evals.py | 12 +++++-- slack_index/rerank.py | 66 ++++++++++++++++++++++++++++++++++++ slack_index/search.py | 29 ++++++++++++---- 6 files changed, 137 insertions(+), 14 deletions(-) create mode 100644 evals/questions.example.yaml create mode 100644 slack_index/rerank.py diff --git a/evals/questions.example.yaml b/evals/questions.example.yaml new file mode 100644 index 0000000..c3740d3 --- /dev/null +++ b/evals/questions.example.yaml @@ -0,0 +1,30 @@ +# Question set for `python -m slack_index.evals`. Copy to questions.yaml and fill +# it with questions whose answers you already know; that file stays untracked +# because its labels quote channel content. +# +# - question: what someone would actually type, NOT the wording of the message +# expected: [source_id, ...] # thread_ts or file id; any hit in top-k counts +# tags: [category, ...] # scored per tag, so weak spots stay visible +# +# Paraphrase away from the message text: a question copied from the message lets +# lexical retrieval win for free and stops measuring anything. +# +# Categories in use: +# concept — about content, worded differently from the message +# lexical — hinges on an exact identifier (part number, hostname, error string) +# decision — the answer is an agreement or conclusion, often a short ack +# temporal — scheduling, or what happened when +# file — the answer lives in a shared file, not a message +# people — who said or owns something + +- question: "did we settle on bumping that library" + expected: ["1700000000.000100"] + tags: [decision] + +- question: "what was the env var the deploy script reads" + expected: ["1700000100.000200"] + tags: [lexical] + +- question: "the design doc shared last week" + expected: ["F00EXAMPLE1", "1700000200.000300"] + tags: [file] diff --git a/pyproject.toml b/pyproject.toml index 26aa7fc..27a9d46 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -14,10 +14,10 @@ dependencies = [ [dependency-groups] dev = [ - "mypy>=1.14", - "ipykernel>=7.3.0", - "pytest>=9.1.1", - "types-pyyaml>=6.0.12.20260906", + "mypy>=1.14", + "ipykernel>=7.3.0", + "pytest>=9.1.1", + "types-pyyaml>=6.0.12.20260906", ] [tool.uv] diff --git a/slack_index/config.py b/slack_index/config.py index b98a174..d9f1040 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -29,6 +29,12 @@ # against this one's 0.79. EMBED_MODEL = "nlpai-lab/KURE-v1" +# Cross-encoder over the shortlist. Korean-capable, same family as the embedder. +RERANK_MODEL = "BAAI/bge-reranker-v2-m3" +# How many distinct sources the retriever hands the reranker. Recall above this is +# unreachable, precision below it is the reranker's to fix. +RERANK_CANDIDATES = 20 + # Bulk extraction over short conversations: the cheapest current model is enough. DISTILL_MODEL = os.environ.get("SLACK_INDEX_DISTILL_MODEL", "claude-haiku-4-5") DISTILL_MAX_TOKENS = 1024 diff --git a/slack_index/evals.py b/slack_index/evals.py index dc4338c..4157362 100644 --- a/slack_index/evals.py +++ b/slack_index/evals.py @@ -128,11 +128,12 @@ def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: return f"{name:<10} {cells} MRR: {report.mrr:.2f}" -async def run(questions_path: pathlib.Path, top_k: int) -> None: +async def run(questions_path: pathlib.Path, top_k: int, *, rerank: bool) -> None: questions = load_questions(questions_path) - searcher = await Searcher.open() + searcher = await Searcher.open(rerank=rerank) overall, per_tag, scores = await evaluate(searcher, questions, top_k=top_k) + print(f"rerank: {'on' if rerank else 'off'}") print(_format("overall", overall, DEFAULT_KS)) for tag in sorted(per_tag): print(_format(tag, per_tag[tag], DEFAULT_KS)) @@ -157,8 +158,13 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--questions", type=pathlib.Path, default=DEFAULT_QUESTIONS) parser.add_argument("--top-k", type=int, default=max(DEFAULT_KS)) + parser.add_argument( + "--no-rerank", + action="store_true", + help="score the retriever alone, to measure what reranking adds", + ) args = parser.parse_args() - asyncio.run(run(args.questions, args.top_k)) + asyncio.run(run(args.questions, args.top_k, rerank=not args.no_rerank)) if __name__ == "__main__": diff --git a/slack_index/rerank.py b/slack_index/rerank.py new file mode 100644 index 0000000..e3ffecf --- /dev/null +++ b/slack_index/rerank.py @@ -0,0 +1,66 @@ +"""Cross-encoder reranking of the candidate pool. + +The retriever embeds query and document separately, so it can only compare them +through one vector each. A cross-encoder reads the pair together and scores the +match directly: slower, so it only ever sees the shortlist the retriever produced. +""" + +from __future__ import annotations + +import asyncio +import os +from typing import TYPE_CHECKING + +import torch +from sentence_transformers import CrossEncoder + +if TYPE_CHECKING: + from slack_index.search import Hit + + +def default_device() -> str: + """Scoring 20 pairs per query is the slowest step in a search; on this laptop + the GPU is an order of magnitude faster than the CPU fallback.""" + override = os.environ.get("SLACK_INDEX_RERANK_DEVICE") + if override: + return override + if torch.backends.mps.is_available(): + return "mps" + if torch.cuda.is_available(): + return "cuda" + return "cpu" + + +class Reranker: + """Loads the cross-encoder lazily: a process that never reranks never pays for it.""" + + def __init__(self, model_name: str, device: str | None = None) -> None: + self._model_name = model_name + self._device = device or default_device() + self._model: CrossEncoder | None = None + + def __coco_memo_key__(self) -> object: + return self._model_name + + def _load(self) -> CrossEncoder: + if self._model is None: + self._model = CrossEncoder(self._model_name, device=self._device) + return self._model + + async def rank(self, query: str, hits: list[Hit], top_k: int) -> list[Hit]: + """Reorder *hits* by cross-encoder score and keep the best *top_k*. + + Reranking can only reorder what it is given — a document the retriever + never returned cannot be rescued here. + """ + if not hits: + return [] + model = self._load() + pairs = [(query, hit.text) for hit in hits] + scores = await asyncio.to_thread(model.predict, pairs) + rescored = [ + hit.with_score(float(score)) + for hit, score in zip(hits, scores, strict=True) + ] + rescored.sort(key=lambda hit: hit.score, reverse=True) + return rescored[:top_k] diff --git a/slack_index/search.py b/slack_index/search.py index 73f53e7..efb4ccb 100644 --- a/slack_index/search.py +++ b/slack_index/search.py @@ -2,13 +2,14 @@ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, replace from cocoindex.connectors import lancedb from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder from lancedb.table import AsyncTable from slack_index import config +from slack_index.rerank import Reranker # One source can own many chunks; over-fetch so that collapsing them still # leaves top_k distinct sources. @@ -25,29 +26,41 @@ class Hit: text: str score: float + def with_score(self, score: float) -> Hit: + return replace(self, score=score) + class Searcher: """Holds the embedder and the open table so a run of queries pays for them once.""" def __init__( - self, table: AsyncTable, embedder: SentenceTransformerEmbedder + self, + table: AsyncTable, + embedder: SentenceTransformerEmbedder, + reranker: Reranker | None, ) -> None: self._table = table self._embedder = embedder + self._reranker = reranker @classmethod - async def open(cls) -> Searcher: + async def open(cls, *, rerank: bool = True) -> Searcher: settings = config.Settings.from_env() conn = await lancedb.connect_async(str(config.LANCEDB_URI)) table = await conn.open_table(config.TABLE_NAME) - return cls(table, SentenceTransformerEmbedder(settings.embed_model)) + return cls( + table, + SentenceTransformerEmbedder(settings.embed_model), + Reranker(config.RERANK_MODEL) if rerank else None, + ) async def search(self, query: str, top_k: int) -> list[Hit]: """Best chunk per source, ranked — a thread that chunked into ten pieces should occupy one result slot, not ten.""" + wanted = max(top_k, config.RERANK_CANDIDATES) if self._reranker else top_k vector = await self._embedder.embed(query) request = await self._table.search(vector, vector_column_name="embedding") - rows = await request.limit(top_k * _CANDIDATE_FACTOR).to_list() + rows = await request.limit(wanted * _CANDIDATE_FACTOR).to_list() hits: list[Hit] = [] seen: set[str] = set() @@ -67,6 +80,8 @@ async def search(self, query: str, top_k: int) -> list[Hit]: score=1.0 - row["_distance"], ) ) - if len(hits) == top_k: + if len(hits) == wanted: break - return hits + if self._reranker is None: + return hits + return await self._reranker.rank(query, hits, top_k) From 9220b7bdb48fe84168689df5c564e5a83aecfc30 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:08:33 +0900 Subject: [PATCH 09/12] slack: shrink the reranker's shortlist to ten MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sweeping the pool over one model load shows the size was chosen too generously: ten and twenty candidates score the same MRR, ten runs in 0.98s per question against 2.33s, and past twenty the extra candidates only give the cross-encoder more to confuse itself with — recall@3 slips from 0.96 to 0.94 at thirty and MRR to 0.87 at fifty. The sweep also settles what to do about the late-interaction embedding model, which was next on the list: a better retriever can only help by lifting the answer into the shortlist, and the current one already does that for 99% of the questions while the reranker needs only ten of them. The headroom is about one question, against the cost of multi-vector storage and MaxSim scoring. The error that is left sits between recall@1 at 0.79 and recall@3 at 0.96 — an ordering problem inside the top three, not a retrieval one. --- slack_index/config.py | 8 +++++--- slack_index/evals.py | 30 ++++++++++++++++++++++++++---- slack_index/search.py | 13 ++++++++++--- 3 files changed, 41 insertions(+), 10 deletions(-) diff --git a/slack_index/config.py b/slack_index/config.py index d9f1040..3acaa05 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -31,9 +31,11 @@ # Cross-encoder over the shortlist. Korean-capable, same family as the embedder. RERANK_MODEL = "BAAI/bge-reranker-v2-m3" -# How many distinct sources the retriever hands the reranker. Recall above this is -# unreachable, precision below it is the reranker's to fix. -RERANK_CANDIDATES = 20 +# How many distinct sources the retriever hands the reranker. Measured: 10 and 20 +# score identically while 10 is 2.4x faster, and 30-50 start costing accuracy — +# extra candidates are extra distractors. The pool never shrinks below the +# requested result count. +RERANK_CANDIDATES = 10 # Bulk extraction over short conversations: the cheapest current model is enough. DISTILL_MODEL = os.environ.get("SLACK_INDEX_DISTILL_MODEL", "claude-haiku-4-5") diff --git a/slack_index/evals.py b/slack_index/evals.py index 4157362..2d411b5 100644 --- a/slack_index/evals.py +++ b/slack_index/evals.py @@ -93,6 +93,7 @@ async def evaluate( *, ks: tuple[int, ...] = DEFAULT_KS, top_k: int | None = None, + candidates: int | None = None, ) -> tuple[Report, dict[str, Report], Scores]: limit = top_k or max(ks) overall = Report() @@ -100,7 +101,7 @@ async def evaluate( scores = Scores() for question in questions: - hits = await searcher.search(question.question, limit) + hits = await searcher.search(question.question, limit, candidates=candidates) top_score = hits[0].score if hits else 0.0 if not question.answerable: scores.unanswerable.append(top_score) @@ -128,10 +129,18 @@ def _format(name: str, report: Report, ks: tuple[int, ...]) -> str: return f"{name:<10} {cells} MRR: {report.mrr:.2f}" -async def run(questions_path: pathlib.Path, top_k: int, *, rerank: bool) -> None: +async def run( + questions_path: pathlib.Path, + top_k: int, + *, + rerank: bool, + candidates: int | None = None, +) -> None: questions = load_questions(questions_path) searcher = await Searcher.open(rerank=rerank) - overall, per_tag, scores = await evaluate(searcher, questions, top_k=top_k) + overall, per_tag, scores = await evaluate( + searcher, questions, top_k=top_k, candidates=candidates + ) print(f"rerank: {'on' if rerank else 'off'}") print(_format("overall", overall, DEFAULT_KS)) @@ -163,8 +172,21 @@ def main() -> None: action="store_true", help="score the retriever alone, to measure what reranking adds", ) + parser.add_argument( + "--candidates", + type=int, + default=None, + help="how many sources the reranker sees (default: config.RERANK_CANDIDATES)", + ) args = parser.parse_args() - asyncio.run(run(args.questions, args.top_k, rerank=not args.no_rerank)) + asyncio.run( + run( + args.questions, + args.top_k, + rerank=not args.no_rerank, + candidates=args.candidates, + ) + ) if __name__ == "__main__": diff --git a/slack_index/search.py b/slack_index/search.py index efb4ccb..15971f3 100644 --- a/slack_index/search.py +++ b/slack_index/search.py @@ -54,10 +54,17 @@ async def open(cls, *, rerank: bool = True) -> Searcher: Reranker(config.RERANK_MODEL) if rerank else None, ) - async def search(self, query: str, top_k: int) -> list[Hit]: + async def search( + self, query: str, top_k: int, *, candidates: int | None = None + ) -> list[Hit]: """Best chunk per source, ranked — a thread that chunked into ten pieces - should occupy one result slot, not ten.""" - wanted = max(top_k, config.RERANK_CANDIDATES) if self._reranker else top_k + should occupy one result slot, not ten. + + *candidates* sizes the shortlist handed to the reranker; it is the knob that + trades reranking latency against the recall the reranker has to work with. + """ + pool = candidates if candidates is not None else config.RERANK_CANDIDATES + wanted = max(top_k, pool) if self._reranker else top_k vector = await self._embedder.embed(query) request = await self._table.search(vector, vector_column_name="embedding") rows = await request.limit(wanted * _CANDIDATE_FACTOR).to_list() From 456068b85f7a999bbd156b9b4da81b548d4c6c28 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:17:43 +0900 Subject: [PATCH 10/12] secrets: keep credentials in a sops file instead of .env MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A dotenv sits in the working tree as plaintext and is invisible to the repository, so the only record of which credentials the pipeline needs was a .env.example nobody validates — and losing the file loses the values with it. secrets.yaml is committed encrypted to the age key this machine already holds for the infrastructure repo, so the set of required credentials is reviewable and the values survive a clean checkout. Nothing reads that file at run time. The dev shell decrypts it into environment variables and a deployment will hand the same variables to the unit, so the application keeps reading only the environment. An empty variable no longer counts as a value: the cocoindex CLI auto-loads the first .env it finds upwards, and a leftover scaffold full of placeholders was blanking out credentials that were set elsewhere. --- .env.example | 19 ----------------- .envrc | 2 +- .gitignore | 3 ++- .sops.yaml | 9 +++++++++ nix/devshell.nix | 1 + secrets.yaml | 31 ++++++++++++++++++++++++++++ secrets.yaml.example | 28 ++++++++++++++++++++++++++ slack_index/app.py | 5 ++++- slack_index/config.py | 47 +++++++++++++++++++++++++++++++++---------- slack_index/rerank.py | 5 +++-- tests/test_config.py | 32 +++++++++++++++++++++++++++++ 11 files changed, 147 insertions(+), 35 deletions(-) delete mode 100644 .env.example create mode 100644 .sops.yaml create mode 100644 secrets.yaml create mode 100644 secrets.yaml.example create mode 100644 tests/test_config.py diff --git a/.env.example b/.env.example deleted file mode 100644 index 75d97d9..0000000 --- a/.env.example +++ /dev/null @@ -1,19 +0,0 @@ -# Bot token (xoxb-...) with channels:history, groups:history, files:read, users:read. -# The bot must be a member of every channel listed below. -SLACK_BOT_TOKEN=xoxb-replace-me - -# Comma-separated channel ids, e.g. C0123ABCD,C0456EFGH -SLACK_CHANNEL_IDS= - -# Distillation. An org-scoped key also needs ANTHROPIC_WORKSPACE_ID. -ANTHROPIC_API_KEY=sk-ant-replace-me -#ANTHROPIC_WORKSPACE_ID= - -# Optional -#SLACK_INDEX_DISTILL_MODEL=claude-haiku-4-5 -#SLACK_INDEX_EMBED_MODEL=nlpai-lab/KURE-v1 -# Unset or 0 indexes the channel from its first message. -#SLACK_INDEX_LOOKBACK_DAYS=90 -#SLACK_INDEX_POLL_SECONDS=60 -#SLACK_INDEX_MAX_FILE_BYTES=5242880 -#SLACK_INDEX_VAR_DIR=var diff --git a/.envrc b/.envrc index 2483236..48f139c 100644 --- a/.envrc +++ b/.envrc @@ -1,3 +1,3 @@ # shellcheck shell=bash use flake -dotenv_if_exists +eval "$("$direnv" dotenv bash <(sops -d --output-type dotenv secrets.yaml))" diff --git a/.gitignore b/.gitignore index 5d609b8..5255b32 100644 --- a/.gitignore +++ b/.gitignore @@ -13,5 +13,6 @@ __pycache__ # runtime state: engine db, vector store var/ -# secrets +# secrets: the encrypted secrets.yaml is committed, plaintext never is .env +secrets.yaml.dec diff --git a/.sops.yaml b/.sops.yaml new file mode 100644 index 0000000..dd1f9b9 --- /dev/null +++ b/.sops.yaml @@ -0,0 +1,9 @@ +creation_rules: + - key_groups: + - age: + - age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + path_regex: ^secrets\.yaml$ + - key_groups: + - age: + - age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + path_regex: ^evals/questions\.enc\.yaml$ diff --git a/nix/devshell.nix b/nix/devshell.nix index 5c41f5d..5a43eef 100644 --- a/nix/devshell.nix +++ b/nix/devshell.nix @@ -13,6 +13,7 @@ let git lmdb ruff + sops uv ]; env = { diff --git a/secrets.yaml b/secrets.yaml new file mode 100644 index 0000000..3f3d836 --- /dev/null +++ b/secrets.yaml @@ -0,0 +1,31 @@ +#ENC[AES256_GCM,data:4nxLgDN+tVSdCkWVbnmeMI0Ro9eo/lSuKsxe6434HNEUnP/Yv6IDw8JWZbrvblc5sZWgnUqI4CStdJet2PXZa+HpfUCx6Sd5WA==,iv:d+1bGA71iMcYFtv5NDOT7Q/k9Vy1+LyOpDuUj6b2KjU=,tag:jCRYG0Wun7yPBaHBi+rlhw==,type:comment] +#ENC[AES256_GCM,data:VpYCLgmUnaBx5AaY7Tj2E9Lh0umhGdUZxb9yZq+9n3pQo8AidJdYUK5KIB+V2AtdsHjOe/gGwso=,iv:fMfx7xlU92GE1NDR2aPmy/Mdbr0ItgMwUxPlRzU6K3s=,tag:BBWoyREI+8kLPCR245iVnw==,type:comment] +SLACK_BOT_TOKEN: ENC[AES256_GCM,data:cVNAnDTK1hQamns/dPXY/zodL04dOvb+O7nW/eahdQ0Onf4TWvDmUe3rWVjguiK4KQ8o++v6VV6Vfw==,iv:uu/Ujpc7IObUJu/Xuu2tcl1P8EQNybKIgGEGTdH2Klw=,tag:/cc0FdN0Gp8EPcuR3XG7dQ==,type:str] +#ENC[AES256_GCM,data:6bfENRrTG/d/4bd6/9AolE3fz6J/1RXqsxHaHFFhgvfLWcm6i8V8Vle9VLdzzmsGUfSMJS0E,iv:NUwJZuwfNaE9CaLcpSTPSyXj2yp6lp6ssKa26y8KLjI=,tag:c0QWWKxvI4BBhkBV896YHA==,type:comment] +SLACK_CHANNEL_IDS: ENC[AES256_GCM,data:0vGH9saX+EGtCCs=,iv:gZ4bIMQr8OeANw8nMat79A4Uy245+TL8sqUf0jGfVuA=,tag:N654rfTrZJltvVs01C5dsw==,type:str] +#ENC[AES256_GCM,data:/87rtbLj3SQpP+kasbwH05zqo30SHHaPtaEcG4eSKQgl9hTW7w9288bnq0pnThP50BnjYORMajyzugxWW+5Ru0zjUQ==,iv:4SIzwgrLP0PzAveX/C5PMbFbLwGNvQ8eb8U6tC3JJZ8=,tag:b7zCQ9NgVUz88FvYQhH15g==,type:comment] +ANTHROPIC_API_KEY: ENC[AES256_GCM,data:LjEV90smqXsLU1HOGD3YShEEmGalXj5cS3A1DHGRingUZH0SKox7ceeOgXWME8ztrhwOOe7BUNmSS1wUiTOgQa9fD7ILhoADzCw6gMcGGAYunC8hrWU2JHhNnSYih1qbjOczssAkf38y6BhW,iv:Uf+MGNzy/PdTus4IloN5tH7ldl8P7VOmZoFU+fnnj6s=,tag:HB8csr+7I2x+9Uk3tK+xiA==,type:str] +#ENC[AES256_GCM,data:DZfAI37opf/HUr0f6dRn/chYXPZAFPyd0AcjXwEUEcsYxTFDIgJgvCFqCl1cfqYUmgUxNHCB+4dhbc1eazsvLneUFyJvemulhk0scg==,iv:P6+Ek/oAJN20IllyUoJ1WGJ6E7jA1Nq0NbmWs3DLlpI=,tag:8Y8VDnwWbEgXYRkTwA/PYw==,type:comment] +#ENC[AES256_GCM,data:HoSUNGy3R1zwmp2LvTHKM37P6U5aRe69qzqCtURXT6BAt1F/pCZrvIMyQXU=,iv:ZRziCRWcA+Hca5I7KY+ov0iBkai8UZXd8gjdPDZ1T/Q=,tag:rnnx7Chm2IRkIshKAWGd6g==,type:comment] +#ENC[AES256_GCM,data:cI58KLZYzOIdZk+eFv9m/cbU1JJg2GOfAYedtkw+jNKbDhT+XV/AuhTh,iv:8NsIc4v4GB9qEBDHycmhu3j1zNSNVpSmtfDx0YbgfIs=,tag:DrCwC/do5WHbijZoztVttw==,type:comment] +#ENC[AES256_GCM,data:0dOad7gf7q4AvBUIsVtvHR3Hx/9x8W9yQjXRKwO5J0bWVskKgiiDW971hA==,iv:mxMy+eo4DmqfoRN8I0ueMa82krSw5Kp+Fgkh3Evqf20=,tag:TlkZgkqY1BO3LQxTW701pA==,type:comment] +#ENC[AES256_GCM,data:aCZNjkgqSQkX3WqYfPFcupWK5O6y+EXMJIbrxUoE,iv:4z+NqPAMJuhfylnubb5VDBg4pY4qV17O/EPWJPKU91o=,tag:7FDJsU1HF8clOQ0KMy+c1g==,type:comment] +#ENC[AES256_GCM,data:J4krqZs26uZbYk9+cL/RbakKUjPMqAIQBsiQKhsY68lDXt9rM9k7znlNpQTxSzCrwWj15wv1GA==,iv:1J13sn/q0HZu4uptsyQ/jCLDKB9SsxyJPIKEtJmNdbg=,tag:24MiA+B0JW/4qq5AhO/btQ==,type:comment] +#ENC[AES256_GCM,data:FqccKI0ZTaiYx/KWfp+J1l3i0EY5bKa08bHOV1Ut,iv:h8eBY6IRPJ4QfwM+k4jYEHxgTAbnEhBUZ20aN043MQQ=,tag:am/VRBvo7YO/CmTVdbB7Ww==,type:comment] +#ENC[AES256_GCM,data:zLaA23xAd1sVwBsVXD+Dun1EX6bJbohsVXctjHHa,iv:7+p50PqD8qbz9RhuqcINFGZMHCqXeu4wC0ixY3SACE4=,tag:4YuSzP1jV1/DeDYHFFZiFw==,type:comment] +#ENC[AES256_GCM,data:5VxRntT1v/dxvxPx49mhTZqjeYOpP/yAuxLilRQDsA2BHk6sdw==,iv:z8jxiqsBwvJATEBNX5jttIHIsKr/3kDJ1AMhIrAl4/Q=,tag:9Ao52YQL1WQA3pJG9mw+/Q==,type:comment] +sops: + age: + - enc: | + -----BEGIN AGE ENCRYPTED FILE----- + YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBuTGRTaDRKR0NOeTU1eEQr + NjNXZlB5bUNqR29qR2w5SjMwTjRPNUlpZkNRCmJsZWtObzcrcmx1YkRzc0ZWcjdr + ZkNOcXNON2IxdmtQaUtGVFEyM2dCK1EKLS0tIHFvajY3dVBzNEJBYW9oTkF4TnU4 + di9rd0FJTGV1TUJPSHRpU0NHancwZFEK48gfvnkzQoPgD0kQGrwxzYFp+Agp4k7K + C1vBUNEgRXrv0zpnNL7CChXKAb2zzfgGN9x1NOkSQ8Yd7NGjz1cfJg== + -----END AGE ENCRYPTED FILE----- + recipient: age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + lastmodified: "2026-09-23T06:33:35Z" + mac: ENC[AES256_GCM,data:PoO2ZUovKek6lSkQhgkofnn4CEoyAL6q3knv1wjI461JaxJ9NYSmdKu9jYlyQy/pC3OzVKtsNLWG8n5Mk98H51gY9NtVFLLHLlrIqyEGMPvi25wfKql2ft4AJVlTd+U8qTcDBb9Dv6rLgMggFl9F4YuVN0lb4V25+AnC2EFiDdA=,iv:B3hOWQh1IKPx0xLUK5usF1Nqu5OJzraTklzukCa3zZs=,tag:idVox70tlxxGQVqcyNThDw==,type:str] + unencrypted_suffix: _unencrypted + version: 3.13.3 diff --git a/secrets.yaml.example b/secrets.yaml.example new file mode 100644 index 0000000..7fd9013 --- /dev/null +++ b/secrets.yaml.example @@ -0,0 +1,28 @@ +# Copy to secrets.yaml and encrypt it: `sops -e -i secrets.yaml`. +# Afterwards edit in place with `sops secrets.yaml`; the encrypted file is +# committed, the plaintext never is. +# +# Nothing reads this file at run time. The dev shell's .envrc decrypts it into +# environment variables; in production the unit supplies the same variables from +# its own credentials. The application only ever reads the environment. + +# Bot token with channels:history, groups:history, files:read, users:read. +# The bot must be a member of every channel listed below. +SLACK_BOT_TOKEN: xoxb-replace-me + +# Comma-separated channel ids, e.g. C0123ABCD,C0456EFGH +SLACK_CHANNEL_IDS: "" + +# Distillation. An org-scoped key also needs ANTHROPIC_WORKSPACE_ID. +ANTHROPIC_API_KEY: sk-ant-replace-me +ANTHROPIC_WORKSPACE_ID: "" + +# Optional overrides; the committed defaults in slack_index/config.py are the +# source of truth, these are for experiments. +#SLACK_INDEX_EMBED_MODEL: nlpai-lab/KURE-v1 +#SLACK_INDEX_DISTILL_MODEL: claude-haiku-4-5 +#SLACK_INDEX_RERANK_DEVICE: mps +# Unset or 0 indexes the channel from its first message. +#SLACK_INDEX_LOOKBACK_DAYS: "0" +#SLACK_INDEX_POLL_SECONDS: "60" +#SLACK_INDEX_MAX_FILE_BYTES: "5242880" diff --git a/slack_index/app.py b/slack_index/app.py index eecd7ef..ad39384 100644 --- a/slack_index/app.py +++ b/slack_index/app.py @@ -34,7 +34,10 @@ async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None] builder.provide( DISTILLER, Distiller( - AsyncAnthropic(default_headers=config.anthropic_headers()), + AsyncAnthropic( + api_key=config.anthropic_api_key(), + default_headers=config.anthropic_headers(), + ), config.DISTILL_MODEL, ), ) diff --git a/slack_index/config.py b/slack_index/config.py index 3acaa05..0828581 100644 --- a/slack_index/config.py +++ b/slack_index/config.py @@ -38,7 +38,7 @@ RERANK_CANDIDATES = 10 # Bulk extraction over short conversations: the cheapest current model is enough. -DISTILL_MODEL = os.environ.get("SLACK_INDEX_DISTILL_MODEL", "claude-haiku-4-5") +DISTILL_MODEL = "claude-haiku-4-5" DISTILL_MAX_TOKENS = 1024 DISTILL_TEMPERATURE = 0.0 @@ -57,41 +57,66 @@ class Settings: @classmethod def from_env(cls) -> Settings: - channels = os.environ.get("SLACK_CHANNEL_IDS", "") + channels = setting("SLACK_CHANNEL_IDS", "") or "" channel_ids = tuple(c.strip() for c in channels.split(",") if c.strip()) if not channel_ids: raise RuntimeError( - "SLACK_CHANNEL_IDS is empty: set it to a comma-separated list of " - "channel ids (e.g. C0123ABCD,C0456EFGH)" + "SLACK_CHANNEL_IDS is empty: set it to a comma-separated list of channel " + "ids, e.g. C0123ABCD,C0456EFGH" ) return cls( channel_ids=channel_ids, - embed_model=os.environ.get("SLACK_INDEX_EMBED_MODEL", EMBED_MODEL), + embed_model=setting("SLACK_INDEX_EMBED_MODEL", EMBED_MODEL) or EMBED_MODEL, lookback=_lookback_from_env(), poll_interval=datetime.timedelta( - seconds=float(os.environ.get("SLACK_INDEX_POLL_SECONDS", "60")) + seconds=float(setting("SLACK_INDEX_POLL_SECONDS", "60") or "60") ), max_file_bytes=int( - os.environ.get("SLACK_INDEX_MAX_FILE_BYTES", str(5 * 1024 * 1024)) + setting("SLACK_INDEX_MAX_FILE_BYTES", str(5 * 1024 * 1024)) + or str(5 * 1024 * 1024) ), ) +def setting(name: str, default: str | None = None) -> str | None: + """Configuration arrives as environment variables — from sops through direnv in + a dev shell, from the unit's credentials in production. + + An empty value counts as unset: a stray .env, which the cocoindex CLI auto-loads + from the first one it finds upwards, would otherwise blank out a credential that + the environment actually carries. + """ + return os.environ.get(name) or default + + def _lookback_from_env() -> datetime.timedelta | None: """Unset or 0 means the whole channel history.""" - days = float(os.environ.get("SLACK_INDEX_LOOKBACK_DAYS", "0")) + days = float(setting("SLACK_INDEX_LOOKBACK_DAYS", "0") or "0") return datetime.timedelta(days=days) if days > 0 else None +def anthropic_api_key() -> str: + """Passed explicitly: the SDK reads only the environment, and the key lives in + the secrets file.""" + key = setting("ANTHROPIC_API_KEY") + if not key: + raise RuntimeError( + "ANTHROPIC_API_KEY is not set; the dev shell loads it from secrets.yaml" + ) + return key + + def anthropic_headers() -> dict[str, str]: """An org-scoped API key must name the workspace on every request; a workspace-scoped key needs nothing.""" - workspace = os.environ.get("ANTHROPIC_WORKSPACE_ID") + workspace = setting("ANTHROPIC_WORKSPACE_ID") return {"anthropic-workspace-id": workspace} if workspace else {} def bot_token() -> str: - token = os.environ.get("SLACK_BOT_TOKEN") + token = setting("SLACK_BOT_TOKEN") if not token: - raise RuntimeError("SLACK_BOT_TOKEN is not set") + raise RuntimeError( + "SLACK_BOT_TOKEN is not set in secrets.yaml or the environment" + ) return token diff --git a/slack_index/rerank.py b/slack_index/rerank.py index e3ffecf..1dc5672 100644 --- a/slack_index/rerank.py +++ b/slack_index/rerank.py @@ -8,12 +8,13 @@ from __future__ import annotations import asyncio -import os from typing import TYPE_CHECKING import torch from sentence_transformers import CrossEncoder +from slack_index import config + if TYPE_CHECKING: from slack_index.search import Hit @@ -21,7 +22,7 @@ def default_device() -> str: """Scoring 20 pairs per query is the slowest step in a search; on this laptop the GPU is an order of magnitude faster than the CPU fallback.""" - override = os.environ.get("SLACK_INDEX_RERANK_DEVICE") + override = config.setting("SLACK_INDEX_RERANK_DEVICE") if override: return override if torch.backends.mps.is_available(): diff --git a/tests/test_config.py b/tests/test_config.py new file mode 100644 index 0000000..a4d5914 --- /dev/null +++ b/tests/test_config.py @@ -0,0 +1,32 @@ +"""Configuration comes from the environment, and an empty variable is not a value.""" + +from __future__ import annotations + +import pytest + +from slack_index import config + + +def test_a_set_variable_is_used(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("SLACK_BOT_TOKEN", "xoxb-from-env") + + assert config.setting("SLACK_BOT_TOKEN") == "xoxb-from-env" + assert config.bot_token() == "xoxb-from-env" + + +def test_an_empty_variable_falls_back_to_the_default( + monkeypatch: pytest.MonkeyPatch, +) -> None: + # A leftover .env — which the cocoindex CLI auto-loads — must not blank a value out. + monkeypatch.setenv("SLACK_INDEX_POLL_SECONDS", "") + + assert config.setting("SLACK_INDEX_POLL_SECONDS", "60") == "60" + + +def test_a_missing_token_says_where_it_comes_from( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("SLACK_BOT_TOKEN", raising=False) + + with pytest.raises(RuntimeError, match="secrets.yaml"): + config.bot_token() From bcd9df01408bcbc21d6b745dd6f7c3bd59feab27 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:17:43 +0900 Subject: [PATCH 11/12] evals: commit the labelled questions encrypted MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The question set is the expensive artefact in this repository — every retrieval decision so far was settled by re-scoring the same hand-labelled questions — and it lived only in an ignored directory, where it was deleted once already. It cannot be committed in the clear because the labels quote a private channel, so it is committed through sops to the same age key as the credentials, and the scorer now says which command writes the plaintext it reads. --- evals/.gitignore | 4 ++++ evals/questions.enc.yaml | 15 +++++++++++++++ slack_index/evals.py | 10 ++++++++++ 3 files changed, 29 insertions(+) create mode 100644 evals/questions.enc.yaml diff --git a/evals/.gitignore b/evals/.gitignore index 72e8ffc..8866ef9 100644 --- a/evals/.gitignore +++ b/evals/.gitignore @@ -1 +1,5 @@ +# The labels quote channel content, so only the encrypted copy is committed. * +!.gitignore +!questions.example.yaml +!questions.enc.yaml diff --git a/evals/questions.enc.yaml b/evals/questions.enc.yaml new file mode 100644 index 0000000..6eac09e --- /dev/null +++ b/evals/questions.enc.yaml @@ -0,0 +1,15 @@ +data: ENC[AES256_GCM,data:jQOm2pPLH5Cbmlvd21T3KBFasl/baPgLE+wja75PaGAfhIvUSRV1PQEo10fBljZQUqUeekQbQr2A5lhzWBVr7wdCMj2euOrJJF3OhJSWZ/mkjhgQP9IHB88Az4eD0FH5O4h1Zbpmj7nOrsLGeZ4mswriJlN5GcjNrSSWTZ2ds3NB7dOMKBbxALJpMMqYRxmU5wkkdzNkEncb00S2FOurMbNRVwXuLkl9UyNWsGgWJSaodaQ4MfubRtLRzzDk/Um/JrtPijc04zLmljBVWC4xoO/rmej3hQByD67dSc/uF45Th07OSj+ZXU/dzhiUoc1mVDSZ7XrilbL+fNzyKJls0dXFg0+lATLDNdb204n1SuRtT7cvAaHWlAxcRB6Q5j853/0GrulTxg4kiq0SyGu+MSOCHfAePRlTxLADUoTC9tznceM5hpaf7I+u08+axmDs5c95URgQmT4orxEVC9kMM+BAwKgQN+Jq8BHhqfYlOFLJ+H66BBX3ftyErJ8Ij0oMFiM2ZuDvsdyE/VaISQtq4d4ekzk8Sl9BRNnVuuPJSPxpSacf8G/REVwwRLi5vpAXfiyQ0mUDdVOaLLP0zv5JyJV5TttDD86zCg3JJO/Dtvld5rm1vIT3Up4nNwhfW6OouMpt2xPWBm8TyLyFjOboNQavbpEl0iHJUteRDheWZM621mK7eK1dDwxgZyFP9trIhpyWGjvDdSROZ+ny7i34+0kYcwwhyiQfb1n/OVp+a5hp9x6MHu32zDjLXges93y2y0Fhflo9FOlUFB9qyuY9fIPNeCX1aSXjMHcVpg1kOvJRW7J31NmOS8tAQrb2Zs72CVPwMA58H3hfxM6oCfQ1TNjXlJ5RU6F/yCP0rCxnxOmkS4l/jnjArD/Lg6BDjRw8/2MHWD8/zIv97oc6/078XQiFJDUOIb16U1ALFayjwN+/SJ5rrgrCl/rRiXmRkMkqgHF5n68XVwm4Kou1Jcb9IjC7eNjl/sB/iDGAUCO5WW9gX0cihLLJtyRq05nXRM3caIOgUI2vYCGbqwwdCyjYuW+RjB8934ACpR6D+9EMQkfdi0xc2tpw+EGctPAaLLdAma7I9DdVzCyhPuGSbA5UA0w4U30iA4xZctHWZz2eex9LAuPndwDPzFaCkKeZaFesZlkz6ubxOXHswFpyUiDsNQ0f7cAOhr0uoRQS/MMsskDEdfS591Q/BuXYZaA2sKyFyEiAGh6FLU9QuI0VJu++8rDcO1SpEJFeTlEU1xNPDId0HvUitURWhGyLXeZ4pLpF13BzRtCcNBvVfldw0EVg38DUa4r7dyyWmoBIjXB2hatqoXfI7BqqMKdqUNF7P0VmpMe1rITE0ZstMt3rkCcSH3Q2zPO6lwgrSo5KMQPVHszGJvDY3EeTnSqo6gUKH3FPxid4ECEGzhHJSbPO1VpWawFF2/lbCP3j37m6IymYs+m4HPJtQzbxQOtO6I7lyGSPYOw+caUtCcEFYxHbx1KFMyNMyJ0QIotzUdRIVcw84c7oIqurb4lf/G4Urxk/BWpPx6NnQ0us3PKq+1n6YAjZIjzPoGZEmLiNoU9pKDASMbRJ+t2soT/CE95z8WcXQ1qPPpfAwFFggnQM81HnqdCiDlEpwdxv1QsQ9bRUl3Njtad+M/CdUl0s+C4k/8vD73RP2aoRwLo+M6rq7RG60Tec1zPrIBXgyEGIYrjZlhHN4lWFgKgOsE6JdQxGOy7zLS+TQAa86C1+ZqqMKqFdXrM2W4DSsBvwwYLb2agN5/s50tAuI/FhPcxNg70BZjfgYe4Xg1iIF//lgrjzwKYEjhXIbURb1xzXS8DhfO9J/4A62KfvrX1rlQuJIm+QO9qOAlxFZWnl988GWIZ7eo6r5XYoBQR8sWgVQtZEm8kTHjL8Qj2d0IceGW2DxC1wxTGAQ1VRJr78KjqICPUtq3WwOIJkRt2yGbpdXss5a7L2hGNdkKi1O+vtfEq7UzE70leEV9t0UBThTC3zhj3MKkS6uvI7Nd5zG/2PePO5uXzHSZyXIy1SPFloFx9QBgMuL7qvsFniklhQKW2mQgB1zuTu9DiTGHU0mup98YM/YLqm3yWgu0pWV5aalWgMUbaufluonNil24Nzww3CwvAEA6tjR2c6aumT6n9CARN4bWlRsmh9DhWEzTwAn8FZGS5MFjp+r1Jkfar79mVJ1njm1dhfYgos0jsM4CxRoQto92G80o2f235w0Gp1flnBn353Z50FJiC3/V+wFF1Y7XUfjGtcXQCg6vZBjVk6jg4J9M3brkJgHX+VERF8aewLvrn+NyhjvoSIeiUbTABTwXs6fbjyIO6uT+9MQrC8JNo/uyXGZselpRj6EhAaVFTbEIaXbkyg6x8mPnsVgwjqVGIughoEROGFNLYG6JeHzmvn3XS10sErvKrfBEgHYSJuRdqGLFgJLGuK+x2EqnyfMUQpvnTh5mnPT7+NZ+iD3/VfKBUohs1Hp493Gbgw+fRPJ9o2kaaGnVTy8N2SN7vo2Eta2QQ74WzHsgEVhVFIIZhLLrsAiPom317+uOuBlSO66rZZwjybHwd9bra6qjNIiVxTkthMys7A67+lwJEWkricSdu3SzEIgCRSlMkRmAkla0BYyCzfbCrp3quQTcz9kZ0mmFy5+SdO07pAWAvolnK249zYL2YPVJObBkVO70x8omco099dba8xKWVP2Xprdnw4zP7sU/lzlwfmuJZprqnJUMwCKGuqwhsClkJP527P1G0OewKw8EUxpr/a3uU5HXZ0U9SC5zFxEoFtlcKnVYfBM/xFsyFmPzZXhowrX9I8Cg7eKklDjhcT3b/gWYEiIR/SnzhGelbUBvCopSOj/uEPoOVn0ReYVSpGwX8T95QBQCUo7DeI2WhG9guuy2hRSbCjfq9XxDzRgD9iNe1IpHVYPvEhplxv1x+IuiaZZmAH9DcJrVNJqxFhd9C65xXOfdY2ZlLLE2khILUpcvsoWPC7mSHNZUwexSQwdMjowCBb5qgCAKi/yDj4H7C60epYtqL4RsvbGDNmw267wEmsOrGrw34DMl1C5PFawia0vnqJdFeloZrdOqlJ/ANZQdoZaf9GgPdyGn5iPoUzU+Sey/XIx+AS69F5X8qBcq0y8e9qtm/QLEgKSdpSopJkygiQiovG8uPgTu5crV86AgFPKKFAp5cSudO4eizFo61q26DbjqdPmunvOdZcZ8XNgzO/igCQk8KLFiQvtD9fwIeQyHfLV8BbzGLndEvzCzuVNmDeuj6jIUBO5wuROH99iO0kudtzYDMcJF2IXLuQETwOV9pMllMAvLujuIU2qVDIvc3osGaM+QssFL2DxQKvRH4DUoAMZ+S92SKRCArL/dXi6AMCZ058MQwXKF5W51DPRa/HWa1ehWkkaGaBbgKWdBHhp6mRvwmWFo2M3c5izMUiKzfnXbcuZQadzlElHTm/gxvcTiil5vUyqO6ctnoRTIIYYITuCSDNgtbPiNzEi8LWZ9dxPA1lFwHkELOVSrNWbHvXKIoWBCW+ocQD++GaDTBIhrnnbVYa0Eosx40oMsd6NRX2iB8jDh/HsfF7KcBwE/FJRq35QHUsbAq9p8I0LCH+emEULdaIFHK8yL6miSoM2QB5I27R2q3cRrnRKIAcrGtDfiQFTtigz1OVGCcGsIISexi+A7Thg9JSRoPdXFv2Y3xFH4N0rURUzlLRo1vmcfNRwVzZtB+Z+bH7odZyKHO3Q6+Z6U0BdUiwiN4SnR7JnLzhCxzYPizcHM6m02bUFsbRae0tca+hs5PIxvyhcLLVPwXIAavNcOj2Pqfx+rlILfJh5SMVD3cci3yYmKRCFoZ13Z+6LRaiFHd2krAxpJpmBTY9cr0LBn5svN3wFSC5wUQxiFhIlrsHaF+y+wXJ/tcWq6qWJEPJz0Po1eQhHDVBvpVJO8J6bg/low4HRB6+PeJQ7IgtkXT/7DPJnlhU5UG3SBm+7nip3sNjgD/kRydo2jNJkNABBnrqrzk9TvXX/wdJoRY9shNhSqqctpaXegglB2jTz44pNCxmS+ip0N/+sqnxaRedEcLgXgwpFwPD6Z7At/eqklm0jrnlAKW01gPzi6rFzvHuPOQ0w1iQhGA48dMbRLK/adqy6TcZBcKKjz0jtw+K9HNcSWXMKNru+AdI2hbK/GkLyZa53awmOYv3sTU7sFbl47bDjOkmVIC0ROKEBXS8oMBeK2AIlfNgSrqrtZXS9J7dSjAWjkN8sQVakf7s3ANuNGrFhQO1OOJX3EyVVhyFwGIFhNFFl/gjyN9Eut3xbCNClLMQPLVRwMSLCUvT1eFFCVZkJ2d9XGl96Ccf8rD0AE/tPJibM5vJYje3q+fhYuKN3t/AYqVr5eaBSrIuYyKiZFGcDcMQY2agzc6wM+VnKZ1hosnJeV+e07uBhFdCAuCcO50c7zsw2u7r/8R/xXQNhSDfUaPw42rMemx3nUgAgg2OL57zkH7a0yR8zII6SufY2EHhMnw8LMK3kYKPLhasAwF73V7wu6hkR5W6sCjy4Mr4fpaf5CHzR2ioRm5/w+Faz/vK8QrI1ii0eRmAN/UirXijsrTdsva68DetR2rG/EhUzdWIwKfkXcjI8yIsA5mdkh71pS/YPg4m2tYz5jwFvkC+/MFhyQLtsD2ANwTtGptdaHVf8zMslHKsFm8znY69pUDovqGjVAtAcEnzHmzIHiTVkyhi7Q6+r5C2GQc3F8VukaqUeaoTmDFKXvJ2CGKpFtQ2vpfPic+HeF46upSjhFBFabi1Nm0GkFWhsEIitY842cFCirMhG3ST2RYxgZMX+7aAl8yxeaMFZC5D1x7Km7EbypWGlIrYAB6hX1nTR7klHvCUDdPNO/w8+VqgpbwurnV0/O3rcoGlS8BrDFBCHZYP3Ym5KlLNthLwWR3+KzDW1ITCRks4hY5s9WeYFe/Zz/591gIuTXDxraEic6yObekZfNW6QNC5hLiha6KBLZrQXSCq1zHjwGvspr8SZN6PomXKRXQ1p5Z9saPRd5e92F/AY+2sT9uGQVsFwZ6Eh8rm0BLgVlNjgiWx7dWZoI6FFyWZBpzZUEBovf2a5iuIxu56h5ULoNtyYo7jlin3VQsjxkk9qxWKCPwrsWLurOtb+9skV2w2k1FbrwKDHpZCbmZpPoZe+geIhK8dALLExFji2Euux6ES59T/3fO36M7AxRd5IZ+ODofPH9QygclRarQam/Nzk/8cX5rFvNkqu1JVsekUzeW8sno7kfdPWeLHEF1/vjADOI5NqYWWEdNWloWArO1iO6IetRjvbrgwDropIBy1c+z1/DKFhjIcBi3K7Vz/2tKTYgfYS9YSP3W/r86nyHIUKOdlLC18M9xa+6hQC05V3NXmG+8ol+TTC/0jjJvnQOTuQluH8uhM5b29MIjjbUdPGXorx+res7xjwxtd0sYskIXuJoDWfCa4s3ene8lncEa2IqSe/MbCfKqYPcmBmgoGaPWyn6jwaUG7al7Wv2H78pSVIcks+t737ZiOebn5hls4v+fiWaREXQuwfetvju5DA6hywJmLzPf5df5IutMv8uZujhzd+FYKEsLLqJ35We/kpBgMAxPijRO5JZdYvP7zNzN98cMmjfUMSx59p3O5io9fg5dBp47QxBeSo9whWucCRcS7vKel6gn8ZEqzusqdjkSXXITt4kvYppSRaG0IW5ZiIoUh8s/o6Ir9unYB44hRMZxhIJsFtwPv3qhzZbBQuuqjmPcjh/9IOvHpsUCAJu6O5+BIOYoS9g6VJfGR2GVgTs2690862XWPkclwGSUxUmi2aiQgEGFtY7zoMjNCnKHkENVQMm+/FDrGFTD3tda498f5BmzgsM1pYghrmP4LLsBrVUc/ng0FfwSOVy/5UEyRQAnhlLkoMUYor+puJUzcLXNRFfh6w6oaV6zMr1MnCoTP87Dwo8VHzSmt/wZhUkhhFQvoWgRBBYPbJZiGsZnwQXGKn3m/t9Fqt0HxKZb5ucteJwqXX6UeOetlEFLJiCvSJBUWw6D0JbsPNvSQPJGgCmxaUtp6x0D5/IvKjjM8lo8nZBw++cF+ycFo+5dPq7UiYYO2X2gS/RYTLWcdWRmIbFk+VnBVvMkJztjhMvvUoXE+0jiet84yReASDqFRH1s6lHsPfmPz+PJE2dudkjEKZzSBr84WcWf4DsWqq7yc8EERTPWfMVQPQmDzL6Kzf5dlcfeTZanhZaM165jJcADpI4aeLShhJlTPiSYxEg6+cvk+T/zVgQmefYQDhSABDKjzRk0nEwjVjOEV9IgJb+YybxuHX6Szs9Mguzbr3n+Lj63UVYmSjBJfuW0lh6acnOLV+7mPmZvgTEhA+hplsNqqufBMa+hURBXJ118XYGJ9v9JoaUuXgRJFmAnddtf4TcYo+7vOeBfizeCELVBmSL5s5KpPVx340KujHwAe5ZQsnDKHdHamHGRaZ7ZpzNJtBkcDlR+EMTbPBzM6tHuGkVZDzbSpID1+rTYhXGZPWpURHvac7C5oPylfJ3Yu9LplpDThUmqWegIfQqdPVGyPBLEdNct9UsCPxqyI0IM31ZueWJtK1iSx340OHrs0ybDV78VbWlB5bxooZWaN8O9JlOrplpdsheQ/AWQUtH2s87YymlY/uaTfw5SJmSAsbyMMOG2pecu4OC65KBp/DEKOgg9L6xtz1ucbPf2ltH9bNE1pZTRh5zVko184+T4WDe27q4/Yor/5iKbo9JUS7MUSDYHKZ/03Yi3kWVPbE1eyuWXt5B3Sf921MJ/OX52wD7wl38cQ76WLJ/96M/a6OZTFblHwj4xqKUumGEiNxg4qQQppHRwkrTd0C9DFHMn2hDffD3wPRclssUR6UdYkFNAoONqf1urrdvJ1HNRMAb+MKcZVyVz2WiiNjIeyNGzwr8GDQdn9zyeAWn0sKgEGVbX4YyxjHGVxiBk45wBZP3NZkXMK8PBgpjBsDdg4OuoBrvds4wDR89Tb52PqcDQyiGP07wHuH8WTDl33jDFiLraFzDNiWMmVJnveabwUewEb0wN2yFg1+Y3c/NxkQ8xvxY+1NsRKDoE0e+2tOiHp21yvECalpsn2fUlVou5OglKrmxGwnqSSXHgY4+9V82iGFKkIozhx4N5e01inBg1598OIHidpc8LoDSH2ZXeK6UXLrv2UxiQ0gPmiSfvBXFXuLLZa5nHvLea3+GWltenPszWu2ZwF69yIlYkBbU4hay536ZZ1mUV/VLsCD599r60zmGzuXQTLn8//1CMIffHQRyJdwrCXbVmJtLG4BjeyfJhXzubXF5EIAOExJZDE9lYSBVLdVXmgj10w6N8c5SaXAR1KKDztWfLTq0GuwC6iKTBAjGQeNsTv/oomdtYDI5f9BvohYf+wNUZDlEhjtOUyYCasVJCxDSCnNzr3+DksIeMtalfhalPGwBUcXfnBTuAOamCsOmb/eeSvR0KbT0H1oDVFjT7XEZClJebHy/0nARH5mmTYtVuGnZrX7cbUtSckWLuu+OGIpY9fB9HWSkKNS5IFbGR8g5zQ0djCj7IvTEOybleJd6C9IHrVWo/7yBDjYMq3KtU2PipnglSBjImBkYYRM6VPheDD0FqC4Mccf4vSPAS4OSHvYODwpty8htpVqgQs3k550uSH9xl3qxrBmT7Ef8YiM+Hu3ONj/gr1SQ/X2BHpXpSfUonO9IErL8mjBPPvxYEGFPApzZyHkJ9VEfX9V2cohFQ2yL7NgC8js3MK5mOCpoLAC4dldo+ZsVhUWFmSVjDyoy1mBdM81UOIrKmxLnjiXWcjh5TsEYqxgU0fibSd1r3qp/XMXwGrNq/wsEcbJn61ualUzUDMy9jkOJgP3J5mneBGS7MoYDYK+ykBFmMXs+yfv5lSoD53myQfO2WW19ECBJNQmlMh3Bgf0U71sEeZwCVv27p+cYCTFHZXB23+vK9+YutcX7UNFL+gGKDl0ZC2+HHdpns7+fp3B0xFs4wC/DnF7l1U2+6kVZC2xU4Rg4BGVpv2huvq8xsIdIYbH4m40Vkjbl8uCErOSIGEstyzDVsTwukRqHvQmYMLA9WQFNigXezgSQCx02EcT7K19LOWBzVR1F/mXEK9IrZyVzwcrbdVvqds/MCkSoujF7WSqDjKIyEav0haw3uhYcD3YcflRvlgt1ZoNKwzj1nvJGQwqKBLwP6bd8aDLy15LzIQnT8JpT65upyum12fcEVBdWdHQM9N15M5nuJdHVuPs6ZYCUpJrsY70yjuPo5sKr9CwnnDh0zKkXE5NH7/OY9apwih6QNI0cSZPz0jpX0QSWO1wBNtdJ49u55yxZG3hyhLgCTQsHSnqhlrKoGRl9HlPuZ0Vtv0LLPolsCN7yNzQoT5ooEUIdVG6e6JDL6Cob+6DeyaF50Q6nHWMzznOg1RY0qXpT4aU9dBF0wqT79MIVA+gNKthM2H7PLUsdzx4r9s8XGTdS26bVH3vnIJsiWWZdAWTEFKjOmuRPsfY7fod7AWClk9Vu53VGpsW0R7+OSYZmLGd4hQ1y9sN/ovQBxnxukVFDXL1wobQI/A5dd4pOGK50zMbHOpcXfSao/HCMOQ98DmVRAwJ5zh5InJQ1FjjKGQdzqPsG9L7AjTKPjtFZvB4CqBkc4S6p0XG3bnI/P5pC89sJ6xYOMJUZpEwDvUWyMC0FjFlkvktJZyr/4TsNY4tWqZSwLimqmzzvFlycdKfpSHQWey78regtQUTmdumignwW/PJqvXB80BlZSaCSln7hRV/REAf8w7AkQUfIDbxGuCrJjn7Ioom/2a1gCNeBCoTjGSM6npMjcwBhXlQHHozUAQ4noVk1zUrTQm0uc/oKSHOM7BGDvT8eHQ5nCQNPbnc73782kbAwpqvBqSN+d3CjBmQyao4ZGf+lj+JESC6z+LwlL1FH6bYwXGxFKJQIqzDLKBpAtL1hX7LmhKgUCkjiQ4qRrrH1SpIyEyok7R/WvujwGsDuVfUrezDrWlQ18mmgQuq31Wc0NtV0zObLAuTgG3tNPungHehXIKC6OZAwDTz8tR0zACBmlfzyrHe0E1N9ILrvpZ9/H+2SRSjmN1tev9/AYNZCgCO25C24RiBN1iFrYLKvoXywvkOvys24jnJ6lRB4Z3fFyPoKKdkYhz7zhvzZPPwUJzsRWU7BBPG22FLAuWXFZtEzeUzREzVezEl53ebsNxC5A9+/2XTTUBYfDDIwFabbCxlIxc8husKszSPGFjwBlD/6ixRqKFx7jxW1lW92qVZD1E6b8jOhrHpyovL+ou5FZ1hp2MLOp1XEnGlZrfwBmQyzUY8upowwQV6ze6okAde7pQLPgNYk1loCXIf2GoETkcHyVX5rtO3g1QlZbaEoTQFqjnVu4wprMJXM9I10l4FWq/ypVSihwU59ZdGI6mG5RdqD0VAPbuuwnrl3N25Nw+NjL8iFu1FM7Mg4qGLT280zCF+XPD+M5RGkgm3n5JHb2WbsCgPCIJCj8th6BbAAmmkuy7oa5cJinslkM+Hbi7yu0/mgaDro0CK9u30G7YMMNCXbWnOaRRHOwRe/6rYYjyOb3Ev0aYAK7eywlE86u0OGjRDYZvtnG9p6lOy7soOABNONFS2wb6rc5ihzY0fUGeNpSAIFLF4CZWXRI7kfCaVaoJAGK+WhW7lipRXwka0Zx/Gvb5YcI8Goq4Q8bFJ9BEjEH+i+ezX4DQUC/lTL+IqGMpvDLIa9WeeTx14vWVslJnOfgxQz8hzASn3y0TIYMmcn6MlipLakqWjUx3LL6OTi8SzND6PE5demhCf5MRllziczVU6HY+CYcQ22X3Z6LmkIa7iKRYqyEUVHvDBKHv8yg79y51mjlQph6GtGigIZ8BpJ2KaxbQNEXuZKdcvzQ3w/rPn2mgJxmOBs44CCeIGmwL6shNdm7ZcpMqV5HMOD82n+irWbFQE0AQz7UtHLPuUfLNx0Yp2fBgIr01sQFcxpn2oxEZAeJIBh76DvtivkE66NUy4oODXsIVbldyvnQJtuZCnW3zpj+c06amP33mRjqhslFLziXc8UNo/dAIZ98bTIupucY7Jq80jJ9rxLdVxpAxCwY3QwaqxWZxqgnAk+mCdutHygpZJbGU1jX2Krr1e9UZ1gjD7jkgw3aEgbWrtRC5VnJkDMTvhk+k9vZOFJgVL+mNliMG3/Kwk4QWKNe3XZzYNxlPo8r5j/zaHZlpYS3aRYdFqGVvfp9CCPGyTXiOxNJRGXb6Jtne8Gv6u7oLMTWV5spXZvEfT2+Nll3++7kho2ros9hOan2KnsIQkqEdShaocNQtHDQG+NbFX9RUF5kCf3qwrwpo3xOZGdgBM1h49v2TD4W/EB3816ALAfU4Y6mFbBaPsyGPOljFoRRuj4ALI+d+ix9IKn3ay+mzF7VEpBBGyY9PerleUqQfRrs7M5STjaAljiOTNcicfgIMQ9InwKaT9d1guxvZaH1snJ9rIFqVz4E8EOowvwEI9iqoN7Kp8rZG4kCDj8Dbox2HfgV1TkKRMzvsELgf0dUWBwPl2yOgnX5mNld8GSIRZr5QCUlJpVfQ2lyID8StrgwaWj/AMtBv3wpHKRT95DSmIf4QgiMxQHL8DrS/5drBPqgwtlRJs4m/QkxHHBEB2k8LbXIpCrSZl8fBQCJyJOekBibIklbHLA3a7fezT2ElFmUN26UnOCH55BJa/5476eLBVfqP1YQ4y3rPfWOl2UIeNuDr1Le37Q+yHrkv26mrZWWOPqQxs+KTagD/G+x9dxDWFMCGnt7Qv95UEDTwGyjO9YfnR01COigdigFyWUDvec8kAYHh1yt997wjweZEUHalWC7sIEa2wiykO6l2MqHHUTMMFrArM81SbE/b1bNxJP4elB1vToQdFMygYvERMqRKjINqCdNhFGERQmIfvoYwiIm5ikO032VXT1EqwG6wkUoQ1AtjAUHkOY3ARptz2Z9XvZLnlN/8dA8sXCwdHjXd6yzi+NpbRvrUzvd9xfzTzC540kGn2Ksnb+ECldEEITn/Yen/baTIrmlK3Iwgf5VYO5jO34GmXUTr0fMaNiJDRIi17Yhkr+Gol9xfoGTGy5a2+277ocWKGFHSB2cLJOV9SjI/TGcTduCZXEsQHsRDiZSgjwnSy9aDN0pz9CQFeDmCP6BS9EheCN7z1do1vwEoBL/z82wvZcMtf9GRS761hie/yrS6fveUbEsIwJqarlK6FZhVrk3czTXA0lgTWcmB1TG5rDkaGzZYSYWwD+K9fzpRveHsvBjt5CHSYDfVnWWkB6nZyaVoc0l1Z0Z945AuNNClDgIYIjD58h5QcVOe6jOrDLzeekfTyej/Mws5+PWcjstXVQYpQHtHbup0FVZM7SQkUSZvz5oZRicS7FbcvzNEbIBq3yOm8Bgdik8LMTGA4quUgXmSOmSJHeBN/mX2FP+44a2L0vhclodk3Ec74QnR0DsCyr8LGjStgwB64vdMBmt0hE1mXDo+NeJdPTG0qB/0GnAy/eZW5B3tkJ+1snZOUgIJf55m3KK41jnWAT9bGb9Vp8fJNC9F314o0O8OzQneIj0+5yJ4kZPZvrITsIigJzEB88FQPhOV1c5x6ssJZrskpYntpzsUReCeoUgmRSeIm6Jl1RWVp1DSkjSSky9rCU6yYpjXoQFLL1GI5jpx4aAa5Te4quXs3Q1jYqP+PzhrjuZMr5VdMkZccDaJezHR8kRIbc2Q34rOkqfWRj3sqOL1EgvW6yFlJQt6nj6ISwyvqxdwY5F8Dpg/KIH5Nd5RA3s8F5CdlzAPGQ4xjKkaLd91ZLg+b63UxPKaOOInzyE7BIwIOIjjORURvxMCFFKL2DY12irPnxEl900/8UntEf5XTZZYKJsgQC/qeNCYbPxaNBldFWn633hLrAz20QKvzLTIoPCFBl+ppl6t+oa4XyDKxqtMCnzW1dgew2n8/ZEx/Av57YSOvd+q+Was0oU9679042JR/n3XqV/LPgegjEunI+fUs8ZmTdsJ7YH+Ptjs2242d4R9gvddlQdytamVGKxOvsN1en4Eldz0GJ88Wt6RgPkzeZHz3zwj4R6b2GQy25ymSlHh7I0cEEoN3EFBTeUqPotLE7oG1cDUzZ52I5WlGrUyLlwOfZTNIBDWrXa56oK9qh1Vy553sehevhYPmRgBDi1ep57icypIy9ZvIVh51q/XnPXLB7KDVXGHM0O/heQq6GbO3E89D34s+86eyM7dy65gb0TQowKPo2hfhe9EUHQ/JQBawZjSujfu6OcrXJLldN4sRQ9XvniKA26zaVAVqoaN/mun6XRngx5Y+QaS7k42VDJsSnvNhRcwWa70cpzVML7H04iczmO4Es62tjv8iTbOlQQgoNKNkPGj/ILHiA92u8us7c+NSM0b1tXt/jGTrUVVCUXsggNYrLo2/U79OkIjrc/bp4/XTSChz7FDSQzmDHrDojq9ojjkXzK0VI6vxzZYgSrsU/39NuVcMmpf68R8xeWJDolU9R4t5W3qYy1TGfGO9NP2E3ld+VRBt4gNPXO4c3U7dAf9FEK/MPR9G4a+m40VrGDMY4V2RPkk7g8ryCe608ATzd9yJCne9nKjO6v7mplCJLVnwmxWBovMVzPPMSCtaNoAcP4bNANswBpc3Vvg2Fa+i7dUMrif3KG06z1sK+sKzXv0Y0k4Nw/I2FTPEJ3L/8C7P1A0Hsgb0OY2vbdUu74HLiUeVhczBv/VHTuw6xhnsVV/AMfgFUSEWglRHXO7oAQ5EJl03EQiOo5P4IG0/LwYCoo/JThcpN1f8C1T1G47EPtNzardmTdokZK9uFrRN+eOX8PBlIm+tebIKbwzSsiFvmzg1CtvXJlqw7PIbkmlGQGiVuKBI8z8dsZTRsZvMRQ+rmWpiDe0YULA0UcMns8iNbv8HOxUHWli37Ij3imsOv5Gndzz0DqdSI1m1KrLDanVHSo8QrDrdQApSnKvP+86q65qRU0uxRGwSQISwT5zP1+ST3IgmN2lEmEKbyFuhW41az0qY9j7gl3CCLNPpLWW3X7+GMrITt/1nifOQjqbUpH2BrZoQ336IAMe9uRXtqPjHsjWPXrQ9eRrEhgIlB576jd9aKcgsEgElL80gRE4AWxozo+hgITS69OeUI3udnQ8wCNoBYskplfJJupo0PdrGvVo4HRz6efbBfnZIF6gHHH8+fb/JP+ruNfNSnMPNumjUE93q7nrfjX0yusTE3jny51um1rUtvflCCxFsxFcJsgbe6HTltake6CMaxhw+Ux3G/GFct/K46TAFHj8c9T8K+bwu7pCKBNvXZIoY6UBTAAA2MouXJw2+F2AtMAyM9KmyxVUQXLc/aYS3c3/x1W7BWSepBF+nk/hWN1kiofIlf79fSjME8tpFxUckuZu8cxO/UNQC6qEYByZigwoBR7Mbdii25kK9/ZhKL0bRJ/FmUNuo5vNKNekuEKclJtXR1Wi3V8Ts/WF1+unrjVSUfjachR25F5CDvvKZzW3ZyREQH+mBnrq82oB9pBaG2VyrW4B73fSKt+odOOrzKo9oBkX7oWeh04NMWTQmeHINbOAQeBhEYQ43gCg3Wwh2fy6BMrqNnHADLjHrNIjs69Vep7E7VK220xIZFRM+OCVyRH5H3HsPM5S21ui9Vi5/1bFtFlQ4dGzTkJ4TB3cCqx2qH2PTJGQxQ14Sl1ox3egSQdakPMyvegm/QHhZkJY3kml8I23trTlvfW/I/xWWvpkKpkAAM7o208XzCZW7Y/17nc2jlAFHKXiWT7um7JQVfE3iDhjE7xGYBWcRVocdzdio/NW/qJB9ov5BcViLt/AfzPlajWQcqmifcUNCOvm3bNiZRIKq/TPR2ziiiXtiQ4Kxn7tUkTEgoX1C3VIGCpMapEwkkCt2Y9CC5CpFgeK72IVeWJYzgSJCtPC/pC+TzmWUC8a7e2QH1BBYM3oivRKDROv+IRB/ku2edwEN2HjGHdKklbE568tk8OpXOIarxcR/fOwcCYzHgf1vyleBfaDN4pbM0WBJxv3c+nk3/tQS744VFYtW2Wybiv4LyZBb2nj7htqnY/mw7FM32abqtCTZdYMlHqM+97Z5pM2Tpq5c6jCqZ2vfRX3N6LSMWmudWsEJxPFSxio7O6bUvkgcq6V0p/r8o+eOhTUzw9VRpQzRaisPYYIZllEmlrIgKcPifCuW3v0vLbZxkBZx70uGxK6F0ZW6a2peLaT0ARAMhq7HmI/obP3fS90kpXtlkqRwM7U7rN03Dl9SYLTCgkTQQoBdSXas3c63oENMA1L9C5o4/5HTrfUFFzwJBstSOdefVRxALOYcrmBRKnkyqCm4MbRS7Kqe0xOg3VISfIRyMMRaV2CKiMmAJE94IgnLIacWGPcJ/2XfKaoJuhoXRf7GkG5GD1UbH7Tu91pxIBzW1oThb+B+XNjMyMmzza+/WHa1UN1po/XVKIGdMUcG3BelyXCiihh2HIrrdFb3RnFAhzlIQ92d8fW1RN82Q4kPCBW7YvYFkLlA7nqr0GML4PXXWJRWbJkdxNcMbHzY2IW1H5mBwYXLCksIHwCCX8ZS0ekEejqoWylJte++V+pAfn1BO+BsoMWam9he2t2qLa36qkI5d3v8Qfh5MS67emmmXaGQk94TAJvg0BA5+IXEehztQ5flAygIsIyCCT47UdCNC9JM8/1a++cV2F/OBXgTS4hVhGqO+OlSKUL3HhZxfe3XJAlYGtkWxcTGAN5p2XBli35eW+IVGQYT+qMI6sDCiw5scaZKZlzZf1huHzCO1s55/LEx8TINHixEA4rC5VHnzIh2eJY05hZkSpf1g1rBC+H6YlzN2BcisfePciAuQ7+/9EA5zRsmw6VBJboRziNGi6gAA/FotrXCbosP/OqZJox+zAGnOfAre/2HfYiOsI+GnG4gkoQfp5LTplq+0dRXw6hYNbUR8qNdTwC2M7FS5210tU2GF44ZctJpzjRsno5+U8cKwFTz+lN3XcGxbciy+6pizY+mY2yY7/TDQDmrJ6p6kmQcMRU44UpvV7WHpoMGa/Fd5Pk/sbgUXaOPGeYYf+adMu2J40IYbTNhmPW1iw8e5o=,iv:BhzkNDko9jS64JhDMjlyAe2o4M7xRZvVRK9w5Kb7UxA=,tag:HNniTwaqQZKJ2k22lLxhGw==,type:str] +sops: + age: + - enc: | + -----BEGIN AGE ENCRYPTED FILE----- + YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA5VEhjYjc5SWNobm9BV0VR + bFU5Y2FhMS9GL2NFRWFBbFVKUFhDdnVVYlg4CjIwTEF2T1U4SHpjd3pjMmxHSzZY + SE80NXA1M0R2MFVPNTZiMi9ZcXZ5YU0KLS0tIHc3MlFQTGZiZVBLTnM1eklUamly + WmFiVFMrMVBRcGprVmlHQURQbDRGa1UKCrO5Fjn4EmOF2+b5kPGlJPaB93oFTJxY + nXPniqdN+jsDc5Z8uoCpGDybOtJn67wiwKZ1l7fniIVu3gAp7t4g1g== + -----END AGE ENCRYPTED FILE----- + recipient: age1730f3cxdyh56zw8xcvlmpa7u2x7353wu4u0e58kyx24rsefgp98sxehm6s + lastmodified: "2026-09-23T07:12:51Z" + mac: ENC[AES256_GCM,data:biR2CcelZsxkSaIIoDg1fnTMgaVfT3V8xE7SlbP/EabudvCPBOLXmuXbJngTLtUINV8WTDaTjqcWFOllCM3sywYXPfLkPg9auKos1t3A1a+825s/dnQMERN6UbCHUhPs9d/w1Uxnw4dbzCTVQY6i4hAqTZip47eVDJw/GZzJOW8=,iv:5qhp9bssLL2VTTBqelwsvBNgFuyLNOp/bfg/8PMpYIo=,tag:BZ6gGQjgWhRdLeYUSkuRlg==,type:str] + version: 3.13.3 diff --git a/slack_index/evals.py b/slack_index/evals.py index 2d411b5..efbd25b 100644 --- a/slack_index/evals.py +++ b/slack_index/evals.py @@ -21,6 +21,9 @@ from slack_index.search import Searcher DEFAULT_QUESTIONS = pathlib.Path("evals/questions.yaml") +# The labels quote channel content, so the repo carries only this encrypted copy; +# `sops -d` writes the plaintext the scorer reads. +ENCRYPTED_QUESTIONS = pathlib.Path("evals/questions.enc.yaml") DEFAULT_KS = (1, 3, 10) @@ -53,6 +56,13 @@ def mrr(self) -> float: def load_questions(path: pathlib.Path) -> list[Question]: + """Read the plaintext question set, or decrypt the committed one when a fresh + checkout has no plaintext yet.""" + if not path.exists() and path == DEFAULT_QUESTIONS: + raise FileNotFoundError( + f"{path} is not there; decrypt the committed copy first:\n" + f" sops -d {ENCRYPTED_QUESTIONS} > {path}" + ) raw = yaml.safe_load(path.read_text()) or [] return [ Question( From 75bb69c5a12cb27f9443942de2de0c8294376db8 Mon Sep 17 00:00:00 2001 From: mulatta <67085791+mulatta@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:44:26 +0900 Subject: [PATCH 12/12] treefmt: exclude encrypted files --- .envrc | 2 ++ evals/questions.example.yaml | 3 --- nix/treefmt.nix | 6 +++++- 3 files changed, 7 insertions(+), 4 deletions(-) diff --git a/.envrc b/.envrc index 48f139c..5a8fc5e 100644 --- a/.envrc +++ b/.envrc @@ -1,3 +1,5 @@ # shellcheck shell=bash use flake + +# shellcheck disable=SC2154 eval "$("$direnv" dotenv bash <(sops -d --output-type dotenv secrets.yaml))" diff --git a/evals/questions.example.yaml b/evals/questions.example.yaml index c3740d3..2eea3ea 100644 --- a/evals/questions.example.yaml +++ b/evals/questions.example.yaml @@ -16,15 +16,12 @@ # temporal — scheduling, or what happened when # file — the answer lives in a shared file, not a message # people — who said or owns something - - question: "did we settle on bumping that library" expected: ["1700000000.000100"] tags: [decision] - - question: "what was the env var the deploy script reads" expected: ["1700000100.000200"] tags: [lexical] - - question: "the design doc shared last week" expected: ["F00EXAMPLE1", "1700000200.000300"] tags: [file] diff --git a/nix/treefmt.nix b/nix/treefmt.nix index bd1ee13..a95986d 100644 --- a/nix/treefmt.nix +++ b/nix/treefmt.nix @@ -24,5 +24,9 @@ settings.formatter.ruff-check.priority = 1; settings.formatter.ruff-format.priority = 2; - settings.global.excludes = [ "**/.direnv/**" ]; + settings.global.excludes = [ + "**/.direnv/**" + "secrets.yaml" + "**/*.enc.yaml" + ]; }