diff --git a/CHANGELOG.md b/CHANGELOG.md index a7282b0..907c574 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Preserve all reported public channel, provider, and plugin names instead of truncating each list to 32 before validation; keep upload and Analytics Engine byte limits covered by regression tests. - Accept strictly validated, identifier-free update outcomes in a separate, explicitly configured dataset; add a side-effect-free capability check while keeping production collection unbound. Thanks @roboclaw-bot, @fuller-stack-dev, and @vincentkoc. - Refresh Wrangler, Cloudflare Worker types, and Vitest with the matching workerd runtime and npm lockfile. - Discard malformed UTF-8 feature-statistics uploads instead of repairing and recording them, while preserving update responses. diff --git a/README.md b/README.md index 894f906..a78f9d6 100644 --- a/README.md +++ b/README.md @@ -131,6 +131,11 @@ before parsing. Identity fields remain length-bounded and character-filtered. Fe complete identifiers of at most 64 characters; malformed or overlength IDs are dropped, never repaired or truncated into another name. +Accepted feature lists retain all known public names after case folding and deduplication; +there is no per-list 32-name cutoff. The upload cap and retained public vocabulary bound +storage, with a byte-budget test guarding vocabulary refreshes against Analytics Engine's +combined blob limit. + The geography fields are co-located with the existing identity and feature columns in the same Analytics Engine row and dataset, not stored separately. Analytics Engine retains data for **three months** under its [published limits](https://developers.cloudflare.com/analytics/analytics-engine/limits/). diff --git a/src/index.ts b/src/index.ts index 42fbfe4..bcd3023 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,4 +1,4 @@ -import { keepKnownNames, normalizeVersion } from "./allowlist.js"; +import { normalizeVersion } from "./allowlist.js"; import { buildDataPoint } from "./analytics.js"; import type { Env } from "./env.js"; import { readCappedBody } from "./feature-stats.js"; @@ -94,20 +94,12 @@ async function mayRecord(request: Request, env: Env): Promise { function recordRequest(request: Request, env: Env, features: FeatureStats | undefined): void { const identity = parseClientIdentity(request.headers.get("user-agent")); - const validated = features - ? { - ...features, - channels: keepKnownNames(features.channels), - providerFamilies: keepKnownNames(features.providerFamilies), - plugins: keepKnownNames(features.plugins), - } - : undefined; try { env.TELEMETRY.writeDataPoint( buildDataPoint( { ...identity, version: normalizeVersion(identity.version) }, - validated, + features, parseRequestGeography(request.cf), ), ); diff --git a/src/payload.ts b/src/payload.ts index c1b9a16..8543e16 100644 --- a/src/payload.ts +++ b/src/payload.ts @@ -7,6 +7,8 @@ * malformed or hostile request can never widen what this service records. */ +import { keepKnownNames } from "./allowlist.js"; + export type ClientIdentity = { version: string; platform: string; @@ -33,7 +35,6 @@ const UNKNOWN = "unknown"; const MAX_FIELD_LENGTH = 64; // Bound matching work before the regex; five stored fields fit within this limit. const MAX_USER_AGENT_LENGTH = 512; -const MAX_LIST_ITEMS = 32; const MAX_COUNT = 1_000_000; function sanitizeField(value: string | undefined): string { @@ -69,8 +70,9 @@ function sanitizeList(value: unknown): string[] { !/[^A-Za-z0-9._/-]/u.test(entry) && entry !== UNKNOWN, ); - // Sorted + de-duplicated so identical installs produce identical rows. - return [...new Set(items)].sort().slice(0, MAX_LIST_ITEMS); + // The retained public vocabulary bounds storage. Truncating input first would + // let unknown names or case duplicates evict valid public names. + return keepKnownNames(items); } function sanitizeCount(value: unknown): number { diff --git a/test/feature-stats.test.ts b/test/feature-stats.test.ts index 02d61a6..52670f4 100644 --- a/test/feature-stats.test.ts +++ b/test/feature-stats.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from "vitest"; import { MAX_BODY_BYTES, readFeatureStats } from "../src/feature-stats.js"; +import { PUBLIC_NAMES } from "../src/public-vocabulary.js"; const FEATURE_BODY = JSON.stringify({ schema: 1, @@ -41,6 +42,20 @@ function countingBody(totalBytes: number, chunkSize: number) { } describe("readFeatureStats", () => { + it("retains the full public inventory while omitting private names", async () => { + const request = new Request("https://telemetry.example/api/latest-version", { + method: "POST", + body: JSON.stringify({ schema: 1, features: { + plugins: [...PUBLIC_NAMES, "CODEX", "private-plugin"], + pluginsEnabled: PUBLIC_NAMES.length + 1, + } }), + }); + await expect(readFeatureStats(request)).resolves.toMatchObject({ + plugins: PUBLIC_NAMES, + pluginsEnabled: PUBLIC_NAMES.length + 1, + }); + }); + it("accepts a documented POST body under the cap", async () => { const request = new Request("https://telemetry.example/api/latest-version", { method: "POST", diff --git a/test/latest-version.test.ts b/test/latest-version.test.ts index b0205bf..584eaf6 100644 --- a/test/latest-version.test.ts +++ b/test/latest-version.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import type { Env } from "../src/env.js"; import worker from "../src/index.js"; +import { PUBLIC_NAMES } from "../src/public-vocabulary.js"; type MemoryCache = { store: Map; @@ -125,6 +126,31 @@ describe("GET | POST /api/latest-version", () => { }); }); + it("records all public feature names beyond 32 while preserving the version answer", async () => { + const names = [ + ...Array.from({ length: 40 }, (_, index) => `aaa-private-${index}`), + ...[...PUBLIC_NAMES].reverse(), "CODEX", "codex", + ]; + const response = await worker.fetch(updateRequest(geography, { + method: "POST", + body: JSON.stringify({ schema: 1, features: { + channels: names, providerFamilies: names, plugins: names, + pluginsEnabled: PUBLIC_NAMES.length, sessionsLast24h: 14, + } }), + }), env); + expect(response.status).toBe(200); + await expect(response.json()).resolves.toEqual({ version: "2026.9.2" }); + expect(writeDataPoint).toHaveBeenCalledExactlyOnceWith({ + indexes: ["2026.9.2"], + blobs: [ + "2026.9.2", "linux", "x64", "node/v24.0.0", "gateway", + ...Array(3).fill(PUBLIC_NAMES.join(",")), + "US", "CA", "San Francisco", "America/Los_Angeles", + ], + doubles: [1, PUBLIC_NAMES.length, 14], + }); + }); + it.each([undefined, null, false, 1, "US", []].map((cf) => ({ cf })))( "preserves features when cf is not a metadata record: $cf", async ({ cf }) => { diff --git a/test/payload.test.ts b/test/payload.test.ts index 8c24af9..b4ee969 100644 --- a/test/payload.test.ts +++ b/test/payload.test.ts @@ -1,8 +1,9 @@ import { describe, expect, it } from "vitest"; -import { keepKnownNames } from "../src/allowlist.js"; import { buildDataPoint } from "../src/analytics.js"; +import { MAX_BODY_BYTES } from "../src/feature-stats.js"; import { parseRequestGeography } from "../src/geography.js"; import { parseClientIdentity, parseFeatureStats } from "../src/payload.js"; +import { PUBLIC_NAMES } from "../src/public-vocabulary.js"; describe("parseClientIdentity", () => { it("reads version, platform, runtime, arch and surface from the client User-Agent", () => { @@ -119,7 +120,7 @@ describe("parseFeatureStats", () => { expect(parsed).toEqual(expectedFeatures); }); - it("bounds list length and coerces hostile counts", () => { + it("drops unknown names and coerces hostile counts", () => { const parsed = parseFeatureStats({ schema: 1, features: { @@ -129,7 +130,7 @@ describe("parseFeatureStats", () => { sessionsLast24h: Number.POSITIVE_INFINITY, }, }); - expect(parsed?.channels).toHaveLength(32); + expect(parsed?.channels).toEqual([]); expect(parsed?.providerFamilies).toEqual(["openai"]); expect(parsed?.pluginsEnabled).toBe(0); expect(parsed?.sessionsLast24h).toBe(0); @@ -153,17 +154,18 @@ describe("parseFeatureStats", () => { ); it("rejects overlength tokens without truncating them to another identifier", () => { - const valid = "a".repeat(64); + const valid = "codex"; + const overlength = valid + "a".repeat(64); expect( parseFeatureStats({ schema: 1, - features: { plugins: [`${valid}b`, valid] }, + features: { plugins: [overlength, valid] }, })?.plugins, ).toEqual([valid]); expect( parseFeatureStats({ schema: 1, - features: { plugins: [`${valid}b`] }, + features: { plugins: [overlength] }, })?.plugins, ).toEqual([]); }); @@ -175,12 +177,36 @@ describe("parseFeatureStats", () => { plugins: ["CODEX", "codex", "Browser", "acme-internal-crm", "cod ex"], }, }); - expect(keepKnownNames(parsed?.plugins ?? [])).toEqual([ + expect(parsed?.plugins).toEqual([ "browser", "codex", ]); }); + it.each(["channels", "providerFamilies", "plugins"] as const)( + "retains more than 32 public %s after validation and canonicalization", + (field) => { + expect(PUBLIC_NAMES.length).toBeGreaterThan(32); + const names = [...PUBLIC_NAMES].reverse(); + const parsed = parseFeatureStats({ schema: 1, features: { [field]: names } }); + expect(parsed?.[field]).toEqual(PUBLIC_NAMES); + }, + ); + + it.each(["channels", "providerFamilies", "plugins"] as const)( + "does not let unknown names or case duplicates evict valid %s", + (field) => { + const names = [ + ...Array.from({ length: 40 }, (_, index) => `aaa-private-${index}`), + ...PUBLIC_NAMES.slice(0, 32).flatMap((name) => [name.toUpperCase(), name]), + "telegram", "TELEGRAM", "zai", "z ai", + ]; + const parsed = parseFeatureStats({ schema: 1, features: { [field]: names } }); + expect(parsed?.[field]) + .toEqual([...PUBLIC_NAMES.slice(0, 32), "telegram", "zai"]); + }, + ); + it("accepts schema-1 feature reports that predate the optional plugins list", () => { const { plugins, ...features } = body.features; const parsed = parseFeatureStats({ schema: 1, features }); @@ -197,6 +223,31 @@ describe("parseFeatureStats", () => { describe("buildDataPoint", () => { const identity = parseClientIdentity("openclaw/2026.8.2 (darwin; node/v26.0.1; arm64; gateway)"); + it("keeps the complete retained vocabulary within upload and Analytics Engine byte budgets", () => { + const body = { + schema: 1, + features: { + channels: PUBLIC_NAMES, + providerFamilies: PUBLIC_NAMES, + plugins: PUBLIC_NAMES, + pluginsEnabled: 1_000_000, + sessionsLast24h: 1_000_000, + }, + }; + const encoder = new TextEncoder(); + expect(encoder.encode(JSON.stringify(body)).byteLength).toBeLessThanOrEqual(MAX_BODY_BYTES); + // Reserve every identity/geography field's full byte allowance. A vocabulary + // refresh must not exceed Analytics Engine's 16 KiB combined blob limit. + const point = buildDataPoint( + { version: "v".repeat(64), platform: "p".repeat(64), arch: "a".repeat(64), runtime: "r".repeat(64), surface: "s".repeat(64) }, + parseFeatureStats(body), + { country: "US", regionCode: "ABC", city: "é".repeat(64), timezone: "T".repeat(64) }, + ); + expect(point.blobs.slice(5, 8)).toEqual(Array(3).fill(PUBLIC_NAMES.join(","))); + expect(point.blobs.reduce((bytes, value) => bytes + encoder.encode(value).byteLength, 0)) + .toBeLessThanOrEqual(16_384); + }); + it("marks rows without feature stats and still records the identity columns", () => { expect(buildDataPoint(identity, undefined, parseRequestGeography(undefined))).toEqual({ indexes: ["2026.8.2"],