Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

## Unreleased

- Preserve all reported public channel, provider, and plugin names instead of truncating each list to 32 before validation; keep upload and Analytics Engine byte limits covered by regression tests.
- Accept strictly validated, identifier-free update outcomes in a separate, explicitly configured dataset; add a side-effect-free capability check while keeping production collection unbound. Thanks @roboclaw-bot, @fuller-stack-dev, and @vincentkoc.
- Refresh Wrangler, Cloudflare Worker types, and Vitest with the matching workerd runtime and npm lockfile.
- Discard malformed UTF-8 feature-statistics uploads instead of repairing and recording them, while preserving update responses.
Expand Down
5 changes: 5 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,11 @@ before parsing. Identity fields remain length-bounded and character-filtered. Fe
complete identifiers of at most 64 characters; malformed or overlength IDs are dropped, never
repaired or truncated into another name.

Accepted feature lists retain all known public names after case folding and deduplication;
there is no per-list 32-name cutoff. The upload cap and retained public vocabulary bound
storage, with a byte-budget test guarding vocabulary refreshes against Analytics Engine's
combined blob limit.

The geography fields are co-located with the existing identity and feature columns in the same
Analytics Engine row and dataset, not stored separately. Analytics Engine retains data for
**three months** under its [published limits](https://developers.cloudflare.com/analytics/analytics-engine/limits/).
Expand Down
12 changes: 2 additions & 10 deletions src/index.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
import { keepKnownNames, normalizeVersion } from "./allowlist.js";
import { normalizeVersion } from "./allowlist.js";
import { buildDataPoint } from "./analytics.js";
import type { Env } from "./env.js";
import { readCappedBody } from "./feature-stats.js";
Expand Down Expand Up @@ -94,20 +94,12 @@ async function mayRecord(request: Request, env: Env): Promise<boolean> {

function recordRequest(request: Request, env: Env, features: FeatureStats | undefined): void {
const identity = parseClientIdentity(request.headers.get("user-agent"));
const validated = features
? {
...features,
channels: keepKnownNames(features.channels),
providerFamilies: keepKnownNames(features.providerFamilies),
plugins: keepKnownNames(features.plugins),
}
: undefined;

try {
env.TELEMETRY.writeDataPoint(
buildDataPoint(
{ ...identity, version: normalizeVersion(identity.version) },
validated,
features,
parseRequestGeography(request.cf),
),
);
Expand Down
8 changes: 5 additions & 3 deletions src/payload.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@
* malformed or hostile request can never widen what this service records.
*/

import { keepKnownNames } from "./allowlist.js";

export type ClientIdentity = {
version: string;
platform: string;
Expand All @@ -33,7 +35,6 @@ const UNKNOWN = "unknown";
const MAX_FIELD_LENGTH = 64;
// Bound matching work before the regex; five stored fields fit within this limit.
const MAX_USER_AGENT_LENGTH = 512;
const MAX_LIST_ITEMS = 32;
const MAX_COUNT = 1_000_000;

function sanitizeField(value: string | undefined): string {
Expand Down Expand Up @@ -69,8 +70,9 @@ function sanitizeList(value: unknown): string[] {
!/[^A-Za-z0-9._/-]/u.test(entry) &&
entry !== UNKNOWN,
);
// Sorted + de-duplicated so identical installs produce identical rows.
return [...new Set(items)].sort().slice(0, MAX_LIST_ITEMS);
// The retained public vocabulary bounds storage. Truncating input first would
// let unknown names or case duplicates evict valid public names.
return keepKnownNames(items);
}

function sanitizeCount(value: unknown): number {
Expand Down
15 changes: 15 additions & 0 deletions test/feature-stats.test.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import { describe, expect, it } from "vitest";
import { MAX_BODY_BYTES, readFeatureStats } from "../src/feature-stats.js";
import { PUBLIC_NAMES } from "../src/public-vocabulary.js";

const FEATURE_BODY = JSON.stringify({
schema: 1,
Expand Down Expand Up @@ -41,6 +42,20 @@ function countingBody(totalBytes: number, chunkSize: number) {
}

describe("readFeatureStats", () => {
it("retains the full public inventory while omitting private names", async () => {
const request = new Request("https://telemetry.example/api/latest-version", {
method: "POST",
body: JSON.stringify({ schema: 1, features: {
plugins: [...PUBLIC_NAMES, "CODEX", "private-plugin"],
pluginsEnabled: PUBLIC_NAMES.length + 1,
} }),
});
await expect(readFeatureStats(request)).resolves.toMatchObject({
plugins: PUBLIC_NAMES,
pluginsEnabled: PUBLIC_NAMES.length + 1,
});
});

it("accepts a documented POST body under the cap", async () => {
const request = new Request("https://telemetry.example/api/latest-version", {
method: "POST",
Expand Down
26 changes: 26 additions & 0 deletions test/latest-version.test.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import type { Env } from "../src/env.js";
import worker from "../src/index.js";
import { PUBLIC_NAMES } from "../src/public-vocabulary.js";

type MemoryCache = {
store: Map<string, Response>;
Expand Down Expand Up @@ -125,6 +126,31 @@ describe("GET | POST /api/latest-version", () => {
});
});

it("records all public feature names beyond 32 while preserving the version answer", async () => {
const names = [
...Array.from({ length: 40 }, (_, index) => `aaa-private-${index}`),
...[...PUBLIC_NAMES].reverse(), "CODEX", "codex",
];
const response = await worker.fetch(updateRequest(geography, {
method: "POST",
body: JSON.stringify({ schema: 1, features: {
channels: names, providerFamilies: names, plugins: names,
pluginsEnabled: PUBLIC_NAMES.length, sessionsLast24h: 14,
} }),
}), env);
expect(response.status).toBe(200);
await expect(response.json()).resolves.toEqual({ version: "2026.9.2" });
expect(writeDataPoint).toHaveBeenCalledExactlyOnceWith({
indexes: ["2026.9.2"],
blobs: [
"2026.9.2", "linux", "x64", "node/v24.0.0", "gateway",
...Array(3).fill(PUBLIC_NAMES.join(",")),
"US", "CA", "San Francisco", "America/Los_Angeles",
],
doubles: [1, PUBLIC_NAMES.length, 14],
});
});

it.each([undefined, null, false, 1, "US", []].map((cf) => ({ cf })))(
"preserves features when cf is not a metadata record: $cf",
async ({ cf }) => {
Expand Down
65 changes: 58 additions & 7 deletions test/payload.test.ts
Original file line number Diff line number Diff line change
@@ -1,8 +1,9 @@
import { describe, expect, it } from "vitest";
import { keepKnownNames } from "../src/allowlist.js";
import { buildDataPoint } from "../src/analytics.js";
import { MAX_BODY_BYTES } from "../src/feature-stats.js";
import { parseRequestGeography } from "../src/geography.js";
import { parseClientIdentity, parseFeatureStats } from "../src/payload.js";
import { PUBLIC_NAMES } from "../src/public-vocabulary.js";

describe("parseClientIdentity", () => {
it("reads version, platform, runtime, arch and surface from the client User-Agent", () => {
Expand Down Expand Up @@ -119,7 +120,7 @@ describe("parseFeatureStats", () => {
expect(parsed).toEqual(expectedFeatures);
});

it("bounds list length and coerces hostile counts", () => {
it("drops unknown names and coerces hostile counts", () => {
const parsed = parseFeatureStats({
schema: 1,
features: {
Expand All @@ -129,7 +130,7 @@ describe("parseFeatureStats", () => {
sessionsLast24h: Number.POSITIVE_INFINITY,
},
});
expect(parsed?.channels).toHaveLength(32);
expect(parsed?.channels).toEqual([]);
expect(parsed?.providerFamilies).toEqual(["openai"]);
expect(parsed?.pluginsEnabled).toBe(0);
expect(parsed?.sessionsLast24h).toBe(0);
Expand All @@ -153,17 +154,18 @@ describe("parseFeatureStats", () => {
);

it("rejects overlength tokens without truncating them to another identifier", () => {
const valid = "a".repeat(64);
const valid = "codex";
const overlength = valid + "a".repeat(64);
expect(
parseFeatureStats({
schema: 1,
features: { plugins: [`${valid}b`, valid] },
features: { plugins: [overlength, valid] },
})?.plugins,
).toEqual([valid]);
expect(
parseFeatureStats({
schema: 1,
features: { plugins: [`${valid}b`] },
features: { plugins: [overlength] },
})?.plugins,
).toEqual([]);
});
Expand All @@ -175,12 +177,36 @@ describe("parseFeatureStats", () => {
plugins: ["CODEX", "codex", "Browser", "acme-internal-crm", "cod ex"],
},
});
expect(keepKnownNames(parsed?.plugins ?? [])).toEqual([
expect(parsed?.plugins).toEqual([
"browser",
"codex",
]);
});

it.each(["channels", "providerFamilies", "plugins"] as const)(
"retains more than 32 public %s after validation and canonicalization",
(field) => {
expect(PUBLIC_NAMES.length).toBeGreaterThan(32);
const names = [...PUBLIC_NAMES].reverse();
const parsed = parseFeatureStats({ schema: 1, features: { [field]: names } });
expect(parsed?.[field]).toEqual(PUBLIC_NAMES);
},
);

it.each(["channels", "providerFamilies", "plugins"] as const)(
"does not let unknown names or case duplicates evict valid %s",
(field) => {
const names = [
...Array.from({ length: 40 }, (_, index) => `aaa-private-${index}`),
...PUBLIC_NAMES.slice(0, 32).flatMap((name) => [name.toUpperCase(), name]),
"telegram", "TELEGRAM", "zai", "z ai",
];
const parsed = parseFeatureStats({ schema: 1, features: { [field]: names } });
expect(parsed?.[field])
.toEqual([...PUBLIC_NAMES.slice(0, 32), "telegram", "zai"]);
},
);

it("accepts schema-1 feature reports that predate the optional plugins list", () => {
const { plugins, ...features } = body.features;
const parsed = parseFeatureStats({ schema: 1, features });
Expand All @@ -197,6 +223,31 @@ describe("parseFeatureStats", () => {
describe("buildDataPoint", () => {
const identity = parseClientIdentity("openclaw/2026.8.2 (darwin; node/v26.0.1; arm64; gateway)");

it("keeps the complete retained vocabulary within upload and Analytics Engine byte budgets", () => {
const body = {
schema: 1,
features: {
channels: PUBLIC_NAMES,
providerFamilies: PUBLIC_NAMES,
plugins: PUBLIC_NAMES,
pluginsEnabled: 1_000_000,
sessionsLast24h: 1_000_000,
},
};
const encoder = new TextEncoder();
expect(encoder.encode(JSON.stringify(body)).byteLength).toBeLessThanOrEqual(MAX_BODY_BYTES);
// Reserve every identity/geography field's full byte allowance. A vocabulary
// refresh must not exceed Analytics Engine's 16 KiB combined blob limit.
const point = buildDataPoint(
{ version: "v".repeat(64), platform: "p".repeat(64), arch: "a".repeat(64), runtime: "r".repeat(64), surface: "s".repeat(64) },
parseFeatureStats(body),
{ country: "US", regionCode: "ABC", city: "é".repeat(64), timezone: "T".repeat(64) },
);
expect(point.blobs.slice(5, 8)).toEqual(Array(3).fill(PUBLIC_NAMES.join(",")));
expect(point.blobs.reduce((bytes, value) => bytes + encoder.encode(value).byteLength, 0))
.toBeLessThanOrEqual(16_384);
});

it("marks rows without feature stats and still records the identity columns", () => {
expect(buildDataPoint(identity, undefined, parseRequestGeography(undefined))).toEqual({
indexes: ["2026.8.2"],
Expand Down
Loading