Skip to content
File

Blob: test/scripts/validate-fixture-indexer.ts

typescript326 lines
1#!/usr/bin/env -S npx tsx
2/**
3 * Validation script for the streaming pack indexer against the real fixture.
4 *
5 * Tests:
6 * - Generated idx matches Git's own output byte-for-byte
7 * - Subrequest count stays within the receive-path indexer budget (5,000)
8 * - Reports timing and subrequest stats
9 *
10 * Usage:
11 * npx tsx test/scripts/validate-fixture-indexer.ts
12 *
13 * This script creates a mock R2 bucket backed by local files so the indexer
14 * can be tested outside the Cloudflare Workers runtime.
15 */
16 
17import * as fs from "node:fs";
18import * as path from "node:path";
19import * as nodeCrypto from "node:crypto";
20 
21// Polyfill crypto.DigestStream for Node.js (Cloudflare Workers-specific API).
22// This creates a WritableStream that computes a hash digest of all written data.
23if (!(globalThis.crypto as unknown as Record<string, unknown>).DigestStream) {
24 class DigestStreamPolyfill extends WritableStream<Uint8Array> {
25 digest: Promise<ArrayBuffer>;
26 constructor(algorithm: string) {
27 const alg = algorithm.replace("-", "").toLowerCase(); // "SHA-1" -> "sha1"
28 const hash = nodeCrypto.createHash(alg);
29 let resolveDigest: (value: ArrayBuffer) => void;
30 const digestPromise = new Promise<ArrayBuffer>((resolve) => {
31 resolveDigest = resolve;
32 });
33 super({
34 write(chunk) {
35 hash.update(chunk);
36 },
37 close() {
38 const buf = hash.digest();
39 resolveDigest!(buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength));
40 },
41 });
42 this.digest = digestPromise;
43 }
44 }
45 (globalThis.crypto as unknown as Record<string, unknown>).DigestStream = DigestStreamPolyfill;
46}
47 
48// Dynamic import to pick up the project's path aliases via tsx
49const { scanPack, resolveDeltasAndWriteIdx } = await import("../../src/worker/git/pack/indexer");
50const { SubrequestLimiter } = await import("../../src/worker/git/operations/limits");
51const { createLogger } = await import("../../src/worker/common/logger");
52 
53const FIXTURE_DIR = path.resolve(import.meta.dirname ?? ".", "../../uncommitted-fixture");
54const PACK_NAME = "pack-395a180893e59dad8ef9d7fa135ecd8b1b399bb1";
55const PACK_PATH = path.join(FIXTURE_DIR, `${PACK_NAME}.pack`);
56const IDX_PATH = path.join(FIXTURE_DIR, `${PACK_NAME}.idx`);
57 
58const R2_PACK_KEY = `test/fixture/${PACK_NAME}.pack`;
59// The platform hard cap is 10,000 subrequests per Worker invocation. The old
60// 900 "soft budget" was set when reads went through DO RPCs + loose objects.
61// For the streaming receive path (which reads directly from R2), the indexer
62// gets a larger share. We budget 5,000 for the indexer so the remaining ~5,000
63// is available for connectivity checks, catalog loads, and other receive-path
64// operations.
65const SUBREQUEST_BUDGET = 5_000;
66 
67// ---------------------------------------------------------------------------
68// Mock R2 bucket backed by local file system
69// ---------------------------------------------------------------------------
70 
71class MockR2Object {
72 key: string;
73 private data: Uint8Array;
74 size: number;
75 constructor(key: string, data: Uint8Array, size: number) {
76 this.key = key;
77 this.data = data;
78 this.size = size;
79 }
80 
81 async arrayBuffer(): Promise<ArrayBuffer> {
82 return this.data.buffer.slice(
83 this.data.byteOffset,
84 this.data.byteOffset + this.data.byteLength
85 ) as ArrayBuffer;
86 }
87}
88 
89function createMockBucket(files: Map<string, Uint8Array>) {
90 return {
91 async get(
92 key: string,
93 opts?: { range?: { offset: number; length: number } }
94 ): Promise<MockR2Object | null> {
95 const data = files.get(key);
96 if (!data) return null;
97 if (opts?.range) {
98 const { offset, length } = opts.range;
99 const slice = data.subarray(offset, offset + length);
100 return new MockR2Object(key, new Uint8Array(slice), slice.length);
101 }
102 return new MockR2Object(key, data, data.length);
103 },
104 async head(key: string): Promise<{ size: number } | null> {
105 const data = files.get(key);
106 return data ? { size: data.length } : null;
107 },
108 async put(key: string, data: Uint8Array | ArrayBuffer): Promise<void> {
109 const bytes = data instanceof Uint8Array ? data : new Uint8Array(data);
110 files.set(key, bytes);
111 },
112 async delete(_key: string): Promise<void> {},
113 };
114}
115 
116type MemorySnapshot = {
117 heapUsed: number;
118 rss: number;
119};
120 
121function takeMemorySnapshot(): MemorySnapshot {
122 const usage = process.memoryUsage();
123 return {
124 heapUsed: usage.heapUsed,
125 rss: usage.rss,
126 };
127}
128 
129// ---------------------------------------------------------------------------
130// Main
131// ---------------------------------------------------------------------------
132 
133async function main() {
134 if (!fs.existsSync(PACK_PATH) || !fs.existsSync(IDX_PATH)) {
135 console.error(`Fixture files not found in ${FIXTURE_DIR}`);
136 process.exit(1);
137 }
138 
139 const packData = new Uint8Array(fs.readFileSync(PACK_PATH));
140 const expectedIdx = new Uint8Array(fs.readFileSync(IDX_PATH));
141 
142 console.log(
143 `Pack: ${packData.byteLength} bytes (${(packData.byteLength / 1024 / 1024).toFixed(1)} MiB)`
144 );
145 console.log(
146 `Idx: ${expectedIdx.byteLength} bytes (${(expectedIdx.byteLength / 1024 / 1024).toFixed(1)} MiB)`
147 );
148 
149 // Create mock R2 bucket with the fixture pack.
150 const files = new Map<string, Uint8Array>();
151 files.set(R2_PACK_KEY, packData);
152 
153 const env = { REPO_BUCKET: createMockBucket(files) } as unknown as Env;
154 const log = createLogger("info", { service: "FixtureValidator" });
155 const limiter = new SubrequestLimiter(6);
156 const counter = { count: 0 };
157 const countSub = (n = 1) => {
158 counter.count += n;
159 };
160 
161 // ---- Memory baseline (after loading fixture, before indexer) ----
162 // The fixture pack and idx are intentionally loaded before this snapshot so
163 // the delta isolates the indexer itself rather than fixture setup cost.
164 global.gc?.(); // --expose-gc makes this available
165 const before = takeMemorySnapshot();
166 
167 // ---- Scan ----
168 console.log("\nScanning pack...");
169 const scanStart = Date.now();
170 const scanResult = await scanPack({
171 env,
172 packKey: R2_PACK_KEY,
173 packSize: packData.byteLength,
174 limiter,
175 countSubrequest: countSub,
176 log,
177 });
178 const scanMs = Date.now() - scanStart;
179 console.log(` Objects: ${scanResult.objectCount}`);
180 console.log(` Scan time: ${scanMs}ms`);
181 console.log(` Subrequests so far: ${counter.count}`);
182 
183 // ---- Resolve + write idx ----
184 console.log("\nResolving deltas and writing idx...");
185 const resolveStart = Date.now();
186 const resolveResult = await resolveDeltasAndWriteIdx({
187 env,
188 packKey: R2_PACK_KEY,
189 packSize: packData.byteLength,
190 limiter,
191 countSubrequest: countSub,
192 log,
193 scanResult,
194 repoId: "fixture/test",
195 lruBudget: 48 * 1024 * 1024, // 48 MiB — safe with the array-backed payload cache
196 });
197 const resolveMs = Date.now() - resolveStart;
198 const totalMs = scanMs + resolveMs;
199 
200 // ---- Memory deltas: measure what remained resident after indexer work ----
201 // heapUsed best approximates the memory pressure we care about on Workers.
202 // RSS is still useful to inspect Node process growth, but it also includes
203 // V8 and runtime pages that do not map cleanly to Workers.
204 global.gc?.();
205 const after = takeMemorySnapshot();
206 const heapDelta = after.heapUsed - before.heapUsed;
207 const rssDelta = after.rss - before.rss;
208 
209 // ---- Read generated idx ----
210 const idxKey = R2_PACK_KEY.replace(/\.pack$/, ".idx");
211 const generatedIdx = files.get(idxKey);
212 if (!generatedIdx) {
213 console.error("ERROR: No generated idx found in mock R2");
214 process.exit(1);
215 }
216 
217 // ---- Compare byte-for-byte ----
218 console.log("\nComparing idx...");
219 if (generatedIdx.byteLength !== expectedIdx.byteLength) {
220 console.error(
221 `ERROR: Size mismatch: generated ${generatedIdx.byteLength}, expected ${expectedIdx.byteLength}`
222 );
223 process.exit(1);
224 }
225 
226 let firstMismatch = -1;
227 for (let i = 0; i < expectedIdx.byteLength; i++) {
228 if (generatedIdx[i] !== expectedIdx[i]) {
229 firstMismatch = i;
230 break;
231 }
232 }
233 
234 if (firstMismatch !== -1) {
235 console.error(`ERROR: Idx mismatch at byte offset ${firstMismatch}`);
236 const ctx = 16;
237 const start = Math.max(0, firstMismatch - ctx);
238 const end = Math.min(expectedIdx.byteLength, firstMismatch + ctx);
239 console.error(
240 ` Expected: ${Array.from(expectedIdx.subarray(start, end))
241 .map((b) => b.toString(16).padStart(2, "0"))
242 .join(" ")}`
243 );
244 console.error(
245 ` Got: ${Array.from(generatedIdx.subarray(start, end))
246 .map((b) => b.toString(16).padStart(2, "0"))
247 .join(" ")}`
248 );
249 process.exit(1);
250 }
251 
252 // ---- Report ----
253 // Compute the indexer's typed-array footprint (the part we control).
254 const entryTableBytes =
255 scanResult.table.offsets.byteLength +
256 scanResult.table.types.byteLength +
257 scanResult.table.headerLens.byteLength +
258 scanResult.table.spanEnds.byteLength +
259 scanResult.table.crc32s.byteLength +
260 scanResult.table.oids.byteLength +
261 scanResult.table.decompressedSizes.byteLength +
262 scanResult.table.ofsBaseOffsets.byteLength +
263 scanResult.table.resolved.byteLength;
264 
265 const scanAuxBytes = scanResult.refBaseOids.byteLength;
266 
267 const idxViewBytes =
268 resolveResult.idxView.fanout.byteLength +
269 resolveResult.idxView.rawNames.byteLength +
270 resolveResult.idxView.offsets.byteLength +
271 resolveResult.idxView.nextOffsetByIndex.byteLength +
272 resolveResult.idxView.sortedOffsets.byteLength +
273 resolveResult.idxView.sortedOffsetIndices.byteLength;
274 
275 // V8 heap gives the best approximation of what the indexer actually uses.
276 // RSS includes V8 engine, JIT code, and Node.js internals that don't exist
277 // on Cloudflare Workers, so it overstates the indexer's cost.
278 // heapUsed is the closest proxy for the Workers runtime memory footprint.
279 const mem = process.memoryUsage();
280 const heapUsedMiB = mem.heapUsed / 1024 / 1024;
281 const rssMiB = mem.rss / 1024 / 1024;
282 
283 console.log("\n=== Results ===");
284 console.log(`Objects: ${resolveResult.objectCount}`);
285 console.log(`Scan time: ${scanMs}ms`);
286 console.log(`Resolve time: ${resolveMs}ms`);
287 console.log(`Total time: ${totalMs}ms`);
288 console.log(`Subrequests: ${counter.count} / ${SUBREQUEST_BUDGET} budget`);
289 console.log(`Idx size: ${resolveResult.idxBytes} bytes`);
290 console.log(
291 `Entry table: ${entryTableBytes} bytes (${(entryTableBytes / 1024 / 1024).toFixed(2)} MiB)`
292 );
293 console.log(
294 `Scan aux: ${scanAuxBytes} bytes (${(scanAuxBytes / 1024 / 1024).toFixed(2)} MiB)`
295 );
296 console.log(
297 `IdxView: ${idxViewBytes} bytes (${(idxViewBytes / 1024 / 1024).toFixed(2)} MiB)`
298 );
299 console.log(
300 `Typed arrays: ${((entryTableBytes + scanAuxBytes + idxViewBytes) / 1024 / 1024).toFixed(2)} MiB (indexer-controlled)`
301 );
302 console.log(
303 `Heap delta: ${(heapDelta / 1024 / 1024).toFixed(1)} MiB (indexer allocation, run with --expose-gc for accuracy)`
304 );
305 console.log(
306 `RSS delta: ${(rssDelta / 1024 / 1024).toFixed(1)} MiB (Node resident-set growth, includes runtime overhead)`
307 );
308 console.log(`Heap used: ${heapUsedMiB.toFixed(1)} MiB (best proxy for Workers memory)`);
309 console.log(
310 `RSS: ${rssMiB.toFixed(1)} MiB (includes V8/Node overhead, not meaningful for Workers)`
311 );
312 console.log(`Idx match: PASS`);
313 
314 if (counter.count >= SUBREQUEST_BUDGET) {
315 console.error(`FAIL: Subrequest count ${counter.count} exceeds budget ${SUBREQUEST_BUDGET}`);
316 process.exit(1);
317 }
318 console.log(`Budget: PASS (${counter.count} < ${SUBREQUEST_BUDGET})`);
319 console.log("\nAll validations passed.");
320}
321 
322main().catch((err) => {
323 console.error(err);
324 process.exit(1);
325});