File
Blob: test/scripts/validate-fixture-indexer.ts
| 1 | #!/usr/bin/env -S npx tsx |
| 2 | /** |
| 3 | * Validation script for the streaming pack indexer against the real fixture. |
| 4 | * |
| 5 | * Tests: |
| 6 | * - Generated idx matches Git's own output byte-for-byte |
| 7 | * - Subrequest count stays within the receive-path indexer budget (5,000) |
| 8 | * - Reports timing and subrequest stats |
| 9 | * |
| 10 | * Usage: |
| 11 | * npx tsx test/scripts/validate-fixture-indexer.ts |
| 12 | * |
| 13 | * This script creates a mock R2 bucket backed by local files so the indexer |
| 14 | * can be tested outside the Cloudflare Workers runtime. |
| 15 | */ |
| 16 | |
| 17 | import * as fs from "node:fs"; |
| 18 | import * as path from "node:path"; |
| 19 | import * as nodeCrypto from "node:crypto"; |
| 20 | |
| 21 | // Polyfill crypto.DigestStream for Node.js (Cloudflare Workers-specific API). |
| 22 | // This creates a WritableStream that computes a hash digest of all written data. |
| 23 | if (!(globalThis.crypto as unknown as Record<string, unknown>).DigestStream) { |
| 24 | class DigestStreamPolyfill extends WritableStream<Uint8Array> { |
| 25 | digest: Promise<ArrayBuffer>; |
| 26 | constructor(algorithm: string) { |
| 27 | const alg = algorithm.replace("-", "").toLowerCase(); // "SHA-1" -> "sha1" |
| 28 | const hash = nodeCrypto.createHash(alg); |
| 29 | let resolveDigest: (value: ArrayBuffer) => void; |
| 30 | const digestPromise = new Promise<ArrayBuffer>((resolve) => { |
| 31 | resolveDigest = resolve; |
| 32 | }); |
| 33 | super({ |
| 34 | write(chunk) { |
| 35 | hash.update(chunk); |
| 36 | }, |
| 37 | close() { |
| 38 | const buf = hash.digest(); |
| 39 | resolveDigest!(buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength)); |
| 40 | }, |
| 41 | }); |
| 42 | this.digest = digestPromise; |
| 43 | } |
| 44 | } |
| 45 | (globalThis.crypto as unknown as Record<string, unknown>).DigestStream = DigestStreamPolyfill; |
| 46 | } |
| 47 | |
| 48 | // Dynamic import to pick up the project's path aliases via tsx |
| 49 | const { scanPack, resolveDeltasAndWriteIdx } = await import("../../src/worker/git/pack/indexer"); |
| 50 | const { SubrequestLimiter } = await import("../../src/worker/git/operations/limits"); |
| 51 | const { createLogger } = await import("../../src/worker/common/logger"); |
| 52 | |
| 53 | const FIXTURE_DIR = path.resolve(import.meta.dirname ?? ".", "../../uncommitted-fixture"); |
| 54 | const PACK_NAME = "pack-395a180893e59dad8ef9d7fa135ecd8b1b399bb1"; |
| 55 | const PACK_PATH = path.join(FIXTURE_DIR, `${PACK_NAME}.pack`); |
| 56 | const IDX_PATH = path.join(FIXTURE_DIR, `${PACK_NAME}.idx`); |
| 57 | |
| 58 | const R2_PACK_KEY = `test/fixture/${PACK_NAME}.pack`; |
| 59 | // The platform hard cap is 10,000 subrequests per Worker invocation. The old |
| 60 | // 900 "soft budget" was set when reads went through DO RPCs + loose objects. |
| 61 | // For the streaming receive path (which reads directly from R2), the indexer |
| 62 | // gets a larger share. We budget 5,000 for the indexer so the remaining ~5,000 |
| 63 | // is available for connectivity checks, catalog loads, and other receive-path |
| 64 | // operations. |
| 65 | const SUBREQUEST_BUDGET = 5_000; |
| 66 | |
| 67 | // --------------------------------------------------------------------------- |
| 68 | // Mock R2 bucket backed by local file system |
| 69 | // --------------------------------------------------------------------------- |
| 70 | |
| 71 | class MockR2Object { |
| 72 | key: string; |
| 73 | private data: Uint8Array; |
| 74 | size: number; |
| 75 | constructor(key: string, data: Uint8Array, size: number) { |
| 76 | this.key = key; |
| 77 | this.data = data; |
| 78 | this.size = size; |
| 79 | } |
| 80 | |
| 81 | async arrayBuffer(): Promise<ArrayBuffer> { |
| 82 | return this.data.buffer.slice( |
| 83 | this.data.byteOffset, |
| 84 | this.data.byteOffset + this.data.byteLength |
| 85 | ) as ArrayBuffer; |
| 86 | } |
| 87 | } |
| 88 | |
| 89 | function createMockBucket(files: Map<string, Uint8Array>) { |
| 90 | return { |
| 91 | async get( |
| 92 | key: string, |
| 93 | opts?: { range?: { offset: number; length: number } } |
| 94 | ): Promise<MockR2Object | null> { |
| 95 | const data = files.get(key); |
| 96 | if (!data) return null; |
| 97 | if (opts?.range) { |
| 98 | const { offset, length } = opts.range; |
| 99 | const slice = data.subarray(offset, offset + length); |
| 100 | return new MockR2Object(key, new Uint8Array(slice), slice.length); |
| 101 | } |
| 102 | return new MockR2Object(key, data, data.length); |
| 103 | }, |
| 104 | async head(key: string): Promise<{ size: number } | null> { |
| 105 | const data = files.get(key); |
| 106 | return data ? { size: data.length } : null; |
| 107 | }, |
| 108 | async put(key: string, data: Uint8Array | ArrayBuffer): Promise<void> { |
| 109 | const bytes = data instanceof Uint8Array ? data : new Uint8Array(data); |
| 110 | files.set(key, bytes); |
| 111 | }, |
| 112 | async delete(_key: string): Promise<void> {}, |
| 113 | }; |
| 114 | } |
| 115 | |
| 116 | type MemorySnapshot = { |
| 117 | heapUsed: number; |
| 118 | rss: number; |
| 119 | }; |
| 120 | |
| 121 | function takeMemorySnapshot(): MemorySnapshot { |
| 122 | const usage = process.memoryUsage(); |
| 123 | return { |
| 124 | heapUsed: usage.heapUsed, |
| 125 | rss: usage.rss, |
| 126 | }; |
| 127 | } |
| 128 | |
| 129 | // --------------------------------------------------------------------------- |
| 130 | // Main |
| 131 | // --------------------------------------------------------------------------- |
| 132 | |
| 133 | async function main() { |
| 134 | if (!fs.existsSync(PACK_PATH) || !fs.existsSync(IDX_PATH)) { |
| 135 | console.error(`Fixture files not found in ${FIXTURE_DIR}`); |
| 136 | process.exit(1); |
| 137 | } |
| 138 | |
| 139 | const packData = new Uint8Array(fs.readFileSync(PACK_PATH)); |
| 140 | const expectedIdx = new Uint8Array(fs.readFileSync(IDX_PATH)); |
| 141 | |
| 142 | console.log( |
| 143 | `Pack: ${packData.byteLength} bytes (${(packData.byteLength / 1024 / 1024).toFixed(1)} MiB)` |
| 144 | ); |
| 145 | console.log( |
| 146 | `Idx: ${expectedIdx.byteLength} bytes (${(expectedIdx.byteLength / 1024 / 1024).toFixed(1)} MiB)` |
| 147 | ); |
| 148 | |
| 149 | // Create mock R2 bucket with the fixture pack. |
| 150 | const files = new Map<string, Uint8Array>(); |
| 151 | files.set(R2_PACK_KEY, packData); |
| 152 | |
| 153 | const env = { REPO_BUCKET: createMockBucket(files) } as unknown as Env; |
| 154 | const log = createLogger("info", { service: "FixtureValidator" }); |
| 155 | const limiter = new SubrequestLimiter(6); |
| 156 | const counter = { count: 0 }; |
| 157 | const countSub = (n = 1) => { |
| 158 | counter.count += n; |
| 159 | }; |
| 160 | |
| 161 | // ---- Memory baseline (after loading fixture, before indexer) ---- |
| 162 | // The fixture pack and idx are intentionally loaded before this snapshot so |
| 163 | // the delta isolates the indexer itself rather than fixture setup cost. |
| 164 | global.gc?.(); // --expose-gc makes this available |
| 165 | const before = takeMemorySnapshot(); |
| 166 | |
| 167 | // ---- Scan ---- |
| 168 | console.log("\nScanning pack..."); |
| 169 | const scanStart = Date.now(); |
| 170 | const scanResult = await scanPack({ |
| 171 | env, |
| 172 | packKey: R2_PACK_KEY, |
| 173 | packSize: packData.byteLength, |
| 174 | limiter, |
| 175 | countSubrequest: countSub, |
| 176 | log, |
| 177 | }); |
| 178 | const scanMs = Date.now() - scanStart; |
| 179 | console.log(` Objects: ${scanResult.objectCount}`); |
| 180 | console.log(` Scan time: ${scanMs}ms`); |
| 181 | console.log(` Subrequests so far: ${counter.count}`); |
| 182 | |
| 183 | // ---- Resolve + write idx ---- |
| 184 | console.log("\nResolving deltas and writing idx..."); |
| 185 | const resolveStart = Date.now(); |
| 186 | const resolveResult = await resolveDeltasAndWriteIdx({ |
| 187 | env, |
| 188 | packKey: R2_PACK_KEY, |
| 189 | packSize: packData.byteLength, |
| 190 | limiter, |
| 191 | countSubrequest: countSub, |
| 192 | log, |
| 193 | scanResult, |
| 194 | repoId: "fixture/test", |
| 195 | lruBudget: 48 * 1024 * 1024, // 48 MiB — safe with the array-backed payload cache |
| 196 | }); |
| 197 | const resolveMs = Date.now() - resolveStart; |
| 198 | const totalMs = scanMs + resolveMs; |
| 199 | |
| 200 | // ---- Memory deltas: measure what remained resident after indexer work ---- |
| 201 | // heapUsed best approximates the memory pressure we care about on Workers. |
| 202 | // RSS is still useful to inspect Node process growth, but it also includes |
| 203 | // V8 and runtime pages that do not map cleanly to Workers. |
| 204 | global.gc?.(); |
| 205 | const after = takeMemorySnapshot(); |
| 206 | const heapDelta = after.heapUsed - before.heapUsed; |
| 207 | const rssDelta = after.rss - before.rss; |
| 208 | |
| 209 | // ---- Read generated idx ---- |
| 210 | const idxKey = R2_PACK_KEY.replace(/\.pack$/, ".idx"); |
| 211 | const generatedIdx = files.get(idxKey); |
| 212 | if (!generatedIdx) { |
| 213 | console.error("ERROR: No generated idx found in mock R2"); |
| 214 | process.exit(1); |
| 215 | } |
| 216 | |
| 217 | // ---- Compare byte-for-byte ---- |
| 218 | console.log("\nComparing idx..."); |
| 219 | if (generatedIdx.byteLength !== expectedIdx.byteLength) { |
| 220 | console.error( |
| 221 | `ERROR: Size mismatch: generated ${generatedIdx.byteLength}, expected ${expectedIdx.byteLength}` |
| 222 | ); |
| 223 | process.exit(1); |
| 224 | } |
| 225 | |
| 226 | let firstMismatch = -1; |
| 227 | for (let i = 0; i < expectedIdx.byteLength; i++) { |
| 228 | if (generatedIdx[i] !== expectedIdx[i]) { |
| 229 | firstMismatch = i; |
| 230 | break; |
| 231 | } |
| 232 | } |
| 233 | |
| 234 | if (firstMismatch !== -1) { |
| 235 | console.error(`ERROR: Idx mismatch at byte offset ${firstMismatch}`); |
| 236 | const ctx = 16; |
| 237 | const start = Math.max(0, firstMismatch - ctx); |
| 238 | const end = Math.min(expectedIdx.byteLength, firstMismatch + ctx); |
| 239 | console.error( |
| 240 | ` Expected: ${Array.from(expectedIdx.subarray(start, end)) |
| 241 | .map((b) => b.toString(16).padStart(2, "0")) |
| 242 | .join(" ")}` |
| 243 | ); |
| 244 | console.error( |
| 245 | ` Got: ${Array.from(generatedIdx.subarray(start, end)) |
| 246 | .map((b) => b.toString(16).padStart(2, "0")) |
| 247 | .join(" ")}` |
| 248 | ); |
| 249 | process.exit(1); |
| 250 | } |
| 251 | |
| 252 | // ---- Report ---- |
| 253 | // Compute the indexer's typed-array footprint (the part we control). |
| 254 | const entryTableBytes = |
| 255 | scanResult.table.offsets.byteLength + |
| 256 | scanResult.table.types.byteLength + |
| 257 | scanResult.table.headerLens.byteLength + |
| 258 | scanResult.table.spanEnds.byteLength + |
| 259 | scanResult.table.crc32s.byteLength + |
| 260 | scanResult.table.oids.byteLength + |
| 261 | scanResult.table.decompressedSizes.byteLength + |
| 262 | scanResult.table.ofsBaseOffsets.byteLength + |
| 263 | scanResult.table.resolved.byteLength; |
| 264 | |
| 265 | const scanAuxBytes = scanResult.refBaseOids.byteLength; |
| 266 | |
| 267 | const idxViewBytes = |
| 268 | resolveResult.idxView.fanout.byteLength + |
| 269 | resolveResult.idxView.rawNames.byteLength + |
| 270 | resolveResult.idxView.offsets.byteLength + |
| 271 | resolveResult.idxView.nextOffsetByIndex.byteLength + |
| 272 | resolveResult.idxView.sortedOffsets.byteLength + |
| 273 | resolveResult.idxView.sortedOffsetIndices.byteLength; |
| 274 | |
| 275 | // V8 heap gives the best approximation of what the indexer actually uses. |
| 276 | // RSS includes V8 engine, JIT code, and Node.js internals that don't exist |
| 277 | // on Cloudflare Workers, so it overstates the indexer's cost. |
| 278 | // heapUsed is the closest proxy for the Workers runtime memory footprint. |
| 279 | const mem = process.memoryUsage(); |
| 280 | const heapUsedMiB = mem.heapUsed / 1024 / 1024; |
| 281 | const rssMiB = mem.rss / 1024 / 1024; |
| 282 | |
| 283 | console.log("\n=== Results ==="); |
| 284 | console.log(`Objects: ${resolveResult.objectCount}`); |
| 285 | console.log(`Scan time: ${scanMs}ms`); |
| 286 | console.log(`Resolve time: ${resolveMs}ms`); |
| 287 | console.log(`Total time: ${totalMs}ms`); |
| 288 | console.log(`Subrequests: ${counter.count} / ${SUBREQUEST_BUDGET} budget`); |
| 289 | console.log(`Idx size: ${resolveResult.idxBytes} bytes`); |
| 290 | console.log( |
| 291 | `Entry table: ${entryTableBytes} bytes (${(entryTableBytes / 1024 / 1024).toFixed(2)} MiB)` |
| 292 | ); |
| 293 | console.log( |
| 294 | `Scan aux: ${scanAuxBytes} bytes (${(scanAuxBytes / 1024 / 1024).toFixed(2)} MiB)` |
| 295 | ); |
| 296 | console.log( |
| 297 | `IdxView: ${idxViewBytes} bytes (${(idxViewBytes / 1024 / 1024).toFixed(2)} MiB)` |
| 298 | ); |
| 299 | console.log( |
| 300 | `Typed arrays: ${((entryTableBytes + scanAuxBytes + idxViewBytes) / 1024 / 1024).toFixed(2)} MiB (indexer-controlled)` |
| 301 | ); |
| 302 | console.log( |
| 303 | `Heap delta: ${(heapDelta / 1024 / 1024).toFixed(1)} MiB (indexer allocation, run with --expose-gc for accuracy)` |
| 304 | ); |
| 305 | console.log( |
| 306 | `RSS delta: ${(rssDelta / 1024 / 1024).toFixed(1)} MiB (Node resident-set growth, includes runtime overhead)` |
| 307 | ); |
| 308 | console.log(`Heap used: ${heapUsedMiB.toFixed(1)} MiB (best proxy for Workers memory)`); |
| 309 | console.log( |
| 310 | `RSS: ${rssMiB.toFixed(1)} MiB (includes V8/Node overhead, not meaningful for Workers)` |
| 311 | ); |
| 312 | console.log(`Idx match: PASS`); |
| 313 | |
| 314 | if (counter.count >= SUBREQUEST_BUDGET) { |
| 315 | console.error(`FAIL: Subrequest count ${counter.count} exceeds budget ${SUBREQUEST_BUDGET}`); |
| 316 | process.exit(1); |
| 317 | } |
| 318 | console.log(`Budget: PASS (${counter.count} < ${SUBREQUEST_BUDGET})`); |
| 319 | console.log("\nAll validations passed."); |
| 320 | } |
| 321 | |
| 322 | main().catch((err) => { |
| 323 | console.error(err); |
| 324 | process.exit(1); |
| 325 | }); |