Skip to content
File

Blob: src/workerd/api/pyodide/pyodide.c++

24.3 KB
1// Copyright (c) 2017-2022 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4#include "pyodide.h"
5 
6#include "requirements.h"
7 
8#include <workerd/io/compatibility-date.h>
9#include <workerd/io/features.h>
10#include <workerd/io/io-context.h>
11#include <workerd/util/autogate.h>
12#include <workerd/util/strings.h>
13 
14#include <pyodide/generated/pyodide_extra.capnp.h>
15 
16#include <capnp/dynamic.h>
17#include <capnp/schema.h>
18#include <kj/array.h>
19#include <kj/common.h>
20#include <kj/compat/gzip.h>
21#include <kj/compat/tls.h>
22#include <kj/debug.h>
23#include <kj/string.h>
24 
25#include <algorithm> // for std::sort
26 
27namespace workerd::api::pyodide {
28 
29// singleton that owns bundle
30 
31const kj::Maybe<jsg::Bundle::Reader> PyodideBundleManager::getPyodideBundle(
32 kj::StringPtr version) const {
33 return bundles.lockShared()->find(version).map(
34 [](const MessageBundlePair& t) { return t.bundle; });
35}
36 
37void PyodideBundleManager::setPyodideBundleData(
38 kj::String version, kj::Array<unsigned char> data) const {
39 auto wordArray = kj::arrayPtr(
40 reinterpret_cast<const capnp::word*>(data.begin()), data.size() / sizeof(capnp::word));
41 // We're going to reuse this in the ModuleRegistry for every Python isolate, so set the traversal
42 // limit to infinity or else eventually a new Python isolate will fail.
43 auto messageReader = kj::heap<capnp::FlatArrayMessageReader>(
44 wordArray, capnp::ReaderOptions{.traversalLimitInWords = kj::maxValue})
45 .attach(kj::mv(data));
46 auto bundle = messageReader->getRoot<jsg::Bundle>();
47 bundles.lockExclusive()->insert(
48 kj::mv(version), {.messageReader = kj::mv(messageReader), .bundle = bundle});
49}
50 
51const kj::Maybe<const kj::Array<unsigned char>&> PyodidePackageManager::getPyodidePackage(
52 kj::StringPtr id) const {
53 return packages.lockShared()->find(id);
54}
55 
56void PyodidePackageManager::setPyodidePackageData(
57 kj::String id, kj::Array<unsigned char> data) const {
58 packages.lockExclusive()->insert(kj::mv(id), kj::mv(data));
59}
60 
61static int readToTarget(
62 kj::ArrayPtr<const kj::byte> source, int offset, kj::ArrayPtr<kj::byte> buf) {
63 int size = source.size();
64 if (offset >= size || offset < 0) {
65 return 0;
66 }
67 int toCopy = buf.size();
68 if (size - offset < toCopy) {
69 toCopy = size - offset;
70 }
71 memcpy(buf.begin(), source.begin() + offset, toCopy);
72 return toCopy;
73}
74 
75int ReadOnlyBuffer::read(jsg::Lock& js, int offset, kj::Array<kj::byte> buf) {
76 return readToTarget(source, offset, buf);
77}
78 
79kj::Array<kj::StringPtr> PyodideMetadataReader::getNames(
80 jsg::Lock& js, jsg::Optional<kj::String> maybeExtFilter) {
81 auto builder = kj::Vector<kj::StringPtr>(state->moduleInfo.names.size());
82 for (auto i: kj::zeroTo(builder.capacity())) {
83 KJ_IF_SOME(ext, maybeExtFilter) {
84 if (!state->moduleInfo.names[i].endsWith(ext)) {
85 continue;
86 }
87 }
88 builder.add(state->moduleInfo.names[i]);
89 }
90 return builder.releaseAsArray();
91}
92 
93void PyodideMetadataReader::setCpuLimitNearlyExceededCallback(
94 jsg::Lock& js, kj::Array<kj::byte> wasm_memory, int sig_clock, int sig_flag) {
95 // This callback has to be implemented in C++ because we don't hold the isolate lock when we call
96 // it. It also has to be signal safe since we call it from the cpu time limiter.
97 Worker::Isolate::from(js).setCpuLimitNearlyExceededCallback(
98 [wasm_memory = kj::mv(wasm_memory), sig_clock, sig_flag]() mutable {
99 // Set signal handling clock to fire on the next check.
100 wasm_memory[sig_clock] = 0;
101 // Set signal handling to on
102 wasm_memory[sig_flag] = 1;
103 });
104}
105 
106kj::Array<kj::String> PythonModuleInfo::getPythonFileContents() {
107 auto builder = kj::Vector<kj::String>(names.size());
108 for (auto i: kj::zeroTo(names.size())) {
109 if (names[i].endsWith(".py")) {
110 builder.add(kj::str(contents[i].asChars()));
111 }
112 }
113 return builder.releaseAsArray();
114}
115 
116kj::HashSet<kj::String> PythonModuleInfo::getWorkerModuleSet() {
117 auto result = kj::HashSet<kj::String>();
118 const auto vendor = "python_modules/"_kj;
119 const auto dotPy = ".py"_kj;
120 const auto dotSo = ".so"_kj;
121 for (auto& item: names) {
122 kj::StringPtr name = item;
123 if (name.startsWith(vendor)) {
124 name = name.slice(vendor.size());
125 }
126 auto firstSlash = name.findFirst('/');
127 KJ_IF_SOME(idx, firstSlash) {
128 result.upsert(kj::str(name.slice(0, idx)), [](auto&&, auto&&) {});
129 continue;
130 }
131 if (name.endsWith(dotPy)) {
132 result.upsert(kj::str(name.slice(0, name.size() - dotPy.size())), [](auto&&, auto&&) {});
133 continue;
134 }
135 if (name.endsWith(dotSo)) {
136 result.upsert(kj::str(name.slice(0, name.size() - dotSo.size())), [](auto&&, auto&&) {});
137 continue;
138 }
139 }
140 return result;
141}
142 
143kj::Array<kj::String> PythonModuleInfo::getPackageSnapshotImports(kj::StringPtr version) {
144 auto workerFiles = this->getPythonFileContents();
145 auto importedNames = parsePythonScriptImports(kj::mv(workerFiles));
146 auto workerModules = getWorkerModuleSet();
147 return PythonModuleInfo::filterPythonScriptImports(
148 kj::mv(workerModules), kj::mv(importedNames), version);
149}
150 
151kj::Array<kj::String> PyodideMetadataReader::getPackageSnapshotImports(kj::String version) {
152 return state->moduleInfo.getPackageSnapshotImports(version);
153}
154 
155kj::Array<jsg::JsRef<jsg::JsString>> PyodideMetadataReader::getRequirements(jsg::Lock& js) {
156 auto builder = kj::heapArrayBuilder<jsg::JsRef<jsg::JsString>>(state->requirements.size());
157 for (auto i: kj::zeroTo(builder.capacity())) {
158 builder.add(js, js.str(state->requirements[i]));
159 }
160 return builder.finish();
161}
162 
163kj::Array<int> PyodideMetadataReader::getSizes(jsg::Lock& js) {
164 auto builder = kj::heapArrayBuilder<int>(state->moduleInfo.names.size());
165 for (auto i: kj::zeroTo(builder.capacity())) {
166 builder.add(state->moduleInfo.contents[i].size());
167 }
168 return builder.finish();
169}
170 
171int PyodideMetadataReader::read(jsg::Lock& js, int index, int offset, kj::Array<kj::byte> buf) {
172 if (index >= state->moduleInfo.contents.size() || index < 0) {
173 return 0;
174 }
175 auto& data = state->moduleInfo.contents[index];
176 return readToTarget(data, offset, buf);
177}
178 
179int PyodideMetadataReader::readMemorySnapshot(int offset, kj::Array<kj::byte> buf) {
180 if (state->memorySnapshot == kj::none) {
181 return 0;
182 }
183 return readToTarget(KJ_REQUIRE_NONNULL(state->memorySnapshot), offset, buf);
184}
185 
186kj::HashSet<kj::String> PyodideMetadataReader::getTransitiveRequirements() {
187 auto packages = parseLockFile(state->packagesLock);
188 auto depMap = getDepMapFromPackagesLock(*packages);
189 
190 return getPythonPackageNames(*packages, depMap, state->requirements, state->packagesVersion);
191}
192 
193int ArtifactBundler::readMemorySnapshot(int offset, kj::Array<kj::byte> buf) {
194 if (inner->existingSnapshot == kj::none) {
195 return 0;
196 }
197 return readToTarget(KJ_REQUIRE_NONNULL(inner->existingSnapshot), offset, buf);
198}
199 
200kj::Array<kj::String> PythonModuleInfo::parsePythonScriptImports(kj::Array<kj::String> files) {
201 auto result = kj::Vector<kj::String>();
202 
203 for (auto& file: files) {
204 // Returns the number of characters skipped. When `oneOf` is not found, skips to the end of
205 // the string.
206 auto skipUntil = [](kj::StringPtr str, std::initializer_list<char> oneOf, int start) -> int {
207 int result = 0;
208 while (start + result < str.size()) {
209 char c = str[start + result];
210 for (char expected: oneOf) {
211 if (c == expected) {
212 return result;
213 }
214 }
215 
216 result++;
217 }
218 
219 return result;
220 };
221 
222 // Skips while current character is in `oneOf`. Returns the number of characters skipped.
223 auto skipWhile = [](kj::StringPtr str, std::initializer_list<char> oneOf, int start) -> int {
224 int result = 0;
225 while (start + result < str.size()) {
226 char c = str[start + result];
227 bool found = false;
228 for (char expected: oneOf) {
229 if (c == expected) {
230 result++;
231 found = true;
232 break;
233 }
234 }
235 
236 if (!found) {
237 break;
238 }
239 }
240 
241 return result;
242 };
243 
244 // Skips one of the characters (specified in `oneOf`) at the current position. Otherwise
245 // throws. Returns the number of characters skipped.
246 auto skipChar = [](kj::StringPtr str, std::initializer_list<char> oneOf, int start) -> int {
247 for (char expected: oneOf) {
248 if (str[start] == expected) {
249 return 1;
250 }
251 }
252 
253 KJ_FAIL_REQUIRE("Expected ", oneOf, "but received", str[start]);
254 };
255 
256 auto parseKeyword = [](kj::StringPtr str, kj::StringPtr ident, int start) -> bool {
257 int i = 0;
258 for (; i < ident.size() && start + i < str.size(); i++) {
259 if (str[start + i] != ident[i]) {
260 return false;
261 }
262 }
263 
264 return i == ident.size();
265 };
266 
267 // Returns the size of the import identifier or 0 if no identifier exists at `start`.
268 auto parseIdent = [](kj::StringPtr str, int start) -> int {
269 // https://docs.python.org/3/reference/lexical_analysis.html#identifiers
270 //
271 // We also accept `.` because import idents can contain it.
272 // TODO: We don't currently support unicode, but if we see packages that utilize it we will
273 // implement that support.
274 if (isDigit(str[start])) {
275 return 0;
276 }
277 int i = 0;
278 for (; start + i < str.size(); i++) {
279 char c = str[start + i];
280 bool validIdentChar = isAlpha(c) || isDigit(c) || c == '_' || c == '.';
281 if (!validIdentChar) {
282 return i;
283 }
284 }
285 
286 return i;
287 };
288 
289 int i = 0;
290 while (i < file.size()) {
291 switch (file[i]) {
292 case 'i':
293 case 'f': {
294 auto keywordToParse = file[i] == 'i' ? "import"_kj : "from"_kj;
295 if (!parseKeyword(file, keywordToParse, i)) {
296 // We cannot simply skip the current char here, doing so would mean that
297 // `iimport x` would be parsed as a valid import.
298 i += skipUntil(file, {'\n', '\r', '"', '\''}, i);
299 continue;
300 }
301 i += keywordToParse.size(); // skip "import" or "from"
302 
303 while (i < file.size()) {
304 // Python expects a `\` to be paired with a newline, but we don't have to be as strict
305 // here because we rely on the fact that the script has gone through validation already.
306 i += skipWhile(
307 file, {'\r', '\n', ' ', '\t', '\\'}, i); // skip whitespace and backslash.
308 
309 if (file[i] == '.') {
310 // ignore relative imports
311 break;
312 }
313 
314 int identLen = parseIdent(file, i);
315 KJ_REQUIRE(identLen > 0);
316 
317 kj::String ident = kj::heapString(file.slice(i, i + identLen));
318 if (ident[identLen - 1] != '.') { // trailing period means the import is invalid
319 result.add(kj::mv(ident));
320 }
321 
322 i += identLen;
323 
324 // If "import" statement then look for comma.
325 if (keywordToParse == "import") {
326 i += skipWhile(
327 file, {'\r', '\n', ' ', '\t', '\\'}, i); // skip whitespace and backslash.
328 // Check if next char is a comma.
329 if (file[i] == ',') {
330 i += 1; // Skip comma.
331 // Allow while loop to continue
332 } else {
333 // No more idents, so break out of loop.
334 break;
335 }
336 } else {
337 // The "from" statement doesn't support commas.
338 break;
339 }
340 }
341 break;
342 }
343 case '"':
344 case '\'': {
345 char quote = file[i];
346 // Detect multi-line string literals `"""` and skip until the corresponding ending `"""`.
347 if (i + 2 < file.size() && file[i + 1] == quote && file[i + 2] == quote) {
348 i += 3; // skip start quotes.
349 // skip until terminating quotes.
350 while (i + 2 < file.size() && file[i + 1] != quote && file[i + 2] != quote) {
351 if (file[i] == quote) {
352 i++;
353 }
354 i += skipUntil(file, {quote}, i);
355 }
356 i += 3; // skip terminating quotes.
357 } else if (i + 2 < file.size() && file[i + 1] == '\\' &&
358 (file[i + 2] == '\n' || file[i + 2] == '\r')) {
359 // Detect string literal with backslash.
360 i += 3; // skip `"\<NL>`
361 // skip until quote, but ignore `\"`.
362 while (file[i] != quote && file[i - 1] != '\\') {
363 if (file[i] == quote) {
364 i++;
365 }
366 i += skipUntil(file, {quote}, i);
367 }
368 i += 1; // skip quote.
369 } else {
370 i += 1; // skip quote.
371 }
372 
373 // skip until EOL so that we don't mistakenly parse and capture `"import x`.
374 i += skipUntil(file, {'\n', '\r', '"', '\''}, i);
375 break;
376 }
377 default:
378 // Skip to the next line or " or '
379 i += skipUntil(file, {'\n', '\r', '"', '\''}, i);
380 if (file[i] == '"' || file[i] == '\'') {
381 continue; // Allow the quotes to be handled above.
382 }
383 if (file[i] != '\0') {
384 i += skipChar(file, {'\n', '\r'}, i); // skip newline.
385 }
386 }
387 }
388 }
389 
390 return result.releaseAsArray();
391}
392 
393const kj::Array<kj::StringPtr> snapshotImports = kj::arr("_pyodide"_kj,
394 "_pyodide.docstring"_kj,
395 "_pyodide._core_docs"_kj,
396 "traceback"_kj,
397 "collections.abc"_kj,
398 // Asyncio is the really slow one here. In native Python on my machine, `import asyncio` takes ~50
399 // ms.
400 "asyncio"_kj,
401 "inspect"_kj,
402 "tarfile"_kj,
403 "importlib",
404 "importlib.metadata"_kj,
405 "re"_kj,
406 "shutil"_kj,
407 "sysconfig"_kj,
408 "importlib.machinery"_kj,
409 "pathlib"_kj,
410 "site"_kj,
411 "tempfile"_kj,
412 "typing"_kj,
413 "zipfile"_kj);
414 
415kj::Array<kj::StringPtr> PyodideMetadataReader::getBaselineSnapshotImports() {
416 return kj::heapArray(snapshotImports.begin(), snapshotImports.size());
417}
418 
419bool PyodideMetadataReader::shouldAbortIsolateOnFatalError() {
420 return util::Autogate::isEnabled(util::AutogateKey::PYTHON_ABORT_ISOLATE_ON_FATAL_ERROR);
421}
422 
423jsg::JsObject PyodideMetadataReader::getCompatibilityFlags(jsg::Lock& js) {
424 auto flags = FeatureFlags::get(js);
425 auto obj = js.objNoProto();
426 auto dynamic = capnp::toDynamic(flags);
427 auto schema = dynamic.getSchema();
428 
429 for (auto field: schema.getFields()) {
430 auto annotations = field.getProto().getAnnotations();
431 
432 // Note that disable flags are not exposed.
433 for (auto annotation: annotations) {
434 if (annotation.getId() == COMPAT_ENABLE_FLAG_ANNOTATION_ID) {
435 obj.setReadOnly(
436 js, annotation.getValue().getText(), js.boolean(dynamic.get(field).as<bool>()));
437 }
438 }
439 }
440 
441 obj.seal(js);
442 return obj;
443}
444 
445PyodideMetadataReader::State::State(const State& other)
446 : mainModule(kj::str(other.mainModule)),
447 moduleInfo(other.moduleInfo.clone()),
448 requirements(KJ_MAP(req, other.requirements) { return kj::str(req); }),
449 pyodideVersion(kj::str(other.pyodideVersion)),
450 packagesVersion(kj::str(other.packagesVersion)),
451 packagesLock(kj::str(other.packagesLock)),
452 isWorkerdFlag(other.isWorkerdFlag),
453 isTracingFlag(other.isTracingFlag),
454 snapshotToDisk(other.snapshotToDisk),
455 createBaselineSnapshot(other.createBaselineSnapshot),
456 memorySnapshot(other.memorySnapshot.map(
457 [](auto& snapshot) { return kj::heapArray<kj::byte>(snapshot); })) {}
458 
459kj::Own<PyodideMetadataReader::State> PyodideMetadataReader::State::clone() {
460 return kj::heap<PyodideMetadataReader::State>(*this);
461}
462 
463void PyodideMetadataReader::State::verifyNoMainModuleInVendor() {
464 // Verify that we don't have module named after the main module in the `python_modules` subdir.
465 // mainModule includes the .py extension, so we need to extract the base name
466 kj::ArrayPtr<const char> mainModuleBase = mainModule;
467 if (mainModule.endsWith(".py")) {
468 mainModuleBase = mainModuleBase.slice(0, mainModuleBase.size() - 3);
469 }
470 
471 for (auto& name: moduleInfo.names) {
472 if (name.startsWith(kj::str("python_modules/", mainModule))) {
473 JSG_FAIL_REQUIRE(
474 Error, kj::str("Python module python_modules/", mainModule, " clashes with main module"));
475 }
476 if (name == kj::str("python_modules/", mainModuleBase, "/__init__.py")) {
477 JSG_FAIL_REQUIRE(Error,
478 kj::str("Python module python_modules/", mainModuleBase,
479 "/__init__.py clashes with main module"));
480 }
481 if (name == kj::str("python_modules/", mainModuleBase, ".so")) {
482 JSG_FAIL_REQUIRE(Error,
483 kj::str("Python module python_modules/", mainModuleBase, ".so clashes with main module"));
484 }
485 }
486}
487 
488kj::Array<kj::String> PythonModuleInfo::filterPythonScriptImports(
489 kj::HashSet<kj::String> workerModules,
490 kj::ArrayPtr<kj::String> imports,
491 kj::StringPtr version) {
492 auto baselineSnapshotImportsSet = kj::HashSet<kj::StringPtr>();
493 for (auto& pkgImport: snapshotImports) {
494 baselineSnapshotImportsSet.upsert(kj::mv(pkgImport), [](auto&&, auto&&) {});
495 }
496 
497 kj::HashSet<kj::String> filteredImportsSet;
498 filteredImportsSet.reserve(imports.size());
499 for (auto& pkgImport: imports) {
500 auto firstDot = pkgImport.findFirst('.').orDefault(pkgImport.size());
501 auto firstComponent = pkgImport.slice(0, firstDot);
502 // Skip duplicates
503 if (filteredImportsSet.contains(pkgImport)) [[unlikely]] {
504 continue;
505 }
506 
507 // don't include modules that we provide and that are likely to be imported by most
508 // workers.
509 if (firstComponent == "js"_kj.asArray() || firstComponent == "asgi"_kj.asArray() ||
510 firstComponent == "workers"_kj.asArray()) {
511 continue;
512 }
513 if (version == "0.26.0a2") {
514 if (firstComponent == "pyodide"_kj.asArray() || firstComponent == "httpx"_kj.asArray() ||
515 firstComponent == "openai"_kj.asArray() || firstComponent == "starlette"_kj.asArray() ||
516 firstComponent == "urllib3"_kj.asArray()) {
517 continue;
518 }
519 }
520 
521 // Don't include anything that went into the baseline snapshot
522 if (baselineSnapshotImportsSet.contains(pkgImport)) {
523 continue;
524 }
525 
526 // Don't include imports from worker files
527 if (workerModules.contains(firstComponent)) {
528 continue;
529 }
530 filteredImportsSet.upsert(kj::mv(pkgImport), [](auto&&, auto&&) {});
531 }
532 
533 auto filteredImportsBuilder = kj::heapArrayBuilder<kj::String>(filteredImportsSet.size());
534 for (auto& pkgImport: filteredImportsSet) {
535 filteredImportsBuilder.add(kj::mv(pkgImport));
536 }
537 return filteredImportsBuilder.finish();
538}
539 
540kj::Maybe<kj::String> getPyodideLock(PythonSnapshotRelease::Reader pythonSnapshotRelease) {
541 for (auto pkgLock: *PACKAGE_LOCKS) {
542 if (pkgLock.getPackageDate() == pythonSnapshotRelease.getPackages()) {
543 return kj::str(pkgLock.getLock());
544 }
545 }
546 
547 return kj::none;
548}
549 
550const kj::Maybe<kj::Own<const kj::Directory>> DiskCache::NULL_CACHE_ROOT = kj::none;
551 
552jsg::Optional<kj::Array<kj::byte>> DiskCache::get(jsg::Lock& js, kj::String key) {
553 KJ_IF_SOME(root, cacheRoot) {
554 kj::Path path(key);
555 auto file = root->tryOpenFile(path);
556 
557 KJ_IF_SOME(f, file) {
558 return f->readAllBytes();
559 } else {
560 return kj::none;
561 }
562 } else {
563 return kj::none;
564 }
565}
566 
567void DiskCache::put(jsg::Lock& js, kj::String key, kj::Array<kj::byte> data) {
568 KJ_IF_SOME(root, cacheRoot) {
569 kj::Path path(key);
570 auto file = root->tryOpenFile(path, kj::WriteMode::CREATE | kj::WriteMode::MODIFY);
571 
572 KJ_IF_SOME(f, file) {
573 f->writeAll(data);
574 } else {
575 KJ_LOG(ERROR, "DiskCache: Failed to open file", key);
576 }
577 } else {
578 return;
579 }
580}
581 
582void DiskCache::putSnapshot(jsg::Lock& js, kj::String key, kj::Array<kj::byte> data) {
583 KJ_IF_SOME(root, snapshotRoot) {
584 kj::Path path(key);
585 auto file = root->tryOpenFile(path, kj::WriteMode::CREATE | kj::WriteMode::MODIFY);
586 
587 KJ_IF_SOME(f, file) {
588 f->writeAll(data);
589 } else {
590 KJ_LOG(ERROR, "DiskCache: Failed to open file", key);
591 }
592 } else {
593 return;
594 }
595}
596 
597} // namespace workerd::api::pyodide
598 
599namespace workerd {
600 
601struct PythonSnapshotParsedField {
602 PythonSnapshotRelease::Reader pythonSnapshotRelease;
603 capnp::StructSchema::Field field;
604};
605 
606kj::Array<const PythonSnapshotParsedField> makePythonSnapshotFieldTable(
607 capnp::StructSchema::FieldList fields) {
608 kj::Vector<PythonSnapshotParsedField> table(fields.size());
609 
610 for (auto field: fields) {
611 bool isPythonField = false;
612 
613 for (auto annotation: field.getProto().getAnnotations()) {
614 if (annotation.getId() == PYTHON_SNAPSHOT_RELEASE_ANNOTATION_ID) {
615 isPythonField = true;
616 break;
617 }
618 }
619 if (!isPythonField) {
620 continue;
621 }
622 
623 auto name = field.getProto().getName();
624 kj::Maybe<PythonSnapshotRelease::Reader> pythonSnapshotRelease;
625 for (auto release: *RELEASES) {
626 if (release.getFlagName() == name) {
627 pythonSnapshotRelease = release;
628 break;
629 }
630 }
631 table.add(PythonSnapshotParsedField{
632 .pythonSnapshotRelease = KJ_REQUIRE_NONNULL(pythonSnapshotRelease),
633 .field = field,
634 });
635 }
636 
637 return table.releaseAsArray();
638}
639 
640kj::Maybe<PythonSnapshotRelease::Reader> getPythonSnapshotRelease(
641 CompatibilityFlags::Reader featureFlags) {
642 uint latestFieldOrdinal = 0;
643 kj::Maybe<PythonSnapshotRelease::Reader> result;
644 
645 static const auto fieldTable =
646 makePythonSnapshotFieldTable(capnp::Schema::from<CompatibilityFlags>().getFields());
647 
648 for (auto field: fieldTable) {
649 bool isEnabled = capnp::toDynamic(featureFlags).get(field.field).as<bool>();
650 if (!isEnabled) {
651 continue;
652 }
653 
654 // We pick the flag with the highest ordinal value that is enabled and has a
655 // pythonSnapshotRelease annotation.
656 //
657 // The fieldTable is probably ordered by the ordinal anyway, but doesn't hurt to be explicit
658 // here.
659 if (latestFieldOrdinal < field.field.getIndex()) {
660 latestFieldOrdinal = field.field.getIndex();
661 result = field.pythonSnapshotRelease;
662 }
663 }
664 
665 return result;
666}
667 
668kj::String getPythonBundleName(PythonSnapshotRelease::Reader pyodideRelease) {
669 if (pyodideRelease.getPyodide() == "dev") {
670 return kj::str("dev");
671 }
672 return kj::str(pyodideRelease.getPyodide(), "_", pyodideRelease.getPyodideRevision(), "_",
673 pyodideRelease.getBackport());
674}
675 
676namespace api::pyodide {
677 
678// Returns a string containing the contents of the hashset, delimited by ", "
679kj::String hashsetToString(const kj::HashSet<kj::String>& set) {
680 if (set.size() == 0) {
681 return kj::String();
682 }
683 
684 kj::Vector<kj::StringPtr> elems;
685 for (const auto& e: set) {
686 elems.add(e);
687 }
688 
689 // Sort the elements for consistent output
690 auto array = elems.releaseAsArray();
691 std::sort(array.begin(), array.end());
692 
693 return kj::str(kj::delimited(array, ", "_kjc));
694}
695 
696kj::Array<kj::String> getPythonPackageFiles(kj::StringPtr lockFileContents,
697 kj::ArrayPtr<kj::String> requirements,
698 kj::StringPtr packagesVersion) {
699 auto packages = parseLockFile(lockFileContents);
700 auto depMap = getDepMapFromPackagesLock(*packages);
701 
702 auto allRequirements = getPythonPackageNames(*packages, depMap, requirements, packagesVersion);
703 
704 // Add the file names of all the requirements to our result array.
705 kj::Vector<kj::String> res;
706 for (const auto& ent: *packages) {
707 auto name = ent.getName();
708 auto obj = ent.getValue().getObject();
709 auto fileName = kj::str(getField(obj, "file_name").getString());
710 
711 auto maybeRow = allRequirements.find(name);
712 KJ_IF_SOME(row, maybeRow) {
713 allRequirements.erase(row);
714 res.add(kj::mv(fileName));
715 } else if (packagesVersion == "20240829.4") {
716 auto packageType = getField(obj, "package_type").getString();
717 if (packageType == "cpython_module") {
718 res.add(kj::mv(fileName));
719 }
720 }
721 }
722 
723 if (allRequirements.size() != 0) {
724 JSG_FAIL_REQUIRE(Error,
725 "Requested Python package(s) that are not supported: ", hashsetToString(allRequirements));
726 }
727 
728 return res.releaseAsArray();
729}
730 
731void WorkerFatalReporter::reportFatal(jsg::Lock& js, kj::String error) {
732 KJ_IF_SOME(ioContext, IoContext::tryCurrent()) {
733 kj::runCatchingExceptions([&]() { ioContext.getMetrics().setWorkerFatal(); });
734 }
735}
736 
737void WorkerFatalReporter::reportPythonWorkersInternalError(jsg::Lock& js) {
738 KJ_IF_SOME(ioContext, IoContext::tryCurrent()) {
739 kj::runCatchingExceptions([&]() { ioContext.getMetrics().setPythonWorkersInternalError(); });
740 }
741}
742 
743} // namespace api::pyodide
744 
745} // namespace workerd