diff --git a/THIRD_PARTY.md b/THIRD_PARTY.md index c23f0ec4df..88a1576eca 100644 --- a/THIRD_PARTY.md +++ b/THIRD_PARTY.md @@ -117,9 +117,10 @@ License summary: | wyhash | `internal/cbm/vendored/wyhash/` | Unlicense (public domain) | [wangyi-fudan/wyhash](https://github.com/wangyi-fudan/wyhash) | Local modifications to these libraries are documented next to the -vendored sources (currently only SQLite: `vendored/sqlite3/PATCHES.md`, -raising the Unix VFS `MAX_PATHNAME` ceiling from 512 to 4096 to match -CBM's 4 KiB path support). Patches must be reapplied on every upstream +vendored sources: `vendored/sqlite3/PATCHES.md` documents the Unix VFS +`MAX_PATHNAME` ceiling increase from 512 to 4096 to match CBM's 4 KiB path +support; `vendored/mimalloc/PATCHES.md` documents the reproducible version +banner patch and upstream source pin. Patches must be reapplied on every upstream refresh and are covered by `scripts/vendored-checksums.txt`. The graph-UI HTTP server is a first-party implementation diff --git a/scripts/ci/generate-sbom.py b/scripts/ci/generate-sbom.py index 31f333c147..706805f9d9 100644 --- a/scripts/ci/generate-sbom.py +++ b/scripts/ci/generate-sbom.py @@ -40,7 +40,7 @@ "packages": [ {"SPDXID": "SPDXRef-Package-sqlite3", "name": "sqlite3", "versionInfo": "3.51.3", "licenseDeclared": "blessing", "downloadLocation": "https://sqlite.org", "filesAnalyzed": False}, {"SPDXID": "SPDXRef-Package-yyjson", "name": "yyjson", "versionInfo": "0.12.0", "licenseDeclared": "MIT", "downloadLocation": "https://github.com/ibireme/yyjson", "filesAnalyzed": False}, - {"SPDXID": "SPDXRef-Package-mimalloc", "name": "mimalloc", "versionInfo": "3.3.2", "licenseDeclared": "MIT", "downloadLocation": "https://github.com/microsoft/mimalloc", "filesAnalyzed": False}, + {"SPDXID": "SPDXRef-Package-mimalloc", "name": "mimalloc", "versionInfo": "3.4.4", "licenseDeclared": "MIT", "downloadLocation": "https://github.com/microsoft/mimalloc", "filesAnalyzed": False}, {"SPDXID": "SPDXRef-Package-xxhash", "name": "xxhash", "versionInfo": "0.8.3", "licenseDeclared": "BSD-2-Clause", "downloadLocation": "https://github.com/Cyan4973/xxHash", "filesAnalyzed": False}, {"SPDXID": "SPDXRef-Package-tre", "name": "tre", "versionInfo": "0.8.0", "licenseDeclared": "BSD-2-Clause", "downloadLocation": "https://github.com/laurikari/tre", "filesAnalyzed": False}, {"SPDXID": "SPDXRef-Package-tree-sitter", "name": "tree-sitter", "versionInfo": "0.24.4", "licenseDeclared": "MIT", "downloadLocation": "https://github.com/tree-sitter/tree-sitter", "filesAnalyzed": False}, diff --git a/scripts/test.sh b/scripts/test.sh index 3a4072064e..fa8c519848 100755 --- a/scripts/test.sh +++ b/scripts/test.sh @@ -476,6 +476,11 @@ make -j"$NPROC" -f Makefile.cbm cbm TEST_SEAMS=1 ${MAKE_ARGS[@]+"${MAKE_ARGS[@]} WATCHDOG_BINARY="$ROOT/$BUILD_DIR/codebase-memory-mcp" CBM_TEST_BINARY="$WATCHDOG_BINARY" bash "$ROOT/tests/test_parent_watchdog.sh" +# The sanitizer runner omits global allocator interposition. Exercise the real +# release allocator before its constructor, independently of the host libc. +echo "=== Step 5a0: production allocator startup regression (mimalloc #1341) ===" +bash "$ROOT/tests/test_mimalloc_startup.sh" "$ROOT/$BUILD_DIR" + # Step 5a: that watchdog is also the SMALLEST-stack thread in the image, which # makes it the first casualty when static TLS grows — glibc takes the TLS block # out of each thread's own stack allocation. Checked here, against the binary diff --git a/scripts/vendored-checksums.txt b/scripts/vendored-checksums.txt index ef424f4a13..a27d06ee38 100644 --- a/scripts/vendored-checksums.txt +++ b/scripts/vendored-checksums.txt @@ -992,50 +992,52 @@ a86b7735823b0408a4f3b922be86d4d25b6f4ab0dab32e03fb77246ca439dcf2 internal/cbm/v 9b4bc8245565c98ccfc61c07749928b57e7c0f6fddb0530c4f6aa1971893d88b internal/cbm/vendored/zstd/zstd.h 66a8c3f71d12ea6e797e4f622f31f3f8f81c41b36f48cad4f5de7d8bfb6aac0a internal/cbm/vendored/zstd/zstd_errors.h 82ade5d7d9b029044b5fbbc326207fc47c2d44ee4dd71e8a559c36d728217de9 vendored/mimalloc/LICENSE -dd4f25cae53209d45d73f8e6a2b9c219e8fc7434d97f20eba9af1a8b850030fd vendored/mimalloc/include/mimalloc-new-delete.h +f39ac1a2ed90a4e1b5590f451efcf2e2b49896a3e7652dd9e08b05b48809dbcd vendored/mimalloc/PATCHES.md +1bc31e20fb0340d9d071c69eaac2f07d0dfe4cdf95849ed8d91fb2bd7538d55b vendored/mimalloc/include/mimalloc-new-delete.h 21fcf61c4443341ac6bf6ea528af31dc7267e8e3456fc64bfd07704503032175 vendored/mimalloc/include/mimalloc-override.h -5260df227ea1612b8678a06b063cc8f48e739b551554541cdeb4d430a7e257b3 vendored/mimalloc/include/mimalloc-stats.h -869161433ab2d807a75752d0762accf5b68b6b24ce67e417af46bf181b9b6c72 vendored/mimalloc/include/mimalloc.h -e9512b521a82cc0cbc262f6060dd70bb4243ac4a7d69d6ab8b1ef8e56f181a21 vendored/mimalloc/include/mimalloc/atomic.h -57a7297e18ae464a7bfd3fc25a46229008602385697305c69ed289b89ece7631 vendored/mimalloc/include/mimalloc/bits.h -5aeeff21d667daf4aadcc66f634aa957ecd688d15b277828f367c8c2274251a8 vendored/mimalloc/include/mimalloc/internal.h -555302d2fcd05e974441d0c9c9208c26e860db4b3f17b9025a08f42e0f4807ba vendored/mimalloc/include/mimalloc/prim.h -1f378b4b6063c078d2aeb5d916c1d30b5db787a44ab9118e7daf673b086882c6 vendored/mimalloc/include/mimalloc/track.h -b54e760da2a8d9d8a9cb708e280268c2d194c99bc518b21f72db70100578ad6a vendored/mimalloc/include/mimalloc/types.h -70420b4430da4bb6b5197fd27c6c0df485249c53a986bc9536c9fb643fc7613b vendored/mimalloc/src/alloc-aligned.c -6c8e9b0d042481c55791edb7330daec64bedaf26f0e70c9ca40de3cc17a690fe vendored/mimalloc/src/alloc-override.c -98d0eb877d7b9cdf7e9f857aaf3c739b69f746b91221e0e427e82bdd7cec90b3 vendored/mimalloc/src/alloc-posix.c -a92d58e47220e5b6d138b83d92498771be4195ee59136a134e4461f6f726b9b8 vendored/mimalloc/src/alloc.c -86fde644ede2e8d582e53499568a1713516a20326e0350eaa06a28029cece91b vendored/mimalloc/src/arena-meta.c -a1d6900d2b9a17461aa2a72a7791dfb22f6167b86f7ac6f15f981d9735aa2adb vendored/mimalloc/src/arena.c -de16033e029488e9ade4152ea1274e1a5ecbdc2c8a8d8ee50b745fd037ee6f6c vendored/mimalloc/src/bitmap.c -2e7a54ff44592a1fe8fe929e691322622f8aa1e0b1e80954781f332a88dd6cd1 vendored/mimalloc/src/bitmap.h -c9a845dd7b9020f26d73761d235fb75e7bc4a3682576de750050ac4098211b43 vendored/mimalloc/src/free.c -ae8e88737c9861e8e071fe5f1e7acc52093bcffd136a359a6b473101e968c14f vendored/mimalloc/src/heap.c -eca1fe77d72ab2aaccc447c8119056e9c4e8851bda9a95472b573539f76107ad vendored/mimalloc/src/init.c -ebd95b475f15d5387e6ef3c88a9cb47b88e64d7e55998ae1dc656614676a443b vendored/mimalloc/src/libc.c -192ce06ac9076445d223205fb340d8a1fc818dbf2b4c4a801dfd2c10bc6c1162 vendored/mimalloc/src/options.c -011e965f28c0e2c885a6834d51bd756a458dd37ebfed6fd4d9ad6607ce0390ed vendored/mimalloc/src/os.c -1be3a208c678fc34187c2a8ffb75f3fe1eecc674bbf8b2af4d1f329506b1fad5 vendored/mimalloc/src/page-map.c -7ec275805c9f00eb66710ffa99817b3ed1dba9826435d08f4866f7935152f6ae vendored/mimalloc/src/page-queue.c -d3614a59f2407a32e75b051deb1aff31b233f587d325dcc5420ba508551b687e vendored/mimalloc/src/page.c -7488e9e562ca85d2d757ec48e4d334935263a2333aa05b6061d8520f427c5890 vendored/mimalloc/src/prim/emscripten/prim.c -ce14eb6dfd55f241095b55f70983931a2d92f7a6a622a9b360cdd70ab3b53208 vendored/mimalloc/src/prim/osx/alloc-override-zone.c +48b31f349120c534b008d7bc43f7e3c9602df5fb3530b10688a19a0f31329d0a vendored/mimalloc/include/mimalloc-stats.h +b457b2365eb25d852efe02f0b5808c99f822b2934d6bcf9acf3a06254c8dcdd1 vendored/mimalloc/include/mimalloc.h +b815499d737a607fec9e4da05de9b93eadc71a7227e7297723d2595ef76bb56b vendored/mimalloc/include/mimalloc/atomic.h +6efb7256ce1e545c89fb4fb9b644559d2048d2856e3a0b896c81bfc449d2fd55 vendored/mimalloc/include/mimalloc/bits.h +d1775f0e8693391d63a076454e2eb9bdc6662c966d7151e566fd205a6f6f7cd6 vendored/mimalloc/include/mimalloc/internal.h +8eaf268579821ba137d9ff14497ffff54684ee3efaf30907d11e9a3586a0d615 vendored/mimalloc/include/mimalloc/prim-tls.h +1dabd75cbc71502d0f88a8d2af94526bf151f337b702bc9c07b9cda90ae218b3 vendored/mimalloc/include/mimalloc/prim.h +e7d94cae6b36f92b934b76aa3d24854845710e798654bdd29cef90e13e0f161e vendored/mimalloc/include/mimalloc/track.h +7561d1224ca585923fa673fd7add6e5b83ec491615f172217554ea89167ae557 vendored/mimalloc/include/mimalloc/types.h +ce91209c4299f627874ae23c9e7e16e36d585742bf29e5bad48350460d7da2f9 vendored/mimalloc/src/alloc-aligned.c +64c145ff98af7ed0a2d4049b0c7ab7adeae816237990e73919cf11d387a8f831 vendored/mimalloc/src/alloc-override.c +cd4cb1b639a7eab39425a669f897558df761b8103cd5f66f8028c259cf62690a vendored/mimalloc/src/alloc-posix.c +2d227b31c62fbfa2b6d8b1603b62e86bb1aecacc80c60fbddc83656b117c8561 vendored/mimalloc/src/alloc.c +6929e3660109969b7b86aa7538dcb4c3ab14e347bfa0a5abbc2ab6fc8adc4a6d vendored/mimalloc/src/arena-meta.c +0919ab2a7a69a90ff88b7cfcdb3d4c67a2bfc791b8a6617cd31b37b7b9295093 vendored/mimalloc/src/arena.c +b148dcfc3b6b5b9f3e6ef33278b7c5c063bdfa555cb84399e7026d1b726ecd01 vendored/mimalloc/src/bitmap.c +69dea8bc6badb8612ce3c10a1a16a7e54aacc7cd1a82b16ef00efe9ab3d00d03 vendored/mimalloc/src/bitmap.h +1abf2f16f7c8b8bc96e3c91adac2793cf2f62cb18d3aa8009b39c00e18da04fd vendored/mimalloc/src/free.c +826493ad3b4b999071d03884e5bd63380a07c2b26e558ef05dc181408917500a vendored/mimalloc/src/heap.c +21e06e19ca2383f98cbacf2629e613a28c4165baf3d0102ca46834ecab5bcdcc vendored/mimalloc/src/init.c +47c9375964be1f93d77e26e7f2216bd2ba3fe02104d684eab439b7f3863849ca vendored/mimalloc/src/libc.c +78213175bcd6f649408dd0b6133cd60bb6640be6dac6f05edffa33211ab7a753 vendored/mimalloc/src/options.c +e55a179cbbe6035133e0d76f5a09b0958044af9cc91b49174ca660505814454c vendored/mimalloc/src/os.c +8e320ee05354b73aec7c786cbf3659bd84efa53cabeed9249626e41fd22d7e5c vendored/mimalloc/src/page-map.c +a7f3e45ecfacf0beab5b8df20dfb7f43990c96ab906f64d50a2a7526c7ac7f86 vendored/mimalloc/src/page-queue.c +ca1ae321f315f4883ffffb648db70a391b181f185f6a2b4bde7d0b2682fc4fde vendored/mimalloc/src/page.c +896fc1016a26e10c6cf96259d0524c57e123ac9dcef1ba04f23d0add445286dd vendored/mimalloc/src/prim/emscripten/prim.c +10dde0c09fb859e2a67151f51d0737a50531c6fd492eb563e1e6ad03a9b0800c vendored/mimalloc/src/prim/osx/alloc-override-zone.c 247a9952465eb105be03a9962e922b085ce5d52034775551a61cd8594275be73 vendored/mimalloc/src/prim/osx/prim.c -cc771788e5ba591efb681829a8724f8662a1f50f3db39cf5ebe7738b8ed57e44 vendored/mimalloc/src/prim/prim.c +6eff9f103ed868bca18d158673e1f6d661cd49b377229d2a87832cc8a1b6e966 vendored/mimalloc/src/prim/prim.c 792531421c247fca0766ff7630057ec491173578ca4d1919ae34f05eb289ccf4 vendored/mimalloc/src/prim/readme.md -c1ae32eaac7ccf18c10058cca99723892f8bd027318590151f5011c929c7e3be vendored/mimalloc/src/prim/unix/prim.c -bee9c655a17e1e93bf4908e0a902fd3b566628447b1446fcbfa92bca2e76ca15 vendored/mimalloc/src/prim/wasi/prim.c +14c9e5a2e6c585bb4e6ab4c34534d2dd15fac7b7def6c5b210cd158420f54d3f vendored/mimalloc/src/prim/unix/prim.c +2bc862cddb522314d58b3c6ef041e101c25cf04dfdda35cf78f91dbb6140cf46 vendored/mimalloc/src/prim/wasi/prim.c f70e61a1f2fe63da2f51c581895cf4a0918a9bffb0cf8600171f297141446962 vendored/mimalloc/src/prim/windows/etw-mimalloc.wprp 4f3110ef2054c95cb275be96cb279224d3d46a728c5117bd1435425f382de778 vendored/mimalloc/src/prim/windows/etw.h 445e6a70f9728fcc76a5ba52079336064d5e2257c2969023f1c64bf943776f54 vendored/mimalloc/src/prim/windows/etw.man -5cde60fcef94cd79ab816030be00f8c1bdf18fab343563508c681ec6e762d1f1 vendored/mimalloc/src/prim/windows/prim.c +0ebbc38867f12c9a091983dbbcc5ba00bef1767520d95a763228cd7f6b26d2a2 vendored/mimalloc/src/prim/windows/prim.c 1bcb43eef7fcddf6d8f18c6d5ee9bd350090ed183a1009d4f1c9b4d81a205bb9 vendored/mimalloc/src/prim/windows/readme.md -a96a1acc75fd6595811b28420cf0495b87fac98c4dbc73aa0d742b6824f607c8 vendored/mimalloc/src/random.c +c833eaf89ebd73c05a47a5ea8b439f4cc8a7ec5b87f85ddad4ddb72640d1c34e vendored/mimalloc/src/random.c 4059be2a6676693be3433f5aedc711f56301772e65f3573654c236d6d1cfd074 vendored/mimalloc/src/static.c -16825bbc252de422982b5728df7eda181d952824f5e3ce958c7fc85feae64b8e vendored/mimalloc/src/stats.c -7f76bcace4b0578f2063477752138adabd99f7b10a276998e296a312c47a7cfe vendored/mimalloc/src/theap.c -49be69a4b674e09690a63870830f3ffb05698ed4210949ef5add11957a08325b vendored/mimalloc/src/threadlocal.c +05a99787ab042f1fc714f0f53ba4ed0d884e3b8aaaaaead73d0a6db3dd4e36ac vendored/mimalloc/src/stats.c +ac9623f2275db103b7c53d7d6f3f55b93bdf5098e3c5e38c638b52f109d0f60d vendored/mimalloc/src/theap.c +7ff67f4b97409e208a8d24d7ea13e7bc4f4d57d40b71bdce912d607a021bbe8d vendored/mimalloc/src/threadlocal.c cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30 vendored/nomic/LICENSE 893a74d9e3f0a74358344edc999e165739df343b36297967f3fd9cd462022f39 vendored/nomic/NOTICE dab3009c0d76b0c5e05ad8abc7c9f8f6effca547fea3c0394be96418a85c1081 vendored/nomic/code_tokens.h diff --git a/tests/test_mimalloc_startup.sh b/tests/test_mimalloc_startup.sh new file mode 100644 index 0000000000..b20ca2b894 --- /dev/null +++ b/tests/test_mimalloc_startup.sh @@ -0,0 +1,56 @@ +#!/usr/bin/env bash +set -euo pipefail + +# The sanitizer runner does not interpose malloc/free. Link the production +# allocator instead and call free(NULL) from ELF preinit, before its constructor. +# This reproduces mimalloc #1341 even on a libc that does not call free(NULL) +# during libstdc++ startup. Run after building cbm: +# bash tests/test_mimalloc_startup.sh [build-directory] +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +BUILD_DIR="${1:-$ROOT/build/c}" +if [[ "$(uname -s)" != Linux ]]; then + echo "SKIP_PLATFORM: mimalloc startup regression requires ELF preinit (Linux)" + exit 0 +fi +[[ -f "$BUILD_DIR/prod_mimalloc.o" ]] || { + echo "FAIL: build the production allocator first: $BUILD_DIR/prod_mimalloc.o" >&2 + exit 1 +} +WORK="$(mktemp -d)" +trap 'rm -rf "$WORK"' EXIT + +cat > "$WORK/startup.c" <<'EOF' +#include +#include +#include "mimalloc.h" + +static bool early_free_called; + +static void early_free(void) { + void (*volatile release)(void *) = free; // Prevent optimizing away free(NULL). + release(NULL); + early_free_called = true; +} + +__attribute__((section(".preinit_array"), used)) +static void (*const preinit)(void) = early_free; + +int main(void) { + if (!early_free_called) { + return 1; + } + void *p = mi_malloc(32); + if (p == NULL) { + return 1; + } + mi_free(p); + return 0; +} +EOF + +read -r -a compiler <<< "${CC:-cc}" +"${compiler[@]}" -std=c11 -O2 -Wall -Wextra -Werror \ + -I"$ROOT/vendored/mimalloc/include" "$WORK/startup.c" \ + "$BUILD_DIR/prod_mimalloc.o" -lm -lpthread -o "$WORK/startup" +"$WORK/startup" +echo "PASS: production free(NULL) before allocator initialization" diff --git a/vendored/mimalloc/PATCHES.md b/vendored/mimalloc/PATCHES.md new file mode 100644 index 0000000000..8e36bd91f1 --- /dev/null +++ b/vendored/mimalloc/PATCHES.md @@ -0,0 +1,24 @@ +# mimalloc vendoring + +Sources: [microsoft/mimalloc v3.4.4](https://github.com/microsoft/mimalloc/tree/v3.4.4), +commit `1f06f694972279bbc7ec72902e8570f5784d0fc9`. +Only `include/`, `src/`, and `LICENSE` are vendored. + +This release includes the fix for +[mimalloc #1341](https://github.com/microsoft/mimalloc/issues/1341): +`free(NULL)` must work before allocator initialization. Without it, glibc 2.44's +`newlocale()` can crash during libstdc++ startup with CBM's Linux global override. + +## Local modifications + +`src/options.c`: remove `__DATE__` and `__TIME__` from the version banner so +identical sources can produce reproducible binaries. The patch is marked +`CBM LOCAL PATCH` inline. Reapply it on every upstream refresh. + +`src/prim/osx/alloc-override-zone.c`: reword the comment about a child crashing +while forking, avoiding a literal function-call spelling that CBM's vendored +security scanner would mistake for an actual subprocess call. No code changes. + +After refreshing, update the mimalloc version in `scripts/ci/generate-sbom.py` +and regenerate `scripts/vendored-checksums.txt` with +`scripts/security-vendored.sh --update`. diff --git a/vendored/mimalloc/include/mimalloc-new-delete.h b/vendored/mimalloc/include/mimalloc-new-delete.h index c16f4a6653..aaf185bb14 100644 --- a/vendored/mimalloc/include/mimalloc-new-delete.h +++ b/vendored/mimalloc/include/mimalloc-new-delete.h @@ -48,7 +48,7 @@ terms of the MIT license. A copy of the license can be found in the file void operator delete[](void* p, std::size_t n) noexcept { mi_free_size(p,n); }; #endif - #if (__cplusplus > 201402L || defined(__cpp_aligned_new)) + #if (__cplusplus > 201402L && defined(__cpp_aligned_new)) void operator delete (void* p, std::align_val_t al) noexcept { mi_free_aligned(p, static_cast(al)); } void operator delete[](void* p, std::align_val_t al) noexcept { mi_free_aligned(p, static_cast(al)); } void operator delete (void* p, std::size_t n, std::align_val_t al) noexcept { mi_free_size_aligned(p, n, static_cast(al)); }; diff --git a/vendored/mimalloc/include/mimalloc-stats.h b/vendored/mimalloc/include/mimalloc-stats.h index 1c620fd1c1..b4e19c09aa 100644 --- a/vendored/mimalloc/include/mimalloc-stats.h +++ b/vendored/mimalloc/include/mimalloc-stats.h @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2024-2025, Microsoft Research, Daan Leijen +Copyright (c) 2024-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -22,7 +22,7 @@ terms of the MIT license. A copy of the license can be found in the file #elif __cplusplus >= 201103L #define mi_decl_align(a) alignas(a) #else -#define mi_decl_align(a) +#define mi_decl_align(a) _Alignas(a) #endif diff --git a/vendored/mimalloc/include/mimalloc.h b/vendored/mimalloc/include/mimalloc.h index f373f5dc87..439bccf1e1 100644 --- a/vendored/mimalloc/include/mimalloc.h +++ b/vendored/mimalloc/include/mimalloc.h @@ -8,7 +8,7 @@ terms of the MIT license. A copy of the license can be found in the file #ifndef MIMALLOC_H #define MIMALLOC_H -#define MI_MALLOC_VERSION 30302 // major + 2 digits minor + 2 digits patch +#define MI_MALLOC_VERSION 30404 // major + 2 digits minor + 2 digits patch // ------------------------------------------------------ // Compiler specific attributes @@ -95,7 +95,7 @@ terms of the MIT license. A copy of the license can be found in the file // Includes // ------------------------------------------------------ -#include // size_t +#include // size_t, wchar_t #include // bool #ifdef __cplusplus @@ -187,7 +187,7 @@ typedef void (mi_cdecl mi_error_fun)(int err, void* arg); mi_decl_export void mi_register_error(mi_error_fun* fun, void* arg); mi_decl_export void mi_collect(bool force) mi_attr_noexcept; -mi_decl_export int mi_version(void) mi_attr_noexcept; +mi_decl_export int mi_version(void); mi_decl_export void mi_options_print(void) mi_attr_noexcept; mi_decl_export void mi_process_info_print(void) mi_attr_noexcept; mi_decl_export void mi_options_print_out(mi_output_fun* out, void* arg) mi_attr_noexcept; @@ -465,7 +465,7 @@ typedef enum mi_option_e { mi_option_deprecated_purge_extend_delay, mi_option_disallow_arena_alloc, // 1 = do not use arena's for allocation (except if using specific arena id's) mi_option_retry_on_oom, // retry on out-of-memory for N milli seconds (=400), set to 0 to disable retries. (only on windows) - mi_option_visit_abandoned, // allow visiting theap blocks from abandoned threads (=0) + mi_option_deprecated_visit_abandoned, // allow visiting theap blocks from abandoned threads (=0) mi_option_guarded_min, // only used when building with MI_GUARDED: minimal rounded object size for guarded objects (=0) mi_option_guarded_max, // only used when building with MI_GUARDED: maximal rounded object size for guarded objects (=0) mi_option_guarded_precise, // disregard minimal alignment requirement to always place guarded blocks exactly in front of a guard page (=0) @@ -519,7 +519,7 @@ mi_decl_nodiscard mi_decl_export size_t mi_malloc_size(const void* p) mi_ mi_decl_nodiscard mi_decl_export size_t mi_malloc_good_size(size_t size) mi_attr_noexcept; mi_decl_nodiscard mi_decl_export size_t mi_malloc_usable_size(const void *p) mi_attr_noexcept; -mi_decl_export int mi_posix_memalign(void** p, size_t alignment, size_t size) mi_attr_noexcept; +mi_decl_export int mi_posix_memalign(void** p, size_t alignment, size_t size); // mi_attr_noexcept; mi_decl_nodiscard mi_decl_export mi_decl_restrict void* mi_memalign(size_t alignment, size_t size) mi_attr_noexcept mi_attr_malloc mi_attr_alloc_size(2) mi_attr_alloc_align(1); mi_decl_nodiscard mi_decl_export mi_decl_restrict void* mi_valloc(size_t size) mi_attr_noexcept mi_attr_malloc mi_attr_alloc_size(1); mi_decl_nodiscard mi_decl_export mi_decl_restrict void* mi_pvalloc(size_t size) mi_attr_noexcept mi_attr_malloc mi_attr_alloc_size(1); @@ -536,7 +536,6 @@ mi_decl_export void mi_free_aligned(void* p, size_t alignment) mi_decl_export int mi_dupenv_s(char** buf, size_t* size, const char* name) mi_attr_noexcept; // wide characters -#include // wchar_t mi_decl_export int mi_wdupenv_s(wchar_t** buf, size_t* size, const wchar_t* name) mi_attr_noexcept; mi_decl_nodiscard mi_decl_export mi_decl_restrict wchar_t* mi_wcsdup(const wchar_t* s) mi_attr_noexcept mi_attr_malloc; mi_decl_nodiscard mi_decl_export mi_decl_restrict unsigned char* mi_mbsdup(const unsigned char* s) mi_attr_noexcept mi_attr_malloc; diff --git a/vendored/mimalloc/include/mimalloc/atomic.h b/vendored/mimalloc/include/mimalloc/atomic.h index e4abde6b04..2aa86d8a92 100644 --- a/vendored/mimalloc/include/mimalloc/atomic.h +++ b/vendored/mimalloc/include/mimalloc/atomic.h @@ -14,8 +14,8 @@ terms of the MIT license. A copy of the license can be found in the file #define WIN32_LEAN_AND_MEAN #endif #include -#elif !defined(__wasi__) && (!defined(__EMSCRIPTEN__) || defined(__EMSCRIPTEN_PTHREADS__)) -#define MI_USE_PTHREADS +#elif MI_TLS_MODEL_PTHREADS || (!defined(__wasi__) && (!defined(__EMSCRIPTEN__) || defined(__EMSCRIPTEN_PTHREADS__))) +#define MI_USE_PTHREADS 1 #include #endif @@ -215,16 +215,12 @@ static inline uintptr_t mi_atomic_exchange_explicit(_Atomic(uintptr_t)*p, uintpt (void)(mo); return (uintptr_t)MI_MSC_64(_InterlockedExchange)((volatile msc_intptr_t*)p, (msc_intptr_t)exchange); } -static inline void mi_atomic_thread_fence(mi_memory_order mo) { - (void)(mo); - _Atomic(uintptr_t) x = 0; - mi_atomic_exchange_explicit(&x, 1, mo); -} static inline uintptr_t mi_atomic_load_explicit(_Atomic(uintptr_t) const* p, mi_memory_order mo) { (void)(mo); // assert(mo<=mi_memory_order_acquire); // others are not used by mimalloc #if defined(_M_IX86) || defined(_M_X64) + // on x86/x64 we have a strong memory model so any load is acquire return (uintptr_t)MI_MSC_XX(__iso_volatile_load)((volatile const intptr_t*)p); #elif defined(_M_ARM) || defined(_M_ARM64) if (mo == mi_memory_order_relaxed) { @@ -335,18 +331,18 @@ static inline void mi_atomic_void_addi64_relaxed(volatile int64_t* p, const vola } } -static inline void mi_atomic_maxi64_relaxed(volatile _Atomic(int64_t)*p, int64_t x) { +static inline void mi_atomic_maxi64_relaxed(volatile _Atomic(int64_t)* p, int64_t x) { int64_t current; do { current = *p; } while (current < x && _InterlockedCompareExchange64(p, x, current) != current); } -static inline void mi_atomic_addi64_acq_rel(volatile _Atomic(int64_t*)p, int64_t i) { +static inline void mi_atomic_addi64_acq_rel(volatile _Atomic(int64_t)* p, int64_t i) { mi_atomic_addi64_relaxed(p, i); } -static inline bool mi_atomic_casi64_strong_acq_rel(volatile _Atomic(int64_t*)p, int64_t* exp, int64_t des) { +static inline bool mi_atomic_casi64_strong_acq_rel(volatile _Atomic(int64_t)* p, int64_t* exp, int64_t des) { const int64_t read = _InterlockedCompareExchange64(p, des, *exp); if (read == *exp) { return true; @@ -496,10 +492,10 @@ static inline void mi_lock_release(mi_lock_t* lock) { lock->mutex.unlock(); } static inline void mi_lock_init(mi_lock_t* lock) { - new(&lock->mutex) std::mutex(); + new(&lock->mutex) std::mutex(); // in-place constructor } static inline void mi_lock_done(mi_lock_t* lock) { - (void)(lock); + lock->mutex.~mutex(); // in-place destructor } #else @@ -524,6 +520,7 @@ static inline bool mi_lock_try_acquire(mi_lock_t* lock) { return mi_atomic_cas_strong_acq_rel(&lock->mutex, &expected, (uintptr_t)1); } static inline void mi_lock_acquire(mi_lock_t* lock) { + size_t ticks = 0; for (int i = 0; i < 10000; i++) { // for at most 10000 tries? if (mi_lock_try_acquire(lock)) return; _mi_prim_thread_yield(); diff --git a/vendored/mimalloc/include/mimalloc/bits.h b/vendored/mimalloc/include/mimalloc/bits.h index 617877283e..f2c6f3f260 100644 --- a/vendored/mimalloc/include/mimalloc/bits.h +++ b/vendored/mimalloc/include/mimalloc/bits.h @@ -16,6 +16,7 @@ terms of the MIT license. A copy of the license can be found in the file #include // size_t #include // int64_t etc #include // bool +#include // LONG_MAX // ------------------------------------------------------ // Size of a pointer. @@ -102,13 +103,13 @@ typedef int32_t mi_ssize_t; #include #endif -#if MI_ARCH_X64 && defined(__AVX2__) && !defined(__BMI2__) // msvc +#if MI_ARCH_X64 && defined(__AVX2__) && !defined(__BMI2__) // avx2 implies bmi2 #define __BMI2__ 1 #endif -#if MI_ARCH_X64 && (defined(__AVX2__) || defined(__BMI2__)) && !defined(__BMI1__) // msvc +#if MI_ARCH_X64 && (defined(__AVX2__) || defined(__BMI2__) || defined(__BMI__)) && !defined(__BMI1__) // bmi2 implies bmi1 #define __BMI1__ 1 #endif -#if MI_ARCH_X64 && defined(__AVX2__) && !defined(__LZCNT__) // msvc +#if MI_ARCH_X64 && defined(__AVX2__) && !defined(__LZCNT__) // avx2 implies lzcnt #define __LZCNT__ 1 #endif @@ -126,6 +127,15 @@ typedef int32_t mi_ssize_t; #define MI_MAX_VABITS (32) #endif +// the MI_MIN_VABITS determine how many bits of the address are always mapped in the page_map +#if MI_MAX_VABITS <= 32 +#define MI_MIN_VABITS (32) +#elif MI_MAX_VABITS <= 43 +#define MI_MIN_VABITS MI_MAX_VABITS +#else +#define MI_MIN_VABITS (43) /* 8 TiB */ +#endif + // use a flat page-map or a 2-level one #ifndef MI_PAGE_MAP_FLAT #if MI_MAX_VABITS <= 40 && !defined(__APPLE__) && MI_SECURE==0 && !MI_PAGE_META_IS_SEPARATED @@ -177,12 +187,16 @@ typedef int32_t mi_ssize_t; -------------------------------------------------------------------------------- */ size_t _mi_popcount_generic(size_t x); +extern bool _mi_cpu_has_popcnt; static inline size_t mi_popcount(size_t x) { #if mi_has_builtinz(popcount) return mi_builtinz(popcount)(x); - #elif defined(_MSC_VER) && (MI_ARCH_X64 || MI_ARCH_X86 || MI_ARCH_ARM64 || MI_ARCH_ARM32) - return mi_msc_builtinz(__popcnt)(x); + #elif defined(_MSC_VER) && (MI_ARCH_ARM64 || MI_ARCH_ARM32) + return mi_msc_builtinz(_CountOneBits)(x); + #elif defined(_MSC_VER) && (MI_ARCH_X64 || MI_ARCH_X86) + if (_mi_cpu_has_popcnt) { return mi_msc_builtinz(__popcnt)(x); } + else { return _mi_popcount_generic(x); } // see issue #1291 #elif MI_ARCH_X64 && defined(__BMI1__) return (size_t)_mm_popcnt_u64(x); #else diff --git a/vendored/mimalloc/include/mimalloc/internal.h b/vendored/mimalloc/include/mimalloc/internal.h index febf2559e3..2174dd511c 100644 --- a/vendored/mimalloc/include/mimalloc/internal.h +++ b/vendored/mimalloc/include/mimalloc/internal.h @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -85,16 +85,6 @@ terms of the MIT license. A copy of the license can be found in the file #define mi_likely(x) (x) #endif -#ifndef __has_builtin -#define __has_builtin(x) 0 -#endif - -#if defined(__cplusplus) -#define mi_decl_externc extern "C" -#else -#define mi_decl_externc -#endif - #if (defined(__GNUC__) && (__GNUC__ >= 7)) || defined(__clang__) // includes clang and icc #define mi_decl_maybe_unused __attribute__((unused)) #elif __cplusplus >= 201703L // c++17 @@ -103,13 +93,16 @@ terms of the MIT license. A copy of the license can be found in the file #define mi_decl_maybe_unused #endif +#ifndef __has_builtin +#define __has_builtin(x) 0 +#endif + #if defined(__cplusplus) -#define mi_decl_externc extern "C" +#define mi_decl_externc extern "C" #else #define mi_decl_externc #endif - #if defined(__EMSCRIPTEN__) && !defined(__wasi__) #define __wasi__ #endif @@ -126,19 +119,18 @@ int _mi_vsnprintf(char* buf, size_t bufsize, const char* fmt, va_list int _mi_snprintf(char* buf, size_t buflen, const char* fmt, ...); char _mi_toupper(char c); int _mi_strnicmp(const char* s, const char* t, size_t n); -void _mi_strlcpy(char* dest, const char* src, size_t dest_size); -void _mi_strlcat(char* dest, const char* src, size_t dest_size); +bool _mi_strlcpy(char* dest, const char* src, size_t dest_size); // returns true if the entire src was copied +bool _mi_strlcat(char* dest, const char* src, size_t dest_size); // returns true if the entire src was appended size_t _mi_strlen(const char* s); size_t _mi_strnlen(const char* s, size_t max_len); char* _mi_strnstr(char* s, size_t max_len, const char* pat); bool _mi_streq(const char* s, const char* t); -bool _mi_getenv(const char* name, char* result, size_t result_size); +int _mi_getenv(const char* name, char* result, size_t result_size); // "options.c" void _mi_fputs(mi_output_fun* out, void* arg, const char* prefix, const char* message); void _mi_fprintf(mi_output_fun* out, void* arg, const char* fmt, ...); void _mi_raw_message(const char* fmt, ...); -void _mi_message(const char* fmt, ...); void _mi_warning_message(const char* fmt, ...); void _mi_verbose_message(const char* fmt, ...); void _mi_trace_message(const char* fmt, ...); @@ -165,8 +157,11 @@ bool _mi_is_redirected(void); bool _mi_allocator_init(const char** message); void _mi_allocator_done(void); bool _mi_is_main_thread(void); +bool _mi_is_process_heap_main(const mi_heap_t* heap); bool _mi_preloading(void); // true while the C runtime is not initialized yet void _mi_thread_done(mi_theap_t* theap); +mi_theap_t* _mi_thread_init(void); +bool _mi_is_empty_theap(const mi_theap_t* theap); mi_subproc_t* _mi_subproc(void); mi_subproc_t* _mi_subproc_main(void); @@ -174,20 +169,15 @@ mi_heap_t* _mi_subproc_heap_main(mi_subproc_t* subproc); mi_subproc_t* _mi_subproc_from_id(mi_subproc_id_t subproc_id); mi_threadid_t _mi_thread_id(void) mi_attr_noexcept; -size_t _mi_thread_seq_id(void) mi_attr_noexcept; -bool _mi_is_heap_main(const mi_heap_t* heap); -bool _mi_is_theap_main(const mi_theap_t* theap); void _mi_theap_guarded_init(mi_theap_t* theap); void _mi_theap_options_init(mi_theap_t* theap); -mi_theap_t* _mi_theap_default_safe(void); // ensure the returned theap is initialized -mi_theap_t* _mi_theap_main_safe(void); - + // os.c void _mi_os_init(void); // called from process init -void* _mi_os_alloc(size_t size, mi_memid_t* memid); -void* _mi_os_zalloc(size_t size, mi_memid_t* memid); -void _mi_os_free(void* p, size_t size, mi_memid_t memid); -void _mi_os_free_ex(void* p, size_t size, bool still_committed, mi_memid_t memid, mi_subproc_t* subproc ); +void* _mi_os_alloc(mi_subproc_t* subproc, size_t size, mi_memid_t* memid); +void* _mi_os_zalloc(mi_subproc_t* subproc, size_t size, mi_memid_t* memid); +void _mi_os_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid); +void _mi_os_free_ex(mi_subproc_t* subproc, void* p, size_t size, bool still_committed, mi_memid_t memid ); size_t _mi_os_page_size(void); size_t _mi_os_guard_page_size(void); @@ -197,34 +187,35 @@ bool _mi_os_has_virtual_reserve(void); size_t _mi_os_virtual_address_bits(void); size_t _mi_os_minimal_purge_size(void); -bool _mi_os_reset(void* addr, size_t size); -bool _mi_os_decommit(void* addr, size_t size); -void _mi_os_reuse(void* p, size_t size); -mi_decl_nodiscard bool _mi_os_commit(void* p, size_t size, bool* is_zero); -mi_decl_nodiscard bool _mi_os_commit_ex(void* addr, size_t size, bool* is_zero, size_t stat_size); +bool _mi_os_reset(mi_subproc_t* subproc, void* addr, size_t size); +bool _mi_os_decommit(mi_subproc_t* subproc, void* addr, size_t size); +void _mi_os_reuse(mi_subproc_t* subproc, void* p, size_t size); +mi_decl_nodiscard bool _mi_os_commit(mi_subproc_t* subproc, void* p, size_t size, bool* is_zero); +mi_decl_nodiscard bool _mi_os_commit_ex(mi_subproc_t* subproc, void* addr, size_t size, bool* is_zero, size_t stat_size); mi_decl_nodiscard bool _mi_os_protect(void* addr, size_t size); bool _mi_os_unprotect(void* addr, size_t size); -bool _mi_os_purge(void* p, size_t size); -bool _mi_os_purge_ex(void* p, size_t size, bool allow_reset, size_t stats_size, mi_commit_fun_t* commit_fun, void* commit_fun_arg); +bool _mi_os_purge(mi_subproc_t* subproc, void* p, size_t size); +bool _mi_os_purge_ex(mi_subproc_t* subproc, void* p, size_t size, bool allow_reset, size_t stats_size, mi_commit_fun_t* commit_fun, void* commit_fun_arg); size_t _mi_os_secure_guard_page_size(void); -bool _mi_os_secure_guard_page_set_at(void* addr, mi_memid_t memid); -bool _mi_os_secure_guard_page_set_before(void* addr, mi_memid_t memid); -bool _mi_os_secure_guard_page_reset_at(void* addr, mi_memid_t memid); -bool _mi_os_secure_guard_page_reset_before(void* addr, mi_memid_t memid); +bool _mi_os_secure_guard_page_set_at(mi_subproc_t* subproc, void* addr, mi_memid_t memid); +bool _mi_os_secure_guard_page_set_before(mi_subproc_t* subproc, void* addr, mi_memid_t memid); +bool _mi_os_secure_guard_page_reset_at(mi_subproc_t* subproc, void* addr, mi_memid_t memid); +bool _mi_os_secure_guard_page_reset_before(mi_subproc_t* subproc, void* addr, mi_memid_t memid); int _mi_os_numa_node(void); int _mi_os_numa_node_count(void); -void* _mi_os_alloc_aligned(size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid); -void* _mi_os_alloc_aligned_at_offset(size_t size, size_t alignment, size_t align_offset, bool commit, bool allow_large, mi_memid_t* memid); +void* _mi_os_alloc_aligned(mi_subproc_t* subproc, size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid); +void* _mi_os_alloc_aligned_at_offset(mi_subproc_t* subproc, size_t size, size_t alignment, size_t align_offset, bool commit, bool allow_large, mi_memid_t* memid); void* _mi_os_get_aligned_hint(size_t try_alignment, size_t size); bool _mi_os_canuse_large_page(size_t size, size_t alignment); size_t _mi_os_large_page_size(void); -void* _mi_os_alloc_huge_os_pages(size_t pages, int numa_node, mi_msecs_t max_secs, size_t* pages_reserved, size_t* psize, mi_memid_t* memid); +void* _mi_os_alloc_huge_os_pages(mi_subproc_t* subproc, size_t pages, int numa_node, mi_msecs_t max_secs, size_t* pages_reserved, size_t* psize, mi_memid_t* memid); // threadlocal.c +#define mi_thread_local_key_fast ((mi_thread_local_t)1) mi_thread_local_t _mi_thread_local_create(void); void _mi_thread_local_free( mi_thread_local_t key ); @@ -241,8 +232,7 @@ bool _mi_arena_memid_is_suitable(mi_memid_t memid, mi_arena_t* request_ void* _mi_arenas_alloc(mi_heap_t* heap, size_t size, bool commit, bool allow_pinned, mi_arena_t* req_arena, size_t tseq, int numa_node, mi_memid_t* memid); void* _mi_arenas_alloc_aligned(mi_heap_t* heap, size_t size, size_t alignment, size_t align_offset, bool commit, bool allow_pinned, mi_arena_t* req_arena, size_t tseq, int numa_node, mi_memid_t* memid); -void _mi_arenas_free(void* p, size_t size, mi_memid_t memid); -bool _mi_arenas_contain(const void* p); +void _mi_arenas_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid); void _mi_arenas_collect(bool force_purge, bool visit_all, mi_tld_t* tld); void _mi_arenas_unsafe_destroy_all(mi_subproc_t* subproc); @@ -253,9 +243,9 @@ void _mi_arenas_page_unabandon(mi_page_t* page, mi_theap_t* current_the bool _mi_arenas_page_try_reabandon_to_mapped(mi_page_t* page); // arena-meta.c -void* _mi_meta_zalloc( size_t size, mi_memid_t* memid ); -void _mi_meta_free(void* p, size_t size, mi_memid_t memid); -bool _mi_meta_is_meta_page(void* p); +void* _mi_meta_zalloc( mi_subproc_t* subproc, size_t size, mi_memid_t* memid ); +void _mi_meta_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid); +bool _mi_meta_is_meta_page(mi_subproc_t* subproc, void* p); // "page-map.c" bool _mi_page_map_init(void); @@ -263,7 +253,7 @@ mi_decl_nodiscard bool _mi_page_map_register(mi_page_t* page); void _mi_page_map_unregister(mi_page_t* page); void _mi_page_map_unregister_range(void* start, size_t size); mi_page_t* _mi_safe_ptr_page(const void* p); -void _mi_page_map_unsafe_destroy(mi_subproc_t* subproc); +void _mi_page_map_unsafe_destroy(void); // "page.c" void* _mi_malloc_generic(mi_theap_t* theap, size_t size, size_t zero_huge_alignment, size_t* usable) mi_attr_noexcept mi_attr_malloc; @@ -272,10 +262,7 @@ void _mi_page_retire(mi_page_t* page) mi_attr_noexcept; // free t void _mi_page_unfull(mi_page_t* page); void _mi_page_free(mi_page_t* page, mi_page_queue_t* pq); // free the page void _mi_page_abandon(mi_page_t* page, mi_page_queue_t* pq); // abandon the page, to be picked up by another thread... - -size_t _mi_page_queue_append(mi_theap_t* theap, mi_page_queue_t* pq, mi_page_queue_t* append); void _mi_deferred_free(mi_theap_t* theap, bool force); - void _mi_page_free_collect(mi_page_t* page, bool force); void _mi_page_free_collect_partly(mi_page_t* page, mi_block_t* head); mi_decl_nodiscard bool _mi_page_init(mi_theap_t* theap, mi_page_t* page); @@ -287,7 +274,6 @@ size_t _mi_bin(size_t size); // for stats // "theap.c" mi_theap_t* _mi_theap_create(mi_heap_t* heap, mi_tld_t* tld); -void _mi_theap_delete(mi_theap_t* theap, bool acquire_tld_theaps_lock); void _mi_theap_default_set(mi_theap_t* theap); void _mi_theap_cached_set(mi_theap_t* theap); void _mi_theap_collect_retired(mi_theap_t* theap, bool force); @@ -301,10 +287,11 @@ void _mi_theap_decref(mi_theap_t* theap); // "heap.c" void _mi_heap_area_init(mi_heap_area_t* area, mi_page_t* page); mi_decl_cold mi_theap_t* _mi_heap_theap_get_or_init(const mi_heap_t* heap); // get (and possible create) the theap belonging to a heap -mi_decl_cold mi_theap_t* _mi_heap_theap_get_peek(const mi_heap_t* heap); // get the theap for a heap without initializing (and return NULL in that case) void _mi_heap_move_pages(mi_heap_t* heap_from, mi_heap_t* heap_to); // in "arena.c" void _mi_heap_destroy_pages(mi_heap_t* heap_from); // in "arena.c" -void _mi_heap_force_destroy(mi_heap_t* heap); // allow destroying the main heap +void _mi_heap_force_destroy(mi_heap_t* heap, bool acquire_heaps_lock); // allow destroying the main heap +mi_heap_t* _mi_heap_new_for_subproc(mi_subproc_t* subproc, mi_arena_id_t exclusive_arena_id, bool is_heap_main); +bool _mi_heap_theap_set(mi_heap_t* heap, mi_theap_t* theap); // "stats.c" void _mi_stats_init(void); @@ -322,6 +309,10 @@ void* _mi_theap_realloc_zero(mi_theap_t* theap, void* p, size_t newsize, mi_block_t* _mi_page_ptr_unalign(const mi_page_t* page, const void* p); void _mi_padding_shrink(const mi_page_t* page, const mi_block_t* block, const size_t min_size); +// "free.c" +void _mi_free_subproc_safe(void* p); +void _mi_page_unguard_all(mi_page_t* page); + #if MI_DEBUG>1 bool _mi_page_is_valid(mi_page_t* page); #endif @@ -384,10 +375,6 @@ void __mi_stat_counter_increase_mt(mi_stat_counter_t* stat, size_t amount); #define mi_subproc_stat_adjust_increase(subproc,stat,amount) __mi_stat_adjust_increase_mt( &(subproc)->stats.stat, amount) #define mi_subproc_stat_adjust_decrease(subproc,stat,amount) __mi_stat_adjust_decrease_mt( &(subproc)->stats.stat, amount) -#define mi_os_stat_counter_increase(stat,amount) mi_subproc_stat_counter_increase(_mi_subproc(),stat,amount) -#define mi_os_stat_increase(stat,amount) mi_subproc_stat_increase(_mi_subproc(),stat,amount) -#define mi_os_stat_decrease(stat,amount) mi_subproc_stat_decrease(_mi_subproc(),stat,amount) - #define mi_theap_stat_counter_increase(theap,stat,amount) __mi_stat_counter_increase( &(theap)->stats.stat, amount) #define mi_theap_stat_increase(theap,stat,amount) __mi_stat_increase( &(theap)->stats.stat, amount) #define mi_theap_stat_decrease(theap,stat,amount) __mi_stat_decrease( &(theap)->stats.stat, amount) @@ -395,6 +382,48 @@ void __mi_stat_counter_increase_mt(mi_stat_counter_t* stat, size_t amount); #define mi_theap_stat_adjust_decrease(theap,stat,amnt) __mi_stat_adjust_decrease( &(theap)->stats.stat, amnt) +/* ----------------------------------------------------------- + pthread thread locals +----------------------------------------------------------- */ + +#if MI_USE_PTHREADS + +#if defined(__APPLE__) && defined(__aarch64__) +#define MI_PTHREAD_KEY_INVALID ((pthread_key_t)(0)) // nicer codegen +#else +#define MI_PTHREAD_KEY_INVALID ((pthread_key_t)(-1)) +#endif + +#if defined(__linux__) && defined(__GLIBC__) +// pthread_getspecific returns NULL for invalid keys. +// see also: +#define MI_PTHREADS_GET_INVALID_KEY_IS_NULL 1 +#endif + +mi_decl_noinline bool _mi_pthread_key_create(pthread_key_t* pkey, void (*destruct)(void*), void* init); + +static inline void* mi_pthread_key_get(pthread_key_t key) { + #if !MI_PTHREADS_GET_INVALID_KEY_IS_NULL + if mi_unlikely(key==MI_PTHREAD_KEY_INVALID) return NULL; + #endif + return pthread_getspecific(key); +} + +static inline bool mi_pthread_key_set(pthread_key_t* pkey, void* val) { + if mi_likely(*pkey!=MI_PTHREAD_KEY_INVALID) { pthread_setspecific(*pkey,val); return true; } + else if (val!=NULL) { return _mi_pthread_key_create(pkey,NULL,val); } + else return true; +} + +static inline void mi_pthread_key_delete(pthread_key_t* pkey) { + const pthread_key_t key = *pkey; + if (key!=MI_PTHREAD_KEY_INVALID) { + *pkey = MI_PTHREAD_KEY_INVALID; + pthread_key_delete(key); + } +} +#endif + /* ----------------------------------------------------------- Options (exposed for the debugger) ----------------------------------------------------------- */ @@ -445,6 +474,11 @@ static inline bool _mi_is_power_of_two(uintptr_t x) { return ((x & (x - 1)) == 0); } +// valid alignment values are as posix memalign: +static inline bool mi_alignment_is_valid(size_t alignment) { + return ((alignment!=0) && _mi_is_power_of_two(alignment)); +} + // Is a pointer aligned? static inline bool _mi_is_aligned(const void* p, size_t alignment) { return (alignment==0 || ((uintptr_t)p % alignment) == 0); @@ -519,10 +553,10 @@ static inline size_t _mi_wsize_from_size(size_t size) { #undef _CLOCK_T #endif static inline bool mi_mul_overflow(size_t count, size_t size, size_t* total) { - #if (SIZE_MAX == ULONG_MAX) - return __builtin_umull_overflow(count, size, (unsigned long *)total); - #elif (SIZE_MAX == UINT_MAX) + #if (SIZE_MAX == UINT_MAX) return __builtin_umul_overflow(count, size, (unsigned int *)total); + #elif (SIZE_MAX == ULONG_MAX) + return __builtin_umull_overflow(count, size, (unsigned long *)total); #else return __builtin_umulll_overflow(count, size, (unsigned long long *)total); #endif @@ -562,16 +596,28 @@ static inline bool mi_count_size_overflow(size_t count, size_t size, size_t* tot Heap functions ------------------------------------------------------------------------------------------- */ -extern mi_decl_hidden const mi_theap_t _mi_theap_empty; // read-only empty theap, initial value of the thread local default theap (in the MI_TLS_MODEL_THREAD_LOCAL) +extern mi_decl_hidden const mi_theap_t _mi_theap_empty; // read-only empty theap, initial value of the thread local default theap (in the MI_TLS_MODEL_LOCAL) extern mi_decl_hidden const mi_theap_t _mi_theap_empty_wrong; // read-only empty theap used to signal that a theap for a heap could not be allocated +static inline mi_heap_t* _mi_theap_heap_peek(const mi_theap_t* theap) { + return mi_atomic_load_ptr_relaxed(mi_heap_t,&theap->heap); +} + static inline mi_heap_t* _mi_theap_heap(const mi_theap_t* theap) { - return mi_atomic_load_ptr_acquire(mi_heap_t,&theap->heap); + mi_heap_t* const heap = _mi_theap_heap_peek(theap); + mi_assert_internal(heap!=NULL); + return heap; } static inline bool mi_theap_is_initialized(const mi_theap_t* theap) { - return (theap != NULL && _mi_theap_heap(theap) != NULL); + return (theap != NULL && _mi_theap_heap_peek(theap) != NULL); +} + +static inline mi_subproc_t* _mi_theap_subproc(const mi_theap_t* theap) { + mi_subproc_t* const subproc = mi_atomic_load_ptr_relaxed(mi_subproc_t,&theap->subproc); + mi_assert_internal(!mi_theap_is_initialized(theap) || _mi_theap_heap(theap)->subproc == subproc); + return subproc; } static inline mi_page_t* _mi_theap_get_free_small_page(mi_theap_t* theap, size_t size) { @@ -581,7 +627,6 @@ static inline mi_page_t* _mi_theap_get_free_small_page(mi_theap_t* theap, size_t return theap->pages_free_direct[idx]; } - //static inline uintptr_t _mi_ptr_cookie(const void* p) { // extern mi_theap_t _mi_theap_main; // mi_assert_internal(_mi_theap_main.cookie != 0); @@ -598,20 +643,30 @@ static inline mi_page_t* _mi_theap_get_free_small_page(mi_theap_t* theap, size_t // flat page-map committed on demand, using one byte per slice (64 KiB). // single indirection and low commit, but large initial virtual reserve (4 GiB with 48 bit virtual addresses) // used by default on <= 40 bit virtual address spaces. -extern mi_decl_hidden uint8_t* _mi_page_map; +extern mi_decl_hidden _Atomic(uint8_t*) _mi_page_map; +extern mi_decl_hidden _Atomic(void*) _mi_page_map_max_address; static inline size_t _mi_page_map_index(const void* p) { return (size_t)((uintptr_t)p >> MI_ARENA_SLICE_SHIFT); } +static inline uint8_t _mi_page_map_at(size_t idx) { + return mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map)[idx]; +} + static inline mi_page_t* _mi_ptr_page_ex(const void* p, bool* valid) { const size_t idx = _mi_page_map_index(p); - const size_t ofs = _mi_page_map[idx]; + const size_t ofs = _mi_page_map_at(idx); if (valid != NULL) { *valid = (ofs != 0); } return (mi_page_t*)((((uintptr_t)p >> MI_ARENA_SLICE_SHIFT) + 1 - ofs) << MI_ARENA_SLICE_SHIFT); } static inline mi_page_t* _mi_checked_ptr_page(const void* p) { + #if MI_MIN_VABITS < MI_INTPTR_BITS + if mi_unlikely(((uintptr_t)p >> MI_MIN_VABITS) != 0) { + if (p > mi_atomic_load_ptr_relaxed(void, &_mi_page_map_max_address)) return NULL; + } + #endif bool valid; mi_page_t* const page = _mi_ptr_page_ex(p, &valid); return (valid ? page : NULL); @@ -637,6 +692,7 @@ static inline mi_page_t* _mi_unchecked_ptr_page(const void* p) { typedef mi_page_t** mi_submap_t; extern mi_decl_hidden _Atomic(mi_submap_t)* _mi_page_map; +extern mi_decl_hidden _Atomic(void*) _mi_page_map_max_address; static inline size_t _mi_page_map_index(const void* p, size_t* sub_idx) { const size_t u = (size_t)((uintptr_t)p / MI_ARENA_SLICE_SIZE); @@ -655,6 +711,11 @@ static inline mi_page_t* _mi_unchecked_ptr_page(const void* p) { } static inline mi_page_t* _mi_checked_ptr_page(const void* p) { + #if MI_MIN_VABITS < MI_INTPTR_BITS + if mi_unlikely(((uintptr_t)p >> MI_MIN_VABITS) != 0) { + if (p > mi_atomic_load_ptr_relaxed(void, &_mi_page_map_max_address)) return NULL; + } + #endif size_t sub_idx; const size_t idx = _mi_page_map_index(p, &sub_idx); mi_submap_t const sub = _mi_page_map_at(idx); @@ -664,10 +725,9 @@ static inline mi_page_t* _mi_checked_ptr_page(const void* p) { #endif - static inline mi_page_t* _mi_ptr_page(const void* p) { mi_assert_internal(p==NULL || mi_is_in_heap_region(p)); - #if MI_DEBUG || MI_SECURE || defined(__APPLE__) + #if MI_DEBUG || MI_SECURE || MI_FREE_IS_CHECKED return _mi_checked_ptr_page(p); #else return _mi_unchecked_ptr_page(p); @@ -683,7 +743,8 @@ static inline size_t mi_page_block_size(const mi_page_t* page) { // Page start static inline uint8_t* mi_page_start(const mi_page_t* page) { - return page->page_start; + // multiplication must be done in `size_t`; in a 32-bit multiplication the offset wraps for pages whose blocks start 4 GiB or more after the page meta info + return (uint8_t*)page + (((size_t)page->page_ma_offset) * MI_MAX_ALIGN_SIZE); } static inline size_t mi_page_size(const mi_page_t* page) { @@ -722,17 +783,17 @@ static inline size_t mi_page_usable_block_size(const mi_page_t* page) { static inline bool mi_page_meta_is_separated(const mi_page_t* page) { #if MI_PAGE_META_IS_SEPARATED // usually separated but can still be in front for direct OS allocations (due to size or alignment) or due to MI_PAGE_META_ALIGNED_FREE_SMALL - return (page->memid.memkind == MI_MEM_ARENA && page != _mi_align_down_ptr(page->page_start, MI_ARENA_SLICE_ALIGN)); + return (page->memid.memkind == MI_MEM_ARENA && page != _mi_align_down_ptr(mi_page_start(page), MI_ARENA_SLICE_ALIGN)); #else MI_UNUSED(page); - return false; + return false; #endif } static inline uint8_t* mi_page_slice_start(const mi_page_t* page) { - if (mi_page_meta_is_separated(page)) { + if (mi_page_meta_is_separated(page)) { // page meta info is at a separate location (at `arena->pages`) - return (uint8_t*)_mi_align_down_ptr(page->page_start, MI_ARENA_SLICE_ALIGN); + return (uint8_t*)_mi_align_down_ptr(mi_page_start(page), MI_ARENA_SLICE_ALIGN); } else { // page meta info is at the start of the page slices @@ -740,9 +801,9 @@ static inline uint8_t* mi_page_slice_start(const mi_page_t* page) { } } -// This gives the offset relative to the start slice of a page. +// This gives the offset relative to the start slice of a page. static inline size_t mi_page_slice_offset_of(const mi_page_t* page, size_t offset_relative_to_page_start) { - return (page->page_start - mi_page_slice_start(page)) + offset_relative_to_page_start; + return (mi_page_start(page) - mi_page_slice_start(page)) + offset_relative_to_page_start; } // Currently committed part of a page @@ -842,7 +903,8 @@ static inline void mi_page_set_in_full(mi_page_t* page, bool in_full) { mi_theap_t* const theap = page->theap; mi_assert_internal(theap!=NULL); if (theap != NULL) { - const size_t size = page->capacity * mi_page_block_size(page); + mi_assert_internal(page->capacity==page->reserved); + const size_t size = page->reserved * mi_page_block_size(page); if (in_full) { theap->pages_full_size += size; } else { mi_assert_internal(size <= theap->pages_full_size); theap->pages_full_size -= size; } } @@ -893,7 +955,7 @@ static inline void mi_page_clear_abandoned_mapped(mi_page_t* page) { static inline mi_theap_t* mi_page_theap(const mi_page_t* page) { mi_assert_internal(!mi_page_is_abandoned(page)); - mi_assert_internal(page->theap != NULL); + mi_assert_internal(page->theap != NULL && page->theap != &_mi_theap_empty); return page->theap; } @@ -906,12 +968,30 @@ static inline mi_tld_t* mi_page_tld(const mi_page_t* page) { static inline mi_heap_t* mi_page_heap(const mi_page_t* page) { mi_heap_t* heap = page->heap; - // we use NULL for the main heap to make `_mi_page_get_associated_theap` fast in `free.c:mi_abandoned_page_try_reclaim`. - if mi_likely(heap==NULL) heap = mi_heap_main(); mi_assert_internal(heap != NULL); return heap; } +static inline mi_subproc_t* mi_page_subproc(const mi_page_t* page) { + mi_heap_t* const heap = mi_page_heap(page); + return heap->subproc; +} + +static inline mi_heap_t* mi_arena_heap_main(const mi_arena_t* arena) { + return _mi_subproc_heap_main(arena->subproc); +} + +static inline mi_heap_t* mi_heap_get_heap_main(const mi_heap_t* heap) { + return _mi_subproc_heap_main(heap->subproc); +} + +static inline bool _mi_is_heap_main(const mi_heap_t* heap) { + mi_assert_internal(heap!=NULL); + return (mi_heap_get_heap_main(heap) == heap); +} + + + //----------------------------------------------------------- // Thread free list and ownership //----------------------------------------------------------- @@ -967,7 +1047,7 @@ static inline bool mi_block_ptr_is_guarded(const mi_block_t* block, const void* #else MI_UNUSED(block); MI_UNUSED(p); return false; -#endif +#endif } #if MI_GUARDED @@ -978,24 +1058,29 @@ static inline bool mi_theap_malloc_use_guarded(mi_theap_t* theap, size_t size) { // no sample theap->guarded_sample_count = count; return false; - } - else if (size >= theap->guarded_size_min && size <= theap->guarded_size_max) { - // use guarded allocation - theap->guarded_sample_count = theap->guarded_sample_rate; // reset - return (theap->guarded_sample_rate != 0); - } - else { - // failed size criteria, rewind count (but don't write to an empty theap) - if (theap->guarded_sample_rate != 0) { theap->guarded_sample_count = 1; } - return false; + } + else { + // count == 0 + const size_t rate = theap->guarded_sample_rate; + if (rate == 0) { + return false; // don't write to an empty theap + } + else if (size >= theap->guarded_size_min && size <= theap->guarded_size_max) { + // use guarded allocation + theap->guarded_sample_count = rate; // reset + return true; + } + else { + // failed size criteria, rewind count + theap->guarded_sample_count = 1; + return false; + } } } -mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, bool zero) mi_attr_noexcept; - +mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, bool zero, size_t* usable) mi_attr_noexcept; #endif - /* ------------------------------------------------------------------- Encoding/Decoding the free list next pointers @@ -1075,7 +1160,7 @@ static inline mi_block_t* mi_block_next(const mi_page_t* page, const mi_block_t* #if MI_ENCODE_FREELIST mi_block_t* next = mi_block_nextx(page,block,page->keys); // check for free list corruption: is `next` at least in the same page? - // TODO: check if `next` is `page->block_size` aligned? + // todo: check if `next` is `page->block_size` aligned? if mi_unlikely(next!=NULL && !mi_is_in_same_page(block, next)) { _mi_error_message(EFAULT, "corrupted free list entry of size %zub at %p: value 0x%zx\n", mi_page_block_size(page), block, (uintptr_t)next); next = NULL; @@ -1096,6 +1181,8 @@ static inline void mi_block_set_next(const mi_page_t* page, mi_block_t* block, c #endif } + + /* ----------------------------------------------------------- arena blocks ----------------------------------------------------------- */ @@ -1136,7 +1223,7 @@ static inline mi_memid_t _mi_memid_create_os(void* base, size_t size, bool commi return memid; } -static inline mi_memid_t _mi_memid_create_meta(void* mpage, size_t block_idx, size_t block_count) { +static inline mi_memid_t _mi_memid_create_meta(mi_meta_page_t* mpage, size_t block_idx, size_t block_count) { mi_memid_t memid = _mi_memid_create(MI_MEM_META); memid.mem.meta.meta_page = mpage; memid.mem.meta.block_index = (uint32_t)block_idx; diff --git a/vendored/mimalloc/include/mimalloc/prim-tls.h b/vendored/mimalloc/include/mimalloc/prim-tls.h new file mode 100644 index 0000000000..9e689e866d --- /dev/null +++ b/vendored/mimalloc/include/mimalloc/prim-tls.h @@ -0,0 +1,483 @@ +/* ---------------------------------------------------------------------------- +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen +This is free software; you can redistribute it and/or modify it under the +terms of the MIT license. A copy of the license can be found in the file +"LICENSE" at the root of this distribution. +-----------------------------------------------------------------------------*/ +#pragma once +#ifndef MIMALLOC_PRIM_TLS_H +#define MIMALLOC_PRIM_TLS_H + +#include "types.h" +#include "internal.h" // mi_decl_hidden + +// -------------------------------------------------------------------------- +// We need fast access to both a unique thread id (in `free.c:mi_free`) and +// to a thread-local theap pointer (in `alloc.c:mi_malloc`). +// +// For performance, we tend to use specialized code for various platforms. +// This leads to quite a few ifdefs but it is just for performance and there +// is always a portable fallback (based on regular thread local variables). +// +// Windows : use NtCurrentTeB and TlsAlloc (MI_TLS_MODEL_WIN32) +// Linux,FreeBSD : use thread locals with the initial-exec model (MI_TLS_MODEL_LOCAL) +// macOS : use pthread locals with assembly for the thread-id (MI_TLS_MODEL_PTHREADS) +// Android,OpenBSD : use pthread locals (MI_TLS_MODEL_PTHREADS). todo: maybe on Android MI_TLS_MODEL_LOCAL is better? +// -------------------------------------------------------------------------- + +// static inline void* mi_prim_tls_slot(size_t slot) mi_attr_noexcept; // directly read an entry from the thread local storage (or thread control block) +// static inline void mi_prim_tls_slot_set(size_t slot, void* value) mi_attr_noexcept; + +static inline mi_threadid_t _mi_prim_thread_id(void) mi_attr_noexcept; // get a unique id for a thread +static inline mi_theap_t* _mi_theap_default(void); // the default thread local theap +static inline mi_theap_t* _mi_theap_cached(void); // last used thread local theap using the _heap_ api +static inline bool _mi_thread_is_initialized(void); // a thread is initialized if it has a default theap +static inline mi_theap_t* _mi_heap_theap(mi_heap_t* heap); // get the thread local theap belonging to a heap +static inline mi_theap_t* _mi_heap_theap_peek(const mi_heap_t* heap); // get the theap but don't update _mi_theap_cached +static inline mi_theap_t* _mi_page_associated_theap_peek(mi_page_t* page); // get the theap associated with a page (used in `mi_free_collect_mt`) + + +// Default TLS model +#if !defined(MI_TLS_MODEL_LOCAL) && !defined(MI_TLS_MODEL_PTHREADS) && !defined(MI_TLS_MODEL_FIXED) && !defined(MI_TLS_MODEL_WIN32) +#if defined(_WIN32) +#define MI_TLS_MODEL_WIN32 1 +#elif defined(__APPLE__) || defined(__OpenBSD__) || defined(__ANDROID__) // and FreeBSD? +#define MI_TLS_MODEL_PTHREADS 1 +#else +#define MI_TLS_MODEL_LOCAL 1 +#endif +#endif + + +//------------------------------------------------------------------- +// Access to TLS (thread local storage) slots. +//------------------------------------------------------------------- + +// On some libc + platform combinations we can directly access a thread-local storage (TLS) slot. +// The TLS layout depends on both the OS and libc implementation so we use specific tests for each main platform. +// If you test on another platform and it works please send a PR :-) +// see also https://akkadia.org/drepper/tls.pdf for more info on the TLS register. +// +// Note: we would like to prefer `__builtin_thread_pointer()` nowadays instead of using assembly, +// but unfortunately we can not detect support reliably (see issue #883) +#if (defined(_WIN32)) || \ + (defined(__GNUC__) && ( \ + (defined(__GLIBC__) && (defined(__x86_64__) || defined(__i386__) || (defined(__arm__) && __ARM_ARCH >= 7) || defined(__aarch64__) || defined(__riscv))) \ + || (defined(__APPLE__) && (defined(__x86_64__) || defined(__aarch64__) || defined(__POWERPC__))) \ + || (defined(__BIONIC__) && (defined(__x86_64__) || defined(__i386__) || (defined(__arm__) && __ARM_ARCH >= 7) || defined(__aarch64__))) \ + || (defined(__FreeBSD__) && (defined(__x86_64__) || defined(__i386__) || defined(__aarch64__))) \ + || (defined(__OpenBSD__) && (defined(__x86_64__) || defined(__i386__) || defined(__aarch64__))) \ + )) + +static inline void* mi_prim_tls_slot(size_t slot) mi_attr_noexcept { + void* res; + const size_t ofs = (slot*sizeof(void*)); + #if defined(_WIN32) + #if (_M_X64 || _M_AMD64) && !defined(_M_ARM64EC) + res = (void*)__readgsqword((unsigned long)ofs); // direct load at offset from gs + #elif _M_IX86 && !defined(_M_ARM64EC) + res = (void*)__readfsdword((unsigned long)ofs); // direct load at offset from fs + #else + res = ((void**)NtCurrentTeb())[slot]; MI_UNUSED(ofs); + #endif + #elif defined(__i386__) + __asm__("movl %%gs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86 32-bit always uses GS + #elif defined(__APPLE__) && defined(__x86_64__) + __asm__("movq %%gs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86_64 macOSX uses GS + #elif defined(__x86_64__) && (MI_INTPTR_SIZE==4) + __asm__("movl %%fs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x32 ABI + #elif defined(__x86_64__) + __asm__("movq %%fs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86_64 Linux, BSD uses FS + #elif defined(__arm__) + void** tcb; MI_UNUSED(ofs); + __asm__ volatile ("mrc p15, 0, %0, c13, c0, 3\nbic %0, %0, #3" : "=r" (tcb)); + res = tcb[slot]; + #elif defined(__aarch64__) + void** tcb; MI_UNUSED(ofs); + #if defined(__APPLE__) // M1, issue #343 + __asm__ volatile ("mrs %0, tpidrro_el0\nbic %0, %0, #7" : "=r" (tcb)); + #else + __asm__ volatile ("mrs %0, tpidr_el0" : "=r" (tcb)); + #endif + res = tcb[slot]; + #elif defined(__riscv) + void** tcb; MI_UNUSED(ofs); + __asm__ volatile ("mv %0, tp" : "=r" (tcb)); + res = tcb[slot]; + #elif defined(__APPLE__) && defined(__POWERPC__) // ppc, issue #781 + MI_UNUSED(ofs); + res = pthread_getspecific(slot); + #else + #define MI_HAS_TLS_SLOT 0 + MI_UNUSED(ofs); + res = NULL; + #endif + return res; +} + +#ifndef MI_HAS_TLS_SLOT +#define MI_HAS_TLS_SLOT 1 +#endif + +// setting a tls slot is only used with TLS_MODEL_FIXED (which is not used by default on any platform) +static inline void mi_prim_tls_slot_set(size_t slot, void* value) mi_attr_noexcept { + const size_t ofs = (slot*sizeof(void*)); + #if defined(_WIN32) + ((void**)NtCurrentTeb())[slot] = value; MI_UNUSED(ofs); + #elif defined(__i386__) + __asm__("movl %1,%%gs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // 32-bit always uses GS + #elif defined(__APPLE__) && defined(__x86_64__) + __asm__("movq %1,%%gs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x86_64 macOS uses GS + #elif defined(__x86_64__) && (MI_INTPTR_SIZE==4) + __asm__("movl %1,%%fs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x32 ABI + #elif defined(__x86_64__) + __asm__("movq %1,%%fs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x86_64 Linux, BSD uses FS + #elif defined(__arm__) + void** tcb; MI_UNUSED(ofs); + __asm__ volatile ("mrc p15, 0, %0, c13, c0, 3\nbic %0, %0, #3" : "=r" (tcb)); + tcb[slot] = value; + #elif defined(__aarch64__) + void** tcb; MI_UNUSED(ofs); + #if defined(__APPLE__) // M1, issue #343 + __asm__ volatile ("mrs %0, tpidrro_el0\nbic %0, %0, #7" : "=r" (tcb)); + #else + __asm__ volatile ("mrs %0, tpidr_el0" : "=r" (tcb)); + #endif + tcb[slot] = value; + #elif defined(__riscv) + void** tcb; MI_UNUSED(ofs); + __asm__ volatile ("mv %0, tp" : "=r" (tcb)); + tcb[slot] = value; + #elif defined(__APPLE__) && defined(__POWERPC__) // ppc, issue #781 + MI_UNUSED(ofs); + pthread_setspecific(slot, value); + #else + MI_UNUSED(ofs); MI_UNUSED(value); + #endif +} + +#endif + + +//------------------------------------------------------------------- +// Get a fast unique thread id. +// +// Getting the thread id should be performant as it is called in the +// fast path of `_mi_free` and we specialize for various platforms as +// inlined definitions. Regular code should call `init.c:_mi_thread_id()`. +// We only require _mi_prim_thread_id() to return a unique id +// for each thread (unequal to zero) with the bottom 2 bits clear. +//------------------------------------------------------------------- + +// Do we have __builtin_thread_pointer? This would be the preferred way to get a unique thread id +// but unfortunately, it seems we cannot test for this reliably at this time (see issue #883) +// Nevertheless, it seems needed on older graviton platforms (see issue #851). +// For now, we only enable this for specific platforms. +#if !defined(MI_USE_BUILTIN_THREAD_POINTER) /* allow user override */ + #if !defined(__APPLE__) /* on apple (M1) the wrong register is read (tpidr_el0 instead of tpidrro_el0) so fall back to TLS slot assembly ()*/ \ + && !defined(__CYGWIN__) \ + && !defined(MI_LIBC_MUSL) \ + && (!defined(__clang_major__) || __clang_major__ >= 14) /* older clang versions emit bad code; fall back to using the TLS slot () */ + #if (defined(__GNUC__) && (__GNUC__ >= 7) && defined(__aarch64__)) /* aarch64 for older gcc versions (issue #851) */ \ + || (defined(__GNUC__) && (__GNUC__ >= 7) && defined(__riscv)) \ + || (defined(__GNUC__) && (__GNUC__ >= 11) && defined(__x86_64__)) \ + || (defined(__clang_major__) && (__clang_major__ >= 14) && (defined(__aarch64__) || defined(__x86_64__))) + #define MI_USE_BUILTIN_THREAD_POINTER 1 + #endif + #endif +#endif + +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept; + +static inline mi_threadid_t _mi_prim_thread_id(void) mi_attr_noexcept { + const mi_threadid_t tid = __mi_prim_thread_id(); + mi_assert_internal(tid > 1); + mi_assert_internal((tid & MI_PAGE_FLAG_MASK) == 0); // bottom 2 bits are clear? + return tid; +} + +// Get a unique id for the current thread. +#if defined(MI_PRIM_THREAD_ID) + +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + return MI_PRIM_THREAD_ID(); // used for example by CPython for a free threaded build (see python/cpython#115488) +} + +#elif defined(_WIN32) + +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + // Windows: works on Intel and ARM in both 32- and 64-bit + return (uintptr_t)NtCurrentTeb(); +} + +#elif MI_USE_BUILTIN_THREAD_POINTER + +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + // Works on most Unix based platforms with recent compilers + return (uintptr_t)__builtin_thread_pointer(); +} + +#elif MI_HAS_TLS_SLOT + +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + #if defined(__BIONIC__) + // issue #384, #495: on the Bionic libc (Android), slot 1 is the thread id + // see: https://github.com/aosp-mirror/platform_bionic/blob/c44b1d0676ded732df4b3b21c5f798eacae93228/libc/platform/bionic/tls_defines.h#L86 + return (uintptr_t)mi_prim_tls_slot(1); + #else + // in all our other targets, slot 0 is the thread id + // glibc: https://sourceware.org/git/?p=glibc.git;a=blob_plain;f=sysdeps/x86_64/nptl/tls.h + // apple: https://github.com/apple/darwin-xnu/blob/main/libsyscall/os/tsd.h#L36 + return (uintptr_t)mi_prim_tls_slot(0); + #endif +} + +#elif defined(MI_USE_PTHREADS) && defined(__APPLE__) + +// on macOS, pthread_t is pointer +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + return (uintptr_t)((void*)pthread_self()); +} + +#else + +extern mi_decl_hidden mi_decl_thread void* __mi_thread_id_helper; + +// otherwise use portable C, taking the address of a thread local variable (this is still very fast on most platforms). +static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { + return (uintptr_t)&__mi_thread_id_helper; +} + +#endif + + + +/* ---------------------------------------------------------------------------------------- +Get the thread local default theap: `_mi_theap_default()` (and the cached heap `_mi_theap_cached`). + +This is inlined here as it is on the fast path for allocation functions. +We have 4 models: + +- MI_TLS_MODEL_LOCAL: use regular thread local (default on Linux, FreeBSD, etc) + On most platforms (Linux, FreeBSD, NetBSD, etc), this just returns a + thread local variable (`__mi_theap_default`). With the initial-exec TLS model this ensures + that the storage will always be available and properly initialized (with an empty theap). + + On some platforms the underlying TLS implementation (or the loader) will call itself `malloc` + on a first access to a thread local and recurse in the MI_TLS_MODEL_LOCAL. + A way around this is to define MI_TLS_RECURSE_GUARD which adds an extra check if the process + is initialized before accessing the thread-local. This is a check in the fast path though + so this should be avoided. + +- MI_TLS_MODEL_PTHREADS: use `pthread_getspecific`. (default on macOS and OpenBSD, maybe good for Android as well?) + Use pthread local storage. Can be as fast as thread locals on many platforms (like recent macOS). + +- MI_TLS_MODEL_FIXED: use a fixed slot in the TLS block. + This reserves an unused and fixed TLS slot. This is fast and avoids the problem + where the underlying TLS implementation (or the loader) will call itself `malloc` + on a first access to a thread local (and recurse in the MI_TLS_MODEL_LOCAL). + This goes wrong though if the OS or a library uses the same fixed slot, and also + prevents multiple instances of mimalloc in the same process. + +- MI_TLS_MODEL_WIN32: use a dynamically allocated slot with TlsAlloc. (default on Windows) + We use TlsAlloc'd slot. First tries to use one of the "direct" first 64 slots which + are the fastest, but falls back to using "expansion" slots when needed (up to 1088 slots). + (If the allocated slot happens to always be under 64 for a particular program, + one might use cmake with `-DMI_WIN_DIRECT_TLS=ON` to skip the expansion slot test in the fast path.) + +Each model should define `MI_THEAP_INITASNULL` to signify that the initial value +returned from `_mi_theap_default()` can be `NULL` (instead of the address of the empty heap). +This incurs an extra check in the fast path (but can often be combined in an existing check). +------------------------------------------------------------------------------------------- */ + +#if !defined(MI_TLS_RECURSE_GUARD) && MI_TLS_MODEL_LOCAL && defined(__APPLE__) +#define MI_TLS_RECURSE_GUARD 1 // macOS can allocate on thread-local initialization +#endif + +// Declared this way to optimize register spills and branches +mi_decl_cold mi_decl_noinline mi_theap_t* _mi_theap_empty_get(void); + +static inline mi_theap_t* __mi_theap_empty(void) { + #if __GNUC__ + __asm(""); // prevent conditional load + return (mi_theap_t*)&_mi_theap_empty; + #else + return _mi_theap_empty_get(); + #endif +} + +#if MI_TLS_MODEL_LOCAL +// Thread local with an initial value (default on Linux). Very efficient. +extern mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_default; // default theap to allocate from +extern mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_cached; // theap from the last used heap + +// defined in `init.c`; do not use these directly +extern mi_decl_hidden bool _mi_process_is_initialized; // has mi_process_init been called? + +static inline mi_theap_t* _mi_theap_default(void) { + #if defined(MI_TLS_RECURSE_GUARD) + if mi_unlikely(!_mi_process_is_initialized) return _mi_theap_empty_get(); + #endif + return __mi_theap_default; +} + +static inline mi_theap_t* _mi_theap_cached(void) { + return __mi_theap_cached; +} + +#elif MI_TLS_MODEL_PTHREADS +// Dynamic pthread slots. This can be fast depending on the platform (default for macOS and OpenBSD) +// On some platforms (like macOS), the loader might allocate on thread local declarations which +// can be avoided with pthreads. +#define MI_THEAP_INITASNULL 1 + +extern mi_decl_hidden pthread_key_t _mi_theap_default_key; +extern mi_decl_hidden pthread_key_t _mi_theap_cached_key; + +static inline mi_theap_t* _mi_theap_default(void) { + #if defined(__APPLE__) && defined(__aarch64__) && MI_HAS_TLS_SLOT + // on apple arm64, the pthread specific slots are direct slots; inline it to avoid a stack frame setup in `mi_malloc` + // todo: this is probably also the case on x64 and power pc? + if (_mi_theap_default_key == MI_PTHREAD_KEY_INVALID) return NULL; + return (mi_theap_t*)mi_prim_tls_slot(_mi_theap_default_key); + #else + return (mi_theap_t*)mi_pthread_key_get(_mi_theap_default_key); + #endif +} + +static inline mi_theap_t* _mi_theap_cached(void) { + #if defined(__APPLE__) && defined(__aarch64__) && MI_HAS_TLS_SLOT + if (_mi_theap_cached_key == MI_PTHREAD_KEY_INVALID) return NULL; + return (mi_theap_t*)mi_prim_tls_slot(_mi_theap_cached_key); + #else + return (mi_theap_t*)mi_pthread_key_get(_mi_theap_cached_key); + #endif +} + +#elif MI_TLS_MODEL_WIN32 +// Dynamic TLS slots -- this is the default on Windows. +#define MI_THEAP_INITASNULL 1 + +// We try to use direct slots (64 available), but can also use the expansion slots (upto 1024 extra available) +// See for the offsets. +#if MI_SIZE_SIZE==4 +#define MI_TLS_EXPANSION_SLOT (0x0F94 / MI_INTPTR_SIZE) +#else +#define MI_TLS_EXPANSION_SLOT (0x1780 / MI_INTPTR_SIZE) +#endif + +extern mi_decl_hidden _Atomic(size_t) _mi_theap_default_slot; +extern mi_decl_hidden _Atomic(size_t) _mi_theap_cached_slot; +extern mi_decl_hidden _Atomic(size_t) _mi_theap_default_expansion_slot; +extern mi_decl_hidden _Atomic(size_t) _mi_theap_cached_expansion_slot; + +static inline mi_theap_t* _mi_theap_default(void) { + const size_t slot = mi_atomic_load_relaxed(&_mi_theap_default_slot); + mi_theap_t* theap = (mi_theap_t*)mi_prim_tls_slot(slot); + #if !MI_WIN_DIRECT_TLS + if mi_unlikely(slot==MI_TLS_EXPANSION_SLOT) { // in TlsExpansionSlots ? + mi_theap_t** const eslots = (mi_theap_t**)theap; // theap is actually the expansion slot entry + if mi_likely(eslots!=NULL) { // is it initialized? (on this thread) + theap = eslots[mi_atomic_load_relaxed(&_mi_theap_default_expansion_slot)]; + } + } + #endif + return theap; +} + +static inline mi_theap_t* _mi_theap_cached(void) { + const size_t slot = mi_atomic_load_relaxed(&_mi_theap_cached_slot); + mi_theap_t* theap = (mi_theap_t*)mi_prim_tls_slot(slot); + #if !MI_WIN_DIRECT_TLS + if mi_unlikely(slot==MI_TLS_EXPANSION_SLOT) { // in TlsExpansionSlots ? + mi_theap_t** const eslots = (mi_theap_t**)theap; // theap is the expansion slot entry + if mi_likely(eslots!=NULL) { // is it initialized? (on this thread) + theap = eslots[mi_atomic_load_relaxed(&_mi_theap_cached_expansion_slot)]; + } + } + #endif + return theap; +} + +#elif MI_TLS_MODEL_FIXED +// Fixed TLS slot. Can be the fastest approach, but does not work if there are multiple instances of +// mimalloc in the same process. Most OS's do not have official user reserved fixed slots so this cannot be +// guaranteed to work in general. +#define MI_THEAP_INITASNULL 1 + +#if !MI_HAS_TLS_SLOT +#error this platform cannot support MI_TLS_MODEL_FIXED without defining mi_prim_tls_slot +#endif + +#if !defined(MI_TLS_MODEL_FIXED_DEFAULT) + #if defined(__APPLE__) && !defined(__POWERPC__) // macOS on arm64 or x64 + // we use the last two swift framework slots which seem unused. + // we may want to use slot 6 and 11 instead which are only used by Windows emulation. + // see for assigned slots + #define MI_TLS_MODEL_FIXED_DEFAULT 108 + #define MI_TLS_MODEL_FIXED_CACHED 109 + #elif defined(_WIN32) + // we use two seemingly unused fields in the Windows TEB. + // see + #define MI_TLS_MODEL_FIXED_DEFAULT 5 // arbitrary user pointer + #define MI_TLS_MODEL_FIXED_CACHED 7 // environment pointer (used by OS2) + #else + #error define the TLS model fixed slots (or change the TLS model away from MI_TLS_MODEL_FIXED) + #endif +#endif + +static inline mi_theap_t* _mi_theap_default(void) { + return (mi_theap_t*)mi_prim_tls_slot(MI_TLS_MODEL_FIXED_DEFAULT); +} + +static inline mi_theap_t* _mi_theap_cached(void) { + return (mi_theap_t*)mi_prim_tls_slot(MI_TLS_MODEL_FIXED_CACHED); +} + +#else +#error "no TLS model is defined for this platform?" +#endif + + +// Check if a thread is initialized (without using a thread-local if using fixed slots) +static inline bool _mi_thread_is_initialized(void) { + return mi_theap_is_initialized(_mi_theap_default()); +} + +// Get (and possible create) the theap belonging to a heap +// We cache the last accessed theap in `_mi_theap_cached` for better performance. +static inline mi_theap_t* _mi_heap_theap(mi_heap_t* heap) { + mi_theap_t* theap = _mi_theap_cached(); + #if MI_THEAP_INITASNULL + if mi_likely(theap!=NULL && _mi_theap_heap_peek(theap)==heap) return theap; + #else + if mi_likely(_mi_theap_heap_peek(theap)==heap) return theap; + #endif + return _mi_heap_theap_get_or_init(heap); +} + +// Get the theap belonging to a heap without creating it if it is not yet initialized. +static inline mi_theap_t* _mi_heap_theap_peek(const mi_heap_t* heap) { + mi_theap_t* theap = _mi_theap_cached(); + #if MI_THEAP_INITASNULL + if mi_likely(theap!=NULL && _mi_theap_heap_peek(theap)==heap) return theap; + #else + if mi_likely(_mi_theap_heap_peek(theap)==heap) return theap; + #endif + theap = (mi_theap_t*)_mi_thread_local_get(heap->theap); // don't update the cache on a query + mi_assert_internal(theap==NULL || (!_mi_is_empty_theap(theap) && theap->heap==heap)); + return theap; +} + +// Find the associated theap or NULL if it does not exist (during shutdown) +// Should be fast as it is called in `free.c:mi_free_try_collect`. +static inline mi_theap_t* _mi_page_associated_theap_peek(mi_page_t* page) { + mi_heap_t* const heap = mi_page_heap(page); + mi_theap_t* const theap = (mi_theap_t*)_mi_thread_local_get(heap->theap); + if (theap==NULL) return NULL; + if (theap->heap != heap) return NULL; // should never happen, but can happen for a free across subprocesses, which can happen during pthread tls storage deallocation + mi_assert_internal(!_mi_is_empty_theap(theap) && _mi_thread_id()==theap->tld->thread_id); + return theap; +} + +#endif // MI_PRIM_TLS_H diff --git a/vendored/mimalloc/include/mimalloc/prim.h b/vendored/mimalloc/include/mimalloc/prim.h index aecfc80742..0f442051da 100644 --- a/vendored/mimalloc/include/mimalloc/prim.h +++ b/vendored/mimalloc/include/mimalloc/prim.h @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -7,7 +7,8 @@ terms of the MIT license. A copy of the license can be found in the file #pragma once #ifndef MIMALLOC_PRIM_H #define MIMALLOC_PRIM_H -#include "internal.h" // mi_decl_hidden + +#include "types.h" // -------------------------------------------------------------------------- // This file specifies the primitive portability API. @@ -107,7 +108,9 @@ void _mi_prim_out_stderr( const char* msg ); // Get an environment variable. (only for options) // name != NULL, result != NULL, result_size >= 64 -bool _mi_prim_getenv(const char* name, char* result, size_t result_size); +// Return 1 for success, 0 if not found, +// and -1 on error (for example, if `getenv` cannot be called yet during preloading). +int _mi_prim_getenv(const char* name, char* result, size_t result_size); // Fill a buffer with strong randomness; return `false` on error or if @@ -126,417 +129,8 @@ void _mi_prim_thread_associate_default_theap(mi_theap_t* theap); // Is this thread part of a thread pool? bool _mi_prim_thread_is_in_threadpool(void); -// Yield to other threads. Should be similar to `sleep(0)`. +// Yield to other threads. Should be similar to `sleep(0)`. // Is called only in rare situations and does not have to be lightning fast. void _mi_prim_thread_yield(void); -//------------------------------------------------------------------- -// Access to TLS (thread local storage) slots. -// We need fast access to both a unique thread id (in `free.c:mi_free`) and -// to a thread-local theap pointer (in `alloc.c:mi_malloc`). -// To achieve this we use specialized code for various platforms. -//------------------------------------------------------------------- - -// On some libc + platform combinations we can directly access a thread-local storage (TLS) slot. -// The TLS layout depends on both the OS and libc implementation so we use specific tests for each main platform. -// If you test on another platform and it works please send a PR :-) -// see also https://akkadia.org/drepper/tls.pdf for more info on the TLS register. -// -// Note: we would like to prefer `__builtin_thread_pointer()` nowadays instead of using assembly, -// but unfortunately we can not detect support reliably (see issue #883) -// We also use it on Apple OS as we use a TLS slot for the default theap there. -#if (defined(_WIN32)) || \ - (defined(__GNUC__) && ( \ - (defined(__GLIBC__) && (defined(__x86_64__) || defined(__i386__) || (defined(__arm__) && __ARM_ARCH >= 7) || defined(__aarch64__))) \ - || (defined(__APPLE__) && (defined(__x86_64__) || defined(__aarch64__) || defined(__POWERPC__))) \ - || (defined(__BIONIC__) && (defined(__x86_64__) || defined(__i386__) || (defined(__arm__) && __ARM_ARCH >= 7) || defined(__aarch64__))) \ - || (defined(__FreeBSD__) && (defined(__x86_64__) || defined(__i386__) || defined(__aarch64__))) \ - || (defined(__OpenBSD__) && (defined(__x86_64__) || defined(__i386__) || defined(__aarch64__))) \ - )) - -static inline void* mi_prim_tls_slot(size_t slot) mi_attr_noexcept { - void* res; - const size_t ofs = (slot*sizeof(void*)); - #if defined(_WIN32) - #if (_M_X64 || _M_AMD64) && !defined(_M_ARM64EC) - res = (void*)__readgsqword((unsigned long)ofs); // direct load at offset from gs - #elif _M_IX86 && !defined(_M_ARM64EC) - res = (void*)__readfsdword((unsigned long)ofs); // direct load at offset from fs - #else - res = ((void**)NtCurrentTeb())[slot]; MI_UNUSED(ofs); - #endif - #elif defined(__i386__) - __asm__("movl %%gs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86 32-bit always uses GS - #elif defined(__APPLE__) && defined(__x86_64__) - __asm__("movq %%gs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86_64 macOSX uses GS - #elif defined(__x86_64__) && (MI_INTPTR_SIZE==4) - __asm__("movl %%fs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x32 ABI - #elif defined(__x86_64__) - __asm__("movq %%fs:%1, %0" : "=r" (res) : "m" (*((void**)ofs)) : ); // x86_64 Linux, BSD uses FS - #elif defined(__arm__) - void** tcb; MI_UNUSED(ofs); - __asm__ volatile ("mrc p15, 0, %0, c13, c0, 3\nbic %0, %0, #3" : "=r" (tcb)); - res = tcb[slot]; - #elif defined(__aarch64__) - void** tcb; MI_UNUSED(ofs); - #if defined(__APPLE__) // M1, issue #343 - __asm__ volatile ("mrs %0, tpidrro_el0\nbic %0, %0, #7" : "=r" (tcb)); - #else - __asm__ volatile ("mrs %0, tpidr_el0" : "=r" (tcb)); - #endif - res = tcb[slot]; - #elif defined(__APPLE__) && defined(__POWERPC__) // ppc, issue #781 - MI_UNUSED(ofs); - res = pthread_getspecific(slot); - #else - #define MI_HAS_TLS_SLOT 0 - MI_UNUSED(ofs); - res = NULL; - #endif - return res; -} - -#ifndef MI_HAS_TLS_SLOT -#define MI_HAS_TLS_SLOT 1 -#endif - -// setting a tls slot is only used on macOS for now -static inline void mi_prim_tls_slot_set(size_t slot, void* value) mi_attr_noexcept { - const size_t ofs = (slot*sizeof(void*)); - #if defined(_WIN32) - ((void**)NtCurrentTeb())[slot] = value; MI_UNUSED(ofs); - #elif defined(__i386__) - __asm__("movl %1,%%gs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // 32-bit always uses GS - #elif defined(__APPLE__) && defined(__x86_64__) - __asm__("movq %1,%%gs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x86_64 macOS uses GS - #elif defined(__x86_64__) && (MI_INTPTR_SIZE==4) - __asm__("movl %1,%%fs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x32 ABI - #elif defined(__x86_64__) - __asm__("movq %1,%%fs:%0" : "=m" (*((void**)ofs)) : "rn" (value) : ); // x86_64 Linux, BSD uses FS - #elif defined(__arm__) - void** tcb; MI_UNUSED(ofs); - __asm__ volatile ("mrc p15, 0, %0, c13, c0, 3\nbic %0, %0, #3" : "=r" (tcb)); - tcb[slot] = value; - #elif defined(__aarch64__) - void** tcb; MI_UNUSED(ofs); - #if defined(__APPLE__) // M1, issue #343 - __asm__ volatile ("mrs %0, tpidrro_el0\nbic %0, %0, #7" : "=r" (tcb)); - #else - __asm__ volatile ("mrs %0, tpidr_el0" : "=r" (tcb)); - #endif - tcb[slot] = value; - #elif defined(__APPLE__) && defined(__POWERPC__) // ppc, issue #781 - MI_UNUSED(ofs); - pthread_setspecific(slot, value); - #else - MI_UNUSED(ofs); MI_UNUSED(value); - #endif -} - -#endif - - -// defined in `init.c`; do not use these directly -extern mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_main; // theap belonging to the main heap -extern mi_decl_hidden bool _mi_process_is_initialized; // has mi_process_init been called? - - -//------------------------------------------------------------------- -// Get a fast unique thread id. -// -// Getting the thread id should be performant as it is called in the -// fast path of `_mi_free` and we specialize for various platforms as -// inlined definitions. Regular code should call `init.c:_mi_thread_id()`. -// We only require _mi_prim_thread_id() to return a unique id -// for each thread (unequal to zero). -//------------------------------------------------------------------- - - -// Do we have __builtin_thread_pointer? This would be the preferred way to get a unique thread id -// but unfortunately, it seems we cannot test for this reliably at this time (see issue #883) -// Nevertheless, it seems needed on older graviton platforms (see issue #851). -// For now, we only enable this for specific platforms. -#if !defined(MI_USE_BUILTIN_THREAD_POINTER) /* allow user override */ - #if !defined(__APPLE__) /* on apple (M1) the wrong register is read (tpidr_el0 instead of tpidrro_el0) so fall back to TLS slot assembly ()*/ \ - && !defined(__CYGWIN__) \ - && !defined(MI_LIBC_MUSL) \ - && (!defined(__clang_major__) || __clang_major__ >= 14) /* older clang versions emit bad code; fall back to using the TLS slot () */ - #if (defined(__GNUC__) && (__GNUC__ >= 7) && defined(__aarch64__)) /* aarch64 for older gcc versions (issue #851) */ \ - || (defined(__GNUC__) && (__GNUC__ >= 11) && defined(__x86_64__)) \ - || (defined(__clang_major__) && (__clang_major__ >= 14) && (defined(__aarch64__) || defined(__x86_64__))) - #define MI_USE_BUILTIN_THREAD_POINTER 1 - #endif - #endif -#endif - -static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept; - -static inline mi_threadid_t _mi_prim_thread_id(void) mi_attr_noexcept { - const mi_threadid_t tid = __mi_prim_thread_id(); - mi_assert_internal(tid > 1); - mi_assert_internal((tid & MI_PAGE_FLAG_MASK) == 0); // bottom 2 bits are clear? - return tid; -} - -// Get a unique id for the current thread. -#if defined(MI_PRIM_THREAD_ID) - -static inline mi_threadid_t _mi_prim_thread_id(void) mi_attr_noexcept { - const mi_threadid_t tid = MI_PRIM_THREAD_ID(); // used for example by CPython for a free threaded build (see python/cpython#115488) - mi_assert_internal( (tid & 0x03) == 0 ); // mimalloc reserves the bottom 2 bits - return tid; -} - -#elif defined(_WIN32) - -static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { - // Windows: works on Intel and ARM in both 32- and 64-bit - return (uintptr_t)NtCurrentTeb(); -} - -#elif MI_USE_BUILTIN_THREAD_POINTER - -static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { - // Works on most Unix based platforms with recent compilers - return (uintptr_t)__builtin_thread_pointer(); -} - -#elif MI_HAS_TLS_SLOT - -static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { - #if defined(__BIONIC__) - // issue #384, #495: on the Bionic libc (Android), slot 1 is the thread id - // see: https://github.com/aosp-mirror/platform_bionic/blob/c44b1d0676ded732df4b3b21c5f798eacae93228/libc/platform/bionic/tls_defines.h#L86 - return (uintptr_t)mi_prim_tls_slot(1); - #else - // in all our other targets, slot 0 is the thread id - // glibc: https://sourceware.org/git/?p=glibc.git;a=blob_plain;f=sysdeps/x86_64/nptl/tls.h - // apple: https://github.com/apple/darwin-xnu/blob/main/libsyscall/os/tsd.h#L36 - return (uintptr_t)mi_prim_tls_slot(0); - #endif -} - -#else - -// otherwise use portable C, taking the address of a thread local variable (this is still very fast on most platforms). -static inline mi_threadid_t __mi_prim_thread_id(void) mi_attr_noexcept { - return (uintptr_t)&__mi_theap_main; -} - -#endif - - - -/* ---------------------------------------------------------------------------------------- -Get the thread local default theap: `_mi_theap_default()` (and the cached heap `_mi_theap_cached`). - -This is inlined here as it is on the fast path for allocation functions. -We have 4 models: - -- MI_TLS_MODEL_THREAD_LOCAL: use regular thread local (default on Linux, FreeBSD, etc) - On most platforms (Linux, FreeBSD, NetBSD, etc), this just returns a - thread local variable (`__mi_theap_default`). With the initial-exec TLS model this ensures - that the storage will always be available and properly initialized (with an empty theap). - - On some platforms the underlying TLS implementation (or the loader) will call itself `malloc` - on a first access to a thread local and recurse in the MI_TLS_MODEL_THREAD_LOCAL. - A way around this is to define MI_TLS_RECURSE_GUARD which adds an extra check if the process - is initialized before accessing the thread-local. This is a check in the fast path though - so this should be avoided. - -- MI_TLS_MODEL_FIXED_SLOT: use a fixed slot in the TLS block (default on macOS) - This reserves an unused and fixed TLS slot. This is fast and avoids the problem - where the underlying TLS implementation (or the loader) will call itself `malloc` - on a first access to a thread local (and recurse in the MI_TLS_MODEL_THREAD_LOCAL). - This goes wrong though if the OS or a library uses the same fixed slot. - -- MI_TLS_MODEL_DYNAMIC_WIN32: use a dynamically allocated slot with TlsAlloc. (default on Windows) - Windows has somewhat slow thread locals so by default we use TlsAlloc'd slots which - can be more efficient. First tries to use one of the "direct" first 64 slots which - are the fastest, but falls back to using "expansion" slots when needed (up to 1088 slots). - (If the allocated slot happens to always be under 64 for a particular program, - one might use cmake with `-DMI_WIN_DIRECT_TLS=ON` to skip the expansion slot test in the fast path.) - -- MI_TLS_MODEL_DYNAMIC_PTHREADS: use `pthread_getspecific`. (default on OpenBSD, maybe good for Android as well?) - Use pthread local storage. Somewhat slow but can work well depending on the platform. - -Each model should define `MI_THEAP_INITASNULL` to signify that the initial value -returned from `_mi_theap_default()` can be `NULL` (instead of the address of the empty heap). -This incurs an extra check in the fast path (but can often be combined in an existing check). -------------------------------------------------------------------------------------------- */ - -static inline mi_theap_t* _mi_theap_default(void); -static inline mi_theap_t* _mi_theap_cached(void); - -#if defined(_WIN32) - #define MI_TLS_MODEL_DYNAMIC_WIN32 1 -#elif defined(__APPLE__) && MI_HAS_TLS_SLOT && !defined(__POWERPC__) // macOS on arm64 or x64 - // #define MI_TLS_MODEL_DYNAMIC_PTHREADS 1 // also works but a bit slower - #define MI_TLS_MODEL_FIXED_SLOT 1 - #define MI_TLS_MODEL_FIXED_SLOT_DEFAULT 108 // seems unused. @apple: it would be great to get 2 official slots for custom allocators :-) - #define MI_TLS_MODEL_FIXED_SLOT_CACHED 109 - // we used before __PTK_FRAMEWORK_OLDGC_KEY9 (89) but that seems used now. - // see -#elif defined(__APPLE__) || defined(__OpenBSD__) || defined(__ANDROID__) - #define MI_TLS_MODEL_DYNAMIC_PTHREADS 1 - // #define MI_TLS_MODEL_DYNAMIC_PTHREADS_DEFAULT_ENTRY_IS_NULL 1 -#else - #define MI_TLS_MODEL_THREAD_LOCAL 1 -#endif - -// Declared this way to optimize register spills and branches -mi_decl_cold mi_decl_noinline mi_theap_t* _mi_theap_empty_get(void); - -static inline mi_theap_t* __mi_theap_empty(void) { - #if __GNUC__ - __asm(""); // prevent conditional load - return (mi_theap_t*)&_mi_theap_empty; - #else - return _mi_theap_empty_get(); - #endif -} - -#if MI_TLS_MODEL_THREAD_LOCAL -// Thread local with an initial value (default on Linux). Very efficient. - -extern mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_default; // default theap to allocate from -extern mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_cached; // theap from the last used heap - -static inline mi_theap_t* _mi_theap_default(void) { - #if defined(MI_TLS_RECURSE_GUARD) - if (mi_unlikely(!_mi_process_is_initialized)) return _mi_theap_empty_get(); - #endif - return __mi_theap_default; -} - -static inline mi_theap_t* _mi_theap_cached(void) { - return __mi_theap_cached; -} - -#elif MI_TLS_MODEL_FIXED_SLOT -// Fixed TLS slot (default on macOS). -#define MI_THEAP_INITASNULL 1 - -static inline mi_theap_t* _mi_theap_default(void) { - return (mi_theap_t*)mi_prim_tls_slot(MI_TLS_MODEL_FIXED_SLOT_DEFAULT); -} - -static inline mi_theap_t* _mi_theap_cached(void) { - return (mi_theap_t*)mi_prim_tls_slot(MI_TLS_MODEL_FIXED_SLOT_CACHED); -} - -#elif MI_TLS_MODEL_DYNAMIC_WIN32 -// Dynamic TLS slot (default on Windows) -#define MI_THEAP_INITASNULL 1 - -// We try to use direct slots (64), but can also use the expansion slots (upto 1024 extra available) -// See for the offsets. -#if MI_SIZE_SIZE==4 -#define MI_TLS_EXPANSION_SLOT (0x0F94 / MI_SIZE_SIZE) -#else -#define MI_TLS_EXPANSION_SLOT (0x1780 / MI_SIZE_SIZE) -#endif - -extern mi_decl_hidden size_t _mi_theap_default_slot; -extern mi_decl_hidden size_t _mi_theap_cached_slot; -extern mi_decl_hidden size_t _mi_theap_default_expansion_slot; -extern mi_decl_hidden size_t _mi_theap_cached_expansion_slot; - -static inline mi_theap_t* _mi_theap_default(void) { - const size_t slot = _mi_theap_default_slot; - mi_theap_t* theap = (mi_theap_t*)mi_prim_tls_slot(slot); - #if !MI_WIN_DIRECT_TLS - if mi_unlikely(slot==MI_TLS_EXPANSION_SLOT) { // in TlsExpansionSlots ? - if mi_likely(theap!=NULL) { // initialized (on this thread)? - theap = ((mi_theap_t**)theap)[_mi_theap_default_expansion_slot]; - } - } - #endif - return theap; -} - -static inline mi_theap_t* _mi_theap_cached(void) { - const size_t slot = _mi_theap_cached_slot; - mi_theap_t* theap = (mi_theap_t*)mi_prim_tls_slot(slot); - #if !MI_WIN_DIRECT_TLS - if mi_unlikely(slot==MI_TLS_EXPANSION_SLOT) { // in TlsExpansionSlots ? - if mi_likely(theap!=NULL) { // initialized (on this thread)? - theap = ((mi_theap_t**)theap)[_mi_theap_cached_expansion_slot]; - } - } - #endif - return theap; -} - -#elif MI_TLS_MODEL_DYNAMIC_PTHREADS -// Dynamic pthread slot on less common platforms. This is not too bad. (default on OpenBSD) -#define MI_THEAP_INITASNULL 1 - -extern mi_decl_hidden pthread_key_t _mi_theap_default_key; -extern mi_decl_hidden pthread_key_t _mi_theap_cached_key; - -static inline mi_theap_t* _mi_theap_default(void) { - #if !MI_TLS_MODEL_DYNAMIC_PTHREADS_DEFAULT_ENTRY_IS_NULL - // we can skip this check if using the initial key will return NULL from pthread_getspecific - if mi_unlikely(_mi_theap_default_key==0) { return NULL; } - #endif - return (mi_theap_t*)pthread_getspecific(_mi_theap_default_key); -} - -static inline mi_theap_t* _mi_theap_cached(void) { - #if !MI_TLS_MODEL_DYNAMIC_PTHREADS_DEFAULT_ENTRY_IS_NULL - // we can skip this check if using the initial key will return NULL from pthread_getspecific - if mi_unlikely(_mi_theap_cached_key==0) { return NULL; } - #endif - return (mi_theap_t*)pthread_getspecific(_mi_theap_cached_key); -} - -#else -#error "no TLS model is defined for this platform?" -#endif - - -// Check if a thread is initialized (without using a thread-local if using fixed slots) -static inline bool _mi_thread_is_initialized(void) { - return (mi_theap_is_initialized(_mi_theap_default())); -} - -// Get (and possible create) the theap belonging to a heap -// We cache the last accessed theap in `_mi_theap_cached` for better performance. -static inline mi_theap_t* _mi_heap_theap(const mi_heap_t* heap) { - mi_theap_t* theap = _mi_theap_cached(); - #if MI_THEAP_INITASNULL - if mi_likely(theap!=NULL && _mi_theap_heap(theap)==heap) return theap; - #else - if mi_likely(_mi_theap_heap(theap)==heap) return theap; - #endif - return _mi_heap_theap_get_or_init(heap); -} - -// Get the theap belonging to a heap without creating in if it is not yet initialized. -static inline mi_theap_t* _mi_heap_theap_peek(const mi_heap_t* heap) { - mi_theap_t* theap = _mi_theap_cached(); - #if MI_THEAP_INITASNULL - if mi_unlikely(theap==NULL || _mi_theap_heap(theap)!=heap) - #else - if mi_unlikely(_mi_theap_heap(theap)!=heap) - #endif - { - theap = _mi_heap_theap_get_peek(heap); // don't update the cache on a query (?) - } - mi_assert(theap==NULL || _mi_theap_heap(theap)==heap); - return theap; -} - -// Find the associated theap or NULL if it does not exist (during shutdown) -// Should be fast as it is called in `free.c:mi_free_try_collect`. -static inline mi_theap_t* _mi_page_associated_theap_peek(mi_page_t* page) { - mi_heap_t* const heap = page->heap; - mi_theap_t* theap; - if mi_likely(heap==NULL) { theap = __mi_theap_main; } // note: on macOS accessing the thread_local can cause allocation during thread shutdown (and reinitialize the thread)! - else { theap = _mi_heap_theap_peek(heap); } - mi_assert_internal(theap==NULL || _mi_thread_id()==theap->tld->thread_id); - return theap; -} - #endif // MI_PRIM_H diff --git a/vendored/mimalloc/include/mimalloc/track.h b/vendored/mimalloc/include/mimalloc/track.h index 1f40212978..753363366b 100644 --- a/vendored/mimalloc/include/mimalloc/track.h +++ b/vendored/mimalloc/include/mimalloc/track.h @@ -135,16 +135,16 @@ defined, undefined, or not accessible at all: #if MI_PADDING #define mi_track_malloc(p,reqsize,zero) \ - if ((p)!=NULL) { \ + do { if ((p)!=NULL) { \ mi_assert_internal(mi_usable_size(p)==(reqsize)); \ mi_track_malloc_size(p,reqsize,reqsize,zero); \ - } + } } while(0) #else #define mi_track_malloc(p,reqsize,zero) \ - if ((p)!=NULL) { \ + do { if ((p)!=NULL) { \ mi_assert_internal(mi_usable_size(p)>=(reqsize)); \ mi_track_malloc_size(p,reqsize,mi_usable_size(p),zero); \ - } + } } while(0) #endif #endif // MI_TRACK_H diff --git a/vendored/mimalloc/include/mimalloc/types.h b/vendored/mimalloc/include/mimalloc/types.h index 31fdace9c7..58c8cc07ef 100644 --- a/vendored/mimalloc/include/mimalloc/types.h +++ b/vendored/mimalloc/include/mimalloc/types.h @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -87,7 +87,7 @@ terms of the MIT license. A copy of the license can be found in the file #endif // Enable guard pages behind objects of a certain size (set by the MIMALLOC_GUARDED_MIN/MAX/SAMPLE_RATE options) -#if !defined(MI_GUARDED) && MI_DEBUG && !defined(NDEBUG) && !MI_PAGE_META_ALIGNED_FREE_SMALL +#if !defined(MI_GUARDED) && MI_DEBUG && !defined(NDEBUG) && !MI_PAGE_META_ALIGNED_FREE_SMALL #define MI_GUARDED 1 #endif @@ -109,6 +109,10 @@ terms of the MIT license. A copy of the license can be found in the file #define MI_ENCODE_FREELIST 1 #endif +#if (MI_ENCODE_FREELIST && (MI_SECURE>=4 || MI_DEBUG!=0)) +#define MI_CHECK_DOUBLE_FREE 1 +#endif + // Enable large pages for objects between 64KiB and 512KiB. // This should perhaps be disabled by default as for many workloads the block sizes above 64 KiB // are quite random which can lead to too many partially used large pages (but see issue #1104). @@ -116,9 +120,9 @@ terms of the MIT license. A copy of the license can be found in the file #define MI_ENABLE_LARGE_PAGES 1 #endif -// Place page meta info at the start of the page area or keep it separate? -// Separate keeps the page info at the arena start (default) which is more secure -// and reduces wasted space due to alignment and block sizes. +// Place page meta info at the start of the page area or keep it separate? +// Separate keeps the page info at the arena start (default) which is more secure +// and reduces wasted space due to alignment and block sizes. // (but also reserves more memory up front (about 2MiB per GiB)) #if !defined(MI_PAGE_META_IS_SEPARATED) #if MI_PAGE_MAP_FLAT @@ -159,14 +163,14 @@ terms of the MIT license. A copy of the license can be found in the file #ifndef MI_ARENA_SLICE_SHIFT #ifdef MI_SMALL_PAGE_SHIFT // backward compatibility #define MI_ARENA_SLICE_SHIFT MI_SMALL_PAGE_SHIFT - #elif MI_SECURE>=5 && __APPLE__ && MI_ARCH_ARM64 + #elif MI_SECURE>=5 && ((__APPLE__ && MI_ARCH_ARM64) || (defined(PAGE_SIZE) && PAGE_SIZE >= 16*MI_KiB)) #define MI_ARENA_SLICE_SHIFT (17) // 128 KiB to not waste too much due to 16 KiB guard pages #else #define MI_ARENA_SLICE_SHIFT (13 + MI_SIZE_SHIFT) // 64 KiB (32 KiB on 32-bit) #endif #endif -#if MI_ARENA_SLICE_SHIFT < 12 -#error Arena slices should be at least 4KiB +#if MI_ARENA_SLICE_SHIFT < 13 +#error Arena slices should be at least 8KiB #endif #ifndef MI_BCHUNK_BITS_SHIFT @@ -267,6 +271,7 @@ static inline bool mi_memkind_needs_no_free(mi_memkind_t memkind) { return (memkind <= MI_MEM_STATIC); } +typedef struct mi_meta_page_s mi_meta_page_t; typedef struct mi_memid_os_info { void* base; // actual base address of the block (used for offset aligned allocations) @@ -281,9 +286,9 @@ typedef struct mi_memid_arena_info { } mi_memid_arena_info_t; typedef struct mi_memid_meta_info { - void* meta_page; // meta-page that contains the block - uint32_t block_index; // block index in the meta-data page - uint32_t block_count; // allocated blocks + mi_meta_page_t* meta_page; // meta-page that contains the block + uint32_t block_index; // block index in the meta-data page + uint32_t block_count; // allocated blocks } mi_memid_meta_info_t; typedef struct mi_memid_s { @@ -393,7 +398,8 @@ typedef struct mi_page_s { _Atomic(mi_thread_free_t) xthread_free; // list of deferred free blocks freed by other threads (= `mi_block_t* | (1 if owned)`) size_t block_size; // const: size available in each block (always `>0`) - uint8_t* page_start; // const: start of the blocks + uint32_t page_ma_offset; // const: offset relative to the page (in MI_MAX_ALIGN_SIZE parts) to the start of the blocks + uint32_t slice_committed; // committed size relative to the first arena slice of the page data (or 0 if the page is fully committed already) #if (MI_ENCODE_FREELIST || MI_PADDING) uintptr_t keys[2]; // const: two random keys to encode the free lists (see `_mi_block_next`) or padding canary @@ -404,7 +410,6 @@ typedef struct mi_page_s { struct mi_page_s* next; // next page owned by the theap with the same `block_size` struct mi_page_s* prev; // previous page owned by the theap with the same `block_size` - size_t slice_committed; // committed size relative to the first arena slice of the page data (or 0 if the page is fully committed already) mi_memid_t memid; // const: provenance of the page memory } mi_page_t; @@ -423,7 +428,7 @@ typedef struct mi_page_s { #define MI_SMALL_MAX_OBJ_SIZE ((MI_SMALL_PAGE_SIZE-MI_PAGE_OSPAGE_BLOCK_ALIGN2)/6) // = 10 KiB #if MI_ENABLE_LARGE_PAGES #define MI_MEDIUM_MAX_OBJ_SIZE ((MI_MEDIUM_PAGE_SIZE-MI_PAGE_OSPAGE_BLOCK_ALIGN2)/6) // ~ 84 KiB -#define MI_LARGE_MAX_OBJ_SIZE (MI_LARGE_PAGE_SIZE/8) // <= 512 KiB // note: this must be a nice power of 2 or we get rounding issues with `_mi_bin` +#define MI_LARGE_MAX_OBJ_SIZE (MI_LARGE_PAGE_SIZE/8) // <= 512 KiB. note: this must be a nice power of 2 or we get rounding issues with `_mi_bin` #else #define MI_MEDIUM_MAX_OBJ_SIZE (MI_MEDIUM_PAGE_SIZE/8) // <= 64 KiB #define MI_LARGE_MAX_OBJ_SIZE MI_MEDIUM_MAX_OBJ_SIZE // note: this must be a nice power of 2 or we get rounding issues with `_mi_bin` @@ -434,6 +439,16 @@ typedef struct mi_page_s { #error "mimalloc internal: define more bins" #endif +// static invariant: MI_MAX_SINGLETON_BIN >= _mi_bin(MI_LARGE_MAX_OBJ_SIZE) (See init.c for the size bins) +#if (MI_LARGE_MAX_OBJ_WSIZE <= 8192) // 64 KiB +#define MI_MAX_SINGLETON_BIN (48) +#elif (MI_LARGE_MAX_OBJ_WSIZE <= 32768) // 256KiB +#define MI_MAX_SINGLETON_BIN (56) +#elif (MI_LARGE_MAX_OBJ_WSIZE <= 65536) // 512KiB +#define MI_MAX_SINGLETON_BIN (60) +#else +#define MI_MAX_SINGLETON_BIN MI_BIN_HUGE +#endif // ------------------------------------------------------ // Page kinds @@ -504,7 +519,9 @@ typedef struct mi_padding_s { struct mi_theap_s { mi_tld_t* tld; // thread-local data _Atomic(mi_heap_t*) heap; // the heap this theap belongs to. + _Atomic(mi_subproc_t*)subproc; // subproc this belongs too (always `subproc == heap->subproc` but needed for safe destruction) _Atomic(size_t) refcount; // reference count + _Atomic(size_t) freed; // ensure atomic free-ing unsigned long long heartbeat; // monotonic heartbeat count uintptr_t cookie; // random cookie to verify pointers (see `_mi_ptr_cookie`) mi_random_ctx_t random; // random number context used for secure allocation @@ -516,12 +533,12 @@ struct mi_theap_s { long generic_collect_count; // how often is `_mi_malloc_generic` called without collecting? mi_theap_t* tnext; // list of theaps in this thread - mi_theap_t* tprev; + mi_theap_t* tprev; mi_theap_t* hnext; // list of theaps of the owning `heap` mi_theap_t* hprev; - + long page_full_retain; // how many full pages can be retained per queue (before abandoning them) - bool allow_page_reclaim; // `true` if this theap should not reclaim abandoned pages + bool allow_page_reclaim; // `true` if this theap can reclaim abandoned pages bool allow_page_abandon; // `true` if this theap can abandon pages to reduce memory footprint #if MI_GUARDED size_t guarded_size_min; // minimal size for guarded objects @@ -590,6 +607,7 @@ struct mi_subproc_s { size_t subproc_seq; // unique id for sub-processes mi_subproc_t* next; // list of all sub-processes mi_subproc_t* prev; + _Atomic(mi_meta_page_t*) meta_pages; // meta data pages _Atomic(size_t) arena_count; // current count of arena's _Atomic(mi_arena_t*) arenas[MI_MAX_ARENAS]; // arena's of this sub-process @@ -607,6 +625,7 @@ struct mi_subproc_s { _Atomic(size_t) heap_total_count; // total created heaps in this sub-process mi_memid_t memid; // provenance of this memory block (meta or static) + mi_subproc_t* parent; // subproc in which this one was allocated mi_decl_align(8) // needed on some 32-bit platforms mi_stats_t stats; // subprocess statistics; updated for arena/OS stats like committed, // and otherwise merged with heap stats when those are deleted @@ -646,18 +665,18 @@ struct mi_tld_s { to reserve large arenas upfront and be able to reuse the memory more effectively. -----------------------------------------------------------------------------*/ -#define MI_ARENA_BIN_COUNT (MI_BIN_COUNT) +#define MI_ARENA_BIN_COUNT (MI_MAX_SINGLETON_BIN+1) #define MI_ARENA_MIN_SIZE (MI_BCHUNK_BITS * MI_ARENA_SLICE_SIZE) // 32 MiB (or 8 MiB on 32-bit) -#define MI_ARENA_MAX_SIZE (MI_BITMAP_MAX_BIT_COUNT * MI_ARENA_SLICE_SIZE) +#define MI_ARENA_MAX_SIZE (MI_BITMAP_MAX_BIT_COUNT * MI_ARENA_SLICE_SIZE) // 16 GiB typedef struct mi_bitmap_s mi_bitmap_t; // atomic bitmap (defined in `src/bitmap.h`) typedef struct mi_bbitmap_s mi_bbitmap_t; // atomic binned bitmap (defined in `src/bitmap.h`) -typedef struct mi_arena_pages_s { +struct mi_arena_pages_s { mi_bitmap_t* pages; // all registered pages (abandoned and owned) mi_bitmap_t* pages_abandoned[MI_ARENA_BIN_COUNT]; // abandoned pages per size bin (a set bit means the start of the page) // followed by the bitmaps (whose siz`es depend on the arena size) -} mi_arena_pages_t; +}; // A memory arena @@ -713,6 +732,10 @@ typedef struct mi_arena_s { #ifndef EOVERFLOW // count*size overflow #define EOVERFLOW (75) #endif +#ifndef ENOENT // environment variable not found +#define ENOENT (2) +#endif + /* ----------------------------------------------------------- Debug constants diff --git a/vendored/mimalloc/src/alloc-aligned.c b/vendored/mimalloc/src/alloc-aligned.c index 97dc6045b9..910ce66744 100644 --- a/vendored/mimalloc/src/alloc-aligned.c +++ b/vendored/mimalloc/src/alloc-aligned.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -7,7 +7,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc.h" #include "mimalloc/internal.h" -#include "mimalloc/prim.h" // _mi_theap_default +#include "mimalloc/prim-tls.h" // _mi_theap_default #include // memset @@ -18,7 +18,7 @@ terms of the MIT license. A copy of the license can be found in the file static bool mi_malloc_is_naturally_aligned( size_t size, size_t alignment ) { // certain blocks are always allocated at a certain natural alignment. // (see also `arena.c:mi_arenas_page_alloc_fresh`). - mi_assert_internal(_mi_is_power_of_two(alignment) && (alignment > 0)); + mi_assert_internal(mi_alignment_is_valid(alignment)); if (alignment > size) return false; const size_t bsize = mi_good_size(size); const bool ok = (bsize <= MI_PAGE_MAX_START_BLOCK_ALIGN2 && _mi_is_power_of_two(bsize)) || // power-of-two under N @@ -28,7 +28,7 @@ static bool mi_malloc_is_naturally_aligned( size_t size, size_t alignment ) { } #if MI_GUARDED -static mi_decl_restrict void* mi_theap_malloc_guarded_aligned(mi_theap_t* theap, size_t size, size_t alignment, bool zero) mi_attr_noexcept { +static mi_decl_noinline mi_decl_restrict void* mi_theap_malloc_guarded_aligned(mi_theap_t* theap, size_t size, size_t alignment, bool zero, size_t* usable) mi_attr_noexcept { // use over allocation for guarded blocksl #if MI_THEAP_INITASNULL if mi_unlikely(theap==NULL) { theap = _mi_theap_empty_get(); } @@ -39,7 +39,7 @@ static mi_decl_restrict void* mi_theap_malloc_guarded_aligned(mi_theap_t* theap, return NULL; } const size_t oversize = size + alignment - 1; - void* const base = _mi_theap_malloc_guarded(theap, oversize, zero); + void* const base = _mi_theap_malloc_guarded(theap, oversize, zero, usable); if (base==NULL) return NULL; void* const p = _mi_align_up_ptr(base, alignment); mi_track_align(base, p, (uint8_t*)p - (uint8_t*)base, size); @@ -69,7 +69,7 @@ static void* mi_theap_malloc_zero_no_guarded(mi_theap_t* theap, size_t size, boo static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_overalloc(mi_theap_t* const theap, const size_t size, const size_t alignment, const size_t offset, const bool zero, size_t* usable) mi_attr_noexcept { mi_assert_internal(size <= (MI_MAX_ALLOC_SIZE - MI_PADDING_SIZE)); - mi_assert_internal(alignment != 0 && _mi_is_power_of_two(alignment)); + mi_assert_internal(mi_alignment_is_valid(alignment)); void* p; size_t oversize; @@ -110,6 +110,11 @@ static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_overalloc(mi_theap mi_page_t* page = _mi_ptr_page(p); if (aligned_p != p) { mi_page_set_has_interior_pointers(page, true); + if (usable!=NULL) { + mi_assert_internal(*usable > adjust); + if (*usable > adjust) { *usable = *usable - adjust; } + mi_assert_internal(*usable >= size); + } #if MI_GUARDED // set tag to aligned so mi_usable_size works with guard pages if (adjust >= sizeof(mi_block_t)) { @@ -136,7 +141,7 @@ static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_overalloc(mi_theap // // for the tracker, on huge aligned allocations only from the start of the large block is defined // mi_track_mem_undefined(aligned_p, size); // if (zero) { - // _mi_memzero_aligned(aligned_p, mi_usable_size(aligned_p)); + // _mi_memzero_(aligned_p, mi_usable_size(aligned_p)); // } //} @@ -152,10 +157,10 @@ static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_overalloc(mi_theap // Generic primitive aligned allocation -- split out for better codegen static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_generic(mi_theap_t* const theap, const size_t size, const size_t alignment, const size_t offset, const bool zero, size_t* usable) mi_attr_noexcept { - mi_assert_internal(alignment != 0 && _mi_is_power_of_two(alignment)); + mi_assert_internal(mi_alignment_is_valid(alignment)); // we don't allocate more than MI_MAX_ALLOC_SIZE (see ) if mi_unlikely(size > (MI_MAX_ALLOC_SIZE - MI_PADDING_SIZE)) { - _mi_error_message(EOVERFLOW, "aligned allocation request is too large (size %zu, alignment %zu)\n", size, alignment); + _mi_error_message(EINVAL, "aligned allocation request is too large (size %zu, alignment %zu)\n", size, alignment); return NULL; } @@ -181,15 +186,17 @@ static mi_decl_noinline void* mi_theap_malloc_zero_aligned_at_generic(mi_theap_t } +static mi_decl_cold mi_decl_noinline void* mi_error_bad_alignment(size_t size, size_t alignment, size_t offset) { + _mi_error_message(EINVAL, "aligned allocation requires the alignment to be a power-of-two (size %zu, alignment %zu, offset %zu)\n", size, alignment, offset); + return NULL; +} + // Primitive aligned allocation static inline void* mi_theap_malloc_zero_aligned_at(mi_theap_t* const theap, const size_t size, const size_t alignment, const size_t offset, const bool zero, size_t* usable) mi_attr_noexcept { // note: we don't require `size > offset`, we just guarantee that the address at offset is aligned regardless of the allocated size. - if mi_unlikely(alignment == 0 || !_mi_is_power_of_two(alignment)) { // require power-of-two (see ) - #if MI_DEBUG > 0 - _mi_error_message(EOVERFLOW, "aligned allocation requires the alignment to be a power-of-two (size %zu, alignment %zu)\n", size, alignment); - #endif - return NULL; + if mi_unlikely(!mi_alignment_is_valid(alignment)) { // require power-of-two and multiple of void* (see ) + return mi_error_bad_alignment(size, alignment, offset); } #if MI_GUARDED @@ -197,7 +204,7 @@ static inline void* mi_theap_malloc_zero_aligned_at(mi_theap_t* const theap, con if mi_likely(theap!=NULL) #endif if (offset==0 && alignment < MI_PAGE_MAX_OVERALLOC_ALIGN && mi_theap_malloc_use_guarded(theap,size)) { - return mi_theap_malloc_guarded_aligned(theap, size, alignment, zero); + return mi_theap_malloc_guarded_aligned(theap, size, alignment, zero, usable); } #endif @@ -330,13 +337,15 @@ mi_decl_nodiscard mi_decl_restrict void* mi_heap_calloc_aligned(mi_heap_t* heap, // ------------------------------------------------------ static void* mi_theap_realloc_zero_aligned_at(mi_theap_t* theap, void* p, size_t newsize, size_t alignment, size_t offset, bool zero) mi_attr_noexcept { - mi_assert(alignment > 0); + mi_assert(mi_alignment_is_valid(alignment)); + if mi_unlikely(!mi_alignment_is_valid(alignment)) { // require power-of-two (see ) + return mi_error_bad_alignment(newsize,alignment,offset); + } if (alignment <= sizeof(uintptr_t) && offset==0) return _mi_theap_realloc_zero(theap,p,newsize,zero,NULL,NULL); if (p == NULL) return mi_theap_malloc_zero_aligned_at(theap,newsize,alignment,offset,zero,NULL); size_t size = mi_usable_size(p); - if (newsize <= size && newsize >= (size - (size / 2)) - && (((uintptr_t)p + offset) % alignment) == 0) { - return p; // reallocation still fits, is aligned and not more than 25% waste + if (newsize <= size && newsize >= (size - (size / 2)) && (((uintptr_t)p + offset) & (alignment-1)) == 0) { + return p; // reallocation still fits, is aligned and not more than 50% waste } else { // note: we don't zero allocate upfront so we only zero initialize the expanded part @@ -347,7 +356,7 @@ static void* mi_theap_realloc_zero_aligned_at(mi_theap_t* theap, void* p, size_t size_t start = (size >= sizeof(intptr_t) ? size - sizeof(intptr_t) : 0); _mi_memzero((uint8_t*)newp + start, newsize - start); } - _mi_memcpy_aligned(newp, p, (newsize > size ? size : newsize)); + _mi_memcpy(newp, p, (newsize > size ? size : newsize)); // cannot be aligned due to abitrary offset... (todo: require offset to be a multiple of sizeof(void*)?) mi_free(p); // only free if successful } return newp; diff --git a/vendored/mimalloc/src/alloc-override.c b/vendored/mimalloc/src/alloc-override.c index 93d066dc14..04da5f2935 100644 --- a/vendored/mimalloc/src/alloc-override.c +++ b/vendored/mimalloc/src/alloc-override.c @@ -127,7 +127,7 @@ typedef void* mi_nothrow_t; #elif defined(_MSC_VER) _Check_return_ _Ret_maybenull_ _Post_writable_byte_size_(_Size) _ACRTIMP _CRTALLOCATOR _CRT_HYBRIDPATCHABLE void* __cdecl _expand(_Pre_notnull_ void* _Block, _In_ _CRT_GUARDOVERFLOW size_t _Size) { - return mi_expand(_Block, _Size); + return mi__expand(_Block, _Size); } _Check_return_ _ACRTIMP size_t __cdecl _msize_base(_Pre_notnull_ void* _Block) _CRT_NOEXCEPT { @@ -324,14 +324,25 @@ typedef void* mi_nothrow_t; extern "C" { #endif +// defined here instead alloc-posix so we can alias it +mi_decl_nodiscard size_t mi_malloc_size(const void* p) mi_attr_noexcept { + if (!mi_is_in_heap_region(p)) return 0; + return mi_usable_size(p); +} + +mi_decl_nodiscard size_t mi_malloc_usable_size(const void *p) mi_attr_noexcept { + if (!mi_is_in_heap_region(p)) return 0; + return mi_usable_size(p); +} + #ifndef MI_OSX_IS_INTERPOSED // Forward Posix/Unix calls as well void* reallocf(void* p, size_t newsize) MI_FORWARD2(mi_reallocf,p,newsize) - size_t malloc_size(const void* p) MI_FORWARD1(mi_usable_size,p) + size_t malloc_size(const void* p) MI_FORWARD1(mi_malloc_size,p) #if !defined(__ANDROID__) && !defined(__FreeBSD__) && !defined(__DragonFly__) - size_t malloc_usable_size(void *p) MI_FORWARD1(mi_usable_size,p) + size_t malloc_usable_size(void *p) MI_FORWARD1(mi_malloc_usable_size,p) #else - size_t malloc_usable_size(const void *p) MI_FORWARD1(mi_usable_size,p) + size_t malloc_usable_size(const void *p) MI_FORWARD1(mi_malloc_usable_size,p) #endif // No forwarding here due to aliasing/name mangling issues diff --git a/vendored/mimalloc/src/alloc-posix.c b/vendored/mimalloc/src/alloc-posix.c index 60639ff497..c7d6e3bb6b 100644 --- a/vendored/mimalloc/src/alloc-posix.c +++ b/vendored/mimalloc/src/alloc-posix.c @@ -31,17 +31,6 @@ terms of the MIT license. A copy of the license can be found in the file #define ENOMEM 12 #endif - -mi_decl_nodiscard size_t mi_malloc_size(const void* p) mi_attr_noexcept { - // if (!mi_is_in_heap_region(p)) return 0; - return mi_usable_size(p); -} - -mi_decl_nodiscard size_t mi_malloc_usable_size(const void *p) mi_attr_noexcept { - // if (!mi_is_in_heap_region(p)) return 0; - return mi_usable_size(p); -} - mi_decl_nodiscard size_t mi_malloc_good_size(size_t size) mi_attr_noexcept { return mi_good_size(size); } @@ -52,13 +41,12 @@ void mi_cfree(void* p) mi_attr_noexcept { } } -int mi_posix_memalign(void** p, size_t alignment, size_t size) mi_attr_noexcept { +int mi_posix_memalign(void** p, size_t alignment, size_t size) { // mi_attr_noexcept (issue #794) // Note: The spec dictates we should not modify `*p` on an error. (issue#27) // if (p == NULL) return EINVAL; - if ((alignment % sizeof(void*)) != 0) return EINVAL; // natural alignment - // it is also required that alignment is a power of 2 and > 0; this is checked in `mi_malloc_aligned` - if (alignment==0 || !_mi_is_power_of_two(alignment)) return EINVAL; // not a power of 2 + // it is required that alignment is a power of 2 and a multiple of sizeof(void*) + if (alignment // memset, strlen (for mi_strdup) #include // malloc, abort @@ -120,10 +120,6 @@ extern void* _mi_page_malloc_zero(mi_theap_t* theap, mi_page_t* page, size_t siz return mi_page_malloc_zero(theap, page, size, zero, NULL); } -#if MI_GUARDED -mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, bool zero) mi_attr_noexcept; -#endif - // main allocation primitives for small and generic allocation // internal small size allocation @@ -140,7 +136,7 @@ static mi_decl_forceinline mi_decl_restrict void* mi_theap_malloc_small_zero_non #endif #if MI_GUARDED if mi_unlikely(mi_theap_malloc_use_guarded(theap,size)) { - return _mi_theap_malloc_guarded(theap, size, zero); + return _mi_theap_malloc_guarded(theap, size, zero, usable); } #endif @@ -165,7 +161,7 @@ static mi_decl_forceinline void* mi_theap_malloc_generic(mi_theap_t* theap, size if (theap!=NULL) #endif if (huge_alignment==0 && mi_theap_malloc_use_guarded(theap, size)) { - return _mi_theap_malloc_guarded(theap, size, zero); + return _mi_theap_malloc_guarded(theap, size, zero, usable); } #endif #if !MI_THEAP_INITASNULL @@ -252,7 +248,7 @@ mi_decl_nodiscard extern inline mi_decl_restrict void* mi_theap_malloc(mi_theap_ } mi_decl_nodiscard mi_decl_restrict void* mi_malloc(size_t size) mi_attr_noexcept { - return mi_theap_malloc(_mi_theap_default(), size); + return mi_theap_malloc(_mi_theap_default(), size); } mi_decl_nodiscard mi_decl_restrict void* mi_heap_malloc(mi_heap_t* heap, size_t size) mi_attr_noexcept { @@ -307,6 +303,10 @@ mi_decl_nodiscard mi_decl_restrict void* mi_umalloc_small(size_t size, size_t* u return mi_theap_malloc_small_zero(_mi_theap_default(), size, false, usable); } +mi_decl_nodiscard mi_decl_restrict void* mi_uzalloc_small(size_t size, size_t* usable) mi_attr_noexcept { + return mi_theap_malloc_small_zero(_mi_theap_default(), size, true, usable); +} + mi_decl_nodiscard mi_decl_restrict void* mi_theap_umalloc(mi_theap_t* theap, size_t size, size_t* usable) mi_attr_noexcept { return _mi_theap_malloc_zero_ex(theap, size, false, 0, usable); } @@ -369,21 +369,34 @@ void* _mi_theap_realloc_zero(mi_theap_t* theap, void* p, size_t newsize, bool ze size = 0; if (usable_pre!=NULL) { *usable_pre = 0; } } - else { - page = mi_validate_ptr_page(p,"mi_realloc"); + else { + page = mi_validate_ptr_page(p,"mi_realloc"); + if mi_unlikely(page==NULL) { // invalid pointer + if (usable_pre!=NULL) { *usable_pre = 0; } + if (usable_post!=NULL) { *usable_post = 0; } + return NULL; + } size = _mi_usable_size(p,page); if (usable_pre!=NULL) { *usable_pre = mi_page_usable_block_size(page); } } - if mi_unlikely(newsize<=size && newsize>=(size/2) && newsize>0 // note: newsize must be > 0 or otherwise we return NULL for realloc(NULL,0) - && mi_page_heap(page)==_mi_theap_heap(theap)) // and within the same heap - { - mi_assert_internal(p!=NULL); - // todo: do not track as the usable size is still the same in the free; adjust potential padding? - // mi_track_resize(p,size,newsize) - // if (newsize < size) { mi_track_mem_noaccess((uint8_t*)p + newsize, size - newsize); } - if (usable_post!=NULL) { *usable_post = mi_page_usable_block_size(page); } - return p; // reallocation still fits and not more than 50% waste + // check if we can reuse the existing block + if mi_unlikely(newsize<=size && newsize>=(size/2) && newsize>0) { // note: newsize must be > 0 or otherwise we return NULL for realloc(NULL,0) + mi_assert_internal(page!=NULL); // note: page!=NULL (since if p==NULL, we have size=0 and size>=newsize>0 + #if MI_THEAP_INITASNULL + if (theap!=NULL) + #endif + { + if (mi_page_heap(page)==_mi_theap_heap(theap)) { // and within the same heap + mi_assert_internal(p!=NULL); + // todo: do not track as the usable size is still the same in the free; adjust potential padding? + // mi_track_resize(p,size,newsize) + // if (newsize < size) { mi_track_mem_noaccess((uint8_t*)p + newsize, size - newsize); } + if (usable_post!=NULL) { *usable_post = mi_page_usable_block_size(page); } + return p; // reallocation still fits and not more than 50% waste + } + } } + // otherwise allocate a fresh block void* newp = mi_theap_umalloc(theap,newsize,usable_post); if mi_likely(newp != NULL) { if (zero && newsize > size) { @@ -535,13 +548,30 @@ mi_decl_nodiscard mi_decl_restrict char* mi_heap_strndup(mi_heap_t* heap, const mi_decl_nodiscard static mi_decl_restrict char* mi_theap_realpath(mi_theap_t* theap, const char* fname, char* resolved_name) mi_attr_noexcept { // todo: use GetFullPathNameW to allow longer file names + if (fname==NULL || *fname==0) { + errno = EINVAL; + return NULL; + } char buf[PATH_MAX]; DWORD res = GetFullPathNameA(fname, PATH_MAX, (resolved_name == NULL ? buf : resolved_name), NULL); if (res == 0) { - errno = GetLastError(); return NULL; + DWORD err = GetLastError(); + switch (err) { + case ERROR_LOCK_VIOLATION: + case ERROR_SHARING_VIOLATION: + case ERROR_INVALID_ACCESS: errno = EACCES; break; + case ERROR_INVALID_HANDLE: + case ERROR_INVALID_FUNCTION: errno = EINVAL; break; + case ERROR_PATH_NOT_FOUND: errno = ENOTDIR; break; + case ERROR_FILE_NOT_FOUND: errno = ENOENT; break; + case ERROR_NOT_ENOUGH_MEMORY: errno = ENOMEM; break; + default: errno = EIO; + } + return NULL; } else if (res > PATH_MAX) { - errno = EINVAL; return NULL; + errno = ENAMETOOLONG; + return NULL; } else if (resolved_name != NULL) { return resolved_name; @@ -615,8 +645,11 @@ The standard requires calling into `get_new_handler` and throwing the bad_alloc exception on failure. If we compile with a C++ compiler we can implement this precisely. If we use a C compiler we cannot throw a `bad_alloc` exception -but we call `exit` instead (i.e. not returning). +but we call `abort` instead (i.e. not returning). +Also, the standard requires calling the new handler until +it returns false, but we limit the total calls. -------------------------------------------------------*/ +#define MI_TRY_NEW_MAX (4) #ifdef __cplusplus #include @@ -638,10 +671,19 @@ static bool mi_try_new_handler(bool nothrow) { #endif return false; } - else { + else if (!nothrow) { h(); return true; } + else { + try { + h(); + } + catch(...) { // swallow std::bad_alloc + return false; // stop trying + } + return true; + } } #else typedef void (*std_new_handler_t)(void); @@ -678,7 +720,8 @@ static bool mi_try_new_handler(bool nothrow) { static mi_decl_noinline void* mi_theap_try_new(mi_theap_t* theap, size_t size, bool nothrow ) { void* p = NULL; - while(p == NULL && mi_try_new_handler(nothrow)) { + for(int i = 0; i < MI_TRY_NEW_MAX && p == NULL && mi_try_new_handler(nothrow); i++) { + if (size > MI_MAX_ALLOC_SIZE) return NULL; // call try_new_handler at least once p = mi_theap_malloc(theap,size); } return p; @@ -692,7 +735,6 @@ static mi_decl_noinline void* mi_heap_try_new(mi_heap_t* heap, size_t size, bool return mi_theap_try_new(_mi_heap_theap(heap), size, nothrow); } - mi_decl_nodiscard static mi_decl_restrict void* mi_theap_alloc_new(mi_theap_t* theap, size_t size) { void* p = mi_theap_malloc(theap,size); if mi_unlikely(p == NULL) return mi_theap_try_new(theap, size, false); @@ -709,7 +751,6 @@ mi_decl_nodiscard mi_decl_restrict void* mi_heap_alloc_new(mi_heap_t* heap, size return p; } - mi_decl_nodiscard static mi_decl_restrict void* mi_theap_alloc_new_n(mi_theap_t* theap, size_t count, size_t size) { size_t total; if mi_unlikely(mi_count_size_overflow(count, size, &total)) { @@ -729,43 +770,52 @@ mi_decl_nodiscard mi_decl_restrict void* mi_heap_alloc_new_n(mi_heap_t* heap, si return mi_theap_alloc_new_n(_mi_heap_theap(heap), count, size); } - mi_decl_nodiscard mi_decl_restrict void* mi_new_nothrow(size_t size) mi_attr_noexcept { void* p = mi_malloc(size); if mi_unlikely(p == NULL) return mi_try_new(size, true); return p; } -mi_decl_nodiscard mi_decl_restrict void* mi_new_aligned(size_t size, size_t alignment) { - void* p; - do { - p = mi_malloc_aligned(size, alignment); +static mi_decl_noinline void* mi_try_new_aligned(size_t size, size_t alignment, bool nothrow) { + void* p = NULL; + for(int i = 0; i < MI_TRY_NEW_MAX && p==NULL && mi_try_new_handler(nothrow); i++) { + if (!mi_alignment_is_valid(alignment)) return NULL; + p = mi_malloc_aligned(size,alignment); } - while(p == NULL && mi_try_new_handler(false)); + return p; +} + +mi_decl_nodiscard mi_decl_restrict void* mi_new_aligned(size_t size, size_t alignment) { + void* p = mi_malloc_aligned(size, alignment); + if mi_unlikely(p==NULL) return mi_try_new_aligned(size,alignment,false); return p; } mi_decl_nodiscard mi_decl_restrict void* mi_new_aligned_nothrow(size_t size, size_t alignment) mi_attr_noexcept { - void* p; - do { - p = mi_malloc_aligned(size, alignment); - } - while(p == NULL && mi_try_new_handler(true)); + void* p = mi_malloc_aligned(size, alignment); + if mi_unlikely(p==NULL) return mi_try_new_aligned(size,alignment,true); return p; } +static mi_decl_noinline void* mi_try_new_realloc(void* p, size_t newsize) { + void* q = NULL; + for(int i = 0; i < MI_TRY_NEW_MAX && q==NULL && mi_try_new_handler(false); i++) { + if (newsize > MI_MAX_ALLOC_SIZE) return NULL; + q = mi_realloc(p,newsize); + } + return q; +} + mi_decl_nodiscard void* mi_new_realloc(void* p, size_t newsize) { - void* q; - do { - q = mi_realloc(p, newsize); - } while (q == NULL && mi_try_new_handler(false)); + void* q = mi_realloc(p, newsize); + if (q == NULL) return mi_try_new_realloc(p,newsize); return q; } mi_decl_nodiscard void* mi_new_reallocn(void* p, size_t newcount, size_t size) { size_t total; if mi_unlikely(mi_count_size_overflow(newcount, size, &total)) { - mi_try_new_handler(false); // on overflow we invoke the try_new_handler once to potentially throw std::bad_alloc + mi_try_new_handler(false); return NULL; } else { @@ -778,8 +828,8 @@ mi_decl_nodiscard void* mi_new_reallocn(void* p, size_t newcount, size_t size) { // We then set the first word of the block to `0` for regular offset aligned allocations (in `alloc-aligned.c`) // and the first word to `~0` for guarded allocations to have a correct `mi_usable_size` -static void* mi_block_ptr_set_guarded(mi_block_t* block, size_t obj_size) { - // TODO: we can still make padding work by moving it out of the guard page area +static void* mi_block_ptr_set_guarded(mi_block_t* block, size_t obj_size, size_t* usable_size) { + // todo: we can still make padding work by moving it out of the guard page area mi_page_t* const page = _mi_ptr_page(block); mi_page_set_has_interior_pointers(page, true); block->next = MI_BLOCK_TAG_GUARDED; @@ -816,13 +866,14 @@ static void* mi_block_ptr_set_guarded(mi_block_t* block, size_t obj_size) { offset = MI_PAGE_MAX_OVERALLOC_ALIGN; } uint8_t* const p = (uint8_t*)block + offset; - mi_assert_internal(p == guard_page - obj_size); + mi_assert_internal(p == guard_page - obj_size || offset >= MI_PAGE_MAX_OVERALLOC_ALIGN); + if (usable_size != NULL) { *usable_size = (guard_page - p); mi_assert_internal(mi_usable_size(p)==*usable_size); } mi_track_align(block, p, offset, obj_size); mi_track_mem_defined(block, sizeof(mi_block_t)); return p; } -mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, bool zero) mi_attr_noexcept +mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, bool zero, size_t* usable) mi_attr_noexcept { // allocate multiple of page size ending in a guard page // ensure minimal alignment requirement? @@ -834,16 +885,17 @@ mi_decl_restrict void* _mi_theap_malloc_guarded(mi_theap_t* theap, size_t size, const size_t obj_size = (mi_option_is_enabled(mi_option_guarded_precise) ? size : _mi_align_up(size, MI_MAX_ALIGN_SIZE)); const size_t bsize = _mi_align_up(_mi_align_up(obj_size, MI_MAX_ALIGN_SIZE) + sizeof(mi_block_t), MI_MAX_ALIGN_SIZE); const size_t req_size = _mi_align_up(bsize + os_page_size, os_page_size); - mi_block_t* const block = (mi_block_t*)_mi_malloc_generic(theap, req_size, 0 /* don't zero */, NULL); + mi_block_t* const block = (mi_block_t*)_mi_malloc_generic(theap, req_size, 0 /* don't zero */, usable); if (block==NULL) return NULL; - void* const p = mi_block_ptr_set_guarded(block, obj_size); + size_t usable_size = 0; + void* const p = mi_block_ptr_set_guarded(block, obj_size, &usable_size); if (p == NULL) return p; if (zero) { _mi_memzero_aligned(p,obj_size); // we have to zero here as padding might have written here (if the blocksize > reqsize + os_page_size) } // stats - mi_track_malloc(p, obj_size, zero); + mi_track_malloc(p, usable_size, zero); if (!mi_theap_is_initialized(theap)) { theap = _mi_theap_default(); } mi_theap_stat_counter_increase(theap, malloc_guarded_count, 1); #if MI_STAT>1 @@ -872,6 +924,8 @@ void* _mi_externs[] = { (void*)&mi_theap_malloc, (void*)&mi_theap_zalloc, (void*)&mi_theap_malloc_small, + (void*)&mi_theap_zalloc_small, + (void*)&mi_theap_calloc, (void*)&mi_malloc, (void*)&mi_malloc_small, (void*)&mi_zalloc, diff --git a/vendored/mimalloc/src/arena-meta.c b/vendored/mimalloc/src/arena-meta.c index cee38caa16..1007a8a04e 100644 --- a/vendored/mimalloc/src/arena-meta.c +++ b/vendored/mimalloc/src/arena-meta.c @@ -34,15 +34,16 @@ terms of the MIT license. A copy of the license can be found in the file #if MI_META_MAX_SIZE <= 4096 #error "max meta object size should be at least 4KiB" #endif +#if MI_META_BLOCK_ALIGN < MI_BCHUNK_SIZE +#error "minimal meta object alignment should be at least MI_BCHUNK_SIZE (for thread locals)" +#endif -typedef struct mi_meta_page_s { - _Atomic(struct mi_meta_page_s*) next; // a linked list of meta-data pages (never released) - mi_memid_t memid; // provenance of the meta-page memory itself - mi_bbitmap_t blocks_free; // a small bitmap with 1 bit per block. -} mi_meta_page_t; - -static mi_decl_cache_align _Atomic(mi_meta_page_t*) mi_meta_pages = MI_ATOMIC_VAR_INIT(NULL); - +struct mi_meta_page_s { + _Atomic(struct mi_meta_page_s*) next; // a linked list of meta-data pages (never released) + mi_memid_t memid; // provenance of the meta-page memory itself + // mi_subproc_t* subproc; // subprocess this page belongs to + mi_bbitmap_t blocks_free; // a small bitmap with 1 bit per block. +}; #if MI_DEBUG > 1 static mi_meta_page_t* mi_meta_page_of_ptr(void* p, size_t* block_idx) { @@ -67,11 +68,13 @@ static void* mi_meta_block_start( mi_meta_page_t* mpage, size_t block_idx ) { } // allocate a fresh meta page and add it to the global list. -static mi_meta_page_t* mi_meta_page_zalloc(void) { +static mi_meta_page_t* mi_meta_page_zalloc(mi_subproc_t* subproc) { + mi_assert_internal(subproc!=NULL); + mi_assert_internal(subproc->heap_main!=NULL); // allocate a fresh arena slice // note: careful with _mi_subproc as it may recurse into mi_tld and meta_page_zalloc again.. (same with _mi_os_numa_node()...) mi_memid_t memid; - uint8_t* base = (uint8_t*)_mi_arenas_alloc_aligned(mi_heap_main(), MI_META_PAGE_SIZE, MI_META_PAGE_ALIGN, 0, + uint8_t* base = (uint8_t*)_mi_arenas_alloc_aligned(subproc->heap_main, MI_META_PAGE_SIZE, MI_META_PAGE_ALIGN, 0, true /* commit*/, (MI_SECURE==0) /* allow large? */, NULL /* req arena */, 0 /* thread_seq */, -1 /* numa node */, &memid); if (base == NULL) return NULL; @@ -82,14 +85,14 @@ static mi_meta_page_t* mi_meta_page_zalloc(void) { // guard pages #if MI_SECURE >= 1 - _mi_os_secure_guard_page_set_at(base, memid); - _mi_os_secure_guard_page_set_before(base + MI_META_PAGE_SIZE, memid); + _mi_os_secure_guard_page_set_at(subproc, base, memid); + _mi_os_secure_guard_page_set_before(subproc, base + MI_META_PAGE_SIZE, memid); #endif // initialize the page and free block bitmap mi_meta_page_t* mpage = (mi_meta_page_t*)(base + _mi_os_secure_guard_page_size()); mpage->memid = memid; - mi_bbitmap_init(&mpage->blocks_free, MI_META_BLOCKS_PER_PAGE, true /* already_zero */); + mi_bbitmap_init(subproc, &mpage->blocks_free, MI_META_BLOCKS_PER_PAGE, true /* already_zero */); const size_t mpage_size = offsetof(mi_meta_page_t,blocks_free) + mi_bbitmap_size(MI_META_BLOCKS_PER_PAGE, NULL); const size_t info_blocks = _mi_divide_up(mpage_size,MI_META_BLOCK_SIZE); const size_t guard_blocks = _mi_divide_up(_mi_os_secure_guard_page_size(), MI_META_BLOCK_SIZE); @@ -98,23 +101,24 @@ static mi_meta_page_t* mi_meta_page_zalloc(void) { // push atomically in front of the meta page list // (note: there is no ABA issue since we never free meta-pages) - mi_meta_page_t* old = mi_atomic_load_ptr_acquire(mi_meta_page_t,&mi_meta_pages); + mi_meta_page_t* old = mi_atomic_load_ptr_acquire(mi_meta_page_t,&subproc->meta_pages); do { mi_atomic_store_ptr_release(mi_meta_page_t, &mpage->next, old); - } while(!mi_atomic_cas_ptr_weak_acq_rel(mi_meta_page_t,&mi_meta_pages,&old,mpage)); + } while(!mi_atomic_cas_ptr_weak_acq_rel(mi_meta_page_t,&subproc->meta_pages,&old,mpage)); return mpage; } // allocate meta-data -mi_decl_noinline void* _mi_meta_zalloc( size_t size, mi_memid_t* pmemid ) +mi_decl_noinline void* _mi_meta_zalloc( mi_subproc_t* subproc, size_t size, mi_memid_t* pmemid ) { mi_assert_internal(pmemid != NULL); + mi_assert_internal(subproc!=NULL); size = _mi_align_up(size,MI_META_BLOCK_SIZE); if (size == 0 || size > MI_META_MAX_SIZE) return NULL; const size_t block_count = _mi_divide_up(size,MI_META_BLOCK_SIZE); mi_assert_internal(block_count > 0 && block_count < MI_BCHUNK_BITS); - mi_meta_page_t* mpage0 = mi_atomic_load_ptr_acquire(mi_meta_page_t,&mi_meta_pages); + mi_meta_page_t* mpage0 = mi_atomic_load_ptr_acquire(mi_meta_page_t,&subproc->meta_pages); mi_meta_page_t* mpage = mpage0; while (mpage != NULL) { size_t block_idx; @@ -128,12 +132,12 @@ mi_decl_noinline void* _mi_meta_zalloc( size_t size, mi_memid_t* pmemid ) } } // failed to find space in existing pages - if (mi_atomic_load_ptr_acquire(mi_meta_page_t,&mi_meta_pages) != mpage0) { + if (mi_atomic_load_ptr_acquire(mi_meta_page_t,&subproc->meta_pages) != mpage0) { // the page list was updated by another thread in the meantime, retry - return _mi_meta_zalloc(size,pmemid); + return _mi_meta_zalloc(subproc,size,pmemid); } // otherwise, allocate a fresh metapage and try once more - mpage = mi_meta_page_zalloc(); + mpage = mi_meta_page_zalloc(subproc); if (mpage != NULL) { size_t block_idx; if (mi_bbitmap_try_find_and_clearN(&mpage->blocks_free, 0, block_count, &block_idx)) { @@ -143,11 +147,11 @@ mi_decl_noinline void* _mi_meta_zalloc( size_t size, mi_memid_t* pmemid ) } } // if all this failed, allocate from the OS - return _mi_os_alloc(size, pmemid); + return _mi_os_zalloc(subproc, size, pmemid); } // free meta-data -mi_decl_noinline void _mi_meta_free(void* p, size_t size, mi_memid_t memid) { +mi_decl_noinline void _mi_meta_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid) { if (p==NULL) return; if (memid.memkind == MI_MEM_META) { mi_assert_internal(_mi_divide_up(size, MI_META_BLOCK_SIZE) == memid.mem.meta.block_count); @@ -162,14 +166,14 @@ mi_decl_noinline void _mi_meta_free(void* p, size_t size, mi_memid_t memid) { mi_bbitmap_setN(&mpage->blocks_free, block_idx, block_count); } else { - _mi_arenas_free(p,size,memid); + _mi_arenas_free(subproc, p, size, memid); } } // used for debug output -bool _mi_meta_is_meta_page(void* p) +bool _mi_meta_is_meta_page(mi_subproc_t* subproc, void* p) { - mi_meta_page_t* mpage0 = mi_atomic_load_ptr_acquire(mi_meta_page_t, &mi_meta_pages); + mi_meta_page_t* mpage0 = mi_atomic_load_ptr_acquire(mi_meta_page_t, &subproc->meta_pages); mi_meta_page_t* mpage = mpage0; while (mpage != NULL) { if ((void*)mpage == p) return true; diff --git a/vendored/mimalloc/src/arena.c b/vendored/mimalloc/src/arena.c index a5686d5549..fc98a3691f 100644 --- a/vendored/mimalloc/src/arena.c +++ b/vendored/mimalloc/src/arena.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2019-2025, Microsoft Research, Daan Leijen +Copyright (c) 2019-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -21,9 +21,13 @@ The arena allocation needs to be thread safe and we use an atomic bitmap to allo #include "mimalloc.h" #include "mimalloc/internal.h" -#include "mimalloc/prim.h" +#include "mimalloc/prim-tls.h" #include "bitmap.h" +#if (MI_ARENA_MAX_SIZE > MI_MAX_ALIGN_SIZE*UINT32_MAX) +#error "The page_t.page_ma_offset field is not large enough to cover a full arena" +#endif + /* ----------------------------------------------------------- Arena id's ----------------------------------------------------------- */ @@ -99,23 +103,24 @@ static size_t mi_arena_max_object_size(void) { if (max_size <= MI_ARENA_MIN_OBJ_SIZE) { return MI_ARENA_MIN_OBJ_SIZE; } - else if (max_size >= MI_ARENA_MAX_SIZE - MI_BCHUNK_SIZE) { // minus a bchunk to accommodate meta info - return (MI_ARENA_MAX_SIZE - MI_BCHUNK_SIZE); + else if (max_size >= MI_ARENA_MAX_SIZE - (MI_BCHUNK_BITS*MI_ARENA_SLICE_SIZE)) { // minus an initial chunk to accommodate meta info + return (MI_ARENA_MAX_SIZE - (MI_BCHUNK_BITS*MI_ARENA_SLICE_SIZE)); } else { return max_size; } } -mi_decl_nodiscard static bool mi_arena_commit(mi_arena_t* arena, void* start, size_t size, bool* is_zero, size_t already_committed) { +mi_decl_nodiscard static bool mi_arena_commit(mi_subproc_t* subproc, mi_arena_t* arena, void* start, size_t size, bool* is_zero, size_t already_committed) { + mi_assert_internal(subproc!=NULL); if (arena != NULL && arena->commit_fun != NULL) { return (*arena->commit_fun)(true, start, size, is_zero, arena->commit_fun_arg); } else if (already_committed > 0) { - return _mi_os_commit_ex(start, size, is_zero, already_committed); + return _mi_os_commit_ex(subproc, start, size, is_zero, already_committed); } else { - return _mi_os_commit(start, size, is_zero); + return _mi_os_commit(subproc, start, size, is_zero); } } @@ -215,6 +220,8 @@ static size_t mi_page_full_size(mi_page_t* page) { static mi_decl_noinline void* mi_arena_try_alloc_at( mi_arena_t* arena, size_t slice_count, bool commit, size_t tseq, mi_memid_t* memid) { + mi_assert_internal(arena!=NULL); + mi_assert_internal(slice_count>0); size_t slice_index; if (!mi_bbitmap_try_find_and_clearN(arena->slices_free, tseq, slice_count, &slice_index)) return NULL; @@ -231,6 +238,10 @@ static mi_decl_noinline void* mi_arena_try_alloc_at( mi_assert_internal(already_dirty <= touched_slices); touched_slices -= already_dirty; } + else { + // todo: properly count touched pages with a separate bitmap? + touched_slices = 0; + } // set commit state if (commit) { @@ -239,7 +250,7 @@ static mi_decl_noinline void* mi_arena_try_alloc_at( if (already_committed < slice_count) { // not all committed, try to commit now bool commit_zero = false; - if (!mi_arena_commit(arena, p, mi_size_of_slices(slice_count), &commit_zero, mi_size_of_slices(slice_count - already_committed))) { + if (!mi_arena_commit(arena->subproc, arena, p, mi_size_of_slices(slice_count), &commit_zero, mi_size_of_slices(slice_count - already_committed))) { // if the commit fails, release ownership, and return NULL; // note: this does not roll back dirty bits but that is ok. mi_bbitmap_setN(arena->slices_free, slice_index, slice_count); @@ -264,7 +275,7 @@ static mi_decl_noinline void* mi_arena_try_alloc_at( } else { // already fully committed. - _mi_os_reuse(p, mi_size_of_slices(slice_count)); + _mi_os_reuse(arena->subproc, p, mi_size_of_slices(slice_count)); // if the OS has overcommit, and this is the first time we access these pages, then // count the commit now (as at arena reserve we didn't count those commits as these are on-demand) if (_mi_os_has_overcommit() && touched_slices > 0 && !arena->memid.is_pinned /* huge pages, issue #1236 */) { @@ -540,6 +551,7 @@ static mi_decl_noinline void* mi_arenas_try_alloc( // Allocate from the OS (if allowed) static void* mi_arena_os_alloc_aligned( + mi_subproc_t* subproc, size_t size, size_t alignment, size_t align_offset, bool commit, bool allow_large, mi_arena_id_t req_arena_id, mi_memid_t* memid) @@ -551,10 +563,10 @@ static void* mi_arena_os_alloc_aligned( } if (align_offset > 0) { - return _mi_os_alloc_aligned_at_offset(size, alignment, align_offset, commit, allow_large, memid); + return _mi_os_alloc_aligned_at_offset(subproc, size, alignment, align_offset, commit, allow_large, memid); } else { - return _mi_os_alloc_aligned(size, alignment, commit, allow_large, memid); + return _mi_os_alloc_aligned(subproc, size, alignment, commit, allow_large, memid); } } @@ -579,7 +591,7 @@ void* _mi_arenas_alloc_aligned( mi_heap_t* heap, } // fall back to the OS - void* p = mi_arena_os_alloc_aligned(size, alignment, align_offset, commit, allow_large, req_arena, memid); + void* p = mi_arena_os_alloc_aligned(heap->subproc, size, alignment, align_offset, commit, allow_large, req_arena, memid); return p; } @@ -655,7 +667,8 @@ static mi_arena_t* mi_page_arena_pages(mi_page_t* page, size_t* slice_index, siz mi_arena_t* const arena = mi_arena_from_memid(page->memid, slice_index, slice_count); mi_assert_internal(arena != NULL); if (parena_pages != NULL) { - mi_arena_pages_t* const arena_pages = mi_heap_arena_pages(mi_page_heap(page), arena); + mi_heap_t* heap = mi_page_heap(page); + mi_arena_pages_t* const arena_pages = mi_heap_arena_pages(heap, arena); mi_assert_internal(arena_pages != NULL); mi_assert_internal(slice_index==NULL || mi_bitmap_is_set(arena_pages->pages, *slice_index)); *parena_pages = arena_pages; @@ -697,7 +710,9 @@ static mi_page_t* mi_arenas_page_try_find_abandoned(mi_theap_t* theap, size_t sl MI_UNUSED(slice_count); const size_t bin = _mi_bin(block_size); - mi_assert_internal(bin < MI_BIN_COUNT); + if (bin >= MI_ARENA_BIN_COUNT) { + return NULL; // singleton page size + } // any abandoned in our size class? mi_assert_internal(heap != NULL); @@ -743,8 +758,11 @@ static mi_page_t* mi_arenas_page_try_find_abandoned(mi_theap_t* theap, size_t sl return NULL; } -static uint8_t* mi_arenas_page_alloc_fresh_area(mi_theap_t* theap, size_t slice_count, size_t block_size, size_t block_alignment, bool os_align, bool commit, mi_memid_t* memid) { +static uint8_t* mi_arenas_page_alloc_fresh_area(mi_theap_t* theap, size_t slice_count, size_t block_size, size_t block_alignment, bool os_align, bool commit, mi_memid_t* memid, mi_arena_pages_t** parena_pages ) { MI_UNUSED_RELEASE(block_size); + mi_assert_internal(parena_pages!=NULL); + + *parena_pages = NULL; const bool allow_large = (MI_SECURE < 5); // 5 = guard page at end of each arena page const size_t page_alignment = MI_ARENA_SLICE_ALIGN; @@ -764,15 +782,16 @@ static uint8_t* mi_arenas_page_alloc_fresh_area(mi_theap_t* theap, size_t slice_ start = (uint8_t*)mi_arenas_try_alloc(heap, slice_count, page_alignment, commit, allow_large, req_arena, tld->thread_seq, numa_node, memid); if (start != NULL) { mi_arena_pages_t* const arena_pages = mi_heap_ensure_arena_pages(heap, memid->mem.arena.arena); + *parena_pages = arena_pages; if (arena_pages==NULL) { - _mi_arenas_free(start, mi_size_of_slices(slice_count), *memid); // roll back + _mi_arenas_free(heap->subproc, start, mi_size_of_slices(slice_count), *memid); // roll back start = NULL; } else { // note: the following assert should hold if we could check it atomically, but in a concurrent setting we may already allocate in slice_count // mi_assert_internal(mi_bitmap_is_clearN(arena_pages->pages, memid->mem.arena.slice_index, memid->mem.arena.slice_count)); mi_assert_internal(mi_bitmap_is_clear(arena_pages->pages, memid->mem.arena.slice_index)); - mi_bitmap_set(arena_pages->pages, memid->mem.arena.slice_index); + // don't set yet: mi_bitmap_set(arena_pages->pages, memid->mem.arena.slice_index); } } } @@ -782,10 +801,10 @@ static uint8_t* mi_arenas_page_alloc_fresh_area(mi_theap_t* theap, size_t slice_ if (os_align) { // note: slice_count already includes the page mi_assert_internal(slice_count >= mi_slice_count_of_size(block_size) + mi_slice_count_of_size(page_alignment)); - start = (uint8_t*)mi_arena_os_alloc_aligned(alloc_size, block_alignment, page_alignment /* align offset */, commit, allow_large, req_arena, memid); + start = (uint8_t*)mi_arena_os_alloc_aligned(heap->subproc, alloc_size, block_alignment, page_alignment /* align offset */, commit, allow_large, req_arena, memid); } else { - start = (uint8_t*)mi_arena_os_alloc_aligned(alloc_size, page_alignment, 0 /* align offset */, commit, allow_large, req_arena, memid); + start = (uint8_t*)mi_arena_os_alloc_aligned(heap->subproc, alloc_size, page_alignment, 0 /* align offset */, commit, allow_large, req_arena, memid); } } @@ -797,49 +816,52 @@ static uint8_t* mi_arenas_page_alloc_fresh_area(mi_theap_t* theap, size_t slice_ static size_t mi_page_block_start(size_t block_size, bool os_align) { + size_t offset; #if MI_GUARDED // in a guarded build, we align pages with blocks a multiple of an OS page size, to the OS page size // this ensures that all blocks in such pages are OS page size aligned (which is needed for the guard pages) const size_t os_page_size = _mi_os_page_size(); mi_assert_internal(MI_PAGE_ALIGN >= os_page_size); if (!os_align && block_size % os_page_size == 0 && block_size > os_page_size /* at least 2 or more */ ) { - return _mi_align_up(mi_page_info_size(), os_page_size); + offset = _mi_align_up(mi_page_info_size(), os_page_size); } else #endif if (os_align) { - return MI_PAGE_ALIGN; + offset = MI_PAGE_ALIGN; } else if (_mi_is_power_of_two(block_size) && block_size <= MI_PAGE_MAX_START_BLOCK_ALIGN2) { // naturally align power-of-2 blocks up to MI_PAGE_MAX_START_BLOCK_ALIGN2 size (4KiB) - return _mi_align_up(mi_page_info_size(), block_size); + offset = _mi_align_up(mi_page_info_size(), block_size); } else if (block_size != 0 && (block_size % MI_PAGE_OSPAGE_BLOCK_ALIGN2) == 0) { // also align large pages that are a multiple of MI_PAGE_OSPAGE_BLOCK_ALIGN2 (4KiB) - return _mi_align_up(mi_page_info_size(), MI_PAGE_OSPAGE_BLOCK_ALIGN2); + offset = _mi_align_up(mi_page_info_size(), MI_PAGE_OSPAGE_BLOCK_ALIGN2); } else { // otherwise start after the info - return mi_page_info_size(); + offset = mi_page_info_size(); } + return _mi_align_up(offset,MI_MAX_ALIGN_SIZE); } +// Free a page without modifying page_bin stats +static void mi_arenas_page_free_prim(mi_page_t* page); + // Allocate a fresh page static mi_page_t* mi_arenas_page_alloc_fresh(mi_theap_t* theap, size_t slice_count, size_t block_size, size_t block_alignment, bool commit) { - const bool os_align = (block_alignment > MI_PAGE_MAX_OVERALLOC_ALIGN); - const size_t alloc_size = mi_size_of_slices(slice_count); - mi_memid_t memid = _mi_memid_none(); - uint8_t* const slice_start = mi_arenas_page_alloc_fresh_area(theap,slice_count,block_size,block_alignment,os_align,commit,&memid); + const bool os_align = (block_alignment > MI_PAGE_MAX_OVERALLOC_ALIGN); + const size_t alloc_size = mi_size_of_slices(slice_count); + mi_memid_t memid = _mi_memid_none(); + mi_arena_pages_t* arena_pages = NULL; + uint8_t* const slice_start = mi_arenas_page_alloc_fresh_area(theap,slice_count,block_size,block_alignment,os_align,commit,&memid,&arena_pages); if (!slice_start) return NULL; // guard page at the end of mimalloc page? #if MI_SECURE>=5 mi_assert(alloc_size > _mi_os_secure_guard_page_size()); const size_t page_noguard_size = alloc_size - _mi_os_secure_guard_page_size(); - if (memid.initially_committed) { - _mi_os_secure_guard_page_set_at(slice_start + page_noguard_size, memid); - } #else const size_t page_noguard_size = alloc_size; #endif @@ -881,6 +903,7 @@ static mi_page_t* mi_arenas_page_alloc_fresh(mi_theap_t* theap, size_t slice_cou page = (mi_page_t*)slice_start; block_start = mi_page_block_start(block_size, os_align); } + mi_assert_internal(block_start % MI_MAX_ALIGN_SIZE == 0); // commit first block? size_t commit_size = 0; @@ -888,15 +911,25 @@ static mi_page_t* mi_arenas_page_alloc_fresh(mi_theap_t* theap, size_t slice_cou commit_size = _mi_align_up(block_start + block_size, MI_PAGE_MIN_COMMIT_SIZE); if (commit_size > page_noguard_size) { commit_size = page_noguard_size; } bool is_zero = false; - if mi_unlikely(!mi_arena_commit( mi_memid_arena(memid), slice_start, commit_size, &is_zero, 0)) { - _mi_arenas_free(slice_start, alloc_size, memid); + if mi_unlikely(!mi_arena_commit( _mi_theap_subproc(theap), mi_memid_arena(memid), slice_start, commit_size, &is_zero, 0)) { + _mi_arenas_free(_mi_theap_subproc(theap), slice_start, alloc_size, memid); return NULL; } } + // now we can finish initalization and use `mi_arenas_free_page_prim` on error + + // zero initialize the page meta data if (!memid.initially_zero && !page_meta_is_separate) { _mi_memzero_aligned(page, sizeof(*page)); } + // set the guard page + #if MI_SECURE>=5 + if (memid.initially_committed) { + _mi_os_secure_guard_page_set_at(_mi_theap_subproc(theap), slice_start + page_noguard_size, memid); + } + #endif + // claimed free slices: initialize the page partly if (!memid.initially_zero && memid.initially_committed) { mi_track_mem_undefined(slice_start, slice_count * MI_ARENA_SLICE_SIZE); @@ -916,23 +949,43 @@ static mi_page_t* mi_arenas_page_alloc_fresh(mi_theap_t* theap, size_t slice_cou const size_t reserved = (os_align ? 1 : (page_noguard_size - block_start) / block_size); mi_assert_internal(reserved > 0 && reserved <= UINT16_MAX); - // initialize - page->reserved = (uint16_t)reserved; - page->page_start = slice_start + block_start; + // initialize the page start + uint8_t* const start = slice_start + block_start; + mi_assert_internal(start > (uint8_t*)page); + const size_t offset = start - (uint8_t*)page; + mi_assert_internal((offset % MI_MAX_ALIGN_SIZE) == 0 && (offset / MI_MAX_ALIGN_SIZE) <= UINT32_MAX); + page->page_ma_offset = (uint32_t)(offset / MI_MAX_ALIGN_SIZE); + + // initialize page meta-data + page->reserved = (uint16_t)reserved; page->block_size = block_size; - page->slice_committed = commit_size; page->memid = memid; page->free_is_zero = memid.initially_zero; + + mi_assert_internal((commit && commit_size==0) || (!commit && commit_size < UINT32_MAX)); + page->slice_committed = (uint32_t)commit_size; + + page->heap = _mi_theap_heap(theap); + mi_page_set_theap(page,theap); + // mi_assert_internal(mi_page_theap(page) == _mi_heap_theap_peek(page->heap)); + mi_assert_internal(page->free==NULL); mi_assert_internal(page_meta_is_separate == mi_page_meta_is_separated(page)); mi_assert_internal(mi_page_slice_start(page) == slice_start); + mi_assert_internal(mi_page_size(page) <= page_noguard_size); + + // now register in the arena_pages + if (arena_pages!=NULL) { + mi_assert_internal(memid.memkind == MI_MEM_ARENA); + mi_bitmap_set(arena_pages->pages, memid.mem.arena.slice_index); + } // and own it mi_page_claim_ownership(page); // register in the page map if mi_unlikely(!_mi_page_map_register(page)) { - _mi_arenas_free( slice_start, alloc_size, memid ); + mi_arenas_page_free_prim(page); return NULL; } @@ -943,7 +996,7 @@ static mi_page_t* mi_arenas_page_alloc_fresh(mi_theap_t* theap, size_t slice_cou mi_assert_internal(_mi_is_aligned(mi_page_slice_start(page),MI_PAGE_ALIGN)); mi_assert_internal(_mi_ptr_page(mi_page_start(page))==page); mi_assert_internal(mi_page_block_size(page) == block_size); - mi_assert_internal(mi_page_is_abandoned(page)); + // mi_assert_internal(mi_page_is_abandoned(page)); mi_assert_internal(mi_page_is_owned(page)); return page; @@ -961,13 +1014,14 @@ static mi_page_t* mi_arenas_page_regular_alloc(mi_theap_t* theap, size_t slice_c // 2. find a free block, potentially allocating a new arena const long commit_on_demand = mi_option_get(mi_option_page_commit_on_demand); const bool commit = (slice_count <= mi_slice_count_of_size(MI_PAGE_MIN_COMMIT_SIZE) || // always commit small pages - (commit_on_demand == 2 && _mi_os_has_overcommit()) || (commit_on_demand == 0)); + (slice_count >= mi_slice_count_of_size(UINT32_MAX)) || // always commit pages too large to hold a 32-bit slice_committed + (commit_on_demand == 2 && _mi_os_has_overcommit()) || (commit_on_demand == 0)); page = mi_arenas_page_alloc_fresh(theap, slice_count, block_size, 1, commit); if (page == NULL) return NULL; mi_assert_internal(page->memid.memkind != MI_MEM_ARENA || page->memid.mem.arena.slice_count == slice_count); if (!_mi_page_init(theap, page)) { - _mi_arenas_free( page, mi_page_full_size(page), page->memid); + _mi_arenas_page_free(page,theap); return NULL; } @@ -990,7 +1044,7 @@ static mi_page_t* mi_arenas_page_singleton_alloc(mi_theap_t* theap, size_t block mi_assert(page->reserved == 1); if (!_mi_page_init(theap, page)) { - _mi_arenas_free( page, mi_page_full_size(page), page->memid); + _mi_arenas_page_free(page,theap); return NULL; } @@ -1000,6 +1054,8 @@ static mi_page_t* mi_arenas_page_singleton_alloc(mi_theap_t* theap, size_t block mi_page_t* _mi_arenas_page_alloc(mi_theap_t* theap, size_t block_size, size_t block_alignment) { mi_page_t* page; + // semi static assert: ensure that all non-singleton block size bins are covered. + mi_assert(_mi_bin(MI_LARGE_MAX_OBJ_SIZE) < MI_ARENA_BIN_COUNT); if mi_unlikely(block_alignment > MI_PAGE_MAX_OVERALLOC_ALIGN) { mi_assert_internal(_mi_is_power_of_two(block_alignment)); page = mi_arenas_page_singleton_alloc(theap, block_size, block_alignment); @@ -1029,25 +1085,13 @@ mi_page_t* _mi_arenas_page_alloc(mi_theap_t* theap, size_t block_size, size_t bl return page; } -void _mi_arenas_page_free(mi_page_t* page, mi_theap_t* current_theapx) { +static void mi_arenas_page_free_prim(mi_page_t* page) { mi_assert_internal(_mi_is_aligned(mi_page_slice_start(page), MI_PAGE_ALIGN)); mi_assert_internal(_mi_ptr_page(mi_page_start(page))==page); mi_assert_internal(mi_page_is_owned(page)); mi_assert_internal(mi_page_all_free(page)); - mi_assert_internal(mi_page_is_abandoned(page)); mi_assert_internal(page->next==NULL && page->prev==NULL); - mi_assert_internal(current_theapx == NULL || _mi_thread_id()==current_theapx->tld->thread_id); - - if (current_theapx != NULL) { - mi_theap_stat_decrease(current_theapx, page_bins[_mi_page_stats_bin(page)], 1); - mi_theap_stat_decrease(current_theapx, pages, 1); - } - else { - mi_heap_t* const heap = mi_page_heap(page); - mi_heap_stat_decrease(heap, page_bins[_mi_page_stats_bin(page)], 1); - mi_heap_stat_decrease(heap, pages, 1); - } - + #if MI_DEBUG>1 if (page->memid.memkind==MI_MEM_ARENA && !mi_page_is_full(page)) { size_t bin = _mi_bin(mi_page_block_size(page)); @@ -1057,7 +1101,7 @@ void _mi_arenas_page_free(mi_page_t* page, mi_theap_t* current_theapx) { mi_arena_t* const arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); mi_assert_internal(mi_bbitmap_is_clearN(arena->slices_free, slice_index, slice_count)); mi_assert_internal(page->slice_committed > 0 || mi_bitmap_is_setN(arena->slices_committed, slice_index, slice_count)); - mi_assert_internal(mi_bitmap_is_clearN(arena_pages->pages_abandoned[bin], slice_index, 1)); + mi_assert_internal(bin >= MI_ARENA_BIN_COUNT || mi_bitmap_is_clearN(arena_pages->pages_abandoned[bin], slice_index, 1)); mi_assert_internal(mi_bitmap_is_setN(arena_pages->pages, slice_index, 1)); // note: we cannot check for `!mi_page_is_abandoned_and_mapped` since that may // be (temporarily) not true if the free happens while trying to reclaim @@ -1065,22 +1109,25 @@ void _mi_arenas_page_free(mi_page_t* page, mi_theap_t* current_theapx) { } #endif + // unregister page + _mi_page_map_unregister(page); + // recommit guard page at the end? // we must do this since we may later allocate large spans over this page and cannot have a guard page in between #if MI_SECURE >= 5 if (!page->memid.is_pinned) { - _mi_os_secure_guard_page_reset_before(mi_page_slice_start(page) + mi_page_full_size(page), page->memid); + _mi_os_secure_guard_page_reset_before(mi_page_subproc(page), mi_page_slice_start(page) + mi_page_full_size(page), page->memid); } #endif - // unregister page - _mi_page_map_unregister(page); + // and free if (page->memid.memkind == MI_MEM_ARENA) { mi_arena_pages_t* arena_pages; size_t slice_index; size_t slice_count; MI_UNUSED(slice_count); mi_arena_t* const arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); mi_assert_internal(arena_pages!=NULL); + mi_assert_internal(arena->subproc == mi_page_subproc(page)); mi_bitmap_clear(arena_pages->pages, slice_index); if (page->slice_committed > 0) { // if committed on-demand, set the commit bits to account commit properly @@ -1103,7 +1150,28 @@ void _mi_arenas_page_free(mi_page_t* page, mi_theap_t* current_theapx) { } } if (mi_page_meta_is_separated(page)) { page->block_size = 0; } // for assertion checking - _mi_arenas_free( mi_page_slice_start(page), mi_page_full_size(page), page->memid); + _mi_arenas_free( mi_page_subproc(page), mi_page_slice_start(page), mi_page_full_size(page), page->memid); +} + +void _mi_arenas_page_free(mi_page_t* page, mi_theap_t* current_theapx) { + mi_assert_internal(_mi_is_aligned(mi_page_slice_start(page), MI_PAGE_ALIGN)); + mi_assert_internal(_mi_ptr_page(mi_page_start(page))==page); + mi_assert_internal(mi_page_is_owned(page)); + mi_assert_internal(mi_page_all_free(page)); + mi_assert_internal(mi_page_is_abandoned(page)); + mi_assert_internal(page->next==NULL && page->prev==NULL); + mi_assert_internal(current_theapx == NULL || _mi_thread_id()==current_theapx->tld->thread_id); + + if (current_theapx != NULL) { + mi_theap_stat_decrease(current_theapx, page_bins[_mi_page_stats_bin(page)], 1); + mi_theap_stat_decrease(current_theapx, pages, 1); + } + else { + mi_heap_t* const heap = mi_page_heap(page); + mi_heap_stat_decrease(heap, page_bins[_mi_page_stats_bin(page)], 1); + mi_heap_stat_decrease(heap, pages, 1); + } + mi_arenas_page_free_prim(page); } /* ----------------------------------------------------------- @@ -1120,41 +1188,47 @@ void _mi_arenas_page_abandon(mi_page_t* page, mi_theap_t* current_theap) { mi_assert_internal(_mi_thread_id()==current_theap->tld->thread_id); // mi_assert_internal(current_theap == _mi_page_associated_theap(page)); - mi_heap_t* heap = mi_page_heap(page); mi_assert_internal(heap==_mi_theap_heap(current_theap)); + // add to abandoned? + mi_heap_t* heap = mi_page_heap(page); + mi_assert_internal(heap==_mi_theap_heap(current_theap)); if (page->memid.memkind==MI_MEM_ARENA && !mi_page_is_full(page)) { // make available for allocations size_t bin = _mi_bin(mi_page_block_size(page)); - size_t slice_index; - size_t slice_count; - mi_arena_pages_t* arena_pages = NULL; - mi_arena_t* const arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); MI_UNUSED(arena); - - mi_assert_internal(!mi_page_is_singleton(page)); - mi_assert_internal(mi_bbitmap_is_clearN(arena->slices_free, slice_index, slice_count)); - mi_assert_internal(page->slice_committed > 0 || mi_bitmap_is_setN(arena->slices_committed, slice_index, slice_count)); - mi_assert_internal(mi_bitmap_is_setN(arena->slices_dirty, slice_index, slice_count)); - - mi_page_set_abandoned_mapped(page); - const bool was_clear = mi_bitmap_set(arena_pages->pages_abandoned[bin], slice_index); - MI_UNUSED(was_clear); mi_assert_internal(was_clear); - mi_atomic_increment_relaxed(&heap->abandoned_count[bin]); - mi_theap_stat_increase(current_theap, pages_abandoned, 1); + mi_assert_internal(bin < MI_ARENA_BIN_COUNT); + if (bin < MI_ARENA_BIN_COUNT) { // paranoia + size_t slice_index; + size_t slice_count; + mi_arena_pages_t* arena_pages = NULL; + mi_arena_t* const arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); MI_UNUSED(arena); + + mi_assert_internal(!mi_page_is_singleton(page)); + mi_assert_internal(mi_bbitmap_is_clearN(arena->slices_free, slice_index, slice_count)); + mi_assert_internal(page->slice_committed > 0 || mi_bitmap_is_setN(arena->slices_committed, slice_index, slice_count)); + mi_assert_internal(mi_bitmap_is_setN(arena->slices_dirty, slice_index, slice_count)); + + mi_page_set_abandoned_mapped(page); + const bool was_clear = mi_bitmap_set(arena_pages->pages_abandoned[bin], slice_index); + MI_UNUSED(was_clear); mi_assert_internal(was_clear); + mi_atomic_increment_relaxed(&heap->abandoned_count[bin]); + mi_theap_stat_increase(current_theap, pages_abandoned, 1); + mi_abandoned_page_unown(page, current_theap); + return; + } } - else { - // page is full (or a singleton), or the page is OS/externally allocated - // leave as is; it will be reclaimed when an object is free'd in the page - // but for non-arena pages, add to the subproc list so these can be visited - if (page->memid.memkind != MI_MEM_ARENA && mi_option_is_enabled(mi_option_visit_abandoned)) { - mi_lock(&heap->os_abandoned_pages_lock) { - // push in front - page->prev = NULL; - page->next = heap->os_abandoned_pages; - if (page->next != NULL) { page->next->prev = page; } - heap->os_abandoned_pages = page; - } + // otherwise, + // page is full (or a singleton), or the page is OS/externally allocated + // leave as is; it will be reclaimed when an object is free'd in the page + // but for non-arena pages, add to the subproc list so these can be visited + if (page->memid.memkind != MI_MEM_ARENA) { + mi_lock(&heap->os_abandoned_pages_lock) { + // push in front + page->prev = NULL; + page->next = heap->os_abandoned_pages; + if (page->next != NULL) { page->next->prev = page; } + heap->os_abandoned_pages = page; } - mi_theap_stat_increase(current_theap, pages_abandoned, 1); } + mi_theap_stat_increase(current_theap, pages_abandoned, 1); mi_abandoned_page_unown(page, current_theap); } @@ -1199,7 +1273,8 @@ void _mi_arenas_page_unabandon(mi_page_t* page, mi_theap_t* current_theapx) { if (mi_page_is_abandoned_mapped(page)) { mi_assert_internal(page->memid.memkind==MI_MEM_ARENA); // remove from the abandoned map - size_t bin = _mi_bin(mi_page_block_size(page)); + const size_t bin = _mi_bin(mi_page_block_size(page)); + mi_assert_internal(bin < MI_ARENA_BIN_COUNT); size_t slice_index; size_t slice_count; mi_arena_pages_t* arena_pages; @@ -1209,14 +1284,14 @@ void _mi_arenas_page_unabandon(mi_page_t* page, mi_theap_t* current_theapx) { mi_assert_internal(page->slice_committed > 0 || mi_bitmap_is_setN(arena->slices_committed, slice_index, slice_count)); // this busy waits until a concurrent reader (from alloc_abandoned) is done - mi_bitmap_clear_once_set(arena_pages->pages_abandoned[bin], slice_index); + mi_bitmap_clear_once_set(arena->subproc, arena_pages->pages_abandoned[bin], slice_index); mi_page_clear_abandoned_mapped(page); mi_atomic_decrement_relaxed(&heap->abandoned_count[bin]); } else { // page is full (or a singleton), page is OS allocated // if not an arena page, remove from the subproc os pages list - if (page->memid.memkind != MI_MEM_ARENA && mi_option_is_enabled(mi_option_visit_abandoned)) { + if (page->memid.memkind != MI_MEM_ARENA) { mi_lock(&heap->os_abandoned_pages_lock) { if (page->prev != NULL) { page->prev->next = page->next; } if (page->next != NULL) { page->next->prev = page->prev; } @@ -1241,7 +1316,7 @@ void _mi_arenas_page_unabandon(mi_page_t* page, mi_theap_t* current_theapx) { static void mi_arena_schedule_purge(mi_arena_t* arena, size_t slice_index, size_t slices); static void mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, size_t tseq); -void _mi_arenas_free(void* p, size_t size, mi_memid_t memid) { +void _mi_arenas_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid) { if (p==NULL) return; if (size==0) return; @@ -1250,26 +1325,28 @@ void _mi_arenas_free(void* p, size_t size, mi_memid_t memid) { if (mi_memkind_is_os(memid.memkind)) { // was a direct OS allocation, pass through - _mi_os_free(p, size, memid); + _mi_os_free(subproc, p, size, memid); } else if (memid.memkind == MI_MEM_ARENA) { // allocated in an arena size_t slice_count; size_t slice_index; mi_arena_t* arena = mi_arena_from_memid(memid, &slice_index, &slice_count); + mi_assert_internal(arena!=NULL); + mi_assert_internal(arena->subproc == subproc); mi_assert_internal((size%MI_ARENA_SLICE_SIZE)==0); mi_assert_internal((slice_count*MI_ARENA_SLICE_SIZE)==size); mi_assert_internal(mi_arena_slice_start(arena,slice_index) <= (uint8_t*)p); mi_assert_internal(mi_arena_slice_start(arena,slice_index) + mi_size_of_slices(slice_count) > (uint8_t*)p); // checks if (arena == NULL) { - _mi_error_message(EINVAL, "trying to free from an invalid arena: %p, size %zu, memid: 0x%zx\n", p, size, memid); + _mi_error_message(EINVAL, "trying to free from an invalid arena: %p, size %zu, memkind: 0x%x\n", p, size, memid.memkind); return; } mi_assert_internal(slice_index < arena->slice_count); mi_assert_internal(slice_index >= mi_arena_info_slices(arena)); if (slice_index < mi_arena_info_slices(arena) || slice_index >= arena->slice_count) { - _mi_error_message(EINVAL, "trying to free from an invalid arena block: %p, size %zu, memid: 0x%zx\n", p, size, memid); + _mi_error_message(EINVAL, "trying to free from an invalid arena block: %p, size %zu, memkind: 0x%x\n", p, size, memid.memkind); return; } @@ -1287,7 +1364,7 @@ void _mi_arenas_free(void* p, size_t size, mi_memid_t memid) { }; } else if (memid.memkind == MI_MEM_META) { - _mi_meta_free(p, size, memid); + _mi_meta_free(subproc, p, size, memid); } else { // arena was none, external, or static; nothing to do @@ -1328,10 +1405,10 @@ static bool mi_arenas_contain_ex(const void* p, mi_arena_t* parent) { return false; } -// Is a pointer inside any of our arenas? -bool _mi_arenas_contain(const void* p) { - return mi_arenas_contain_ex(p, NULL); -} +// // Is a pointer inside any of our arenas? +// bool _mi_arenas_contain(const void* p) { +// return mi_arenas_contain_ex(p, NULL); +// } // Is a pointer contained in the given arena area? bool mi_arena_contains(mi_arena_id_t arena_id, const void* p) { @@ -1357,7 +1434,7 @@ static void mi_arenas_unsafe_destroy(mi_subproc_t* subproc) { // mi_lock_done(&arena->abandoned_visit_lock); mi_atomic_store_ptr_release(mi_arena_t, &subproc->arenas[i], NULL); if (mi_memkind_is_os(arena->memid.memkind)) { - _mi_os_free_ex(mi_arena_start(arena), mi_arena_size(arena), true, arena->memid, subproc); // pass `subproc` to avoid accessing the theap pointer (in `_mi_subproc()`) + _mi_os_free_ex(subproc, mi_arena_start(arena), mi_arena_size(arena), true, arena->memid); } } } @@ -1457,9 +1534,9 @@ static mi_bitmap_t* mi_arena_bitmap_init(size_t slice_count, uint8_t** base) { return bitmap; } -static mi_bbitmap_t* mi_arena_bbitmap_init(size_t slice_count, uint8_t** base) { +static mi_bbitmap_t* mi_arena_bbitmap_init(mi_subproc_t* subproc, size_t slice_count, uint8_t** base) { mi_bbitmap_t* bbitmap = (mi_bbitmap_t*)(*base); - *base = (*base) + mi_bbitmap_init(bbitmap, slice_count, true /* already zero */); + *base = (*base) + mi_bbitmap_init(subproc, bbitmap, slice_count, true /* already zero */); return bbitmap; } @@ -1467,7 +1544,7 @@ static mi_arena_pages_t* mi_arena_pages_alloc(mi_arena_t* arena) { const size_t slice_count = arena->slice_count; size_t bitmap_base = 0; const size_t size = mi_arena_pages_size(slice_count, &bitmap_base); - mi_arena_pages_t* arena_pages = (mi_arena_pages_t*)mi_heap_zalloc_aligned(mi_heap_main(), size, MI_BCHUNK_SIZE); + mi_arena_pages_t* arena_pages = (mi_arena_pages_t*)mi_heap_zalloc_aligned(arena->subproc->heap_main, size, MI_BCHUNK_SIZE); if (arena_pages==NULL) return NULL; uint8_t* base = (uint8_t*)arena_pages + bitmap_base; mi_assert_internal(_mi_is_aligned(base, MI_BCHUNK_SIZE)); @@ -1498,10 +1575,10 @@ static mi_arena_t* mi_arena_initialize(mi_subproc_t* subproc, void* start, _mi_warning_message("cannot use OS memory since it is not large enough (size %zu KiB, minimum required is %zu KiB)", mi_size_of_slices(slice_count)/MI_KiB, mi_size_of_slices(info_slices+1)/MI_KiB); return NULL; } - else if (info_slices >= MI_ARENA_MAX_CHUNK_OBJ_SLICES) { - _mi_warning_message("cannot use OS memory since it is too large with respect to the maximum object size (size %zu MiB, meta-info slices %zu, maximum object slices are %zu)", mi_size_of_slices(slice_count)/MI_MiB, info_slices, MI_ARENA_MAX_CHUNK_OBJ_SLICES); - return NULL; - } + // else if (info_slices >= MI_ARENA_MAX_CHUNK_OBJ_SLICES) { + // _mi_warning_message("cannot use OS memory since it is too large with respect to the maximum object size (size %zu MiB, meta-info slices %zu, maximum object slices are %zu)", mi_size_of_slices(slice_count)/MI_MiB, info_slices, MI_ARENA_MAX_CHUNK_OBJ_SLICES); + // return NULL; + // } mi_arena_t* arena = (mi_arena_t*)start; @@ -1515,7 +1592,7 @@ static mi_arena_t* mi_arena_initialize(mi_subproc_t* subproc, void* start, ok = (*commit_fun)(true /* commit */, arena, commit_size, NULL, commit_fun_arg); } else { - ok = _mi_os_commit(arena, commit_size, NULL); + ok = _mi_os_commit(subproc, arena, commit_size, NULL); } if (!ok) { _mi_warning_message("unable to commit meta-data for OS memory"); @@ -1525,7 +1602,7 @@ static mi_arena_t* mi_arena_initialize(mi_subproc_t* subproc, void* start, else if (!memid.is_pinned) { // if MI_SECURE, set a guard page at the end of the arena info // todo: this does not respect the commit_fun as the memid is of external memory - _mi_os_secure_guard_page_set_before((uint8_t*)arena + mi_size_of_slices(info_slices), memid); + _mi_os_secure_guard_page_set_before(subproc, (uint8_t*)arena + mi_size_of_slices(info_slices), memid); } if (!memid.initially_zero) { _mi_memzero(arena, mi_size_of_slices(info_slices) - _mi_os_secure_guard_page_size()); @@ -1551,7 +1628,7 @@ static mi_arena_t* mi_arena_initialize(mi_subproc_t* subproc, void* start, // init bitmaps uint8_t* base = mi_arena_start(arena) + bitmap_base; - arena->slices_free = mi_arena_bbitmap_init(slice_count, &base); + arena->slices_free = mi_arena_bbitmap_init(subproc, slice_count, &base); arena->slices_committed = mi_arena_bitmap_init(slice_count, &base); arena->slices_dirty = mi_arena_bitmap_init(slice_count, &base); arena->slices_purge = mi_arena_bitmap_init(slice_count, &base); @@ -1584,7 +1661,6 @@ static bool mi_manage_os_memory_ex2(mi_subproc_t* subproc, void* start, size_t s mi_memid_t memid, mi_commit_fun_t* commit_fun, void* commit_fun_arg, mi_arena_id_t* arena_id) mi_attr_noexcept { // checks - mi_assert(_mi_is_aligned(start, MI_ARENA_SLICE_SIZE)); mi_assert(start!=NULL); if (arena_id != NULL) { *arena_id = _mi_arena_id_none(); } if (start==NULL) return false; @@ -1675,12 +1751,18 @@ bool mi_manage_memory(void* start, size_t size, bool is_committed, bool is_pinne // Reserve a range of regular OS memory static int mi_reserve_os_memory_ex2(mi_subproc_t* subproc, size_t size, bool commit, bool allow_large, bool exclusive, mi_arena_id_t* arena_id) { if (arena_id != NULL) *arena_id = _mi_arena_id_none(); - size = _mi_align_up(size, MI_ARENA_SLICE_SIZE); // at least one slice + if (size <= MI_MAX_ALLOC_SIZE) { + size = _mi_align_up(size, MI_ARENA_SLICE_SIZE); // at least one slice + } + if (size > MI_MAX_ALLOC_SIZE) { + _mi_error_message(EOVERFLOW, "memory reservation request is too large (size %zu)\n", size); + return ENOMEM; + } mi_memid_t memid; - void* start = _mi_os_alloc_aligned(size, MI_ARENA_SLICE_ALIGN, commit, allow_large, &memid); + void* start = _mi_os_alloc_aligned(subproc, size, MI_ARENA_SLICE_ALIGN, commit, allow_large, &memid); if (start == NULL) return ENOMEM; if (!mi_manage_os_memory_ex2(subproc, start, size, -1 /* numa node */, exclusive, memid, NULL, NULL, arena_id)) { - _mi_os_free_ex(start, size, commit, memid, NULL); + _mi_os_free_ex(subproc, start, size, commit, memid); _mi_verbose_message("failed to reserve %zu KiB memory\n", _mi_divide_up(size, 1024)); return ENOMEM; } @@ -1790,7 +1872,7 @@ static size_t mi_debug_show_page_bfield(char* buf, size_t* k, mi_arena_t* arena, else { c = '?'; if (bit_of_page > 0) { c = '-'; } - else if (_mi_meta_is_meta_page(start)) { c = 'm'; color = MI_GRAY; } + else if (_mi_meta_is_meta_page(arena->subproc,start)) { c = 'm'; color = MI_GRAY; } else if (slice_index + bit < arena->info_slices) { c = 'i'; color = MI_GRAY; } // else if (mi_bitmap_is_setN(arena->pages_purge, slice_index + bit, NULL)) { c = '*'; } else if (mi_bbitmap_is_setN(arena->slices_free, slice_index+bit,1)) { @@ -1898,13 +1980,13 @@ static void mi_debug_show_arenas_ex(mi_heap_t* heap, bool show_pages, bool narro size_t page_total = 0; for (size_t i = 0; i < max_arenas; i++) { mi_arena_t* arena = mi_atomic_load_ptr_acquire(mi_arena_t, &subproc->arenas[i]); - if (arena == NULL) break; + if (arena == NULL) continue; mi_assert(arena->subproc == subproc); // slice_total += arena->slice_count; - _mi_raw_message("%sarena %zu at %p: %zu slices (%zu MiB)%s%s, subproc: %p, numa: %i\n", + _mi_raw_message("%sarena %zu at %p: %zu slices (%zu MiB)%s%s, subproc: %zu, numa: %i\n", (arena->parent==NULL ? "" : "(sub)"), i, arena, arena->slice_count, (size_t)(mi_size_of_slices(arena->slice_count)/MI_MiB), (arena->memid.is_pinned ? ", pinned" : ""), (arena->is_exclusive ? ", exclusive" : ""), - arena->subproc, arena->numa_node); + arena->subproc->subproc_seq, arena->numa_node); //if (show_inuse) { // free_total += mi_debug_show_bbitmap("in-use slices", arena->slice_count, arena->slices_free, true, NULL); //} @@ -1951,18 +2033,19 @@ int mi_reserve_huge_os_pages_at_ex(size_t pages, int numa_node, size_t timeout_m if (pages==0) return 0; if (numa_node < -1) numa_node = -1; if (numa_node >= 0) numa_node = numa_node % _mi_os_numa_node_count(); + mi_subproc_t* subproc = _mi_subproc(); size_t hsize = 0; size_t pages_reserved = 0; mi_memid_t memid; - void* p = _mi_os_alloc_huge_os_pages(pages, numa_node, timeout_msecs, &pages_reserved, &hsize, &memid); + void* p = _mi_os_alloc_huge_os_pages(subproc, pages, numa_node, timeout_msecs, &pages_reserved, &hsize, &memid); if (p==NULL || pages_reserved==0) { _mi_warning_message("failed to reserve %zu GiB huge pages\n", pages); return ENOMEM; } _mi_verbose_message("numa node %i: reserved %zu GiB huge pages (of the %zu GiB requested)\n", numa_node, pages_reserved, pages); - if (!mi_manage_os_memory_ex2(_mi_subproc(), p, hsize, numa_node, exclusive, memid, NULL, NULL, arena_id)) { - _mi_os_free(p, hsize, memid); + if (!mi_manage_os_memory_ex2(subproc, p, hsize, numa_node, exclusive, memid, NULL, NULL, arena_id)) { + _mi_os_free(subproc, p, hsize, memid); return ENOMEM; } return 0; @@ -2019,7 +2102,14 @@ int mi_reserve_huge_os_pages(size_t pages, double max_secs, size_t* pages_reserv static long mi_arena_purge_delay(void) { // <0 = no purging allowed, 0=immediate purging, >0=milli-second delay - return (mi_option_get(mi_option_purge_delay) * mi_option_get(mi_option_arena_purge_mult)); + const long delay = mi_option_get(mi_option_purge_delay); + const long mult = mi_option_get(mi_option_arena_purge_mult); + if (delay<0 || mult<0) { return -1; } + if (delay==0 || mult==0) { return 0; } + size_t total; + if (mi_mul_overflow((size_t)delay, (size_t)mult, &total)) { return delay; } + if (total > LONG_MAX) { return delay; } + return (long)total; } // reset or decommit in an arena and update the commit bitmap @@ -2035,7 +2125,7 @@ static bool mi_arena_purge(mi_arena_t* arena, size_t slice_index, size_t slice_c size_t already_committed; mi_bitmap_setN(arena->slices_committed, slice_index, slice_count, &already_committed); // pretend all committed.. (as we lack a clearN call that counts the already set bits..) const bool all_committed = (already_committed == slice_count); - const bool needs_recommit = _mi_os_purge_ex(p, size, all_committed /* allow reset? */, mi_size_of_slices(already_committed), arena->commit_fun, arena->commit_fun_arg); + const bool needs_recommit = _mi_os_purge_ex(arena->subproc, p, size, all_committed /* allow reset? */, mi_size_of_slices(already_committed), arena->commit_fun, arena->commit_fun_arg); if (needs_recommit) { // no longer committed @@ -2254,7 +2344,7 @@ bool _mi_heap_visit_blocks(mi_heap_t* heap, bool abandoned_only, bool visit_bloc mi_arena_pages_t* arena_pages = mi_heap_arena_pages(heap, arena); if (ok && arena_pages != NULL) { if (abandoned_only) { - for (size_t bin = 0; ok && bin < MI_BIN_COUNT; bin++) { + for (size_t bin = 0; ok && bin < MI_ARENA_BIN_COUNT; bin++) { // todo: if we had a single abandoned page map as well, this can be faster. if (mi_atomic_load_relaxed(&heap->abandoned_count[bin]) > 0) { ok = _mi_bitmap_forall_set(arena_pages->pages_abandoned[bin], &mi_heap_visit_page_at, arena, &visit_info); @@ -2322,6 +2412,9 @@ static bool mi_heap_delete_page(const mi_heap_t* heap, const mi_heap_area_t* are _mi_arenas_page_free(page, theap); } else if (heap_target==NULL) { + #if MI_GUARDED + _mi_page_unguard_all(page); // remove potential interior guard pages + #endif // destroy the page page->used=0; // note: invariant `|local_free| + |free| == reserved - used` does not hold in this case _mi_arenas_page_free(page, theap); @@ -2330,12 +2423,19 @@ static bool mi_heap_delete_page(const mi_heap_t* heap, const mi_heap_area_t* are // move the page to `heap_target` as an abandoned page // first remove it from the current heap const size_t sbin = _mi_page_stats_bin(page); - size_t slice_index; - size_t slice_count; - mi_arena_pages_t* arena_pages = NULL; - mi_arena_t* const arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); - mi_assert_internal(mi_bitmap_is_set(arena_pages->pages, slice_index)); - mi_bitmap_clear(arena_pages->pages, slice_index); + mi_arena_t* arena = NULL; + size_t slice_index = 0; + if (page->memid.memkind == MI_MEM_ARENA) { + size_t slice_count; + mi_arena_pages_t* arena_pages = NULL; + arena = mi_page_arena_pages(page, &slice_index, &slice_count, &arena_pages); + mi_assert_internal(mi_bitmap_is_set(arena_pages->pages, slice_index)); + mi_bitmap_clear(arena_pages->pages, slice_index); + } + else { + // os allocated + mi_assert_internal(mi_memid_is_os(page->memid) && page->next == NULL); + } if (theap != NULL) { mi_theap_stat_decrease(theap, page_bins[sbin], 1); mi_theap_stat_decrease(theap, pages, 1); @@ -2347,16 +2447,18 @@ static bool mi_heap_delete_page(const mi_heap_t* heap, const mi_heap_area_t* are mi_theap_t* theap_target = info->theap_target; // and then add it to the new target heap - mi_arena_pages_t* arena_pages_target = mi_heap_ensure_arena_pages(heap_target, arena); - if mi_unlikely(arena_pages_target==NULL) { - // if we cannot allocate this, we move it to the main heap instead (which does not require allocation) - heap_target = mi_heap_main(); - theap_target = mi_heap_theap(heap_target); - arena_pages_target = mi_heap_ensure_arena_pages(heap_target, arena); - mi_assert_internal(arena_pages_target!=NULL); + if (arena != NULL) { + mi_arena_pages_t* arena_pages_target = mi_heap_ensure_arena_pages(heap_target, arena); + if mi_unlikely(arena_pages_target==NULL) { + // if we cannot allocate this, we move it to the main heap instead (which does not require allocation) + heap_target = mi_arena_heap_main(arena); + theap_target = mi_heap_theap(heap_target); // todo: find through theap_target tld? + arena_pages_target = mi_heap_ensure_arena_pages(heap_target, arena); + mi_assert_internal(arena_pages_target!=NULL); + } + mi_assert_internal(mi_bitmap_is_clear(arena_pages_target->pages, slice_index)); + mi_bitmap_set(arena_pages_target->pages, slice_index); } - mi_assert_internal(mi_bitmap_is_clear(arena_pages_target->pages, slice_index)); - mi_bitmap_set(arena_pages_target->pages, slice_index); page->heap = heap_target; mi_theap_stat_increase(theap_target, page_bins[sbin], 1); mi_theap_stat_increase(theap_target, pages, 1); @@ -2374,7 +2476,7 @@ static void mi_heap_delete_pages(mi_heap_t* heap, mi_heap_t* heap_target) { _mi_heap_visit_blocks(heap, false, false, &mi_heap_delete_page, &info); #if MI_DEBUG>1 // no more arena pages? - for (size_t i = 0; i < MI_ARENA_BIN_COUNT; i++) { + for (size_t i = 0; i < MI_MAX_ARENAS; i++) { mi_arena_pages_t* const arena_pages = mi_atomic_load_ptr_relaxed(mi_arena_pages_t, &heap->arena_pages[i]); if (arena_pages!=NULL) { mi_assert_internal(mi_bitmap_is_all_clear(arena_pages->pages)); @@ -2386,7 +2488,7 @@ static void mi_heap_delete_pages(mi_heap_t* heap, mi_heap_t* heap_target) { mi_assert_internal(heap->os_abandoned_pages == NULL); } // nor arena abandoned pages? - for (size_t i = 0; i < MI_BIN_COUNT; i++) { + for (size_t i = 0; i < MI_ARENA_BIN_COUNT; i++) { mi_assert_internal(mi_atomic_load_relaxed(&heap->abandoned_count[i])==0); } #endif @@ -2394,7 +2496,7 @@ static void mi_heap_delete_pages(mi_heap_t* heap, mi_heap_t* heap_target) { void _mi_heap_move_pages(mi_heap_t* heap_from, mi_heap_t* heap_to) { if (_mi_is_heap_main(heap_from)) return; - if (heap_to==NULL) { heap_to = mi_heap_main(); } + if (heap_to==NULL) { heap_to = heap_from->subproc->heap_main; } mi_heap_delete_pages(heap_from, heap_to); } @@ -2443,7 +2545,7 @@ mi_decl_export bool mi_arena_unload(mi_arena_id_t arena_id, void** base, size_t* // adjust abandoned page count mi_subproc_t* const subproc = arena->subproc; - for (size_t bin = 0; bin < MI_BIN_COUNT; bin++) { + for (size_t bin = 0; bin < MI_ARENA_BIN_COUNT; bin++) { const size_t count = mi_bitmap_popcount(arena->pages_abandoned[bin]); if (count > 0) { mi_atomic_decrement_acq_rel(&subproc->abandoned_count[bin]); } } @@ -2503,7 +2605,7 @@ mi_decl_export bool mi_arena_reload(void* start, size_t size, mi_commit_fun_t* c } // adjust abandoned page count - for (size_t bin = 0; bin < MI_BIN_COUNT; bin++) { + for (size_t bin = 0; bin < MI_ARENA_BIN_COUNT; bin++) { const size_t count = mi_bitmap_popcount(arena->pages_abandoned[bin]); if (count > 0) { mi_atomic_decrement_acq_rel(&arena->subproc->abandoned_count[bin]); } } diff --git a/vendored/mimalloc/src/bitmap.c b/vendored/mimalloc/src/bitmap.c index 67269e8bf2..aad7a5559a 100644 --- a/vendored/mimalloc/src/bitmap.c +++ b/vendored/mimalloc/src/bitmap.c @@ -109,7 +109,7 @@ static inline bool mi_bfield_atomic_clear(_Atomic(mi_bfield_t)*b, size_t idx, bo // Clear a bit but only when/once it is set. This is used by concurrent free's while // the page is abandoned and mapped. This can incure a busy wait :-( but it should // happen almost never (and is accounted for in the stats) -static inline void mi_bfield_atomic_clear_once_set(_Atomic(mi_bfield_t)*b, size_t idx) { +static inline void mi_bfield_atomic_clear_once_set(mi_subproc_t* subproc, _Atomic(mi_bfield_t)*b, size_t idx) { mi_assert_internal(idx < MI_BFIELD_BITS); const mi_bfield_t mask = mi_bfield_mask(1, idx);; mi_bfield_t old = mi_atomic_load_relaxed(b); @@ -117,7 +117,7 @@ static inline void mi_bfield_atomic_clear_once_set(_Atomic(mi_bfield_t)*b, size_ if mi_unlikely((old&mask) == 0) { old = mi_atomic_load_acquire(b); if ((old&mask)==0) { - mi_subproc_stat_counter_increase(_mi_subproc(), pages_unabandon_busy_wait, 1); + mi_subproc_stat_counter_increase(subproc, pages_unabandon_busy_wait, 1); } while ((old&mask)==0) { // busy wait _mi_prim_thread_yield(); @@ -331,7 +331,7 @@ mi_decl_noinline static bool mi_bchunk_xsetNC(mi_xset_t set, mi_bchunk_t* chunk, bool all_clear = false; const bool transition = (set ? mi_bfield_atomic_set_mask(&chunk->bfields[field], mask, &already_set) : mi_bfield_atomic_clear_mask(&chunk->bfields[field], mask, &all_clear)); - mi_assert_internal((transition && already_set == 0) || (!transition && already_set > 0)); + mi_assert_internal(!set || ((transition && already_set == 0) || (!transition && already_set > 0))); all_transition = all_transition && transition; total_already_set += already_set; maybe_all_clear = maybe_all_clear && all_clear; @@ -369,7 +369,7 @@ static inline bool mi_bchunk_clearN(mi_bchunk_t* chunk, size_t cidx, size_t n, b if (n==1) return mi_bchunk_clear(chunk, cidx, maybe_all_clear); // if (n==8) return mi_bchunk_clear8(chunk, cidx, maybe_all_clear); // if (n==MI_BFIELD_BITS) return mi_bchunk_clearX(chunk, cidx, maybe_all_clear); - // TODO: implement mi_bchunk_xsetNX instead of setNX + // todo: implement mi_bchunk_xsetNX instead of setNX return mi_bchunk_xsetNC(MI_BIT_CLEAR, chunk, cidx, n, NULL, maybe_all_clear); } @@ -445,7 +445,7 @@ static inline bool mi_bchunk_is_xsetN(mi_xset_t set, const mi_bchunk_t* chunk, s // ------- mi_bchunk_try_clear --------------------------------------- // Clear `0 < n <= MI_BITFIELD_BITS`. Can cross over a bfield boundary. -static inline bool mi_bchunk_try_clearNX(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* pmaybe_all_clear) { +static inline bool mi_bchunk_try_clearNX(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* pmaybe_all_clear, bool* did_temp_clear_bits) { mi_assert_internal(cidx < MI_BCHUNK_BITS); mi_assert_internal(n <= MI_BFIELD_BITS); const size_t i = cidx / MI_BFIELD_BITS; @@ -468,6 +468,7 @@ static inline bool mi_bchunk_try_clearNX(mi_bchunk_t* chunk, size_t cidx, size_t if (!mi_bfield_atomic_try_clear_mask(&chunk->bfields[i+1], mi_bfield_mask(n - m, 0), &field2_is_clear)) { // we failed to clear the second field, restore the first one mi_bfield_atomic_set_mask(&chunk->bfields[i], mi_bfield_mask(m, idx), NULL); + if (did_temp_clear_bits != NULL) { *did_temp_clear_bits = true; } return false; } if (pmaybe_all_clear != NULL) { *pmaybe_all_clear = field1_is_clear && field2_is_clear; } @@ -488,7 +489,7 @@ static inline bool mi_bchunk_try_clearNX(mi_bchunk_t* chunk, size_t cidx, size_t // and false otherwise leaving all bit fields as is. // Note: this is the complex one as we need to unwind partial atomic operations if we fail halfway.. // `maybe_all_clear` is set to `true` if all the bfields involved become zero. -mi_decl_noinline static bool mi_bchunk_try_clearNC(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* pmaybe_all_clear) { +mi_decl_noinline static bool mi_bchunk_try_clearNC(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* pmaybe_all_clear, bool* did_temp_clear_bits) { mi_assert_internal(cidx + n <= MI_BCHUNK_BITS); mi_assert_internal(n>0); if (pmaybe_all_clear != NULL) { *pmaybe_all_clear = true; } @@ -538,6 +539,7 @@ mi_decl_noinline static bool mi_bchunk_try_clearNC(mi_bchunk_t* chunk, size_t ci restore: // `field` is the index of the field that failed to set atomically; we need to restore all previous fields mi_assert_internal(field > start_field); + if (did_temp_clear_bits != NULL) { *did_temp_clear_bits = true; } while( field > start_field) { field--; if (field == start_field) { @@ -551,11 +553,11 @@ mi_decl_noinline static bool mi_bchunk_try_clearNC(mi_bchunk_t* chunk, size_t ci } -static inline bool mi_bchunk_try_clearN(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* maybe_all_clear) { +static inline bool mi_bchunk_try_clearN(mi_bchunk_t* chunk, size_t cidx, size_t n, bool* maybe_all_clear, bool* did_temp_clear_bits) { mi_assert_internal(n>0); // if (n==MI_BFIELD_BITS) return mi_bchunk_try_clearX(chunk, cidx, maybe_all_clear); - if (n<=MI_BFIELD_BITS) return mi_bchunk_try_clearNX(chunk, cidx, n, maybe_all_clear); - return mi_bchunk_try_clearNC(chunk, cidx, n, maybe_all_clear); + if (n<=MI_BFIELD_BITS) return mi_bchunk_try_clearNX(chunk, cidx, n, maybe_all_clear, did_temp_clear_bits); + return mi_bchunk_try_clearNC(chunk, cidx, n, maybe_all_clear, did_temp_clear_bits); } @@ -686,8 +688,8 @@ static inline bool mi_bchunk_try_find_and_clear(mi_bchunk_t* chunk, size_t* pidx return false; } -static inline bool mi_bchunk_try_find_and_clear_1(mi_bchunk_t* chunk, size_t n, size_t* pidx) { - mi_assert_internal(n==1); MI_UNUSED(n); +static inline bool mi_bchunk_try_find_and_clear_1(mi_bchunk_t* chunk, size_t n, size_t* pidx, bool* did_temp_clear_bits) { + mi_assert_internal(n==1); MI_UNUSED(n); MI_UNUSED(did_temp_clear_bits); return mi_bchunk_try_find_and_clear(chunk, pidx); } @@ -749,8 +751,8 @@ static mi_decl_noinline bool mi_bchunk_try_find_and_clear8(mi_bchunk_t* chunk, s #endif } -static inline bool mi_bchunk_try_find_and_clear_8(mi_bchunk_t* chunk, size_t n, size_t* pidx) { - mi_assert_internal(n==8); MI_UNUSED(n); +static inline bool mi_bchunk_try_find_and_clear_8(mi_bchunk_t* chunk, size_t n, size_t* pidx, bool* did_temp_clear_bits) { + mi_assert_internal(n==8); MI_UNUSED(n); MI_UNUSED(did_temp_clear_bits); return mi_bchunk_try_find_and_clear8(chunk, pidx); } @@ -759,7 +761,7 @@ static inline bool mi_bchunk_try_find_and_clear_8(mi_bchunk_t* chunk, size_t n, // and try to clear them atomically. // set `*pidx` to its bit index (0 <= *pidx <= MI_BCHUNK_BITS - n) on success. // will cross bfield boundaries. -mi_decl_noinline static bool mi_bchunk_try_find_and_clearNX(mi_bchunk_t* chunk, size_t n, size_t* pidx) { +mi_decl_noinline static bool mi_bchunk_try_find_and_clearNX(mi_bchunk_t* chunk, size_t n, size_t* pidx, bool* did_temp_clear_bits) { if (n == 0 || n > MI_BFIELD_BITS) return false; const mi_bfield_t mask = mi_bfield_mask(n, 0); // for all fields in the chunk @@ -803,7 +805,7 @@ mi_decl_noinline static bool mi_bchunk_try_find_and_clearNX(mi_bchunk_t* chunk, if (post + pre >= n) { // it fits -- try to claim it atomically const size_t cidx = (i*MI_BFIELD_BITS) + (MI_BFIELD_BITS - post); - if (mi_bchunk_try_clearNX(chunk, cidx, n, NULL)) { + if (mi_bchunk_try_clearNX(chunk, cidx, n, NULL, did_temp_clear_bits)) { // we cleared all atomically *pidx = cidx; mi_assert_internal(*pidx < MI_BCHUNK_BITS); @@ -821,7 +823,7 @@ mi_decl_noinline static bool mi_bchunk_try_find_and_clearNX(mi_bchunk_t* chunk, // and try to clear them atomically. // set `*pidx` to its bit index (0 <= *pidx <= MI_BCHUNK_BITS - n) on success. // This can cross bfield boundaries. -static mi_decl_noinline bool mi_bchunk_try_find_and_clearNC(mi_bchunk_t* chunk, size_t n, size_t* pidx) { +static mi_decl_noinline bool mi_bchunk_try_find_and_clearNC(mi_bchunk_t* chunk, size_t n, size_t* pidx, bool* did_temp_clear_bits) { if (n == 0 || n > MI_BCHUNK_BITS) return false; // cannot be more than a chunk // we first scan ahead to see if there is a range of `n` set bits, and only then try to clear atomically @@ -870,7 +872,7 @@ static mi_decl_noinline bool mi_bchunk_try_find_and_clearNC(mi_bchunk_t* chunk, // did we find a range? if (m==0) { - if (mi_bchunk_try_clearN(chunk, cidx, n, NULL)) { + if (mi_bchunk_try_clearN(chunk, cidx, n, NULL, did_temp_clear_bits)) { // we cleared all atomically *pidx = cidx; mi_assert_internal(*pidx < MI_BCHUNK_BITS); @@ -888,11 +890,11 @@ static mi_decl_noinline bool mi_bchunk_try_find_and_clearNC(mi_bchunk_t* chunk, // ------- mi_bchunk_clear_once_set --------------------------------------- -static inline void mi_bchunk_clear_once_set(mi_bchunk_t* chunk, size_t cidx) { +static inline void mi_bchunk_clear_once_set(mi_subproc_t* subproc, mi_bchunk_t* chunk, size_t cidx) { mi_assert_internal(cidx < MI_BCHUNK_BITS); const size_t i = cidx / MI_BFIELD_BITS; const size_t idx = cidx % MI_BFIELD_BITS; - mi_bfield_atomic_clear_once_set(&chunk->bfields[i], idx); + mi_bfield_atomic_clear_once_set(subproc, &chunk->bfields[i], idx); } @@ -963,7 +965,7 @@ static bool mi_bchunk_bsr(mi_bchunk_t* chunk, size_t* pidx) { return false; } -static bool mi_bchunk_bsr_inv(mi_bchunk_t* chunk, size_t* pidx) { +static bool mi_bchunk_bsr_inv(mi_bchunk_t* chunk, size_t* pidx) { for (size_t i = MI_BCHUNK_FIELDS; i > 0; ) { i--; mi_bfield_t b = mi_atomic_load_relaxed(&chunk->bfields[i]); @@ -1323,9 +1325,10 @@ static bool mi_bitmap_try_find_and_claim_visit(mi_bitmap_t* bitmap, size_t chunk return true; } else { - // failed to claim it, set abandoned mapping again (unless the page was freed) + // failed to claim it, set abandoned mapping again (unless the page was freed and keep_set will be false) if (keep_set) { const bool wasclear = mi_bchunk_set(&bitmap->chunks[chunk_idx], cidx, NULL); + mi_bitmap_chunkmap_set(bitmap, chunk_idx); mi_assert_internal(wasclear); MI_UNUSED(wasclear); } } @@ -1355,12 +1358,15 @@ bool mi_bitmap_bsr(mi_bitmap_t* bitmap, size_t* idx) { mi_bfield_t cmap = mi_atomic_load_relaxed(&bitmap->chunkmap.bfields[i]); size_t cmap_idx; if (mi_bsr(cmap,&cmap_idx)) { - // highest chunk - const size_t chunk_idx = i*MI_BFIELD_BITS + cmap_idx; - size_t cidx; - if (mi_bchunk_bsr(&bitmap->chunks[chunk_idx], &cidx)) { - *idx = (chunk_idx * MI_BCHUNK_BITS) + cidx; - return true; + // from highest chunk to lowest (scan all in case the cmap entry was stale) + for (size_t j = cmap_idx+1; j>0; ) { + j--; + const size_t chunk_idx = (i*MI_BFIELD_BITS) + j; + size_t cidx; + if (mi_bchunk_bsr(&bitmap->chunks[chunk_idx], &cidx)) { + *idx = (chunk_idx * MI_BCHUNK_BITS) + cidx; + return true; + } } } } @@ -1388,12 +1394,12 @@ size_t mi_bitmap_popcount(mi_bitmap_t* bitmap) { // Clear a bit once it is set. -void mi_bitmap_clear_once_set(mi_bitmap_t* bitmap, size_t idx) { +void mi_bitmap_clear_once_set(mi_subproc_t* subproc, mi_bitmap_t* bitmap, size_t idx) { mi_assert_internal(idx < mi_bitmap_max_bits(bitmap)); const size_t chunk_idx = idx / MI_BCHUNK_BITS; const size_t cidx = idx % MI_BCHUNK_BITS; mi_assert_internal(chunk_idx < mi_bitmap_chunk_count(bitmap)); - mi_bchunk_clear_once_set(&bitmap->chunks[chunk_idx], cidx); + mi_bchunk_clear_once_set(subproc, &bitmap->chunks[chunk_idx], cidx); } @@ -1456,6 +1462,8 @@ bool _mi_bitmap_forall_setc_ranges(mi_bitmap_t* bitmap, mi_forall_set_fun_t* vis mi_assert_internal(rng>=1 && rng<=MI_BFIELD_BITS); mi_assert_internal((idx % MI_BFIELD_BITS) + rng <= MI_BFIELD_BITS); mi_assert_internal((idx / MI_BCHUNK_BITS) < mi_bitmap_chunk_count(bitmap)); + // clear rng bits in b + b = b & ~mi_bfield_mask(rng, bidx); if (!visit(idx, rng, arena, arg)) { // break early: reset the non-visited bits if (b!=0) { @@ -1463,8 +1471,6 @@ bool _mi_bitmap_forall_setc_ranges(mi_bitmap_t* bitmap, mi_forall_set_fun_t* vis } return false; } - // clear rng bits in b - b = b & ~mi_bfield_mask(rng, bidx); } mi_assert_internal(rngcount == bpopcount); } @@ -1502,14 +1508,15 @@ bool _mi_bitmap_forall_setc_rangesn(mi_bitmap_t* bitmap, size_t rngslices, mi_fo const size_t base_idx = (chunk_idx*MI_BCHUNK_BITS) + (j*MI_BFIELD_BITS); mi_bfield_t b = mi_atomic_exchange_relaxed(&chunk->bfields[j], (mi_bfield_t)0); // atomic clear mi_bfield_t skipped = 0; // but track which bits we skip so we can restore them - for(size_t shift = 0; rngslices + shift <= MI_BFIELD_BITS; shift += rngslices) { // per `rngslices` to keep alignment + size_t shift; + for(shift = 0; rngslices + shift <= MI_BFIELD_BITS; shift += rngslices) { // per `rngslices` to keep alignment const mi_bfield_t rngmask = mi_bfield_mask(rngslices, shift); if ((b & rngmask) == rngmask) { const size_t idx = base_idx + shift; if (!visit(idx, rngslices, arena, arg)) { // break early: restore non-visited entries mi_bfield_t notyet_visited = 0; - if (shift + rngslices < MI_BFIELD_BITS) { + if (rngslices + shift < MI_BFIELD_BITS) { notyet_visited = (b & (~(mi_bfield_t)0 << (shift + rngslices))); } mi_assert_internal((notyet_visited & skipped) == 0); @@ -1523,8 +1530,13 @@ bool _mi_bitmap_forall_setc_rangesn(mi_bitmap_t* bitmap, size_t rngslices, mi_fo skipped = skipped | (b & rngmask); } } - + if (shift < MI_BFIELD_BITS) { + // there are some non-visited top bits when `MI_BFIELD_BITS % rngslices != 0`. + mi_assert_internal(MI_BFIELD_BITS % rngslices != 0); + skipped = skipped | (b & (~(mi_bfield_t)0 << shift)); + } if (skipped != 0) { + // restore non-visited entries mi_atomic_or_relaxed(&chunk->bfields[j], skipped); } } @@ -1554,7 +1566,7 @@ size_t mi_bbitmap_size(size_t bit_count, size_t* pchunk_count) { // initialize a bitmap to all unset; avoid a mem_zero if `already_zero` is true // returns the size of the bitmap -size_t mi_bbitmap_init(mi_bbitmap_t* bbitmap, size_t bit_count, bool already_zero) { +size_t mi_bbitmap_init(mi_subproc_t* subproc, mi_bbitmap_t* bbitmap, size_t bit_count, bool already_zero) { size_t chunk_count; const size_t size = mi_bbitmap_size(bit_count, &chunk_count); if (!already_zero) { @@ -1562,6 +1574,7 @@ size_t mi_bbitmap_init(mi_bbitmap_t* bbitmap, size_t bit_count, bool already_zer } mi_atomic_store_release(&bbitmap->chunk_count, chunk_count); mi_assert_internal(mi_atomic_load_relaxed(&bbitmap->chunk_count) <= MI_BITMAP_MAX_CHUNK_COUNT); + bbitmap->subproc = subproc; return size; } @@ -1572,27 +1585,16 @@ void mi_bbitmap_unsafe_setN(mi_bbitmap_t* bbitmap, size_t idx, size_t n) { } bool mi_bbitmap_bsr_inv(mi_bbitmap_t* bbitmap, size_t* idx) { + // scan for highest zero bit in the bitmap + // note: we cannot use the chunkmap since that only conservatively denotes if there might be a set bit in a chuck + // todo: bbitmap_init rounds up the bitcount to BCHUNK_BITS and we should skip the top-padding! const size_t chunk_count = mi_bbitmap_chunk_count(bbitmap); - const size_t chunkmap_max = _mi_divide_up(chunk_count, MI_BFIELD_BITS); - size_t skip_at_top = chunk_count % MI_BFIELD_BITS; - for (size_t i = chunkmap_max; i > 0; ) { + for(size_t i = chunk_count; i > 0; ) { i--; - mi_bfield_t cmap = mi_atomic_load_relaxed(&bbitmap->chunkmap.bfields[i]); - size_t cmap_idx; - // don't consider top 0 bits; set those to 1 here - if (skip_at_top > 0) { - const size_t mask_top = (~mi_bfield_zero()) << (MI_BFIELD_BITS - skip_at_top); - skip_at_top = 0; // only for the first iteration - cmap |= mask_top; - } - if (mi_bsr(~cmap, &cmap_idx)) { - // highest chunk - const size_t chunk_idx = i*MI_BFIELD_BITS + cmap_idx; - size_t cidx; - if (mi_bchunk_bsr_inv(&bbitmap->chunks[chunk_idx], &cidx)) { - *idx = (chunk_idx * MI_BCHUNK_BITS) + cidx; - return true; - } + size_t cidx; + if (mi_bchunk_bsr_inv(&bbitmap->chunks[i], &cidx)) { + *idx = (i * MI_BCHUNK_BITS) + cidx; + return true; } } return false; @@ -1609,11 +1611,11 @@ static void mi_bbitmap_set_chunk_bin(mi_bbitmap_t* bbitmap, size_t chunk_idx, mi for (mi_chunkbin_t ibin = MI_CBIN_SMALL; ibin < MI_CBIN_NONE; ibin = mi_chunkbin_inc(ibin)) { if (ibin == bin) { const bool was_clear = mi_bchunk_set(& bbitmap->chunkmap_bins[ibin], chunk_idx, NULL); - if (was_clear) { mi_os_stat_increase(chunk_bins[ibin],1); } + if (was_clear) { mi_subproc_stat_increase(bbitmap->subproc, chunk_bins[ibin],1); } } else { const bool was_set = mi_bchunk_clear(&bbitmap->chunkmap_bins[ibin], chunk_idx, NULL); - if (was_set) { mi_os_stat_decrease(chunk_bins[ibin],1); } + if (was_set) { mi_subproc_stat_decrease(bbitmap->subproc,chunk_bins[ibin],1); } } } } @@ -1710,8 +1712,15 @@ bool mi_bbitmap_try_clearNC(mi_bbitmap_t* bbitmap, size_t idx, size_t n) { mi_assert_internal(chunk_idx < mi_bbitmap_chunk_count(bbitmap)); if (cidx + n > MI_BCHUNK_BITS) return false; bool maybe_all_clear = false; - const bool cleared = mi_bchunk_try_clearN(&bbitmap->chunks[chunk_idx], cidx, n, &maybe_all_clear); - if (cleared && maybe_all_clear) { mi_bbitmap_chunkmap_try_clear(bbitmap, chunk_idx); } + bool did_temp_clear_bits = false; + const bool cleared = mi_bchunk_try_clearN(&bbitmap->chunks[chunk_idx], cidx, n, &maybe_all_clear, &did_temp_clear_bits); + if (cleared && maybe_all_clear) { + mi_assert_internal(!did_temp_clear_bits); + mi_bbitmap_chunkmap_try_clear(bbitmap, chunk_idx); + } else if (did_temp_clear_bits) { + // may have raced with a clearer (in on_find) so set the chunkmap bit conservatively + mi_bbitmap_chunkmap_set(bbitmap, chunk_idx, false); + } // note: we don't set the size class for an explicit try_clearN (only used by purging) return cleared; } @@ -1753,7 +1762,7 @@ bool mi_bbitmap_is_xsetN(mi_xset_t set, mi_bbitmap_t* bbitmap, size_t idx, size_ (used to find free pages) -------------------------------------------------------------------------------- */ -typedef bool (mi_bchunk_try_find_and_clear_fun_t)(mi_bchunk_t* chunk, size_t n, size_t* idx); +typedef bool (mi_bchunk_try_find_and_clear_fun_t)(mi_bchunk_t* chunk, size_t n, size_t* idx, bool* did_temp_clear_bits); // Go through the bbitmap and for every sequence of `n` set bits, call the visitor function. // If it returns `true` stop the search. @@ -1814,7 +1823,8 @@ static inline bool mi_bbitmap_try_find_and_clear_generic(mi_bbitmap_t* bbitmap, mi_bchunk_t* chunk = &bbitmap->chunks[chunk_idx]; size_t cidx; - if ((*on_find)(chunk, n, &cidx)) { + bool did_temp_clear_bits = false; + if ((*on_find)(chunk, n, &cidx, &did_temp_clear_bits)) { if (cidx==0 && ibin == MI_CBIN_NONE) { // only the first block determines the size bin // this chunk is now reserved for the `bbin` size class mi_bbitmap_set_chunk_bin(bbitmap, chunk_idx, bbin); @@ -1826,7 +1836,13 @@ static inline bool mi_bbitmap_try_find_and_clear_generic(mi_bbitmap_t* bbitmap, else { // todo: should _on_find_ return a boolean if there is a chance all are clear to avoid calling `try_clear?` // we may find that all are cleared only on a second iteration but that is ok as the chunkmap is a conservative approximation. - mi_bbitmap_chunkmap_try_clear(bbitmap, chunk_idx); + if (did_temp_clear_bits) { + // a concurrent find_and_claim may have cleared the chunkmap bit, restore it now + mi_bbitmap_chunkmap_set(bbitmap, chunk_idx, false); + } + else { + mi_bbitmap_chunkmap_try_clear(bbitmap, chunk_idx); + } } } mi_bfield_cycle_iterate_end(Y); @@ -1871,12 +1887,11 @@ bool mi_bbitmap_try_find_and_clearNC(mi_bbitmap_t* bbitmap, size_t tseq, size_t // Try to atomically clear `n` bits starting at `chunk_idx` where `n` can span over multiple chunks static bool mi_bchunk_try_clearN_(mi_bbitmap_t* bbitmap, size_t chunk_idx, size_t n) { mi_assert_internal((chunk_idx * MI_BCHUNK_BITS) + n <= mi_bbitmap_max_bits(bbitmap)); - size_t m = n; // bits to go size_t count = 0; // chunk count while (m > 0) { mi_bchunk_t* chunk = &bbitmap->chunks[chunk_idx + count]; - if (!mi_bchunk_try_clearN(chunk, 0, (m > MI_BCHUNK_BITS ? MI_BCHUNK_BITS : m), NULL)) { + if (!mi_bchunk_try_clearN(chunk, 0, (m > MI_BCHUNK_BITS ? MI_BCHUNK_BITS : m), NULL, NULL)) { goto rollback; } m = (m <= MI_BCHUNK_BITS ? 0 : m - MI_BCHUNK_BITS); @@ -1890,6 +1905,8 @@ static bool mi_bchunk_try_clearN_(mi_bbitmap_t* bbitmap, size_t chunk_idx, size_ count--; mi_bchunk_t* chunk = &bbitmap->chunks[chunk_idx + count]; mi_bchunk_setN(chunk, 0, MI_BCHUNK_BITS, NULL); + // since we may race with clearing, we need to set the chunkmap conservatively + mi_bbitmap_chunkmap_set(bbitmap, chunk_idx + count, false); } return false; } @@ -1926,7 +1943,7 @@ bool mi_bbitmap_try_find_and_clearN_(mi_bbitmap_t* bbitmap, size_t tseq, size_t // did we find a suitable range? if (count == chunk_req) { - // now try to claim it! + // now try to claim it! if (mi_bchunk_try_clearN_(bbitmap, chunk_idx, n)) { *pidx = (chunk_idx * MI_BCHUNK_BITS); for (size_t i = 0; i < count; i++) { @@ -1935,6 +1952,11 @@ bool mi_bbitmap_try_find_and_clearN_(mi_bbitmap_t* bbitmap, size_t tseq, size_t mi_assert_internal(*pidx + n <= mi_bbitmap_max_bits(bbitmap)); return true; } + else { + // contended: we reset count to retry from the first + // (we still skip the first chunk to guarantee progress) + count = 0; + } } // keep searching but skip the scanned range diff --git a/vendored/mimalloc/src/bitmap.h b/vendored/mimalloc/src/bitmap.h index ca51a4e047..8522d346d7 100644 --- a/vendored/mimalloc/src/bitmap.h +++ b/vendored/mimalloc/src/bitmap.h @@ -199,7 +199,7 @@ mi_decl_nodiscard bool mi_bitmap_try_find_and_claim(mi_bitmap_t* bitmap, size_t // Atomically clear a bit but only if it is set. Will block otherwise until the bit is set. // This is used to delay free-ing a page that it at the same time being considered to be // allocated from `mi_arena_try_abandoned` (and is in the `claim` function of `mi_bitmap_try_find_and_claim`). -void mi_bitmap_clear_once_set(mi_bitmap_t* bitmap, size_t idx); +void mi_bitmap_clear_once_set(mi_subproc_t* subproc, mi_bitmap_t* bitmap, size_t idx); // If a bit is set in the bitmap, return `true` and set `idx` to the index of the highest bit. @@ -260,8 +260,9 @@ static inline mi_chunkbin_t mi_chunkbin_of(size_t slice_count) { typedef mi_decl_bchunk_align struct mi_bbitmap_s { _Atomic(size_t) chunk_count; // total count of chunks (0 < N <= MI_BCHUNKMAP_BITS) _Atomic(size_t) chunk_max_accessed; // max chunk index that was once cleared or set - #if (MI_BCHUNK_SIZE / MI_SIZE_SIZE) > 2 - size_t _padding[MI_BCHUNK_SIZE/MI_SIZE_SIZE - 2]; // suppress warning on msvc by aligning manually + mi_subproc_t* subproc; // constant, for stats + #if (MI_BCHUNK_SIZE / MI_SIZE_SIZE) > 3 + size_t _padding[MI_BCHUNK_SIZE/MI_SIZE_SIZE - 3]; // suppress warning on msvc by aligning manually #endif mi_bchunkmap_t chunkmap; mi_bchunkmap_t chunkmap_bins[MI_CBIN_COUNT - 1]; // chunkmaps with bit set if the chunk is in that size class (excluding MI_CBIN_NONE) @@ -288,7 +289,7 @@ bool mi_bbitmap_bsr_inv(mi_bbitmap_t* bbitmap, size_t* idx); // Initialize a bitmap to all clear; avoid a mem_zero if `already_zero` is true // returns the size of the bitmap. -size_t mi_bbitmap_init(mi_bbitmap_t* bbitmap, size_t bit_count, bool already_zero); +size_t mi_bbitmap_init(mi_subproc_t* subproc, mi_bbitmap_t* bbitmap, size_t bit_count, bool already_zero); // Set/clear a sequence of `n` bits in the bitmap (and can cross chunks). // Not atomic so only use if still local to a thread. diff --git a/vendored/mimalloc/src/free.c b/vendored/mimalloc/src/free.c index 6f33340d04..0c830c6f06 100644 --- a/vendored/mimalloc/src/free.c +++ b/vendored/mimalloc/src/free.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -9,7 +9,7 @@ terms of the MIT license. A copy of the license can be found in the file // add includes help an IDE #include "mimalloc.h" #include "mimalloc/internal.h" -#include "mimalloc/prim.h" // _mi_prim_thread_id() +#include "mimalloc/prim-tls.h" // _mi_prim_thread_id() #endif // forward declarations @@ -33,14 +33,21 @@ static inline void mi_free_block_local(mi_page_t* page, mi_block_t* block, bool if (!was_guarded) { mi_check_padding(page, block); } if (track_stats) { mi_stat_free(page, block); } #if (MI_DEBUG>0) && !MI_TRACK_ENABLED && !MI_TSAN - memset(block, MI_DEBUG_FREED, mi_page_block_size(page)); + size_t dbgsize = mi_page_block_size(page); + if (dbgsize > 1*MI_MiB) { dbgsize = 1*MI_MiB; } + _mi_memset_aligned(block, MI_DEBUG_FREED, dbgsize); #endif if (track_stats) { mi_track_free_size(block, mi_page_usable_size_of(page, block, was_guarded)); } // faster then mi_usable_size as we already know the page and that p is unaligned // actual free: push on the local free list mi_block_set_next(page, block, page->local_free); page->local_free = block; - if mi_unlikely(--page->used == 0) { + #if defined(__clang__) && defined(__aarch64__) + if mi_unlikely(page->used-- == 1) // better code on arm64 than using `--page->used == 0` + #else + if mi_unlikely(--page->used == 0) + #endif + { if (page->retire_expire==0) { // no need to re-retire retired pages (happens when we alloc/free one block repeatedly in an empty page) _mi_page_retire(page); } @@ -54,42 +61,46 @@ static inline void mi_free_block_local(mi_page_t* page, mi_block_t* block, bool static void mi_decl_noinline mi_free_try_collect_mt(mi_page_t* page, mi_block_t* mt_free) mi_attr_noexcept; // Free a block multi-threaded -static inline void mi_free_block_mt(mi_page_t* page, mi_block_t* block, bool was_guarded) mi_attr_noexcept +static inline void mi_free_block_mt(mi_page_t* page, mi_block_t* block, bool was_guarded, bool allow_collect) mi_attr_noexcept { - MI_UNUSED(was_guarded); - // adjust stats (after padding check and potentially recursive `mi_free` above) + // todo: we cannot safely check for double free in _mt -- should check when collecting the thread_free list + if (!was_guarded) { mi_check_padding(page, block); } // checking padding is safe for mt + // adjust stats (after padding check ) mi_stat_free(page, block); // stat_free may access the padding mi_track_free_size(block, mi_page_usable_size_of(page, block, was_guarded)); // _mi_padding_shrink(page, block, sizeof(mi_block_t)); -#if (MI_DEBUG>0) && !MI_TRACK_ENABLED && !MI_TSAN // note: when tracking, cannot use mi_usable_size with multi-threading + #if (MI_DEBUG>0) && !MI_TRACK_ENABLED && !MI_TSAN // note: when tracking, cannot use mi_usable_size with multi-threading if (!was_guarded) { size_t dbgsize = mi_usable_size(block); - if (dbgsize > MI_MiB) { dbgsize = MI_MiB; } + if (dbgsize > 1*MI_MiB) { dbgsize = 1*MI_MiB; } _mi_memset_aligned(block, MI_DEBUG_FREED, dbgsize); } -#endif + #endif // push atomically on the page thread free list mi_thread_free_t tf_new; mi_thread_free_t tf_old = mi_atomic_load_relaxed(&page->xthread_free); do { mi_block_set_next(page, block, mi_tf_block(tf_old)); - tf_new = mi_tf_create(block, true /* always use owned: try to claim it if the page is abandoned */); + const bool new_owned = (allow_collect ? true : mi_tf_is_owned(tf_old)); // if allow collection then always try to claim it if the page is abandoned + tf_new = mi_tf_create(block, new_owned); } while (!mi_atomic_cas_weak_acq_rel(&page->xthread_free, &tf_old, tf_new)); // todo: release is enough? // and atomically try to collect the page if it was abandoned - const bool is_owned_now = !mi_tf_is_owned(tf_old); - if (is_owned_now) { - mi_assert_internal(mi_page_is_abandoned(page)); - mi_free_try_collect_mt(page,block); + if (allow_collect) { + const bool is_owned_now = !mi_tf_is_owned(tf_old); + if (is_owned_now) { + mi_assert_internal(mi_page_is_abandoned(page)); + mi_free_try_collect_mt(page,block); + } } } // Adjust a block that was allocated aligned, to the actual start of the block in the page. // note: this can be called from `mi_free_generic_mt` where a non-owning thread accesses the -// `page_start` and `block_size` fields; however these are constant and the page won't be +// `page_woffset` and `block_size` fields; however these are constant and the page won't be // deallocated (as the block we are freeing keeps it alive) and thus safe to read concurrently. mi_block_t* _mi_page_ptr_unalign(const mi_page_t* page, const void* p) { mi_assert_internal(page!=NULL && p!=NULL); @@ -100,6 +111,17 @@ mi_block_t* _mi_page_ptr_unalign(const mi_page_t* page, const void* p) { return (mi_block_t*)((uintptr_t)p - adjust); } +static inline mi_block_t* mi_validate_block_from_ptr( const mi_page_t* page, const void* p ) { + mi_assert(_mi_page_ptr_unalign(page,p) == (mi_block_t*)p); // should never be an interior pointer + #if MI_SECURE > 0 + // in secure mode we always unalign to guard against free-ing interior pointers + return _mi_page_ptr_unalign(page,p); + #else + MI_UNUSED(page); + return (mi_block_t*)p; + #endif +} + // forward declaration for a MI_GUARDED build #if MI_GUARDED static void mi_block_unguard(mi_page_t* page, mi_block_t* block, void* p); // forward declaration @@ -119,17 +141,6 @@ static inline bool mi_block_check_unguard(mi_page_t* page, mi_block_t* block, vo } #endif -static inline mi_block_t* mi_validate_block_from_ptr( const mi_page_t* page, void* p ) { - mi_assert(_mi_page_ptr_unalign(page,p) == (mi_block_t*)p); // should never be an interior pointer - #if MI_SECURE > 0 - // in secure mode we always unalign to guard against free-ing interior pointers - return _mi_page_ptr_unalign(page,p); - #else - MI_UNUSED(page); - return (mi_block_t*)p; - #endif -} - // free a local pointer (page parameter comes first for better codegen) static void mi_decl_noinline mi_free_generic_local(mi_page_t* page, void* p) mi_attr_noexcept { @@ -140,17 +151,17 @@ static void mi_decl_noinline mi_free_generic_local(mi_page_t* page, void* p) mi_ } // free a pointer owned by another thread (page parameter comes first for better codegen) -static void mi_decl_noinline mi_free_generic_mt(mi_page_t* page, void* p) mi_attr_noexcept { +static void mi_decl_noinline mi_free_generic_mt(mi_page_t* page, void* p, bool allow_collect) mi_attr_noexcept { mi_assert_internal(p!=NULL && page != NULL); mi_block_t* const block = (mi_page_has_interior_pointers(page) ? _mi_page_ptr_unalign(page, p) : mi_validate_block_from_ptr(page,p)); const bool was_guarded = mi_block_check_unguard(page, block, p); - mi_free_block_mt(page, block, was_guarded); + mi_free_block_mt(page, block, was_guarded, allow_collect); } // generic free (for runtime integration) void mi_decl_noinline _mi_free_generic(mi_page_t* page, bool is_local, void* p) mi_attr_noexcept { if (is_local) mi_free_generic_local(page,p); - else mi_free_generic_mt(page,p); + else mi_free_generic_mt(page,p,true); } @@ -176,9 +187,12 @@ static inline mi_page_t* mi_validate_ptr_page(const void* p, const char* msg) // Free a block // Fast path written carefully to prevent register spilling on the stack -static mi_decl_forceinline void mi_free_ex(void* p, size_t* usable, mi_page_t* page) +static mi_decl_forceinline void mi_free_ex(void* p, size_t* usable, mi_page_t* page, bool allow_collect) { - if mi_unlikely(page==NULL) return; // page will be NULL if p==NULL + if mi_unlikely(page==NULL) { // page will be NULL if p==NULL + if (usable!=NULL) { *usable = 0; } + return; + } mi_assert_internal(p!=NULL && page!=NULL); if (usable!=NULL) { *usable = mi_page_usable_block_size(page); } @@ -196,22 +210,22 @@ static mi_decl_forceinline void mi_free_ex(void* p, size_t* usable, mi_page_t* p else if ((xtid & MI_PAGE_FLAG_MASK) == 0) { // `tid != mi_page_thread_id(page) && mi_page_flags(page) == 0` // blocks are aligned (and not a full page); push on the thread_free list mi_block_t* const block = mi_validate_block_from_ptr(page,p); - mi_free_block_mt(page,block,false /* was_guarded */); + mi_free_block_mt(page,block,false /* was_guarded */, allow_collect); } else { // page is full or contains (inner) aligned blocks; use generic multi-thread path - mi_free_generic_mt(page, p); + mi_free_generic_mt(page, p, allow_collect); } } void mi_free(void* p) mi_attr_noexcept { mi_page_t* const page = mi_validate_ptr_page(p,"mi_free"); - mi_free_ex(p, NULL, page); + mi_free_ex(p, NULL, page, true); } void mi_ufree(void* p, size_t* usable) mi_attr_noexcept { mi_page_t* const page = mi_validate_ptr_page(p,"mi_ufree"); - mi_free_ex(p, usable, page); + mi_free_ex(p, usable, page, true); } void mi_free_small(void* p) mi_attr_noexcept { @@ -228,15 +242,21 @@ void mi_free_small(void* p) mi_attr_noexcept { #else mi_page_t* const page = (mi_page_t*)_mi_align_down_ptr(p,MI_SMALL_PAGE_SIZE); mi_assert(page == mi_validate_ptr_page(p,"mi_free_small")); - mi_assert((void*)page == _mi_align_down_ptr(page->page_start,MI_SMALL_PAGE_SIZE)); + mi_assert((void*)page == _mi_align_down_ptr(mi_page_start(page),MI_SMALL_PAGE_SIZE)); mi_assert(page->block_size <= MI_SMALL_SIZE_MAX); // note: not `MI_SMALL_MAX_OBJ_SIZE` as we need to match `mi_(heap_)malloc_small` - mi_free_ex(p, NULL, page); + mi_free_ex(p, NULL, page, true); #endif #else mi_free(p); #endif } +// Free a pointer that is potentially allocated in a different sub-process +void _mi_free_subproc_safe(void* p) { + mi_page_t* const page = mi_validate_ptr_page(p,"_mi_free_subproc_safe"); + mi_free_ex(p, NULL, page, false); +} + // -------------------------------------------------------------------------------------------- // `mi_free_try_collect_mt`: Potentially collect a page in a free in an abandoned page. @@ -315,14 +335,14 @@ static mi_decl_noinline bool mi_abandoned_page_try_reclaim(mi_page_t* page, long mi_assert_internal(!mi_page_all_free(page)); mi_assert_internal(page->block_size <= MI_MEDIUM_MAX_OBJ_SIZE); mi_assert_internal(reclaim_on_free >= 0); - + // dont reclaim if we just have terminated this thread and we should // not reinitialize the theap for this thread. (can happen due to thread-local destructors for example -- issue #944) if (!_mi_thread_is_initialized()) return false; // get our theap mi_theap_t* const theap = _mi_page_associated_theap_peek(page); - if (theap==NULL || !theap->allow_page_reclaim) return false; + if (theap==NULL || theap->tld==NULL || !theap->allow_page_reclaim) return false; // see issue #1289 // todo: cache `is_in_threadpool` and `exclusive_arena` directly in the theap for performance? // set max_reclaim limit @@ -359,6 +379,8 @@ static void mi_decl_noinline mi_free_try_collect_mt(mi_page_t* page, mi_block_t* mi_assert_internal(mi_page_is_owned(page)); mi_assert_internal(mi_page_is_abandoned(page)); mi_assert_internal(mt_free != NULL); + // mi_assert_internal(_mi_subproc() == mi_page_subproc(page)); // never collect across subprocesses + // we own the page now, and it is safe to collect the thread atomic free list if (page->block_size <= MI_SMALL_SIZE_MAX) { // use the `_partly` version to avoid atomic operations since we already have the `mt_free` pointing into the thread free list @@ -391,7 +413,7 @@ static void mi_decl_noinline mi_free_try_collect_mt(mi_page_t* page, mi_block_t* // ------------------------------------------------------ -// Usable size +// Usable size // ------------------------------------------------------ // Bytes available in a block @@ -399,16 +421,17 @@ static size_t mi_decl_noinline mi_page_usable_aligned_size_of(const mi_page_t* p const mi_block_t* block = _mi_page_ptr_unalign(page, p); const bool is_guarded = mi_block_ptr_is_guarded(block,p); const size_t size = mi_page_usable_size_of(page, block, is_guarded); - const ptrdiff_t adjust = (uint8_t*)p - (uint8_t*)block; - mi_assert_internal(adjust >= 0 && (size_t)adjust <= size); - const size_t aligned_size = (size - adjust); + mi_assert_internal((void*)p >= (void*)block); + const size_t adjust = (uint8_t*)p - (uint8_t*)block; + mi_assert_internal(adjust <= size); + const size_t aligned_size = (adjust <= size ? size - adjust : 0); // size can be zero if the padding is corrupted return aligned_size; } static inline size_t _mi_usable_size(const void* p, const mi_page_t* page) mi_attr_noexcept { if mi_unlikely(page==NULL) return 0; if mi_likely(!mi_page_has_interior_pointers(page)) { - const mi_block_t* block = (const mi_block_t*)p; + const mi_block_t* block = mi_validate_block_from_ptr(page,p); return mi_page_usable_size_of(page, block, false /* is guarded */); } else { @@ -455,12 +478,18 @@ void mi_free_aligned(void* p, size_t alignment) mi_attr_noexcept { // This is somewhat expensive so only enabled for secure mode 4 // ------------------------------------------------------ -#if (MI_ENCODE_FREELIST && (MI_SECURE>=4 || MI_DEBUG!=0)) +#if MI_CHECK_DOUBLE_FREE // linear check if the free list contains a specific element -static bool mi_list_contains(const mi_page_t* page, const mi_block_t* list, const mi_block_t* elem) { - while (list != NULL) { +static bool mi_list_contains(const mi_page_t* page, const mi_block_t* list, const mi_block_t* elem, const char* list_kind) { + const size_t max_count = page->capacity; // can never hold more blocks than the capacity + size_t count = 0; + while (list != NULL && count <= max_count) { // double-free can create cycles so we limit the number of iterations if (elem==list) return true; list = mi_block_next(page, list); + count++; + } + if mi_unlikely(count > max_count) { + _mi_error_message(EFAULT, "corrupted %s list (possibly due to a double free)\n", list_kind); } return false; } @@ -468,9 +497,9 @@ static bool mi_list_contains(const mi_page_t* page, const mi_block_t* list, cons static mi_decl_noinline bool mi_check_is_double_freex(const mi_page_t* page, const mi_block_t* block) { // The decoded value is in the same page (or NULL). // Walk the free lists to verify positively if it is already freed - if (mi_list_contains(page, page->free, block) || - mi_list_contains(page, page->local_free, block) || - mi_list_contains(page, mi_page_thread_free(page), block)) + if (mi_list_contains(page, page->free, block, "free") || + mi_list_contains(page, page->local_free, block, "local free") || + mi_list_contains(page, mi_page_thread_free(page), block, "thread free")) { _mi_error_message(EAGAIN, "double free detected of block %p with size %zu\n", block, mi_page_block_size(page)); return true; @@ -478,19 +507,22 @@ static mi_decl_noinline bool mi_check_is_double_freex(const mi_page_t* page, con return false; } -#define mi_track_page(page,access) { size_t psize; void* pstart = _mi_page_start(_mi_page_segment(page),page,&psize); mi_track_mem_##access( pstart, psize); } +// Used for double free checking to avoid checking free lists too frequently +static inline bool mi_block_could_be_double_free(const mi_page_t* page, const mi_block_t* block) { + mi_block_t* n = mi_block_nextx(page,block,page->keys); + return (((uintptr_t)n & (MI_INTPTR_SIZE-1))==0 && // quick check: aligned pointer? + (n==NULL || mi_is_in_same_page(block,n))); // quick check: in the same page or NULL? +} +// check if `block` was free'd before static inline bool mi_check_is_double_free(const mi_page_t* page, const mi_block_t* block) { - bool is_double_free = false; - mi_block_t* n = mi_block_nextx(page, block, page->keys); // pretend it is freed, and get the decoded first field - if (((uintptr_t)n & (MI_INTPTR_SIZE-1))==0 && // quick check: aligned pointer? - (n==NULL || mi_is_in_same_page(block, n))) // quick check: in same page or NULL? + if mi_unlikely(mi_block_could_be_double_free(page,block)) // quick check: next field is aligned in the same page or NULL? { // Suspicious: decoded value a in block is in the same page (or NULL) -- maybe a double free? // (continue in separate function to improve code generation) - is_double_free = mi_check_is_double_freex(page, block); + return mi_check_is_double_freex(page, block); } - return is_double_free; + else return false; } #else static inline bool mi_check_is_double_free(const mi_page_t* page, const mi_block_t* block) { @@ -522,7 +554,7 @@ static bool mi_page_decode_padding(const mi_page_t* page, const mi_block_t* bloc // Return the exact usable size of a block. static size_t mi_page_usable_size_of(const mi_page_t* page, const mi_block_t* block, bool is_guarded) { - if (is_guarded) { + if mi_unlikely(is_guarded) { const size_t bsize = mi_page_block_size(page); return (bsize - _mi_os_page_size()); } @@ -555,9 +587,15 @@ void _mi_padding_shrink(const mi_page_t* page, const mi_block_t* block, const si mi_track_mem_noaccess(padding,sizeof(mi_padding_t)); } #else -static size_t mi_page_usable_size_of(const mi_page_t* page, const mi_block_t* block, bool is_guarded) { - MI_UNUSED(is_guarded); MI_UNUSED(block); - return mi_page_usable_block_size(page); +static inline size_t mi_page_usable_size_of(const mi_page_t* page, const mi_block_t* block, bool is_guarded) { + MI_UNUSED(block); + if mi_unlikely(is_guarded) { + const size_t bsize = mi_page_block_size(page); + return (bsize - _mi_os_page_size()); + } + else { + return mi_page_usable_block_size(page); + } } void _mi_padding_shrink(const mi_page_t* page, const mi_block_t* block, const size_t min_size) { @@ -655,4 +693,17 @@ static void mi_block_unguard(mi_page_t* page, mi_block_t* block, void* p) { mi_assert_internal(_mi_is_aligned(gpage, psize)); _mi_os_unprotect(gpage, psize); } + +// unguard a whole page (called from `mi_heap_destroy`) +void _mi_page_unguard_all(mi_page_t* page) { + if mi_likely(!mi_page_has_interior_pointers(page)) return; + uint8_t* const start = mi_page_start(page); + const size_t psize = mi_page_committed(page); + _mi_os_unprotect(start,psize); // unprotect all at once as we cannot know which blocks are guarded +} +#else +void _mi_page_unguard_all(mi_page_t* page) { + MI_UNUSED(page); + // nothing to do +} #endif diff --git a/vendored/mimalloc/src/heap.c b/vendored/mimalloc/src/heap.c index f0a016527a..d3a9d42edb 100644 --- a/vendored/mimalloc/src/heap.c +++ b/vendored/mimalloc/src/heap.c @@ -1,5 +1,5 @@ /*---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -7,7 +7,8 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc.h" #include "mimalloc/internal.h" -#include "mimalloc/prim.h" // _mi_theap_default +#include "mimalloc/prim.h" // _mi_prim_thread_yield +#include "mimalloc/prim-tls.h" // _mi_heap_theap /* ----------------------------------------------------------- @@ -15,7 +16,7 @@ terms of the MIT license. A copy of the license can be found in the file ----------------------------------------------------------- */ mi_theap_t* mi_heap_theap(mi_heap_t* heap) { - return _mi_heap_theap(heap); + return _mi_heap_theap(heap); // in prim.h } void mi_heap_set_numa_affinity(mi_heap_t* heap, int numa_node) { @@ -30,64 +31,74 @@ void mi_heap_stats_merge_to_subproc(mi_heap_t* heap) { void mi_heap_stats_merge_to_main(mi_heap_t* heap) { if (heap==NULL) return; - _mi_stats_merge_into(&mi_heap_main()->stats, &heap->stats); + _mi_stats_merge_into(&mi_heap_get_heap_main(heap)->stats, &heap->stats); } +bool _mi_heap_theap_set(mi_heap_t* heap, mi_theap_t* theap) { + mi_assert_internal((uintptr_t)theap == 1 || _mi_theap_heap(theap)==heap); + mi_assert_internal(!_mi_is_empty_theap(theap)); + mi_assert_internal(heap->theap != 0); + return _mi_thread_local_set(heap->theap,theap); +} + +// mi_theap_t* _mi_heap_theap_get_peek(const mi_heap_t* heap) { +// mi_theap_t* theap; +// mi_assert_internal(heap->theap != 0); +// if mi_likely(heap->theap!=0) { // paranoia +// theap = (mi_theap_t*)_mi_thread_local_get(heap->theap); +// } +// else { +// _mi_error_message(EFAULT, "no thread-local reserved for heap (%p)\n", heap); +// return NULL; +// } +// mi_assert_internal(!_mi_is_empty_theap(theap)); +// mi_assert_internal(theap->heap == heap); // this goes wrong if using main heaps across subprocesses (as all share the same key) +// return theap; +// } + + static mi_decl_noinline mi_theap_t* mi_heap_init_theap(const mi_heap_t* const_heap) { mi_heap_t* heap = (mi_heap_t*)const_heap; mi_assert_internal(heap!=NULL); - if (_mi_is_heap_main(heap)) { - // this can be called if the (main) thread is not yet initialized (as no allocation happened) - // but `theap_main_init_get()` will call `mi_thread_init()` - mi_theap_t* const theap = _mi_theap_main_safe(); - mi_assert_internal(theap!=NULL && _mi_is_heap_main(_mi_theap_heap(theap))); - return theap; + // if (_mi_is_process_heap_main(heap)) { + // // this can be called if the (main) thread is not yet initialized (as no allocation happened) + // // but `theap_main_init_get()` will call `mi_thread_init()` + // mi_theap_t* const theap = _mi_theap_main_safe(); + // mi_assert_internal(theap!=NULL && _mi_is_heap_main(_mi_theap_heap(theap))); + // return theap; + // } + + // initialize thread first in case this is the main heap + // (which may allocate the default theap already for the main heap) + if (!_mi_thread_is_initialized()) { + mi_thread_init(); } - // otherwise initialize the theap for this heap - // get the thread local - mi_assert_internal(heap->theap != 0); - if (heap->theap==0) { // paranoia - _mi_error_message(EFAULT, "no thread-local reserved for heap (%p)\n", heap); - return NULL; - } + // get the thread local theap mi_theap_t* theap = (mi_theap_t*)_mi_thread_local_get(heap->theap); // create a fresh theap? if (theap==NULL) { - // set first an invalid value to ensure the thread local storage is allocated - if (!_mi_thread_local_set(heap->theap, (mi_theap_t*)1)) { - _mi_error_message(EFAULT, "unable to allocate memory for thread local storage\n"); - return NULL; - } - // then allocate the theap - theap = _mi_theap_create(heap, _mi_theap_default_safe()->tld); - _mi_thread_local_set(heap->theap, theap); // Cannot fail now as it was set before. Always set so the local is valid or NULL (and not 1) + // allocate a fresh theap + theap = _mi_theap_create(heap, mi_theap_get_default()->tld); // sets the theap thread local if (theap==NULL) { _mi_error_message(EFAULT, "unable to allocate memory for a thread local heap\n"); return NULL; - } + } + _mi_heap_theap_set(heap, theap); + mi_assert_internal(theap == (mi_theap_t*)_mi_thread_local_get(heap->theap)); } return theap; } -// get the theap for a heap without initializing (and return NULL in that case) -mi_theap_t* _mi_heap_theap_get_peek(const mi_heap_t* heap) { - if (heap==NULL || _mi_is_heap_main(heap)) { - return _mi_theap_main_safe(); - } - else { - return (mi_theap_t*)_mi_thread_local_get(heap->theap); - } -} - // get (and possibly create) the theap belonging to a heap mi_theap_t* _mi_heap_theap_get_or_init(const mi_heap_t* heap) { - mi_theap_t* theap = _mi_heap_theap_peek(heap); + mi_assert_internal(heap->theap != 0); + mi_theap_t* theap = (mi_theap_t*)_mi_thread_local_get(heap->theap); if mi_unlikely(theap==NULL) { theap = mi_heap_init_theap(heap); if (theap==NULL) { return (mi_theap_t*)&_mi_theap_empty_wrong; } // this will return NULL from page.c:_mi_malloc_generic @@ -96,26 +107,12 @@ mi_theap_t* _mi_heap_theap_get_or_init(const mi_heap_t* heap) return theap; } - -mi_heap_t* mi_heap_new_in_arena(mi_arena_id_t exclusive_arena_id) { - // always allocate heap data in the (subprocess) main heap - mi_heap_t* const heap_main = mi_heap_main(); - // todo: allocate heap data in the exclusive arena ? - mi_heap_t* const heap = (mi_heap_t*)mi_heap_zalloc( heap_main, sizeof(mi_heap_t) ); - if (heap==NULL) return NULL; - - // reserve a thread local slot for this heap (see also issue #1230) - const mi_thread_local_t theap_slot = _mi_thread_local_create(); - if (theap_slot == 0) { - _mi_error_message(EFAULT, "unable to dynamically create a thread local for a heap\n"); - mi_free(heap); - return NULL; - } - +static void mi_heap_initialize(mi_heap_t* heap, mi_thread_local_t theap_slot, mi_subproc_t* subproc, mi_arena_id_t exclusive_arena_id) +{ // init fields heap->theap = theap_slot; - heap->subproc = heap_main->subproc; - heap->heap_seq = mi_atomic_increment_relaxed(&heap_main->subproc->heap_total_count); + heap->subproc = subproc; + heap->heap_seq = mi_atomic_increment_relaxed(&subproc->heap_total_count); heap->exclusive_arena = _mi_arena_from_id(exclusive_arena_id); heap->numa_node = -1; // no initial affinity mi_stats_header_init(&heap->stats); @@ -131,48 +128,74 @@ mi_heap_t* mi_heap_new_in_arena(mi_arena_id_t exclusive_arena_id) { if (head!=NULL) { head->prev = heap; } heap->subproc->heaps = heap; } - mi_atomic_increment_relaxed(&heap_main->subproc->heap_count); - mi_subproc_stat_increase(heap_main->subproc, heaps, 1); + mi_atomic_increment_relaxed(&subproc->heap_count); + mi_subproc_stat_increase(subproc, heaps, 1); + mi_assert_internal(_mi_is_heap_main(heap) ? heap->theap == mi_thread_local_key_fast : heap->theap != 0); +} + +mi_heap_t* _mi_heap_new_for_subproc(mi_subproc_t* subproc, mi_arena_id_t exclusive_arena_id, bool is_main_heap) { + mi_assert_internal(is_main_heap ? (subproc->heap_main == NULL && subproc->parent != NULL) : subproc->heap_main != NULL); + // heap data is allocated in the current subproc + mi_heap_t* const heap_main = (is_main_heap ? subproc->parent->heap_main : subproc->heap_main); + // todo: allocate heap data in the exclusive arena ? + mi_heap_t* const heap = (mi_heap_t*)mi_heap_zalloc( heap_main, sizeof(mi_heap_t) ); + if (heap==NULL) return NULL; + // reserve a thread local slot for this heap (see also issue #1230) + mi_thread_local_t theap_slot = (is_main_heap ? mi_thread_local_key_fast : _mi_thread_local_create()); + if (theap_slot == 0) { + _mi_error_message(EFAULT, "unable to dynamically create a thread local for a heap\n"); + mi_free(heap); + return NULL; + } + if (is_main_heap) { + mi_assert_internal(subproc->heap_main == NULL); + subproc->heap_main = heap; + } + mi_heap_initialize(heap, theap_slot, subproc, exclusive_arena_id); return heap; } +mi_heap_t* mi_heap_new_in_arena(mi_arena_id_t exclusive_arena_id) { + return _mi_heap_new_for_subproc(_mi_subproc(), exclusive_arena_id, false); +} + mi_heap_t* mi_heap_new(void) { return mi_heap_new_in_arena(0); } // free all theaps belonging to this heap (without deleting their pages as we do this arena wise for efficiency) static void mi_heap_free_theaps(mi_heap_t* heap) { - // This can run concurrently with a thread that terminates (see `init.c:mi_thread_theaps_done`), + // This can run concurrently with a thread that terminates (see `init.c:mi_thread_theaps_done`), // and we need to ensure we free theaps atomically. - // We do this in a loop where we release the theaps_lock at every potential re-iteration to unblock + // We do this in a loop where we release the theaps_lock at every potential re-iteration to unblock // potential concurrent thread termination which tries to remove the theap from our theaps list. bool all_freed; do { all_freed = true; mi_theap_t* theap = NULL; - mi_lock(&heap->theaps_lock) { - theap = heap->theaps; + mi_lock(&heap->theaps_lock) { + theap = heap->theaps; while(theap != NULL) { mi_theap_t* next = theap->hnext; if (!_mi_theap_free(theap, false /* dont re-acquire the heap->theaps_lock */, true /* acquire the tld->theaps_lock though */ )) { all_freed = false; } theap = next; - } + } } - if (!all_freed) { - mi_heap_stat_counter_increase(heap,heaps_delete_wait,1); + if (!all_freed) { + mi_heap_stat_counter_increase(heap,heaps_delete_wait,1); _mi_prim_thread_yield(); } - else { - mi_assert_internal(heap->theaps==NULL); - } + else { + mi_assert_internal(heap->theaps==NULL); + } } while(!all_freed); } // free the heap resources (assuming the pages are already moved/destroyed, and all theaps have been freed) -static void mi_heap_free(mi_heap_t* heap) { +static void mi_heap_free(mi_heap_t* heap, bool acquire_heaps_lock) { mi_assert_internal(heap!=NULL && !_mi_is_heap_main(heap)); // free all arena pages infos @@ -181,7 +204,7 @@ static void mi_heap_free(mi_heap_t* heap) { mi_arena_pages_t* arena_pages = mi_atomic_load_ptr_relaxed(mi_arena_pages_t, &heap->arena_pages[i]); if (arena_pages!=NULL) { mi_atomic_store_ptr_relaxed(mi_arena_pages_t, &heap->arena_pages[i], NULL); - mi_free(arena_pages); + _mi_free_subproc_safe(arena_pages); } } } @@ -190,7 +213,7 @@ static void mi_heap_free(mi_heap_t* heap) { mi_heap_stats_merge_to_main(heap); mi_atomic_decrement_relaxed(&heap->subproc->heap_count); mi_subproc_stat_decrease(heap->subproc, heaps, 1); - mi_lock(&heap->subproc->heaps_lock) { + mi_lock_maybe(&heap->subproc->heaps_lock, acquire_heaps_lock) { if (heap->next!=NULL) { heap->next->prev = heap->prev; } if (heap->prev!=NULL) { heap->prev->next = heap->next; } else { heap->subproc->heaps = heap->next; } @@ -200,25 +223,26 @@ static void mi_heap_free(mi_heap_t* heap) { mi_lock_done(&heap->theaps_lock); mi_lock_done(&heap->os_abandoned_pages_lock); mi_lock_done(&heap->arena_pages_lock); - mi_free(heap); + _mi_free_subproc_safe(heap); } void mi_heap_delete(mi_heap_t* heap) { if (heap==NULL) return; - if (_mi_is_heap_main(heap)) { + mi_heap_t* heap_main = mi_heap_get_heap_main(heap); + if (heap == heap_main) { _mi_warning_message("cannot delete the main heap\n"); return; } mi_heap_free_theaps(heap); - _mi_heap_move_pages(heap, mi_heap_main()); - mi_heap_free(heap); + _mi_heap_move_pages(heap, heap_main); + mi_heap_free(heap,true /* acquire subproc->heaps_lock */); } -void _mi_heap_force_destroy(mi_heap_t* heap) { +void _mi_heap_force_destroy(mi_heap_t* heap, bool acquire_heaps_lock) { if (heap==NULL) return; mi_heap_free_theaps(heap); _mi_heap_destroy_pages(heap); - if (!_mi_is_heap_main(heap)) { mi_heap_free(heap); } // todo: release locks of the main heap? + if (!_mi_is_heap_main(heap)) { mi_heap_free(heap, acquire_heaps_lock); } // todo: release locks of the main heap? } void mi_heap_destroy(mi_heap_t* heap) { @@ -227,17 +251,18 @@ void mi_heap_destroy(mi_heap_t* heap) { _mi_warning_message("cannot destroy the main heap\n"); return; } - _mi_heap_force_destroy(heap); + _mi_heap_force_destroy(heap,true /* acquire subproc->heaps_lock */); } mi_heap_t* mi_heap_of(const void* p) { - mi_page_t* page = _mi_safe_ptr_page(p); + mi_page_t* const page = _mi_safe_ptr_page(p); if (page==NULL) return NULL; return mi_page_heap(page); } bool mi_any_heap_contains(const void* p) { - return (mi_heap_of(p)!=NULL); + mi_page_t* const page = _mi_safe_ptr_page(p); + return (page!=NULL); } bool mi_heap_contains(const mi_heap_t* heap, const void* p) { @@ -257,7 +282,7 @@ bool mi_unsafe_heap_page_is_under_utilized(mi_heap_t* heap, void* p, size_t perc if (p==NULL) return false; const mi_page_t* const page = _mi_safe_ptr_page(p); // Get the page containing this pointer if (page==NULL || page->used==page->capacity || page->capacity < page->reserved) return false; - // If the page is the head of the queue, it is currently being used for + // If the page is the head of the queue, it is currently being used for // allocations; we skip it to avoid immediate thrashing. if (page->prev == NULL) return false; @@ -265,7 +290,7 @@ bool mi_unsafe_heap_page_is_under_utilized(mi_heap_t* heap, void* p, size_t perc const mi_heap_t* const page_heap = mi_page_heap(page); if (page_heap==NULL) return false; if (heap!=NULL && page_heap!=heap) return false; - + // check utilization if (page->capacity==0) return false; if (perc_threshold>=100) return true; diff --git a/vendored/mimalloc/src/init.c b/vendored/mimalloc/src/init.c index 1fabd89188..37bf358300 100644 --- a/vendored/mimalloc/src/init.c +++ b/vendored/mimalloc/src/init.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -7,6 +7,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc.h" #include "mimalloc/internal.h" #include "mimalloc/prim.h" +#include "mimalloc/prim-tls.h" #include // memcpy, memset #include // atexit @@ -26,14 +27,14 @@ const mi_page_t _mi_page_empty = { NULL, // local_free MI_ATOMIC_VAR_INIT(0), // xthread_free 0, // block_size - NULL, // page_start + 0, // page_woffset + MI_ARENA_SLICE_SIZE, // page_committed #if (MI_PADDING || MI_ENCODE_FREELIST) { 0, 0 }, // keys #endif NULL, // theap NULL, // heap NULL, NULL, // next, prev - MI_ARENA_SLICE_SIZE, // page_committed MI_MEMID_STATIC // memid }; @@ -114,7 +115,9 @@ static mi_decl_cache_align mi_tld_t tld_empty = { mi_decl_cache_align const mi_theap_t _mi_theap_empty = { &tld_empty, // tld MI_ATOMIC_VAR_INIT(NULL), // heap + MI_ATOMIC_VAR_INIT(NULL), // subproc MI_ATOMIC_VAR_INIT(1), // refcount + MI_ATOMIC_VAR_INIT(0), // freed 0, // heartbeat 0, // cookie { {0}, {0}, 0, true }, // random @@ -128,7 +131,7 @@ mi_decl_cache_align const mi_theap_t _mi_theap_empty = { false, // allow reclaim true, // allow abandon #if MI_GUARDED - 0, 0, 0, 1, // sample count is 1 so we never write to it (see `internal.h:mi_theap_malloc_use_guarded`) + 0, 0, 0, 1, // rate is 0 and count is 1 so we never write to it (see `internal.h:mi_heap_malloc_use_guarded`) #endif MI_SMALL_PAGES_EMPTY, MI_PAGE_QUEUES_EMPTY, @@ -139,9 +142,11 @@ mi_decl_cache_align const mi_theap_t _mi_theap_empty = { mi_decl_cache_align const mi_theap_t _mi_theap_empty_wrong = { &tld_empty, // tld MI_ATOMIC_VAR_INIT(NULL), // heap + MI_ATOMIC_VAR_INIT(NULL), // subproc MI_ATOMIC_VAR_INIT(1), // refcount + MI_ATOMIC_VAR_INIT(0), // freed 0, // heartbeat - 0, // cookie + 1, // cookie (see issue #1343) { {0}, {0}, 0, true }, // random 0, // page count MI_BIN_FULL, 0, // page retired min/max @@ -153,7 +158,7 @@ mi_decl_cache_align const mi_theap_t _mi_theap_empty_wrong = { false, // allow reclaim true, // allow abandon #if MI_GUARDED - 0, 0, 0, 1, // sample count is 1 so we never write to it (see `internal.h:mi_theap_malloc_use_guarded`) + 0, 0, 0, 1, // rate is 0 and count is 1 so we never write to it (see `internal.h:mi_heap_malloc_use_guarded`) #endif MI_SMALL_PAGES_EMPTY, MI_PAGE_QUEUES_EMPTY, @@ -163,25 +168,29 @@ mi_decl_cache_align const mi_theap_t _mi_theap_empty_wrong = { // Heap for the main thread -extern mi_decl_hidden mi_decl_cache_align mi_theap_t theap_main; -extern mi_decl_hidden mi_decl_cache_align mi_heap_t heap_main; +#define MI_THREADID_INVALID ((mi_threadid_t)(~0)) + +extern mi_decl_hidden mi_decl_cache_align mi_theap_t mi_theap_main; // theap of the main thread (belonging to the `mi_process_heap_main`) +extern mi_decl_hidden mi_decl_cache_align mi_heap_t mi_process_heap_main; // main heap of the main subproc -static mi_decl_cache_align mi_tld_t tld_main = { +static mi_decl_cache_align mi_tld_t mi_process_tld_main = { 0, // thread_id 0, // thread_seq 0, // numa node &subproc_main, // subproc - &theap_main, // theaps list + &mi_theap_main, // theaps list MI_LOCK_INITIALIZER, // theaps lock false, // recurse false, // is_in_threadpool MI_MEMID_STATIC // memid }; -mi_decl_cache_align mi_theap_t theap_main = { - &tld_main, // thread local data - MI_ATOMIC_VAR_INIT(&heap_main), // main heap +mi_decl_cache_align mi_theap_t mi_theap_main = { + &mi_process_tld_main, // thread local data + MI_ATOMIC_VAR_INIT(&mi_process_heap_main), // main heap + MI_ATOMIC_VAR_INIT(&subproc_main), // main subproc MI_ATOMIC_VAR_INIT(1), // refcount + MI_ATOMIC_VAR_INIT(0), // freed 0, // heartbeat 0, // initial cookie { {0x846ca68b}, {0}, 0, true }, // random @@ -203,30 +212,29 @@ mi_decl_cache_align mi_theap_t theap_main = { { sizeof(mi_stats_t), MI_STAT_VERSION, MI_STATS_NULL }, // stats }; -mi_decl_cache_align mi_heap_t heap_main +mi_decl_cache_align mi_heap_t mi_process_heap_main #if __cplusplus = { }; // empty initializer to prevent running the constructor (with msvc) #else = { 0 }; // C zero initialize #endif -// the theap belonging to the main heap -mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_main = NULL; - mi_threadid_t _mi_thread_id(void) mi_attr_noexcept { - mi_threadid_t tid = _mi_prim_thread_id(); + const mi_threadid_t tid = _mi_prim_thread_id(); mi_assert_internal( (tid & 0x03) == 0 ); // mimalloc reserves the bottom 2 bits return tid; } -#if MI_TLS_MODEL_THREAD_LOCAL +mi_decl_hidden mi_decl_thread void* __mi_thread_id_helper = NULL; + +#if MI_TLS_MODEL_LOCAL // the thread-local main theap for allocation mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_default = (mi_theap_t*)&_mi_theap_empty; // the last used non-main theap mi_decl_hidden mi_decl_thread mi_theap_t* __mi_theap_cached = (mi_theap_t*)&_mi_theap_empty; #endif -bool _mi_process_is_initialized = false; // set to `true` in `mi_process_init`. +mi_decl_hidden bool _mi_process_is_initialized = false; // set to `true` in `mi_process_init`. mi_stats_t _mi_stats_main = { sizeof(mi_stats_t), MI_STAT_VERSION, MI_STATS_NULL }; @@ -283,24 +291,24 @@ void _mi_theap_guarded_init(mi_theap_t* theap) { static void mi_subproc_main_init(void) { if (subproc_main.memid.memkind != MI_MEM_STATIC) { subproc_main.memid = _mi_memid_create(MI_MEM_STATIC); - subproc_main.heaps = &heap_main; + subproc_main.heaps = &mi_process_heap_main; subproc_main.heap_total_count = 1; subproc_main.heap_count = 1; - mi_atomic_store_ptr_release(mi_heap_t, &subproc_main.heap_main, &heap_main); + mi_atomic_store_ptr_release(mi_heap_t, &subproc_main.heap_main, &mi_process_heap_main); __mi_stat_increase_mt(&subproc_main.stats.heaps, 1); mi_stats_header_init(&subproc_main.stats); mi_lock_init(&subproc_main.arena_reserve_lock); mi_lock_init(&subproc_main.heaps_lock); mi_lock_init(&subprocs_lock); - mi_lock_init(&tld_empty.theaps_lock); + mi_lock_init(&tld_empty.theaps_lock); } } // Initialize main tld static void mi_tld_main_init(void) { - if (tld_main.thread_id == 0) { - tld_main.thread_id = _mi_prim_thread_id(); - mi_lock_init(&tld_main.theaps_lock); + if (mi_process_tld_main.thread_id == 0) { + mi_process_tld_main.thread_id = _mi_prim_thread_id(); + mi_lock_init(&mi_process_tld_main.theaps_lock); } } @@ -312,33 +320,35 @@ void _mi_theap_options_init(mi_theap_t* theap) { // Initialization of the (statically allocated) main theap, and the main tld and subproc. static void mi_theap_main_init(void) { - if mi_unlikely(theap_main.memid.memkind != MI_MEM_STATIC) { + if mi_unlikely(mi_theap_main.memid.memkind != MI_MEM_STATIC) { // theap - theap_main.memid = _mi_memid_create(MI_MEM_STATIC); - #if defined(__APPLE__) || defined(_WIN32) && !defined(MI_SHARED_LIB) - _mi_random_init_weak(&theap_main.random); // prevent allocation failure during bcrypt dll initialization with static linking (issue #1185) + mi_theap_main.memid = _mi_memid_create(MI_MEM_STATIC); + #if defined(__APPLE__) || (defined(_WIN32) && !defined(MI_SHARED_LIB)) + _mi_random_init_weak(&mi_theap_main.random); // prevent allocation failure during bcrypt dll initialization with static linking (issue #1185) #else - _mi_random_init(&theap_main.random); + _mi_random_init(&mi_theap_main.random); #endif - theap_main.cookie = _mi_theap_random_next(&theap_main); - _mi_theap_options_init(&theap_main); - _mi_theap_guarded_init(&theap_main); + mi_theap_main.cookie = _mi_theap_random_next(&mi_theap_main); + _mi_theap_options_init(&mi_theap_main); + _mi_theap_guarded_init(&mi_theap_main); } } // Initialize main heap static void mi_heap_main_init(void) { - if mi_unlikely(heap_main.subproc == NULL) { - heap_main.subproc = &subproc_main; - heap_main.theaps = &theap_main; + if mi_unlikely(mi_process_heap_main.subproc == NULL) { + mi_process_heap_main.subproc = &subproc_main; + mi_process_heap_main.theaps = &mi_theap_main; + mi_process_heap_main.theap = mi_thread_local_key_fast; mi_theap_main_init(); mi_subproc_main_init(); mi_tld_main_init(); + // mi_heap_theap_set(&mi_process_heap_main,&mi_theap_main); // set in `mi_thread_init(_theap_default)` - mi_lock_init(&heap_main.theaps_lock); - mi_lock_init(&heap_main.os_abandoned_pages_lock); - mi_lock_init(&heap_main.arena_pages_lock); + mi_lock_init(&mi_process_heap_main.theaps_lock); + mi_lock_init(&mi_process_heap_main.os_abandoned_pages_lock); + mi_lock_init(&mi_process_heap_main.arena_pages_lock); } } @@ -348,17 +358,18 @@ static void mi_heap_main_init(void) { ----------------------------------------------------------- */ // Allocate fresh tld -static mi_tld_t* mi_tld_alloc(void) { - if (_mi_is_main_thread()) { - mi_atomic_increment_relaxed(&tld_main.subproc->thread_count); - return &tld_main; - } - else { +static mi_tld_t* mi_tld_alloc(mi_subproc_t* subproc) { + // if (_mi_is_main_thread()) { + // mi_atomic_increment_relaxed(&tld_main.subproc->thread_count); + // return &tld_main; + // } + // else + { // allocate tld meta-data // note: we need to be careful to not access the tld from `_mi_meta_zalloc` // (and in turn from `_mi_arena_alloc_aligned` and `_mi_os_alloc_aligned`). mi_memid_t memid; - mi_tld_t* tld = (mi_tld_t*)_mi_meta_zalloc(sizeof(mi_tld_t), &memid); + mi_tld_t* tld = (mi_tld_t*)_mi_meta_zalloc(subproc, sizeof(mi_tld_t), &memid); if (tld==NULL) { _mi_error_message(ENOMEM, "unable to allocate memory for thread local data\n"); return NULL; @@ -366,7 +377,7 @@ static mi_tld_t* mi_tld_alloc(void) { tld->memid = memid; tld->theaps = NULL; mi_lock_init(&tld->theaps_lock); - tld->subproc = &subproc_main; + tld->subproc = subproc; tld->numa_node = _mi_os_numa_node(); tld->thread_id = _mi_prim_thread_id(); tld->thread_seq = mi_atomic_increment_relaxed(&tld->subproc->thread_total_count); @@ -379,42 +390,13 @@ static mi_tld_t* mi_tld_alloc(void) { #define MI_TLD_INVALID ((mi_tld_t*)1) mi_decl_noinline static void mi_tld_free(mi_tld_t* tld) { + if (tld==NULL || tld==MI_TLD_INVALID) return; + mi_atomic_decrement_relaxed(&tld->subproc->thread_count); + tld->thread_id = MI_THREADID_INVALID; // note: not 0 as that would re-initialize tld_main + // we also need to set an invalid tid for tld_main as sometimes the same thread-id + // is reused by the OS after a thread has terminated. (see issue #1287) mi_lock_done(&tld->theaps_lock); - if (tld != NULL && tld != MI_TLD_INVALID) { - mi_atomic_decrement_relaxed(&tld->subproc->thread_count); - _mi_meta_free(tld, sizeof(mi_tld_t), tld->memid); - } - #if 0 - // do not read/write to `thread_tld` on older macOS <= 14 as that will re-initialize the thread local storage - // (since we are calling this during pthread shutdown) - // (and this could happen on other systems as well, so let's never do it) - thread_tld = MI_TLD_INVALID; - #endif -} - -// return the thread local heap ensuring it is initialized (and not `NULL` or `&_mi_theap_empty`); -mi_theap_t* _mi_theap_default_safe(void) { - mi_theap_t* theap = _mi_theap_default(); - if mi_likely(mi_theap_is_initialized(theap)) return theap; - mi_thread_init(); - mi_assert_internal(mi_theap_is_initialized(_mi_theap_default())); - return _mi_theap_default(); -} - -// return the main theap ensuring it is initialized. -mi_theap_t* _mi_theap_main_safe(void) { - mi_theap_t* theap = __mi_theap_main; - if mi_unlikely(theap==NULL) { // if thread_init or default_set was never called - mi_thread_init(); // sets the default slot to the main theap - theap = _mi_theap_default(); - mi_assert_internal(theap!=NULL); - mi_assert_internal(_mi_is_theap_main(theap)); - if (_mi_is_theap_main(theap)) { - __mi_theap_main = theap; - } - } - mi_assert_internal(theap!=NULL && _mi_is_theap_main(theap)); - return theap; + _mi_meta_free(tld->subproc, tld, sizeof(mi_tld_t), tld->memid); // note: safe for static tld_main } @@ -428,7 +410,7 @@ mi_subproc_t* _mi_subproc(void) { // on such systems we can check for this with the _mi_prim_get_default_theap as those are protected (by being // stored in a TLS slot for example) mi_theap_t* theap = _mi_theap_default(); - if (theap == NULL) { + if (theap == NULL || theap->tld == NULL) { // see issue #1289 return _mi_subproc_main(); } else { @@ -437,14 +419,18 @@ mi_subproc_t* _mi_subproc(void) { } mi_heap_t* _mi_subproc_heap_main(mi_subproc_t* subproc) { - mi_heap_t* heap = mi_atomic_load_ptr_relaxed(mi_heap_t,&subproc->heap_main); + mi_heap_t* heap = mi_atomic_load_ptr_acquire(mi_heap_t,&subproc->heap_main); if mi_likely(heap!=NULL) { return heap; } - else { + else if (subproc==_mi_subproc_main()) { mi_heap_main_init(); - mi_assert_internal(mi_atomic_load_relaxed(&subproc->heap_main) != NULL); - return mi_atomic_load_ptr_relaxed(mi_heap_t,&subproc->heap_main); + mi_assert_internal(mi_atomic_load_ptr_acquire(mi_heap_t,&subproc->heap_main) != NULL); + return mi_atomic_load_ptr_acquire(mi_heap_t,&subproc->heap_main); + } + else { + mi_assert_internal(false); + return &mi_process_heap_main; } } @@ -452,14 +438,10 @@ mi_heap_t* mi_heap_main(void) { return _mi_subproc_heap_main(_mi_subproc()); // don't use mi_theap_main_init_get() so this call works during process_init } -bool _mi_is_heap_main(const mi_heap_t* heap) { - mi_assert_internal(heap!=NULL); - return (_mi_subproc_heap_main(heap->subproc) == heap); +bool _mi_is_process_heap_main(const mi_heap_t* heap) { + return (heap == NULL || heap == &mi_process_heap_main); } -bool _mi_is_theap_main(const mi_theap_t* theap) { - return (mi_theap_is_initialized(theap) && _mi_is_heap_main(_mi_theap_heap(theap))); -} /* ----------------------------------------------------------- Sub process @@ -485,10 +467,14 @@ mi_subproc_id_t mi_subproc_current(void) { mi_subproc_id_t mi_subproc_new(void) { static _Atomic(size_t) subproc_total_count; + mi_subproc_t* const parent = _mi_subproc(); mi_memid_t memid; - mi_subproc_t* subproc = (mi_subproc_t*)_mi_meta_zalloc(sizeof(mi_subproc_t),&memid); - if (subproc == NULL) return _mi_subproc_to_id(NULL); + mi_subproc_t* subproc = (mi_subproc_t*)_mi_meta_zalloc(parent, sizeof(mi_subproc_t),&memid); + if (subproc == NULL) { return _mi_subproc_to_id(NULL); } + + // init subproc subproc->memid = memid; + subproc->parent = parent; subproc->subproc_seq = mi_atomic_increment_relaxed(&subproc_total_count) + 1; mi_stats_header_init(&subproc->stats); mi_lock_init(&subproc->arena_reserve_lock); @@ -499,6 +485,15 @@ mi_subproc_id_t mi_subproc_new(void) { if (subprocs!=NULL) { subprocs->prev = subproc; } subprocs = subproc; } + + // init main heap + mi_heap_t* heap_main = _mi_heap_new_for_subproc(subproc,0,true); + if (heap_main==NULL) { + mi_subproc_destroy(_mi_subproc_to_id(subproc)); + return _mi_subproc_to_id(NULL); + } + mi_assert_internal(subproc->heap_main == heap_main); + return _mi_subproc_to_id(subproc); } @@ -519,11 +514,18 @@ static void mi_subproc_unsafe_destroy(mi_subproc_t* subproc, bool acquire_subpro mi_heap_t* heap = subproc->heaps; while (heap != NULL) { mi_heap_t* next = heap->next; - if (heap!=subproc->heap_main) { mi_heap_destroy(heap); } + if (heap!=subproc->heap_main) { _mi_heap_force_destroy(heap,false /* do not re-acquire the heaps_lock */); } heap = next; } - mi_assert_internal(subproc->heaps == subproc->heap_main); - _mi_heap_force_destroy(subproc->heap_main); // no warning if destroying the main heap + mi_assert_internal(subproc->heap_main==NULL || subproc->heaps == subproc->heap_main); + if (subproc->heap_main!=NULL) { + _mi_heap_force_destroy(subproc->heap_main,false /* do not re-acquire the heaps_lock */); // no warning if destroying the main heap + } + } + + if (subproc==&subproc_main) { + // for the main subproc, release the thread locals now (as they may free memory) + _mi_thread_locals_done(); } // remove associated arenas @@ -539,11 +541,11 @@ static void mi_subproc_unsafe_destroy(mi_subproc_t* subproc, bool acquire_subpro mi_lock_done(&subproc->arena_reserve_lock); mi_lock_done(&subproc->heaps_lock); if (subproc!=&subproc_main) { - _mi_meta_free(subproc, sizeof(mi_subproc_t), subproc->memid); + _mi_meta_free( subproc->parent, subproc, sizeof(mi_subproc_t), subproc->memid); } else { // for the main subproc, also release the global page map - _mi_page_map_unsafe_destroy(&subproc_main); + _mi_page_map_unsafe_destroy(); } } @@ -563,23 +565,28 @@ static void mi_subprocs_unsafe_destroy_all(void) { } subproc = next; } - } + } mi_subproc_unsafe_destroy(&subproc_main, true /* take subprocs lock */); } +static mi_theap_t* mi_thread_init_ex(mi_heap_t* heap_main) mi_attr_noexcept; void mi_subproc_add_current_thread(mi_subproc_id_t subproc_id) { mi_subproc_t* subproc = _mi_subproc_from_id(subproc_id); - mi_tld_t* const tld = _mi_theap_default_safe()->tld; - mi_assert(tld->subproc== &subproc_main); - if (tld->subproc != &subproc_main) { - _mi_warning_message("unable to add thread to the subprocess as it was already in another subprocess (id: %p)\n", subproc); + mi_assert_internal(subproc!=NULL); + if (subproc==NULL) return; + mi_assert_internal(subproc->heap_main!=NULL); + if (subproc->heap_main==NULL) return; + mi_theap_t* theap = _mi_theap_default(); + if (mi_theap_is_initialized(theap)) { + if (theap->tld!=NULL && theap->tld->subproc != subproc) { + _mi_warning_message("unable to add thread to the subprocess as it was already in another subprocess (at %p)\n", theap->tld->subproc); + } return; } - tld->subproc = subproc; - tld->thread_seq = mi_atomic_increment_relaxed(&subproc->thread_total_count); - mi_atomic_decrement_relaxed(&subproc_main.thread_count); - mi_atomic_increment_relaxed(&subproc->thread_count); + + // initialize this thread tld & theap + mi_thread_init_ex(subproc->heap_main); } @@ -600,26 +607,62 @@ bool mi_subproc_visit_heaps(mi_subproc_id_t subproc_id, mi_heap_visit_fun* visit Allocate theap data ----------------------------------------------------------- */ +static mi_theap_t* mi_heap_check_for_existing_theap(mi_heap_t* heap) { + const mi_threadid_t tid = _mi_thread_id(); + mi_theap_t* thread_theap = NULL; + mi_lock(&heap->theaps_lock) { + for(mi_theap_t* theap = heap->theaps; theap != NULL; theap = theap->hnext ) { + if (theap->tld->thread_id == tid) { + thread_theap = theap; + break; + } + } + } + return thread_theap; +} + // Initialize the thread local default theap, called from `mi_thread_init` -static mi_theap_t* _mi_thread_init_theap_default(void) { +static mi_theap_t* _mi_thread_init_theap_default(mi_heap_t* heap_main) { mi_theap_t* theap = _mi_theap_default(); if (mi_theap_is_initialized(theap)) return theap; - if (_mi_is_main_thread()) { + if (_mi_is_main_thread() && heap_main==NULL) { + heap_main = &mi_process_heap_main; + theap = &mi_theap_main; mi_heap_main_init(); - theap = &theap_main; } else { // allocates tld data // note: we cannot access thread-locals yet as that can cause (recursive) allocation // (on macOS <= 14 for example where the loader allocates thread-local data on demand). - mi_tld_t* tld = mi_tld_alloc(); - if (tld==NULL) return NULL; // things are very wrong if this fails (out of memory) - // allocate and initialize the theap for the main heap - theap = _mi_theap_create(mi_heap_main(), tld); + if (heap_main==NULL) { + heap_main = mi_heap_main(); + mi_assert_internal(heap_main == &mi_process_heap_main); + } + mi_assert_internal(heap_main!=NULL); + theap = mi_heap_check_for_existing_theap(heap_main); + if (theap==NULL) + { + // allocated the tld + mi_tld_t* tld = mi_tld_alloc(heap_main->subproc); + if (tld==NULL) return NULL; // out-of-memory on tld allocation + + // allocate and initialize the theap for the main heap + theap = _mi_theap_create( heap_main, tld); + if (theap==NULL) { + mi_tld_free(tld); + return NULL; // out-of-memory on theap allocation + } + } } - // associate the theap with this thread - // (this is safe, on macOS for example, the theap is set in a dedicated TLS slot and thus does not cause recursive allocation) + // now initialize the thread _mi_theap_default_set(theap); + + // and only then set the heap_theap field as that accesses thread locals + _mi_heap_theap_set(heap_main, theap); // todo: can fail! + + mi_assert_internal(mi_theap_is_initialized(theap)); + mi_theap_t* const heap_theap = (heap_main==NULL ? NULL : (mi_theap_t*)_mi_thread_local_get(heap_main->theap)); + mi_assert_internal(heap_main==NULL || heap_theap == theap); MI_UNUSED_RELEASE(heap_theap); return theap; } @@ -627,16 +670,11 @@ static mi_theap_t* _mi_thread_init_theap_default(void) { // Free the thread local theaps static void mi_thread_theaps_done(mi_tld_t* tld) { - // reset the thread local theaps - _mi_theap_default_set((mi_theap_t*)&_mi_theap_empty); - _mi_theap_cached_set((mi_theap_t*)&_mi_theap_empty); - __mi_theap_main = NULL; - // abandon the pages of all theaps in this thread mi_lock(&tld->theaps_lock) { mi_theap_t* theap = tld->theaps; while (theap != NULL) { - mi_theap_t* next = theap->tnext; + mi_theap_t* next = theap->tnext; // never destroy theaps; if a dll is linked statically with mimalloc, // there may still be delete/free calls after the mi_fls_done is called. Issue #207 _mi_theap_collect_abandon(theap); @@ -645,9 +683,15 @@ static void mi_thread_theaps_done(mi_tld_t* tld) } } + // reset the thread local theaps + // note: do this after abandon as page->heap may be NULL and mi_heap_main should return the heap + // belonging to the right subprocess + _mi_theap_default_set((mi_theap_t*)&_mi_theap_empty); + _mi_theap_cached_set((mi_theap_t*)&_mi_theap_empty); + // free the theaps of this thread. // This can run concurrently with a `mi_heap_free_theaps` and we need to ensure we free theaps atomically. - // We do this in a loop where we release the theaps_lock at every potential re-iteration to unblock + // We do this in a loop where we release the theaps_lock at every potential re-iteration to unblock // potential concurrent `mi_heap_free_theaps` which tries to remove the theap from our theaps list. bool all_freed; do { @@ -663,12 +707,12 @@ static void mi_thread_theaps_done(mi_tld_t* tld) theap = next; } } - if (!all_freed) { - mi_subproc_stat_counter_increase(tld->subproc,heaps_delete_wait,1); - _mi_prim_thread_yield(); + if (!all_freed) { + mi_subproc_stat_counter_increase(tld->subproc,heaps_delete_wait,1); + _mi_prim_thread_yield(); } - else { - mi_assert_internal(tld->theaps==NULL); + else { + mi_assert_internal(tld->theaps==NULL); } } while (!all_freed); @@ -677,7 +721,6 @@ static void mi_thread_theaps_done(mi_tld_t* tld) } - // -------------------------------------------------------- // Try to run `mi_thread_done()` automatically so any memory // owned by the thread but not yet released can be abandoned @@ -698,29 +741,40 @@ static void mi_thread_theaps_done(mi_tld_t* tld) static void mi_process_setup_auto_thread_done(void) { mi_atomic_do_once { _mi_prim_thread_init_auto_done(); - _mi_theap_default_set(&theap_main); + _mi_theap_default_set(&mi_theap_main); } } bool _mi_is_main_thread(void) { - return (tld_main.thread_id==0 || tld_main.thread_id == _mi_thread_id()); + return (mi_process_tld_main.thread_id==0 || mi_process_tld_main.thread_id == _mi_thread_id()); } - // Initialize thread -void mi_thread_init(void) mi_attr_noexcept +static mi_theap_t* mi_thread_init_ex(mi_heap_t* heap_main) mi_attr_noexcept { // ensure our process has started already mi_process_init(); + // if the theap_default is already set we have already initialized - if (_mi_thread_is_initialized()) return; + mi_theap_t* theap = _mi_theap_default(); + if (mi_theap_is_initialized(theap)) return theap; - // initialize the default theap - if (_mi_thread_init_theap_default() == NULL) return; // out-of-memory on tld/theap allocation + // otherwise initialize the default theap + theap = _mi_thread_init_theap_default(heap_main); + if (theap == NULL) return NULL; // out-of-memory on tld/theap allocation - mi_heap_stat_increase(mi_heap_main(), threads, 1); + mi_subproc_stat_increase(_mi_theap_subproc(theap), threads, 1); // or theap stats and wait for merge? // _mi_verbose_message("thread init: 0x%zx\n", _mi_thread_id()); + return theap; +} + +mi_theap_t* _mi_thread_init(void) { + return mi_thread_init_ex(NULL); +} + +void mi_decl_noinline mi_thread_init(void) mi_attr_noexcept { + _mi_thread_init(); } void mi_thread_done(void) mi_attr_noexcept { @@ -731,11 +785,7 @@ void _mi_thread_done(mi_theap_t* _theap_main) { // NULL can be passed on some platforms if (_theap_main==NULL) { - _theap_main = __mi_theap_main; // don't call `mi_theap_main_safe` as that re-initializes the thread - if (_theap_main==NULL) { // can happen if `mi_theap_main_safe` is never called; but then the default is main - _theap_main = _mi_theap_default(); - mi_assert_internal(_theap_main==NULL || _mi_is_theap_main(_theap_main)); - } + _theap_main = _mi_theap_default(); } // prevent re-entrancy through theap_done/theap_set_default_direct (issue #699) @@ -743,14 +793,14 @@ void _mi_thread_done(mi_theap_t* _theap_main) return; } - // release dynamic thread_local's - _mi_thread_locals_thread_done(); - // note: we store the tld as we should avoid reading `thread_tld` at this point (to avoid reinitializing the thread local storage) mi_tld_t* const tld = _theap_main->tld; + // release dynamic thread_local's + _mi_thread_locals_thread_done(); + // adjust stats - mi_heap_stat_decrease(_mi_subproc_heap_main(tld->subproc), threads, 1); // todo: or `_theap_main->heap`? + mi_subproc_stat_decrease(tld->subproc, threads, 1); // todo: or `_theap_main->heap`? // check thread-id as on Windows shutdown with FLS the main (exit) thread may call this on thread-local theaps... if (tld->thread_id != _mi_prim_thread_id()) return; @@ -767,71 +817,85 @@ mi_decl_cold mi_decl_noinline mi_theap_t* _mi_theap_empty_get(void) { return (mi_theap_t*)&_mi_theap_empty; } -#if MI_TLS_MODEL_DYNAMIC_WIN32 +bool _mi_is_empty_theap(const mi_theap_t* theap) { + return (theap == &_mi_theap_empty); +} + + +#if MI_TLS_MODEL_WIN32 // If we can, we use one of the 64 direct TLS slots (but fall back to expansion slots if needed) // See for the offsets. #if MI_SIZE_SIZE==4 -#define MI_TLS_DIRECT_FIRST (0x0E10 / MI_SIZE_SIZE) +#define MI_TLS_DIRECT_FIRST (0x0E10 / MI_INTPTR_SIZE) #else -#define MI_TLS_DIRECT_FIRST (0x1480 / MI_SIZE_SIZE) +#define MI_TLS_DIRECT_FIRST (0x1480 / MI_INTPTR_SIZE) #endif #define MI_TLS_DIRECT_SLOTS (64) #define MI_TLS_EXPANSION_SLOTS (1024) #if !MI_WIN_DIRECT_TLS +// We initially use the last of the expansion slots as the default NULL. +// note: this will fail if the program allocates exactly 1024+64 slots with TlsAlloc +// before we are initialized :-( (but this seems quite unlikely). +// (todo: another approach could be to use slot 7 (EnvironmentPointer) as the initial slot as that seems to be always NULL) #define MI_TLS_INITIAL_SLOT MI_TLS_EXPANSION_SLOT #define MI_TLS_INITIAL_EXPANSION_SLOT (MI_TLS_EXPANSION_SLOTS-1) #else -// with only direct entries, use the "arbitrary user data" field -// and assume it is NULL (see also ) -#define MI_TLS_INITIAL_SLOT (5) -#define MI_TLS_INITIAL_EXPANSION_SLOT (0) +// With direct tls we need an initial NULL slot outside the expansion slots +#define MI_TLS_INITIAL_SLOT (5) // Arbitrary user pointer +#define MI_TLS_INITIAL_EXPANSION_SLOT (MI_TLS_EXPANSION_SLOTS-1) // unused #endif -// we initially use the last of the expansion slots as the default NULL. -// note: this will fail if the program allocates exactly 1024+64 slots with TlsAlloc (which is quite unlikely) -mi_decl_hidden mi_decl_cache_align size_t _mi_theap_default_slot = MI_TLS_INITIAL_SLOT; -mi_decl_hidden size_t _mi_theap_default_expansion_slot = MI_TLS_INITIAL_EXPANSION_SLOT; -mi_decl_hidden size_t _mi_theap_cached_slot = MI_TLS_INITIAL_SLOT; -mi_decl_hidden size_t _mi_theap_cached_expansion_slot = MI_TLS_INITIAL_EXPANSION_SLOT; +// in case of errors assign fixed slots (but since we use EFAULT the program should fail anyways) +#define MI_TLS_ERROR_SLOT (5) // arbitrary user pointer +#define MI_TLS_ERROR_EXPANSION_SLOT (7) // environment pointer (only used for OS/2 emulation) + + +mi_decl_hidden mi_decl_cache_align _Atomic(size_t) _mi_theap_default_slot = MI_ATOMIC_VAR_INIT(MI_TLS_INITIAL_SLOT); +mi_decl_hidden _Atomic(size_t) _mi_theap_default_expansion_slot = MI_ATOMIC_VAR_INIT(MI_TLS_INITIAL_EXPANSION_SLOT); +mi_decl_hidden _Atomic(size_t) _mi_theap_cached_slot = MI_ATOMIC_VAR_INIT(MI_TLS_INITIAL_SLOT); +mi_decl_hidden _Atomic(size_t) _mi_theap_cached_expansion_slot = MI_ATOMIC_VAR_INIT(MI_TLS_INITIAL_EXPANSION_SLOT); static DWORD mi_tls_raw_index_default = TLS_OUT_OF_INDEXES; static DWORD mi_tls_raw_index_cached = TLS_OUT_OF_INDEXES; -static bool mi_win_tls_slot_alloc(size_t* slot, size_t* extended, DWORD* raw_index) { +static bool mi_win_tls_slot_alloc(_Atomic(size_t)* slot, _Atomic(size_t)* extended, DWORD* raw_index) { + // always write slot before extended due to concurrent readers const DWORD index = TlsAlloc(); *raw_index = index; if (index==TLS_OUT_OF_INDEXES) { - *extended = 0; - *slot = 0; + mi_atomic_store_release(slot,MI_TLS_ERROR_SLOT); + mi_atomic_store_release(extended,MI_TLS_ERROR_EXPANSION_SLOT); return false; } else if (indextld != NULL); mi_assert_internal(theap->tld->thread_id==0 || theap->tld->thread_id==_mi_thread_id()); mi_tls_slots_init(); - #if MI_TLS_MODEL_THREAD_LOCAL + #if MI_TLS_MODEL_LOCAL __mi_theap_default = theap; - #elif MI_TLS_MODEL_FIXED_SLOT - mi_prim_tls_slot_set(MI_TLS_MODEL_FIXED_SLOT_DEFAULT, theap); - #elif MI_TLS_MODEL_DYNAMIC_WIN32 + #elif MI_TLS_MODEL_FIXED + mi_prim_tls_slot_set(MI_TLS_MODEL_FIXED_DEFAULT, theap); + #elif MI_TLS_MODEL_WIN32 mi_win_tls_slot_set(_mi_theap_default_slot, _mi_theap_default_expansion_slot, theap); - #elif MI_TLS_MODEL_DYNAMIC_PTHREADS - if (_mi_theap_default_key!=0) pthread_setspecific(_mi_theap_default_key, theap); + #elif MI_TLS_MODEL_PTHREADS + mi_pthread_key_set(&_mi_theap_default_key, theap); #endif // set theap main if needed if (mi_theap_is_initialized(theap)) { // ensure the default theap is passed to `_mi_thread_done` as on some platforms we cannot access TLS at thread termination (as it would allocate again) _mi_prim_thread_associate_default_theap(theap); - if (_mi_is_heap_main(_mi_theap_heap(theap))) { - __mi_theap_main = theap; - } - } - - // ensure either the default slot contains the main theap, or __mi_theap_main is initialized - if (mi_theap_is_initialized(theap_old) && _mi_is_heap_main(_mi_theap_heap(theap_old))) { - __mi_theap_main = theap_old; } } void mi_thread_set_in_threadpool(void) mi_attr_noexcept { - mi_theap_t* theap = _mi_theap_default_safe(); + mi_theap_t* theap = mi_theap_get_default(); theap->tld->is_in_threadpool = true; } @@ -1005,8 +1059,10 @@ void _mi_auto_process_init(void) { mi_assert_internal(_mi_is_main_thread()); mi_process_init(); + mi_tls_slots_init(); mi_process_setup_auto_thread_done(); _mi_thread_locals_init(); + _mi_options_post_init(); // now we can print to stderr if (_mi_is_redirected()) _mi_verbose_message("malloc is redirected.\n"); @@ -1018,7 +1074,7 @@ void _mi_auto_process_init(void) { } // reseed random - _mi_random_reinit_if_weak(&theap_main.random); + _mi_random_reinit_if_weak(&mi_theap_main.random); } // CPU features @@ -1095,7 +1151,6 @@ static void mi_detect_cpu_features(void) { // Initialize the process; called by thread_init or the process loader static void mi_process_init_once(void) mi_attr_noexcept { - _mi_process_is_initialized = true; _mi_verbose_message("process init: 0x%zx\n", _mi_thread_id()); mi_detect_cpu_features(); @@ -1109,7 +1164,7 @@ static void mi_process_init_once(void) mi_attr_noexcept { mi_thread_init(); _mi_process_is_initialized = true; - #if defined(_WIN32) && defined(MI_WIN_USE_FLS) + #if defined(_WIN32) && defined(MI_WIN_INIT_USE_FLS) // On windows, when building as a static lib the FLS cleanup happens to early for the main thread. // To avoid this, set the FLS value for the main thread to NULL so the fls cleanup // will not call _mi_thread_done on the (still executing) main thread. See issue #508. @@ -1145,18 +1200,16 @@ void mi_process_init(void) mi_attr_noexcept { } } -// Called when the process is done (cdecl as it is used with `at_exit` on some platforms) -void mi_cdecl mi_process_done(void) mi_attr_noexcept { + +// Called when the process is done +static void mi_process_done_once(void) { // only shutdown if we were initialized if (!_mi_process_is_initialized) return; // ensure we are called once static bool process_done = false; if (process_done) return; process_done = true; - - // free dynamic thread locals (if used at all) - _mi_thread_locals_done(); - + // release any thread specific resources and ensure _mi_thread_done is called on all but the main thread _mi_prim_thread_done_auto_done(); @@ -1170,29 +1223,41 @@ void mi_cdecl mi_process_done(void) mi_attr_noexcept { #endif // done with tracking tools - mi_track_done() + mi_track_done(); // Forcefully release all retained memory; this can be dangerous in general if overriding regular malloc/free // since after process_done there might still be other code running that calls `free` (like at_exit routines, // or C-runtime termination code. if (mi_option_is_enabled(mi_option_destroy_on_exit)) { - mi_subprocs_unsafe_destroy_all(); // destroys all subprocs, arenas, and the page_map! + mi_subprocs_unsafe_destroy_all(); // destroys all subprocs, arenas, thread locals, and the page_map! } else { - mi_heap_stats_merge_to_subproc(mi_heap_main()); + // free dynamic thread locals (if used at all) + _mi_thread_locals_done(); + if (subproc_main.heap_main != NULL) { + mi_heap_stats_merge_to_subproc(subproc_main.heap_main); + } } - - // careful now to no longer access any allocator functionality + + // careful now to no longer access any allocator functionality if (mi_option_is_enabled(mi_option_show_stats) || mi_option_is_enabled(mi_option_verbose)) { mi_subproc_stats_print_out(mi_subproc_main(), NULL, NULL); } mi_lock_done(&subprocs_lock); mi_tls_slots_done(); _mi_allocator_done(); - _mi_verbose_message("process done: 0x%zx\n", tld_main.thread_id); + _mi_verbose_message("process done: 0x%zx\n", mi_process_tld_main.thread_id); os_preloading = true; // don't call the C runtime anymore } + +// Called when the process is done (cdecl as it is used with `at_exit` on some platforms) +void mi_cdecl mi_process_done(void) mi_attr_noexcept { + mi_atomic_do_once { + mi_process_done_once(); + } +} + void mi_cdecl _mi_auto_process_done(void) mi_attr_noexcept { if (_mi_option_get_fast(mi_option_destroy_on_exit)>1) return; mi_process_done(); diff --git a/vendored/mimalloc/src/libc.c b/vendored/mimalloc/src/libc.c index c74003a149..b6eaaba409 100644 --- a/vendored/mimalloc/src/libc.c +++ b/vendored/mimalloc/src/libc.c @@ -39,8 +39,8 @@ bool _mi_streq(const char* s, const char* t) { return (*s == *t); } -void _mi_strlcpy(char* dest, const char* src, size_t dest_size) { - if (dest==NULL || src==NULL || dest_size == 0) return; +bool _mi_strlcpy(char* dest, const char* src, size_t dest_size) { + if (dest==NULL || src==NULL || dest_size == 0) return (src==NULL || *src==0); // copy until end of src, or when dest is (almost) full while (*src != 0 && dest_size > 1) { *dest++ = *src++; @@ -48,23 +48,24 @@ void _mi_strlcpy(char* dest, const char* src, size_t dest_size) { } // always zero terminate *dest = 0; + return (*src == 0); } -void _mi_strlcat(char* dest, const char* src, size_t dest_size) { - if (dest==NULL || src==NULL || dest_size == 0) return; +bool _mi_strlcat(char* dest, const char* src, size_t dest_size) { + if (dest==NULL || src==NULL || dest_size == 0) return (src==NULL || *src==0); // find end of string in the dest buffer while (*dest != 0 && dest_size > 1) { dest++; dest_size--; } // and catenate - _mi_strlcpy(dest, src, dest_size); + return _mi_strlcpy(dest, src, dest_size); } size_t _mi_strnlen(const char* s, size_t max_len) { if (s==NULL) return 0; size_t len = 0; - while(s[len] != 0 && len < max_len) { len++; } + while(len < max_len && s[len] != 0) { len++; } return len; } @@ -76,7 +77,7 @@ char* _mi_strnstr(char* s, size_t max_len, const char* pat) { if (s==NULL) return NULL; if (pat==NULL) return s; const size_t m = _mi_strnlen(s, max_len); - const size_t n = _mi_strlen(pat); + const size_t n = _mi_strlen(pat); for (size_t start = 0; start + n <= m; start++) { size_t i = 0; while (i 0 ? 0 : (res == 0 ? ENOENT : EAGAIN)); } #endif @@ -117,13 +120,13 @@ bool _mi_atomic_once_enter(mi_atomic_once_t* once) { const mi_threadid_t current_tid = _mi_thread_id(); if (once_tid == current_tid) { return false; // recursive invocation; we need this for process_init for example - } + } mi_lock_acquire(&once->lock); uintptr_t expected = 0; if (mi_atomic_cas_strong_acq_rel(&once->tid, &expected, current_tid)) { // could use atomic_load/store as well return true; // should execute and release - } + } else { mi_lock_release(&once->lock); return false; // already another thread entered and released @@ -134,8 +137,25 @@ void _mi_atomic_once_release(mi_atomic_once_t* once) { if (mi_atomic_load_acquire(&once->tid)>1) { // paranoia mi_atomic_store_release(&once->tid,1); // done executing mi_lock_release(&once->lock); - } + } +} + +#if MI_USE_PTHREADS +mi_decl_noinline bool _mi_pthread_key_create(pthread_key_t* pkey, void (*destruct)(void*), void* init) { + int err = pthread_key_create(pkey,destruct); + if mi_unlikely(err!=0) { + *pkey = MI_PTHREAD_KEY_INVALID; + _mi_error_message(ENOMEM,"unable to allocate a thread local variable (error %d)\n", err); + return false; + } + mi_assert_internal(*pkey != MI_PTHREAD_KEY_INVALID); + if (init!=NULL) { + pthread_setspecific(*pkey,init); + }; + mi_assert_internal(pthread_getspecific(*pkey)==init); + return true; } +#endif // -------------------------------------------------------- @@ -196,22 +216,20 @@ static void mi_out_num(uintmax_t x, size_t base, char prefix, char** out, char* mi_outc('0',out,end); } else { - // output digits in reverse - char* start = *out; - while (x > 0) { + #define MI_MAX_OUT_DIGITS (160) /* a 512 bit number has 155 digits */ + char num[MI_MAX_OUT_DIGITS]; + int dcount = 0; + while(x>0 && dcount < MI_MAX_OUT_DIGITS) { char digit = (char)(x % base); - mi_outc((digit <= 9 ? '0' + digit : 'A' + digit - 10),out,end); + num[dcount++] = (digit <= 9 ? '0' + digit : 'A' + digit - 10); x = x / base; } + if (dcount>=MI_MAX_OUT_DIGITS) return; // don't output anything? if (prefix != 0) { mi_outc(prefix, out, end); } - size_t len = *out - start; - // and reverse in-place - for (size_t i = 0; i < (len / 2); i++) { - char c = start[len - i - 1]; - start[len - i - 1] = start[i]; - start[i] = c; + while(dcount-- > 0) { + mi_outc(num[dcount], out, end); } } } @@ -258,7 +276,10 @@ int _mi_vsnprintf(char* buf, size_t bufsize, const char* fmt, va_list args) { if (c >= '1' && c <= '9') { width = (c - '0'); MI_NEXTC(); while (c >= '0' && c <= '9') { - width = (10 * width) + (c - '0'); MI_NEXTC(); + if (width < SIZE_MAX/1024) { // no overflow + width = (10 * width) + (c - '0'); + } + MI_NEXTC(); } if (c == 0) break; // extra check due to while } @@ -297,7 +318,7 @@ int _mi_vsnprintf(char* buf, size_t bufsize, const char* fmt, va_list args) { if (width == 0 && (c == 'x' || c == 'p')) { if (c == 'p') { width = 2 * (x <= UINT32_MAX ? 4 : ((x >> 16) <= UINT32_MAX ? 6 : sizeof(void*))); } if (width == 0) { width = 2; } - fill = '0'; + if (alignright) { fill = '0'; } } mi_out_num(x, (c == 'x' || c == 'p' ? 16 : 10), numplus, &out, end); } @@ -350,6 +371,7 @@ int _mi_snprintf(char* buf, size_t buflen, const char* fmt, ...) { return written; } +#undef MI_NEXTC // -------------------------------------------------------- @@ -365,7 +387,7 @@ static size_t mi_ctz_generic32(uint32_t x) { 31, 27, 13, 23, 21, 19, 16, 7, 26, 12, 18, 6, 11, 5, 10, 9 }; if (x==0) return 32; - return debruijn[(uint32_t)((x & -(int32_t)x) * (uint32_t)(0x077CB531U)) >> 27]; + return debruijn[(uint32_t)((x & (~x + 1U)) * (uint32_t)(0x077CB531U)) >> 27]; } static size_t mi_clz_generic32(uint32_t x) { diff --git a/vendored/mimalloc/src/options.c b/vendored/mimalloc/src/options.c index c2d8fa9148..3b54cba031 100644 --- a/vendored/mimalloc/src/options.c +++ b/vendored/mimalloc/src/options.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -17,7 +17,7 @@ static long mi_max_warning_count = 16; // stop outputting warnings after this (u static void mi_add_stderr_output(void); -int mi_version(void) mi_attr_noexcept { +int mi_version(void) { return MI_MALLOC_VERSION; } @@ -121,7 +121,7 @@ static mi_option_desc_t mi_options[_mi_option_last] = { MI_DEFAULT_VERBOSE, MI_OPTION_UNINIT, MI_OPTION(verbose) }, // some of the following options are experimental and not all combinations are allowed. - { 1, MI_OPTION_UNINIT, MI_OPTION(deprecated_eager_commit) }, + { 1, MI_OPTION_UNINIT, MI_OPTION(deprecated_eager_commit) }, { MI_DEFAULT_ARENA_EAGER_COMMIT, MI_OPTION_UNINIT, MI_OPTION_LEGACY(arena_eager_commit,eager_region_commit) }, // eager commit arena's? 2 is used to enable this only on an OS that has overcommit (i.e. linux) { 1, MI_OPTION_UNINIT, MI_OPTION_LEGACY(purge_decommits,reset_decommits) }, // purge decommits memory (instead of reset) (note: on linux this uses MADV_DONTNEED for decommit) @@ -134,9 +134,9 @@ static mi_option_desc_t mi_options[_mi_option_last] = MI_OPTION_UNINIT, MI_OPTION(reserve_os_memory) }, // reserve N KiB OS memory in advance (use `option_get_size`) { 0, MI_OPTION_UNINIT, MI_OPTION(deprecated_segment_cache) }, // cache N segments per thread { 0, MI_OPTION_UNINIT, MI_OPTION(deprecated_page_reset) }, // reset page memory on free - { 0, MI_OPTION_UNINIT, MI_OPTION(deprecated_abandoned_page_purge) }, + { 0, MI_OPTION_UNINIT, MI_OPTION(deprecated_abandoned_page_purge) }, { 0, MI_OPTION_UNINIT, MI_OPTION(deprecated_segment_reset) }, // reset segment memory on free (needs eager commit) - { 1, MI_OPTION_UNINIT, MI_OPTION(deprecated_eager_commit_delay) }, + { 1, MI_OPTION_UNINIT, MI_OPTION(deprecated_eager_commit_delay) }, { 1000,MI_OPTION_UNINIT, MI_OPTION_LEGACY(purge_delay,reset_delay) }, // purge delay in milli-seconds { 0, MI_OPTION_UNINIT, MI_OPTION(use_numa_nodes) }, // 0 = use available numa nodes, otherwise use at most N nodes. { 0, MI_OPTION_UNINIT, MI_OPTION_LEGACY(disallow_os_alloc,limit_os_alloc) }, // 1 = do not use OS memory for allocation (but only reserved arenas) @@ -150,11 +150,7 @@ static mi_option_desc_t mi_options[_mi_option_last] = { 1, MI_OPTION_UNINIT, MI_OPTION_LEGACY(deprecated_purge_extend_delay, decommit_extend_delay) }, { MI_DEFAULT_DISALLOW_ARENA_ALLOC, MI_OPTION_UNINIT, MI_OPTION(disallow_arena_alloc) }, // 1 = do not use arena's for allocation (except if using specific arena id's) { 400, MI_OPTION_UNINIT, MI_OPTION(retry_on_oom) }, // windows only: retry on out-of-memory for N milli seconds (=400), set to 0 to disable retries. -#if defined(MI_VISIT_ABANDONED) - { 1, MI_OPTION_INITIALIZED, MI_OPTION(visit_abandoned) }, // allow visiting theap blocks in abandoned segments; requires taking locks during reclaim. -#else - { 0, MI_OPTION_UNINIT, MI_OPTION(visit_abandoned) }, -#endif + { 1, MI_OPTION_UNINIT, MI_OPTION(deprecated_visit_abandoned) }, { 0, MI_OPTION_UNINIT, MI_OPTION(guarded_min) }, // only used when building with MI_GUARDED: minimal rounded object size for guarded objects { MI_GiB, MI_OPTION_UNINIT, MI_OPTION(guarded_max) }, // only used when building with MI_GUARDED: maximal rounded object size for guarded objects { 0, MI_OPTION_UNINIT, MI_OPTION(guarded_precise) }, // disregard minimal alignment requirement to always place guarded blocks exactly in front of a guard page (=0) @@ -176,15 +172,15 @@ static mi_option_desc_t mi_options[_mi_option_last] = { MI_DEFAULT_ALLOW_THP, MI_OPTION_UNINIT, MI_OPTION(allow_thp) }, // allow transparent huge pages? (=1) (on Android =0 by default). Set to 0 to disable THP for the process. { 0, MI_OPTION_UNINIT, MI_OPTION(minimal_purge_size) }, // set minimal purge size (in KiB) (=0). Using 0 resolves to either 64 (or 2048 if `mi_option_allow_thp==2`). - { MI_DEFAULT_ARENA_MAX_OBJECT_SIZE, - MI_OPTION_UNINIT, MI_OPTION(arena_max_object_size) }, // set maximal object size that can be allocated in an arena (in KiB) (=2GiB on 64-bit). + { MI_DEFAULT_ARENA_MAX_OBJECT_SIZE, + MI_OPTION_UNINIT, MI_OPTION(arena_max_object_size) }, // set maximal object size that can be allocated in an arena (in KiB) (=2GiB on 64-bit). { 0, MI_OPTION_UNINIT, MI_OPTION(arena_is_numa_local) }, // associate local numa node with an initial arena allocation }; static void mi_option_init(mi_option_desc_t* desc); static bool mi_option_has_size_in_kib(mi_option_t option) { - return (option == mi_option_reserve_os_memory || option == mi_option_arena_reserve || + return (option == mi_option_reserve_os_memory || option == mi_option_arena_reserve || option == mi_option_minimal_purge_size || option == mi_option_arena_max_object_size); } @@ -203,7 +199,7 @@ void _mi_options_init(void) { _mi_warning_message("option 'allow_large_os_pages' is disabled to allow for guarded objects\n"); } } - #endif + #endif } // called at actual process load, it should be safe to print now @@ -482,7 +478,7 @@ static void mi_recurse_exit(void) { } void _mi_fputs(mi_output_fun* out, void* arg, const char* prefix, const char* message) { - if (out==NULL || (void*)out==(void*)stdout || (void*)out==(void*)stderr) { // TODO: use mi_out_stderr for stderr? + if (out==NULL || (void*)out==(void*)stdout || (void*)out==(void*)stderr) { // todo: use mi_out_stderr for stderr? if (!mi_recurse_enter()) return; out = mi_out_get_default(&arg); if (prefix != NULL) out(prefix, arg); @@ -531,13 +527,6 @@ void _mi_raw_message(const char* fmt, ...) { va_end(args); } -void _mi_message(const char* fmt, ...) { - va_list args; - va_start(args, fmt); - mi_vfprintf_thread(NULL, NULL, "mimalloc: ", fmt, args); - va_end(args); -} - void _mi_trace_message(const char* fmt, ...) { if (mi_option_get(mi_option_verbose) <= 1) return; // only with verbose level 2 or higher va_list args; @@ -590,24 +579,27 @@ static _Atomic(void*) mi_error_arg; // = NULL static void mi_error_default(int err) { MI_UNUSED(err); -#if (MI_DEBUG>0) - if (err==EFAULT) { - #ifdef _MSC_VER - __debugbreak(); - #endif - abort(); - } -#endif -#if (MI_SECURE>0) - if (err==EFAULT) { // abort on serious errors in secure mode (corrupted meta-data) - abort(); - } -#endif -#if defined(MI_XMALLOC) - if (err==ENOMEM || err==EOVERFLOW) { // abort on memory allocation fails in xmalloc mode - abort(); + #if (MI_DEBUG>0) + if (err==EFAULT) { + #ifdef _MSC_VER + __debugbreak(); + #endif + abort(); + } + #endif + #if (MI_SECURE>0) + if (err==EFAULT) { // abort on serious errors in secure mode (corrupted meta-data) + abort(); + } + #endif + #if defined(MI_XMALLOC) + if (err==ENOMEM || err==EOVERFLOW || err=EINVAL) { // abort on memory allocation fails in xmalloc mode + abort(); + } + #endif + if (errno==0) { + errno = (err==EINVAL ? EINVAL : ENOMEM /* compatibility */ ); } -#endif } void mi_register_error(mi_error_fun* fun, void* arg) { @@ -621,7 +613,7 @@ void _mi_error_message(int err, const char* fmt, ...) { va_start(args, fmt); mi_show_error_message(fmt, args); va_end(args); - // and call the error handler which may abort (or return normally) + // and call the error handler which may abort (or return normally, potentially setting errno) if (mi_error_handler != NULL) { mi_error_handler(err, mi_atomic_load_ptr_acquire(void,&mi_error_arg)); } @@ -636,8 +628,6 @@ void _mi_error_message(int err, const char* fmt, ...) { // TODO: implement ourselves to reduce dependencies on the C runtime #include // strtol -#include // strstr - static void mi_option_init(mi_option_desc_t* desc) { // Read option value from the environment @@ -645,17 +635,17 @@ static void mi_option_init(mi_option_desc_t* desc) { char buf[64+1]; _mi_strlcpy(buf, "mimalloc_", sizeof(buf)); _mi_strlcat(buf, desc->name, sizeof(buf)); - bool found = _mi_getenv(buf, s, sizeof(s)); - if (!found && desc->legacy_name != NULL) { + int err = _mi_getenv(buf, s, sizeof(s)); + if (err==ENOENT && desc->legacy_name != NULL) { _mi_strlcpy(buf, "mimalloc_", sizeof(buf)); _mi_strlcat(buf, desc->legacy_name, sizeof(buf)); - found = _mi_getenv(buf, s, sizeof(s)); - if (found) { + err = _mi_getenv(buf, s, sizeof(s)); + if (err==0) { _mi_warning_message("environment option \"mimalloc_%s\" is deprecated -- use \"mimalloc_%s\" instead.\n", desc->legacy_name, desc->name); } } - if (found) { + if (err==0) { size_t len = _mi_strnlen(s, sizeof(buf) - 1); for (size_t i = 0; i < len; i++) { buf[i] = _mi_toupper(s[i]); @@ -665,14 +655,15 @@ static void mi_option_init(mi_option_desc_t* desc) { desc->value = 1; desc->init = MI_OPTION_INITIALIZED; } - else if (_mi_streq(buf,"0") || _mi_streq(buf,"FALSE") || _mi_streq(buf,"NO") || _mi_streq(buf,"OFF")) { + else if (_mi_streq(buf,"0") || _mi_streq(buf,"FALSE") || _mi_streq(buf,"NO") || _mi_streq(buf,"OFF")) { desc->value = 0; desc->init = MI_OPTION_INITIALIZED; } else { char* end = buf; + errno = 0; long value = strtol(buf, &end, 10); - if (mi_option_has_size_in_kib(desc->option)) { + if (errno==0 && mi_option_has_size_in_kib(desc->option)) { // this option is interpreted in KiB to prevent overflow of `long` for large allocations // (long is 32-bit on 64-bit windows, which allows for 4TiB max.) size_t size = (value < 0 ? 0 : (size_t)value); @@ -687,7 +678,7 @@ static void mi_option_init(mi_option_desc_t* desc) { if (overflow || size > (MI_MAX_ALLOC_SIZE / MI_KiB)) { size = (MI_MAX_ALLOC_SIZE / MI_KiB); } value = (size > LONG_MAX ? LONG_MAX : (long)size); } - if (*end == 0) { + if (errno==0 && *end == 0) { mi_option_set(desc->option, value); } else { @@ -707,7 +698,8 @@ static void mi_option_init(mi_option_desc_t* desc) { } mi_assert_internal(desc->init != MI_OPTION_UNINIT); } - else if (!_mi_preloading()) { + else if (err==ENOENT) { desc->init = MI_OPTION_DEFAULTED; } + // and on another error, keep unitialized to try again (can happen during preloading if getenv is not available) } diff --git a/vendored/mimalloc/src/os.c b/vendored/mimalloc/src/os.c index febecb0c9c..f01379f5e7 100644 --- a/vendored/mimalloc/src/os.c +++ b/vendored/mimalloc/src/os.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -8,6 +8,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc/internal.h" #include "mimalloc/atomic.h" #include "mimalloc/prim.h" +#include "mimalloc/prim-tls.h" // _mi_theap_default for random /* ----------------------------------------------------------- Initialization. @@ -103,8 +104,8 @@ void _mi_os_init(void) { /* ----------------------------------------------------------- Util -------------------------------------------------------------- */ -bool _mi_os_decommit(void* addr, size_t size); -bool _mi_os_commit(void* addr, size_t size, bool* is_zero); +bool _mi_os_decommit(mi_subproc_t* subproc, void* addr, size_t size); +bool _mi_os_commit(mi_subproc_t* subproc, void* addr, size_t size, bool* is_zero); // On systems with enough virtual address bits, we can do efficient aligned allocation by using // the 2TiB to 30TiB area to allocate those. If we have at least 46 bits of virtual address @@ -114,7 +115,7 @@ bool _mi_os_commit(void* addr, size_t size, bool* is_zero); // Return a MI_HINT_ALIGN (4MiB) aligned address that is probably available. // If this returns NULL, the OS will determine the address but on some OS's that may not be // properly aligned which can be more costly as it needs to be adjusted afterwards. -// For a size > 16GiB this always returns NULL in order to guarantee good ASLR randomization; +// In secure mode, for a size > 16GiB this always returns NULL in order to guarantee good ASLR randomization; // (otherwise an initial large allocation of say 2TiB has a 50% chance to include (known) addresses // in the middle of the 2TiB - 6TiB address range (see issue #372)) @@ -131,7 +132,9 @@ void* _mi_os_get_aligned_hint(size_t try_alignment, size_t size) if (try_alignment <= mi_os_mem_config.alloc_granularity || try_alignment > MI_HINT_ALIGN) return NULL; if (mi_os_mem_config.virtual_address_bits < 46) return NULL; // < 64TiB virtual address space size = _mi_align_up(size, MI_HINT_ALIGN); - if (size > 16*MI_GiB) return NULL; // guarantee the chance of fixed valid address is at least 1/(MI_HINT_AREA / 1<<34) + #if (MI_SECURE>=1) + if (size > 16*MI_GiB) return NULL; // guarantee the chance of fixed valid address is at most 1/(MI_HINT_AREA / 1<<34) = 1/256 + #endif size += MI_HINT_ALIGN; // put in virtual gaps between hinted blocks; this splits VLA's but increases guarded areas. uintptr_t hint = mi_atomic_add_acq_rel(&aligned_base, size); @@ -173,7 +176,7 @@ size_t _mi_os_secure_guard_page_size(void) { } // In secure mode, try to decommit an area and output a warning if this fails. -bool _mi_os_secure_guard_page_set_at(void* addr, mi_memid_t memid) { +bool _mi_os_secure_guard_page_set_at(mi_subproc_t* subproc, void* addr, mi_memid_t memid) { if (addr == NULL) return true; #if MI_SECURE > 0 bool ok = false; @@ -183,7 +186,7 @@ bool _mi_os_secure_guard_page_set_at(void* addr, mi_memid_t memid) { ok = (*(arena->commit_fun))(false /* decommit */, addr, _mi_os_secure_guard_page_size(), NULL, arena->commit_fun_arg); } else { - ok = _mi_os_decommit(addr, _mi_os_secure_guard_page_size()); + ok = _mi_os_decommit(subproc, addr, _mi_os_secure_guard_page_size()); } } if (!ok) { @@ -191,18 +194,18 @@ bool _mi_os_secure_guard_page_set_at(void* addr, mi_memid_t memid) { } return ok; #else - MI_UNUSED(memid); + MI_UNUSED(subproc); MI_UNUSED(memid); return true; #endif } // In secure mode, try to decommit an area and output a warning if this fails. -bool _mi_os_secure_guard_page_set_before(void* addr, mi_memid_t memid) { - return _mi_os_secure_guard_page_set_at((uint8_t*)addr - _mi_os_secure_guard_page_size(), memid); +bool _mi_os_secure_guard_page_set_before(mi_subproc_t* subproc, void* addr, mi_memid_t memid) { + return _mi_os_secure_guard_page_set_at(subproc, (uint8_t*)addr - _mi_os_secure_guard_page_size(), memid); } // In secure mode, try to recommit an area -bool _mi_os_secure_guard_page_reset_at(void* addr, mi_memid_t memid) { +bool _mi_os_secure_guard_page_reset_at(mi_subproc_t* subproc, void* addr, mi_memid_t memid) { if (addr == NULL) return true; #if MI_SECURE > 0 if (!memid.is_pinned) { @@ -211,18 +214,18 @@ bool _mi_os_secure_guard_page_reset_at(void* addr, mi_memid_t memid) { return (*(arena->commit_fun))(true, addr, _mi_os_secure_guard_page_size(), NULL, arena->commit_fun_arg); } else { - return _mi_os_commit(addr, _mi_os_secure_guard_page_size(), NULL); + return _mi_os_commit(subproc, addr, _mi_os_secure_guard_page_size(), NULL); } } #else - MI_UNUSED(memid); + MI_UNUSED(subproc); MI_UNUSED(memid); #endif return true; } // In secure mode, try to recommit an area -bool _mi_os_secure_guard_page_reset_before(void* addr, mi_memid_t memid) { - return _mi_os_secure_guard_page_reset_at((uint8_t*)addr - _mi_os_secure_guard_page_size(), memid); +bool _mi_os_secure_guard_page_reset_before(mi_subproc_t* subproc, void* addr, mi_memid_t memid) { + return _mi_os_secure_guard_page_reset_at(subproc, (uint8_t*)addr - _mi_os_secure_guard_page_size(), memid); } @@ -230,23 +233,23 @@ bool _mi_os_secure_guard_page_reset_before(void* addr, mi_memid_t memid) { Free memory -------------------------------------------------------------- */ -static void mi_os_free_huge_os_pages(void* p, size_t size, mi_subproc_t* subproc); +static void mi_os_free_huge_os_pages(mi_subproc_t* subproc, void* p, size_t size); -static void mi_os_prim_free(void* addr, size_t size, size_t commit_size, mi_subproc_t* subproc) { +static void mi_os_prim_free(mi_subproc_t* subproc, void* addr, size_t size, size_t commit_size) { + mi_assert_internal(subproc!=NULL); mi_assert_internal((size % _mi_os_page_size()) == 0); if (addr == NULL) return; // || _mi_os_is_huge_reserved(addr) int err = _mi_prim_free(addr, size); // allow size==0 (issue #1041) if (err != 0) { _mi_warning_message("unable to free OS memory (error: %d (0x%x), size: 0x%zx bytes, address: %p)\n", err, err, size, addr); } - if (subproc == NULL) { subproc = _mi_subproc(); } // from `mi_arenas_unsafe_destroy` we pass subproc_main explicitly as we can no longer use the theap pointer if (commit_size > 0) { mi_subproc_stat_decrease(subproc, committed, commit_size); } mi_subproc_stat_decrease(subproc, reserved, size); } -void _mi_os_free_ex(void* addr, size_t size, bool still_committed, mi_memid_t memid, mi_subproc_t* subproc /* can be NULL */) { +void _mi_os_free_ex(mi_subproc_t* subproc, void* addr, size_t size, bool still_committed, mi_memid_t memid) { if (mi_memkind_is_os(memid.memkind)) { size_t csize = memid.mem.os.size; if (csize==0) { csize = _mi_os_good_alloc_size(size); } @@ -268,10 +271,10 @@ void _mi_os_free_ex(void* addr, size_t size, bool still_committed, mi_memid_t me // free it if (memid.memkind == MI_MEM_OS_HUGE) { mi_assert(memid.is_pinned); - mi_os_free_huge_os_pages(base, csize, subproc); + mi_os_free_huge_os_pages(subproc, base, csize); } else { - mi_os_prim_free(base, csize, (still_committed ? commit_size : 0), subproc); + mi_os_prim_free(subproc, base, csize, (still_committed ? commit_size : 0)); } } else { @@ -280,8 +283,8 @@ void _mi_os_free_ex(void* addr, size_t size, bool still_committed, mi_memid_t me } } -void _mi_os_free(void* p, size_t size, mi_memid_t memid) { - _mi_os_free_ex(p, size, true, memid, NULL); +void _mi_os_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid) { + _mi_os_free_ex(subproc, p, size, true, memid); } @@ -291,7 +294,7 @@ void _mi_os_free(void* p, size_t size, mi_memid_t memid) { // Note: the `try_alignment` is just a hint and the returned pointer is not guaranteed to be aligned. // Also `hint_addr` is a hint and may be ignored. -static void* mi_os_prim_alloc_at(void* hint_addr, size_t size, size_t try_alignment, bool commit, bool allow_large, bool* is_large, bool* is_zero) { +static void* mi_os_prim_alloc_at(mi_subproc_t* subproc, void* hint_addr, size_t size, size_t try_alignment, bool commit, bool allow_large, bool* is_large, bool* is_zero) { mi_assert_internal(size > 0 && (size % _mi_os_page_size()) == 0); mi_assert_internal(is_zero != NULL); mi_assert_internal(is_large != NULL); @@ -312,11 +315,11 @@ static void* mi_os_prim_alloc_at(void* hint_addr, size_t size, size_t try_alignm _mi_warning_message("unable to allocate OS memory (error: %d (0x%x), addr: %p, size: 0x%zx bytes, align: 0x%zx, commit: %d, allow large: %d)\n", err, err, hint_addr, size, try_alignment, commit, allow_large); } - mi_os_stat_counter_increase(mmap_calls, 1); + mi_subproc_stat_counter_increase(subproc, mmap_calls, 1); if (p != NULL) { - mi_os_stat_increase(reserved, size); + mi_subproc_stat_increase(subproc, reserved, size); if (commit) { - mi_os_stat_increase(committed, size); + mi_subproc_stat_increase(subproc, committed, size); // seems needed for asan (or `mimalloc-test-api` fails) #ifdef MI_TRACK_ASAN if (*is_zero) { mi_track_mem_defined(p,size); } @@ -327,14 +330,14 @@ static void* mi_os_prim_alloc_at(void* hint_addr, size_t size, size_t try_alignm return p; } -static void* mi_os_prim_alloc(size_t size, size_t try_alignment, bool commit, bool allow_large, bool* is_large, bool* is_zero) { - return mi_os_prim_alloc_at(NULL, size, try_alignment, commit, allow_large, is_large, is_zero); +static void* mi_os_prim_alloc(mi_subproc_t* subproc, size_t size, size_t try_alignment, bool commit, bool allow_large, bool* is_large, bool* is_zero) { + return mi_os_prim_alloc_at(subproc, NULL /* hint addr */, size, try_alignment, commit, allow_large, is_large, is_zero); } // Primitive aligned allocation from the OS. // This function guarantees the allocated memory is aligned. -static void* mi_os_prim_alloc_aligned(size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid) { +static void* mi_os_prim_alloc_aligned(mi_subproc_t* subproc, size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid) { mi_assert_internal(memid!=NULL); mi_assert_internal(alignment >= _mi_os_page_size() && ((alignment & (alignment - 1)) == 0)); mi_assert_internal(size > 0 && (size % _mi_os_page_size()) == 0); @@ -352,7 +355,7 @@ static void* mi_os_prim_alloc_aligned(size_t size, size_t alignment, bool commit size_t os_size = size; void* p = NULL; if (try_direct_alloc) { - p = mi_os_prim_alloc(size, alignment, commit, allow_large, &os_is_large, &os_is_zero); + p = mi_os_prim_alloc(subproc, size, alignment, commit, allow_large, &os_is_large, &os_is_zero); } // aligned already? @@ -366,33 +369,33 @@ static void* mi_os_prim_alloc_aligned(size_t size, size_t alignment, bool commit _mi_warning_message("unable to allocate aligned OS memory directly, fall back to over-allocation (size: 0x%zx bytes, address: %p, alignment: 0x%zx, commit: %d)\n", size, p, alignment, commit); } #endif - if (p != NULL) { mi_os_prim_free(p, size, (commit ? size : 0), NULL); } + if (p != NULL) { mi_os_prim_free(subproc, p, size, (commit ? size : 0)); } if (size >= (SIZE_MAX - alignment)) return NULL; // overflow const size_t over_size = size + alignment; if (!mi_os_mem_config.has_partial_free) { // win32 virtualAlloc cannot free parts of an allocated block // over-allocate uncommitted (virtual) memory - p = mi_os_prim_alloc(over_size, 1 /*alignment*/, false /* commit? */, false /* allow_large */, &os_is_large, &os_is_zero); + p = mi_os_prim_alloc(subproc, over_size, 1 /*alignment*/, false /* commit? */, false /* allow_large */, &os_is_large, &os_is_zero); if (p == NULL) return NULL; // set p to the aligned part in the full region // note: Windows VirtualFree needs the actual base pointer // this is handled though by having the `base` field in the memid os_base = p; // remember the base - os_size = over_size; + os_size = over_size; // todo: use size instead as now we over-decrement commit stats on free? p = _mi_align_up_ptr(p, alignment); // explicitly commit only the aligned part if (commit) { - if (!_mi_os_commit(p, size, NULL)) { - mi_os_prim_free(os_base, over_size, 0, NULL); + if (!_mi_os_commit(subproc, p, size, NULL)) { + mi_os_prim_free(subproc, os_base, over_size, 0); return NULL; } } } else { // mmap can free inside an allocation // overallocate... - p = mi_os_prim_alloc(over_size, 1, commit, false, &os_is_large, &os_is_zero); + p = mi_os_prim_alloc(subproc, over_size, 1, commit, false, &os_is_large, &os_is_zero); if (p == NULL) return NULL; // and selectively unmap parts around the over-allocated area. @@ -401,8 +404,8 @@ static void* mi_os_prim_alloc_aligned(size_t size, size_t alignment, bool commit const size_t mid_size = _mi_align_up(size, _mi_os_page_size()); const size_t post_size = over_size - pre_size - mid_size; mi_assert_internal(pre_size < over_size&& post_size < over_size&& mid_size >= size); - if (pre_size > 0) { mi_os_prim_free(p, pre_size, (commit ? pre_size : 0), NULL); } - if (post_size > 0) { mi_os_prim_free((uint8_t*)aligned_p + mid_size, post_size, (commit ? post_size : 0), NULL); } + if (pre_size > 0) { mi_os_prim_free(subproc, p, pre_size, (commit ? pre_size : 0)); } + if (post_size > 0) { mi_os_prim_free(subproc, (uint8_t*)aligned_p + mid_size, post_size, (commit ? post_size : 0)); } // we can return the aligned pointer on `mmap` systems p = aligned_p; os_base = aligned_p; // since we freed the pre part, `*base == p`. @@ -421,13 +424,13 @@ static void* mi_os_prim_alloc_aligned(size_t size, size_t alignment, bool commit OS API: alloc and alloc_aligned ----------------------------------------------------------- */ -void* _mi_os_alloc(size_t size, mi_memid_t* memid) { +void* _mi_os_alloc(mi_subproc_t* subproc, size_t size, mi_memid_t* memid) { *memid = _mi_memid_none(); if (size == 0) return NULL; size = _mi_os_good_alloc_size(size); bool os_is_large = false; bool os_is_zero = false; - void* p = mi_os_prim_alloc(size, 0, true, false, &os_is_large, &os_is_zero); + void* p = mi_os_prim_alloc(subproc, size, 0, true, false, &os_is_large, &os_is_zero); if (p == NULL) return NULL; *memid = _mi_memid_create_os(p, size, true, os_is_zero, os_is_large); @@ -436,7 +439,7 @@ void* _mi_os_alloc(size_t size, mi_memid_t* memid) { return p; } -void* _mi_os_alloc_aligned(size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid) +void* _mi_os_alloc_aligned(mi_subproc_t* subproc, size_t size, size_t alignment, bool commit, bool allow_large, mi_memid_t* memid) { MI_UNUSED(&_mi_os_get_aligned_hint); // suppress unused warnings *memid = _mi_memid_none(); @@ -444,7 +447,7 @@ void* _mi_os_alloc_aligned(size_t size, size_t alignment, bool commit, bool allo size = _mi_os_good_alloc_size(size); alignment = _mi_align_up(alignment, _mi_os_page_size()); - void* p = mi_os_prim_alloc_aligned(size, alignment, commit, allow_large, memid ); + void* p = mi_os_prim_alloc_aligned(subproc, size, alignment, commit, allow_large, memid ); if (p == NULL) return NULL; mi_assert_internal(memid->mem.os.size >= size); @@ -454,13 +457,13 @@ void* _mi_os_alloc_aligned(size_t size, size_t alignment, bool commit, bool allo } -mi_decl_nodiscard static void* mi_os_ensure_zero(void* p, size_t size, mi_memid_t* memid) { +mi_decl_nodiscard static void* mi_os_ensure_zero(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t* memid) { if (p==NULL || size==0) return p; // ensure committed if (!memid->initially_committed) { bool is_zero = false; - if (!_mi_os_commit(p, size, &is_zero)) { - _mi_os_free(p, size, *memid); + if (!_mi_os_commit(subproc, p, size, &is_zero)) { + _mi_os_free(subproc, p, size, *memid); return NULL; } memid->initially_committed = true; @@ -472,9 +475,9 @@ mi_decl_nodiscard static void* mi_os_ensure_zero(void* p, size_t size, mi_memid_ return p; } -void* _mi_os_zalloc(size_t size, mi_memid_t* memid) { - void* p = _mi_os_alloc(size,memid); - return mi_os_ensure_zero(p, size, memid); +void* _mi_os_zalloc(mi_subproc_t* subproc, size_t size, mi_memid_t* memid) { + void* p = _mi_os_alloc(subproc, size,memid); + return mi_os_ensure_zero(subproc, p, size, memid); } /* ----------------------------------------------------------- @@ -485,28 +488,28 @@ void* _mi_os_zalloc(size_t size, mi_memid_t* memid) { to use the actual start of the memory region. ----------------------------------------------------------- */ -void* _mi_os_alloc_aligned_at_offset(size_t size, size_t alignment, size_t offset, bool commit, bool allow_large, mi_memid_t* memid) { +void* _mi_os_alloc_aligned_at_offset(mi_subproc_t* subproc, size_t size, size_t alignment, size_t offset, bool commit, bool allow_large, mi_memid_t* memid) { mi_assert(offset <= size); mi_assert((alignment % _mi_os_page_size()) == 0); *memid = _mi_memid_none(); if (offset > size) return NULL; if (offset == 0) { // regular aligned allocation - return _mi_os_alloc_aligned(size, alignment, commit, allow_large, memid); + return _mi_os_alloc_aligned(subproc, size, alignment, commit, allow_large, memid); } else { // overallocate to align at an offset const size_t extra = _mi_align_up(offset, alignment) - offset; if (size >= SIZE_MAX - extra) return NULL; // too large const size_t oversize = size + extra; - void* const start = _mi_os_alloc_aligned(oversize, alignment, commit, allow_large, memid); + void* const start = _mi_os_alloc_aligned(subproc, oversize, alignment, commit, allow_large, memid); if (start == NULL) return NULL; void* const p = (uint8_t*)start + extra; mi_assert(_mi_is_aligned((uint8_t*)p + offset, alignment)); // decommit the overallocation at the start if (commit && extra >= _mi_os_page_size()) { - _mi_os_decommit(start, extra); + _mi_os_decommit(subproc, start, extra); } return p; } @@ -540,9 +543,9 @@ static void* mi_os_page_align_area_conservative(void* addr, size_t size, size_t* return mi_os_page_align_areax(true, addr, size, newsize); } -bool _mi_os_commit_ex(void* addr, size_t size, bool* is_zero, size_t stat_size) { +bool _mi_os_commit_ex(mi_subproc_t* subproc, void* addr, size_t size, bool* is_zero, size_t stat_size) { if (is_zero != NULL) { *is_zero = false; } - mi_os_stat_counter_increase(commit_calls, 1); + mi_subproc_stat_counter_increase(subproc, commit_calls, 1); // page align range size_t csize; @@ -565,15 +568,15 @@ bool _mi_os_commit_ex(void* addr, size_t size, bool* is_zero, size_t stat_size) if (os_is_zero) { mi_track_mem_defined(start,csize); } else { mi_track_mem_undefined(start,csize); } #endif - mi_os_stat_increase(committed, stat_size); // use size for precise commit vs. decommit + mi_subproc_stat_increase(subproc, committed, stat_size); // use size for precise commit vs. decommit return true; } -bool _mi_os_commit(void* addr, size_t size, bool* is_zero) { - return _mi_os_commit_ex(addr, size, is_zero, size); +bool _mi_os_commit(mi_subproc_t* subproc, void* addr, size_t size, bool* is_zero) { + return _mi_os_commit_ex(subproc, addr, size, is_zero, size); } -static bool mi_os_decommit_ex(void* addr, size_t size, bool* needs_recommit, size_t stat_size) { +static bool mi_os_decommit_ex(mi_subproc_t* subproc, void* addr, size_t size, bool* needs_recommit, size_t stat_size) { mi_assert_internal(needs_recommit!=NULL); // page align @@ -588,15 +591,15 @@ static bool mi_os_decommit_ex(void* addr, size_t size, bool* needs_recommit, siz _mi_warning_message("cannot decommit OS memory (error: %d (0x%x), address: %p, size: 0x%zx bytes)\n", err, err, start, csize); } else if (*needs_recommit) { - mi_os_stat_decrease(committed, stat_size); + mi_subproc_stat_decrease(subproc, committed, stat_size); } mi_assert_internal(err == 0); return (err == 0); } -bool _mi_os_decommit(void* addr, size_t size) { +bool _mi_os_decommit(mi_subproc_t* subproc, void* addr, size_t size) { bool needs_recommit; - return mi_os_decommit_ex(addr, size, &needs_recommit, size); + return mi_os_decommit_ex(subproc, addr, size, &needs_recommit, size); } @@ -604,13 +607,13 @@ bool _mi_os_decommit(void* addr, size_t size) { // but may be used later again. This will release physical memory // pages and reduce swapping while keeping the memory committed. // We page align to a conservative area inside the range to reset. -bool _mi_os_reset(void* addr, size_t size) { +bool _mi_os_reset(mi_subproc_t* subproc, void* addr, size_t size) { // page align conservatively within the range size_t csize; void* start = mi_os_page_align_area_conservative(addr, size, &csize); if (csize == 0) return true; // || _mi_os_is_huge_reserved(addr) - mi_os_stat_counter_increase(reset, csize); - mi_os_stat_counter_increase(reset_calls, 1); + mi_subproc_stat_counter_increase(subproc, reset, csize); + mi_subproc_stat_counter_increase(subproc, reset_calls, 1); #if (MI_DEBUG>1) && !MI_SECURE && !MI_TRACK_ENABLED // && !MI_TSAN memset(start, 0, csize); // pretend it is eagerly reset @@ -624,7 +627,8 @@ bool _mi_os_reset(void* addr, size_t size) { } -void _mi_os_reuse( void* addr, size_t size ) { +void _mi_os_reuse( mi_subproc_t* subproc, void* addr, size_t size ) { + MI_UNUSED(subproc); // page align conservatively within the range size_t csize = 0; void* const start = mi_os_page_align_area_conservative(addr, size, &csize); @@ -637,11 +641,11 @@ void _mi_os_reuse( void* addr, size_t size ) { // either resets or decommits memory, returns true if the memory needs // to be recommitted if it is to be re-used later on. -bool _mi_os_purge_ex(void* p, size_t size, bool allow_reset, size_t stat_size, mi_commit_fun_t* commit_fun, void* commit_fun_arg) +bool _mi_os_purge_ex(mi_subproc_t* subproc, void* p, size_t size, bool allow_reset, size_t stat_size, mi_commit_fun_t* commit_fun, void* commit_fun_arg) { if (mi_option_get(mi_option_purge_delay) < 0) return false; // is purging allowed? - mi_os_stat_counter_increase(purge_calls, 1); - mi_os_stat_counter_increase(purged, size); + mi_subproc_stat_counter_increase(subproc, purge_calls, 1); + mi_subproc_stat_counter_increase(subproc, purged, size); if (commit_fun != NULL) { bool decommitted = (*commit_fun)(false, p, size, NULL, commit_fun_arg); @@ -651,12 +655,12 @@ bool _mi_os_purge_ex(void* p, size_t size, bool allow_reset, size_t stat_size, m !_mi_preloading()) // don't decommit during preloading (unsafe) { bool needs_recommit = true; - mi_os_decommit_ex(p, size, &needs_recommit, stat_size); + mi_os_decommit_ex(subproc, p, size, &needs_recommit, stat_size); return needs_recommit; } else { if (allow_reset) { // this can sometimes be not allowed if the range is not fully committed (on Windows, we cannot reset uncommitted memory) - _mi_os_reset(p, size); + _mi_os_reset(subproc, p, size); } return false; // needs no recommit } @@ -664,8 +668,8 @@ bool _mi_os_purge_ex(void* p, size_t size, bool allow_reset, size_t stat_size, m // either resets or decommits memory, returns true if the memory needs // to be recommitted if it is to be re-used later on. -bool _mi_os_purge(void* p, size_t size) { - return _mi_os_purge_ex(p, size, true, size, NULL, NULL); +bool _mi_os_purge(mi_subproc_t* subproc, void* p, size_t size) { + return _mi_os_purge_ex(subproc, p, size, true, size, NULL, NULL); } @@ -711,7 +715,11 @@ static mi_decl_cache_align _Atomic(uintptr_t) mi_huge_start; // = 0 // Claim an aligned address range for huge pages static uint8_t* mi_os_claim_huge_pages(size_t pages, size_t* total_size) { if (total_size != NULL) *total_size = 0; - const size_t size = pages * MI_HUGE_OS_PAGE_SIZE; + size_t size = 0; + if (mi_mul_overflow(pages,MI_HUGE_OS_PAGE_SIZE,&size)) { + _mi_warning_message("too many huge pages requested: %zu\n", pages); + return NULL; + } uintptr_t start = 0; uintptr_t end = 0; @@ -720,7 +728,7 @@ static uint8_t* mi_os_claim_huge_pages(size_t pages, size_t* total_size) { start = huge_start; if (start == 0) { // Initialize the start address after the 32TiB area - start = ((uintptr_t)8 << 40); // 8TiB virtual start address + start = ((uintptr_t)32 << 40); // 32TiB virtual start address (after addresses returned by _mi_os_get_aligned_hint) #if (MI_SECURE>0 || MI_DEBUG==0) // security: randomize start of huge pages unless in debug mode mi_theap_t* const theap = _mi_theap_default(); // don't use `mi_theap_get_default()` as that can cause allocation recursively (issue #1267) if (mi_theap_is_initialized(theap)) { // todo: or no hint at all if we lack randomness? @@ -728,7 +736,7 @@ static uint8_t* mi_os_claim_huge_pages(size_t pages, size_t* total_size) { start = start + ((uintptr_t)MI_HUGE_OS_PAGE_SIZE * ((r>>17) & 0x0FFF)); // (randomly 12bits)*1GiB == between 0 to 4TiB } else { - _mi_warning_message("failed to randomize the start address of huge pages allocation (%zu bytes at %p)", size, start); + _mi_warning_message("failed to randomize the start address of huge pages allocation (%zu bytes at %p)", size, (void*)start); } #endif } @@ -747,7 +755,7 @@ static uint8_t* mi_os_claim_huge_pages(size_t pages, size_t* total_size) { #endif // Allocate MI_ARENA_SLICE_ALIGN aligned huge pages -void* _mi_os_alloc_huge_os_pages(size_t pages, int numa_node, mi_msecs_t max_msecs, size_t* pages_reserved, size_t* psize, mi_memid_t* memid) { +void* _mi_os_alloc_huge_os_pages(mi_subproc_t* subproc, size_t pages, int numa_node, mi_msecs_t max_msecs, size_t* pages_reserved, size_t* psize, mi_memid_t* memid) { *memid = _mi_memid_none(); if (psize != NULL) *psize = 0; if (pages_reserved != NULL) *pages_reserved = 0; @@ -778,21 +786,21 @@ void* _mi_os_alloc_huge_os_pages(size_t pages, int numa_node, mi_msecs_t max_mse // no success, issue a warning and break if (p != NULL) { _mi_warning_message("could not allocate contiguous huge OS page %zu at %p\n", page, addr); - mi_os_prim_free(p, MI_HUGE_OS_PAGE_SIZE, MI_HUGE_OS_PAGE_SIZE, NULL); + mi_os_prim_free(subproc, p, MI_HUGE_OS_PAGE_SIZE, MI_HUGE_OS_PAGE_SIZE); } break; } // success, record it page++; // increase before timeout check (see issue #711) - mi_os_stat_increase(committed, MI_HUGE_OS_PAGE_SIZE); - mi_os_stat_increase(reserved, MI_HUGE_OS_PAGE_SIZE); + mi_subproc_stat_increase(subproc, committed, MI_HUGE_OS_PAGE_SIZE); + mi_subproc_stat_increase(subproc, reserved, MI_HUGE_OS_PAGE_SIZE); // check for timeout if (max_msecs > 0) { mi_msecs_t elapsed = _mi_clock_end(start_t); if (page >= 1) { - mi_msecs_t estimate = ((elapsed / (page+1)) * pages); + mi_msecs_t estimate = ((elapsed / (page==0 ? 1 : page)) * pages); if (estimate > 2*max_msecs) { // seems like we are going to timeout, break elapsed = max_msecs + 1; } @@ -821,11 +829,11 @@ void* _mi_os_alloc_huge_os_pages(size_t pages, int numa_node, mi_msecs_t max_mse // free every huge page in a range individually (as we allocated per page) // note: needed with VirtualAlloc but could potentially be done in one go on mmap'd systems. -static void mi_os_free_huge_os_pages(void* p, size_t size, mi_subproc_t* subproc) { +static void mi_os_free_huge_os_pages(mi_subproc_t* subproc, void* p, size_t size) { if (p==NULL || size==0) return; uint8_t* base = (uint8_t*)p; while (size >= MI_HUGE_OS_PAGE_SIZE) { - mi_os_prim_free(base, MI_HUGE_OS_PAGE_SIZE, MI_HUGE_OS_PAGE_SIZE, subproc); + mi_os_prim_free(subproc, base, MI_HUGE_OS_PAGE_SIZE, MI_HUGE_OS_PAGE_SIZE); size -= MI_HUGE_OS_PAGE_SIZE; base += MI_HUGE_OS_PAGE_SIZE; } diff --git a/vendored/mimalloc/src/page-map.c b/vendored/mimalloc/src/page-map.c index acc0ee9f74..5d688a4106 100644 --- a/vendored/mimalloc/src/page-map.c +++ b/vendored/mimalloc/src/page-map.c @@ -1,5 +1,5 @@ /*---------------------------------------------------------------------------- -Copyright (c) 2023-2025, Microsoft Research, Daan Leijen +Copyright (c) 2023-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -25,8 +25,11 @@ static void mi_page_map_cannot_commit(void) { // A full 256 TiB address space (48 bit) needs a 4 GiB page map. // A full 4 GiB address space (32 bit) needs only a 64 KiB page map. -mi_decl_cache_align uint8_t* _mi_page_map = NULL; -static void* mi_page_map_max_address = NULL; +// Use an initial empty page map so `free(NULL)` works even if mimalloc is not yet initialized (issue #1341) +static uint8_t mi_page_map_empty[1] = { 1 }; // _mi_ptr_page(NULL) == NULL + +mi_decl_hidden mi_decl_cache_align _Atomic(uint8_t*) _mi_page_map = mi_page_map_empty; +mi_decl_hidden _Atomic(void*) _mi_page_map_max_address = NULL; static mi_memid_t mi_page_map_memid; #define MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT MI_ARENA_SLICE_SIZE @@ -35,7 +38,7 @@ static mi_bitmap_t* mi_page_map_commit; // one bit per committed 64 KiB entries mi_decl_nodiscard static bool mi_page_map_ensure_committed(size_t idx, size_t slice_count); bool _mi_page_map_init(void) { - size_t vbits = (size_t)mi_option_get_clamp(mi_option_max_vabits, 0, MI_SIZE_BITS); + size_t vbits = (size_t)mi_option_get_clamp(mi_option_max_vabits, 0, MI_MAX_VABITS); if (vbits == 0) { vbits = _mi_os_virtual_address_bits(); #if MI_ARCH_X64 // canonical address is limited to the first 128 TiB @@ -45,15 +48,22 @@ bool _mi_page_map_init(void) { if (vbits < MI_ARENA_SLICE_SHIFT) { vbits = MI_ARENA_SLICE_SHIFT; } + if (vbits < MI_MIN_VABITS) { // cover at least this much for a faster _mi_checked_ptr + vbits = MI_MIN_VABITS; + } + if (vbits > MI_MAX_VABITS) { // limit page map size even if more virtual addresses are available + vbits = MI_MAX_VABITS; + } // Allocate the page map and commit bits - mi_page_map_max_address = (void*)(vbits >= MI_SIZE_BITS ? (SIZE_MAX - MI_ARENA_SLICE_SIZE + 1) : (MI_PU(1) << vbits)); + mi_atomic_store_ptr_release(void, &_mi_page_map_max_address, (void*)(vbits >= MI_SIZE_BITS ? (SIZE_MAX - MI_ARENA_SLICE_SIZE + 1) : (MI_PU(1) << vbits))); const size_t page_map_size = (MI_ZU(1) << (vbits - MI_ARENA_SLICE_SHIFT)); const bool commit = (page_map_size <= 1*MI_MiB || mi_option_is_enabled(mi_option_pagemap_commit)); // _mi_os_has_overcommit(); // commit on-access on Linux systems? const size_t commit_bits = _mi_divide_up(page_map_size, MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT); const size_t bitmap_size = (commit ? 0 : mi_bitmap_size(commit_bits, NULL)); const size_t reserve_size = bitmap_size + page_map_size; - uint8_t* const base = (uint8_t*)_mi_os_alloc_aligned(reserve_size, 1, commit, true /* allow large */, &mi_page_map_memid); + mi_subproc_t* const subproc = _mi_subproc_main(); + uint8_t* const base = (uint8_t*)_mi_os_alloc_aligned(subproc, reserve_size, 1, commit, true /* allow large */, &mi_page_map_memid); if (base==NULL) { _mi_error_message(ENOMEM, "unable to reserve virtual memory for the page map (%zu KiB)\n", page_map_size / MI_KiB); return false; @@ -64,13 +74,13 @@ bool _mi_page_map_init(void) { } if (bitmap_size > 0) { mi_page_map_commit = (mi_bitmap_t*)base; - if (!_mi_os_commit(mi_page_map_commit, bitmap_size, NULL)) { + if (!_mi_os_commit(subproc, mi_page_map_commit, bitmap_size, NULL)) { mi_page_map_cannot_commit(); return false; } mi_bitmap_init(mi_page_map_commit, commit_bits, true); } - _mi_page_map = base + bitmap_size; + mi_atomic_store_ptr_release(uint8_t,&_mi_page_map, base + bitmap_size); // commit the first part so NULL pointers get resolved without an access violation if (!commit) { @@ -79,19 +89,18 @@ bool _mi_page_map_init(void) { return false; } } - _mi_page_map[0] = 1; // so _mi_ptr_page(NULL) == NULL + mi_atomic_load_ptr_relaxed(uint8_t, &_mi_page_map)[0] = 1; // so _mi_ptr_page(NULL) == NULL mi_assert_internal(_mi_ptr_page(NULL)==NULL); return true; } -void _mi_page_map_unsafe_destroy(mi_subproc_t* subproc) { - mi_assert_internal(subproc != NULL); - mi_assert_internal(_mi_page_map != NULL); - if (_mi_page_map == NULL) return; - _mi_os_free_ex(mi_page_map_memid.mem.os.base, mi_page_map_memid.mem.os.size, true, mi_page_map_memid, subproc); - _mi_page_map = NULL; +void _mi_page_map_unsafe_destroy(void) { + mi_assert_internal(mi_atomic_load_ptr_relaxed(uint8_t, &_mi_page_map) != NULL); + if (mi_atomic_load_ptr_relaxed(uint8_t, &_mi_page_map) == NULL) return; + _mi_os_free_ex(_mi_subproc_main(), mi_page_map_memid.mem.os.base, mi_page_map_memid.mem.os.size, true, mi_page_map_memid); + mi_atomic_store_ptr_release(uint8_t, &_mi_page_map, NULL); mi_page_map_commit = NULL; - mi_page_map_max_address = NULL; + mi_atomic_store_ptr_release(void, &_mi_page_map_max_address, NULL); mi_page_map_memid = _mi_memid_none(); } @@ -100,6 +109,7 @@ static bool mi_page_map_ensure_committed(size_t idx, size_t slice_count) { // is the page map area that contains the page address committed? // we always set the commit bits so we can track what ranges are in-use. // we only actually commit if the map wasn't committed fully already. + uint8_t* const page_map = mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map); if (mi_page_map_commit != NULL) { const size_t commit_idx = idx / MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT; const size_t commit_idx_hi = (idx + slice_count - 1) / MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT; @@ -107,9 +117,9 @@ static bool mi_page_map_ensure_committed(size_t idx, size_t slice_count) { if (mi_bitmap_is_clear(mi_page_map_commit, i)) { // this may race, in which case we do multiple commits (which is ok) bool is_zero; - uint8_t* const start = _mi_page_map + (i * MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT); + uint8_t* const start = page_map + (i * MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT); const size_t size = MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT; - if (!_mi_os_commit(start, size, &is_zero)) { + if (!_mi_os_commit(_mi_subproc_main(), start, size, &is_zero)) { mi_page_map_cannot_commit(); return false; } @@ -119,8 +129,8 @@ static bool mi_page_map_ensure_committed(size_t idx, size_t slice_count) { } } #if MI_DEBUG > 0 - _mi_page_map[idx] = 0; - _mi_page_map[idx+slice_count-1] = 0; + page_map[idx] = 0; + page_map[idx+slice_count-1] = 0; #endif return true; } @@ -137,11 +147,13 @@ static size_t mi_page_map_get_idx(mi_page_t* page, uint8_t** page_start, size_t* bool _mi_page_map_register(mi_page_t* page) { mi_assert_internal(page != NULL); mi_assert_internal(_mi_is_aligned(mi_page_slice_start(page), MI_PAGE_ALIGN)); - mi_assert_internal(_mi_page_map != NULL); // should be initialized before multi-thread access! - if mi_unlikely(_mi_page_map == NULL) { + mi_assert_internal(mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map) != NULL); // should be initialized before multi-thread access! + uint8_t* page_map = mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map); + if mi_unlikely(mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map) == NULL) { if (!_mi_page_map_init()) return false; + page_map = mi_atomic_load_ptr_acquire(uint8_t,&_mi_page_map); } - mi_assert(_mi_page_map!=NULL); + mi_assert(page_map!=NULL); uint8_t* page_start; size_t slice_count; const size_t idx = mi_page_map_get_idx(page, &page_start, &slice_count); @@ -153,37 +165,42 @@ bool _mi_page_map_register(mi_page_t* page) { // set the offsets for (size_t i = 0; i < slice_count; i++) { mi_assert_internal(i < 128); - _mi_page_map[idx + i] = (uint8_t)(i+1); + page_map[idx + i] = (uint8_t)(i+1); } return true; } void _mi_page_map_unregister(mi_page_t* page) { - mi_assert_internal(_mi_page_map != NULL); + uint8_t* const page_map = mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map); + mi_assert_internal(page_map != NULL); + if (page_map == NULL) return; // get index and count uint8_t* page_start; size_t slice_count; const size_t idx = mi_page_map_get_idx(page, &page_start, &slice_count); // unset the offsets - _mi_memzero(_mi_page_map + idx, slice_count); + _mi_memzero(page_map + idx, slice_count); } void _mi_page_map_unregister_range(void* start, size_t size) { + uint8_t* const page_map = mi_atomic_load_ptr_relaxed(uint8_t,&_mi_page_map); + mi_assert_internal(page_map!=NULL); + if (page_map == NULL) return; const size_t slice_count = _mi_divide_up(size, MI_ARENA_SLICE_SIZE); const uintptr_t index = _mi_page_map_index(start); // todo: scan the commit bits and clear only those ranges? if (!mi_page_map_ensure_committed(index, slice_count)) { // we commit the range in total; return; } - _mi_memzero(&_mi_page_map[index], slice_count); + _mi_memzero(&page_map[index], slice_count); } mi_page_t* _mi_safe_ptr_page(const void* p) { - if mi_unlikely(p >= mi_page_map_max_address) return NULL; + if mi_unlikely(p >= mi_atomic_load_ptr_relaxed(void, &_mi_page_map_max_address)) return NULL; const uintptr_t idx = _mi_page_map_index(p); if mi_unlikely(mi_page_map_commit != NULL && !mi_bitmap_is_set(mi_page_map_commit, idx/MI_PAGE_MAP_ENTRIES_PER_COMMIT_BIT)) return NULL; - const uintptr_t ofs = _mi_page_map[idx]; + const uintptr_t ofs = _mi_page_map_at(idx); if mi_unlikely(ofs == 0) return NULL; return (mi_page_t*)((((uintptr_t)p >> MI_ARENA_SLICE_SHIFT) - ofs + 1) << MI_ARENA_SLICE_SHIFT); } @@ -198,9 +215,13 @@ mi_decl_nodiscard mi_decl_export bool mi_is_in_heap_region(const void* p) mi_att #define MI_PAGE_MAP_SUB_SIZE (MI_PAGE_MAP_SUB_COUNT * sizeof(mi_page_t*)) #define MI_PAGE_MAP_ENTRIES_PER_CBIT (MI_PAGE_MAP_COUNT < MI_BFIELD_BITS ? 1 : (MI_PAGE_MAP_COUNT / MI_BFIELD_BITS)) -mi_decl_cache_align _Atomic(mi_submap_t)* _mi_page_map; +// Use an initial empty page map so `free(NULL)` works even if mimalloc is not yet initialized (issue #1341) +static mi_page_t* mi_submap_empty[1] = { NULL }; +static _Atomic(mi_submap_t) mi_page_map_empty[1] = { MI_ATOMIC_VAR_INIT(mi_submap_empty) }; + +mi_decl_hidden mi_decl_cache_align _Atomic(mi_submap_t)* _mi_page_map = mi_page_map_empty; +mi_decl_hidden _Atomic(void*) _mi_page_map_max_address = NULL; static size_t mi_page_map_count; -static void* mi_page_map_max_address; static mi_memid_t mi_page_map_memid; static mi_lock_t mi_page_map_lock; @@ -220,7 +241,7 @@ mi_decl_nodiscard static bool mi_page_map_ensure_committed(size_t idx, mi_submap size_t bit_idx; if mi_unlikely(!mi_page_map_is_committed(idx, &bit_idx)) { uint8_t* start = (uint8_t*)&_mi_page_map[bit_idx * MI_PAGE_MAP_ENTRIES_PER_CBIT]; - if (!_mi_os_commit(start, MI_PAGE_MAP_ENTRIES_PER_CBIT * sizeof(mi_submap_t), NULL)) { + if (!_mi_os_commit(_mi_subproc_main(), start, MI_PAGE_MAP_ENTRIES_PER_CBIT * sizeof(mi_submap_t), NULL)) { mi_page_map_cannot_commit(); return false; } @@ -232,7 +253,7 @@ mi_decl_nodiscard static bool mi_page_map_ensure_committed(size_t idx, mi_submap // initialize the page map bool _mi_page_map_init(void) { - size_t vbits = (size_t)mi_option_get_clamp(mi_option_max_vabits, 0, MI_SIZE_BITS); + size_t vbits = (size_t)mi_option_get_clamp(mi_option_max_vabits, 0, MI_MAX_VABITS); if (vbits == 0) { vbits = _mi_os_virtual_address_bits(); #if MI_ARCH_X64 // canonical address is limited to the first 128 TiB @@ -242,10 +263,16 @@ bool _mi_page_map_init(void) { if (vbits < MI_PAGE_MAP_SUB_SHIFT + MI_ARENA_SLICE_SHIFT) { vbits = MI_PAGE_MAP_SUB_SHIFT + MI_ARENA_SLICE_SHIFT; } + if (vbits < MI_MIN_VABITS) { // cover at least this much for a faster _mi_checked_ptr + vbits = MI_MIN_VABITS; + } + if (vbits > MI_MAX_VABITS) { // limit page map size even if more virtual addresses are available + vbits = MI_MAX_VABITS; + } // Allocate the page map and commit bits mi_assert(MI_MAX_VABITS >= vbits); - mi_page_map_max_address = (void*)(vbits >= MI_SIZE_BITS ? (SIZE_MAX - MI_ARENA_SLICE_SIZE + 1) : (MI_PU(1) << vbits)); + mi_atomic_store_ptr_release(void, &_mi_page_map_max_address, (void*)(vbits >= MI_SIZE_BITS ? (SIZE_MAX - MI_ARENA_SLICE_SIZE + 1) : (MI_PU(1) << vbits))); mi_page_map_count = (MI_ZU(1) << (vbits - MI_PAGE_MAP_SUB_SHIFT - MI_ARENA_SLICE_SHIFT)); mi_assert(mi_page_map_count <= MI_PAGE_MAP_COUNT); const size_t os_page_size = _mi_os_page_size(); @@ -258,9 +285,11 @@ bool _mi_page_map_init(void) { const bool commit = page_map_size <= 64*MI_KiB || mi_option_is_enabled(mi_option_pagemap_commit) || _mi_os_has_overcommit(); #endif - _mi_page_map = (_Atomic(mi_page_t**)*)_mi_os_alloc_aligned(reserve_size, 1, commit, true /* allow large */, &mi_page_map_memid); + mi_subproc_t* const subproc = _mi_subproc_main(); + _mi_page_map = (_Atomic(mi_page_t**)*)_mi_os_alloc_aligned(subproc, reserve_size, 1, commit, true /* allow large */, &mi_page_map_memid); if (_mi_page_map==NULL) { _mi_error_message(ENOMEM, "unable to reserve virtual memory for the page map (%zu KiB)\n", page_map_size / MI_KiB); + _mi_page_map = mi_page_map_empty; return false; } if (mi_page_map_memid.initially_committed && !mi_page_map_memid.initially_zero) { @@ -272,7 +301,7 @@ bool _mi_page_map_init(void) { // ensure there is a submap for the NULL address mi_page_t** const sub0 = (mi_page_t**)((uint8_t*)_mi_page_map + page_map_size); // we reserved a submap part at the end already if (!mi_page_map_memid.initially_committed) { - if (!_mi_os_commit(sub0, submap_size, NULL)) { // commit full submap (issue #1087) + if (!_mi_os_commit(subproc, sub0, submap_size, NULL)) { // commit full submap (issue #1087) mi_page_map_cannot_commit(); return false; } @@ -293,10 +322,10 @@ bool _mi_page_map_init(void) { } -void _mi_page_map_unsafe_destroy(mi_subproc_t* subproc) { - mi_assert_internal(subproc != NULL); +void _mi_page_map_unsafe_destroy(void) { mi_assert_internal(_mi_page_map != NULL); if (_mi_page_map == NULL) return; + mi_subproc_t* const subproc = _mi_subproc_main(); mi_lock_done(&mi_page_map_lock); for (size_t idx = 1; idx < mi_page_map_count; idx++) { // skip entry 0 (as we allocate that submap at the end of the page_map) // free all sub-maps @@ -304,16 +333,16 @@ void _mi_page_map_unsafe_destroy(mi_subproc_t* subproc) { mi_submap_t sub = _mi_page_map_at(idx); if (sub != NULL) { mi_memid_t memid = _mi_memid_create_os(sub, MI_PAGE_MAP_SUB_SIZE, true, false, false); - _mi_os_free_ex(memid.mem.os.base, memid.mem.os.size, true, memid, subproc); + _mi_os_free_ex(subproc, memid.mem.os.base, memid.mem.os.size, true, memid); mi_atomic_store_ptr_release(mi_page_t*, &_mi_page_map[idx], NULL); } } } - _mi_os_free_ex(_mi_page_map, mi_page_map_memid.mem.os.size, true, mi_page_map_memid, subproc); + _mi_os_free_ex(subproc, _mi_page_map, mi_page_map_memid.mem.os.size, true, mi_page_map_memid); _mi_page_map = NULL; mi_page_map_count = 0; mi_page_map_memid = _mi_memid_none(); - mi_page_map_max_address = NULL; + mi_atomic_store_ptr_release(void, &_mi_page_map_max_address, NULL); mi_atomic_store_release(&mi_page_map_commit, (mi_bfield_t)0); } @@ -331,9 +360,10 @@ mi_decl_nodiscard static bool mi_page_map_ensure_submap_at(size_t idx, mi_submap sub = mi_atomic_load_ptr_acquire(mi_page_t*, &_mi_page_map[idx]); // reload if (sub==NULL) // not yet allocated by another thread? { + mi_subproc_t* const subproc = _mi_subproc_main(); mi_memid_t memid; - const size_t submap_size = MI_PAGE_MAP_SUB_SIZE; - sub = (mi_submap_t)_mi_os_zalloc(submap_size, &memid); + const size_t submap_size = MI_PAGE_MAP_SUB_SIZE; + sub = (mi_submap_t)_mi_os_zalloc(subproc, submap_size, &memid); if (sub==NULL) { _mi_warning_message("internal error: unable to extend the page map\n"); } @@ -341,7 +371,7 @@ mi_decl_nodiscard static bool mi_page_map_ensure_submap_at(size_t idx, mi_submap mi_submap_t expect = NULL; if (!mi_atomic_cas_ptr_strong_acq_rel(mi_page_t*, &_mi_page_map[idx], &expect, sub)) { // another thread already allocated it.. free and continue - _mi_os_free(sub, submap_size, memid); + _mi_os_free(subproc, sub, submap_size, memid); sub = expect; } } @@ -411,6 +441,7 @@ void _mi_page_map_unregister(mi_page_t* page) { mi_assert_internal(_mi_page_map != NULL); mi_assert_internal(page != NULL); mi_assert_internal(_mi_is_aligned(mi_page_slice_start(page), MI_PAGE_ALIGN)); + // note: should proceed even if the page was not registered yet (for failure paths in page allocation in `arena.c`) if mi_unlikely(_mi_page_map == NULL) return; // get index and count size_t slice_count; @@ -431,7 +462,7 @@ void _mi_page_map_unregister_range(void* start, size_t size) { // Return NULL for invalid pointers mi_page_t* _mi_safe_ptr_page(const void* p) { if (p==NULL) return NULL; - if mi_unlikely(p >= mi_page_map_max_address) return NULL; + if mi_unlikely(p >= mi_atomic_load_ptr_relaxed(void, &_mi_page_map_max_address)) return NULL; size_t sub_idx; const size_t idx = _mi_page_map_index(p,&sub_idx); if mi_unlikely(!mi_page_map_is_committed(idx,NULL)) return NULL; diff --git a/vendored/mimalloc/src/page-queue.c b/vendored/mimalloc/src/page-queue.c index 731672b025..ab1ca85098 100644 --- a/vendored/mimalloc/src/page-queue.c +++ b/vendored/mimalloc/src/page-queue.c @@ -112,10 +112,10 @@ size_t _mi_bin_size(size_t bin) { // Good size for allocation mi_decl_nodiscard mi_decl_export size_t mi_good_size(size_t size) mi_attr_noexcept { - if (size <= MI_LARGE_MAX_OBJ_SIZE) { + if (size <= MI_LARGE_MAX_OBJ_SIZE - MI_PADDING_SIZE) { return _mi_bin_size(mi_bin(size + MI_PADDING_SIZE)); } - else if (size <= MI_MAX_ALLOC_SIZE) { + else if (size <= MI_MAX_ALLOC_SIZE - MI_PADDING_SIZE) { return _mi_align_up(size + MI_PADDING_SIZE,_mi_os_page_size()); } else { @@ -421,38 +421,3 @@ static void mi_page_queue_enqueue_from_full(mi_page_queue_t* to, mi_page_queue_t // note: we could insert at the front to increase reuse, but it slows down certain benchmarks (like `alloc-test`) mi_page_queue_enqueue_from_ex(to, from, true /* enqueue at the end of the `to` queue? */, page); } - -// Only called from `mi_theap_absorb`. -size_t _mi_page_queue_append(mi_theap_t* theap, mi_page_queue_t* pq, mi_page_queue_t* append) { - mi_assert_internal(mi_theap_contains_queue(theap,pq)); - mi_assert_internal(pq->block_size == append->block_size); - - if (append->first==NULL) return 0; - - // set append pages to new theap and count - size_t count = 0; - for (mi_page_t* page = append->first; page != NULL; page = page->next) { - mi_page_set_theap(page, theap); - count++; - } - mi_assert_internal(count == append->count); - - if (pq->last==NULL) { - // take over afresh - mi_assert_internal(pq->first==NULL); - pq->first = append->first; - pq->last = append->last; - mi_theap_queue_first_update(theap, pq); - } - else { - // append to end - mi_assert_internal(pq->last!=NULL); - mi_assert_internal(append->first!=NULL); - pq->last->next = append->first; - append->first->prev = pq->last; - pq->last = append->last; - } - pq->count += append->count; - - return count; -} diff --git a/vendored/mimalloc/src/page.c b/vendored/mimalloc/src/page.c index 97d2fc2797..31a610b4b0 100644 --- a/vendored/mimalloc/src/page.c +++ b/vendored/mimalloc/src/page.c @@ -15,6 +15,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc/internal.h" #include "mimalloc/atomic.h" #include "mimalloc/prim.h" +#include "mimalloc/prim-tls.h" /* ----------------------------------------------------------- Definition of page queues for each block size @@ -41,7 +42,7 @@ static bool mi_page_extend_free(mi_theap_t* theap, mi_page_t* page); #if (MI_DEBUG>=3) static size_t mi_page_list_count(mi_page_t* page, mi_block_t* head) { - mi_assert_internal(_mi_ptr_page(page->page_start) == page); + mi_assert_internal(_mi_ptr_page(mi_page_start(page)) == page); const uint8_t* slice_start = mi_page_slice_start(page); mi_assert_internal(_mi_is_aligned(slice_start,MI_PAGE_ALIGN)); size_t count = 0; @@ -84,6 +85,10 @@ static bool mi_page_is_valid_init(mi_page_t* page) { mi_assert_internal(mi_page_block_size(page) > 0); mi_assert_internal(page->used <= page->capacity); mi_assert_internal(page->capacity <= page->reserved); + + mi_assert_internal(page->heap!=NULL); + mi_theap_t* const page_theap = _mi_heap_theap_peek(page->heap); + mi_assert_internal(page_theap == NULL || mi_page_theap(page)==page_theap); // const size_t bsize = mi_page_block_size(page); // uint8_t* start = mi_page_start(page); @@ -99,7 +104,7 @@ static bool mi_page_is_valid_init(mi_page_t* page) { mi_assert_expensive(mi_mem_is_zero(block + 1, ubsize - sizeof(mi_block_t))); } } - #endif + #endif #if !MI_TRACK_ENABLED && !MI_TSAN mi_block_t* tfree = mi_page_thread_free(page); @@ -123,6 +128,9 @@ bool _mi_page_is_valid(mi_page_t* page) { #endif if (!mi_page_is_abandoned(page)) { //mi_assert_internal(!_mi_process_is_initialized); + mi_assert_internal(page->heap!=NULL); + mi_theap_t* const page_theap = _mi_heap_theap_peek(page->heap); + mi_assert_internal(page_theap == NULL || mi_page_theap(page)==page_theap); { mi_page_queue_t* pq = mi_page_queue_of(page); mi_assert_internal(mi_page_queue_contains(pq, page)); @@ -155,7 +163,7 @@ static void mi_page_thread_collect_to_local(mi_page_t* page, mi_block_t* head) // if `count > max_count` there was a memory corruption (possibly infinite list due to double multi-threaded free) if mi_unlikely(count > max_count) { - _mi_error_message(EFAULT, "corrupted thread-free list\n"); + _mi_error_message(EFAULT, "corrupted thread-free list (possibly due to a cross-thread double free)\n"); return; // the thread-free items cannot be freed } // if `count > page->used` there was another kind memory corruption (either in the page meta-data or in the linked list) @@ -265,24 +273,6 @@ void _mi_page_free_collect_partly(mi_page_t* page, mi_block_t* head) { Page fresh and retire ----------------------------------------------------------- */ -/* -// called from segments when reclaiming abandoned pages -void _mi_page_reclaim(mi_theap_t* theap, mi_page_t* page) { - // mi_page_set_theap(page, theap); - // _mi_page_use_delayed_free(page, MI_USE_DELAYED_FREE, true); // override never (after theap is set) - _mi_page_free_collect(page, false); // ensure used count is up to date - - mi_assert_expensive(mi_page_is_valid_init(page)); - // mi_assert_internal(mi_page_theap(page) == theap); - // mi_assert_internal(mi_page_thread_free_flag(page) != MI_NEVER_DELAYED_FREE); - - // TODO: push on full queue immediately if it is full? - mi_page_queue_t* pq = mi_theap_page_queue_of(theap, page); - mi_page_queue_push(theap, pq, page); - mi_assert_expensive(_mi_page_is_valid(page)); -} -*/ - // called from `mi_free` on a reclaim, and fresh_alloc if we get an abandoned page void _mi_theap_page_reclaim(mi_theap_t* theap, mi_page_t* page) { @@ -331,7 +321,9 @@ static mi_page_t* mi_page_fresh_alloc(mi_theap_t* theap, mi_page_queue_t* pq, si if (!mi_page_immediate_available(page)) { if (mi_page_is_expandable(page)) { if (!mi_page_extend_free(theap, page)) { - return NULL; // cannot commit + // cannot commit + _mi_page_abandon(page,pq); + return NULL; }; } else { @@ -419,7 +411,6 @@ void _mi_page_free(mi_page_t* page, mi_page_queue_t* pq) { _mi_arenas_collect(false, false, theap->tld); // allow purging } -#define MI_MAX_RETIRE_SIZE MI_LARGE_OBJ_SIZE_MAX // should be less than size for MI_BIN_HUGE #define MI_RETIRE_CYCLES (16) // Retire a page with no more used blocks @@ -530,7 +521,7 @@ static void mi_theap_collect_full_pages(mi_theap_t* theap) { #define MI_MIN_SLICES (2) static void mi_page_free_list_extend_secure(mi_theap_t* const theap, mi_page_t* const page, const size_t bsize, const size_t extend) { - #if (MI_SECURE<3) + #if (MI_SECURE < 2) mi_assert_internal(page->free == NULL); mi_assert_internal(page->local_free == NULL); #endif @@ -587,7 +578,7 @@ static void mi_page_free_list_extend_secure(mi_theap_t* const theap, mi_page_t* static mi_decl_noinline void mi_page_free_list_extend( mi_page_t* const page, const size_t bsize, const size_t extend) { - #if (MI_SECURE<3) + #if (MI_SECURE < 2) mi_assert_internal(page->free == NULL); mi_assert_internal(page->local_free == NULL); #endif @@ -628,7 +619,7 @@ static mi_decl_noinline void mi_page_free_list_extend( mi_page_t* const page, co // extra test in malloc? or cache effects?) static bool mi_page_extend_free(mi_theap_t* theap, mi_page_t* page) { mi_assert_expensive(mi_page_is_valid_init(page)); - #if (MI_SECURE<3) + #if (MI_SECURE < 2) mi_assert(page->free == NULL); mi_assert(page->local_free == NULL); if (page->free != NULL) return true; @@ -662,14 +653,28 @@ static bool mi_page_extend_free(mi_theap_t* theap, mi_page_t* page) { // commit on demand? if (page->slice_committed > 0) { + // reduce extend if it commits more than an arena slice + if ((extend * bsize) > MI_ARENA_SLICE_SIZE) { + extend = _mi_divide_up(MI_ARENA_SLICE_SIZE, bsize); + } + // commit required size const size_t needed_size = (page->capacity + extend)*bsize; - const size_t needed_commit = _mi_align_up( mi_page_slice_offset_of(page, needed_size), MI_PAGE_MIN_COMMIT_SIZE ); + mi_assert_internal(needed_size <= page_size); + size_t needed_commit = _mi_align_up( mi_page_slice_offset_of(page, needed_size), MI_PAGE_MIN_COMMIT_SIZE ); + #if MI_SECURE>=5 + // the previous alignup could extend the commit into the guard page; re-adjust if needed + const size_t page_size_commit = _mi_align_up( mi_page_slice_offset_of(page, page_size), _mi_os_page_size() ); + if (needed_commit > page_size_commit) { + needed_commit = page_size_commit; + } + #endif if (needed_commit > page->slice_committed) { mi_assert_internal(((needed_commit - page->slice_committed) % _mi_os_page_size()) == 0); - if (!_mi_os_commit(mi_page_slice_start(page) + page->slice_committed, needed_commit - page->slice_committed, NULL)) { + if (!_mi_os_commit(_mi_theap_subproc(theap), mi_page_slice_start(page) + page->slice_committed, needed_commit - page->slice_committed, NULL)) { return false; } - page->slice_committed = needed_commit; + mi_assert_internal(needed_commit < UINT32_MAX); + page->slice_committed = (uint32_t)needed_commit; } } @@ -693,8 +698,8 @@ static bool mi_page_extend_free(mi_theap_t* theap, mi_page_t* page) { mi_decl_nodiscard bool _mi_page_init(mi_theap_t* theap, mi_page_t* page) { mi_assert(page != NULL); mi_assert(theap!=NULL); - page->heap = (_mi_is_heap_main(_mi_theap_heap(theap)) ? NULL : _mi_theap_heap(theap)); // faster for `mi_page_associated_theap` - mi_page_set_theap(page, theap); + // page->heap = (_mi_is_heap_main(_mi_theap_heap(theap)) ? NULL : _mi_theap_heap(theap)); // faster for `mi_page_associated_theap` + // mi_page_set_theap(page, theap); size_t page_size; uint8_t* page_start = mi_page_area(page, &page_size); MI_UNUSED(page_start); @@ -707,11 +712,13 @@ mi_decl_nodiscard bool _mi_page_init(mi_theap_t* theap, mi_page_t* page) { #endif #if MI_DEBUG>2 if (page->memid.initially_zero) { - mi_track_mem_defined(page->page_start, mi_page_committed(page)); + mi_track_mem_defined(mi_page_start(page), mi_page_committed(page)); mi_assert_expensive(mi_mem_is_zero(page_start, mi_page_committed(page))); } #endif + mi_assert_internal(page->heap != NULL); + mi_assert_internal(page->heap == _mi_theap_heap(theap)); mi_assert_internal(page->theap!=NULL); mi_assert_internal(page->theap == mi_page_theap(page)); mi_assert_internal(page->capacity == 0); @@ -984,10 +991,8 @@ void* _mi_malloc_generic(mi_theap_t* theap, size_t size, size_t zero_huge_alignm return NULL; } // otherwise we initialize the thread and its default theap - mi_thread_init(); - theap = _mi_theap_default(); - if mi_unlikely(!mi_theap_is_initialized(theap)) { return NULL; } - mi_assert_internal(_mi_theap_default()==theap); + theap = _mi_thread_init(); + if mi_unlikely(!mi_theap_is_initialized(theap)) { return NULL; } } mi_assert_internal(mi_theap_is_initialized(theap)); diff --git a/vendored/mimalloc/src/prim/emscripten/prim.c b/vendored/mimalloc/src/prim/emscripten/prim.c index 1ba0936e09..992965d56f 100644 --- a/vendored/mimalloc/src/prim/emscripten/prim.c +++ b/vendored/mimalloc/src/prim/emscripten/prim.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen, Alon Zakai +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen, Alon Zakai This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -61,7 +61,6 @@ void _mi_prim_mem_init( mi_os_mem_config_t* config) { extern void emmalloc_free(void*); int _mi_prim_free(void* addr, size_t size) { - if (size==0) return 0; emmalloc_free(addr); return 0; } @@ -77,7 +76,7 @@ extern void* emmalloc_memalign(size_t alignment, size_t size); int _mi_prim_alloc(void* hint_addr, size_t size, size_t try_alignment, bool commit, bool allow_large, bool* is_large, bool* is_zero, void** addr) { MI_UNUSED(try_alignment); MI_UNUSED(allow_large); MI_UNUSED(commit); MI_UNUSED(hint_addr); *is_large = false; - // TODO: Track the highest address ever seen; first uses of it are zeroes. + // todo: Track the highest address ever seen; first uses of it are zeroes. // That assumes no one else uses sbrk but us (they could go up, // scribble, and then down), but we could assert on that perhaps. *is_zero = false; @@ -101,7 +100,7 @@ int _mi_prim_alloc(void* hint_addr, size_t size, size_t try_alignment, bool comm int _mi_prim_commit(void* addr, size_t size, bool* is_zero) { MI_UNUSED(addr); MI_UNUSED(size); - // See TODO above. + // See todo above. *is_zero = false; return 0; } @@ -155,7 +154,8 @@ size_t _mi_prim_numa_node_count(void) { #include mi_msecs_t _mi_prim_clock_now(void) { - return emscripten_date_now(); + // todo: use a monotonic clock instead + return emscripten_date_now(); } @@ -185,12 +185,12 @@ void _mi_prim_out_stderr( const char* msg) { // Environment //---------------------------------------------------------------- -bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { +int _mi_prim_getenv(const char* name, char* result, size_t result_size) { // For code size reasons, do not support environ customization for now. MI_UNUSED(name); MI_UNUSED(result); MI_UNUSED(result_size); - return false; + return 0; // not found } @@ -212,7 +212,7 @@ bool _mi_prim_random_buf(void* buf, size_t buf_len) { // use pthread local storage keys to detect thread ending // (and used with MI_TLS_PTHREADS for the default theap) -pthread_key_t _mi_heap_default_key = (pthread_key_t)(-1); +pthread_key_t _mi_heap_default_key = MI_PTHREAD_KEY_INVALID; static void mi_pthread_done(void* value) { if (value!=NULL) { @@ -221,18 +221,20 @@ static void mi_pthread_done(void* value) { } void _mi_prim_thread_init_auto_done(void) { - mi_assert_internal(_mi_heap_default_key == (pthread_key_t)(-1)); + mi_assert_internal(_mi_heap_default_key == MI_PTHREAD_KEY_INVALID); pthread_key_create(&_mi_heap_default_key, &mi_pthread_done); } void _mi_prim_thread_done_auto_done(void) { - if (_mi_heap_default_key != (pthread_key_t)(-1)) { // do not leak the key, see issue #809 - pthread_key_delete(_mi_heap_default_key); + pthread_key_t key = _mi_heap_default_key; + if (key != MI_PTHREAD_KEY_INVALID) { // do not leak the key, see issue #809 + _mi_heap_default_key = MI_PTHREAD_KEY_INVALID; + pthread_key_delete(key); } } void _mi_prim_thread_associate_default_theap(mi_theap_t* theap) { - if (_mi_heap_default_key != (pthread_key_t)(-1)) { // can happen during recursive invocation on freeBSD + if (_mi_heap_default_key != MI_PTHREAD_KEY_INVALID) { // can happen during recursive invocation on freeBSD pthread_setspecific(_mi_heap_default_key, theap); } } diff --git a/vendored/mimalloc/src/prim/osx/alloc-override-zone.c b/vendored/mimalloc/src/prim/osx/alloc-override-zone.c index fdbe095c2e..f529b545dd 100644 --- a/vendored/mimalloc/src/prim/osx/alloc-override-zone.c +++ b/vendored/mimalloc/src/prim/osx/alloc-override-zone.c @@ -7,7 +7,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc.h" #include "mimalloc/internal.h" - +#include "mimalloc/prim-tls.h" // _mi_thread_is_initialized #if defined(MI_MALLOC_OVERRIDE) #if !defined(__APPLE__) @@ -71,8 +71,14 @@ static void* zone_valloc(malloc_zone_t* zone, size_t size) { } static void zone_free(malloc_zone_t* zone, void* p) { - if (mi_any_heap_contains(p)) { - mi_free(p); // with the page_map and pagemap_commit=1 we can use the regular free + if mi_likely(mi_any_heap_contains(p)) { + if mi_likely(_mi_thread_is_initialized()) { + mi_free(p); // with the page_map and pagemap_commit=1 we can use the regular free + } + else { + // during thread shutdown `_pthread_tsd_cleanup` may call `zone_free` on a pointer that was allocated in another subproc. + _mi_free_subproc_safe(p); + } } else if (!is_mimalloc_zone(zone)) { // can happen due to interpose zone->free(zone,p); @@ -134,7 +140,6 @@ static boolean_t zone_claimed_address(malloc_zone_t* zone, void* p) { return mi_is_in_heap_region(p); } - /* ------------------------------------------------------ Introspection members ------------------------------------------------------ */ @@ -194,6 +199,14 @@ static boolean_t intro_zone_locked(malloc_zone_t* zone) { return false; } +// Required whenever the zone advertises version >= 9: macOS calls this from the +// atfork_child handler (_malloc_fork_child) without a NULL check. mimalloc keeps +// no zone-level locks that need reinitializing after fork, so a no-op is safe. +// Leaving it NULL makes the forked child jump to address 0 and crash while forking. +static void intro_reinit_lock(malloc_zone_t* zone) { + MI_UNUSED(zone); +} + /* ------------------------------------------------------ At process start, override the default allocator @@ -218,6 +231,9 @@ static malloc_introspection_t mi_introspect = { #if defined(MAC_OS_X_VERSION_10_6) && (MAC_OS_X_VERSION_MAX_ALLOWED >= MAC_OS_X_VERSION_10_6) && !defined(__ppc__) .statistics = &intro_statistics, .zone_locked = &intro_zone_locked, +#endif +#if defined(MAC_OS_X_VERSION_10_12) && (MAC_OS_X_VERSION_MAX_ALLOWED >= MAC_OS_X_VERSION_10_12) && !defined(__ppc__) + .reinit_lock = &intro_reinit_lock, #endif }; diff --git a/vendored/mimalloc/src/prim/prim.c b/vendored/mimalloc/src/prim/prim.c index 5147bae81f..9e2afdfc88 100644 --- a/vendored/mimalloc/src/prim/prim.c +++ b/vendored/mimalloc/src/prim/prim.c @@ -27,7 +27,7 @@ terms of the MIT license. A copy of the license can be found in the file #endif // Generic process initialization -#ifndef MI_PRIM_HAS_PROCESS_ATTACH +#if !defined(MI_PRIM_HAS_PROCESS_ATTACH) #if defined(__GNUC__) || defined(__clang__) // gcc,clang: use the constructor/destructor attribute // which for both seem to run before regular constructors/destructors diff --git a/vendored/mimalloc/src/prim/unix/prim.c b/vendored/mimalloc/src/prim/unix/prim.c index bec22c0ba7..31a68e21da 100644 --- a/vendored/mimalloc/src/prim/unix/prim.c +++ b/vendored/mimalloc/src/prim/unix/prim.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -41,6 +41,13 @@ terms of the MIT license. A copy of the license can be found in the file #else #include #endif + #if defined(__riscv) || defined(_M_RISCV) + #if defined(MI_HAS_SYS_HWPROBEH) + #include + #elif defined(MI_HAS_ASM_HWPROBEH) + #include + #endif + #endif #elif defined(__APPLE__) #include #include @@ -72,7 +79,7 @@ terms of the MIT license. A copy of the license can be found in the file #define MADV_FREE POSIX_MADV_FREE #endif -#define MI_UNIX_LARGE_PAGE_SIZE (2*MI_MiB) // TODO: can we query the OS for this? +#define MI_UNIX_LARGE_PAGE_SIZE (2*MI_MiB) // todo: can we query the OS for this? //------------------------------------------------------------------------------------ // Use syscalls for some primitives to allow for libraries that override open/read/close etc. @@ -150,13 +157,14 @@ static bool unix_detect_thp(void) { #if defined(__linux__) int fd = mi_prim_open("/sys/kernel/mm/transparent_hugepage/enabled", O_RDONLY); if (fd >= 0) { - char buf[32]; + char buf[64]; ssize_t nread = mi_prim_read(fd, &buf, sizeof(buf)); mi_prim_close(fd); // // between brackets is the current value, for example: always [madvise] never if (nread >= 1) { - thp_enabled = (_mi_strnstr(buf,32,"[never]") == NULL); + if (nread > 64) { nread = 64; } + thp_enabled = (_mi_strnstr(buf,nread,"[never]") == NULL); } } #endif @@ -205,6 +213,40 @@ static void unix_detect_physical_memory( size_t page_size, size_t* physical_memo #endif } +// Detect the virtual address bits (currently Linux/RISC-V only) +static size_t unix_detect_virtual_address_bits(void) { + #if defined(__riscv) || defined(_M_RISCV) + #if defined(RISCV_HWPROBE_KEY_HIGHEST_VIRT_ADDRESS) + struct riscv_hwprobe probe = { .key = RISCV_HWPROBE_KEY_HIGHEST_VIRT_ADDRESS, }; + // Prefer the GNU libc interface if available, as it can also use the VDSO + #if defined(MI_HAS_SYS_HWPROBEH) + if (__riscv_hwprobe(&probe, 1, 0, NULL, 0) == 0) + #else + if (syscall(__NR_riscv_hwprobe, &probe, 1, 0, NULL, 0) == 0) + #endif + { + if (probe.key != -1) { // If a key is unknown to the kernel, its key field will be cleared to -1. + return (MI_SIZE_BITS - mi_clz((uintptr_t)probe.value)); + } + } + #endif + // Fallback to checking /proc/cpuinfo for older kernels + const int fd = mi_prim_open("/proc/cpuinfo", O_RDONLY); + if (fd >= 0) { + char buf[2048]; + const ssize_t nread = mi_prim_read(fd, &buf, sizeof(buf)); + mi_prim_close(fd); + if ((nread >= 1) && (nread <= (ssize_t)sizeof(buf))) { + if (_mi_strnstr(buf, nread, "sv39")) { return 39; } + else if (_mi_strnstr(buf, nread, "sv48")) { return 48; } + else if (_mi_strnstr(buf, nread, "sv57")) { return 57; } + } + } + #endif // riscv + // default + return MI_MAX_VABITS; +} + void _mi_prim_mem_init( mi_os_mem_config_t* config ) { long psize = sysconf(_SC_PAGESIZE); @@ -218,6 +260,7 @@ void _mi_prim_mem_init( mi_os_mem_config_t* config ) config->has_partial_free = true; // mmap can free in parts config->has_virtual_reserve = true; // todo: check if this true for NetBSD? (for anonymous mmap with PROT_NONE) config->has_transparent_huge_pages = unix_detect_thp(); + config->virtual_address_bits = unix_detect_virtual_address_bits(); // disable transparent huge pages for this process? #if (defined(__linux__) || defined(__ANDROID__)) && defined(PR_GET_THP_DISABLE) @@ -333,6 +376,10 @@ static int unix_mmap_fd(void) { #endif } +#if defined(MAP_ALIGNED_SUPER) || defined(MAP_HUGETLB) || defined(MAP_HUGE_1GB) || defined(MAP_HUGE_2MB) || defined(VM_FLAGS_SUPERPAGE_SIZE_2MB) +#define MI_OS_HAS_HUGE_PAGES 1 +#endif + static void* unix_mmap(void* addr, size_t size, size_t try_alignment, int protect_flags, bool large_only, bool allow_large, bool* is_large) { #if !defined(MAP_ANONYMOUS) #define MAP_ANONYMOUS MAP_ANON @@ -350,6 +397,7 @@ static void* unix_mmap(void* addr, size_t size, size_t try_alignment, int protec protect_flags |= PROT_MAX(PROT_READ | PROT_WRITE); // BSD #endif // huge page allocation + #if MI_OS_HAS_HUGE_PAGES if (allow_large && (large_only || (_mi_os_canuse_large_page(size, try_alignment) && mi_option_is_enabled(mi_option_allow_large_os_pages)))) { static _Atomic(size_t) large_page_try_ok; // = 0; size_t try_ok = mi_atomic_load_acquire(&large_page_try_ok); @@ -370,8 +418,8 @@ static void* unix_mmap(void* addr, size_t size, size_t try_alignment, int protec lflags |= MAP_HUGETLB; #endif #ifdef MAP_HUGE_1GB - static bool mi_huge_pages_available = true; - if (large_only && (size % MI_GiB) == 0 && mi_huge_pages_available) { + static _Atomic(size_t) mi_huge_1gib_pages_unavailable; + if (large_only && (size % MI_GiB) == 0 && (mi_atomic_load_relaxed(&mi_huge_1gib_pages_unavailable)==0)) { lflags |= MAP_HUGE_1GB; } else @@ -390,7 +438,7 @@ static void* unix_mmap(void* addr, size_t size, size_t try_alignment, int protec p = unix_mmap_prim_aligned(addr, size, try_alignment, protect_flags, lflags, lfd); #ifdef MAP_HUGE_1GB if (p == NULL && (lflags & MAP_HUGE_1GB) == MAP_HUGE_1GB) { - mi_huge_pages_available = false; // don't try huge 1GiB pages again + mi_atomic_store_relaxed(&mi_huge_1gib_pages_unavailable,1); // don't try huge 1GiB pages again if (large_only) { _mi_warning_message("unable to allocate huge (1GiB) page, trying large (2MiB) pages instead (errno: %i)\n", errno); } @@ -404,7 +452,8 @@ static void* unix_mmap(void* addr, size_t size, size_t try_alignment, int protec } } } - } + } // huge pages + #endif // regular allocation if (p == NULL) { *is_large = false; @@ -492,27 +541,28 @@ int _mi_prim_reuse(void* start, size_t size) { int _mi_prim_decommit(void* start, size_t size, bool* needs_recommit) { int err = 0; - #if defined(__APPLE__) && defined(MADV_FREE_REUSABLE) - // decommit on macOS: use MADV_FREE_REUSABLE as it does immediate rss accounting (issue #1097) - err = unix_madvise(start, size, MADV_FREE_REUSABLE); - if (err) { err = unix_madvise(start, size, MADV_DONTNEED); } - #else - // decommit: use MADV_DONTNEED as it decreases rss immediately (unlike MADV_FREE) - err = unix_madvise(start, size, MADV_DONTNEED); - #endif - #if !MI_DEBUG && MI_SECURE<=2 - *needs_recommit = false; + #if 1 + #if defined(__APPLE__) && defined(MADV_FREE_REUSABLE) + // decommit on macOS: use MADV_FREE_REUSABLE as it does immediate rss accounting (issue #1097) + err = unix_madvise(start, size, MADV_FREE_REUSABLE); + if (err) { err = unix_madvise(start, size, MADV_DONTNEED); } + #else + // decommit: use MADV_DONTNEED as it decreases rss immediately (unlike MADV_FREE) + err = unix_madvise(start, size, MADV_DONTNEED); + #endif + #if !MI_DEBUG && MI_SECURE<=2 + *needs_recommit = false; + #else + *needs_recommit = true; + mprotect(start, size, PROT_NONE); + #endif #else + // decommit: use mmap with MAP_FIXED and PROT_NONE to discard the existing memory (and reduce rss) *needs_recommit = true; - mprotect(start, size, PROT_NONE); + const int fd = unix_mmap_fd(); + void* p = mmap(start, size, PROT_NONE, (MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE), fd, 0); + if (p != start) { err = errno; } #endif - /* - // decommit: use mmap with MAP_FIXED and PROT_NONE to discard the existing memory (and reduce rss) - *needs_recommit = true; - const int fd = unix_mmap_fd(); - void* p = mmap(start, size, PROT_NONE, (MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE), fd, 0); - if (p != start) { err = errno; } - */ return err; } @@ -579,15 +629,15 @@ int _mi_prim_alloc_huge_os_pages(void* hint_addr, size_t size, int numa_node, bo bool is_large = true; *is_zero = true; *addr = unix_mmap(hint_addr, size, MI_ARENA_SLICE_ALIGN, PROT_READ | PROT_WRITE, true, true, &is_large); - if (*addr != NULL && numa_node >= 0 && numa_node < 8*MI_INTPTR_SIZE) { // at most 64 nodes + if (*addr != NULL && numa_node >= 0 && numa_node < (8*MI_INTPTR_SIZE - 1)) { // at most 63 nodes unsigned long numa_mask = (1UL << numa_node); - // TODO: does `mbind` work correctly for huge OS pages? should we + // todo: does `mbind` work correctly for huge OS pages? should we // use `set_mempolicy` before calling mmap instead? // see: long err = mi_prim_mbind(*addr, size, MPOL_PREFERRED, &numa_mask, 8*MI_INTPTR_SIZE, 0); if (err != 0) { err = errno; - _mi_warning_message("failed to bind huge (1GiB) pages to numa node %d (error: %d (0x%x))\n", numa_node, err, err); + _mi_warning_message("failed to bind huge (1GiB) pages to numa node %d (error: %ld (0x%lx))\n", numa_node, err, err); } } return (*addr != NULL ? 0 : errno); @@ -612,9 +662,9 @@ int _mi_prim_alloc_huge_os_pages(void* hint_addr, size_t size, int numa_node, bo size_t _mi_prim_numa_node(void) { #if defined(MI_HAS_SYSCALL_H) && defined(SYS_getcpu) - unsigned long node = 0; - unsigned long ncpu = 0; - long err = syscall(SYS_getcpu, &ncpu, &node, NULL); + unsigned int node = 0; + unsigned int ncpu = 0; + int err = syscall(SYS_getcpu, &ncpu, &node, NULL); if (err != 0) return 0; return node; #else @@ -625,10 +675,15 @@ size_t _mi_prim_numa_node(void) { size_t _mi_prim_numa_node_count(void) { char buf[128]; unsigned node = 0; + size_t skipped = 0; for(node = 0; node < 256; node++) { // enumerate node entries -- todo: it there a more efficient way to do this? (but ensure there is no allocation) _mi_snprintf(buf, 127, "/sys/devices/system/node/node%u", node + 1); - if (mi_prim_access(buf,R_OK) != 0) break; + if (mi_prim_access(buf,R_OK) != 0) { + skipped++; + if (skipped > 4) break; // allow some sparseness of nodes but not more than 4 + } + else { skipped = 0; } // reset skipped count } return (node+1); } @@ -656,7 +711,7 @@ size_t _mi_prim_numa_node_count(void) { #elif defined(__DragonFly__) size_t _mi_prim_numa_node(void) { - // TODO: DragonFly does not seem to provide any userland means to get this information. + // todo: DragonFly does not seem to provide any userland means to get this information. return 0ul; } @@ -686,35 +741,39 @@ size_t _mi_prim_numa_node_count(void) { #include -#if defined(CLOCK_REALTIME) || defined(CLOCK_MONOTONIC) - -mi_msecs_t _mi_prim_clock_now(void) { - struct timespec t; - #ifdef CLOCK_MONOTONIC - clock_gettime(CLOCK_MONOTONIC, &t); +// low resolution timer +static mi_msecs_t mi_prim_clock_now_lowres(void) { + const int64_t ticks = (int64_t)clock(); + #if !defined(CLOCKS_PER_SEC) + return ticks; #else - clock_gettime(CLOCK_REALTIME, &t); + if (CLOCKS_PER_SEC <= 0 || CLOCKS_PER_SEC == 1000) { + return ticks; + } + else if (CLOCKS_PER_SEC > 0 && CLOCKS_PER_SEC < 1000) { + return ticks * (1000 / (mi_msecs_t)CLOCKS_PER_SEC); + } + else { + return ticks / ((mi_msecs_t)CLOCKS_PER_SEC / 1000); + } #endif - return ((mi_msecs_t)t.tv_sec * 1000) + ((mi_msecs_t)t.tv_nsec / 1000000); } -#else - -// low resolution timer mi_msecs_t _mi_prim_clock_now(void) { - #if !defined(CLOCKS_PER_SEC) || (CLOCKS_PER_SEC == 1000) || (CLOCKS_PER_SEC == 0) - return (mi_msecs_t)clock(); - #elif (CLOCKS_PER_SEC < 1000) - return (mi_msecs_t)clock() * (1000 / (mi_msecs_t)CLOCKS_PER_SEC); - #else - return (mi_msecs_t)clock() / ((mi_msecs_t)CLOCKS_PER_SEC / 1000); + #if defined(CLOCK_REALTIME) || defined(CLOCK_MONOTONIC) + #ifdef CLOCK_MONOTONIC + const clockid_t clockid = CLOCK_MONOTONIC; + #else + const clockid_t clockid = CLOCK_REALTIME; + #endif + struct timespec t; + if (clock_gettime(clockid,&t) == 0) { + return ((mi_msecs_t)t.tv_sec * 1000) + ((mi_msecs_t)t.tv_nsec / 1000000L); + } #endif + return mi_prim_clock_now_lowres(); } -#endif - - - //---------------------------------------------------------------- // Process info @@ -738,43 +797,48 @@ static mi_msecs_t timeval_secs(const struct timeval* tv) { } void _mi_prim_process_info(mi_process_info_t* pinfo) -{ +{ struct rusage rusage; - getrusage(RUSAGE_SELF, &rusage); - pinfo->utime = timeval_secs(&rusage.ru_utime); - pinfo->stime = timeval_secs(&rusage.ru_stime); -#if !defined(__HAIKU__) - pinfo->page_faults = rusage.ru_majflt; -#endif -#if defined(__HAIKU__) - // Haiku does not have (yet?) a way to - // get these stats per process - thread_info tid; - area_info mem; - ssize_t c; - get_thread_info(find_thread(0), &tid); - while (get_next_area_info(tid.team, &c, &mem) == B_OK) { - pinfo->peak_rss += mem.ram_size; - } - pinfo->page_faults = 0; -#elif defined(__APPLE__) - pinfo->peak_rss = rusage.ru_maxrss; // macos reports in bytes - #ifdef MACH_TASK_BASIC_INFO - struct mach_task_basic_info info; - mach_msg_type_number_t infoCount = MACH_TASK_BASIC_INFO_COUNT; - if (task_info(mach_task_self(), MACH_TASK_BASIC_INFO, (task_info_t)&info, &infoCount) == KERN_SUCCESS) { - pinfo->current_rss = (size_t)info.resident_size; - } - #else - struct task_basic_info info; - mach_msg_type_number_t infoCount = TASK_BASIC_INFO_COUNT; - if (task_info(mach_task_self(), TASK_BASIC_INFO, (task_info_t)&info, &infoCount) == KERN_SUCCESS) { - pinfo->current_rss = (size_t)info.resident_size; + if (getrusage(RUSAGE_SELF, &rusage) == 0) { + pinfo->utime = timeval_secs(&rusage.ru_utime); + pinfo->stime = timeval_secs(&rusage.ru_stime); + #if !defined(__HAIKU__) + pinfo->page_faults = rusage.ru_majflt; + #endif + #if defined(__APPLE__) + pinfo->peak_rss = rusage.ru_maxrss; // macos reports in bytes + #else + pinfo->peak_rss = rusage.ru_maxrss * 1024; // Linux/BSD report in KiB + #endif } + + #if defined(__HAIKU__) + // Haiku does not have (yet?) a way to + // get these stats per process + thread_info tid; + if (get_thread_info(find_thread(0), &tid) == B_OK) { + area_info mem; + ssize_t c; + while (get_next_area_info(tid.team, &c, &mem) == B_OK) { + pinfo->peak_rss += mem.ram_size; + } + } + pinfo->page_faults = 0; + #elif defined(__APPLE__) + #ifdef MACH_TASK_BASIC_INFO + struct mach_task_basic_info info; + mach_msg_type_number_t infoCount = MACH_TASK_BASIC_INFO_COUNT; + if (task_info(mach_task_self(), MACH_TASK_BASIC_INFO, (task_info_t)&info, &infoCount) == KERN_SUCCESS) { + pinfo->current_rss = (size_t)info.resident_size; + } + #else + struct task_basic_info info; + mach_msg_type_number_t infoCount = TASK_BASIC_INFO_COUNT; + if (task_info(mach_task_self(), TASK_BASIC_INFO, (task_info_t)&info, &infoCount) == KERN_SUCCESS) { + pinfo->current_rss = (size_t)info.resident_size; + } + #endif #endif -#else - pinfo->peak_rss = rusage.ru_maxrss * 1024; // Linux/BSD report in KiB -#endif // use defaults for commit } @@ -821,28 +885,28 @@ static char** mi_get_environ(void) { return environ; } #endif -bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { - if (name==NULL) return false; +int _mi_prim_getenv(const char* name, char* result, size_t result_size) { + if (name==NULL) return -1; const size_t len = _mi_strlen(name); - if (len == 0) return false; + if (len == 0) return -1; char** env = mi_get_environ(); - if (env == NULL) return false; + if (env == NULL) return -1; // compare up to 10000 entries for (int i = 0; i < 10000 && env[i] != NULL; i++) { const char* s = env[i]; if (_mi_strnicmp(name, s, len) == 0 && s[len] == '=') { // case insensitive // found it - _mi_strlcpy(result, s + len + 1, result_size); - return true; + if (!_mi_strlcpy(result, s + len + 1, result_size)) return -1; + return 1; // success } } - return false; + return 0; // not found } #else // fallback: use standard C `getenv` but this cannot be used while initializing the C runtime -bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { +int _mi_prim_getenv(const char* name, char* result, size_t result_size) { // cannot call getenv() when still initializing the C runtime. - if (_mi_preloading()) return false; + if (_mi_preloading()) return -1; // error, try again later const char* s = getenv(name); if (s == NULL) { // we check the upper case name too. @@ -854,9 +918,9 @@ bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { buf[len] = 0; s = getenv(buf); } - if (s == NULL || _mi_strnlen(s,result_size) >= result_size) return false; - _mi_strlcpy(result, s, result_size); - return true; + if (s == NULL || _mi_strnlen(s,result_size) >= result_size) return 0; // not found + if (!_mi_strlcpy(result, s, result_size)) return -1; + return 1; // success } #endif // !MI_USE_ENVIRON @@ -917,7 +981,10 @@ bool _mi_prim_random_buf(void* buf, size_t buf_len) { size_t count = 0; while(count < buf_len) { ssize_t ret = mi_prim_read(fd, (char*)buf + count, buf_len - count); - if (ret<=0) { + if (ret==0) { + break; + } + else if (ret<0) { if (errno!=EAGAIN && errno!=EINTR) break; } else { @@ -945,7 +1012,7 @@ bool _mi_prim_random_buf(void* buf, size_t buf_len) { // use pthread local storage keys to detect thread ending // (and used with MI_TLS_PTHREADS for the default theap) -pthread_key_t _mi_heap_default_key = (pthread_key_t)(-1); +pthread_key_t _mi_heap_default_key = MI_PTHREAD_KEY_INVALID; static void mi_pthread_done(void* value) { if (value!=NULL) { @@ -954,22 +1021,25 @@ static void mi_pthread_done(void* value) { } void _mi_prim_thread_init_auto_done(void) { - mi_assert_internal(_mi_heap_default_key == (pthread_key_t)(-1)); + mi_assert_internal(_mi_heap_default_key == MI_PTHREAD_KEY_INVALID); pthread_key_create(&_mi_heap_default_key, &mi_pthread_done); } void _mi_prim_thread_done_auto_done(void) { - if (_mi_heap_default_key != (pthread_key_t)(-1)) { // do not leak the key, see issue #809 - pthread_key_delete(_mi_heap_default_key); + pthread_key_t key = _mi_heap_default_key; + if (key != MI_PTHREAD_KEY_INVALID) { // do not leak the key, see issue #809 + _mi_heap_default_key = MI_PTHREAD_KEY_INVALID; + pthread_key_delete(key); } } void _mi_prim_thread_associate_default_theap(mi_theap_t* theap) { - if (_mi_heap_default_key != (pthread_key_t)(-1)) { // can happen during recursive invocation on freeBSD + if (_mi_heap_default_key != MI_PTHREAD_KEY_INVALID) { // can happen during recursive invocation on freeBSD pthread_setspecific(_mi_heap_default_key, theap); } } + #else void _mi_prim_thread_init_auto_done(void) { diff --git a/vendored/mimalloc/src/prim/wasi/prim.c b/vendored/mimalloc/src/prim/wasi/prim.c index 790daff02f..2d69b7245f 100644 --- a/vendored/mimalloc/src/prim/wasi/prim.c +++ b/vendored/mimalloc/src/prim/wasi/prim.c @@ -185,33 +185,39 @@ size_t _mi_prim_numa_node_count(void) { #include -#if defined(CLOCK_REALTIME) || defined(CLOCK_MONOTONIC) - -mi_msecs_t _mi_prim_clock_now(void) { - struct timespec t; - #ifdef CLOCK_MONOTONIC - clock_gettime(CLOCK_MONOTONIC, &t); +// low resolution timer +static mi_msecs_t mi_prim_clock_now_lowres(void) { + const int64_t ticks = (int64_t)clock(); + #if !defined(CLOCKS_PER_SEC) + return ticks; #else - clock_gettime(CLOCK_REALTIME, &t); + if (CLOCKS_PER_SEC <= 0 || CLOCKS_PER_SEC == 1000) { + return ticks; + } + else if (CLOCKS_PER_SEC > 0 && CLOCKS_PER_SEC < 1000) { + return ticks * (1000 / (mi_msecs_t)CLOCKS_PER_SEC); + } + else { + return ticks / ((mi_msecs_t)CLOCKS_PER_SEC / 1000); + } #endif - return ((mi_msecs_t)t.tv_sec * 1000) + ((mi_msecs_t)t.tv_nsec / 1000000); } -#else - -// low resolution timer mi_msecs_t _mi_prim_clock_now(void) { - #if !defined(CLOCKS_PER_SEC) || (CLOCKS_PER_SEC == 1000) || (CLOCKS_PER_SEC == 0) - return (mi_msecs_t)clock(); - #elif (CLOCKS_PER_SEC < 1000) - return (mi_msecs_t)clock() * (1000 / (mi_msecs_t)CLOCKS_PER_SEC); - #else - return (mi_msecs_t)clock() / ((mi_msecs_t)CLOCKS_PER_SEC / 1000); - #endif + #if defined(CLOCK_REALTIME) || defined(CLOCK_MONOTONIC) + #ifdef CLOCK_MONOTONIC + const clockid_t clockid = CLOCK_MONOTONIC; + #else + const clockid_t clockid = CLOCK_REALTIME; + #endif + struct timespec t; + if (clock_gettime(clockid,&t) == 0) { + return ((mi_msecs_t)t.tv_sec * 1000) + ((mi_msecs_t)t.tv_nsec / 1000000L); + } + #endif + return mi_prim_clock_now_lowres(); } -#endif - //---------------------------------------------------------------- // Process info @@ -237,9 +243,9 @@ void _mi_prim_out_stderr( const char* msg ) { // Environment //---------------------------------------------------------------- -bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { +int _mi_prim_getenv(const char* name, char* result, size_t result_size) { // cannot call getenv() when still initializing the C runtime. - if (_mi_preloading()) return false; + if (_mi_preloading()) return -1; // error, try again later const char* s = getenv(name); if (s == NULL) { // we check the upper case name too. @@ -251,9 +257,9 @@ bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { buf[len] = 0; s = getenv(buf); } - if (s == NULL || _mi_strnlen(s,result_size) >= result_size) return false; + if (s == NULL || _mi_strnlen(s,result_size) >= result_size) return 0; // not found _mi_strlcpy(result, s, result_size); - return true; + return 1; // found } diff --git a/vendored/mimalloc/src/prim/windows/prim.c b/vendored/mimalloc/src/prim/windows/prim.c index 51ad3104fe..7fcdf0693b 100644 --- a/vendored/mimalloc/src/prim/windows/prim.c +++ b/vendored/mimalloc/src/prim/windows/prim.c @@ -13,9 +13,9 @@ terms of the MIT license. A copy of the license can be found in the file #include // fputs, stderr #include // atexit -// xbox has no console IO -#if !defined(WINAPI_FAMILY_PARTITION) || WINAPI_FAMILY_PARTITION(WINAPI_PARTITION_APP | WINAPI_PARTITION_SYSTEM) -#define MI_HAS_CONSOLE_IO +// xbox has no console IO and cannot use LoadLibrary or GetModuleHandle +#if !defined(WINAPI_FAMILY_PARTITION) || WINAPI_FAMILY_PARTITION(WINAPI_PARTITION_DESKTOP | WINAPI_PARTITION_SYSTEM) +#define MI_WIN_DESKTOP 1 #endif //--------------------------------------------- @@ -87,6 +87,34 @@ typedef BOOL (__stdcall *PGetPhysicallyInstalledSystemMemory)( PULONGLONG TotalM typedef BOOL (__stdcall* PGetVersionExW)(LPOSVERSIONINFOW lpVersionInformation); +// Load a library +static HMODULE mi_win_loadlibrary(const TCHAR* library) { + #if MI_WIN_DESKTOP + return LoadLibrary(library); + #else + return LoadPackagedLibrary(library, 0); + #endif +} + +// Get a library handle (and possibly load it) +static HMODULE mi_win_getlibrary(const TCHAR* library, bool* should_free) { + #if MI_WIN_DESKTOP + // avoid calling LoadLibrary for "kernel32", "ntdll", and "kernelbase" (also to avoid hitting the loader lock) + HMODULE mod = GetModuleHandle(library); + if (mod!=NULL) { + *should_free = false; + return mod; + } + #endif + *should_free = true; + return mi_win_loadlibrary(library); +} + +static void mi_win_freelibrary(HMODULE mod, bool should_free) { + if (should_free) { + FreeLibrary(mod); + } +} //--------------------------------------------- // Enable large page support dynamically (if possible) @@ -103,15 +131,17 @@ static bool win_enable_large_os_pages_once(size_t* large_page_size) unsigned long err = 0; HANDLE token = NULL; BOOL ok = OpenProcessToken(GetCurrentProcess(), TOKEN_ADJUST_PRIVILEGES | TOKEN_QUERY, &token); + err = GetLastError(); if (ok) { TOKEN_PRIVILEGES tp; ok = LookupPrivilegeValue(NULL, TEXT("SeLockMemoryPrivilege"), &tp.Privileges[0].Luid); + err = GetLastError(); if (ok) { tp.PrivilegeCount = 1; tp.Privileges[0].Attributes = SE_PRIVILEGE_ENABLED; ok = AdjustTokenPrivileges(token, FALSE, &tp, 0, (PTOKEN_PRIVILEGES)NULL, 0); + err = GetLastError(); if (ok) { - err = GetLastError(); ok = (err == ERROR_SUCCESS); if (ok && large_page_size != NULL && pGetLargePageMinimum != NULL) { *large_page_size = (*pGetLargePageMinimum)(); @@ -121,17 +151,19 @@ static bool win_enable_large_os_pages_once(size_t* large_page_size) CloseHandle(token); } if (!ok) { - if (err == 0) err = GetLastError(); + if (err == 0) { err = GetLastError(); } _mi_warning_message("cannot enable large OS page support, error %lu\n", err); } return (ok!=0); } static bool win_enable_large_os_pages(size_t* large_page_size) { + static size_t win_large_page_size = 0; mi_atomic_do_once { - win_enable_large_os_pages_once(large_page_size); + win_enable_large_os_pages_once(&win_large_page_size); } - return (_mi_os_large_page_size() > 0); + if (large_page_size != NULL) { *large_page_size = win_large_page_size; } + return (win_large_page_size > 0); } @@ -162,21 +194,22 @@ void _mi_prim_mem_init( mi_os_mem_config_t* config ) } // get the VirtualAlloc2 function - HINSTANCE hDll = LoadLibrary(TEXT("kernelbase.dll")); + bool hDllFree; + HINSTANCE hDll = mi_win_getlibrary(TEXT("kernelbase.dll"), &hDllFree); if (hDll != NULL) { // use VirtualAlloc2FromApp if possible as it is available to Windows store apps pVirtualAlloc2 = (PVirtualAlloc2)(void (*)(void))GetProcAddress(hDll, "VirtualAlloc2FromApp"); if (pVirtualAlloc2==NULL) pVirtualAlloc2 = (PVirtualAlloc2)(void (*)(void))GetProcAddress(hDll, "VirtualAlloc2"); - FreeLibrary(hDll); + mi_win_freelibrary(hDll, hDllFree); } // NtAllocateVirtualMemoryEx is used for huge page allocation - hDll = LoadLibrary(TEXT("ntdll.dll")); + hDll = mi_win_getlibrary(TEXT("ntdll.dll"), &hDllFree); if (hDll != NULL) { pNtAllocateVirtualMemoryEx = (PNtAllocateVirtualMemoryEx)(void (*)(void))GetProcAddress(hDll, "NtAllocateVirtualMemoryEx"); - FreeLibrary(hDll); + mi_win_freelibrary(hDll, hDllFree); } // Try to use Win7+ numa API - hDll = LoadLibrary(TEXT("kernel32.dll")); + hDll = mi_win_getlibrary(TEXT("kernel32.dll"), &hDllFree); if (hDll != NULL) { pGetCurrentProcessorNumberEx = (PGetCurrentProcessorNumberEx)(void (*)(void))GetProcAddress(hDll, "GetCurrentProcessorNumberEx"); pGetNumaProcessorNodeEx = (PGetNumaProcessorNodeEx)(void (*)(void))GetProcAddress(hDll, "GetNumaProcessorNodeEx"); @@ -205,7 +238,7 @@ void _mi_prim_mem_init( mi_os_mem_config_t* config ) win_minor_version = version.dwMinorVersion; } } - FreeLibrary(hDll); + mi_win_freelibrary(hDll, hDllFree); } // Enable large/huge OS page support? if (mi_option_is_enabled(mi_option_allow_large_os_pages) || mi_option_is_enabled(mi_option_reserve_huge_os_pages)) { @@ -228,8 +261,9 @@ int _mi_prim_free(void* addr, size_t size ) { // the memory region returned by VirtualAlloc; in that case we need to free using // the start of the region. MEMORY_BASIC_INFORMATION info; _mi_memzero_var(info); - VirtualQuery(addr, &info, sizeof(info)); - if (info.AllocationBase < addr && ((uint8_t*)addr - (uint8_t*)info.AllocationBase) < (ptrdiff_t)(4*MI_MiB)) { + err = (VirtualQuery(addr, &info, sizeof(info)) == 0); + if (err) { errcode = GetLastError(); } + if (!err && info.AllocationBase < addr && ((uint8_t*)addr - (uint8_t*)info.AllocationBase) < (ptrdiff_t)(4*MI_MiB)) { errcode = 0; err = (VirtualFree(info.AllocationBase, 0, MEM_RELEASE) == 0); if (err) { errcode = GetLastError(); } @@ -525,8 +559,9 @@ static mi_msecs_t mi_to_msecs(LARGE_INTEGER t) { static LARGE_INTEGER mfreq; // = 0 if (mfreq.QuadPart == 0LL) { LARGE_INTEGER f; - QueryPerformanceFrequency(&f); - mfreq.QuadPart = f.QuadPart/1000LL; + if (QueryPerformanceFrequency(&f)) { + mfreq.QuadPart = f.QuadPart/1000LL; + } if (mfreq.QuadPart == 0) mfreq.QuadPart = 1; } return (mi_msecs_t)(t.QuadPart / mfreq.QuadPart); @@ -534,8 +569,12 @@ static mi_msecs_t mi_to_msecs(LARGE_INTEGER t) { mi_msecs_t _mi_prim_clock_now(void) { LARGE_INTEGER t; - QueryPerformanceCounter(&t); - return mi_to_msecs(t); + if (QueryPerformanceCounter(&t)) { + return mi_to_msecs(t); + } + else { + return 0; + } } @@ -562,29 +601,31 @@ void _mi_prim_process_info(mi_process_info_t* pinfo) FILETIME ut; FILETIME st; FILETIME et; - GetProcessTimes(GetCurrentProcess(), &ct, &et, &st, &ut); - pinfo->utime = filetime_msecs(&ut); - pinfo->stime = filetime_msecs(&st); + if (GetProcessTimes(GetCurrentProcess(), &ct, &et, &st, &ut)) { + pinfo->utime = filetime_msecs(&ut); + pinfo->stime = filetime_msecs(&st); + } // load psapi on demand - mi_atomic_do_once { - HINSTANCE hDll = LoadLibrary(TEXT("psapi.dll")); + mi_atomic_do_once{ + HINSTANCE hDll = mi_win_loadlibrary(TEXT("psapi.dll")); if (hDll != NULL) { pGetProcessMemoryInfo = (PGetProcessMemoryInfo)(void (*)(void))GetProcAddress(hDll, "GetProcessMemoryInfo"); - // FreeLibrary(hDll); // don't free + // mi_win_freelibrary(hDll, true); // don't free } } - // get process info - PROCESS_MEMORY_COUNTERS info; _mi_memzero_var(info); + // get process info if (pGetProcessMemoryInfo != NULL) { - pGetProcessMemoryInfo(GetCurrentProcess(), &info, sizeof(info)); + PROCESS_MEMORY_COUNTERS info; _mi_memzero_var(info); + if (pGetProcessMemoryInfo(GetCurrentProcess(), &info, sizeof(info))) { + pinfo->current_rss = (size_t)info.WorkingSetSize; + pinfo->peak_rss = (size_t)info.PeakWorkingSetSize; + pinfo->current_commit = (size_t)info.PagefileUsage; + pinfo->peak_commit = (size_t)info.PeakPagefileUsage; + pinfo->page_faults = (size_t)info.PageFaultCount; + } } - pinfo->current_rss = (size_t)info.WorkingSetSize; - pinfo->peak_rss = (size_t)info.PeakWorkingSetSize; - pinfo->current_commit = (size_t)info.PagefileUsage; - pinfo->peak_commit = (size_t)info.PeakPagefileUsage; - pinfo->page_faults = (size_t)info.PageFaultCount; } //---------------------------------------------------------------- @@ -600,21 +641,21 @@ void _mi_prim_out_stderr( const char* msg ) static HANDLE hcon = INVALID_HANDLE_VALUE; static bool hconIsConsole = false; if (hcon == INVALID_HANDLE_VALUE) { - hcon = GetStdHandle(STD_ERROR_HANDLE); - #ifdef MI_HAS_CONSOLE_IO + hcon = GetStdHandle(STD_ERROR_HANDLE); // returns NULL on error + #if MI_WIN_DESKTOP CONSOLE_SCREEN_BUFFER_INFO sbi; - hconIsConsole = ((hcon != INVALID_HANDLE_VALUE) && GetConsoleScreenBufferInfo(hcon, &sbi)); + hconIsConsole = ((hcon != NULL && hcon != INVALID_HANDLE_VALUE) && GetConsoleScreenBufferInfo(hcon, &sbi)); #endif } const size_t len = _mi_strlen(msg); if (len > 0 && len < UINT32_MAX) { DWORD written = 0; if (hconIsConsole) { - #ifdef MI_HAS_CONSOLE_IO + #if MI_WIN_DESKTOP WriteConsoleA(hcon, msg, (DWORD)len, &written, NULL); #endif } - else if (hcon != INVALID_HANDLE_VALUE) { + else if (hcon != NULL && hcon != INVALID_HANDLE_VALUE) { // use direct write if stderr was redirected WriteFile(hcon, msg, (DWORD)len, &written, NULL); } @@ -635,10 +676,10 @@ void _mi_prim_out_stderr( const char* msg ) // reliably even when this is invoked before the C runtime is initialized. // i.e. when `_mi_preloading() == true`. // Note: on windows, environment names are not case sensitive. -bool _mi_prim_getenv(const char* name, char* result, size_t result_size) { +int _mi_prim_getenv(const char* name, char* result, size_t result_size) { result[0] = 0; const size_t len = GetEnvironmentVariableA(name, result, (DWORD)result_size); - return (len > 0 && len < result_size); + return (len < result_size ? (len > 0 ? 1 /* success */ : 0 /* not found */) : -1 /* error */); } @@ -673,10 +714,10 @@ bool _mi_prim_random_buf(void* buf, size_t buf_len) { mi_assert(buf_len <= ULONG_MAX); if (buf_len > ULONG_MAX) return false; mi_atomic_do_once { - HINSTANCE hDll = LoadLibrary(TEXT("bcrypt.dll")); + HINSTANCE hDll = mi_win_loadlibrary(TEXT("bcrypt.dll")); if (hDll != NULL) { pBCryptGenRandom = (PBCryptGenRandom)(void (*)(void))GetProcAddress(hDll, "BCryptGenRandom"); - // FreeLibrary(hDll); // don't free + // mi_win_freelibrary(hDll); // don't free } } if (pBCryptGenRandom == NULL) return false; @@ -737,10 +778,10 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { both static and dynamic linkage (`MI_WIN_INIT_USE_CRT_TLS`). ------------------------------------------------------------------------- */ #if !defined(MI_WIN_INIT_USE_CRT_TLS) && !defined(MI_WIN_INIT_USE_RAW_DLLMAIN) && !defined(MI_WIN_INIT_USE_TLS_DLLMAIN) && !defined(MI_WIN_INIT_USE_FLS) - #if !defined(__INTEL_LLVM_COMPILER) && !defined(__INTEL_COMPILER) - #define MI_WIN_INIT_USE_CRT_TLS 1 + #if defined(__INTEL_LLVM_COMPILER) || defined(__INTEL_COMPILER) + #define MI_WIN_INIT_USE_TLS_DLLMAIN 1 /* needed for Intel ICX, see issue #1268 */ #else - #define MI_WIN_INIT_USE_TLS_DLLMAIN 1 /* default for Intel ICX, see issue #1268 */ + #define MI_WIN_INIT_USE_CRT_TLS 1 /* default */ #endif #endif @@ -818,10 +859,12 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { #endif typedef int (mi_cdecl* mi_crt_callback_t)(void); - #if defined(_WIN64) + + #if defined(_WIN64) && defined(_MSC_VER) // 64-bit #pragma comment(linker, "/INCLUDE:_tls_used") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_post") + #pragma comment(linker, "/INCLUDE:_mi_crt_callback_init") #pragma const_seg(".CRT$XLB") extern const PIMAGE_TLS_CALLBACK _mi_tls_callback_pre[]; const PIMAGE_TLS_CALLBACK _mi_tls_callback_pre[] = { &mi_tls_attach }; @@ -834,10 +877,11 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { extern const mi_crt_callback_t _mi_crt_callback_init[]; const mi_crt_callback_t _mi_crt_callback_init[] = { &mi_crt_init }; #pragma const_seg() - #else + #elif defined(_MSC_VER) // 32-bit #pragma comment(linker, "/INCLUDE:__tls_used") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_post") + #pragma comment(linker, "/INCLUDE:__mi_crt_callback_init") #pragma data_seg(".CRT$XLB") PIMAGE_TLS_CALLBACK _mi_tls_callback_pre[] = { &mi_tls_attach }; #pragma data_seg() @@ -847,6 +891,12 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { #pragma data_seg(".CRT$XIB") mi_crt_callback_t _mi_crt_callback_init[] = { &mi_crt_init }; #pragma data_seg() + #elif defined(__GCC__) // mingw + extern const IMAGE_TLS_DIRECTORY _tls_used; + __attribute__((used)) static const void* const mi_tls_used_ref = &_tls_used; // pull in the CRT tls + __attribute__((used, section(".CRT$XLB"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_pre = &mi_tls_attach; + __attribute__((used, section(".CRT$XLY"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_post = &mi_tls_detach; + __attribute__((used, section(".CRT$XIB"))) mi_crt_callback_t _mi_crt_callback_init = &mi_crt_init; #endif #if defined(__cplusplus) @@ -916,7 +966,7 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { extern "C" { #endif - #if defined(_WIN64) + #if defined(_WIN64) && defined(_MSC_VER) // 64-bit #pragma comment(linker, "/INCLUDE:_tls_used") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_post") @@ -928,7 +978,7 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { extern const PIMAGE_TLS_CALLBACK _mi_tls_callback_post[]; const PIMAGE_TLS_CALLBACK _mi_tls_callback_post[] = { &mi_tls_detach }; #pragma const_seg() - #else + #elif defined(_MSC_VER) // 32-bit #pragma comment(linker, "/INCLUDE:__tls_used") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_post") @@ -938,6 +988,11 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { #pragma data_seg(".CRT$XLY") PIMAGE_TLS_CALLBACK _mi_tls_callback_post[] = { &mi_tls_detach }; #pragma data_seg() + #elif defined(__GCC__) // mingw + extern const IMAGE_TLS_DIRECTORY _tls_used; + __attribute__((used)) static const void* const mi_tls_used_ref = &_tls_used; // pull in the CRT tls + __attribute__((used, section(".CRT$XLB"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_pre = &mi_tls_attach; + __attribute__((used, section(".CRT$XLY"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_post = &mi_tls_detach; #endif #if defined(__cplusplus) @@ -985,7 +1040,7 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { extern "C" { #endif - #if defined(_WIN64) + #if defined(_WIN64) && defined(_MSC_VER) // 64-bit #pragma comment(linker, "/INCLUDE:_tls_used") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:_mi_tls_callback_post") @@ -997,16 +1052,21 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { extern const PIMAGE_TLS_CALLBACK _mi_tls_callback_post[]; const PIMAGE_TLS_CALLBACK _mi_tls_callback_post[] = { &mi_win_main_detach }; #pragma const_seg() - #else + #elif defined(_MSC_VER) // 32-bit #pragma comment(linker, "/INCLUDE:__tls_used") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_pre") #pragma comment(linker, "/INCLUDE:__mi_tls_callback_post") #pragma data_seg(".CRT$XLB") PIMAGE_TLS_CALLBACK _mi_tls_callback_pre[] = { &mi_win_main_attach }; #pragma data_seg() - #pragma data_seg(".CRT$XIY") + #pragma data_seg(".CRT$XLY") PIMAGE_TLS_CALLBACK _mi_tls_callback_post[] = { &mi_win_main_detach }; #pragma data_seg() + #elif defined(__GCC__) // mingw + extern const IMAGE_TLS_DIRECTORY _tls_used; + __attribute__((used)) static const void* const mi_tls_used_ref = &_tls_used; // pull in the CRT tls + __attribute__((used, section(".CRT$XLB"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_pre = &mi_tls_attach; + __attribute__((used, section(".CRT$XLY"))) PIMAGE_TLS_CALLBACK _mi_tls_callback_post = &mi_tls_detach; #endif #if defined(__cplusplus) @@ -1027,21 +1087,31 @@ static void NTAPI mi_win_main(PVOID module, DWORD reason, LPVOID reserved) { // See #define MI_PRIM_HAS_PROCESS_ATTACH 1 - static int mi_process_attach(void) { + static int mi_cdecl mi_crt_init(void) { mi_win_main(NULL,DLL_PROCESS_ATTACH,NULL); atexit(&_mi_auto_process_done); return 0; } - typedef int(*mi_crt_callback_t)(void); - #if defined(_WIN64) - #pragma comment(linker, "/INCLUDE:_mi_tls_callback") - #pragma section(".CRT$XIU", long, read) - #else - #pragma comment(linker, "/INCLUDE:__mi_tls_callback") + + #if defined(__cplusplus) + extern "C" { + #endif + typedef int (mi_cdecl* mi_crt_callback_t)(void); + #if defined(_WIN64) // 64-bit + #pragma comment(linker, "/INCLUDE:_mi_crt_callback_init") + #pragma const_seg(".CRT$XIU") + extern const mi_crt_callback_t _mi_crt_callback_init[]; + const mi_crt_callback_t _mi_crt_callback_init[] = { &mi_crt_init }; + #pragma const_seg() + #else // 32-bit + #pragma comment(linker, "/INCLUDE:__mi_crt_callback_init") + #pragma data_seg(".CRT$XIU") + mi_crt_callback_t _mi_crt_callback_init[] = { &mi_crt_init }; + #pragma data_seg() + #endif + #if defined(__cplusplus) + } #endif - #pragma data_seg(".CRT$XIU") - mi_decl_externc mi_crt_callback_t _mi_tls_callback[] = { &mi_process_attach }; - #pragma data_seg() #endif // use the fiber api for calling `_mi_thread_done`. diff --git a/vendored/mimalloc/src/random.c b/vendored/mimalloc/src/random.c index 75b5cfbb0b..464627acfb 100644 --- a/vendored/mimalloc/src/random.c +++ b/vendored/mimalloc/src/random.c @@ -134,8 +134,9 @@ static bool mi_random_is_initialized(mi_random_ctx_t* ctx) { void _mi_random_split(mi_random_ctx_t* ctx, mi_random_ctx_t* ctx_new) { mi_assert_internal(mi_random_is_initialized(ctx)); - mi_assert_internal(ctx != ctx_new); - chacha_split(ctx, (uintptr_t)ctx_new /*nonce*/, ctx_new); + mi_assert_internal(ctx != ctx_new); + const uintptr_t nonce_rnd = _mi_random_next(ctx); + chacha_split(ctx, (uintptr_t)ctx_new ^ nonce_rnd /*nonce*/, ctx_new); } uintptr_t _mi_random_next(mi_random_ctx_t* ctx) { @@ -193,6 +194,7 @@ static void mi_random_init_ex(mi_random_ctx_t* ctx, bool use_weak) { ctx->weak = false; } chacha_init(ctx, key, (uintptr_t)ctx /*nonce*/ ); + _mi_memzero(key, sizeof(key)); } void _mi_random_init(mi_random_ctx_t* ctx) { diff --git a/vendored/mimalloc/src/stats.c b/vendored/mimalloc/src/stats.c index 7ca18e2385..f2ccb4f874 100644 --- a/vendored/mimalloc/src/stats.c +++ b/vendored/mimalloc/src/stats.c @@ -1,5 +1,5 @@ /* ---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -8,7 +8,8 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc-stats.h" #include "mimalloc/internal.h" #include "mimalloc/atomic.h" -#include "mimalloc/prim.h" +#include "mimalloc/prim.h" // _mi_prim_clock_now, mi_process_info_t +#include "mimalloc/prim-tls.h" #include // memset @@ -96,15 +97,17 @@ void __mi_stat_adjust_decrease(mi_stat_count_t* stat, size_t amount) { static void mi_stat_count_add_mt(mi_stat_count_t* stat, const mi_stat_count_t* src) { if (stat==src) return; mi_atomic_void_addi64_relaxed(&stat->total, &src->total); - const int64_t prev_current = mi_atomic_addi64_relaxed(&stat->current, src->current); + const int64_t src_peak = mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)&src->peak); + const int64_t src_current = mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)&src->current); + const int64_t prev_current = mi_atomic_addi64_relaxed(&stat->current, src_current); // Global current plus thread peak approximates new global peak // note: peak scores do really not work across threads. // we used to just add them together but that often overestimates in practice. // similarly, max does not seem to work well. The current approach // by Artem Kharytoniuk (@artem-lunarg) seems to work better, see PR#1112 - // for a longer description. - mi_atomic_maxi64_relaxed(&stat->peak, prev_current + src->peak); + // for a longer description. + mi_atomic_maxi64_relaxed(&stat->peak, prev_current + src_peak); } static void mi_stat_counter_add_mt(mi_stat_counter_t* stat, const mi_stat_counter_t* src) { @@ -130,6 +133,9 @@ static void mi_stats_add(mi_stats_t* stats, const mi_stats_t* src) { for (size_t i = 0; i <= MI_BIN_HUGE; i++) { mi_stat_count_add_mt(&stats->page_bins[i], &src->page_bins[i]); } + for (size_t i = 0; i < MI_CBIN_COUNT; i++) { + mi_stat_count_add_mt(&stats->chunk_bins[i], &src->chunk_bins[i]); + } } #undef MI_STAT_COUNT @@ -142,7 +148,7 @@ static void mi_stats_add(mi_stats_t* stats, const mi_stats_t* src) { // unit > 0 : size in binary bytes // unit == 0: count as decimal // unit < 0 : count in binary -static void mi_printf_amount(int64_t n, int64_t unit, mi_output_fun* out, void* arg, const char* fmt) { +static void mi_printf_amount(int64_t n, int64_t unit, mi_output_fun* out, void* arg, bool limitwidth) { char buf[32]; _mi_memzero_var(buf); int len = 32; const char* suffix = (unit <= 0 ? " " : "B"); @@ -167,12 +173,17 @@ static void mi_printf_amount(int64_t n, int64_t unit, mi_output_fun* out, void* _mi_snprintf(unitdesc, 8, "%s%s%s", magnitude, (base==1024 ? "i" : ""), suffix); _mi_snprintf(buf, len, "%ld.%ld %-3s", whole, (frac1 < 0 ? -frac1 : frac1), unitdesc); } - _mi_fprintf(out, arg, (fmt==NULL ? "%12s" : fmt), buf); + if (limitwidth) { + _mi_fprintf(out, arg, "%12s", buf); + } + else { + _mi_fprintf(out, arg, "%s", buf); + } } static void mi_print_amount(int64_t n, int64_t unit, mi_output_fun* out, void* arg) { - mi_printf_amount(n,unit,out,arg,NULL); + mi_printf_amount(n,unit,out,arg,true); } static void mi_print_count(int64_t n, int64_t unit, mi_output_fun* out, void* arg) { @@ -206,7 +217,7 @@ static void mi_stat_print_ex(const mi_stat_count_t* stat, const char* msg, int64 } if (stat->current != 0) { _mi_fprintf(out, arg, " "); - _mi_fprintf(out, arg, (notok == NULL ? "not all freed" : notok)); + _mi_fprintf(out, arg, "%s", (notok == NULL ? "not all freed" : notok)); _mi_fprintf(out, arg, "\n"); } else { @@ -331,10 +342,10 @@ mi_decl_export void mi_process_info_print_out(mi_output_fun* out, void* arg) mi_ _mi_fprintf(out, arg, " %-10s: %5zu.%03zu s\n", "elapsed", elapsed/1000, elapsed%1000); _mi_fprintf(out, arg, " %-10s: user: %zu.%03zu s, system: %zu.%03zu s, faults: %zu, peak rss: ", "process", user_time/1000, user_time%1000, sys_time/1000, sys_time%1000, page_faults); - mi_printf_amount((int64_t)peak_rss, 1, out, arg, "%s"); + mi_printf_amount((int64_t)peak_rss, 1, out, arg, false); if (peak_commit > 0) { _mi_fprintf(out, arg, ", peak commit: "); - mi_printf_amount((int64_t)peak_commit, 1, out, arg, "%s"); + mi_printf_amount((int64_t)peak_commit, 1, out, arg, false); } _mi_fprintf(out, arg, "\n"); } @@ -432,7 +443,7 @@ void _mi_stats_merge_into(mi_stats_t* to, mi_stats_t* from) { mi_assert_internal(to != NULL && from != NULL); if (to == from) return; mi_stats_add(to, from); - _mi_memzero(from, sizeof(mi_stats_t)); + mi_stats_init(from); // zero field and keep the header } static const mi_stats_t* mi_stats_merge_theap_to_heap(mi_theap_t* theap) mi_attr_noexcept { @@ -452,8 +463,9 @@ static const mi_stats_t* mi_heap_get_stats(mi_heap_t* heap) { // deprecated void mi_stats_reset(void) mi_attr_noexcept { if (!mi_theap_is_initialized(_mi_theap_default())) return; - mi_heap_get_stats(mi_heap_main()); - mi_heap_stats_merge_to_subproc(mi_heap_main()); + mi_heap_t* heap_main = mi_heap_main(); + mi_heap_get_stats(heap_main); + mi_heap_stats_merge_to_subproc(heap_main); } @@ -548,8 +560,10 @@ mi_decl_export void mi_process_info(size_t* elapsed_msecs, size_t* user_msecs, s pinfo.elapsed = _mi_clock_end(mi_process_start); { const mi_subproc_t* subproc = _mi_subproc_main(); if (subproc!=NULL) { - pinfo.current_commit = (size_t)(mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)(&subproc->stats.committed.current))); - pinfo.peak_commit = (size_t)(mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)(&subproc->stats.committed.peak))); + const int64_t current = mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)(&subproc->stats.committed.current)); + const int64_t peak = mi_atomic_loadi64_relaxed((_Atomic(int64_t)*)(&subproc->stats.committed.peak)); + pinfo.current_commit = (current < 0 ? 0 : (current < PTRDIFF_MAX ? (size_t)current : PTRDIFF_MAX)); + pinfo.peak_commit = (peak < 0 ? 0 : (peak < PTRDIFF_MAX ? (size_t)peak : PTRDIFF_MAX)); } } pinfo.current_rss = pinfo.current_commit; @@ -722,9 +736,6 @@ static void mi_json_buf_print_counter_value(mi_json_buf_t* hbuf, const char* nam mi_json_buf_print_value(hbuf, name, stat->total); } -#define MI_STAT_COUNT(stat) mi_json_buf_print_count_value(&hbuf, #stat, &stats->stat); -#define MI_STAT_COUNTER(stat) mi_json_buf_print_counter_value(&hbuf, #stat, &stats->stat); - static char* mi_stats_get_json_from(const mi_stats_t* stats, size_t output_size, char* output_buf) mi_attr_noexcept { if (stats==NULL || stats->size!=sizeof(mi_stats_t) || stats->version!=MI_STAT_VERSION) return NULL; mi_json_buf_t hbuf = { NULL, 0, 0, true }; @@ -763,7 +774,13 @@ static char* mi_stats_get_json_from(const mi_stats_t* stats, size_t output_size, mi_json_buf_print(&hbuf, " },\n"); // statistics + #define MI_STAT_COUNT(stat) mi_json_buf_print_count_value(&hbuf, #stat, &stats->stat); + #define MI_STAT_COUNTER(stat) mi_json_buf_print_counter_value(&hbuf, #stat, &stats->stat); + MI_STAT_FIELDS() + + #undef MI_STAT_COUNT + #undef MI_STAT_COUNTER // size bins mi_json_buf_print(&hbuf, " \"malloc_bins\": [\n"); @@ -782,7 +799,7 @@ static char* mi_stats_get_json_from(const mi_stats_t* stats, size_t output_size, } mi_json_buf_print(&hbuf, " ]\n"); mi_json_buf_print(&hbuf, "}\n"); - if (hbuf.used >= hbuf.size) { + if (hbuf.used + 1 >= hbuf.size) { // failed if (hbuf.can_realloc) { mi_free(hbuf.buf); } return NULL; diff --git a/vendored/mimalloc/src/theap.c b/vendored/mimalloc/src/theap.c index ef53c1e07d..f27316f66f 100644 --- a/vendored/mimalloc/src/theap.c +++ b/vendored/mimalloc/src/theap.c @@ -1,5 +1,5 @@ /*---------------------------------------------------------------------------- -Copyright (c) 2018-2025, Microsoft Research, Daan Leijen +Copyright (c) 2018-2026, Microsoft Research, Daan Leijen This is free software; you can redistribute it and/or modify it under the terms of the MIT license. A copy of the license can be found in the file "LICENSE" at the root of this distribution. @@ -7,7 +7,7 @@ terms of the MIT license. A copy of the license can be found in the file #include "mimalloc.h" #include "mimalloc/internal.h" -#include "mimalloc/prim.h" // _mi_theap_default +#include "mimalloc/prim-tls.h" // _mi_theap_default #if defined(_MSC_VER) && (_MSC_VER < 1920) #pragma warning(disable:4204) // non-constant aggregate initializer @@ -23,7 +23,7 @@ typedef bool (theap_page_visitor_fun)(mi_theap_t* theap, mi_page_queue_t* pq, mi // Visit all pages in a theap; returns `false` if break was called. static bool mi_theap_visit_pages(mi_theap_t* theap, theap_page_visitor_fun* fn, bool include_full, void* arg1, void* arg2) { - if (theap==NULL || theap->page_count==0) return 0; + if (theap==NULL || theap->page_count==0) return true; // visit all pages #if MI_DEBUG>1 @@ -50,19 +50,24 @@ static bool mi_theap_visit_pages(mi_theap_t* theap, theap_page_visitor_fun* fn, } -#if MI_DEBUG>=2 +#if MI_DEBUG>=3 static bool mi_theap_page_is_valid(mi_theap_t* theap, mi_page_queue_t* pq, mi_page_t* page, void* arg1, void* arg2) { MI_UNUSED(arg1); MI_UNUSED(arg2); MI_UNUSED(pq); mi_assert_internal(mi_page_theap(page) == theap); + mi_theap_t* const page_theap = _mi_heap_theap_peek(page->heap); + mi_assert_internal(page_theap == NULL || theap == page_theap); mi_assert_expensive(_mi_page_is_valid(page)); return true; } -#endif -#if MI_DEBUG>=3 + static bool mi_theap_is_valid(mi_theap_t* theap) { mi_assert_internal(theap!=NULL); + mi_heap_t* const heap = _mi_theap_heap_peek(theap); + mi_assert_internal(heap != NULL); + mi_theap_t* const heap_theap = _mi_heap_theap_peek(heap); // don't use mi_heap_theap as that may re-initialize the thread + mi_assert_internal(heap_theap==NULL || heap_theap == theap); mi_theap_visit_pages(theap, &mi_theap_page_is_valid, true, NULL, NULL); for (size_t bin = 0; bin < MI_BIN_COUNT; bin++) { mi_assert_internal(_mi_page_queue_is_valid(theap, &theap->pages[bin])); @@ -91,7 +96,7 @@ typedef enum mi_collect_e { static bool mi_theap_page_collect(mi_theap_t* theap, mi_page_queue_t* pq, mi_page_t* page, void* arg_collect, void* arg2 ) { MI_UNUSED(arg2); MI_UNUSED(theap); - mi_assert_internal(mi_theap_page_is_valid(theap, pq, page, NULL, NULL)); + mi_assert_expensive(mi_theap_page_is_valid(theap, pq, page, NULL, NULL)); mi_collect_t collect = *((mi_collect_t*)arg_collect); _mi_page_free_collect(page, collect >= MI_FORCE); if (mi_page_all_free(page)) { @@ -110,7 +115,8 @@ static bool mi_theap_page_collect(mi_theap_t* theap, mi_page_queue_t* pq, mi_pag static void mi_theap_merge_stats(mi_theap_t* theap) { mi_assert_internal(mi_theap_is_initialized(theap)); - _mi_stats_merge_into(&_mi_theap_heap(theap)->stats, &theap->stats); + mi_heap_t* const heap = _mi_theap_heap(theap); + _mi_stats_merge_into(&heap->stats, &theap->stats); } static void mi_theap_collect_ex(mi_theap_t* theap, mi_collect_t collect) @@ -170,24 +176,35 @@ mi_theap_t* mi_theap_get_default(void) { return theap; } +mi_theap_t* mi_theap_set_default(mi_theap_t* theap) { + mi_theap_t* const previous = mi_theap_get_default(); + if (mi_theap_is_initialized(theap)) { + _mi_theap_default_set(theap); + } + return previous; +} + // todo: make order of parameters consistent (but would that break compat with CPython?) void _mi_theap_init(mi_theap_t* theap, mi_heap_t* heap, mi_tld_t* tld) { mi_assert_internal(theap!=NULL); mi_assert_internal(heap!=NULL); + mi_assert_internal(tld!=NULL); mi_memid_t memid = theap->memid; _mi_memcpy_aligned(theap, &_mi_theap_empty, sizeof(mi_theap_t)); theap->memid = memid; - theap->refcount = 1; theap->tld = tld; // avoid reading the thread-local tld during initialization - mi_atomic_store_ptr_relaxed(mi_heap_t,&theap->heap,heap); + mi_atomic_store_release(&theap->refcount,1); + mi_atomic_store_release(&theap->freed,0); + mi_atomic_store_ptr_release(mi_subproc_t,&theap->subproc,heap->subproc); mi_assert_internal(theap->stats.size == sizeof(mi_stats_t)); _mi_theap_options_init(theap); + if (theap->tld->is_in_threadpool) { // if we run as part of a thread pool it is better to not arbitrarily reclaim abandoned pages into our theap. // this is checked in `free.c:mi_free_try_collect_mt` - // .. but abandoning is good in this case: halve the full page retain (possibly to 0) + // .. but abandoning is good in this case: quarter the full page retain (possibly to 0) // (so blocked threads do not hold on to too much memory) if (theap->page_full_retain > 0) { theap->page_full_retain = theap->page_full_retain / 4; @@ -196,29 +213,40 @@ void _mi_theap_init(mi_theap_t* theap, mi_heap_t* heap, mi_tld_t* tld) // push on the thread local theaps list mi_theap_t* head = NULL; + mi_random_ctx_t head_random; mi_lock(&theap->tld->theaps_lock) { head = theap->tld->theaps; theap->tprev = NULL; theap->tnext = head; - if (head!=NULL) { head->tprev = theap; } theap->tld->theaps = theap; + if (head!=NULL) { + head->tprev = theap; + head_random = head->random; + } } - // initialize random - if (head == NULL) { // first theap in this thread? + // initialize random if heap==NULL + if (head==NULL) { // first theap of the first thread? #if defined(_WIN32) && !defined(MI_SHARED_LIB) + if (tld->thread_seq==0) { _mi_random_init_weak(&theap->random); // prevent allocation failure during bcrypt dll initialization with static linking (issue #1185) - #else - _mi_random_init(&theap->random); + } + else #endif + { + _mi_random_init(&theap->random); + } } else { - _mi_random_split(&head->random, &theap->random); + _mi_random_split(&head_random, &theap->random); // &theap->random is used as nonce so it is ok if threads capture the same head->random } - theap->cookie = _mi_theap_random_next(theap) | 1; - _mi_theap_guarded_init(theap); - mi_subproc_stat_increase(_mi_subproc(),theaps,1); + theap->cookie = _mi_theap_random_next(theap) | 1; + _mi_theap_guarded_init(theap); // needs theap->random + mi_subproc_stat_increase(_mi_theap_subproc(theap),theaps,1); // on subproc to match theap_free_mem + // only now set the heap member as it is used to determine if a theap is initialized + mi_atomic_store_ptr_release(mi_heap_t,&theap->heap,heap); + // push on the heap's theap list mi_lock(&heap->theaps_lock) { head = heap->theaps; @@ -232,31 +260,29 @@ void _mi_theap_init(mi_theap_t* theap, mi_heap_t* heap, mi_tld_t* tld) mi_theap_t* _mi_theap_create(mi_heap_t* heap, mi_tld_t* tld) { mi_assert_internal(tld!=NULL); mi_assert_internal(heap!=NULL); + mi_assert_internal(_mi_thread_id() == tld->thread_id); + // mi_assert_internal(_mi_heap_theap_peek(heap)==NULL); // don't access thread locals as this is called on thread init + // allocate and initialize a theap mi_memid_t memid; mi_theap_t* theap; - //if (!_mi_is_heap_main(heap)) { - // theap = (mi_theap_t*)mi_heap_zalloc(mi_heap_main(),sizeof(mi_theap_t)); - // memid = _mi_memid_create(MI_MEM_HEAP_MAIN); - // memid.initially_zero = memid.initially_committed = true; - //} - //else + if (heap->exclusive_arena == NULL) { - theap = (mi_theap_t*)_mi_meta_zalloc(sizeof(mi_theap_t), &memid); + theap = (mi_theap_t*)_mi_meta_zalloc(heap->subproc, sizeof(mi_theap_t), &memid); } else { // theaps associated with a specific arena are allocated in that arena // note: takes up at least one slice which is quite wasteful... const size_t size = _mi_align_up(sizeof(mi_theap_t),MI_ARENA_MIN_OBJ_SIZE); - theap = (mi_theap_t*)_mi_arenas_alloc(heap, size, true, true, heap->exclusive_arena, tld->thread_seq, tld->numa_node, &memid); - mi_assert_internal(memid.mem.os.size >= size); + theap = (mi_theap_t*)_mi_arenas_alloc(heap, size, true, true, heap->exclusive_arena, tld->thread_seq, tld->numa_node, &memid); } if (theap==NULL) { _mi_error_message(ENOMEM, "unable to allocate theap meta-data\n"); return NULL; } + theap->memid = memid; - _mi_theap_init(theap, heap, tld); + _mi_theap_init(theap, heap, tld); return theap; } @@ -266,29 +292,30 @@ uintptr_t _mi_theap_random_next(mi_theap_t* theap) { static void mi_theap_free_mem(mi_theap_t* theap) { if (theap!=NULL) { - mi_subproc_stat_decrease(_mi_subproc(),theaps,1); + mi_subproc_stat_decrease(_mi_theap_subproc(theap),theaps,1); // free the used memory if (theap->memid.memkind == MI_MEM_HEAP_MAIN) { // note: for now unused as it would access theap_default stats in mi_free of the current theap mi_assert_internal(_mi_is_heap_main(mi_heap_of(theap))); - mi_free(theap); + _mi_free_subproc_safe(theap); } else if (theap->memid.memkind == MI_MEM_META) { - _mi_meta_free(theap, sizeof(*theap), theap->memid); + _mi_meta_free(_mi_theap_subproc(theap), theap, sizeof(*theap), theap->memid); } else { - _mi_arenas_free(theap, _mi_align_up(sizeof(*theap),MI_ARENA_MIN_OBJ_SIZE), theap->memid ); // issue #1168, avoid assertion failure + _mi_arenas_free(_mi_theap_subproc(theap), theap, _mi_align_up(sizeof(*theap),MI_ARENA_MIN_OBJ_SIZE), theap->memid ); // issue #1168, avoid assertion failure } } } +// we need to reference count theaps due to the _mi_theap_cached thread locals void _mi_theap_incref(mi_theap_t* theap) { - if (theap!=NULL && theap->memid.memkind > MI_MEM_STATIC) { + if (theap!=NULL && !mi_memid_needs_no_free(theap->memid)) { mi_atomic_increment_acq_rel(&theap->refcount); } } void _mi_theap_decref(mi_theap_t* theap) { - if (theap!=NULL && theap->memid.memkind > MI_MEM_STATIC) { + if (theap!=NULL && !mi_memid_needs_no_free(theap->memid)) { if (mi_atomic_decrement_acq_rel(&theap->refcount) == 1) { mi_theap_free_mem(theap); } @@ -301,13 +328,15 @@ bool _mi_theap_free(mi_theap_t* theap, bool acquire_heap_theaps_lock, bool acqui mi_assert(theap != NULL); if (theap==NULL) return true; - mi_heap_t* const heap = mi_atomic_exchange_ptr_acq_rel(mi_heap_t, &theap->heap, NULL); - if (heap==NULL) { + // ensure only one thread actually frees the theap + const size_t freed = mi_atomic_exchange_acq_rel( &theap->freed, 1 ); + if (freed!=0) { // concurrent interaction, retry in an outer loop (as the other thread may be blocked on our lock) return false; } else { // merge stats to the owning heap + mi_heap_t* const heap = _mi_theap_heap(theap); _mi_stats_merge_into(&heap->stats, &theap->stats); // remove ourselves from the heap theaps list @@ -323,9 +352,16 @@ bool _mi_theap_free(mi_theap_t* theap, bool acquire_heap_theaps_lock, bool acqui if (theap->tnext != NULL) { theap->tnext->tprev = theap->tprev; } if (theap->tprev != NULL) { theap->tprev->tnext = theap->tnext; } else { mi_assert_internal(theap->tld->theaps == theap); theap->tld->theaps = theap->tnext; } - theap->tnext = theap->tprev = NULL; + theap->tnext = theap->tprev = NULL; } + + // Set heap to NULL only after we are removed from the thread local theaps list since + // we may concurrently traverse it to collect (in `init.c:mi_thread_theaps_done`) + // (We need to set it to NULL to avoid an ABA problem where the _mi_theap_cached + // has a heap address that is reused for a newly allocated heap.) + mi_atomic_store_ptr_release(mi_heap_t, &theap->heap, NULL); theap->tld = NULL; + // leave subproc field as is for free-ing _mi_theap_decref(theap); return true; } @@ -333,138 +369,23 @@ bool _mi_theap_free(mi_theap_t* theap, bool acquire_heap_theaps_lock, bool acqui /* ----------------------------------------------------------- - Heap destroy ------------------------------------------------------------ */ -/* - -// zero out the page queues -static void mi_theap_reset_pages(mi_theap_t* theap) { - mi_assert_internal(theap != NULL); - mi_assert_internal(mi_theap_is_initialized(theap)); - // TODO: copy full empty theap instead? - _mi_memset(&theap->pages_free_direct, 0, sizeof(theap->pages_free_direct)); - _mi_memcpy_aligned(&theap->pages, &_mi_theap_empty.pages, sizeof(theap->pages)); - // theap->thread_delayed_free = NULL; - theap->page_count = 0; -} - -static bool _mi_theap_page_destroy(mi_theap_t* theap, mi_page_queue_t* pq, mi_page_t* page, void* arg1, void* arg2) { - MI_UNUSED(arg1); - MI_UNUSED(arg2); - MI_UNUSED(pq); - - // ensure no more thread_delayed_free will be added - //_mi_page_use_delayed_free(page, MI_NEVER_DELAYED_FREE, false); - - // stats - const size_t bsize = mi_page_block_size(page); - if (bsize > MI_LARGE_MAX_OBJ_SIZE) { - mi_theap_stat_decrease(theap, malloc_huge, bsize); - } - #if (MI_STAT>0) - _mi_page_free_collect(page, false); // update used count - const size_t inuse = page->used; - if (bsize <= MI_LARGE_MAX_OBJ_SIZE) { - mi_theap_stat_decrease(theap, malloc_normal, bsize * inuse); - #if (MI_STAT>1) - mi_theap_stat_decrease(theap, malloc_bins[_mi_bin(bsize)], inuse); - #endif - } - // mi_theap_stat_decrease(theap, malloc_requested, bsize * inuse); // todo: off for aligned blocks... - #endif - - /// pretend it is all free now - mi_assert_internal(mi_page_thread_free(page) == NULL); - page->used = 0; - - // and free the page - // mi_page_free(page,false); - page->next = NULL; - page->prev = NULL; - mi_page_set_theap(page, NULL); - _mi_arenas_page_free(page, theap); - - return true; // keep going -} - -void _mi_theap_destroy_pages(mi_theap_t* theap) { - mi_theap_visit_pages(theap, &_mi_theap_page_destroy, NULL, NULL); - mi_theap_reset_pages(theap); -} - -#if MI_TRACK_HEAP_DESTROY -static bool mi_cdecl mi_theap_track_block_free(const mi_theap_t* theap, const mi_theap_area_t* area, void* block, size_t block_size, void* arg) { - MI_UNUSED(theap); MI_UNUSED(area); MI_UNUSED(arg); MI_UNUSED(block_size); - mi_track_free_size(block,mi_usable_size(block)); - return true; -} -#endif - -void mi_theap_destroy(mi_theap_t* theap) { - mi_assert(theap != NULL); - mi_assert(mi_theap_is_initialized(theap)); - mi_assert(!theap->allow_page_reclaim); - mi_assert(!theap->allow_page_abandon); - mi_assert_expensive(mi_theap_is_valid(theap)); - if (theap==NULL || !mi_theap_is_initialized(theap)) return; - #if MI_GUARDED - // _mi_warning_message("'mi_theap_destroy' called but MI_GUARDED is enabled -- using `mi_theap_delete` instead (theap at %p)\n", theap); - mi_theap_delete(theap); - return; - #else - if (theap->allow_page_reclaim) { - _mi_warning_message("'mi_theap_destroy' called but ignored as the theap was not created with 'allow_destroy' (theap at %p)\n", theap); - // don't free in case it may contain reclaimed pages, - mi_theap_delete(theap); - } - else { - // track all blocks as freed - #if MI_TRACK_HEAP_DESTROY - mi_theap_visit_blocks(theap, true, mi_theap_track_block_free, NULL); - #endif - // free all pages - _mi_theap_destroy_pages(theap); - mi_theap_free(theap,true); - } - #endif -} - -// forcefully destroy all theaps in the current thread -void _mi_theap_unsafe_destroy_all(mi_theap_t* theap) { - mi_assert_internal(theap != NULL); - if (theap == NULL) return; - mi_theap_t* curr = theap->tld->theaps; - while (curr != NULL) { - mi_theap_t* next = curr->next; - if (!curr->allow_page_reclaim) { - mi_theap_destroy(curr); - } - else { - _mi_theap_destroy_pages(curr); - } - curr = next; - } -} -*/ - -/* ----------------------------------------------------------- - Safe Heap delete + Safe theap delete ----------------------------------------------------------- */ // Safe delete a theap without freeing any still allocated blocks in that theap. -void _mi_theap_delete(mi_theap_t* theap, bool acquire_tld_theaps_lock) -{ - mi_assert(theap != NULL); - mi_assert(mi_theap_is_initialized(theap)); - mi_assert_expensive(mi_theap_is_valid(theap)); - if (theap==NULL || !mi_theap_is_initialized(theap)) return; +// void _mi_theap_delete(mi_theap_t* theap, bool acquire_tld_theaps_lock) +// { +// mi_assert(theap != NULL); +// mi_assert(mi_theap_is_initialized(theap)); +// mi_assert_expensive(mi_theap_is_valid(theap)); +// if (theap==NULL || !mi_theap_is_initialized(theap)) return; - // abandon all pages - _mi_theap_collect_abandon(theap); +// // abandon all pages +// _mi_theap_collect_abandon(theap); - mi_assert_internal(theap->page_count==0); - _mi_theap_free(theap, true /* acquire heap->theaps_lock */, acquire_tld_theaps_lock); -} +// mi_assert_internal(theap->page_count==0); +// _mi_theap_free(theap, true /* acquire heap->theaps_lock */, acquire_tld_theaps_lock); +// } @@ -663,6 +584,11 @@ bool _mi_theap_area_visit_blocks(const mi_heap_area_t* area, mi_page_t* page, mi return true; } +// bool _mi_page_visit_blocks( mi_page_t* page, mi_block_visit_fun* visitor, void* arg ) { +// mi_heap_area_t area; +// _mi_heap_area_init(&area, page); +// return _mi_theap_area_visit_blocks(&area, page, visitor, arg); +// } // Separate struct to keep `mi_page_t` out of the public interface diff --git a/vendored/mimalloc/src/threadlocal.c b/vendored/mimalloc/src/threadlocal.c index eea3329bfa..ca94307a83 100644 --- a/vendored/mimalloc/src/threadlocal.c +++ b/vendored/mimalloc/src/threadlocal.c @@ -32,7 +32,35 @@ typedef struct mi_thread_locals_s { static mi_thread_locals_t mi_thread_locals_empty = { 0, {{0,NULL}} }; -mi_decl_thread mi_thread_locals_t* mi_thread_locals = &mi_thread_locals_empty; // always point to a valid `mi_thread_locals_t` + +/* ----------------------------------------------------------- + We have 2 thread local variable which we implement with either + a C thread local declaration or using pthread keys. + - mi_thread_locals: points to an array of thread locals for most keys + - mi_slot_fast: a single dedicated thread local for slightly faster access. (used for the main heap's theap) +----------------------------------------------------------- */ + +#if MI_TLS_MODEL_PTHREADS || defined(__APPLE__) // macOS has fast pthreads +// Use pthreads +#define mi_define_thread_local(tp,name,initval) \ + static pthread_key_t __##name##_key = MI_PTHREAD_KEY_INVALID; \ + static inline tp name##_peek(void) { return (tp)mi_pthread_key_get(__##name##_key); } \ + static inline tp name##_get(void) { tp result = name##_peek(); return (result!=NULL ? result : initval); } \ + static inline bool name##_set(tp val) { return mi_pthread_key_set(&__##name##_key,val); } \ + static inline void name##_delete(void) { mi_pthread_key_delete(&__##name##_key); } + +#else +// Direct thread locals +#define mi_define_thread_local(tp,name,initval) \ + static mi_decl_thread tp __##name = initval; \ + static inline tp name##_peek(void) { return __##name; } \ + static inline tp name##_get(void) { return __##name; } \ + static inline bool name##_set(tp val) { __##name = val; return true; } \ + static inline void name##_delete(void) { } +#endif + +mi_define_thread_local(mi_thread_locals_t*, mi_thread_locals, &mi_thread_locals_empty) +mi_define_thread_local(void*, mi_slot_fast, NULL) /* ----------------------------------------------------------- @@ -42,15 +70,18 @@ mi_decl_thread mi_thread_locals_t* mi_thread_locals = &mi_thread_locals_empty; a value, we also set the version of the key. ----------------------------------------------------------- */ -#if MI_SIZE_BITS < 64 -#define MI_TLS_IDX_BITS (MI_SIZE_BITS/2) // half for the index, half for the version +#if MI_SIZE_BITS >= 64 +#define MI_TLS_IDX_BITS (MI_SIZE_BITS/4) /* 16 bits for the index, 48 bits for the version */ +#elif MI_SIZE_BITS >= 32 +#define MI_TLS_IDX_BITS (12) /* 12 bits for index, 20 for the version? */ #else -#define MI_TLS_IDX_BITS (MI_SIZE_BITS/4) // 16 bits for the index, 48 bits for the version +#error not enough bits for the version for thread locals #endif #define MI_TLS_IDX_MASK ((MI_ZU(1)<count; size_t count; if (count_old==0) { tls_old = NULL; // so we allocate fresh from mi_thread_locals_empty count = 16; // start with 16 slots - } + } else if (count_old >= 1024) { count = count_old + 1024; // at some point increase linearly } else { count = 2*count_old; // and double initially } - if (count <= least_idx) { + if (count <= least_idx) { count = least_idx + 1; } if (count > MI_TLS_IDX_MAX) { return NULL; } // too large mi_thread_locals_t* tls = (mi_thread_locals_t*)mi_rezalloc(tls_old, sizeof(mi_thread_locals_t) + count*sizeof(mi_tls_slot_t)); if mi_unlikely(tls==NULL) return NULL; tls->count = count; - mi_thread_locals = tls; + mi_thread_locals_set(tls); return tls; } static mi_decl_noinline bool mi_thread_local_set_expand( mi_thread_local_t key, void* val ) { if (val==NULL) return true; - const size_t idx = mi_key_index(key); + const size_t idx = mi_key_index(key); mi_thread_locals_t* tls = mi_thread_locals_expand(idx); - if (tls==NULL) return false; - mi_assert_internal(tls == mi_thread_locals); + if (tls==NULL) { + _mi_error_message(EFAULT,"unable to allocate thread local variables\n"); + return false; + } + mi_assert_internal(tls == mi_thread_locals_get()); mi_assert_internal(idx < tls->count); tls->slots[idx].value = val; tls->slots[idx].version = mi_key_version(key); @@ -108,8 +142,8 @@ static mi_decl_noinline bool mi_thread_local_set_expand( mi_thread_local_t key, // set a tls slot; returns `true` if successful. // Can return `false` if we could not reallocate the slots array. -bool _mi_thread_local_set( mi_thread_local_t key, void* val ) { - mi_thread_locals_t* tls = mi_thread_locals; +static mi_decl_noinline bool mi_thread_local_set_regular( mi_thread_local_t key, void* val ) { + mi_thread_locals_t* tls = mi_thread_locals_get(); mi_assert_internal(tls!=NULL); mi_assert_internal(key!=0); const size_t idx = mi_key_index(key); @@ -123,25 +157,49 @@ bool _mi_thread_local_set( mi_thread_local_t key, void* val ) { } } +bool _mi_thread_local_set( mi_thread_local_t key, void* val ) { + mi_assert_internal(key!=0); + if (key == mi_thread_local_key_fast) { + return mi_slot_fast_set(val); + } + else { + return mi_thread_local_set_regular(key,val); + } +} + // get a tls slot value -void* _mi_thread_local_get( mi_thread_local_t key ) { - const mi_thread_locals_t* const tls = mi_thread_locals; - mi_assert_internal(tls!=NULL); +static mi_decl_noinline void* mi_thread_local_get_regular( mi_thread_local_t key ) { mi_assert_internal(key!=0); + const mi_thread_locals_t* const tls = mi_thread_locals_get(); + mi_assert_internal(tls!=NULL); const size_t idx = mi_key_index(key); if mi_likely(idx < tls->count && mi_key_version(key) == tls->slots[idx].version) { return tls->slots[idx].value; } else { - return NULL; + return NULL; + } +} + +// get a thread local value +void* _mi_thread_local_get( mi_thread_local_t key ) { + mi_assert_internal(key!=0); + if mi_likely(key == mi_thread_local_key_fast) { + return mi_slot_fast_get(); + } + else { + return mi_thread_local_get_regular(key); } } void _mi_thread_locals_thread_done(void) { - mi_thread_locals_t* const tls = mi_thread_locals; + mi_thread_locals_t* const tls = mi_thread_locals_peek(); if (tls!=NULL && tls->count > 0) { mi_free(tls); - mi_thread_locals = &mi_thread_locals_empty; + mi_thread_locals_set(NULL); + } + if (mi_slot_fast_peek() != NULL) { + mi_slot_fast_set(NULL); } } @@ -152,6 +210,7 @@ Create and free fresh TLS key's static mi_lock_t mi_thread_locals_lock; // we need a lock in order to re-allocate the slot bits static mi_bitmap_t* mi_thread_locals_free; // reuse an arena bitmap to track which slots were assigned (1=free, 0=in-use) +static mi_memid_t mi_thread_locals_memid; // provenance of mi_thread_locals_free static size_t mi_thread_locals_version; // version to be able to reuse slots safely void _mi_thread_locals_init(void) { @@ -161,9 +220,15 @@ void _mi_thread_locals_init(void) { void _mi_thread_locals_done(void) { mi_lock(&mi_thread_locals_lock) { mi_bitmap_t* const slots = mi_thread_locals_free; - mi_free(slots); + if (slots!=NULL) { + const size_t slots_count = mi_bitmap_max_bits(slots); + const size_t slots_size = mi_bitmap_size(slots_count,NULL); + _mi_meta_free(_mi_subproc_main(), slots,slots_size,mi_thread_locals_memid); + } } mi_lock_done(&mi_thread_locals_lock); + mi_thread_locals_delete(); + mi_slot_fast_delete(); } // strange signature but allows us to reuse the arena code for claiming free pages @@ -173,7 +238,7 @@ static bool mi_thread_local_claim_fun(size_t _slice_index, mi_arena_t* _arena, b return true; } -// When we claim a free slot, we increase the global version counter +// When we claim a free slot, we increase the global version counter // (so if we reuse a slot it will be returning NULL initially when a thread tries to get it) static mi_thread_local_t mi_thread_local_claim(void) { size_t idx = 0; @@ -194,16 +259,21 @@ static bool mi_thread_local_create_expand(void) { const size_t newcount = 1024 + oldcount; if (newcount > MI_TLS_IDX_MAX) { return false; } const size_t newsize = mi_bitmap_size( newcount, NULL ); - mi_bitmap_t* newslots = (mi_bitmap_t*)mi_zalloc_aligned(newsize, MI_BCHUNK_SIZE); + // mi_bitmap_t* newslots = (mi_bitmap_t*)mi_zalloc_aligned(newsize, MI_BCHUNK_SIZE); + mi_memid_t memid; + mi_bitmap_t* newslots = (mi_bitmap_t*)_mi_meta_zalloc(_mi_subproc_main(), newsize, &memid); // always allocate thread locals in the main subprocess + mi_assert_internal(_mi_is_aligned(newslots,MI_BCHUNK_SIZE)); if (newslots==NULL) { return false; } if (slots!=NULL) { // copy over the previous bitmap - _mi_memcpy_aligned(newslots, slots, mi_bitmap_size(oldcount, NULL)); - mi_free(slots); + const size_t oldsize = mi_bitmap_size(oldcount,NULL); + _mi_memcpy_aligned(newslots, slots, oldsize); + _mi_meta_free(_mi_subproc_main(), slots,oldsize,mi_thread_locals_memid); } mi_bitmap_init(newslots, newcount, true /* pretend already zero'd so we do not zero out the copied old entries */); mi_bitmap_unsafe_setN(newslots, oldcount, newcount - oldcount); /* set the new expanded slots as available */ mi_thread_locals_free = newslots; + mi_thread_locals_memid = memid; return true; } @@ -219,6 +289,8 @@ mi_thread_local_t _mi_thread_local_create(void) { } } } + mi_assert_internal(key!=0); + mi_assert_internal(key!=mi_thread_local_key_fast); return key; }