diff --git a/docs/packages/ladybug.yaml b/docs/packages/ladybug.yaml index c06ea16c00b..b8e2aaa6907 100644 --- a/docs/packages/ladybug.yaml +++ b/docs/packages/ladybug.yaml @@ -23,3 +23,10 @@ versions: - filename: ladybug-0.19.1-cp314-cp314t-manylinux_2_39_riscv64.whl sha256: 6f365ce0a83d6fcd6b1abc76e78f34759de07a1a7b823f4f3e4119b2f6382537 requires-python: <3.15,>=3.10 +- version: 0.20.0 +- version: 0.20.1 +- version: 0.20.2 +- version: 0.20.3 +- version: 0.20.4 +- version: 0.21.0 +- version: 0.21.1 diff --git a/patches/ladybug/0.20.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.20.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.20.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.20.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.20.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.20.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.20.0/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.20.0/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.20.0/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.20.0/0004-factorized_table-clear-early-return-for-empty-schema.patch b/patches/ladybug/0.20.0/0004-factorized_table-clear-early-return-for-empty-schema.patch new file mode 100644 index 00000000000..0abbdcd869a --- /dev/null +++ b/patches/ladybug/0.20.0/0004-factorized_table-clear-early-return-for-empty-schema.patch @@ -0,0 +1,47 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Thu, 1 Oct 2026 00:00:00 +0000 +Subject: [PATCH] factorized_table: clear() early-return for empty schema + +Upstream-Status: Backport [https://github.com/LadybugDB/ladybug/commit/d09008e75168e2d204b62cbd12029b621631d6f6] + +Re-executing the same parameterized write query string (e.g. +`CREATE (:Log {id: $id, value: $val})`) through the implicit +prepared-statement cache SIGSEGVs on the second execution. The +cached-physical-plan fast path calls prepareForReuse(), which reaches +ResultCollector::prepareForReuse() and FactorizedTable::clear(). For a +write statement the root ResultCollector's FactorizedTable has an empty +result schema (writes return no columns), and the constructor skips +allocating flatTupleBlockCollection / unFlatTupleBlockCollection / +inMemOverflowBuffer entirely for an empty schema. clear() +unconditionally dereferenced the null block collection. + +Reproduces identically on cp312/cp313/cp314 and cp314t (not a +free-threading or riscv64 issue): see e.g. the python_api test suite's +test_blob_parameter.py::test_bytes_param and +test_datatype.py::test_large_array, both of which loop a parameterized +CREATE over the same connection. Fixed upstream before 0.20.1. +--- + src/processor/result/factorized_table.cpp | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/src/processor/result/factorized_table.cpp b/src/processor/result/factorized_table.cpp +index cc2fd52..78a41ab 100644 +--- a/src/processor/result/factorized_table.cpp ++++ b/src/processor/result/factorized_table.cpp +@@ -333,6 +333,13 @@ void FactorizedTable::setNonOverflowColNull(uint8_t* nullBuffer, ft_col_idx_t col + } + + void FactorizedTable::clear() { ++ if (tableSchema.isEmpty()) { ++ // Tables with an empty schema (e.g. the root ResultCollector of a CREATE / DDL ++ // statement) never allocate block collections or an overflow buffer; there is ++ // nothing to reset. Mirrors the constructor, which skips allocation entirely for ++ // an empty schema. ++ return; ++ } + numTuples = 0; + // Reuse the first DataBlock (zeroed) and drop the rest. This skips the + // 256KB malloc on the next append while preserving the dense-packing +-- +2.43.0 diff --git a/patches/ladybug/0.20.0/0005-result_collector-dont-clobber-live-queryresults-on-reuse.patch b/patches/ladybug/0.20.0/0005-result_collector-dont-clobber-live-queryresults-on-reuse.patch new file mode 100644 index 00000000000..6511e0256cf --- /dev/null +++ b/patches/ladybug/0.20.0/0005-result_collector-dont-clobber-live-queryresults-on-reuse.patch @@ -0,0 +1,91 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Thu, 1 Oct 2026 00:00:00 +0000 +Subject: [PATCH] result_collector: don't clobber live QueryResults on reuse + +Upstream-Status: Backport [https://github.com/LadybugDB/ladybug/commit/47443cd683adaf6b34d717cf751ecd6cb22c0869] + +The cached-physical-plan fast path clones the template operator tree +per execution and calls prepareForReuse(). The plan template shares the +ResultCollectorSharedState (and its FactorizedTable) with every +executed clone, and prepareForReuse() unconditionally clear()ed that +table. A QueryResult from a previous execution holds a shared_ptr to +the same table, so overlapping executions (e.g. AsyncConnection's pool, +running the same parameterized statement concurrently) corrupted live +results: queries returned other queries' rows or empty tables. + +This is exactly what test_async_connection.py:: +test_async_prepare_and_execute_concurrent hits on 0.20.0 (upstream's own +commit message cites this same test by name: "asserted [96] == [1]"). +Not free-threading- or riscv64-specific. Fixed upstream before 0.20.1. + +The companion simple_table_function.h fix in the same upstream commit +(resetState() for pandas/polars/arrow table-function rescans) isn't +known to be exercised by anything failing in this port's test suite on +0.20.0, but it's included here too since it's part of the same commit +and trivially small. +--- + src/include/function/table/simple_table_function.h | 4 ++++ + src/include/processor/operator/result_collector.h | 2 ++ + src/processor/operator/result_collector.cpp | 18 +++++++++++++++--- + 3 files changed, 21 insertions(+), 3 deletions(-) + +diff --git a/src/include/function/table/simple_table_function.h b/src/include/function/table/simple_table_function.h +index 5314b73..32abebc 100644 +--- a/src/include/function/table/simple_table_function.h ++++ b/src/include/function/table/simple_table_function.h +@@ -39,6 +39,10 @@ public: + + virtual TableFuncMorsel getMorsel(); + ++ // Reset the scan position so a reused physical plan (cached-plan fast path) ++ // rescans from the beginning instead of finding an exhausted morsel cursor. ++ void resetState() override { curRowIdx = 0; } ++ + common::row_idx_t curRowIdx = 0; + common::offset_t maxMorselSize = common::DEFAULT_VECTOR_CAPACITY; + }; +diff --git a/src/include/processor/operator/result_collector.h b/src/include/processor/operator/result_collector.h +index bad840e..1b52db1 100644 +--- a/src/include/processor/operator/result_collector.h ++++ b/src/include/processor/operator/result_collector.h +@@ -22,6 +22,8 @@ public: + + std::shared_ptr getTable() { return table; } + ++ void setTable(std::shared_ptr newTable) { table = std::move(newTable); } ++ + private: + std::mutex mtx; + std::shared_ptr table; +diff --git a/src/processor/operator/result_collector.cpp b/src/processor/operator/result_collector.cpp +index 2685d0f..ee45bef 100644 +--- a/src/processor/operator/result_collector.cpp ++++ b/src/processor/operator/result_collector.cpp +@@ -58,9 +58,21 @@ void ResultCollector::executeInternal(ExecutionContext* context) { + } + + void ResultCollector::prepareForReuse(storage::MemoryManager* memoryManager) { +- // Clear the existing result table instead of freeing + re-allocating. +- // This keeps the DataBlocks alive so the next execution reuses them. +- sharedState->getTable()->clear(); ++ auto table = sharedState->getTable(); ++ if (table.use_count() <= 1) { ++ // No QueryResult outside this shared state references the table, so we can ++ // keep the DataBlocks alive and reset the bookkeeping (Phase 2 fast path). ++ table->clear(); ++ } else { ++ // A previous execution's QueryResult still holds this table (e.g. overlapping ++ // AsyncConnection executions of the same prepared statement on the cached-plan ++ // fast path, which shares one ResultCollectorSharedState with the plan template). ++ // Clearing it would corrupt that live result, so hand this execution a fresh ++ // table with the same schema instead. The old table stays alive until the ++ // QueryResult that references it is destroyed. ++ sharedState->setTable( ++ std::make_shared(memoryManager, info.tableSchema.copy())); ++ } + PhysicalOperator::prepareForReuse(memoryManager); + } + +-- +2.43.0 diff --git a/patches/ladybug/0.20.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.20.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.20.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.20.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.20.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.20.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.20.1/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.20.1/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.20.1/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.20.2/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.20.2/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.20.2/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.20.2/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.20.2/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.20.2/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.20.2/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.20.2/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.20.2/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.20.3/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.20.3/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.20.3/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.20.3/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.20.3/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.20.3/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.20.3/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.20.3/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.20.3/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.20.4/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.20.4/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.20.4/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.20.4/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.20.4/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.20.4/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.20.4/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.20.4/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.20.4/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.21.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.21.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.21.0/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.21.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.21.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.21.0/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.21.0/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.21.0/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.21.0/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + } diff --git a/patches/ladybug/0.21.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch b/patches/ladybug/0.21.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch new file mode 100644 index 00000000000..9c58b6d45f1 --- /dev/null +++ b/patches/ladybug/0.21.1/0001-storage-shrink-the-VMRegion-reservation-until-it-fits.patch @@ -0,0 +1,48 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Fri, 25 Sep 2026 00:00:00 +0000 +Subject: [PATCH] storage: shrink the VMRegion reservation until it fits + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +Every Database reserves its buffer-manager region with one mmap of +max_db_size bytes, and max_db_size defaults to 1 << 43 (8TB). riscv64 +Sv39 gives a process 256GB of user address space (the T-Head C910/C920 +cores the riscv64 runners use only implement Sv39), so the reservation +fails with ENOMEM and every Database() created with the default +settings throws "Mmap for size 8796093022208 failed." The same happens +under a 39-bit VA arm64 kernel, or on x86-64 under +`ulimit -v 268435456`. Still present in v0.20.4. + +On ENOMEM, halve the reservation until it fits. Where the full region +fits nothing changes; elsewhere the database is capped at the largest +power-of-two region the address space can hold, the limit +max_db_size already expresses. +--- +diff --git a/src/storage/buffer_manager/vm_region.cpp b/src/storage/buffer_manager/vm_region.cpp +index 429bc61..d102eef 100644 +--- a/src/storage/buffer_manager/vm_region.cpp ++++ b/src/storage/buffer_manager/vm_region.cpp +@@ -13,6 +13,8 @@ + #else + #include + #include ++ ++#include + #endif + + #include "common/assert.h" +@@ -61,6 +63,13 @@ VMRegion::VMRegion(PageSizeClass pageSizeClass, uint64_t maxRegionSize) : numFra + // backed by any file, and its content are initialized to zero. + region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ // The default 8TB region does not fit in a smaller user address space (256GB under riscv64 ++ // Sv39, 512GB under a 39-bit VA arm64 kernel), so halve the reservation until it does. ++ while (region == MAP_FAILED && errno == ENOMEM && maxNumFrameGroups > 1) { ++ maxNumFrameGroups /= 2; ++ region = static_cast(mmap(NULL, getMaxRegionSize(), PROT_READ | PROT_WRITE, ++ MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1 /* fd */, 0 /* offset */)); ++ } + if (region == MAP_FAILED) { + throw BufferManagerException( + "Mmap for size " + std::to_string(getMaxRegionSize()) + " failed."); diff --git a/patches/ladybug/0.21.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch b/patches/ladybug/0.21.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch new file mode 100644 index 00000000000..fda78c4815a --- /dev/null +++ b/patches/ladybug/0.21.1/0002-test-repeat-the-interrupt-until-the-query-stops.patch @@ -0,0 +1,41 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sat, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] test: repeat the interrupt until the query stops + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +test_connection_interrupt starts a long query on a thread, sleeps 5s, +calls conn.interrupt() once and expects the thread to end within 100s. +Binding folds each RANGE(1, 1000000) into a million-element list +literal, and ClientContext::executeNoLock() calls resetActiveQuery(), +which clears the interrupted flag, only once compilation is done. On +the riscv64 runners compiling the query takes longer than 5s, so the +interrupt lands during compilation, is wiped, and the query keeps +running. The fixture teardown's close() then waits on it for about 4 +hours, and the query thread segfaults freeing its FactorizedTable +after the database is gone. The same loss reproduces on x86-64 with +upstream's 0.19.1 wheel when the sleep is shorter than the compile. +Still present on ladybug-python main. + +Re-issue the interrupt every second until the thread ends, within the +same 100s budget. On a fast machine the first interrupt still sticks +and the test behaves as before. +--- +diff --git a/tools/python_api/test/test_connection.py b/tools/python_api/test/test_connection.py +index dcc8ee5..aeb2d54 100644 +--- a/tools/python_api/test/test_connection.py ++++ b/tools/python_api/test/test_connection.py +@@ -46,6 +46,10 @@ def test_connection_interrupt(conn_db_readwrite: ConnDB) -> None: + execute_thread = threading.Thread(target=run_long_query, args=(conn,)) + execute_thread.start() + time.sleep(5) +- conn.interrupt() +- execute_thread.join(timeout=100) ++ # Compiling this query can outlast the sleep on a slow machine, and the interrupt flag is ++ # cleared when execution starts, so an interrupt sent during compilation is lost; repeat it. ++ deadline = time.monotonic() + 100 ++ while execute_thread.is_alive() and time.monotonic() < deadline: ++ conn.interrupt() ++ execute_thread.join(timeout=1) + assert not execute_thread.is_alive() diff --git a/patches/ladybug/0.21.1/0003-extension-report-riscv64-as-its-own-platform.patch b/patches/ladybug/0.21.1/0003-extension-report-riscv64-as-its-own-platform.patch new file mode 100644 index 00000000000..ba6c25260cc --- /dev/null +++ b/patches/ladybug/0.21.1/0003-extension-report-riscv64-as-its-own-platform.patch @@ -0,0 +1,33 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Sun, 27 Sep 2026 00:00:00 +0000 +Subject: [PATCH] extension: report riscv64 as its own platform + +Upstream-Status: To upstream [not yet submitted; python-wheels does not open issues/PRs on third-party repos] + +getArch() starts from "amd64" and only overrides it for x86 and arm64, +so on riscv64 getPlatform() returns "linux_amd64". INSTALL then +downloads the x86-64 build of an extension from +extension.ladybugdb.com into ~/.lbdb/extension//linux_amd64/, +and LOAD fails with "cannot open shared object file: No such file or +directory", which is how glibc's dlopen reports an ELF for another +machine. Seen in test_json.py's INSTALL json; LOAD json on the riscv64 +runners. Still present on main. + +Return "riscv64" there, so INSTALL looks for linux_riscv64 builds +(upstream publishes none yet) and the extension cache is keyed by the +right platform. +--- +diff --git a/src/extension/extension.cpp b/src/extension/extension.cpp +index e87004f..e9e2891 100644 +--- a/src/extension/extension.cpp ++++ b/src/extension/extension.cpp +@@ -144,6 +144,8 @@ std::string getArch() { + arch = "x86"; + #elif defined(__aarch64__) || defined(__ARM_ARCH_ISA_A64) + arch = "arm64"; ++#elif defined(__riscv) && __riscv_xlen == 64 ++ arch = "riscv64"; + #endif + return arch; + }