From ffed7ec3ae7a85be05bd77d812eab22b3ae02a84 Mon Sep 17 00:00:00 2001 From: "renovate[bot]" <29139614+renovate[bot]@users.noreply.github.com> Date: Thu, 13 Aug 2026 12:49:10 -0400 Subject: [PATCH 1/5] Update dependency sharp to v0.35.0 [SECURITY] (#3740) --- frontends/main/package.json | 2 +- yarn.lock | 354 +++++++++++++++++++++++++++++++++++- 2 files changed, 353 insertions(+), 3 deletions(-) diff --git a/frontends/main/package.json b/frontends/main/package.json index fb27627cff..3a36261a1b 100644 --- a/frontends/main/package.json +++ b/frontends/main/package.json @@ -70,7 +70,7 @@ "react-hotkeys-hook": "^5.2.1", "react-markdown": "^10.0.0", "react-slick": "^0.31.0", - "sharp": "0.34.5", + "sharp": "0.35.0", "slick-carousel": "^1.8.1", "tiny-invariant": "^1.3.3", "video.js": "^8.23.7", diff --git a/yarn.lock b/yarn.lock index 7ef8c32d93..917a94d14b 100644 --- a/yarn.lock +++ b/yarn.lock @@ -2024,6 +2024,15 @@ __metadata: languageName: node linkType: hard +"@emnapi/runtime@npm:^1.11.0": + version: 1.11.3 + resolution: "@emnapi/runtime@npm:1.11.3" + dependencies: + tslib: "npm:^2.4.0" + checksum: 10/ec834cc0fe248e06dd84f6dee10162133f8a692636189e99aa992303c791d4be1679536ee91aeee6661c4478609768cee9c085acd858caa92784c67acaab44a5 + languageName: node + linkType: hard + "@emnapi/runtime@npm:^1.4.0": version: 1.4.5 resolution: "@emnapi/runtime@npm:1.4.5" @@ -2618,6 +2627,13 @@ __metadata: languageName: node linkType: hard +"@img/colour@npm:^1.1.0": + version: 1.1.0 + resolution: "@img/colour@npm:1.1.0" + checksum: 10/2a29be7b06b046bd33c80ffa0f3493b7535b0841a69f54af81bf5e2d8867f21704fab42e9cf24ec30c089de34f790bae4dad1fc617c3163fbf264998dc316f0a + languageName: node + linkType: hard + "@img/sharp-darwin-arm64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-darwin-arm64@npm:0.34.5" @@ -2630,6 +2646,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-darwin-arm64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-darwin-arm64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-darwin-arm64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-darwin-arm64": + optional: true + conditions: os=darwin & cpu=arm64 + languageName: node + linkType: hard + "@img/sharp-darwin-x64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-darwin-x64@npm:0.34.5" @@ -2642,6 +2670,27 @@ __metadata: languageName: node linkType: hard +"@img/sharp-darwin-x64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-darwin-x64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-darwin-x64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-darwin-x64": + optional: true + conditions: os=darwin & cpu=x64 + languageName: node + linkType: hard + +"@img/sharp-freebsd-wasm32@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-freebsd-wasm32@npm:0.35.0" + dependencies: + "@img/sharp-wasm32": "npm:0.35.0" + conditions: os=freebsd + languageName: node + linkType: hard + "@img/sharp-libvips-darwin-arm64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-darwin-arm64@npm:1.2.4" @@ -2649,6 +2698,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-darwin-arm64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-darwin-arm64@npm:1.3.0" + conditions: os=darwin & cpu=arm64 + languageName: node + linkType: hard + "@img/sharp-libvips-darwin-x64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-darwin-x64@npm:1.2.4" @@ -2656,6 +2712,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-darwin-x64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-darwin-x64@npm:1.3.0" + conditions: os=darwin & cpu=x64 + languageName: node + linkType: hard + "@img/sharp-libvips-linux-arm64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-arm64@npm:1.2.4" @@ -2663,6 +2726,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-arm64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-arm64@npm:1.3.0" + conditions: os=linux & cpu=arm64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linux-arm@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-arm@npm:1.2.4" @@ -2670,6 +2740,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-arm@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-arm@npm:1.3.0" + conditions: os=linux & cpu=arm & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linux-ppc64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-ppc64@npm:1.2.4" @@ -2677,6 +2754,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-ppc64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-ppc64@npm:1.3.0" + conditions: os=linux & cpu=ppc64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linux-riscv64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-riscv64@npm:1.2.4" @@ -2684,6 +2768,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-riscv64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-riscv64@npm:1.3.0" + conditions: os=linux & cpu=riscv64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linux-s390x@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-s390x@npm:1.2.4" @@ -2691,6 +2782,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-s390x@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-s390x@npm:1.3.0" + conditions: os=linux & cpu=s390x & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linux-x64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linux-x64@npm:1.2.4" @@ -2698,6 +2796,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linux-x64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linux-x64@npm:1.3.0" + conditions: os=linux & cpu=x64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-libvips-linuxmusl-arm64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linuxmusl-arm64@npm:1.2.4" @@ -2705,6 +2810,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linuxmusl-arm64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linuxmusl-arm64@npm:1.3.0" + conditions: os=linux & cpu=arm64 & libc=musl + languageName: node + linkType: hard + "@img/sharp-libvips-linuxmusl-x64@npm:1.2.4": version: 1.2.4 resolution: "@img/sharp-libvips-linuxmusl-x64@npm:1.2.4" @@ -2712,6 +2824,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-libvips-linuxmusl-x64@npm:1.3.0": + version: 1.3.0 + resolution: "@img/sharp-libvips-linuxmusl-x64@npm:1.3.0" + conditions: os=linux & cpu=x64 & libc=musl + languageName: node + linkType: hard + "@img/sharp-linux-arm64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-arm64@npm:0.34.5" @@ -2724,6 +2843,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-arm64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-arm64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-arm64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-arm64": + optional: true + conditions: os=linux & cpu=arm64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linux-arm@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-arm@npm:0.34.5" @@ -2736,6 +2867,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-arm@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-arm@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-arm": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-arm": + optional: true + conditions: os=linux & cpu=arm & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linux-ppc64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-ppc64@npm:0.34.5" @@ -2748,6 +2891,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-ppc64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-ppc64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-ppc64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-ppc64": + optional: true + conditions: os=linux & cpu=ppc64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linux-riscv64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-riscv64@npm:0.34.5" @@ -2760,6 +2915,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-riscv64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-riscv64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-riscv64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-riscv64": + optional: true + conditions: os=linux & cpu=riscv64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linux-s390x@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-s390x@npm:0.34.5" @@ -2772,6 +2939,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-s390x@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-s390x@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-s390x": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-s390x": + optional: true + conditions: os=linux & cpu=s390x & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linux-x64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linux-x64@npm:0.34.5" @@ -2784,6 +2963,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linux-x64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linux-x64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linux-x64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linux-x64": + optional: true + conditions: os=linux & cpu=x64 & libc=glibc + languageName: node + linkType: hard + "@img/sharp-linuxmusl-arm64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linuxmusl-arm64@npm:0.34.5" @@ -2796,6 +2987,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linuxmusl-arm64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linuxmusl-arm64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linuxmusl-arm64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linuxmusl-arm64": + optional: true + conditions: os=linux & cpu=arm64 & libc=musl + languageName: node + linkType: hard + "@img/sharp-linuxmusl-x64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-linuxmusl-x64@npm:0.34.5" @@ -2808,6 +3011,18 @@ __metadata: languageName: node linkType: hard +"@img/sharp-linuxmusl-x64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-linuxmusl-x64@npm:0.35.0" + dependencies: + "@img/sharp-libvips-linuxmusl-x64": "npm:1.3.0" + dependenciesMeta: + "@img/sharp-libvips-linuxmusl-x64": + optional: true + conditions: os=linux & cpu=x64 & libc=musl + languageName: node + linkType: hard + "@img/sharp-wasm32@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-wasm32@npm:0.34.5" @@ -2817,6 +3032,24 @@ __metadata: languageName: node linkType: hard +"@img/sharp-wasm32@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-wasm32@npm:0.35.0" + dependencies: + "@emnapi/runtime": "npm:^1.11.0" + checksum: 10/58173a1b3a596b4a450cc23561413aca76741ea9e03534daaea907c9d1438a3f765273216876fe3e03372cfd8a5b72506228f3a20d5eccb6aec1f3b7cdb0c710 + languageName: node + linkType: hard + +"@img/sharp-webcontainers-wasm32@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-webcontainers-wasm32@npm:0.35.0" + dependencies: + "@img/sharp-wasm32": "npm:0.35.0" + conditions: cpu=wasm32 + languageName: node + linkType: hard + "@img/sharp-win32-arm64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-win32-arm64@npm:0.34.5" @@ -2824,6 +3057,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-win32-arm64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-win32-arm64@npm:0.35.0" + conditions: os=win32 & cpu=arm64 + languageName: node + linkType: hard + "@img/sharp-win32-ia32@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-win32-ia32@npm:0.34.5" @@ -2831,6 +3071,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-win32-ia32@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-win32-ia32@npm:0.35.0" + conditions: os=win32 & cpu=ia32 + languageName: node + linkType: hard + "@img/sharp-win32-x64@npm:0.34.5": version: 0.34.5 resolution: "@img/sharp-win32-x64@npm:0.34.5" @@ -2838,6 +3085,13 @@ __metadata: languageName: node linkType: hard +"@img/sharp-win32-x64@npm:0.35.0": + version: 0.35.0 + resolution: "@img/sharp-win32-x64@npm:0.35.0" + conditions: os=win32 & cpu=x64 + languageName: node + linkType: hard + "@isaacs/cliui@npm:^8.0.2": version: 8.0.2 resolution: "@isaacs/cliui@npm:8.0.2" @@ -16715,7 +16969,7 @@ __metadata: react-hotkeys-hook: "npm:^5.2.1" react-markdown: "npm:^10.0.0" react-slick: "npm:^0.31.0" - sharp: "npm:0.34.5" + sharp: "npm:0.35.0" slick-carousel: "npm:^1.8.1" tiny-invariant: "npm:^1.3.3" ts-jest: "npm:^29.2.4" @@ -20939,6 +21193,15 @@ __metadata: languageName: node linkType: hard +"semver@npm:^7.8.4": + version: 7.8.5 + resolution: "semver@npm:7.8.5" + bin: + semver: bin/semver.js + checksum: 10/9b01d2ff11e6e4a4539b7ca3c5f280c8704cb397a28504469f2ed4f00ad2194748d756647362a9712fff30984d15772ab7f083108c2fb508e2096ae9e708f22c + languageName: node + linkType: hard + "serialize-javascript@npm:^6.0.1, serialize-javascript@npm:^6.0.2": version: 6.0.2 resolution: "serialize-javascript@npm:6.0.2" @@ -21005,7 +21268,94 @@ __metadata: languageName: node linkType: hard -"sharp@npm:0.34.5, sharp@npm:^0.34.5": +"sharp@npm:0.35.0": + version: 0.35.0 + resolution: "sharp@npm:0.35.0" + dependencies: + "@img/colour": "npm:^1.1.0" + "@img/sharp-darwin-arm64": "npm:0.35.0" + "@img/sharp-darwin-x64": "npm:0.35.0" + "@img/sharp-freebsd-wasm32": "npm:0.35.0" + "@img/sharp-libvips-darwin-arm64": "npm:1.3.0" + "@img/sharp-libvips-darwin-x64": "npm:1.3.0" + "@img/sharp-libvips-linux-arm": "npm:1.3.0" + "@img/sharp-libvips-linux-arm64": "npm:1.3.0" + "@img/sharp-libvips-linux-ppc64": "npm:1.3.0" + "@img/sharp-libvips-linux-riscv64": "npm:1.3.0" + "@img/sharp-libvips-linux-s390x": "npm:1.3.0" + "@img/sharp-libvips-linux-x64": "npm:1.3.0" + "@img/sharp-libvips-linuxmusl-arm64": "npm:1.3.0" + "@img/sharp-libvips-linuxmusl-x64": "npm:1.3.0" + "@img/sharp-linux-arm": "npm:0.35.0" + "@img/sharp-linux-arm64": "npm:0.35.0" + "@img/sharp-linux-ppc64": "npm:0.35.0" + "@img/sharp-linux-riscv64": "npm:0.35.0" + "@img/sharp-linux-s390x": "npm:0.35.0" + "@img/sharp-linux-x64": "npm:0.35.0" + "@img/sharp-linuxmusl-arm64": "npm:0.35.0" + "@img/sharp-linuxmusl-x64": "npm:0.35.0" + "@img/sharp-webcontainers-wasm32": "npm:0.35.0" + "@img/sharp-win32-arm64": "npm:0.35.0" + "@img/sharp-win32-ia32": "npm:0.35.0" + "@img/sharp-win32-x64": "npm:0.35.0" + detect-libc: "npm:^2.1.2" + semver: "npm:^7.8.4" + dependenciesMeta: + "@img/sharp-darwin-arm64": + optional: true + "@img/sharp-darwin-x64": + optional: true + "@img/sharp-freebsd-wasm32": + optional: true + "@img/sharp-libvips-darwin-arm64": + optional: true + "@img/sharp-libvips-darwin-x64": + optional: true + "@img/sharp-libvips-linux-arm": + optional: true + "@img/sharp-libvips-linux-arm64": + optional: true + "@img/sharp-libvips-linux-ppc64": + optional: true + "@img/sharp-libvips-linux-riscv64": + optional: true + "@img/sharp-libvips-linux-s390x": + optional: true + "@img/sharp-libvips-linux-x64": + optional: true + "@img/sharp-libvips-linuxmusl-arm64": + optional: true + "@img/sharp-libvips-linuxmusl-x64": + optional: true + "@img/sharp-linux-arm": + optional: true + "@img/sharp-linux-arm64": + optional: true + "@img/sharp-linux-ppc64": + optional: true + "@img/sharp-linux-riscv64": + optional: true + "@img/sharp-linux-s390x": + optional: true + "@img/sharp-linux-x64": + optional: true + "@img/sharp-linuxmusl-arm64": + optional: true + "@img/sharp-linuxmusl-x64": + optional: true + "@img/sharp-webcontainers-wasm32": + optional: true + "@img/sharp-win32-arm64": + optional: true + "@img/sharp-win32-ia32": + optional: true + "@img/sharp-win32-x64": + optional: true + checksum: 10/391d1212df0a8ac61c4bdf3b745a8530b050ff71c198038ece5eccfa013cbb5bc0735dba71d47aeeb4f82a86d2bc4dc212e44b397cbb6afa285624a6bd069516 + languageName: node + linkType: hard + +"sharp@npm:^0.34.5": version: 0.34.5 resolution: "sharp@npm:0.34.5" dependencies: From 22beb081679165a258f20d9ecfb518614799577b Mon Sep 17 00:00:00 2001 From: Shankar Ambady Date: Thu, 13 Aug 2026 16:33:28 -0400 Subject: [PATCH 2/5] Vector search performance enhancements (#3754) * adding toggle setting for retrieving resources results from payload * add constants for excluding items from payload * fetch directly from payload * ensure vector search does not prefetch opensearch results * make sure nested content file is retrieved * remove irrelevant field --- .../main/src/app/(site)/search/page.test.tsx | 77 ++++++ frontends/main/src/app/(site)/search/page.tsx | 31 ++- .../SearchDisplay/HybridSearchDisplay.tsx | 105 +------ .../SearchDisplay/vectorSearchParams.ts | 97 +++++++ main/settings.py | 7 + vector_search/constants.py | 31 +++ vector_search/utils.py | 91 ++++++ vector_search/utils_test.py | 261 +++++++++++++++++- vector_search/views.py | 14 +- vector_search/views_test.py | 167 ++++++++++- 10 files changed, 760 insertions(+), 121 deletions(-) create mode 100644 frontends/main/src/app/(site)/search/page.test.tsx create mode 100644 frontends/main/src/page-components/SearchDisplay/vectorSearchParams.ts diff --git a/frontends/main/src/app/(site)/search/page.test.tsx b/frontends/main/src/app/(site)/search/page.test.tsx new file mode 100644 index 0000000000..d99a03042a --- /dev/null +++ b/frontends/main/src/app/(site)/search/page.test.tsx @@ -0,0 +1,77 @@ +import { factories, makeRequest, setMockResponse, urls } from "api/test-utils" +import Page from "./page" + +jest.mock("@/app/getQueryClient", () => { + const { makeBrowserQueryClient } = jest.requireActual("@/app/getQueryClient") + return { getQueryClient: () => makeBrowserQueryClient({ maxRetries: 0 }) } +}) + +const SEARCH_RESPONSE = { + count: 0, + next: null, + previous: null, + results: [], + metadata: { + aggregations: {}, + suggestions: [], + }, +} + +beforeEach(() => { + setMockResponse.get( + urls.offerors.list(), + factories.learningResources.offerors({ count: 0 }), + ) + setMockResponse.get(expect.stringContaining(urls.search.resources()), { + ...SEARCH_RESPONSE, + }) + setMockResponse.get(expect.stringContaining(urls.search.vectorResources()), { + ...SEARCH_RESPONSE, + }) +}) + +test("prefetches OpenSearch results by default", async () => { + await Page({ + params: Promise.resolve({}), + searchParams: Promise.resolve({ q: "test" }), + }) + + expect( + makeRequest.mock.calls.some(([args]) => + args.url.startsWith(urls.search.resources()), + ), + ).toBe(true) + expect( + makeRequest.mock.calls.some(([args]) => + args.url.startsWith(urls.search.vectorResources()), + ), + ).toBe(false) +}) + +test("prefetches vector results when vector_search is enabled", async () => { + await Page({ + params: Promise.resolve({}), + searchParams: Promise.resolve({ + q: "test", + vector_search: "true", + topic: "Physics", + }), + }) + + const vectorCall = makeRequest.mock.calls.find(([args]) => + args.url.startsWith(urls.search.vectorResources()), + ) + expect(vectorCall).toBeDefined() + expect( + makeRequest.mock.calls.some(([args]) => + args.url.startsWith(urls.search.resources()), + ), + ).toBe(false) + + const searchParams = new URL(vectorCall?.[0].url ?? "").searchParams + expect(searchParams.get("hybrid_search")).toBe("true") + expect(searchParams.get("q")).toBe("test") + expect(searchParams.has("topic")).toBe(false) + expect(searchParams.has("limit")).toBe(false) + expect(searchParams.has("offset")).toBe(false) +}) diff --git a/frontends/main/src/app/(site)/search/page.tsx b/frontends/main/src/app/(site)/search/page.tsx index f222ddc07f..c2861c44c9 100644 --- a/frontends/main/src/app/(site)/search/page.tsx +++ b/frontends/main/src/app/(site)/search/page.tsx @@ -11,6 +11,10 @@ import { getExtraFacetNames, } from "@/app-pages/SearchPage/searchRequests" import getSearchParams from "@/page-components/SearchDisplay/getSearchParams" +import { + toUnfacetedVectorSearchParams, + toVectorSearchParams, +} from "@/page-components/SearchDisplay/vectorSearchParams" import validateRequestParams from "@/page-components/SearchDisplay/validateRequestParams" import type { ResourceSearchRequest } from "@/page-components/SearchDisplay/validateRequestParams" import { LearningResourcesSearchApiLearningResourcesSearchRetrieveRequest as LRSearchRequest } from "api" @@ -50,13 +54,28 @@ const Page: React.FC> = async ({ searchParams }) => { }) const queryClient = getQueryClient() + const isVectorSearch = urlParams.get("vector_search") === "true" + const hasSearchTerm = typeof params.q === "string" && params.q.trim() !== "" - await Promise.all([ - queryClient.prefetchQuery(offerorQueries.list({})), - queryClient.prefetchQuery( - learningResourceQueries.search(params as LRSearchRequest), - ), - ]) + if (isVectorSearch) { + await Promise.all([ + queryClient.prefetchQuery(offerorQueries.list({})), + queryClient.prefetchQuery( + learningResourceQueries.vectorSearch( + hasSearchTerm + ? toUnfacetedVectorSearchParams(params) + : toVectorSearchParams(params), + ), + ), + ]) + } else { + await Promise.all([ + queryClient.prefetchQuery(offerorQueries.list({})), + queryClient.prefetchQuery( + learningResourceQueries.search(params as LRSearchRequest), + ), + ]) + } return ( diff --git a/frontends/main/src/page-components/SearchDisplay/HybridSearchDisplay.tsx b/frontends/main/src/page-components/SearchDisplay/HybridSearchDisplay.tsx index 49857e5ab8..075f1b705e 100644 --- a/frontends/main/src/page-components/SearchDisplay/HybridSearchDisplay.tsx +++ b/frontends/main/src/page-components/SearchDisplay/HybridSearchDisplay.tsx @@ -1,107 +1,14 @@ import React, { useMemo } from "react" import { learningResourceQueries } from "api/hooks/learningResources" import type { LearningResource } from "api" -import type { Facets, BooleanFacets } from "@mitodl/course-search-utils" -import type { - LearningResourcesVectorSearchResponse, - VectorLearningResourcesSearchApiVectorLearningResourcesSearchRetrieveRequest as VectorSearchRequest, -} from "api/v0" +import type { LearningResourcesVectorSearchResponse } from "api/v0" import getSearchParams from "./getSearchParams" import SearchDisplay, { SearchDisplayProps } from "./SearchDisplay" - -const mapVectorSortby = ( - sortby?: string, -): VectorSearchRequest["sortby"] | undefined => { - switch (sortby) { - case "-views": - case "popular": - return "-views" - case "upcoming": - return "next_start_date" - case "new": - return "-created_on" - default: - return undefined - } -} - -/** - * Extracts only the fields supported by the vector search API from a broader - * search params object, dropping admin-only params (e.g., content_file_score_weight) - * that the vector endpoint does not accept. - * - * The `as` casts for enum arrays are safe because the v0 and v1 generated - * clients define separate (but structurally identical) enum types for the same - * string-literal values (e.g., delivery: 'online' | 'hybrid' | ...). - */ -const toVectorSearchParams = ( - params: ReturnType & { sortby?: string }, - cutoffScore?: number, -): VectorSearchRequest => ({ - aggregations: params.aggregations as VectorSearchRequest["aggregations"], - certification: params.certification, - certification_type: - params.certification_type as VectorSearchRequest["certification_type"], - course_feature: params.course_feature, - delivery: params.delivery as VectorSearchRequest["delivery"], - department: params.department as VectorSearchRequest["department"], - free: params.free, - level: params.level as VectorSearchRequest["level"], - limit: params.limit, - ocw_topic: params.ocw_topic, - offered_by: params.offered_by as VectorSearchRequest["offered_by"], - offset: params.offset, - platform: params.platform as VectorSearchRequest["platform"], - professional: params.professional, - q: params.q, - resource_category: - params.resource_category as VectorSearchRequest["resource_category"], - resource_type: params.resource_type as VectorSearchRequest["resource_type"], - resource_type_group: - params.resource_type_group as VectorSearchRequest["resource_type_group"], - score_cutoff: cutoffScore, - sortby: mapVectorSortby(params.sortby), - topic: params.topic, - hybrid_search: true, -}) - -const VECTOR_CLIENT_FILTER_FACETS = [ - "resource_type", - "certification_type", - "delivery", - "department", - "topic", - "offered_by", - "free", - "professional", - "resource_category", - "resource_type_group", - "level", - "platform", - "course_feature", -] as const - -type VectorClientFilterFacet = (typeof VECTOR_CLIENT_FILTER_FACETS)[number] - -const toUnfacetedVectorSearchParams = ( - params: ReturnType & { sortby?: string }, - constantSearchParams: Facets & BooleanFacets = {}, - cutoffScore?: number, -): VectorSearchRequest => { - const { - offset: _offset, - limit: _limit, - ...vectorParams - } = toVectorSearchParams(params, cutoffScore) - - return Object.fromEntries( - Object.entries(vectorParams).filter( - ([key]) => - !VECTOR_CLIENT_FILTER_FACETS.includes(key as VectorClientFilterFacet) || - key in constantSearchParams, - ), - ) as VectorSearchRequest -} +import { + VECTOR_CLIENT_FILTER_FACETS, + toUnfacetedVectorSearchParams, + toVectorSearchParams, +} from "./vectorSearchParams" const normalizeParamValues = (value: unknown): string[] => { if (Array.isArray(value)) { diff --git a/frontends/main/src/page-components/SearchDisplay/vectorSearchParams.ts b/frontends/main/src/page-components/SearchDisplay/vectorSearchParams.ts new file mode 100644 index 0000000000..edf587f283 --- /dev/null +++ b/frontends/main/src/page-components/SearchDisplay/vectorSearchParams.ts @@ -0,0 +1,97 @@ +import type { Facets, BooleanFacets } from "@mitodl/course-search-utils" +import type { VectorLearningResourcesSearchApiVectorLearningResourcesSearchRetrieveRequest as VectorSearchRequest } from "api/v0" +import getSearchParams from "./getSearchParams" + +const mapVectorSortby = ( + sortby?: string, +): VectorSearchRequest["sortby"] | undefined => { + switch (sortby) { + case "-views": + case "popular": + return "-views" + case "upcoming": + return "next_start_date" + case "new": + return "-created_on" + default: + return undefined + } +} + +/** + * Extracts only the fields supported by the vector search API from a broader + * search params object, dropping admin-only params (e.g., content_file_score_weight) + * that the vector endpoint does not accept. + * + * The `as` casts for enum arrays are safe because the v0 and v1 generated + * clients define separate (but structurally identical) enum types for the same + * string-literal values (e.g., delivery: 'online' | 'hybrid' | ...). + */ +export const toVectorSearchParams = ( + params: ReturnType & { sortby?: string }, + cutoffScore?: number, +): VectorSearchRequest => ({ + aggregations: params.aggregations as VectorSearchRequest["aggregations"], + certification: params.certification, + certification_type: + params.certification_type as VectorSearchRequest["certification_type"], + course_feature: params.course_feature, + delivery: params.delivery as VectorSearchRequest["delivery"], + department: params.department as VectorSearchRequest["department"], + free: params.free, + level: params.level as VectorSearchRequest["level"], + limit: params.limit, + ocw_topic: params.ocw_topic, + offered_by: params.offered_by as VectorSearchRequest["offered_by"], + offset: params.offset, + platform: params.platform as VectorSearchRequest["platform"], + professional: params.professional, + q: params.q, + resource_category: + params.resource_category as VectorSearchRequest["resource_category"], + resource_type: params.resource_type as VectorSearchRequest["resource_type"], + resource_type_group: + params.resource_type_group as VectorSearchRequest["resource_type_group"], + score_cutoff: cutoffScore, + sortby: mapVectorSortby(params.sortby), + topic: params.topic, + hybrid_search: true, +}) + +export const VECTOR_CLIENT_FILTER_FACETS = [ + "resource_type", + "certification_type", + "delivery", + "department", + "topic", + "offered_by", + "free", + "professional", + "resource_category", + "resource_type_group", + "level", + "platform", + "course_feature", +] as const + +type VectorClientFilterFacet = (typeof VECTOR_CLIENT_FILTER_FACETS)[number] + +export const toUnfacetedVectorSearchParams = ( + params: ReturnType & { sortby?: string }, + constantSearchParams: Facets & BooleanFacets = {}, + cutoffScore?: number, +): VectorSearchRequest => { + const { + offset: _offset, + limit: _limit, + ...vectorParams + } = toVectorSearchParams(params, cutoffScore) + + return Object.fromEntries( + Object.entries(vectorParams).filter( + ([key]) => + !VECTOR_CLIENT_FILTER_FACETS.includes(key as VectorClientFilterFacet) || + key in constantSearchParams, + ), + ) as VectorSearchRequest +} diff --git a/main/settings.py b/main/settings.py index df5afa6c74..2d8419677e 100644 --- a/main/settings.py +++ b/main/settings.py @@ -835,6 +835,13 @@ def get_all_config_keys(): # hard limit for special cases where we need to return all results without pagination VECTOR_SEARCH_PAGE_MAX_LIMIT = get_int("VECTOR_SEARCH_PAGE_MAX_LIMIT", 200) +# serve learning resource search hits from the Qdrant payload instead of +# re-hydrating them from the database. Set to False to fall back to database +# hydration without a deploy. +VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = get_bool( + name="VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD", default=True +) + # toggle to use requests (default for local) or webdriver which renders js elements EMBEDDINGS_EXTERNAL_FETCH_USE_WEBDRIVER = get_bool( "EMBEDDINGS_EXTERNAL_FETCH_USE_WEBDRIVER", default=False diff --git a/vector_search/constants.py b/vector_search/constants.py index 85ef1f42dc..9d082b4b5f 100644 --- a/vector_search/constants.py +++ b/vector_search/constants.py @@ -149,6 +149,37 @@ CONTENT_FILES_RETRIEVE_PAYLOAD = True RESOURCES_RETRIEVE_PAYLOAD = ["readable_id", "platform"] +# Payload keys dropped when resource hits are served straight from the Qdrant +# payload (VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD): what the indexing serializer +# adds on top of the LearningResourceSerializer shape the API returns, plus +# video.transcript, which the response never renders. +# +# content_files is NOT excluded. Document and video responses declare it +# (NestedContentFileSerializer), and search cards fall back to +# content_files[0].image_src for the thumbnail when the resource has no image. +# The indexing serializer re-serializes it with the *full* ContentFileSerializer, +# so its large text fields are trimmed in Python instead -- see +# _trim_indexing_only_list_fields. +# +# This is deliberately not a copy of SOURCE_EXCLUDED_FIELDS: vector_embedding is +# an OpenSearch-only field (grafted on by +# serialize_bulk_learning_resources_with_embeddings), and in Qdrant the dense +# vector lives on the point, not in the payload. +RESOURCES_PAYLOAD_EXCLUDE = [ + "_id", + "resource_relations", + "is_learning_material", + "resource_age_date", + "featured_rank", + "is_incomplete_or_stale", + "video.transcript", +] + +# Qdrant payload selectors descend into objects but not into lists of objects, +# so the extra fields SearchCourseNumberSerializer puts on each course number +# cannot be named in RESOURCES_PAYLOAD_EXCLUDE and are trimmed in Python. +COURSE_NUMBER_INDEXING_ONLY_FIELDS = frozenset({"sort_coursenum", "primary"}) + COLLECTION_PARAM_MAP = { RESOURCES_COLLECTION_NAME: QDRANT_RESOURCE_PARAM_MAP, diff --git a/vector_search/utils.py b/vector_search/utils.py index accd72d1ca..ba3d574944 100644 --- a/vector_search/utils.py +++ b/vector_search/utils.py @@ -12,6 +12,7 @@ from qdrant_client import AsyncQdrantClient, QdrantClient, models from learning_resources.constants import ( + CONTENT_FILE_LARGE_FIELDS, PROGRAM_COURSE_CACHE_KEY_TEST_MODE, ) from learning_resources.models import ( @@ -42,6 +43,7 @@ from vector_search.constants import ( COLLECTION_PARAM_MAP, CONTENT_FILES_COLLECTION_NAME, + COURSE_NUMBER_INDEXING_ONLY_FIELDS, QDRANT_CONTENT_FILE_INDEXES, QDRANT_CONTENT_FILE_PARAM_MAP, QDRANT_LEARNING_RESOURCE_INDEXES, @@ -60,6 +62,8 @@ QDRANT_RESOURCE_PARAM_MAP, QDRANT_TOPIC_INDEXES, RESOURCES_COLLECTION_NAME, + RESOURCES_PAYLOAD_EXCLUDE, + RESOURCES_RETRIEVE_PAYLOAD, TOPICS_COLLECTION_NAME, VECTOR_SEARCH_SCORE_BOOST, ) @@ -1150,6 +1154,93 @@ def process_batch(docs_batch): ) +def resources_payload_selector(): + """ + Return the `with_payload` value to use for the resources collection. + + When hits are served from the payload we want everything the API response + needs, minus the indexing-only keys. Otherwise we only need the two fields + the database hydration path looks resources up by. + """ + if settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD: + return models.PayloadSelectorExclude(exclude=RESOURCES_PAYLOAD_EXCLUDE) + return RESOURCES_RETRIEVE_PAYLOAD + + +def _without_keys(items, drop_keys): + """Drop drop_keys from every dict in a list, leaving non-dicts alone""" + return [ + {key: value for key, value in item.items() if key not in drop_keys} + if isinstance(item, dict) + else item + for item in items + ] + + +def _trim_indexing_only_list_fields(payload): + """ + Drop indexing-only keys the Qdrant payload selector cannot reach. + + Selectors descend into objects but not into lists of objects, so anything + the indexing serializer adds *inside* a list survives the exclude and has + to be removed here: + + - course.course_numbers[] carries sort_coursenum and primary, which + SearchCourseNumberSerializer adds on top of CourseNumberSerializer. + - content_files[] is re-serialized with the full ContentFileSerializer for + nested search, so it carries the large text fields that the API's + NestedContentFileSerializer omits. The rest of the field must survive: + document and video responses declare it, and search cards use + content_files[0].image_src as the thumbnail fallback. + """ + trimmed = payload + + course = payload.get("course") + if isinstance(course, dict) and isinstance(course.get("course_numbers"), list): + trimmed = { + **trimmed, + "course": { + **course, + "course_numbers": _without_keys( + course["course_numbers"], COURSE_NUMBER_INDEXING_ONLY_FIELDS + ), + }, + } + + content_files = payload.get("content_files") + if isinstance(content_files, list): + trimmed = { + **trimmed, + "content_files": _without_keys(content_files, CONTENT_FILE_LARGE_FIELDS), + } + + return trimmed + + +def _resource_payload_hits(search_result): + """ + Build resource hits from the Qdrant payloads themselves. + + The payload is the resource as the indexing serializer wrote it, so it + already carries every field the API response needs -- no database + hydration required. Dedupes on platform:readable_id and preserves the + Qdrant ranking, the same way the hydrated path does. + """ + hits = [] + seen = set() + for hit in search_result: + payload = hit.payload or {} + readable_id = payload.get("readable_id") + if not readable_id: + continue + key = f"{(payload.get('platform') or {}).get('code', '')}:{readable_id}" + if key in seen: + continue + seen.add(key) + hits.append(_trim_indexing_only_list_fields(payload)) + return hits + + def _resource_vector_hits(search_result): readable_ids = [ hit.payload.get("readable_id") diff --git a/vector_search/utils_test.py b/vector_search/utils_test.py index faf88c7796..e0b10faded 100644 --- a/vector_search/utils_test.py +++ b/vector_search/utils_test.py @@ -15,7 +15,9 @@ import vector_search.utils as vs_utils from learning_resources.constants import ( + CONTENT_FILE_LARGE_FIELDS, GROUP_CONTENT_FILE_CONTENT_VIEWERS, + LearningResourceType, ) from learning_resources.factories import ( ContentFileFactory, @@ -25,7 +27,7 @@ LearningResourceRunFactory, LearningResourceTopicFactory, ) -from learning_resources.models import LearningResource +from learning_resources.models import ContentFile, LearningResource from learning_resources.serializers import LearningResourceMetadataDisplaySerializer from learning_resources_search.constants import ( CONTENT_FILE_TYPE, @@ -55,6 +57,8 @@ QDRANT_OPTIMIZER_THRESHOLD_SMALL, QDRANT_RESOURCE_PARAM_MAP, RESOURCES_COLLECTION_NAME, + RESOURCES_PAYLOAD_EXCLUDE, + RESOURCES_RETRIEVE_PAYLOAD, ) from vector_search.encoders.utils import dense_encoder, sparse_encoder from vector_search.utils import ( @@ -65,6 +69,7 @@ _generate_content_file_points, _get_text_splitter, _is_markdown_content, + _resource_payload_hits, _resource_vector_hits, _set_payload, async_qdrant_aggregations, @@ -77,6 +82,7 @@ filter_existing_qdrant_points, qdrant_query_conditions, remove_qdrant_records, + resources_payload_selector, should_generate_content_embeddings, should_generate_resource_embeddings, update_content_file_payload, @@ -2343,6 +2349,259 @@ def test_resource_vector_hits_duplicate_readable_ids_different_platforms(): assert result_2[1]["platform"]["code"] == "xpro" +def test_resources_payload_selector_excludes_indexing_fields(settings): + """The selector should ask for the whole payload minus indexing-only keys""" + settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = True + selector = resources_payload_selector() + assert isinstance(selector, models.PayloadSelectorExclude) + assert selector.exclude == RESOURCES_PAYLOAD_EXCLUDE + + +def test_resources_payload_selector_kill_switch(settings): + """With payload hits disabled we only fetch the DB hydration lookup fields""" + settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = False + assert resources_payload_selector() == RESOURCES_RETRIEVE_PAYLOAD + + +def test_resource_payload_hits_preserves_order_and_dedupes(): + """Hits come straight from the payloads, in Qdrant order, deduped by platform:id""" + search_result = [ + MagicMock( + payload={ + "readable_id": "course-2", + "platform": {"code": "ocw"}, + "title": "Second", + } + ), + MagicMock( + payload={ + "readable_id": "course-1", + "platform": {"code": "ocw"}, + "title": "First", + } + ), + # same readable_id as the first hit, different platform: kept + MagicMock( + payload={ + "readable_id": "course-2", + "platform": {"code": "xpro"}, + "title": "Second on xpro", + } + ), + # exact duplicate of the first hit: dropped + MagicMock( + payload={ + "readable_id": "course-2", + "platform": {"code": "ocw"}, + "title": "Second", + } + ), + # unusable without a readable_id: dropped + MagicMock(payload={"platform": {"code": "ocw"}, "title": "No readable id"}), + ] + + hits = _resource_payload_hits(search_result) + + assert [(hit["readable_id"], hit["platform"]["code"]) for hit in hits] == [ + ("course-2", "ocw"), + ("course-1", "ocw"), + ("course-2", "xpro"), + ] + assert hits[0]["title"] == "Second" + + +def test_resource_payload_hits_handles_null_platform(): + """A resource indexed without a platform should still produce a hit""" + hits = _resource_payload_hits( + [MagicMock(payload={"readable_id": "course-1", "platform": None})] + ) + assert [hit["readable_id"] for hit in hits] == ["course-1"] + + +def test_resource_payload_hits_trims_indexing_only_course_number_fields(): + """ + Qdrant payload selectors cannot descend into lists of objects, so the extra + course number fields the indexing serializer adds are trimmed in Python. + """ + payload = { + "readable_id": "course-1", + "platform": {"code": "ocw"}, + "course": { + "course_numbers": [ + { + "value": "6.006", + "listing_type": "Primary", + "department": {"department_id": "6"}, + "primary": True, + "sort_coursenum": "06.006", + } + ] + }, + } + + hits = _resource_payload_hits([MagicMock(payload=payload)]) + + assert hits[0]["course"]["course_numbers"] == [ + { + "value": "6.006", + "listing_type": "Primary", + "department": {"department_id": "6"}, + } + ] + # the payload dict Qdrant handed us is not mutated + assert "sort_coursenum" in payload["course"]["course_numbers"][0] + + +@pytest.mark.parametrize( + "course", + [None, {}, {"course_numbers": None}], +) +def test_resource_payload_hits_tolerates_missing_course_numbers(course): + """Non-course resources pass through the course number trim untouched""" + hits = _resource_payload_hits( + [MagicMock(payload={"readable_id": "video-1", "course": course})] + ) + assert hits[0]["course"] == course + + +def _add_direct_content_files(resource, count=2, **kwargs): + """ + Attach the direct content files that video/document responses nest. + + ContentFileFactory._create always fills in run or learning_resource, but the + model's check constraint requires a direct content file to have neither, so + the foreign key is moved after creation. + """ + content_files = ContentFileFactory.create_batch( + count, learning_resource=resource, **kwargs + ) + ContentFile.objects.filter(id__in=[cf.id for cf in content_files]).update( + learning_resource=None, direct_learning_resource=resource + ) + return content_files + + +def _payload_as_search_sees_it(resource_id): + """ + Return the indexed payload minus what PayloadSelectorExclude strips, + i.e. exactly what _resource_payload_hits receives from a search. + """ + payload = next(iter(serialize_bulk_learning_resources([resource_id]))) + for excluded in RESOURCES_PAYLOAD_EXCLUDE: + top_level, _, nested = excluded.partition(".") + if nested: + if isinstance(payload.get(top_level), dict): + payload[top_level].pop(nested, None) + else: + payload.pop(top_level, None) + return payload + + +def test_content_files_is_not_excluded_from_the_payload(): + """ + content_files must stay in the payload: document and video responses declare + it, and search cards fall back to content_files[0].image_src for the + thumbnail. Its large text fields are trimmed in Python instead, because a + Qdrant payload selector cannot descend into a list of objects. + """ + assert "content_files" not in RESOURCES_PAYLOAD_EXCLUDE + + +@pytest.mark.parametrize( + ("factory_kwargs", "has_content_files"), + [ + ({"is_course": True}, False), + ({"is_video": True}, True), + ({"resource_type": LearningResourceType.document.name}, True), + ], +) +def test_resource_payload_hits_matches_hydrated_hits(factory_kwargs, has_content_files): + """ + The payload path should return what the database hydration path returns, + modulo the fields the indexing serializer adds on top of the API shape -- + including the nested content_files that document and video responses + declare. + """ + resource = LearningResourceFactory.create(**factory_kwargs) + if has_content_files: + _add_direct_content_files( + resource, image_src="https://img.youtube.com/thumb.jpg" + ) + + payload = _payload_as_search_sees_it(resource.id) + hydrated = _resource_vector_hits( + [ + MagicMock( + payload={ + "readable_id": resource.readable_id, + "platform": { + "code": resource.platform.code if resource.platform else "" + }, + } + ) + ] + ) + from_payload = _resource_payload_hits([MagicMock(payload=payload)]) + + assert len(from_payload) == 1 + assert set(from_payload[0]) == set(hydrated[0]) + + if has_content_files: + # the nested field must carry the API's shape, not the indexing shape + assert from_payload[0]["content_files"] + assert {frozenset(cf) for cf in from_payload[0]["content_files"]} == { + frozenset(cf) for cf in hydrated[0]["content_files"] + } + + +@pytest.mark.parametrize( + "resource_type", + [LearningResourceType.video.name, LearningResourceType.document.name], +) +def test_resource_payload_hits_keeps_content_files_thumbnail_fallback(resource_type): + """ + Search cards use content_files[0].image_src as the thumbnail when the + resource has no image, so the payload path must keep the nested content + files -- minus the large text the indexing serializer re-adds. + """ + payload = { + "readable_id": f"{resource_type}-1", + "platform": {"code": "youtube"}, + "resource_type": resource_type, + "image": None, + "content_files": [ + { + "id": 1, + "key": "lecture.pdf", + "title": "Lecture", + "image_src": "https://img.youtube.com/thumb.jpg", + "content": "the full extracted text, many kilobytes of it", + "summary": "a generated summary", + "flashcards": [{"question": "q", "answer": "a"}], + } + ], + } + + hits = _resource_payload_hits([MagicMock(payload=payload)]) + content_file = hits[0]["content_files"][0] + + assert content_file["image_src"] == "https://img.youtube.com/thumb.jpg" + assert content_file["key"] == "lecture.pdf" + assert content_file["title"] == "Lecture" + assert set(CONTENT_FILE_LARGE_FIELDS).isdisjoint(content_file) + # the payload dict Qdrant handed us is not mutated + assert "content" in payload["content_files"][0] + + +@pytest.mark.parametrize("content_files", [None, [], "not-a-list"]) +def test_resource_payload_hits_tolerates_odd_content_files(content_files): + """Resources without nested content files pass through untouched""" + hits = _resource_payload_hits( + [MagicMock(payload={"readable_id": "c-1", "content_files": content_files})] + ) + assert hits[0]["content_files"] == content_files + + def _make_facet_hit(count=0, value="test"): """Build a minimal mock that looks like a Qdrant FacetHit.""" hit = MagicMock() diff --git a/vector_search/views.py b/vector_search/views.py index 92d140bffd..bb8c747356 100644 --- a/vector_search/views.py +++ b/vector_search/views.py @@ -23,7 +23,6 @@ CONTENT_FILES_RETRIEVE_PAYLOAD, QDRANT_RESOURCE_PARAM_MAP, RESOURCES_COLLECTION_NAME, - RESOURCES_RETRIEVE_PAYLOAD, ) from vector_search.serializers import ( ContentFileVectorSearchRequestSerializer, @@ -34,6 +33,7 @@ from vector_search.utils import ( _content_file_vector_hits, _merge_dicts, + _resource_payload_hits, _resource_vector_hits, async_qdrant_aggregations, async_qdrant_client, @@ -43,6 +43,7 @@ db_sync_to_async, dense_encoder, qdrant_query_conditions, + resources_payload_selector, sparse_encoder, ) @@ -150,7 +151,7 @@ async def _build_search_params( # noqa: PLR0913 "collection_name": search_collection, "query_filter": search_filter, "with_vectors": False, - "with_payload": RESOURCES_RETRIEVE_PAYLOAD + "with_payload": resources_payload_selector() if search_collection == RESOURCES_COLLECTION_NAME else CONTENT_FILES_RETRIEVE_PAYLOAD, "search_params": models.SearchParams( @@ -276,6 +277,11 @@ async def _execute_scroll_search( # noqa: PLR0913 "collection_name": search_collection, "scroll_filter": search_filter, "with_vectors": False, + # Scroll otherwise defaults to the entire payload, transcripts and + # all -- ask for the same fields the query path does. + "with_payload": resources_payload_selector() + if search_collection == RESOURCES_COLLECTION_NAME + else CONTENT_FILES_RETRIEVE_PAYLOAD, } if order_by: @@ -389,6 +395,10 @@ async def _async_vector_hits( # noqa: PLR0913 ) if search_collection == RESOURCES_COLLECTION_NAME: + if settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD: + # Payloads are already the serialized resources -- no database + # round trip, so no thread hop either. + return _resource_payload_hits(search_result) return await db_sync_to_async(_resource_vector_hits)(search_result) else: return await db_sync_to_async(_content_file_vector_hits)(search_result) diff --git a/vector_search/views_test.py b/vector_search/views_test.py index 1d1ec2e239..50038d1784 100644 --- a/vector_search/views_test.py +++ b/vector_search/views_test.py @@ -14,6 +14,12 @@ LearningResourceFactory, LearningResourceRunFactory, ) +from learning_resources_search.serializers import serialize_bulk_learning_resources +from vector_search.constants import ( + CONTENT_FILES_RETRIEVE_PAYLOAD, + RESOURCES_PAYLOAD_EXCLUDE, + RESOURCES_RETRIEVE_PAYLOAD, +) from vector_search.encoders.utils import dense_encoder, sparse_encoder from vector_search.views import QdrantView @@ -749,11 +755,11 @@ def test_vector_search_sortby_with_score_cutoff_manually_sorted(mocker, client): mock_result = mocker.MagicMock() mock_point_1 = mocker.MagicMock() - mock_point_1.payload = {"readable_id": "course-1"} + mock_point_1.payload = {"readable_id": "course-1", "views": 100} mock_point_2 = mocker.MagicMock() - mock_point_2.payload = {"readable_id": "course-2"} + mock_point_2.payload = {"readable_id": "course-2", "views": 50} mock_point_3 = mocker.MagicMock() - mock_point_3.payload = {"readable_id": "course-3"} + mock_point_3.payload = {"readable_id": "course-3", "views": 200} mock_result.points = [mock_point_1, mock_point_2, mock_point_3] mock_qdrant.query_points = mocker.AsyncMock(return_value=mock_result) @@ -763,16 +769,6 @@ def test_vector_search_sortby_with_score_cutoff_manually_sorted(mocker, client): return_value=mock_qdrant, ) - mock_hits = [ - {"readable_id": "course-1", "views": 100}, - {"readable_id": "course-2", "views": 50}, - {"readable_id": "course-3", "views": 200}, - ] - mocker.patch( - "vector_search.views._resource_vector_hits", - return_value=mock_hits, - ) - # Test descending sort: sortby=-views params = { "hybrid_search": "true", @@ -1256,3 +1252,148 @@ def test_content_file_vector_search_count_is_approximate( mock_qdrant.count.assert_awaited() assert mock_qdrant.count.await_args.kwargs["exact"] is False + + +@pytest.mark.parametrize("from_payload", [True, False]) +def test_vector_search_payload_selector(mocker, client, settings, from_payload): + """ + Resource searches request the trimmed full payload when payload hits are + enabled, and only the two hydration lookup fields when they are not. + """ + settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = from_payload + mock_qdrant = mocker.patch( + "qdrant_client.AsyncQdrantClient", return_value=mocker.AsyncMock() + )() + empty = mocker.MagicMock() + empty.points = [] + mock_qdrant.query_points = mocker.AsyncMock(return_value=empty) + mock_qdrant.scroll = mocker.AsyncMock(return_value=([], None)) + mock_qdrant.count = mocker.AsyncMock(return_value=CountResult(count=0)) + mocker.patch("vector_search.views.async_qdrant_client", return_value=mock_qdrant) + + client.get( + reverse("vector_search:v0:vector_learning_resources_search"), + data={"q": "test"}, + ) + + with_payload = mock_qdrant.query_points.mock_calls[0].kwargs["with_payload"] + if from_payload: + assert with_payload == models.PayloadSelectorExclude( + exclude=RESOURCES_PAYLOAD_EXCLUDE + ) + else: + assert with_payload == RESOURCES_RETRIEVE_PAYLOAD + + +@pytest.mark.parametrize("from_payload", [True, False]) +def test_vector_search_scroll_payload_selector(mocker, client, settings, from_payload): + """ + The scroll path (no query string) must use the same selector; it otherwise + defaults to the entire payload, transcripts included. + """ + settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = from_payload + mock_qdrant = mocker.patch( + "qdrant_client.AsyncQdrantClient", return_value=mocker.AsyncMock() + )() + mock_qdrant.scroll = mocker.AsyncMock(return_value=([], None)) + mock_qdrant.count = mocker.AsyncMock(return_value=CountResult(count=0)) + mocker.patch("vector_search.views.async_qdrant_client", return_value=mock_qdrant) + + client.get( + reverse("vector_search:v0:vector_learning_resources_search"), + data={"q": ""}, + ) + + with_payload = mock_qdrant.scroll.mock_calls[0].kwargs["with_payload"] + if from_payload: + assert with_payload == models.PayloadSelectorExclude( + exclude=RESOURCES_PAYLOAD_EXCLUDE + ) + else: + assert with_payload == RESOURCES_RETRIEVE_PAYLOAD + + +@pytest.mark.django_db(transaction=True) +def test_content_file_vector_search_scroll_keeps_full_payload( + mocker, client, content_file_viewer +): + """Content file search is unaffected: it still scrolls the whole payload""" + mock_qdrant = mocker.patch( + "qdrant_client.AsyncQdrantClient", return_value=mocker.AsyncMock() + )() + mock_qdrant.scroll = mocker.AsyncMock(return_value=([], None)) + mock_qdrant.count = mocker.AsyncMock(return_value=CountResult(count=0)) + mocker.patch("vector_search.views.async_qdrant_client", return_value=mock_qdrant) + + client.get(reverse("vector_search:v0:vector_content_files_search"), data={"q": ""}) + + assert ( + mock_qdrant.scroll.mock_calls[0].kwargs["with_payload"] + == CONTENT_FILES_RETRIEVE_PAYLOAD + ) + + +@pytest.mark.django_db(transaction=True) +def test_vector_search_returns_payload_is_not_hydrated(mocker, client): + """ + With payload hits enabled the response is built from the Qdrant payload + rather than re-fetched from the database. + """ + resource = LearningResourceFactory.create(is_course=True) + payload = next(iter(serialize_bulk_learning_resources([resource.id]))) + + mock_qdrant = mocker.patch( + "qdrant_client.AsyncQdrantClient", return_value=mocker.AsyncMock() + )() + mock_result = mocker.MagicMock() + point = mocker.MagicMock() + point.payload = payload + mock_result.points = [point] + mock_qdrant.query_points = mocker.AsyncMock(return_value=mock_result) + mock_qdrant.scroll = mocker.AsyncMock(return_value=([], None)) + mock_qdrant.count = mocker.AsyncMock(return_value=CountResult(count=1)) + mocker.patch("vector_search.views.async_qdrant_client", return_value=mock_qdrant) + hydrate = mocker.patch("vector_search.views._resource_vector_hits") + + response = client.get( + reverse("vector_search:v0:vector_learning_resources_search"), + data={"q": "test"}, + ) + + assert response.status_code == 200 + results = response.json()["results"] + assert [result["readable_id"] for result in results] == [resource.readable_id] + assert results[0]["title"] == resource.title + hydrate.assert_not_called() + + +@pytest.mark.django_db(transaction=True) +def test_vector_search_kill_switch_hydrates_from_database(mocker, client, settings): + """Turning the setting off restores database hydration""" + settings.VECTOR_SEARCH_RESOURCES_FROM_PAYLOAD = False + resource = LearningResourceFactory.create(is_course=True) + + mock_qdrant = mocker.patch( + "qdrant_client.AsyncQdrantClient", return_value=mocker.AsyncMock() + )() + mock_result = mocker.MagicMock() + point = mocker.MagicMock() + point.payload = { + "readable_id": resource.readable_id, + "platform": {"code": resource.platform.code}, + } + mock_result.points = [point] + mock_qdrant.query_points = mocker.AsyncMock(return_value=mock_result) + mock_qdrant.scroll = mocker.AsyncMock(return_value=([], None)) + mock_qdrant.count = mocker.AsyncMock(return_value=CountResult(count=1)) + mocker.patch("vector_search.views.async_qdrant_client", return_value=mock_qdrant) + payload_hits = mocker.patch("vector_search.views._resource_payload_hits") + + response = client.get( + reverse("vector_search:v0:vector_learning_resources_search"), + data={"q": "test"}, + ) + + assert response.status_code == 200 + assert [result["id"] for result in response.json()["results"]] == [resource.id] + payload_hits.assert_not_called() From b161ffc4ffbede494eac1148b2f82c6615d41be9 Mon Sep 17 00:00:00 2001 From: Tobias Macey Date: Thu, 13 Aug 2026 16:36:19 -0400 Subject: [PATCH 3/5] perf(metadata): skip the jsdom sanitize for markup-free descriptions (#3772) * perf(metadata): skip the jsdom sanitize for markup-free descriptions standardizeMetadata calls htmlToPlainText on every server-rendered page, and on the server isomorphic-dompurify is jsdom, so every render allocated a real DOM fragment -- including for descriptions containing no markup at all. The homepage pays it for the constant "Learn with MIT". Return early when the input has neither a tag nor an entity to decode. The guard tests for `<` or `&` rather than `<` alone: "Tom & Jerry" has no tags but still needs decoding, and skipping that would silently leave the entity in the meta description. collapseWhitespace still runs on the fast path, so trimming behaviour is unchanged. This narrows the per-render allocation, not the module-level jsdom import, which is a one-time cost. Refs https://github.com/mitodl/mit-learn/issues/3771 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01XBP7CuQXmxLnPNjQvk4aCz * test(metadata): assert the fast path skips sanitize, not just its output The previous assertions could not fail if the fast path were removed: DOMPurify round-trips markup-free input unchanged, so "2 > 1" and friends produce the same string down either branch. Verified by deleting the guard -- all 11 output-level tests still passed. Spy on DOMPurify.sanitize instead and assert it does not run for markup-free input, which is the allocation this change exists to avoid. Also assert the inverse for tags and entities, so widening the guard until entities skip decoding fails loudly rather than silently regressing meta descriptions. With the guard removed, exactly the 4 new negative assertions fail. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01XBP7CuQXmxLnPNjQvk4aCz --------- Co-authored-by: Claude Opus 5 --- .../main/src/common/htmlToPlainText.test.ts | 52 +++++++++++++++++++ frontends/main/src/common/htmlToPlainText.ts | 14 +++++ 2 files changed, 66 insertions(+) diff --git a/frontends/main/src/common/htmlToPlainText.test.ts b/frontends/main/src/common/htmlToPlainText.test.ts index f1762ced8c..4413e507db 100644 --- a/frontends/main/src/common/htmlToPlainText.test.ts +++ b/frontends/main/src/common/htmlToPlainText.test.ts @@ -1,3 +1,4 @@ +import DOMPurify from "isomorphic-dompurify" import { htmlToPlainText } from "./htmlToPlainText" describe("htmlToPlainText", () => { @@ -40,4 +41,55 @@ describe("htmlToPlainText", () => { it("returns an empty string for empty input", () => { expect(htmlToPlainText("")).toBe("") }) + + // Output-level guards on the two behaviours the fast path must not drop. + // The entity case is the one that genuinely distinguishes the paths: skipping + // sanitize for "&" input would leave the entity undecoded. + it("decodes entities in text that has no tags", () => { + expect(htmlToPlainText("Tom & Jerry")).toBe("Tom & Jerry") + }) + + it("collapses whitespace in text that has no tags or entities", () => { + expect(htmlToPlainText(" Spaced out\n\ttext ")).toBe("Spaced out text") + }) + + it("leaves a bare > in markup-free text alone", () => { + expect(htmlToPlainText("2 > 1")).toBe("2 > 1") + }) + + // Avoiding the jsdom allocation is this change's actual contract, and no + // assertion on the return value can pin it -- DOMPurify round-trips + // markup-free input unchanged, so an output-only test passes either way. + // These assert on whether sanitize ran. + describe("markup-free fast path", () => { + let sanitize: jest.SpyInstance + + beforeEach(() => { + sanitize = jest.spyOn(DOMPurify, "sanitize") + }) + + afterEach(() => { + sanitize.mockRestore() + }) + + it.each([ + ["plain text", "Just plain text"], + ["the default site description", "Learn with MIT"], + ["a bare greater-than", "2 > 1"], + ["text needing whitespace collapsing", " Spaced out\n\ttext "], + ])("does not sanitize %s", (_label, input) => { + htmlToPlainText(input) + expect(sanitize).not.toHaveBeenCalled() + }) + + // The other direction: widening the guard so entities or tags skip + // sanitizing would be a correctness bug, not an optimisation. + it.each([ + ["tags", "

Hello

"], + ["entities", "Tom & Jerry"], + ])("still sanitizes input containing %s", (_label, input) => { + htmlToPlainText(input) + expect(sanitize).toHaveBeenCalled() + }) + }) }) diff --git a/frontends/main/src/common/htmlToPlainText.ts b/frontends/main/src/common/htmlToPlainText.ts index d2f540e882..36a4bc790d 100644 --- a/frontends/main/src/common/htmlToPlainText.ts +++ b/frontends/main/src/common/htmlToPlainText.ts @@ -4,6 +4,12 @@ import { collapseWhitespace } from "@/common/utils" const BLOCK_BOUNDARY_TAGS = /<\/(?:p|div|li|h[1-6]|td|th|tr|blockquote|pre)>|/gi +// A string containing neither of these has no tag to strip and no entity to +// decode, so sanitizing it is a round trip. Both characters matter: guarding on +// `<` alone would send "Tom & Jerry" down the fast path and leave the entity +// undecoded. +const MARKUP_OR_ENTITY = /[<&]/ + /** * Converts a sanitized-HTML string (e.g. a resource `description`) to plain * text suitable for contexts that must not contain markup, like { if (!html) return "" + if (!MARKUP_OR_ENTITY.test(html)) return collapseWhitespace(html) const withBreaks = html.replace(BLOCK_BOUNDARY_TAGS, (match) => `${match} `) // RETURN_DOM_FRAGMENT gives back real DOM nodes rather than a serialized // HTML string, so reading .textContent decodes entities for free (a From 41c62282eaf404f2d2567aea186198befa0a4ae9 Mon Sep 17 00:00:00 2001 From: Shankar Ambady Date: Thu, 13 Aug 2026 16:37:10 -0400 Subject: [PATCH 4/5] constrain percolation to percolate index (#3764) * add fix by constraining to percolate index * paginate results --- learning_resources_search/api.py | 7 +++--- learning_resources_search/api_test.py | 32 +++++++++++++-------------- 2 files changed, 19 insertions(+), 20 deletions(-) diff --git a/learning_resources_search/api.py b/learning_resources_search/api.py index cac955cb0b..e6da9e70d0 100644 --- a/learning_resources_search/api.py +++ b/learning_resources_search/api.py @@ -35,6 +35,7 @@ LEARNING_RESOURCE_QUERY_FIELDS, LEARNING_RESOURCE_SEARCH_SORTBY_OPTIONS, LEARNING_RESOURCE_TYPES, + PERCOLATE_INDEX_TYPE, PROGRAM_TYPE, RUN_INSTRUCTORS_QUERY_FIELDS, RUN_LEVEL_QUERY_FIELDS, @@ -668,13 +669,13 @@ def percolate_matches_for_document(document_id): """ resource = LearningResource.objects.get(id=document_id) index = get_default_alias_name(resource.resource_type) - search = Search() + search = Search(index=get_default_alias_name(PERCOLATE_INDEX_TYPE)) percolate_ids = [] try: results = search.query( Percolate(field="query", index=index, id=str(document_id)) - ).execute() - percolate_ids = [result.id for result in results.hits] + ).scan() + percolate_ids = [result.id for result in results] except NotFoundError: log.info("document %s not found in index", document_id) percolated_queries = PercolateQuery.objects.filter(id__in=percolate_ids) diff --git a/learning_resources_search/api_test.py b/learning_resources_search/api_test.py index 4db0d93692..c8437a01e9 100644 --- a/learning_resources_search/api_test.py +++ b/learning_resources_search/api_test.py @@ -5,7 +5,6 @@ import pytest from freezegun import freeze_time from opensearch_dsl import response -from opensearch_dsl.query import Percolate from learning_resources.constants import OCW_CONTENT_CATEGORY_OPEN_TEXTBOOKS from learning_resources.factories import LearningResourceFactory @@ -31,6 +30,7 @@ CONTENT_FILE_TYPE, COURSE_TYPE, LEARNING_RESOURCE, + PERCOLATE_INDEX_TYPE, PROGRAM_TYPE, ) from learning_resources_search.factories import PercolateQueryFactory @@ -4371,31 +4371,29 @@ def test_document_percolation(opensearch, mocker): { "_index": "test-index", "_id": f"{query.id}", - "id": f"{query.id}", + "_source": {"id": query.id}, + "id": query.id, "_score": 12.0, } ) plugin_log_handler = mocker.patch("learning_resources_search.plugins.log") - mocker.patch.object(Search, "execute") - - Search.execute.return_value = response.Response( - Search().query(Percolate(field="query", index="test", id="test")), - { - "_shards": {"failed": 0, "successful": 10, "total": 10}, - "hits": { - "hits": percolate_hits, - "max_score": 12.0, - "total": 123, - }, - "timed_out": False, - "took": 123, - }, - ).hits + executed_searches = [] + + def mock_scan(search_self, *args, **kwargs): + executed_searches.append(search_self) + for hit in percolate_hits: + yield response.Hit(hit) + + mocker.patch.object(Search, "scan", autospec=True, side_effect=mock_scan) lr = LearningResourceFactory.create() percolate_matches_for_document(lr.id) + assert executed_searches[0]._index == [ # noqa: SLF001 + get_default_alias_name(PERCOLATE_INDEX_TYPE) + ] + plugin_log_handler.debug.assert_called_once_with( "document %i percolated - %s", lr.id, From 6aac001c67403cec98fbc668d22ce4e8420fcb55 Mon Sep 17 00:00:00 2001 From: Doof Date: Mon, 17 Aug 2026 13:46:16 +0000 Subject: [PATCH 5/5] Release 0.77.5 --- RELEASE.rst | 8 ++++++++ main/settings.py | 2 +- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/RELEASE.rst b/RELEASE.rst index 9c00de96c6..cf9201f27b 100644 --- a/RELEASE.rst +++ b/RELEASE.rst @@ -1,6 +1,14 @@ Release Notes ============= +Version 0.77.5 +-------------- + +- constrain percolation to percolate index (#3764) +- perf(metadata): skip the jsdom sanitize for markup-free descriptions (#3772) +- Vector search performance enhancements (#3754) +- Update dependency sharp to v0.35.0 [SECURITY] (#3740) + Version 0.77.4 (Released August 17, 2026) -------------- diff --git a/main/settings.py b/main/settings.py index 6a57f9cc18..d6749f0842 100644 --- a/main/settings.py +++ b/main/settings.py @@ -36,7 +36,7 @@ from main.settings_pluggy import * # noqa: F403 from openapi.settings_spectacular import open_spectacular_settings -VERSION = "0.77.4" +VERSION = "0.77.5" log = logging.getLogger()