Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions RELEASE.rst
Original file line number Diff line number Diff line change
@@ -1,6 +1,14 @@
Release Notes
=============

Version 0.77.5
--------------

- constrain percolation to percolate index (#3764)
- perf(metadata): skip the jsdom sanitize for markup-free descriptions (#3772)
- Vector search performance enhancements (#3754)
- Update dependency sharp to v0.35.0 [SECURITY] (#3740)

Version 0.77.4 (Released August 17, 2026)
--------------

Expand Down
2 changes: 1 addition & 1 deletion frontends/main/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -70,7 +70,7 @@
"react-hotkeys-hook": "^5.2.1",
"react-markdown": "^10.0.0",
"react-slick": "^0.31.0",
"sharp": "0.34.5",
"sharp": "0.35.0",
"slick-carousel": "^1.8.1",
"tiny-invariant": "^1.3.3",
"video.js": "^8.23.7",
Expand Down
77 changes: 77 additions & 0 deletions frontends/main/src/app/(site)/search/page.test.tsx
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
import { factories, makeRequest, setMockResponse, urls } from "api/test-utils"
import Page from "./page"

jest.mock("@/app/getQueryClient", () => {
const { makeBrowserQueryClient } = jest.requireActual("@/app/getQueryClient")
return { getQueryClient: () => makeBrowserQueryClient({ maxRetries: 0 }) }
})

const SEARCH_RESPONSE = {
count: 0,
next: null,
previous: null,
results: [],
metadata: {
aggregations: {},
suggestions: [],
},
}

beforeEach(() => {
setMockResponse.get(
urls.offerors.list(),
factories.learningResources.offerors({ count: 0 }),
)
setMockResponse.get(expect.stringContaining(urls.search.resources()), {
...SEARCH_RESPONSE,
})
setMockResponse.get(expect.stringContaining(urls.search.vectorResources()), {
...SEARCH_RESPONSE,
})
})

test("prefetches OpenSearch results by default", async () => {
await Page({
params: Promise.resolve({}),
searchParams: Promise.resolve({ q: "test" }),
})

expect(
makeRequest.mock.calls.some(([args]) =>
args.url.startsWith(urls.search.resources()),
),
).toBe(true)
expect(
makeRequest.mock.calls.some(([args]) =>
args.url.startsWith(urls.search.vectorResources()),
),
).toBe(false)
})

test("prefetches vector results when vector_search is enabled", async () => {
await Page({
params: Promise.resolve({}),
searchParams: Promise.resolve({
q: "test",
vector_search: "true",
topic: "Physics",
}),
})

const vectorCall = makeRequest.mock.calls.find(([args]) =>
args.url.startsWith(urls.search.vectorResources()),
)
expect(vectorCall).toBeDefined()
expect(
makeRequest.mock.calls.some(([args]) =>
args.url.startsWith(urls.search.resources()),
),
).toBe(false)

const searchParams = new URL(vectorCall?.[0].url ?? "").searchParams
expect(searchParams.get("hybrid_search")).toBe("true")
expect(searchParams.get("q")).toBe("test")
expect(searchParams.has("topic")).toBe(false)
expect(searchParams.has("limit")).toBe(false)
expect(searchParams.has("offset")).toBe(false)
})
31 changes: 25 additions & 6 deletions frontends/main/src/app/(site)/search/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,10 @@ import {
getExtraFacetNames,
} from "@/app-pages/SearchPage/searchRequests"
import getSearchParams from "@/page-components/SearchDisplay/getSearchParams"
import {
toUnfacetedVectorSearchParams,
toVectorSearchParams,
} from "@/page-components/SearchDisplay/vectorSearchParams"
import validateRequestParams from "@/page-components/SearchDisplay/validateRequestParams"
import type { ResourceSearchRequest } from "@/page-components/SearchDisplay/validateRequestParams"
import { LearningResourcesSearchApiLearningResourcesSearchRetrieveRequest as LRSearchRequest } from "api"
Expand Down Expand Up @@ -50,13 +54,28 @@ const Page: React.FC<PageProps<"/search">> = async ({ searchParams }) => {
})

const queryClient = getQueryClient()
const isVectorSearch = urlParams.get("vector_search") === "true"
const hasSearchTerm = typeof params.q === "string" && params.q.trim() !== ""

await Promise.all([
queryClient.prefetchQuery(offerorQueries.list({})),
queryClient.prefetchQuery(
learningResourceQueries.search(params as LRSearchRequest),
),
])
if (isVectorSearch) {
await Promise.all([
queryClient.prefetchQuery(offerorQueries.list({})),
queryClient.prefetchQuery(
learningResourceQueries.vectorSearch(
hasSearchTerm
? toUnfacetedVectorSearchParams(params)
: toVectorSearchParams(params),
),
),
])
} else {
await Promise.all([
queryClient.prefetchQuery(offerorQueries.list({})),
queryClient.prefetchQuery(
learningResourceQueries.search(params as LRSearchRequest),
),
])
}

return (
<HydrationBoundary state={dehydrate(queryClient)}>
Expand Down
52 changes: 52 additions & 0 deletions frontends/main/src/common/htmlToPlainText.test.ts
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import DOMPurify from "isomorphic-dompurify"
import { htmlToPlainText } from "./htmlToPlainText"

describe("htmlToPlainText", () => {
Expand Down Expand Up @@ -40,4 +41,55 @@ describe("htmlToPlainText", () => {
it("returns an empty string for empty input", () => {
expect(htmlToPlainText("")).toBe("")
})

// Output-level guards on the two behaviours the fast path must not drop.
// The entity case is the one that genuinely distinguishes the paths: skipping
// sanitize for "&" input would leave the entity undecoded.
it("decodes entities in text that has no tags", () => {
expect(htmlToPlainText("Tom &amp; Jerry")).toBe("Tom & Jerry")
})

it("collapses whitespace in text that has no tags or entities", () => {
expect(htmlToPlainText(" Spaced out\n\ttext ")).toBe("Spaced out text")
})

it("leaves a bare > in markup-free text alone", () => {
expect(htmlToPlainText("2 > 1")).toBe("2 > 1")
})

// Avoiding the jsdom allocation is this change's actual contract, and no
// assertion on the return value can pin it -- DOMPurify round-trips
// markup-free input unchanged, so an output-only test passes either way.
// These assert on whether sanitize ran.
describe("markup-free fast path", () => {
let sanitize: jest.SpyInstance

beforeEach(() => {
sanitize = jest.spyOn(DOMPurify, "sanitize")
})

afterEach(() => {
sanitize.mockRestore()
})

it.each([
["plain text", "Just plain text"],
["the default site description", "Learn with MIT"],
["a bare greater-than", "2 > 1"],
["text needing whitespace collapsing", " Spaced out\n\ttext "],
])("does not sanitize %s", (_label, input) => {
htmlToPlainText(input)
expect(sanitize).not.toHaveBeenCalled()
})

// The other direction: widening the guard so entities or tags skip
// sanitizing would be a correctness bug, not an optimisation.
it.each([
["tags", "<p>Hello</p>"],
["entities", "Tom &amp; Jerry"],
])("still sanitizes input containing %s", (_label, input) => {
htmlToPlainText(input)
expect(sanitize).toHaveBeenCalled()
})
})
})
14 changes: 14 additions & 0 deletions frontends/main/src/common/htmlToPlainText.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,12 @@ import { collapseWhitespace } from "@/common/utils"
const BLOCK_BOUNDARY_TAGS =
/<\/(?:p|div|li|h[1-6]|td|th|tr|blockquote|pre)>|<br\s*\/?>/gi

// A string containing neither of these has no tag to strip and no entity to
// decode, so sanitizing it is a round trip. Both characters matter: guarding on
// `<` alone would send "Tom &amp; Jerry" down the fast path and leave the entity
// undecoded.
const MARKUP_OR_ENTITY = /[<&]/

/**
* Converts a sanitized-HTML string (e.g. a resource `description`) to plain
* text suitable for contexts that must not contain markup, like <meta
Expand All @@ -16,9 +22,17 @@ const BLOCK_BOUNDARY_TAGS =
* metadata.ts): isomorphic-dompurify has no `sideEffects: false`, so any
* client component importing anything from utils.ts would otherwise pull
* DOMPurify into its bundle even when htmlToPlainText itself is unused.
*
* `standardizeMetadata` calls this on every server-rendered page, and on the
* server `isomorphic-dompurify` is jsdom -- so every call allocated a real DOM
* fragment, including for descriptions that contain no markup at all (the
* default "Learn with MIT" among them). The markup-free fast path below skips
* that. Note it avoids the per-render allocation, not the module-level jsdom
* import, which is a one-time cost.
*/
const htmlToPlainText = (html: string): string => {
if (!html) return ""
if (!MARKUP_OR_ENTITY.test(html)) return collapseWhitespace(html)
const withBreaks = html.replace(BLOCK_BOUNDARY_TAGS, (match) => `${match} `)
// RETURN_DOM_FRAGMENT gives back real DOM nodes rather than a serialized
// HTML string, so reading .textContent decodes entities for free (a
Expand Down
Original file line number Diff line number Diff line change
@@ -1,107 +1,14 @@
import React, { useMemo } from "react"
import { learningResourceQueries } from "api/hooks/learningResources"
import type { LearningResource } from "api"
import type { Facets, BooleanFacets } from "@mitodl/course-search-utils"
import type {
LearningResourcesVectorSearchResponse,
VectorLearningResourcesSearchApiVectorLearningResourcesSearchRetrieveRequest as VectorSearchRequest,
} from "api/v0"
import type { LearningResourcesVectorSearchResponse } from "api/v0"
import getSearchParams from "./getSearchParams"
import SearchDisplay, { SearchDisplayProps } from "./SearchDisplay"

const mapVectorSortby = (
sortby?: string,
): VectorSearchRequest["sortby"] | undefined => {
switch (sortby) {
case "-views":
case "popular":
return "-views"
case "upcoming":
return "next_start_date"
case "new":
return "-created_on"
default:
return undefined
}
}

/**
* Extracts only the fields supported by the vector search API from a broader
* search params object, dropping admin-only params (e.g., content_file_score_weight)
* that the vector endpoint does not accept.
*
* The `as` casts for enum arrays are safe because the v0 and v1 generated
* clients define separate (but structurally identical) enum types for the same
* string-literal values (e.g., delivery: 'online' | 'hybrid' | ...).
*/
const toVectorSearchParams = (
params: ReturnType<typeof getSearchParams> & { sortby?: string },
cutoffScore?: number,
): VectorSearchRequest => ({
aggregations: params.aggregations as VectorSearchRequest["aggregations"],
certification: params.certification,
certification_type:
params.certification_type as VectorSearchRequest["certification_type"],
course_feature: params.course_feature,
delivery: params.delivery as VectorSearchRequest["delivery"],
department: params.department as VectorSearchRequest["department"],
free: params.free,
level: params.level as VectorSearchRequest["level"],
limit: params.limit,
ocw_topic: params.ocw_topic,
offered_by: params.offered_by as VectorSearchRequest["offered_by"],
offset: params.offset,
platform: params.platform as VectorSearchRequest["platform"],
professional: params.professional,
q: params.q,
resource_category:
params.resource_category as VectorSearchRequest["resource_category"],
resource_type: params.resource_type as VectorSearchRequest["resource_type"],
resource_type_group:
params.resource_type_group as VectorSearchRequest["resource_type_group"],
score_cutoff: cutoffScore,
sortby: mapVectorSortby(params.sortby),
topic: params.topic,
hybrid_search: true,
})

const VECTOR_CLIENT_FILTER_FACETS = [
"resource_type",
"certification_type",
"delivery",
"department",
"topic",
"offered_by",
"free",
"professional",
"resource_category",
"resource_type_group",
"level",
"platform",
"course_feature",
] as const

type VectorClientFilterFacet = (typeof VECTOR_CLIENT_FILTER_FACETS)[number]

const toUnfacetedVectorSearchParams = (
params: ReturnType<typeof getSearchParams> & { sortby?: string },
constantSearchParams: Facets & BooleanFacets = {},
cutoffScore?: number,
): VectorSearchRequest => {
const {
offset: _offset,
limit: _limit,
...vectorParams
} = toVectorSearchParams(params, cutoffScore)

return Object.fromEntries(
Object.entries(vectorParams).filter(
([key]) =>
!VECTOR_CLIENT_FILTER_FACETS.includes(key as VectorClientFilterFacet) ||
key in constantSearchParams,
),
) as VectorSearchRequest
}
import {
VECTOR_CLIENT_FILTER_FACETS,
toUnfacetedVectorSearchParams,
toVectorSearchParams,
} from "./vectorSearchParams"

const normalizeParamValues = (value: unknown): string[] => {
if (Array.isArray(value)) {
Expand Down
Loading
Loading