diff --git a/.github/instructions/frontend-tests.instructions.md b/.github/instructions/frontend-tests.instructions.md index fad0232604..a2914568ab 100644 --- a/.github/instructions/frontend-tests.instructions.md +++ b/.github/instructions/frontend-tests.instructions.md @@ -183,6 +183,18 @@ expect(consoleError).toHaveBeenCalled() // In rare cases, TestingErrorBoundary can be used to test thrown errors. ``` +**Mutation failures and the global error toast:** + +Every mutation failure raises a global error toast unless the call site opts out (`meta: SILENCE_ERROR_TOAST`), and an `afterEach` fails any test that leaves a toast unacknowledged. + +```tsx +import { expectErrorToast } from "@/test-utils" +// After driving a mutation failure whose intended error surface is the toast: +await expectErrorToast("Something went wrong") +// If the component shows its own inline error instead, don't acknowledge — +// opt the component out of the toast with `meta: SILENCE_ERROR_TOAST`. +``` + ## Troubleshooting - **"No response specified"** → Mock the API call with `setMockResponse` diff --git a/.github/instructions/frontend.instructions.md b/.github/instructions/frontend.instructions.md index b289f49599..747eadea46 100644 --- a/.github/instructions/frontend.instructions.md +++ b/.github/instructions/frontend.instructions.md @@ -8,3 +8,11 @@ applyTo: "**/*.ts,**/*.tsx,**/package.json" - `api` contains generated API client code and react-query hooks - For reusable UI, use components from `@mitodl/smoot-design`, `ol-components` preferentially - Within `main`, use `@/` for root-relative imports + +## Mutation error handling + +Every mutation failure in the browser shows a global error toast by default (`MutationCache.onError` in `main/src/app/getQueryClient.ts`), so failures are never silent. Tune it per call site via React Query `meta` (typed in `api/mutation-meta`): + +- Component renders its own inline error for the failure → pass `meta: SILENCE_ERROR_TOAST` to the mutation hook, or the user sees a double alert. Required even if you catch the `mutateAsync` rejection yourself — catching does not suppress the toast. +- Custom toast copy → `meta: { errorMessage: "Could not save your changes." }`, or `getErrorMessage(error, variables)` for data-driven copy. +- Shared `api/` hooks accept `{ meta }` (`MutationHookOptions`) and forward it; the opt-out belongs at the _consumer_, never baked into a shared hook (many hooks are surfaced in one place and silent in another). diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6409bb3f9a..c25f3df688 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -27,7 +27,7 @@ jobs: - 5432:5432 redis: - image: redis:8.2.2 + image: redis:8.10.0@sha256:344e3945a0b431c8ff1eecd58c5573538126bd756f02fc7e218ddf1fc2546366 ports: - 6379:6379 @@ -272,3 +272,62 @@ jobs: run: | diff $GENERATOR_OUTPUT_DIR_CI $GENERATOR_OUTPUT_DIR_VC \ || { echo "OpenAPI spec is out of date. Please regenerate via ./scripts/generate_openapi.sh"; exit 1; } + + # ONE required status check standing for this whole workflow. + # + # GitHub's required status checks take an exact list of context names -- no + # wildcards, no "all checks must pass" option -- so requiring these six jobs + # directly means restating all six names in mitodl/ol-infrastructure, where the + # ruleset lives. That list then goes stale in two ways: + # + # A renamed job keeps being required under its old name, which GitHub will never + # report again, so every PR waits forever on it. mitxonline hit exactly this when + # `python-tests` became a 4-way matrix and the checks turned into + # `python-tests (1)`..`(4)`. + # + # A newly added job is not required until somebody remembers to go and add it, so + # new CI silently cannot block a merge. + # + # Requiring `ci-gate` instead keeps the list of what must pass in the same file as + # the jobs it names, so adding or renaming a job is one edit in one repo. + # + # `openapi-diff` is absent from `needs` because `needs` can only reference jobs in + # this same workflow file, and `openapi-diff` runs in its own workflow (triggered on + # `pull_request` rather than `push`). It now has an `--err-ignore` allowlist + # (openapi/oasdiff-err-ignore.txt) for intentional breaking changes, so it + # should be required directly, alongside `ci-gate`, in the ol-infrastructure ruleset + # -- its job name is stable and isn't subject to the renaming/matrix drift `ci-gate` + # exists to solve. + ci-gate: + needs: + - python-tests + - javascript-tests + - build-nextjs-container + - build-storybook + - openapi-generated-client-check-v0 + - openapi-generated-client-check-v1 + # Without this the gate is skipped when a dependency fails, and a skipped required + # check leaves the PR pending rather than failing it. + if: always() + runs-on: ubuntu-24.04 + permissions: {} + steps: + - name: Evaluate CI result + env: + NEEDS: ${{ toJSON(needs) }} + run: | + set -euo pipefail + echo "$NEEDS" | jq -r 'to_entries[] | "\(.key): \(.value.result)"' + # Read from `needs` rather than naming each job again: adding a job to the + # list above is then the only edit, which is the entire point of this job. + # `skipped` passes so that a job legitimately gated behind an `if:` does not + # wedge the branch; `failure` and `cancelled` do not. + failed=$(echo "$NEEDS" | jq -r ' + to_entries[] + | select(.value.result != "success" and .value.result != "skipped") + | .key') + if [ -n "$failed" ]; then + echo "::error::CI did not succeed: $(echo "$failed" | paste -sd, -)" + exit 1 + fi + echo "All CI jobs succeeded." diff --git a/.github/workflows/openapi-diff.yml b/.github/workflows/openapi-diff.yml index b0c571520d..b54ff415d3 100644 --- a/.github/workflows/openapi-diff.yml +++ b/.github/workflows/openapi-diff.yml @@ -67,6 +67,7 @@ jobs: --volume ${{ github.workspace }}:${{ github.workspace }}:ro \ -e GITHUB_WORKSPACE=${{ github.workspace }} \ tufin/oasdiff breaking \ + --err-ignore head/openapi/oasdiff-err-ignore.txt \ --fail-on ERR \ --format githubactions \ --composed --flatten-allof \ diff --git a/RELEASE.rst b/RELEASE.rst index be020a4ea8..12e9fbc53a 100644 --- a/RELEASE.rst +++ b/RELEASE.rst @@ -1,6 +1,30 @@ Release Notes ============= +Version 0.78.1 +-------------- + +- adding initial fix (#3872) +- Per-product Stay Updated HubSpot form id (frontend) (#3844) +- Update dependency tiktoken to >=0.13,<0.14 (#3860) +- feat(b2b): render the analytics dashboard scoped to a contract (#3773) +- fix: report the real status from handle_error instead of 405 (#3791) +- Use the green token instead of a hardcoded hex on product pages (#3847) +- feat: show podcast episode transcripts from the podcast:transcript tag (#3821) +- Remove remaining API endpoint N+1 queries (#3845) +- Update dependency drf-spectacular to >=0.30,<0.31 (#3858) +- Update dependency ruff to v0.16.3 (#3859) +- Submit and link each resource under one URL, from learn_url (#3843) +- Update dependency litellm to v1.96.2 (#3857) +- Update redis Docker tag to v8.10.0 (#2706) +- feat(cohort-1): MicroMasters/MITx Online certificate warehouse-pull sync (#3808) +- ci: add a ci-gate job so one required check can cover the whole suite (#3825) +- Overridable Mutation Error Toast (#3837) +- Show Stay Updated based only on the CMS page flag (#3841) +- staleness penalty for vector search results (#3834) +- Update dependency Django to v5.2.17 [SECURITY] (#3849) +- Update dependency social-auth-app-django to v5.6.0 [SECURITY] (#3850) + Version 0.77.15 (Released August 27, 2026) --------------- diff --git a/channels/models.py b/channels/models.py index d44c159f9b..07078f851b 100644 --- a/channels/models.py +++ b/channels/models.py @@ -69,10 +69,21 @@ def with_detail_relations(self) -> "ChannelQuerySet": "sub_channels__channel", queryset=Channel.objects.annotate_channel_url(), ), + # LearningResourceOfferor.channel_url reads + # channel_unit_details.first(); ordering the prefetch by pk + # lets that resolve from the cache and pick the same row it + # would have queried for. + Prefetch( + "unit_detail__unit__channel_unit_details", + queryset=ChannelUnitDetail.objects.select_related( + "channel" + ).order_by("pk"), + ), ) .annotate_channel_url() .select_related( "featured_list", + "unit_detail__unit", "topic_detail", "department_detail", "unit_detail", diff --git a/channels/views_test.py b/channels/views_test.py index fd8ce1c8bb..4dd671c8d6 100644 --- a/channels/views_test.py +++ b/channels/views_test.py @@ -7,7 +7,12 @@ from django.urls import reverse from channels.constants import ChannelType -from channels.factories import ChannelFactory, ChannelListFactory, SubChannelFactory +from channels.factories import ( + ChannelFactory, + ChannelListFactory, + ChannelUnitDetailFactory, + SubChannelFactory, +) from channels.models import Channel from channels.serializers import ChannelSerializer from learning_resources.factories import LearningResourceFactory @@ -16,7 +21,6 @@ pytestmark = pytest.mark.django_db -@pytest.mark.skip_nplusone_check def test_list_channels(user_client): """Test that all published channels are returned.""" ChannelFactory.create_batch(2, published=False) @@ -33,6 +37,44 @@ def test_list_channels(user_client): assert response_channels[idx] == ChannelSerializer(instance=channel).data +@pytest.mark.parametrize("channel_count", [2, 6]) +def test_list_unit_channels_query_count( + client, django_assert_num_queries, channel_count +): + """Unit channels cost the same number of queries however many are listed. + + unit_detail.unit is nested by the serializer and its channel_url is a + cached_property, so without the prefetch each row costs two extra queries. + """ + channels = ChannelFactory.create_batch(channel_count, is_unit=True) + # An offeror shared with later, unpublished channels: channel_url has to + # resolve to the same row the serializer would pick on its own, so the + # prefetch behind it can't be an unordered subquery. + for _ in range(3): + extra = ChannelFactory.create( + is_unit=True, published=False, create_unit_detail=False + ) + ChannelUnitDetailFactory.create( + channel=extra, unit=channels[0].unit_detail.unit + ) + + url = reverse("channels:v0:channels_api-list") + with django_assert_num_queries(5): + results = client.get(url).json()["results"] + + assert len(results) == channel_count + assert all(item["unit_detail"]["unit"]["channel_url"] for item in results) + # The listing must agree with serializing each instance directly. + for channel in channels: + listed = next(item for item in results if item["id"] == channel.id) + assert ( + listed["unit_detail"]["unit"]["channel_url"] + == ChannelSerializer(instance=channel).data["unit_detail"]["unit"][ + "channel_url" + ] + ) + + def test_channel_detail_has_no_is_moderator(client): """Channel detail no longer exposes moderator-specific fields.""" channel = ChannelFactory.create() @@ -153,7 +195,6 @@ def test_no_excess_list_queries(client, user, django_assert_num_queries, channel assert channel["channel_url"] is not None -@pytest.mark.skip_nplusone_check def test_channel_counts_view(client): """Channel counts should return per-channel resource counts.""" url = reverse( @@ -194,7 +235,6 @@ def enabled_view_cache(settings, request): } -@pytest.mark.skip_nplusone_check @pytest.mark.usefixtures("enabled_view_cache") def test_channel_detail_cache_is_global(client): """Cached detail responses are shared globally across users.""" @@ -210,7 +250,6 @@ def test_channel_detail_cache_is_global(client): assert client.get(url).json()["title"] == "Original title" -@pytest.mark.skip_nplusone_check @pytest.mark.usefixtures("enabled_view_cache") def test_channel_by_type_name_cache_is_global(client): """Cached by-type-name responses are shared globally across users.""" @@ -229,7 +268,6 @@ def test_channel_by_type_name_cache_is_global(client): assert client.get(url).json()["title"] == "Original title" -@pytest.mark.skip_nplusone_check @pytest.mark.usefixtures("enabled_view_cache") @pytest.mark.parametrize("is_authenticated", [False, True]) def test_channel_counts_view_is_cached(client, is_authenticated): diff --git a/docker-compose.services.yml b/docker-compose.services.yml index e00646fb64..6ab1ac488e 100644 --- a/docker-compose.services.yml +++ b/docker-compose.services.yml @@ -31,7 +31,7 @@ services: redis: profiles: - backend - image: redis:8.2.2 + image: redis:8.10.0@sha256:344e3945a0b431c8ff1eecd58c5573538126bd756f02fc7e218ddf1fc2546366 healthcheck: test: ["CMD", "redis-cli", "ping", "|", "grep", "PONG"] interval: 3s diff --git a/drf_lint_baseline.json b/drf_lint_baseline.json index 2657099b78..2222d8bddd 100644 --- a/drf_lint_baseline.json +++ b/drf_lint_baseline.json @@ -2,8 +2,5 @@ "channels/serializers.py:132:24:ORM002", "channels/serializers.py:134:24:ORM002", "channels/serializers.py:136:24:ORM002", - "profiles/serializers.py:136:31:ORM002", - "profiles/serializers.py:196:16:ORM002", - "profiles/serializers.py:446:15:ORM001", - "profiles/serializers.py:447:27:ORM001" + "profiles/serializers.py:205:16:ORM002" ] diff --git a/env/codespaces.env b/env/codespaces.env index d0bf428e6d..61b2f521d3 100644 --- a/env/codespaces.env +++ b/env/codespaces.env @@ -50,6 +50,5 @@ NEXT_PUBLIC_ORIGIN=${MITOL_APP_BASE_URL} NEXT_PUBLIC_MITOL_API_BASE_URL=${MITOL_API_BASE_URL} NEXT_PUBLIC_CSRF_COOKIE_NAME=${CSRF_COOKIE_NAME} NEXT_PUBLIC_MITOL_SUPPORT_EMAIL=${MITOL_SUPPORT_EMAIL} -NEXT_PUBLIC_STAY_UPDATED_HUBSPOT_FORM_ID=${STAY_UPDATED_HUBSPOT_FORM_ID} NEXT_PUBLIC_SITE_NAME="MIT Learn" NEXT_PUBLIC_MITOL_AXIOS_WITH_CREDENTIALS=true diff --git a/env/frontend.env b/env/frontend.env index aeae1d863a..2bea1ee743 100644 --- a/env/frontend.env +++ b/env/frontend.env @@ -14,7 +14,6 @@ NEXT_PUBLIC_POSTHOG_API_KEY=${POSTHOG_PROJECT_API_KEY} NEXT_PUBLIC_POSTHOG_UI_HOST=${POSTHOG_UI_HOST} NEXT_PUBLIC_RECAPTCHA_SITE_KEY=${RECAPTCHA_SITE_KEY} -NEXT_PUBLIC_STAY_UPDATED_HUBSPOT_FORM_ID=${STAY_UPDATED_HUBSPOT_FORM_ID} NEXT_PUBLIC_SITE_NAME="MIT Learn" NEXT_PUBLIC_MITOL_AXIOS_WITH_CREDENTIALS=true diff --git a/env/frontend.local.example.env b/env/frontend.local.example.env index 763bdbaf3e..2e2ab3ab8c 100644 --- a/env/frontend.local.example.env +++ b/env/frontend.local.example.env @@ -1,5 +1,3 @@ -NEXT_PUBLIC_STAY_UPDATED_HUBSPOT_FORM_ID="" - # Optional local GTM overrides (server-side runtime) # GTM_TRACKING_ID= # GTM_AUTH= diff --git a/frontends/api/package.json b/frontends/api/package.json index f8b3df4903..ee613bc0ee 100644 --- a/frontends/api/package.json +++ b/frontends/api/package.json @@ -16,6 +16,7 @@ "./test-utils/mockAxios": "./src/test-utils/mockAxios.ts", "./test-utils": "./src/test-utils/index.ts", "./mitxonline-hooks/*": "./src/mitxonline/hooks/*/index.ts", + "./mutation-meta": "./src/mutations/mutationMeta.ts", "./mitxonline-test-utils": "./src/mitxonline/test-utils/index.ts", "./analytics-hooks/*": "./src/analytics/hooks/*/index.ts", "./analytics-types": "./src/analytics/types.ts", @@ -35,7 +36,7 @@ }, "dependencies": { "@mitodl/mit-learn-api-axios": "2026.8.17", - "@mitodl/mitxonline-api-axios": "2026.8.18", + "@mitodl/mitxonline-api-axios": "2026.8.31", "@tanstack/react-query": "^5.66.0", "axios": "^1.12.2", "tiny-invariant": "^1.3.3" diff --git a/frontends/api/src/analytics/clients.ts b/frontends/api/src/analytics/clients.ts index 2d67bc87fe..3c38224928 100644 --- a/frontends/api/src/analytics/clients.ts +++ b/frontends/api/src/analytics/clients.ts @@ -3,6 +3,8 @@ import axiosInstance from "./axios" import type { AnalyticsPageParams, ContentEngagementDepth, + ContractContentEngagementDepth, + ContractMonthlyEngagementTrend, ContractUtilization, EnrollmentCompletionFunnel, MonthlyEngagementTrend, @@ -104,4 +106,111 @@ const analyticsOrganizationsApi = { ), } -export { analyticsOrganizationsApi, B2B_DASHBOARD_ROOT } +/** + * `contractId` is MITx Online's `ContractPage.page_ptr_id` — the same value the + * manager dashboard puts in its own URLs, and what the analytics API filters + * on. NOT the contract slug the MIT Learn route carries, and NOT the + * warehouse's `contract_pk` surrogate; resolve the slug to an id from the + * org's `contracts` list before calling any of these. + * + * The org segment stays in the path even though a contract id identifies a + * contract on its own: the API filters on both, so that a manager of one org + * cannot read another's contract by naming it. + */ +const contractRoot = (organizationId: string, contractId: string) => + `${orgRoot(organizationId)}/contracts/${encodeURIComponent(contractId)}` + +const getContractResource = ( + organizationId: string, + contractId: string, + resource: string, + page: AnalyticsPageParams | undefined, + signal: AbortSignal | undefined, +): Promise>> => + axiosInstance.get>( + `${contractRoot(organizationId, contractId)}/${resource}`, + { params: page, signal }, + ) + +/** + * The contract-scoped half of the same five endpoints, mirroring MITx Online's + * manager dashboard (which nests contracts under an organization). Same + * envelope, same paging, same suppression — only the scope differs. + * + * Two of the five read materialized views that exist only at contract grain + * (engagement-trend, content-engagement); the rest read the same views as the + * org endpoints, filtered down. + */ +const analyticsContractsApi = { + contractUtilization: ( + organizationId: string, + contractId: string, + page?: AnalyticsPageParams, + signal?: AbortSignal, + ) => + getContractResource( + organizationId, + contractId, + "contract-utilization", + page, + signal, + ), + + enrollmentFunnel: ( + organizationId: string, + contractId: string, + page?: AnalyticsPageParams, + signal?: AbortSignal, + ) => + getContractResource( + organizationId, + contractId, + "enrollment-funnel", + page, + signal, + ), + + engagementTrend: ( + organizationId: string, + contractId: string, + page?: AnalyticsPageParams, + signal?: AbortSignal, + ) => + getContractResource( + organizationId, + contractId, + "engagement-trend", + page, + signal, + ), + + programFunnel: ( + organizationId: string, + contractId: string, + page?: AnalyticsPageParams, + signal?: AbortSignal, + ) => + getContractResource( + organizationId, + contractId, + "program-funnel", + page, + signal, + ), + + contentEngagement: ( + organizationId: string, + contractId: string, + page?: AnalyticsPageParams, + signal?: AbortSignal, + ) => + getContractResource( + organizationId, + contractId, + "content-engagement", + page, + signal, + ), +} + +export { analyticsOrganizationsApi, analyticsContractsApi, B2B_DASHBOARD_ROOT } diff --git a/frontends/api/src/analytics/hooks/organizations/index.ts b/frontends/api/src/analytics/hooks/organizations/index.ts index 5c172ba9a8..da0e4aa108 100644 --- a/frontends/api/src/analytics/hooks/organizations/index.ts +++ b/frontends/api/src/analytics/hooks/organizations/index.ts @@ -1,11 +1,15 @@ export { analyticsOrganizationQueries, analyticsOrganizationKeys, + analyticsContractQueries, + analyticsContractKeys, } from "./queries" export type { AnalyticsPageParams, ContentEngagementDepth, + ContractContentEngagementDepth, + ContractMonthlyEngagementTrend, ContractUtilization, EnrollmentCompletionFunnel, MonthlyEngagementTrend, diff --git a/frontends/api/src/analytics/hooks/organizations/queries.test.ts b/frontends/api/src/analytics/hooks/organizations/queries.test.ts index 46839f969a..a6dfd68999 100644 --- a/frontends/api/src/analytics/hooks/organizations/queries.test.ts +++ b/frontends/api/src/analytics/hooks/organizations/queries.test.ts @@ -3,7 +3,10 @@ import { useQuery, type UseQueryOptions } from "@tanstack/react-query" import { setupReactQueryTest } from "../../../hooks/test-utils" import { setMockResponse, makeRequest } from "../../../test-utils" import { factories, urls } from "../../test-utils" -import { analyticsOrganizationQueries } from "./queries" +import { + analyticsContractQueries, + analyticsOrganizationQueries, +} from "./queries" const ORG_UUID = "3fa85f64-5717-4562-b3fc-2c963f66afa6" @@ -115,3 +118,89 @@ describe("analyticsOrganizationQueries", () => { expect(urls.organizations.contractUtilization("a/b")).toContain("a%2Fb") }) }) + +const CONTRACT_ID = "101" + +describe("analyticsContractQueries", () => { + test.each([ + { + name: "contractUtilization", + query: () => + erase( + analyticsContractQueries.contractUtilization(ORG_UUID, CONTRACT_ID), + ), + url: urls.contracts.contractUtilization(ORG_UUID, CONTRACT_ID), + response: factories.envelope([factories.contractUtilization()]), + }, + { + name: "enrollmentFunnel", + query: () => + erase(analyticsContractQueries.enrollmentFunnel(ORG_UUID, CONTRACT_ID)), + url: urls.contracts.enrollmentFunnel(ORG_UUID, CONTRACT_ID), + response: factories.envelope([factories.enrollmentCompletionFunnel()]), + }, + { + name: "engagementTrend", + query: () => + erase(analyticsContractQueries.engagementTrend(ORG_UUID, CONTRACT_ID)), + url: urls.contracts.engagementTrend(ORG_UUID, CONTRACT_ID), + response: factories.envelope([ + factories.contractMonthlyEngagementTrend(), + ]), + }, + { + name: "programFunnel", + query: () => + erase(analyticsContractQueries.programFunnel(ORG_UUID, CONTRACT_ID)), + url: urls.contracts.programFunnel(ORG_UUID, CONTRACT_ID), + response: factories.envelope([factories.programFunnel()]), + }, + { + name: "contentEngagement", + query: () => + erase( + analyticsContractQueries.contentEngagement(ORG_UUID, CONTRACT_ID), + ), + url: urls.contracts.contentEngagement(ORG_UUID, CONTRACT_ID), + response: factories.envelope([ + factories.contractContentEngagementDepth(), + ]), + }, + ])( + "$name requests the contract-nested path", + async ({ query, url, response }) => { + const { wrapper } = setupReactQueryTest() + setMockResponse.get(url, response) + + const { result } = renderHook(() => useQuery(query()), { wrapper }) + + await waitFor(() => expect(result.current.isSuccess).toBe(true)) + expect(makeRequest).toHaveBeenCalledWith( + expect.objectContaining({ method: "get", url }), + ) + }, + ) + + test("contract keys nest under the org's, and differ per contract", () => { + const key = analyticsContractQueries.contractUtilization( + ORG_UUID, + CONTRACT_ID, + ).queryKey + // The org prefix is intact, so invalidating an org drops its contracts too. + expect(key.slice(0, 3)).toEqual(["analytics", "organizations", ORG_UUID]) + expect(key).toContain(CONTRACT_ID) + expect( + analyticsContractQueries.contractUtilization(ORG_UUID, "202").queryKey, + ).not.toEqual(key) + // And a contract-scoped result never satisfies an org-scoped read. + expect(key).not.toEqual( + analyticsOrganizationQueries.contractUtilization(ORG_UUID).queryKey, + ) + }) + + test("the contract id is url-encoded rather than spliced into the path raw", () => { + expect(urls.contracts.contractUtilization(ORG_UUID, "a/b")).toContain( + "a%2Fb", + ) + }) +}) diff --git a/frontends/api/src/analytics/hooks/organizations/queries.ts b/frontends/api/src/analytics/hooks/organizations/queries.ts index 0679c78880..5abf59c694 100644 --- a/frontends/api/src/analytics/hooks/organizations/queries.ts +++ b/frontends/api/src/analytics/hooks/organizations/queries.ts @@ -1,5 +1,5 @@ import { queryOptions } from "@tanstack/react-query" -import { analyticsOrganizationsApi } from "../../clients" +import { analyticsContractsApi, analyticsOrganizationsApi } from "../../clients" import type { AnalyticsPageParams } from "../../types" /** @@ -96,4 +96,133 @@ const analyticsOrganizationQueries = { }), } -export { analyticsOrganizationQueries, analyticsOrganizationKeys } +/** + * Contract-scoped keys nest under the org's, so invalidating an organization + * drops its contracts' cached sections too — they are views of the same + * underlying data and can never be stale independently. + * + * `contractId` is MITx Online's contract id, not the route's slug. + */ +const analyticsContractKeys = { + contract: (orgId: string, contractId: string) => + [ + ...analyticsOrganizationKeys.organization(orgId), + "contracts", + contractId, + ] as const, + resource: ( + orgId: string, + contractId: string, + resource: string, + page?: AnalyticsPageParams, + ) => + [ + ...analyticsContractKeys.contract(orgId, contractId), + resource, + page, + ] as const, +} + +const analyticsContractQueries = { + contractUtilization: ( + orgId: string, + contractId: string, + page?: AnalyticsPageParams, + ) => + queryOptions({ + queryKey: analyticsContractKeys.resource( + orgId, + contractId, + "contract-utilization", + page, + ), + staleTime: ANALYTICS_STALE_TIME, + queryFn: async ({ signal }) => + analyticsContractsApi + .contractUtilization(orgId, contractId, page, signal) + .then((res) => res.data), + }), + + enrollmentFunnel: ( + orgId: string, + contractId: string, + page?: AnalyticsPageParams, + ) => + queryOptions({ + queryKey: analyticsContractKeys.resource( + orgId, + contractId, + "enrollment-funnel", + page, + ), + staleTime: ANALYTICS_STALE_TIME, + queryFn: async ({ signal }) => + analyticsContractsApi + .enrollmentFunnel(orgId, contractId, page, signal) + .then((res) => res.data), + }), + + engagementTrend: ( + orgId: string, + contractId: string, + page?: AnalyticsPageParams, + ) => + queryOptions({ + queryKey: analyticsContractKeys.resource( + orgId, + contractId, + "engagement-trend", + page, + ), + staleTime: ANALYTICS_STALE_TIME, + queryFn: async ({ signal }) => + analyticsContractsApi + .engagementTrend(orgId, contractId, page, signal) + .then((res) => res.data), + }), + + programFunnel: ( + orgId: string, + contractId: string, + page?: AnalyticsPageParams, + ) => + queryOptions({ + queryKey: analyticsContractKeys.resource( + orgId, + contractId, + "program-funnel", + page, + ), + staleTime: ANALYTICS_STALE_TIME, + queryFn: async ({ signal }) => + analyticsContractsApi + .programFunnel(orgId, contractId, page, signal) + .then((res) => res.data), + }), + + contentEngagement: ( + orgId: string, + contractId: string, + page?: AnalyticsPageParams, + ) => + queryOptions({ + queryKey: analyticsContractKeys.resource( + orgId, + contractId, + "content-engagement", + page, + ), + staleTime: ANALYTICS_STALE_TIME, + queryFn: async ({ signal }) => + analyticsContractsApi + .contentEngagement(orgId, contractId, page, signal) + .then((res) => res.data), + }), +} + +export { + analyticsOrganizationQueries, + analyticsOrganizationKeys, + analyticsContractQueries, + analyticsContractKeys, +} diff --git a/frontends/api/src/analytics/test-utils/factories.ts b/frontends/api/src/analytics/test-utils/factories.ts index abf30a78ad..d7efb2cb4f 100644 --- a/frontends/api/src/analytics/test-utils/factories.ts +++ b/frontends/api/src/analytics/test-utils/factories.ts @@ -1,6 +1,8 @@ import { faker } from "@faker-js/faker/locale/en" import type { ContentEngagementDepth, + ContractContentEngagementDepth, + ContractMonthlyEngagementTrend, ContractUtilization, EnrollmentCompletionFunnel, MonthlyEngagementTrend, @@ -36,7 +38,12 @@ const contractUtilization = ( ): ContractUtilization => ({ organization_key: faker.string.alphanumeric(6).toUpperCase(), organization_name: faker.company.name(), - contract_pk: faker.number.int({ min: 1, max: 10000 }), + contract_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), + contract_id: String(faker.number.int({ min: 1, max: 10000 })), b2b_contract_name: `${faker.company.name()} Contract`, b2b_contract_is_active: true, b2b_contract_start_date: "2026-01-01", @@ -56,9 +63,18 @@ const enrollmentCompletionFunnel = ( ): EnrollmentCompletionFunnel => ({ organization_key: faker.string.alphanumeric(6).toUpperCase(), organization_name: faker.company.name(), - contract_pk: faker.number.int({ min: 1, max: 10000 }), + contract_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), + contract_id: String(faker.number.int({ min: 1, max: 10000 })), b2b_contract_name: `${faker.company.name()} Contract`, - courserun_pk: faker.number.int({ min: 1, max: 10000 }), + courserun_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), courserun_readable_id: `course-v1:MITx+${faker.string.alphanumeric(5)}+2026`, courserun_title: faker.commerce.productName(), enrolled_learners: 40, @@ -78,10 +94,15 @@ const monthlyEngagementTrend = ( activity_year_and_month: "2026-01", monthly_active_learners: 30, new_enrollments: 12, + enrolling_learners: 9, certificates_earned: 5, + certified_learners: 5, total_videos_watched: 900, + video_watchers: 22, total_problems_attempted: 1200, + problem_attempters: 19, total_chatbot_interactions: 80, + chatbot_users: 11, ...overrides, }) @@ -90,9 +111,18 @@ const programFunnel = ( ): ProgramFunnel => ({ organization_key: faker.string.alphanumeric(6).toUpperCase(), organization_name: faker.company.name(), - contract_pk: faker.number.int({ min: 1, max: 10000 }), + contract_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), + contract_id: String(faker.number.int({ min: 1, max: 10000 })), b2b_contract_name: `${faker.company.name()} Contract`, - program_pk: faker.number.int({ min: 1, max: 10000 }), + program_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), program_title: `${faker.commerce.department()} Program`, total_courses: 6, enrolled_in_contract_courses: 50, @@ -127,8 +157,36 @@ const contentEngagementDepth = ( ...overrides, }) +const contractIdentity = () => ({ + contract_pk: faker.string.hexadecimal({ + length: 32, + casing: "lower", + prefix: "", + }), + contract_id: String(faker.number.int({ min: 1, max: 10000 })), + b2b_contract_name: `${faker.company.name()} Contract`, +}) + +const contractMonthlyEngagementTrend = ( + overrides: Partial = {}, +): ContractMonthlyEngagementTrend => ({ + ...monthlyEngagementTrend(), + ...contractIdentity(), + ...overrides, +}) + +const contractContentEngagementDepth = ( + overrides: Partial = {}, +): ContractContentEngagementDepth => ({ + ...contentEngagementDepth(), + ...contractIdentity(), + ...overrides, +}) + export { contentEngagementDepth, + contractContentEngagementDepth, + contractMonthlyEngagementTrend, contractUtilization, enrollmentCompletionFunnel, envelope, diff --git a/frontends/api/src/analytics/test-utils/urls.ts b/frontends/api/src/analytics/test-utils/urls.ts index f87534403e..1fb8f95b6c 100644 --- a/frontends/api/src/analytics/test-utils/urls.ts +++ b/frontends/api/src/analytics/test-utils/urls.ts @@ -18,6 +18,16 @@ const orgResource = ( organizationId, )}/${resource}${queryify(params)}` +const contractResource = ( + organizationId: string, + contractId: string, + resource: string, + params?: AnalyticsPageParams, +) => + `${getApiBaseUrl()}${B2B_DASHBOARD_ROOT}/organizations/${encodeURIComponent( + organizationId, + )}/contracts/${encodeURIComponent(contractId)}/${resource}${queryify(params)}` + const organizations = { contractUtilization: (organizationId: string, params?: AnalyticsPageParams) => orgResource(organizationId, "contract-utilization", params), @@ -31,4 +41,40 @@ const organizations = { orgResource(organizationId, "content-engagement", params), } -export { organizations } +const contracts = { + contractUtilization: ( + organizationId: string, + contractId: string, + params?: AnalyticsPageParams, + ) => + contractResource( + organizationId, + contractId, + "contract-utilization", + params, + ), + enrollmentFunnel: ( + organizationId: string, + contractId: string, + params?: AnalyticsPageParams, + ) => + contractResource(organizationId, contractId, "enrollment-funnel", params), + engagementTrend: ( + organizationId: string, + contractId: string, + params?: AnalyticsPageParams, + ) => contractResource(organizationId, contractId, "engagement-trend", params), + programFunnel: ( + organizationId: string, + contractId: string, + params?: AnalyticsPageParams, + ) => contractResource(organizationId, contractId, "program-funnel", params), + contentEngagement: ( + organizationId: string, + contractId: string, + params?: AnalyticsPageParams, + ) => + contractResource(organizationId, contractId, "content-engagement", params), +} + +export { organizations, contracts } diff --git a/frontends/api/src/analytics/types.ts b/frontends/api/src/analytics/types.ts index c59a992404..ed5388576b 100644 --- a/frontends/api/src/analytics/types.ts +++ b/frontends/api/src/analytics/types.ts @@ -40,7 +40,8 @@ export type OrgAnalyticsResponse = { export type ContractUtilization = { organization_key: string organization_name: string - contract_pk: number + contract_pk: string + contract_id: string b2b_contract_name: string b2b_contract_is_active: boolean b2b_contract_start_date: string | null @@ -58,9 +59,10 @@ export type ContractUtilization = { export type EnrollmentCompletionFunnel = { organization_key: string organization_name: string - contract_pk: number + contract_pk: string + contract_id: string b2b_contract_name: string - courserun_pk: number + courserun_pk: string courserun_readable_id: string courserun_title: string enrolled_learners: number @@ -82,19 +84,25 @@ export type MonthlyEngagementTrend = { activity_year_and_month: string monthly_active_learners: number new_enrollments: number | null + enrolling_learners: number | null certificates_earned: number | null - total_videos_watched: number - total_problems_attempted: number - total_chatbot_interactions: number + certified_learners: number | null + total_videos_watched: number | null + video_watchers: number | null + total_problems_attempted: number | null + problem_attempters: number | null + total_chatbot_interactions: number | null + chatbot_users: number | null } /** `mv_b2b_program_funnel` — grain: org x contract x program. */ export type ProgramFunnel = { organization_key: string organization_name: string - contract_pk: number + contract_pk: string + contract_id: string b2b_contract_name: string - program_pk: number + program_pk: string program_title: string total_courses: number enrolled_in_contract_courses: number @@ -137,6 +145,41 @@ export type ContentEngagementDepth = { certificates_earned: number | null } +/** + * Contract identity, carried by every row of a contract-grained view. + * + * `contract_id` is MITx Online's `ContractPage.page_ptr_id` — the value in that + * dashboard's URLs, and the only one the analytics API will filter on. + * `contract_pk` is the warehouse's own md5 surrogate: useful as a stable row + * key, never as a path segment. + */ +type ContractIdentity = { + contract_pk: string + contract_id: string + b2b_contract_name: string +} + +/** + * `mv_b2b_contract_monthly_engagement_trend` — grain: org x contract x month. + * + * The same columns as {@link MonthlyEngagementTrend} plus the contract. Note + * these rows do NOT partition the org-level ones: a learner active under two + * of an org's contracts is counted in both, so summing + * `monthly_active_learners` across contracts can exceed the org's own figure. + */ +export type ContractMonthlyEngagementTrend = MonthlyEngagementTrend & + ContractIdentity + +/** + * `mv_b2b_contract_content_engagement_depth` — grain: org x contract x run. + * + * Unlike the trend view these rows ARE a partition of the org-level ones: a + * course run belongs to exactly one contract, so naming the contract labels a + * row rather than splitting it. + */ +export type ContractContentEngagementDepth = ContentEngagementDepth & + ContractIdentity + /** * LIMIT/OFFSET paging, shared by every multi-row endpoint. The API caps `limit` * at its own `max_page_size` and rejects anything larger with a 422. diff --git a/frontends/api/src/clients.ts b/frontends/api/src/clients.ts index 7e92e68292..9bb8b646f6 100644 --- a/frontends/api/src/clients.ts +++ b/frontends/api/src/clients.ts @@ -13,6 +13,7 @@ import { FeaturedApi, MediaApi, VideoPlaylistsApi, + PodcastEpisodesApi, HubspotApi, } from "./generated/v1/api" @@ -102,6 +103,11 @@ const videoPlaylistsApi = new VideoPlaylistsApi( BASE_PATH, axiosInstance, ) +const podcastEpisodesApi = new PodcastEpisodesApi( + undefined, + BASE_PATH, + axiosInstance, +) const vectorLearningResourcesSearchApi = new VectorLearningResourcesSearchApi( undefined, BASE_PATH, @@ -133,6 +139,7 @@ export { testimonialsApi, learningResourcesSearchAdminParamsApi, videoPlaylistsApi, + podcastEpisodesApi, vectorLearningResourcesSearchApi, unsubscribeApi, } diff --git a/frontends/api/src/generated/v0/api.ts b/frontends/api/src/generated/v0/api.ts index 30e6ddce9a..5026ab9861 100644 --- a/frontends/api/src/generated/v0/api.ts +++ b/frontends/api/src/generated/v0/api.ts @@ -429,10 +429,10 @@ export interface ContentFeedback { unit_title?: string /** * - * @type {string} + * @type {ContentFeedbackUrl} * @memberof ContentFeedback */ - url?: string + url?: ContentFeedbackUrl /** * * @type {ContentFeedbackSentimentEnum} @@ -491,10 +491,10 @@ export interface ContentFeedbackRequest { unit_title?: string /** * - * @type {string} + * @type {ContentFeedbackUrl} * @memberof ContentFeedbackRequest */ - url?: string + url?: ContentFeedbackUrl /** * * @type {ContentFeedbackSentimentEnum} @@ -539,6 +539,12 @@ export const ContentFeedbackSentimentEnum = { export type ContentFeedbackSentimentEnum = (typeof ContentFeedbackSentimentEnum)[keyof typeof ContentFeedbackSentimentEnum] +/** + * @type ContentFeedbackUrl + * @export + */ +export type ContentFeedbackUrl = string + /** * Serializer class for course run ContentFiles * @export @@ -679,10 +685,10 @@ export interface ContentFile { checksum?: string /** * - * @type {string} + * @type {ContentFileImageSrc} * @memberof ContentFile */ - image_src?: string | null + image_src?: ContentFileImageSrc | null /** * * @type {string} @@ -798,6 +804,12 @@ export const ContentFileContentTypeEnum = { export type ContentFileContentTypeEnum = (typeof ContentFileContentTypeEnum)[keyof typeof ContentFileContentTypeEnum] +/** + * @type ContentFileImageSrc + * @export + */ +export type ContentFileImageSrc = string + /** * SearchResponseSerializer with OpenAPI annotations for Content Files search * @export @@ -2921,10 +2933,10 @@ export interface LearningResourceOfferorDetail { content_types?: Array /** * - * @type {string} + * @type {LearningResourceOfferorDetailMoreInformation} * @memberof LearningResourceOfferorDetail */ - more_information?: string + more_information?: LearningResourceOfferorDetailMoreInformation /** * * @type {string} @@ -2938,6 +2950,12 @@ export interface LearningResourceOfferorDetail { */ display_facet?: boolean } +/** + * @type LearningResourceOfferorDetailMoreInformation + * @export + */ +export type LearningResourceOfferorDetailMoreInformation = string + /** * Serializer for LearningResourcePlatform * @export @@ -3461,10 +3479,10 @@ export interface NestedContentFile { checksum?: string /** * - * @type {string} + * @type {ContentFileImageSrc} * @memberof NestedContentFile */ - image_src?: string | null + image_src?: ContentFileImageSrc | null /** * * @type {string} @@ -4113,11 +4131,11 @@ export interface PodcastEpisode { */ parent_podcasts: Array /** - * - * @type {string} + * Whether a transcript is available from the transcript endpoint. The text itself is excluded from this serializer, so this is how a client knows whether to fetch it. + * @type {boolean} * @memberof PodcastEpisode */ - transcript?: string + has_transcript: boolean /** * * @type {string} @@ -4136,12 +4154,6 @@ export interface PodcastEpisode { * @memberof PodcastEpisode */ duration?: string | null - /** - * - * @type {string} - * @memberof PodcastEpisode - */ - rss?: string | null } /** * Minimal parent-podcast summary embedded in an episode. @@ -6216,16 +6228,16 @@ export interface Video { caption_urls: Array /** * - * @type {string} + * @type {VideoStreamingUrl} * @memberof Video */ - streaming_url: string | null + streaming_url: VideoStreamingUrl | null /** * - * @type {string} + * @type {VideoStreamingUrl} * @memberof Video */ - cover_image_url: string | null + cover_image_url: VideoStreamingUrl | null /** * * @type {string} @@ -6951,6 +6963,12 @@ export const VideoResourceResourceTypeEnum = { export type VideoResourceResourceTypeEnum = (typeof VideoResourceResourceTypeEnum)[keyof typeof VideoResourceResourceTypeEnum] +/** + * @type VideoStreamingUrl + * @export + */ +export type VideoStreamingUrl = string + /** * WidgetInstance serializer * @export diff --git a/frontends/api/src/generated/v1/api.ts b/frontends/api/src/generated/v1/api.ts index deffdac0c9..c4f4f53d4e 100644 --- a/frontends/api/src/generated/v1/api.ts +++ b/frontends/api/src/generated/v1/api.ts @@ -374,10 +374,10 @@ export interface ContentFile { checksum?: string /** * - * @type {string} + * @type {ContentFileImageSrc} * @memberof ContentFile */ - image_src?: string | null + image_src?: ContentFileImageSrc | null /** * * @type {string} @@ -493,6 +493,12 @@ export const ContentFileContentTypeEnum = { export type ContentFileContentTypeEnum = (typeof ContentFileContentTypeEnum)[keyof typeof ContentFileContentTypeEnum] +/** + * @type ContentFileImageSrc + * @export + */ +export type ContentFileImageSrc = string + /** * SearchResponseSerializer with OpenAPI annotations for Content Files search * @export @@ -4311,10 +4317,10 @@ export interface LearningResourceOfferorDetail { content_types?: Array /** * - * @type {string} + * @type {LearningResourceOfferorDetailMoreInformation} * @memberof LearningResourceOfferorDetail */ - more_information?: string + more_information?: LearningResourceOfferorDetailMoreInformation /** * * @type {string} @@ -4328,6 +4334,12 @@ export interface LearningResourceOfferorDetail { */ display_facet?: boolean } +/** + * @type LearningResourceOfferorDetailMoreInformation + * @export + */ +export type LearningResourceOfferorDetailMoreInformation = string + /** * Serializer for LearningResourceOfferor with basic details * @export @@ -5384,10 +5396,10 @@ export interface NestedContentFile { checksum?: string /** * - * @type {string} + * @type {ContentFileImageSrc} * @memberof NestedContentFile */ - image_src?: string | null + image_src?: ContentFileImageSrc | null /** * * @type {string} @@ -6514,12 +6526,18 @@ export interface PatchedWebsiteContentRequest { is_published?: boolean /** * - * @type {string} + * @type {PatchedWebsiteContentRequestSlug} * @memberof PatchedWebsiteContentRequest */ - slug?: string + slug?: PatchedWebsiteContentRequestSlug } +/** + * @type PatchedWebsiteContentRequestSlug + * @export + */ +export type PatchedWebsiteContentRequestSlug = string + /** * Serializer for PercolateQuery objects * @export @@ -6994,11 +7012,11 @@ export interface PodcastEpisode { */ parent_podcasts: Array /** - * - * @type {string} + * Whether a transcript is available from the transcript endpoint. The text itself is excluded from this serializer, so this is how a client knows whether to fetch it. + * @type {boolean} * @memberof PodcastEpisode */ - transcript?: string + has_transcript: boolean /** * * @type {string} @@ -7017,12 +7035,6 @@ export interface PodcastEpisode { * @memberof PodcastEpisode */ duration?: string | null - /** - * - * @type {string} - * @memberof PodcastEpisode - */ - rss?: string | null } /** * Minimal parent-podcast summary embedded in an episode. @@ -7055,12 +7067,6 @@ export interface PodcastEpisodeParent { * @interface PodcastEpisodeRequest */ export interface PodcastEpisodeRequest { - /** - * - * @type {string} - * @memberof PodcastEpisodeRequest - */ - transcript?: string /** * * @type {string} @@ -7079,12 +7085,6 @@ export interface PodcastEpisodeRequest { * @memberof PodcastEpisodeRequest */ duration?: string | null - /** - * - * @type {string} - * @memberof PodcastEpisodeRequest - */ - rss?: string | null } /** * Serializer for podcast episode resources @@ -7545,6 +7545,25 @@ export const PodcastEpisodeResourceResourceTypeEnum = { export type PodcastEpisodeResourceResourceTypeEnum = (typeof PodcastEpisodeResourceResourceTypeEnum)[keyof typeof PodcastEpisodeResourceResourceTypeEnum] +/** + * Serializer for a single podcast episode\'s transcript. Kept out of PodcastEpisodeSerializer so the text is only ever sent when a client asks for this one episode\'s transcript. + * @export + * @interface PodcastEpisodeTranscript + */ +export interface PodcastEpisodeTranscript { + /** + * + * @type {number} + * @memberof PodcastEpisodeTranscript + */ + id: number + /** + * + * @type {string} + * @memberof PodcastEpisodeTranscript + */ + transcript?: string +} /** * Serializer for Podcasts * @export @@ -9380,16 +9399,16 @@ export interface Video { caption_urls: Array /** * - * @type {string} + * @type {VideoStreamingUrl} * @memberof Video */ - streaming_url: string | null + streaming_url: VideoStreamingUrl | null /** * - * @type {string} + * @type {VideoStreamingUrl} * @memberof Video */ - cover_image_url: string | null + cover_image_url: VideoStreamingUrl | null /** * * @type {string} @@ -10434,6 +10453,12 @@ export const VideoResourceResourceTypeEnum = { export type VideoResourceResourceTypeEnum = (typeof VideoResourceResourceTypeEnum)[keyof typeof VideoResourceResourceTypeEnum] +/** + * @type VideoStreamingUrl + * @export + */ +export type VideoStreamingUrl = string + /** * Serializer for webhook responses. * @export @@ -10527,16 +10552,16 @@ export interface WebsiteContent { is_published?: boolean /** * - * @type {string} + * @type {PatchedWebsiteContentRequestSlug} * @memberof WebsiteContent */ - slug?: string + slug?: PatchedWebsiteContentRequestSlug /** * - * @type {string} + * @type {WebsiteContentCoverImage} * @memberof WebsiteContent */ - cover_image: string + cover_image: WebsiteContentCoverImage } /** @@ -10564,6 +10589,12 @@ export const WebsiteContentContentTypeEnum = { export type WebsiteContentContentTypeEnum = (typeof WebsiteContentContentTypeEnum)[keyof typeof WebsiteContentContentTypeEnum] +/** + * @type WebsiteContentCoverImage + * @export + */ +export type WebsiteContentCoverImage = string + /** * Serializer for WebsiteContent model. * @export @@ -10602,10 +10633,10 @@ export interface WebsiteContentRequest { is_published?: boolean /** * - * @type {string} + * @type {PatchedWebsiteContentRequestSlug} * @memberof WebsiteContentRequest */ - slug?: string + slug?: PatchedWebsiteContentRequestSlug } /** @@ -26825,6 +26856,52 @@ export const PodcastEpisodesApiAxiosParamCreator = function ( ...options.headers, } + return { + url: toPathString(localVarUrlObj), + options: localVarRequestOptions, + } + }, + /** + * Fetch one episode\'s transcript. Served separately from the episode payload because the text runs tens of kilobytes; `podcast_episode.has_transcript` says whether there is anything here to fetch. Args: id (integer): The id of the podcast episode Returns: The episode id and its transcript text + * @summary Get a podcast episode transcript + * @param {number} id + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + podcastEpisodesTranscriptRetrieve: async ( + id: number, + options: RawAxiosRequestConfig = {}, + ): Promise => { + // verify required parameter 'id' is not null or undefined + assertParamExists("podcastEpisodesTranscriptRetrieve", "id", id) + const localVarPath = `/api/v1/podcast_episodes/{id}/transcript/`.replace( + `{${"id"}}`, + encodeURIComponent(String(id)), + ) + // use dummy base URL string because the URL constructor only accepts absolute URLs. + const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL) + let baseOptions + if (configuration) { + baseOptions = configuration.baseOptions + } + + const localVarRequestOptions = { + method: "GET", + ...baseOptions, + ...options, + } + const localVarHeaderParameter = {} as any + const localVarQueryParameter = {} as any + + setSearchParams(localVarUrlObj, localVarQueryParameter) + let headersFromBaseOptions = + baseOptions && baseOptions.headers ? baseOptions.headers : {} + localVarRequestOptions.headers = { + ...localVarHeaderParameter, + ...headersFromBaseOptions, + ...options.headers, + } + return { url: toPathString(localVarUrlObj), options: localVarRequestOptions, @@ -26956,6 +27033,40 @@ export const PodcastEpisodesApiFp = function (configuration?: Configuration) { configuration, )(axios, operationBasePath || basePath) }, + /** + * Fetch one episode\'s transcript. Served separately from the episode payload because the text runs tens of kilobytes; `podcast_episode.has_transcript` says whether there is anything here to fetch. Args: id (integer): The id of the podcast episode Returns: The episode id and its transcript text + * @summary Get a podcast episode transcript + * @param {number} id + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + async podcastEpisodesTranscriptRetrieve( + id: number, + options?: RawAxiosRequestConfig, + ): Promise< + ( + axios?: AxiosInstance, + basePath?: string, + ) => AxiosPromise + > { + const localVarAxiosArgs = + await localVarAxiosParamCreator.podcastEpisodesTranscriptRetrieve( + id, + options, + ) + const index = configuration?.serverIndex ?? 0 + const operationBasePath = + operationServerMap[ + "PodcastEpisodesApi.podcastEpisodesTranscriptRetrieve" + ]?.[index]?.url + return (axios, basePath) => + createRequestFunction( + localVarAxiosArgs, + globalAxios, + BASE_PATH, + configuration, + )(axios, operationBasePath || basePath) + }, } } @@ -27020,6 +27131,21 @@ export const PodcastEpisodesApiFactory = function ( .podcastEpisodesRetrieve(requestParameters.id, options) .then((request) => request(axios, basePath)) }, + /** + * Fetch one episode\'s transcript. Served separately from the episode payload because the text runs tens of kilobytes; `podcast_episode.has_transcript` says whether there is anything here to fetch. Args: id (integer): The id of the podcast episode Returns: The episode id and its transcript text + * @summary Get a podcast episode transcript + * @param {PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest} requestParameters Request parameters. + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + podcastEpisodesTranscriptRetrieve( + requestParameters: PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest, + options?: RawAxiosRequestConfig, + ): AxiosPromise { + return localVarFp + .podcastEpisodesTranscriptRetrieve(requestParameters.id, options) + .then((request) => request(axios, basePath)) + }, } } @@ -27170,6 +27296,20 @@ export interface PodcastEpisodesApiPodcastEpisodesRetrieveRequest { readonly id: number } +/** + * Request parameters for podcastEpisodesTranscriptRetrieve operation in PodcastEpisodesApi. + * @export + * @interface PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest + */ +export interface PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest { + /** + * + * @type {number} + * @memberof PodcastEpisodesApiPodcastEpisodesTranscriptRetrieve + */ + readonly id: number +} + /** * PodcastEpisodesApi - object-oriented interface * @export @@ -27230,6 +27370,23 @@ export class PodcastEpisodesApi extends BaseAPI { .podcastEpisodesRetrieve(requestParameters.id, options) .then((request) => request(this.axios, this.basePath)) } + + /** + * Fetch one episode\'s transcript. Served separately from the episode payload because the text runs tens of kilobytes; `podcast_episode.has_transcript` says whether there is anything here to fetch. Args: id (integer): The id of the podcast episode Returns: The episode id and its transcript text + * @summary Get a podcast episode transcript + * @param {PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest} requestParameters Request parameters. + * @param {*} [options] Override http request option. + * @throws {RequiredError} + * @memberof PodcastEpisodesApi + */ + public podcastEpisodesTranscriptRetrieve( + requestParameters: PodcastEpisodesApiPodcastEpisodesTranscriptRetrieveRequest, + options?: RawAxiosRequestConfig, + ) { + return PodcastEpisodesApiFp(this.configuration) + .podcastEpisodesTranscriptRetrieve(requestParameters.id, options) + .then((request) => request(this.axios, this.basePath)) + } } /** diff --git a/frontends/api/src/hooks/hubspot/index.ts b/frontends/api/src/hooks/hubspot/index.ts index 85af0439a9..5073f57f03 100644 --- a/frontends/api/src/hooks/hubspot/index.ts +++ b/frontends/api/src/hooks/hubspot/index.ts @@ -10,6 +10,7 @@ import type { } from "../../generated/v1" import type { HubspotFormDetailResponse } from "./queries" import { hubspotKeys, hubspotQueries } from "./queries" +import type { MutationHookOptions } from "../../mutations/mutationMeta" const HUBSPOT_UTK_COOKIE = "hubspotutk" const HUBSPOT_UTK_MAX_AGE = 34190000 // ~13 months, matching HubSpot's tracking script @@ -84,8 +85,9 @@ const useHubspotFormDetail = ( }) } -const useHubspotFormSubmit = () => { +const useHubspotFormSubmit = ({ meta }: MutationHookOptions = {}) => { return useMutation({ + meta, mutationFn: ({ formId, fields, diff --git a/frontends/api/src/hooks/learningPaths/index.ts b/frontends/api/src/hooks/learningPaths/index.ts index ada723a6eb..98a96c0c86 100644 --- a/frontends/api/src/hooks/learningPaths/index.ts +++ b/frontends/api/src/hooks/learningPaths/index.ts @@ -15,6 +15,7 @@ import { learningPathsApi } from "../../clients" import { learningPathQueries, learningPathKeys } from "./queries" import { learningResourceKeys } from "../learningResources/queries" import { useUserHasPermission, Permission } from "api/hooks/user" +import type { MutationHookOptions } from "../../mutations/mutationMeta" const useLearningPathsList = ( params: ListRequest = {}, @@ -44,7 +45,7 @@ type LearningPathCreateRequest = Omit< CreateRequest["LearningPathResourceRequest"], "readable_id" | "resource_type" > -const useLearningPathCreate = () => { +const useLearningPathCreate = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (params: LearningPathCreateRequest) => @@ -54,10 +55,11 @@ const useLearningPathCreate = () => { onSettled: () => { queryClient.invalidateQueries({ queryKey: learningPathKeys.listRoot() }) }, + meta, }) } -const useLearningPathUpdate = () => { +const useLearningPathUpdate = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( @@ -76,6 +78,7 @@ const useLearningPathUpdate = () => { queryKey: learningResourceKeys.featuredRoot(), }) }, + meta, }) } diff --git a/frontends/api/src/hooks/learningResources/index.ts b/frontends/api/src/hooks/learningResources/index.ts index 3c9d42d343..8c5733f2d3 100644 --- a/frontends/api/src/hooks/learningResources/index.ts +++ b/frontends/api/src/hooks/learningResources/index.ts @@ -27,6 +27,7 @@ import { platformsQueries, learningResourceKeys, videoPlaylistQueries, + podcastEpisodeQueries, } from "./queries" import { userlistKeys } from "../userLists/queries" import { learningPathKeys } from "../learningPaths/queries" @@ -241,5 +242,6 @@ export { topicQueries, learningResourceKeys, videoPlaylistQueries, + podcastEpisodeQueries, LearningResource, } diff --git a/frontends/api/src/hooks/learningResources/queries.ts b/frontends/api/src/hooks/learningResources/queries.ts index 913093a21b..fbf6d41c8a 100644 --- a/frontends/api/src/hooks/learningResources/queries.ts +++ b/frontends/api/src/hooks/learningResources/queries.ts @@ -7,6 +7,7 @@ import { schoolsApi, featuredApi, videoPlaylistsApi, + podcastEpisodesApi, vectorLearningResourcesSearchApi, } from "../../clients" @@ -21,6 +22,7 @@ import type { LearningResourcesApiLearningResourcesSummaryListRequest as LearningResourcesSummaryListRequest, PaginatedLearningResourceRelationshipList, VideoPlaylistResource, + PodcastEpisodeTranscript, LearningResourcesApiLearningResourcesVectorSimilarListRequest, } from "../../generated/v1" import type { VectorLearningResourcesSearchApiVectorLearningResourcesSearchRetrieveRequest as VectorLearningResourcesSearchRetrieveRequest } from "../../generated/v0" @@ -289,6 +291,30 @@ const videoPlaylistQueries = { }), } +const podcastEpisodeKeys = { + root: ["podcast_episodes"], + detailRoot: () => [...podcastEpisodeKeys.root, "detail"], + detail: (id: number) => [...podcastEpisodeKeys.detailRoot(), id], + transcript: (id: number) => [...podcastEpisodeKeys.detail(id), "transcript"], +} + +const podcastEpisodeQueries = { + /** + * An episode's transcript, served separately from the episode payload + * because the text runs tens of kilobytes. Gate this on + * `podcast_episode.has_transcript` so the ~95% of episodes with no + * transcript never issue the request. + */ + transcript: (id: number) => + queryOptions({ + queryKey: podcastEpisodeKeys.transcript(id), + queryFn: () => + podcastEpisodesApi + .podcastEpisodesTranscriptRetrieve({ id }) + .then((res) => res.data), + }), +} + export { learningResourceKeys, learningResourceQueries, @@ -297,4 +323,6 @@ export { schoolQueries, offerorQueries, videoPlaylistQueries, + podcastEpisodeKeys, + podcastEpisodeQueries, } diff --git a/frontends/api/src/hooks/unsubscribe/index.ts b/frontends/api/src/hooks/unsubscribe/index.ts index 4b7dda9ff3..0f26f96730 100644 --- a/frontends/api/src/hooks/unsubscribe/index.ts +++ b/frontends/api/src/hooks/unsubscribe/index.ts @@ -1,10 +1,12 @@ import { useMutation } from "@tanstack/react-query" import { unsubscribeApi } from "../../clients" +import type { MutationHookOptions } from "../../mutations/mutationMeta" -const useUnsubscribe = () => +const useUnsubscribe = ({ meta }: MutationHookOptions = {}) => useMutation({ mutationFn: (token: string) => unsubscribeApi.unsubscribeCreate({ token }).then((res) => res.data), + meta, }) export { useUnsubscribe } diff --git a/frontends/api/src/hooks/userLists/index.ts b/frontends/api/src/hooks/userLists/index.ts index 83c29f24d5..315a8ed650 100644 --- a/frontends/api/src/hooks/userLists/index.ts +++ b/frontends/api/src/hooks/userLists/index.ts @@ -14,6 +14,7 @@ import type { } from "../../generated/v1" import { userlistKeys, userlistQueries } from "./queries" import { useUserIsAuthenticated } from "api/hooks/user" +import type { MutationHookOptions } from "../../mutations/mutationMeta" const useUserListList = ( params: ListRequest = {}, @@ -29,7 +30,7 @@ const useUserListsDetail = (id: number) => { return useQuery(userlistQueries.detail(id)) } -const useUserListCreate = () => { +const useUserListCreate = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (params: CreateRequest["UserListRequest"]) => @@ -39,9 +40,10 @@ const useUserListCreate = () => { onSettled: () => { queryClient.invalidateQueries({ queryKey: userlistKeys.listRoot() }) }, + meta, }) } -const useUserListUpdate = () => { +const useUserListUpdate = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (params: Pick & Partial) => @@ -53,6 +55,7 @@ const useUserListUpdate = () => { queryClient.invalidateQueries({ queryKey: userlistKeys.listRoot() }) queryClient.invalidateQueries({ queryKey: userlistKeys.detail(vars.id) }) }, + meta, }) } diff --git a/frontends/api/src/hooks/website_content/index.ts b/frontends/api/src/hooks/website_content/index.ts index 2ed18c7ee8..354d0cf197 100644 --- a/frontends/api/src/hooks/website_content/index.ts +++ b/frontends/api/src/hooks/website_content/index.ts @@ -8,6 +8,7 @@ import type { WebsiteContent, } from "../../generated/v1" import { websiteContentQueries, websiteContentKeys } from "./queries" +import type { MutationHookOptions } from "../../mutations/mutationMeta" const useWebsiteContentList = ( params: WebsiteContentListRequest = {}, @@ -36,7 +37,7 @@ const useWebsiteContentDetailRetrieve = (identifier: string | undefined) => { }) } -const useWebsiteContentCreate = () => { +const useWebsiteContentCreate = ({ meta }: MutationHookOptions = {}) => { const client = useQueryClient() return useMutation({ mutationFn: ( @@ -56,15 +57,17 @@ const useWebsiteContentCreate = () => { onSuccess: () => { client.invalidateQueries({ queryKey: websiteContentKeys.listRoot() }) }, + meta, }) } -export const useMediaUpload = () => { +export const useMediaUpload = ({ meta }: MutationHookOptions = {}) => { const nextProgressCb = useRef<((percent: number) => void) | undefined>( undefined, ) const mutation = useMutation({ + meta, mutationFn: async (data: { file: File }) => { const response = await mediaApi.mediaUpload( { image_file: data.file }, @@ -111,7 +114,7 @@ const useWebsiteContentDestroy = () => { }, }) } -const useWebsiteContentPartialUpdate = () => { +const useWebsiteContentPartialUpdate = ({ meta }: MutationHookOptions = {}) => { const client = useQueryClient() return useMutation({ mutationFn: ({ @@ -124,6 +127,7 @@ const useWebsiteContentPartialUpdate = () => { PatchedWebsiteContentRequest: data, }) .then((response) => response.data), + meta, onSuccess: (websiteContent: WebsiteContent) => { client.invalidateQueries({ queryKey: websiteContentKeys.detail(websiteContent.id), diff --git a/frontends/api/src/mitxonline/hooks/baskets/index.ts b/frontends/api/src/mitxonline/hooks/baskets/index.ts index a603caf661..78c9cd4ee0 100644 --- a/frontends/api/src/mitxonline/hooks/baskets/index.ts +++ b/frontends/api/src/mitxonline/hooks/baskets/index.ts @@ -2,12 +2,13 @@ import { basketQueries } from "./queries" import { useMutation, useQueryClient } from "@tanstack/react-query" import { basketsApi } from "../../clients" import type { BasketWithProduct } from "@mitodl/mitxonline-api-axios/v2" +import type { MutationHookOptions } from "../../../mutations/mutationMeta" /** * Hook to add a product to the user's basket. * Creates or updates the basket, adding the specified product. */ -const useAddToBasket = () => { +const useAddToBasket = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: async (productId: number): Promise => { @@ -22,13 +23,14 @@ const useAddToBasket = () => { queryKey: basketQueries.basketState().queryKey, }) }, + meta, }) } /** * Hook to clear the user's basket. */ -const useClearBasket = () => { +const useClearBasket = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: async (): Promise => { @@ -39,6 +41,7 @@ const useClearBasket = () => { queryKey: basketQueries.basketState().queryKey, }) }, + meta, }) } diff --git a/frontends/api/src/mitxonline/hooks/enrollment/index.ts b/frontends/api/src/mitxonline/hooks/enrollment/index.ts index 94315f6e87..8693a686ff 100644 --- a/frontends/api/src/mitxonline/hooks/enrollment/index.ts +++ b/frontends/api/src/mitxonline/hooks/enrollment/index.ts @@ -1,5 +1,6 @@ import { enrollmentQueries, enrollmentKeys } from "./queries" import { useMutation, useQueryClient } from "@tanstack/react-query" +import type { MutationHookOptions } from "../../../mutations/mutationMeta" import { b2bApi, courseRunEnrollmentsApi, @@ -14,7 +15,7 @@ import { VerifiedProgramEnrollmentsApiVerifiedProgramEnrollmentsCreateRequest, } from "@mitodl/mitxonline-api-axios/v2" -const useCreateB2bEnrollment = () => { +const useCreateB2bEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (opts: B2bApiB2bEnrollCreateRequest) => @@ -24,10 +25,11 @@ const useCreateB2bEnrollment = () => { queryKey: enrollmentKeys.courseRunEnrollmentsList(), }) }, + meta, }) } -const useCreateEnrollment = () => { +const useCreateEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (opts: CourseRunEnrollmentRequest) => { @@ -43,10 +45,11 @@ const useCreateEnrollment = () => { queryKey: enrollmentKeys.programEnrollmentsList(), }) }, + meta, }) } -const useUpdateEnrollment = () => { +const useUpdateEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (opts: EnrollmentsApiEnrollmentsPartialUpdateRequest) => @@ -56,10 +59,11 @@ const useUpdateEnrollment = () => { queryKey: enrollmentKeys.courseRunEnrollmentsList(), }) }, + meta, }) } -const useDestroyEnrollment = () => { +const useDestroyEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (enrollmentId: number) => @@ -75,10 +79,11 @@ const useDestroyEnrollment = () => { queryKey: enrollmentKeys.courseRunEnrollmentsList(), }) }, + meta, }) } -const useDestroyProgramEnrollment = () => { +const useDestroyProgramEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (programId: number) => @@ -96,10 +101,11 @@ const useDestroyProgramEnrollment = () => { queryKey: enrollmentKeys.programEnrollmentsList(), }) }, + meta, }) } -const useCreateProgramEnrollment = () => { +const useCreateProgramEnrollment = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( @@ -110,10 +116,13 @@ const useCreateProgramEnrollment = () => { queryKey: enrollmentKeys.programEnrollmentsList(), }) }, + meta, }) } -const useCreateVerifiedProgramEnrollment = () => { +const useCreateVerifiedProgramEnrollment = ({ + meta, +}: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( @@ -127,6 +136,7 @@ const useCreateVerifiedProgramEnrollment = () => { queryKey: enrollmentKeys.programEnrollmentsList(), }) }, + meta, }) } diff --git a/frontends/api/src/mitxonline/hooks/organizations/index.ts b/frontends/api/src/mitxonline/hooks/organizations/index.ts index 83d08aabef..82ebff845b 100644 --- a/frontends/api/src/mitxonline/hooks/organizations/index.ts +++ b/frontends/api/src/mitxonline/hooks/organizations/index.ts @@ -13,8 +13,12 @@ import { managerOrganizationQueries, managerOrganizationKeys, } from "./queries" +import type { MutationHookOptions } from "../../../mutations/mutationMeta" -const useB2BAttachMutation = (opts: B2bApiB2bAttachCreateRequest) => { +const useB2BAttachMutation = ( + opts: B2bApiB2bAttachCreateRequest, + { meta }: MutationHookOptions = {}, +) => { const queryClient = useQueryClient() return useMutation({ mutationFn: async () => { @@ -24,6 +28,7 @@ const useB2BAttachMutation = (opts: B2bApiB2bAttachCreateRequest) => { onSuccess: () => { queryClient.invalidateQueries({ queryKey: ["mitxonline"] }) }, + meta, }) } @@ -32,12 +37,13 @@ const useB2BAttachMutation = (opts: B2bApiB2bAttachCreateRequest) => { * auto-allocated by the backend (one per record); the response reports which * addresses were assigned and which failed. */ -const useBulkAssignSeats = () => { +const useBulkAssignSeats = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( opts: B2bApiB2bManagerOrganizationsContractsCodesBulkAssignCreateRequest, ) => b2bApi.b2bManagerOrganizationsContractsCodesBulkAssignCreate(opts), + meta, onSettled: (_data, _err, vars) => { queryClient.invalidateQueries({ queryKey: managerOrganizationKeys.contractCodesForContract( @@ -58,12 +64,13 @@ const useBulkAssignSeats = () => { } /** Resend the claim email for an assigned-but-unredeemed enrollment code. */ -const useRemindCode = () => { +const useRemindCode = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( opts: B2bApiB2bManagerOrganizationsContractsCodesRemindCreateRequest, ) => b2bApi.b2bManagerOrganizationsContractsCodesRemindCreate(opts), + meta, onSettled: (_data, _err, vars) => { queryClient.invalidateQueries({ queryKey: managerOrganizationKeys.contractCodesForContract( @@ -76,12 +83,13 @@ const useRemindCode = () => { } /** Revoke a code assignment, returning the code to the unassigned pool. */ -const useRevokeCode = () => { +const useRevokeCode = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( opts: B2bApiB2bManagerOrganizationsContractsCodesRevokeDestroyRequest, ) => b2bApi.b2bManagerOrganizationsContractsCodesRevokeDestroy(opts), + meta, onSettled: (_data, _err, vars) => { queryClient.invalidateQueries({ queryKey: managerOrganizationKeys.contractCodesForContract( @@ -106,12 +114,13 @@ const useRevokeCode = () => { * The backend updates the existing assignment in place and re-sends the claim * email. Returns 409 if the code has already been redeemed. */ -const useReassignCode = () => { +const useReassignCode = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: ( opts: B2bApiB2bManagerOrganizationsContractsCodesReassignUpdateRequest, ) => b2bApi.b2bManagerOrganizationsContractsCodesReassignUpdate(opts), + meta, onSettled: (_data, _err, vars) => { queryClient.invalidateQueries({ queryKey: managerOrganizationKeys.contractCodesForContract( @@ -124,11 +133,12 @@ const useReassignCode = () => { } /** Send a test enrollment code to the given email address. */ -const useSendTestEmail = () => +const useSendTestEmail = ({ meta }: MutationHookOptions = {}) => useMutation({ mutationFn: ( opts: B2bApiB2bManagerOrganizationsContractsCodesSendTestEmailCreateRequest, ) => b2bApi.b2bManagerOrganizationsContractsCodesSendTestEmailCreate(opts), + meta, }) export { diff --git a/frontends/api/src/mitxonline/hooks/user/index.ts b/frontends/api/src/mitxonline/hooks/user/index.ts index eabdf349a7..2f5625d935 100644 --- a/frontends/api/src/mitxonline/hooks/user/index.ts +++ b/frontends/api/src/mitxonline/hooks/user/index.ts @@ -5,6 +5,7 @@ import { useQueryClient, } from "@tanstack/react-query" import { countriesApi, usersApi } from "../../clients" +import type { MutationHookOptions } from "../../../mutations/mutationMeta" import type { User } from "@mitodl/mitxonline-api-axios/v2" import { UsersApiUsersMePartialUpdateRequest } from "@mitodl/mitxonline-api-axios/v2" @@ -39,7 +40,7 @@ const queries = { const useMitxOnlineUserMe = (opts: { enabled?: boolean } = {}) => useQuery({ ...queries.me(), ...opts }) -const useUpdateUserMutation = () => { +const useUpdateUserMutation = ({ meta }: MutationHookOptions = {}) => { const queryClient = useQueryClient() return useMutation({ mutationFn: (opts: UsersApiUsersMePartialUpdateRequest) => @@ -47,6 +48,7 @@ const useUpdateUserMutation = () => { onSuccess: () => { queryClient.invalidateQueries({ queryKey: userKeys.me() }) }, + meta, }) } diff --git a/frontends/api/src/mitxonline/test-utils/factories/pages.ts b/frontends/api/src/mitxonline/test-utils/factories/pages.ts index 9a6fdf1577..8a1a79d4f1 100644 --- a/frontends/api/src/mitxonline/test-utils/factories/pages.ts +++ b/frontends/api/src/mitxonline/test-utils/factories/pages.ts @@ -133,7 +133,7 @@ const coursePageItem: PartialFactory = (override) => { id: uniquePageId.enforce(() => faker.number.int()), include_in_learn_catalog: faker.datatype.boolean(), ingest_content_files_for_ai: faker.datatype.boolean(), - show_stay_updated: faker.datatype.boolean(), + hubspot_form_id: faker.datatype.boolean() ? faker.string.uuid() : "", length: `${faker.number.int({ min: 1, max: 12 })} weeks`, max_price: `${faker.number.int({ min: 50, max: 500 })}`, max_weekly_hours: `${faker.number.int({ min: 1, max: 20 })}`, @@ -239,7 +239,7 @@ const programPageItem: PartialFactory = (override) => { ], min_price: `${faker.number.int({ min: 1000, max: 2500 })}`, max_price: `${faker.number.int({ min: 2500, max: 10000 })}`, - show_stay_updated: faker.datatype.boolean(), + hubspot_form_id: faker.datatype.boolean() ? faker.string.uuid() : "", prerequisites: makeHTMLParagraph(1), faq_url: faker.internet.url(), about: makeHTMLParagraph(3), diff --git a/frontends/api/src/mutations/mutationMeta.ts b/frontends/api/src/mutations/mutationMeta.ts new file mode 100644 index 0000000000..f1e0e0cc68 --- /dev/null +++ b/frontends/api/src/mutations/mutationMeta.ts @@ -0,0 +1,72 @@ +/** + * Typed React Query mutation `meta`, read by the global mutation-error handler + * (a `MutationCache.onError` wired up in the `main` app's query client). + * + * The default behavior is: any mutation error shows a top-center error toast. + * A call site overrides that through `meta` — declared here as plain data so + * the `api` package stays UI-free (the toast itself lives in `main`). + * + * Augmenting `Register.mutationMeta` makes `mutation.meta` typed everywhere it + * is set (here in `api`) and read (in `main`), instead of `Record`. + */ + +/** + * NOTE: this must stay a `type` alias. As an `interface` it no longer satisfies + * the `TMutationMeta extends Record` constraint in React + * Query's `Register` lookup (interfaces lack an implicit index signature), and + * `meta` silently degrades to `Record` everywhere — with zero + * compiler errors. + */ +export type MutationErrorMeta = { + /** + * Set `false` to suppress the default error toast — e.g. when the call site + * renders its own inline error colocated with the action. + * + * Catching a `mutateAsync` rejection does NOT suppress the toast — the + * cache-level `onError` runs before the error is rethrown to the caller — so + * a call site that handles the error itself still needs this opt-out. + */ + showErrorToast?: boolean + /** Static toast copy for this mutation. */ + errorMessage?: string + /** + * Data-driven toast copy; takes precedence over `errorMessage`. Declared in + * the typed `api` hook where `TVariables` is known; the global handler passes + * the raw (`unknown`) error and variables, so cast inside the implementation: + * + * ```ts + * meta: { + * getErrorMessage: (_error, variables) => + * `Could not remove "${(variables as DestroyRequest).title}".`, + * } + * ``` + */ + getErrorMessage?: (error: unknown, variables: unknown) => string +} + +/** + * Options a shared mutation hook forwards to `useMutation`, letting a *consumer* + * tune error handling without the hook baking in a policy. Chiefly used to pass + * `meta: { showErrorToast: false }` from a call site that renders its own inline + * error, so the same hook can stay silent there and toast by default elsewhere. + */ +export type MutationHookOptions = { + meta?: MutationErrorMeta +} + +/** + * Canonical `meta` for a call site that renders its own inline error and so + * opts out of the global error toast. Named (not inlined) so every opt-out site + * is greppable and reads as a deliberate choice rather than a magic boolean. + */ +export const SILENCE_ERROR_TOAST: MutationErrorMeta = Object.freeze({ + // Frozen: this one object is shared by identity across every opt-out site, + // so a stray mutation would poison all of them. + showErrorToast: false, +}) + +declare module "@tanstack/react-query" { + interface Register { + mutationMeta: MutationErrorMeta + } +} diff --git a/frontends/api/src/test-utils/factories/learningResources.ts b/frontends/api/src/test-utils/factories/learningResources.ts index 3d5d63bfea..6323b33dc6 100644 --- a/frontends/api/src/test-utils/factories/learningResources.ts +++ b/frontends/api/src/test-utils/factories/learningResources.ts @@ -661,6 +661,9 @@ const podcastEpisode: LearningResourceFactory = ( duration: faker.helpers.arrayElement(["PT1H13M44S", "PT2H30M", "PT1M"]), audio_url: faker.internet.url(), episode_link: faker.internet.url(), + // Most episodes have no transcript: only 9 of the 38 feeds Learn + // ingests publish a podcast:transcript tag. Opt in per test. + has_transcript: false, }, }, overrides, diff --git a/frontends/api/src/test-utils/urls.ts b/frontends/api/src/test-utils/urls.ts index 11dce2a50e..8883ab5ff5 100644 --- a/frontends/api/src/test-utils/urls.ts +++ b/frontends/api/src/test-utils/urls.ts @@ -264,9 +264,15 @@ const videoPlaylists = { `${getApiBaseUrl()}/api/v1/video_playlists/${params.id}/`, } +const podcastEpisodes = { + transcript: (id: number) => + `${getApiBaseUrl()}/api/v1/podcast_episodes/${id}/transcript/`, +} + export { learningResources, videoPlaylists, + podcastEpisodes, topics, learningPaths, articles, diff --git a/frontends/main/package.json b/frontends/main/package.json index 950c928dbc..08a242128c 100644 --- a/frontends/main/package.json +++ b/frontends/main/package.json @@ -18,8 +18,8 @@ "@mitodl/arithmix": "^0.2.5", "@mitodl/course-search-utils": "^3.8.0", "@mitodl/hacksnack": "^0.1.2", - "@mitodl/mitxonline-api-axios": "2026.8.18", - "@mitodl/smoot-design": "6.33.1", + "@mitodl/mitxonline-api-axios": "2026.8.31", + "@mitodl/smoot-design": "6.33.4", "@mui/base": "5.0.0-beta.70", "@mui/material": "^6.4.5", "@mui/material-nextjs": "^6.4.3", diff --git a/frontends/main/src/app-pages/ContractAdminPage/AssignSeatsSection.tsx b/frontends/main/src/app-pages/ContractAdminPage/AssignSeatsSection.tsx index c801c88d56..b32d7deff8 100644 --- a/frontends/main/src/app-pages/ContractAdminPage/AssignSeatsSection.tsx +++ b/frontends/main/src/app-pages/ContractAdminPage/AssignSeatsSection.tsx @@ -15,6 +15,7 @@ import { useBulkAssignSeats, useSendTestEmail, } from "api/mitxonline-hooks/organizations" +import { SILENCE_ERROR_TOAST } from "api/mutation-meta" import { mitxUserQueries } from "api/mitxonline-hooks/user" import { useQuery } from "@tanstack/react-query" import type { BulkAssignError } from "@mitodl/mitxonline-api-axios/v2" @@ -232,8 +233,10 @@ const AssignSeatsSection: React.FC = ({ const [errorAnnouncement, setErrorAnnouncement] = useState("") const fileInputRef = useRef(null) - const bulkAssign = useBulkAssignSeats() - const sendTestEmail = useSendTestEmail() + // Both surface their outcome via inline Alerts (bulk-assign result Alert and + // the send-test-email alert), so suppress the global error toast. + const bulkAssign = useBulkAssignSeats({ meta: SILENCE_ERROR_TOAST }) + const sendTestEmail = useSendTestEmail({ meta: SILENCE_ERROR_TOAST }) const { data: user } = useQuery(mitxUserQueries.me()) const submitResult = useMemo( diff --git a/frontends/main/src/app-pages/ContractAdminPage/ContractAdminPage.tsx b/frontends/main/src/app-pages/ContractAdminPage/ContractAdminPage.tsx index b049ea102e..1b813dd88e 100644 --- a/frontends/main/src/app-pages/ContractAdminPage/ContractAdminPage.tsx +++ b/frontends/main/src/app-pages/ContractAdminPage/ContractAdminPage.tsx @@ -54,7 +54,7 @@ import { matchOrganizationBySlug } from "@/common/utils" import { ForbiddenError } from "@/common/errors" import { FeatureFlags } from "@/common/feature_flags" import { useFeatureFlagsLoaded } from "@/common/useFeatureFlagsLoaded" -import { organizationAnalyticsView } from "@/common/urls" +import { contractAnalyticsView } from "@/common/urls" import { ErrorContent } from "../ErrorPage/ErrorPageTemplate" import graduateLogo from "@/public/images/dashboard/graduate.png" @@ -696,15 +696,15 @@ const ContractAdminPageInternal: React.FC = ({ ) : null} - {/* Analytics is the reporting half of this dashboard and is - org-scoped, so it lives at its own URL rather than under this - contract. Behind its own flag: the analytics API is not - deployed everywhere this page is. */} + {/* Analytics is the reporting half of this dashboard, and is + now scoped to this contract rather than the whole org, so it + lives under this contract's URL. Behind its own flag: the + analytics API is not deployed everywhere this page is. */} {analyticsEnabled ? ( View analytics diff --git a/frontends/main/src/app-pages/ContractAdminPage/RowActionMenu.tsx b/frontends/main/src/app-pages/ContractAdminPage/RowActionMenu.tsx index e01eabc567..4c5c2924b1 100644 --- a/frontends/main/src/app-pages/ContractAdminPage/RowActionMenu.tsx +++ b/frontends/main/src/app-pages/ContractAdminPage/RowActionMenu.tsx @@ -21,6 +21,7 @@ import { useRevokeCode, } from "api/mitxonline-hooks/organizations" import type { ManagerEnrollmentCode } from "api/mitxonline-hooks/organizations" +import { SILENCE_ERROR_TOAST } from "api/mutation-meta" import type { AxiosError } from "axios" const ActionMenuItem = styled(MenuItem)(({ theme }) => ({ @@ -96,9 +97,11 @@ const RowActionMenu: React.FC = ({ const reassignDescId = useId() const open = Boolean(anchorEl) - const remind = useRemindCode() - const revoke = useRevokeCode() - const reassign = useReassignCode() + // These actions surface their outcome via the page-level result Alert (see + // onResult), so suppress the global error toast. + const remind = useRemindCode({ meta: SILENCE_ERROR_TOAST }) + const revoke = useRevokeCode({ meta: SILENCE_ERROR_TOAST }) + const reassign = useReassignCode({ meta: SILENCE_ERROR_TOAST }) const isRedeemed = code.redemption_status === "redeemed" const hasAssignedEmail = Boolean(code.assigned_to?.trim()) diff --git a/frontends/main/src/app-pages/DashboardPage/Analytics/CoursePerformanceTable.tsx b/frontends/main/src/app-pages/DashboardPage/Analytics/CoursePerformanceTable.tsx index e284a13fdd..08f8377d2d 100644 --- a/frontends/main/src/app-pages/DashboardPage/Analytics/CoursePerformanceTable.tsx +++ b/frontends/main/src/app-pages/DashboardPage/Analytics/CoursePerformanceTable.tsx @@ -98,7 +98,7 @@ const CoursePerformanceTable: React.FC<{ // The view's grain includes the contract, so a course run appears once per // contract it is offered under. Grouping by contract keeps those rows from // reading as duplicates; the label is dropped when there is only one. - const contracts = new Map() + const contracts = new Map() rows.forEach((row) => { const existing = contracts.get(row.contract_pk) if (existing) { diff --git a/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.test.tsx b/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.test.tsx index 86813b5194..e5979ce38b 100644 --- a/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.test.tsx +++ b/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.test.tsx @@ -1,6 +1,6 @@ import React from "react" import { renderWithProviders, screen, TestingErrorBoundary } from "@/test-utils" -import { setMockResponse } from "api/test-utils" +import { setMockResponse, makeRequest } from "api/test-utils" import { factories, urls } from "api/mitxonline-test-utils" import { factories as analyticsFactories, @@ -13,6 +13,7 @@ import type { OrganizationPage } from "@mitodl/mitxonline-api-axios/v2" import { useFeatureFlagEnabled } from "posthog-js/react" import { allowConsoleErrors } from "ol-test-utilities" import { ForbiddenError } from "@/common/errors" +import { contractAdminView, organizationAnalyticsView } from "@/common/urls" import { useFeatureFlagsLoaded } from "@/common/useFeatureFlagsLoaded" import AnalyticsContent from "./AnalyticsContent" @@ -565,7 +566,7 @@ describe("AnalyticsContent", () => { const courseRows = (count: number) => Array.from({ length: count }, (_, index) => analyticsFactories.enrollmentCompletionFunnel({ - courserun_pk: index + 1, + courserun_pk: String(index + 1), courserun_title: `Course ${index + 1}`, }), ) @@ -743,3 +744,237 @@ describe("AnalyticsContent", () => { }) }) }) + +describe("AnalyticsContent, contract-scoped", () => { + beforeEach(() => { + mockedUseFeatureFlagsLoaded.mockReturnValue(true) + mockedUseFeatureFlagEnabled.mockReturnValue(true) + setMockResponse.get( + urls.userMe.get(), + factories.user.user({ email: "manager@test.com" }), + ) + }) + + test("requests the contract-nested endpoints, resolving the slug to a contract id", async () => { + const contract = factories.contracts.contract() + const org = orgWithUuid({ contracts: [contract] }) + setManagerOrgs([org]) + + const contractId = String(contract.id) + const page = { limit: 200 } + // Only the contract-nested URLs are mocked. An org-scoped request would + // find no mock and fail the render, which is the assertion: this route must + // not silently fall back to org-wide numbers for a contract-scoped page. + setMockResponse.get( + analyticsUrls.contracts.contractUtilization(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.contractUtilization()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.engagementTrend(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractMonthlyEngagementTrend()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.enrollmentFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.enrollmentCompletionFunnel()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.programFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.programFunnel()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.contentEngagement(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractContentEngagementDepth()], + { as_of: AS_OF }, + ), + ) + + renderWithProviders( + , + ) + + await waitFor(() => { + expect(makeRequest).toHaveBeenCalledWith( + expect.objectContaining({ + url: analyticsUrls.contracts.contractUtilization( + ORG_UUID, + contractId, + page, + ), + }), + ) + }) + }) + + test("a contract slug that is not in this org requests nothing at all", async () => { + const org = orgWithUuid() + setManagerOrgs([org]) + + renderWithProviders( + , + ) + + // The slug resolves to no contract, so there is no id to put in the path. + // Better an empty state than a request with `undefined` in the URL, which + // the API would answer 403 — indistinguishable from a real denial. + // + // Await the settled unavailable state FIRST. A bare `waitFor` around a + // negative assertion passes on its first tick, before the manager-org + // lookup has even resolved, so it would hold even if a contract request + // were issued a moment later. + await screen.findByText(/This contract link could not be found/) + expect(makeRequest).not.toHaveBeenCalledWith( + expect.objectContaining({ + url: expect.stringContaining("/contracts/"), + }), + ) + // Falling back to the org's first contract here would send a manager on + // a stale or mistyped link to a different contract's seat admin. + expect( + screen.queryByRole("link", { name: "Manage seats" }), + ).not.toBeInTheDocument() + }) + + test("an unresolved contract slug points the manager at org-wide analytics, not a dead end", async () => { + const org = orgWithUuid() + setManagerOrgs([org]) + const orgSlug = org.slug.replace(/^org-/, "") + + renderWithProviders( + , + ) + + const link = await screen.findByRole("link", { + name: "organization-wide analytics", + }) + expect(link).toHaveAttribute("href", organizationAnalyticsView(orgSlug)) + }) + + test("names the contract being viewed, so a manager on a multi-contract org knows which one this is", async () => { + const contract = factories.contracts.contract({ name: "Fall 2026 Cohort" }) + const org = orgWithUuid({ contracts: [contract] }) + setManagerOrgs([org]) + + const contractId = String(contract.id) + const page = { limit: 200 } + setMockResponse.get( + analyticsUrls.contracts.contractUtilization(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.contractUtilization()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.engagementTrend(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractMonthlyEngagementTrend()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.enrollmentFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.enrollmentCompletionFunnel()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.programFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.programFunnel()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.contentEngagement(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractContentEngagementDepth()], + { as_of: AS_OF }, + ), + ) + + renderWithProviders( + , + ) + + await screen.findByText("Analytics · Fall 2026 Cohort") + }) + + test("'Manage seats' targets the contract being viewed, not the org's first", async () => { + const [first, second] = [ + factories.contracts.contract(), + factories.contracts.contract(), + ] + const org = orgWithUuid({ contracts: [first, second] }) + setManagerOrgs([org]) + + const contractId = String(second.id) + const page = { limit: 200 } + setMockResponse.get( + analyticsUrls.contracts.contractUtilization(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.contractUtilization()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.engagementTrend(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractMonthlyEngagementTrend()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.enrollmentFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.enrollmentCompletionFunnel()], + { as_of: AS_OF }, + ), + ) + setMockResponse.get( + analyticsUrls.contracts.programFunnel(ORG_UUID, contractId, page), + analyticsFactories.envelope([analyticsFactories.programFunnel()], { + as_of: AS_OF, + }), + ) + setMockResponse.get( + analyticsUrls.contracts.contentEngagement(ORG_UUID, contractId, page), + analyticsFactories.envelope( + [analyticsFactories.contractContentEngagementDepth()], + { as_of: AS_OF }, + ), + ) + + const orgSlug = org.slug.replace(/^org-/, "") + renderWithProviders( + , + ) + + // Pointing at contracts[0] here would send a manager viewing the second + // contract's analytics to the first contract's seat admin. + const link = await screen.findByRole("link", { name: "Manage seats" }) + expect(link).toHaveAttribute( + "href", + contractAdminView(orgSlug, second.slug), + ) + }) +}) diff --git a/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.tsx b/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.tsx index 770a319c48..2613baf2ce 100644 --- a/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.tsx +++ b/frontends/main/src/app-pages/DashboardPage/AnalyticsContent.tsx @@ -2,19 +2,37 @@ import React from "react" import Image from "next/image" -import { keepPreviousData, useQuery } from "@tanstack/react-query" +import Link from "next/link" +import { + keepPreviousData, + useQuery, + type UseQueryOptions, +} from "@tanstack/react-query" import type { AxiosError } from "axios" import { useFeatureFlagEnabled } from "posthog-js/react" import { Skeleton, Stack, styled, Typography } from "ol-components" import { ButtonLink } from "@mitodl/smoot-design" import { managerOrganizationQueries } from "api/mitxonline-hooks/organizations" -import { analyticsOrganizationQueries } from "api/analytics-hooks/organizations" +import { + analyticsContractQueries, + analyticsOrganizationQueries, +} from "api/analytics-hooks/organizations" +import type { + ContentEngagementDepth, + ContractContentEngagementDepth, + ContractMonthlyEngagementTrend, + ContractUtilization, + EnrollmentCompletionFunnel, + MonthlyEngagementTrend, + OrgAnalyticsResponse, + ProgramFunnel, +} from "api/analytics-hooks/organizations" import { isAnalyticsConfigured } from "api/runtime" import { matchOrganizationBySlug } from "@/common/utils" import { ForbiddenError } from "@/common/errors" import { FeatureFlags } from "@/common/feature_flags" import { useFeatureFlagsLoaded } from "@/common/useFeatureFlagsLoaded" -import { contractAdminView } from "@/common/urls" +import { contractAdminView, organizationAnalyticsView } from "@/common/urls" import { ErrorContent } from "../ErrorPage/ErrorPageTemplate" import graduateLogo from "@/public/images/dashboard/graduate.png" import ContentEngagementTable from "./Analytics/ContentEngagementTable" @@ -139,12 +157,94 @@ const isForbidden = (error: unknown) => { return status === 401 || status === 403 } +/** + * One section's query options, with the row type pinned but the query key + * widened. Org- and contract-scoped factories build keys of different lengths; + * this is what lets a single `useQuery` call accept either. + */ +type SectionQuery = (page: { + limit: number +}) => UseQueryOptions, Error> + +type SectionQueries = { + utilization: SectionQuery + trend: SectionQuery + courses: SectionQuery + programs: SectionQuery + content: SectionQuery +} + +/** + * Widens only a factory's query-key type param, leaving its row type checked + * against `SectionQueries`. Same trade `erase` makes in + * `hooks/organizations/queries.test.ts`, narrowed to the key alone: wiring a + * section to a factory whose row type doesn't match still fails to compile. + */ +const eraseKey = ( + // Nothing here reads the query key; only the row type above is meant to + // stay checked against `SectionQueries`. + factory: (page: { limit: number }) => UseQueryOptions< + OrgAnalyticsResponse, + Error, + OrgAnalyticsResponse, + // eslint-disable-next-line @typescript-eslint/no-explicit-any + any + >, +): SectionQuery => factory + +/** + * `analyticsContractQueries.engagementTrend` returns + * `ContractMonthlyEngagementTrend` -- the same columns as + * `MonthlyEngagementTrend` plus the contract identity -- rather than + * `MonthlyEngagementTrend` itself, the one section where the org- and + * contract-scoped factories disagree on row type. `eraseKey` can't bridge + * that the way it does for the other three sections: `UseQueryOptions` uses + * the row type contravariantly in `select`, so a superset row type isn't + * assignable through generics alone. `EngagementTrendChart` only reads the + * shared columns, so erasing it here is safe. + */ +const eraseContractTrendRow = ( + // Nothing here reads the query key. + factory: (page: { limit: number }) => UseQueryOptions< + OrgAnalyticsResponse, + Error, + OrgAnalyticsResponse, + // eslint-disable-next-line @typescript-eslint/no-explicit-any + any + >, +): SectionQuery => + factory as unknown as SectionQuery + +/** + * Same trade as `eraseContractTrendRow`, for the one other section where the + * contract-scoped row (`ContractContentEngagementDepth`) is a superset of the + * org-scoped one: it adds the contract identity that `ContentEngagementTable` + * never reads. + */ +const eraseContractContentRow = ( + // Nothing here reads the query key. + factory: (page: { limit: number }) => UseQueryOptions< + OrgAnalyticsResponse, + Error, + OrgAnalyticsResponse, + // eslint-disable-next-line @typescript-eslint/no-explicit-any + any + >, +): SectionQuery => + factory as unknown as SectionQuery + type AnalyticsContentInternalProps = { orgSlug: string + /** + * When present, the page is scoped to one contract: the same four sections, + * narrowed. Absent, it stays org-wide. Both routes render this component. + */ + contractSlug?: string } const AnalyticsContentInternal: React.FC = ({ orgSlug, + contractSlug, }) => { const { data: managerOrgs, @@ -160,7 +260,24 @@ const AnalyticsContentInternal: React.FC = ({ // deploy that predates mitodl/mitxonline#3789 — both must read as "analytics // unavailable" rather than a request with `undefined` in the path. const orgUuid = org?.sso_organization_id ?? null - const analyticsAvailable = isAnalyticsConfigured() && !!orgUuid + // MITx Online's contract id, resolved from the slug the route carries -- the + // same lookup ContractAdminPage does. The analytics API filters on this id, + // never on the slug, and never on the warehouse's `contract_pk` surrogate. + const contract = contractSlug + ? org?.contracts.find((c) => c.slug === contractSlug) + : undefined + const contractId = contract ? String(contract.id) : null + // A contract route whose slug matches nothing in this org is as unavailable + // as a missing org UUID: better an empty state than a request with + // `undefined` in the path, which the API would answer with a 403 for a + // contract that simply does not exist. + const analyticsAvailable = + isAnalyticsConfigured() && !!orgUuid && (!contractSlug || !!contractId) + // The narrower case of the above: the org itself is fine, only the + // contract slug doesn't resolve (a stale or mistyped link). Kept distinct + // from `analyticsAvailable` so the notice can point at the broken link + // instead of telling a manager on a perfectly valid org to contact support. + const contractUnresolved = !!contractSlug && !!orgUuid && !contractId // Per-section page size. Raised only by that section's "Show all", so // expanding a truncated course table never refetches the other three. @@ -182,38 +299,82 @@ const AnalyticsContentInternal: React.FC = ({ // back to its skeleton, which is exactly what the option was there to prevent. // It behaves as documented on `useQuery` (as it does on ContractAdminPage). // The count is fixed and the order never changes, so this is hook-safe. + // + // Org- and contract-scoped queries return the identical envelope and row + // shapes, so the scope is chosen once here and nothing below this line + // changes with it. The hook count and order stay fixed either way. + // + // The `eraseKey` wrap on each factory is load-bearing. The two branches + // build query keys of different lengths (contract keys carry two more + // segments) and `queryOptions` is invariant in the key type, so the + // inferred union leaves `useQuery` unable to pick an overload. Widening + // just the key -- rather than the whole options object -- keeps each + // section's row type checked against `SectionQueries`. + const scoped = React.useMemo(() => { + const orgId = orgUuid ?? "" + return contractId + ? { + utilization: eraseKey((page) => + analyticsContractQueries.contractUtilization( + orgId, + contractId, + page, + ), + ), + trend: eraseContractTrendRow((page) => + analyticsContractQueries.engagementTrend(orgId, contractId, page), + ), + courses: eraseKey((page) => + analyticsContractQueries.enrollmentFunnel(orgId, contractId, page), + ), + programs: eraseKey((page) => + analyticsContractQueries.programFunnel(orgId, contractId, page), + ), + content: eraseContractContentRow((page) => + analyticsContractQueries.contentEngagement(orgId, contractId, page), + ), + } + : { + utilization: eraseKey((page) => + analyticsOrganizationQueries.contractUtilization(orgId, page), + ), + trend: eraseKey((page) => + analyticsOrganizationQueries.engagementTrend(orgId, page), + ), + courses: eraseKey((page) => + analyticsOrganizationQueries.enrollmentFunnel(orgId, page), + ), + programs: eraseKey((page) => + analyticsOrganizationQueries.programFunnel(orgId, page), + ), + content: eraseKey((page) => + analyticsOrganizationQueries.contentEngagement(orgId, page), + ), + } + }, [orgUuid, contractId]) satisfies SectionQueries + const utilization = useQuery({ - ...analyticsOrganizationQueries.contractUtilization(orgUuid ?? "", { - limit: limits.utilization, - }), + ...scoped.utilization({ limit: limits.utilization }), enabled: analyticsAvailable, placeholderData: keepPreviousData, }) const trend = useQuery({ - ...analyticsOrganizationQueries.engagementTrend(orgUuid ?? "", { - limit: limits.trend, - }), + ...scoped.trend({ limit: limits.trend }), enabled: analyticsAvailable, placeholderData: keepPreviousData, }) const courses = useQuery({ - ...analyticsOrganizationQueries.enrollmentFunnel(orgUuid ?? "", { - limit: limits.courses, - }), + ...scoped.courses({ limit: limits.courses }), enabled: analyticsAvailable, placeholderData: keepPreviousData, }) const programs = useQuery({ - ...analyticsOrganizationQueries.programFunnel(orgUuid ?? "", { - limit: limits.programs, - }), + ...scoped.programs({ limit: limits.programs }), enabled: analyticsAvailable, placeholderData: keepPreviousData, }) const content = useQuery({ - ...analyticsOrganizationQueries.contentEngagement(orgUuid ?? "", { - limit: limits.content, - }), + ...scoped.content({ limit: limits.content }), enabled: analyticsAvailable, placeholderData: keepPreviousData, }) @@ -275,7 +436,13 @@ const AnalyticsContentInternal: React.FC = ({ return } - const firstContractSlug = org.contracts[0]?.slug + // On a contract-scoped page this must be the contract being viewed, not the + // org's first one, or "Manage seats" silently sends the manager to a + // different contract's admin page. And when that contract fails to + // resolve, it must stay undefined rather than falling back to the org's + // first contract, or the unresolved-link notice renders a "Manage seats" + // button pointing at an unrelated contract. + const manageSeatsSlug = contractSlug ? contract?.slug : org.contracts[0]?.slug const header = ( @@ -285,14 +452,16 @@ const AnalyticsContentInternal: React.FC = ({
{org.name} - Analytics + + {contract ? `Analytics · ${contract.name}` : "Analytics"} +
- {firstContractSlug ? ( + {manageSeatsSlug ? ( Manage seats @@ -305,11 +474,21 @@ const AnalyticsContentInternal: React.FC = ({ {header} - {isAnalyticsConfigured() - ? // The org record has no Keycloak organization UUID, which is - // what the analytics API keys on. Nothing the manager can fix. - "Analytics is not available for this organization yet. Please contact support if you expect to see data here." - : "Analytics is not available in this environment."} + {contractUnresolved ? ( + <> + This contract link could not be found for {org.name}. Try{" "} + + organization-wide analytics + {" "} + instead. + + ) : isAnalyticsConfigured() ? ( + // The org record has no Keycloak organization UUID, which is + // what the analytics API keys on. Nothing the manager can fix. + "Analytics is not available for this organization yet. Please contact support if you expect to see data here." + ) : ( + "Analytics is not available in this environment." + )} ) @@ -415,9 +594,13 @@ const AnalyticsContentInternal: React.FC = ({ type AnalyticsContentProps = { orgSlug: string + contractSlug?: string } -const AnalyticsContent: React.FC = ({ orgSlug }) => { +const AnalyticsContent: React.FC = ({ + orgSlug, + contractSlug, +}) => { const flagEnabled = useFeatureFlagEnabled(FeatureFlags.B2BAnalyticsDashboard) const flagsLoaded = useFeatureFlagsLoaded() @@ -431,7 +614,9 @@ const AnalyticsContent: React.FC = ({ orgSlug }) => { throw new ForbiddenError("Not enabled.") } - return + return ( + + ) } export default AnalyticsContent diff --git a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.test.tsx b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.test.tsx index 00dad8c511..37803725c8 100644 --- a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.test.tsx +++ b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.test.tsx @@ -110,6 +110,83 @@ describe("DashboardDialogs", () => { ) }) + test("The email settings dialog shows an inline error when the update fails", async () => { + const { enrollments } = setupApis() + const enrollment = faker.helpers.arrayElement(enrollments) + + setMockResponse.patch( + mitxonline.urls.enrollment.courseEnrollment(enrollment.id), + {}, + { code: 500 }, + ) + renderWithProviders() + + const cards = await screen.findAllByTestId("enrollment-card-desktop") + const card = cards.find( + (c) => !!within(c).queryByText(enrollment.run.title), + ) + invariant(card) + + await user.click(await within(card).findByLabelText("More options")) + await user.click( + await screen.findByRole("menuitem", { name: "Email Settings" }), + ) + + const dialog = await screen.findByRole("dialog", { name: "Email Settings" }) + await user.click( + within(dialog).getByRole("checkbox", { name: "Receive course emails" }), + ) + await user.click( + within(dialog).getByRole("button", { name: "Save Settings" }), + ) + + // The dialog always renders a warning alert about unchecking the box, so the + // failure adds a second alert; pinning the count keeps a stray extra alert + // from passing as the inline error. + await waitFor(() => + expect(within(dialog).getAllByRole("alert")).toHaveLength(2), + ) + const [, errorAlert] = within(dialog).getAllByRole("alert") + expect(errorAlert).toHaveTextContent( + "There was a problem updating your email settings. Please try again later.", + ) + // Dialog stays open so the user can retry. + expect(dialog).toBeInTheDocument() + }) + + test("The unenroll dialog shows an inline error when the unenroll fails", async () => { + const { enrollments } = setupApis() + const enrollment = faker.helpers.arrayElement(enrollments) + + setMockResponse.delete( + mitxonline.urls.enrollment.courseEnrollment(enrollment.id), + {}, + { code: 500 }, + ) + renderWithProviders() + + const cards = await screen.findAllByTestId("enrollment-card-desktop") + const card = cards.find( + (c) => !!within(c).queryByText(enrollment.run.title), + ) + invariant(card) + + await user.click(await within(card).findByLabelText("More options")) + await user.click(await screen.findByRole("menuitem", { name: "Unenroll" })) + + const dialog = await screen.findByRole("dialog", { + name: `Unenroll from ${enrollment.run.title}`, + }) + await user.click(within(dialog).getByRole("button", { name: "Unenroll" })) + + expect(await within(dialog).findByRole("alert")).toHaveTextContent( + "There was a problem unenrolling you from this course. Please try again later.", + ) + // The card survives a failed unenroll. + expect(card).toBeInTheDocument() + expect(trackCourseUnenrolled).not.toHaveBeenCalled() + }) + test("Opening the unenroll dialog and confirming the unenroll fires the proper API call", async () => { const { enrollments } = setupApis() const enrollment = faker.helpers.arrayElement(enrollments) @@ -335,6 +412,39 @@ describe("UnenrollProgramDialog", () => { expect(trackCourseUnenrolled).not.toHaveBeenCalled() }) + test("Shows an inline error when the unenroll fails", async () => { + const { programEnrollment } = setupProgramCard("audit", null) + + setMockResponse.delete( + mitxonline.urls.programEnrollments.programEnrollment( + programEnrollment.program.id, + ), + {}, + { code: 500 }, + ) + + renderWithProviders( + , + ) + + const desktopCard = await screen.findByTestId("enrollment-card-desktop") + await user.click(within(desktopCard).getByLabelText("More options")) + await user.click(await screen.findByRole("menuitem", { name: "Unenroll" })) + + const dialog = await screen.findByRole("dialog", { + name: `Unenroll from ${programEnrollment.program.title}`, + }) + await user.click(within(dialog).getByRole("button", { name: "Unenroll" })) + + expect(await within(dialog).findByRole("alert")).toHaveTextContent( + "There was a problem unenrolling you from this program. Please try again later.", + ) + expect(trackProgramUnenrolled).not.toHaveBeenCalled() + }) + test("Cancelling the dialog does not fire the API call", async () => { const { programEnrollment } = setupProgramCard("audit", null) diff --git a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.tsx b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.tsx index b41781c19a..5262a4f3f3 100644 --- a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.tsx +++ b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/DashboardDialogs.tsx @@ -16,6 +16,7 @@ import { useDestroyProgramEnrollment, useUpdateEnrollment, } from "api/mitxonline-hooks/enrollment" +import { SILENCE_ERROR_TOAST } from "api/mutation-meta" import { CourseRunEnrollmentV3 } from "@mitodl/mitxonline-api-axios/v2" import { trackCourseUnenrolled, @@ -42,19 +43,21 @@ const EmailSettingsDialogInner: React.FC = ({ initialValues: { receive_emails: enrollment.edx_emails_subscription ?? true, }, - onSubmit: async () => { - await updateEnrollment.mutateAsync({ - id: enrollment.id, - PatchedUpdateCourseRunEnrollmentRequest: { - receive_emails: formik.values.receive_emails, + onSubmit: () => { + updateEnrollment.mutate( + { + id: enrollment.id, + PatchedUpdateCourseRunEnrollmentRequest: { + receive_emails: formik.values.receive_emails, + }, }, - }) - if (!updateEnrollment.isError) { - modal.hide() - } + { onSuccess: () => modal.hide() }, + ) }, }) - const updateEnrollment = useUpdateEnrollment() + // Renders its own inline error below (updateEnrollment.isError), so suppress + // the global error toast. + const updateEnrollment = useUpdateEnrollment({ meta: SILENCE_ERROR_TOAST }) return ( = ({ enrollment, }) => { const modal = NiceModal.useModal() - const destroyEnrollment = useDestroyEnrollment() + // Renders its own inline error below (destroyEnrollment.isError), so suppress + // the global error toast. + const destroyEnrollment = useDestroyEnrollment({ meta: SILENCE_ERROR_TOAST }) const formik = useFormik({ enableReinitialize: true, validateOnChange: false, validateOnBlur: false, initialValues: {}, - onSubmit: async () => { - await destroyEnrollment.mutateAsync(enrollment.id) - if (!destroyEnrollment.isError) { - trackCourseUnenrolled(title) - modal.hide() - } + onSubmit: () => { + destroyEnrollment.mutate(enrollment.id, { + onSuccess: () => { + trackCourseUnenrolled(title) + modal.hide() + }, + }) }, }) return ( @@ -189,16 +195,23 @@ const UnenrollProgramDialogInner: React.FC = ({ programId, }) => { const modal = NiceModal.useModal() - const destroyProgramEnrollment = useDestroyProgramEnrollment() + // Renders its own inline error below (destroyProgramEnrollment.isError), so + // suppress the global error toast. + const destroyProgramEnrollment = useDestroyProgramEnrollment({ + meta: SILENCE_ERROR_TOAST, + }) const formik = useFormik({ enableReinitialize: true, validateOnChange: false, validateOnBlur: false, initialValues: {}, - onSubmit: async () => { - await destroyProgramEnrollment.mutateAsync(programId) - trackProgramUnenrolled(title) - modal.hide() + onSubmit: () => { + destroyProgramEnrollment.mutate(programId, { + onSuccess: () => { + trackProgramUnenrolled(title) + modal.hide() + }, + }) }, }) return ( diff --git a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/EnrolledCourseCard.tsx b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/EnrolledCourseCard.tsx index 0dbf8eaae9..87bcdf47d2 100644 --- a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/EnrolledCourseCard.tsx +++ b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/EnrolledCourseCard.tsx @@ -30,6 +30,7 @@ import { RiArrowUpCircleLine, RiAwardLine, RiMore2Line } from "@remixicon/react" import { useReplaceBasketItem } from "@/common/mitxonline/useReplaceBasketItem" import { useComplianceGate } from "@/common/mitxonline/useComplianceGate" import { useCreateVerifiedProgramEnrollment } from "api/mitxonline-hooks/enrollment" +import { SILENCE_ERROR_TOAST } from "api/mutation-meta" import { isInPast, calendarDaysUntil, NoSSR } from "ol-utilities" import { SiblingRunsPanel, SiblingRunsToggle } from "./SiblingRunsAccordion" import { EnrollmentStatusIcon } from "./EnrollmentStatus" @@ -98,8 +99,16 @@ const UpgradeBanner: React.FC< onUpgradeFailure, ...others }) => { - const replaceBasketItem = useReplaceBasketItem() - const createVerifiedProgramEnrollment = useCreateVerifiedProgramEnrollment() + // Upgrade failures are caught below and surfaced via onUpgradeFailure (an + // inline alert in the parent), so suppress the global error toast. A caught + // mutateAsync rejection does NOT suppress the cache-level onError, so this + // opt-out is what prevents a double alert. `onUpgradeFailure` is optional — + // a caller that omits it has no error surface of its own, so only silence + // the toast when the callback is actually wired. + const upgradeErrorMeta = onUpgradeFailure ? { meta: SILENCE_ERROR_TOAST } : {} + const replaceBasketItem = useReplaceBasketItem(upgradeErrorMeta) + const createVerifiedProgramEnrollment = + useCreateVerifiedProgramEnrollment(upgradeErrorMeta) const { ensureCompliance } = useComplianceGate() const programRequestBody = programReadableIds?.length diff --git a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/UnenrolledCourseCard.test.tsx b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/UnenrolledCourseCard.test.tsx index 45b099a984..5d799ce52b 100644 --- a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/UnenrolledCourseCard.test.tsx +++ b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/UnenrolledCourseCard.test.tsx @@ -1,5 +1,6 @@ import React from "react" import { + expectErrorToast, renderWithProviders, screen, setMockResponse, @@ -822,6 +823,75 @@ describe.each([ }) }) +describe("UnenrolledCourseCard enrollment error toast", () => { + setupLocationMock() + + const getCard = () => screen.getByTestId("enrollment-card-desktop") + + test("Failed course enrollment surfaces a course-specific error toast", async () => { + setupUserApis() + const run = mitxonline.factories.courses.courseRun({ + b2b_contract: null, + is_enrollable: true, + enrollment_modes: [ + mitxonline.factories.courses.enrollmentMode({ + requires_payment: false, + }), + ], + }) + const course = mitxOnlineCourse({ courseruns: [run], next_run_id: run.id }) + + setMockResponse.get(mitxonline.urls.enrollment.enrollmentsListV3(), []) + setMockResponse.post( + mitxonline.urls.enrollment.enrollmentsListV1(), + {}, + { code: 500 }, + ) + + renderWithProviders() + + await user.click(within(getCard()).getByTestId("courseware-button")) + + await expectErrorToast( + "Something went wrong enrolling you in this course. Please try again.", + ) + }) + + test("Failed verified program enrollment surfaces a program-specific error toast", async () => { + setupUserApis() + const run = mitxonline.factories.courses.courseRun({ + b2b_contract: null, + is_enrollable: true, + courseware_url: faker.internet.url(), + }) + const course = mitxOnlineCourse({ courseruns: [run], next_run_id: run.id }) + + const programEnrollment = + mitxonline.factories.enrollment.programEnrollmentV3({ + enrollment_mode: "verified", + }) + + setMockResponse.post( + mitxonline.urls.verifiedProgramEnrollments.create(run.courseware_id), + {}, + { code: 500 }, + ) + + renderWithProviders( + , + ) + + await user.click(within(getCard()).getByTestId("courseware-button")) + + await expectErrorToast( + "Something went wrong enrolling you in this program. Please try again.", + ) + }) +}) + describe("UnenrolledCourseCard card type label", () => { setupLocationMock() diff --git a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/hooks/useEnrollmentHandler.ts b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/hooks/useEnrollmentHandler.ts index fca020debd..4728e13f36 100644 --- a/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/hooks/useEnrollmentHandler.ts +++ b/frontends/main/src/app-pages/DashboardPage/CoursewareDisplay/hooks/useEnrollmentHandler.ts @@ -15,10 +15,21 @@ import { useComplianceGate } from "@/common/mitxonline/useComplianceGate" import CourseEnrollmentDialog from "@/page-components/EnrollmentDialogs/CourseEnrollmentDialog" import { trackCourseEnrolled } from "@/common/analytics/gtm" +const ENROLL_COURSE_ERROR = + "Something went wrong enrolling you in this course. Please try again." +const ENROLL_PROGRAM_ERROR = + "Something went wrong enrolling you in this program. Please try again." + export const useEnrollmentHandler = () => { - const createB2bEnrollment = useCreateB2bEnrollment() - const createEnrollment = useCreateEnrollment() - const createVerifiedProgramEnrollment = useCreateVerifiedProgramEnrollment() + const createB2bEnrollment = useCreateB2bEnrollment({ + meta: { errorMessage: ENROLL_COURSE_ERROR }, + }) + const createEnrollment = useCreateEnrollment({ + meta: { errorMessage: ENROLL_COURSE_ERROR }, + }) + const createVerifiedProgramEnrollment = useCreateVerifiedProgramEnrollment({ + meta: { errorMessage: ENROLL_PROGRAM_ERROR }, + }) const replaceBasketItem = useReplaceBasketItem() const { ensureCompliance } = useComplianceGate() diff --git a/frontends/main/src/app-pages/EnrollmentCodePage/EnrollmentCodePage.tsx b/frontends/main/src/app-pages/EnrollmentCodePage/EnrollmentCodePage.tsx index d46ffd48eb..2e0fd160bc 100644 --- a/frontends/main/src/app-pages/EnrollmentCodePage/EnrollmentCodePage.tsx +++ b/frontends/main/src/app-pages/EnrollmentCodePage/EnrollmentCodePage.tsx @@ -8,6 +8,7 @@ import { } from "@/common/mitxonline" import * as urls from "@/common/urls" import { useB2BAttachMutation } from "api/mitxonline-hooks/organizations" +import { SILENCE_ERROR_TOAST } from "api/mutation-meta" import { userQueries } from "api/hooks/user" import { useQuery } from "@tanstack/react-query" import { useRouter } from "next-nprogress-bar" @@ -24,9 +25,14 @@ const InterstitialMessage = styled(Typography)(({ theme }) => ({ const EnrollmentCodePage: React.FC = ({ code }) => { const router = useRouter() - const enrollment = useB2BAttachMutation({ - enrollment_code: code, - }) + // Failure redirects to an error page (onError below), so suppress the global + // error toast. + const enrollment = useB2BAttachMutation( + { + enrollment_code: code, + }, + { meta: SILENCE_ERROR_TOAST }, + ) const { isLoading: userLoading, data: user } = useQuery({ ...userQueries.me(), diff --git a/frontends/main/src/app-pages/PodcastPage/EpisodeContentTabs.tsx b/frontends/main/src/app-pages/PodcastPage/EpisodeContentTabs.tsx new file mode 100644 index 0000000000..1e19ea196e --- /dev/null +++ b/frontends/main/src/app-pages/PodcastPage/EpisodeContentTabs.tsx @@ -0,0 +1,255 @@ +"use client" + +import React, { useMemo, useState } from "react" +import { + Skeleton, + TabContext, + TabPanel, + Typography, + TypographyProps, + styled, +} from "ol-components" +import { TabButton, TabButtonList, VisuallyHidden } from "@mitodl/smoot-design" + +const DESCRIPTION_TAB = "description" +const TRANSCRIPT_TAB = "transcript" + +const TabsList = styled(TabButtonList)({ + // TabButtonList defaults to variant="scrollable", which renders scroll + // buttons that are permanently disabled for two tabs. + ".MuiTabScrollButton-root.Mui-disabled": { + display: "none", + }, + marginTop: "32px", +}) + +// The Pick generic is what allows component="div" +// below; both panels need it, since the sanitized description contains block +// elements that are invalid inside Typography's default . +const Body = styled(Typography)>( + ({ theme }) => ({ + color: theme.custom.colors.darkGray2, + display: "block", + marginBottom: "32px", + marginTop: "32px", + fontSize: "18px", + fontStyle: "normal", + lineHeight: "32px", + a: { + textDecoration: "underline", + color: theme.custom.colors.darkGray2, + fontWeight: theme.typography.fontWeightMedium, + }, + "a:hover": { + textDecoration: "none", + }, + [theme.breakpoints.down("sm")]: { + ...theme.typography.body1, + lineHeight: "24px", + marginTop: "16px", + }, + }), +) + +const TranscriptBody = styled(Body)({ + p: { + margin: 0, + }, + "p + p": { + marginTop: "24px", + }, +}) as typeof Body + +// Focus lands on the panel rather than on a child, so the ring belongs here. +const ContentPanel = styled(TabPanel)({ + "&:focus-visible": { + outlineOffset: "4px", + }, +}) + +// A tabpanel whose content holds nothing focusable takes a tab stop of its own +// so keyboard users can reach it; one that does hold something focusable must +// not (WAI-ARIA APG). Descriptions are nh3-sanitized with allowed, so a +// link is the only focusable thing one can contain. +const CONTAINS_LINK_RE = /]/i + +const SkeletonLine = styled(Skeleton)({ + marginBottom: "16px", +}) + +const TRANSCRIPT_SKELETON_WIDTHS = ["100%", "97%", "92%", "100%", "60%"] + +const TranscriptSkeleton = () => ( +
+ {TRANSCRIPT_SKELETON_WIDTHS.map((width, index) => ( + + ))} +
+) + +/** + * The transcript's fetch state, as the tabs need to see it. + * + * A discriminated union rather than separate `hasTranscript` / `isLoading` / + * `isError` props: the combinations those would allow (loading *and* error, + * text *and* absent) have no meaning here, and the panel renders exactly one + * of these four cases. + */ +export type TranscriptState = + /** No transcript to show, so no Transcript tab. */ + | { status: "absent" } + /** The episode reports a transcript and the request is in flight. */ + | { status: "loading" } + /** The episode reports a transcript and the request failed. */ + | { status: "error" } + | { status: "ready"; text: string } + +// Announced by the panel's live region as the state changes. +const TRANSCRIPT_STATUS_MESSAGES: Record = { + absent: "", + loading: "Loading transcript", + error: "The transcript could not be loaded.", + ready: "Transcript loaded", +} + +type EpisodeContentTabsProps = { + /** + * Sanitized description HTML. Sanitized with nh3 during ETL, so it is safe to + * render verbatim -- the same trust model as resource descriptions elsewhere. + */ + descriptionHtml: string | null + /** + * The transcript and its fetch state. `ready` carries normalized plain text + * with paragraphs separated by a blank line, rendered as escaped text and + * never as HTML: unlike the description this is third-party text that was + * never sanitized for markup. + */ + transcript: TranscriptState +} + +/** + * Description and Transcript as tabs, defaulting to Description. + * + * Both panels stay mounted at all times and only the `hidden` attribute + * toggles. That is what keeps the transcript crawlable: search engines index + * DOM content hidden with CSS or `hidden`, but never content that appears only + * after a click. `keepMounted` is load-bearing -- @mui/lab's TabPanel unmounts + * the inactive panel without it. + * + * The tablist appears as soon as the episode reports a transcript, not once the + * text arrives, so the tab set does not shift under someone already reading the + * description. Until then the Transcript panel carries a skeleton and + * `aria-busy`, and a failed fetch says so rather than silently dropping the tab. + * + * With no transcript at all there is nothing to switch between, so the + * description renders on its own with no tablist. + */ +const EpisodeContentTabs: React.FC = ({ + descriptionHtml, + transcript, +}) => { + const [tab, setTab] = useState(DESCRIPTION_TAB) + + const paragraphs = useMemo( + () => + transcript.status === "ready" + ? transcript.text.split("\n\n").filter(Boolean) + : [], + [transcript], + ) + + const descriptionIsFocusable = + !!descriptionHtml && CONTAINS_LINK_RE.test(descriptionHtml) + + const description = descriptionHtml ? ( + // Rendered as a
, not the default

: the sanitized description + // contains block elements which are invalid inside a

, and the browser + // would reparent them and break hydration. + + ) : null + + const transcriptContent = ( + <> + {/* + A live region that stays mounted across all three states, so changing + its text is what announces the outcome. Replacing one element with + another would not: removing a live region announces nothing. It sits + inside the panel, so nothing is announced while the reader is on the + Description tab -- a `hidden` panel is out of the accessibility tree. + */} + + {TRANSCRIPT_STATUS_MESSAGES[transcript.status]} + + {transcript.status === "ready" ? ( + + {paragraphs.map((paragraph, index) => ( + // eslint-disable-next-line react/no-array-index-key +

{paragraph}

+ ))} + + ) : ( + + {transcript.status === "loading" ? ( + + ) : ( + "The transcript could not be loaded. Reload the page to try again." + )} + + )} + + ) + + // Tabs only earn their place when there are two things to switch between. + // With no transcript, the description stands alone; with a transcript but no + // description, showing tabs would open on an empty Description panel, so the + // transcript stands alone instead. + if (transcript.status === "absent") { + return description + } + if (!descriptionHtml) { + return transcriptContent + } + + return ( + + setTab(value)} + > + + + + + {description} + + + {transcriptContent} + + + ) +} + +export default EpisodeContentTabs diff --git a/frontends/main/src/app-pages/PodcastPage/PodcastEpisodeDetailPage.test.tsx b/frontends/main/src/app-pages/PodcastPage/PodcastEpisodeDetailPage.test.tsx index 4576fdfa0e..5f25f8c595 100644 --- a/frontends/main/src/app-pages/PodcastPage/PodcastEpisodeDetailPage.test.tsx +++ b/frontends/main/src/app-pages/PodcastPage/PodcastEpisodeDetailPage.test.tsx @@ -2,7 +2,7 @@ import React from "react" import { factories, setMockResponse, urls } from "api/test-utils" import { ResourceTypeEnum } from "api/v1" import type { LearningResource, PodcastEpisodeResource } from "api/v1" -import { renderWithProviders, screen, user } from "@/test-utils" +import { renderWithProviders, screen, user, waitFor } from "@/test-utils" import { PodcastEpisodeDetailPage } from "./PodcastEpisodeDetailPage" jest.mock("./PodcastPlayer", () => @@ -44,12 +44,25 @@ type SetupOptions = { episodeOverrides?: Partial podcastOverrides?: Partial moreEpisodes?: LearningResource[] + /** + * When set, the episode reports has_transcript and the transcript endpoint + * returns this text. Paragraphs are separated by a blank line, as the ETL + * normalizer emits them. + */ + transcript?: string + /** Report has_transcript, but leave the transcript request in flight. */ + transcriptPending?: boolean + /** Report has_transcript, but fail the transcript request. */ + transcriptFails?: boolean } const setupApis = ({ episodeOverrides = {}, podcastOverrides = {}, moreEpisodes, + transcript, + transcriptPending = false, + transcriptFails = false, }: SetupOptions = {}) => { const podcast = makePodcast(podcastOverrides) const episodeOverridesEpisode = ( @@ -66,10 +79,33 @@ const setupApis = ({ readable_id: podcast.readable_id, }, ], + has_transcript: + transcript !== undefined || transcriptPending || transcriptFails, ...episodeOverridesEpisode, }, } as Partial) + if (transcript !== undefined) { + setMockResponse.get(urls.podcastEpisodes.transcript(episode.id), { + id: episode.id, + transcript, + }) + } + if (transcriptPending) { + // A promise that never settles keeps the query in its loading state. + setMockResponse.get( + urls.podcastEpisodes.transcript(episode.id), + new Promise(() => {}), + ) + } + if (transcriptFails) { + setMockResponse.get( + urls.podcastEpisodes.transcript(episode.id), + "Server error", + { code: 500 }, + ) + } + setMockResponse.get( urls.learningResources.details({ id: episode.id }), episode, @@ -283,6 +319,9 @@ describe("PodcastEpisodeDetailPage", () => { test("names the URL's podcast (not the first parent) for a multi-parent episode", async () => { const episode = makePodcastEpisode() episode.podcast_episode.audio_url = "https://example.com/ep.mp3" + // The resource factory leaves last_modified unset, and the JSON-LD is + // omitted without it. + episode.last_modified = "2026-01-02T03:04:05Z" const podcastA = makePodcast({ title: "Podcast A" }) const podcastB = makePodcast({ title: "Podcast B" }) // The episode belongs to both A and B; the user is on B's URL. @@ -328,6 +367,55 @@ describe("PodcastEpisodeDetailPage", () => { expect(screen.getByTestId("player-podcast-name")).toHaveTextContent( "Podcast B", ) + + // So must the JSON-LD: partOfSeries takes its url from the podcast in the + // current route, so taking the name from parent_podcasts[0] instead would + // publish Podcast A's name against Podcast B's url. + const jsonLd = JSON.parse( + document.querySelector('script[type="application/ld+json"]')!.innerHTML, + ) + expect(jsonLd.partOfSeries).toEqual( + expect.objectContaining({ + "@type": "PodcastSeries", + name: "Podcast B", + url: expect.stringContaining(`/podcast/${podcastB.id}/`), + }), + ) + }) + + test("escapes every < in the JSON-LD, not just { + // Episode titles come straight from third-party RSS with no sanitization + // (podcast.transform_episode reads rss_data.title.text verbatim). Escaping + // only `` but not `\s*[\d:.,]+") +# A line that is only a timestamp, e.g. Captivate's . +STAMP_ONLY_RE = re.compile(r"^[\[(]?\d{1,3}:\d{2}(?::\d{2})?(?:[.,]\d{1,3})?[\])]?$") +VTT_VOICE_RE = re.compile(r"]*)?\s+([^>]+)>", re.IGNORECASE) +VTT_TAG_RE = re.compile(r"]*)?(?:\s[^>]*)?>|<\d{2}:[\d:.]+>", re.I) +INLINE_SPEAKER_RE = re.compile(r"^([^:\n]{2,60}?):\s+(?=\S)") +SENTENCE_END_RE = re.compile(r"[.!?][\"')\]]*$") + + +def parse_transcript_tags(item) -> list[dict]: + """ + Read every ```` tag off one feed ````. + + The tags are matched on local name with a prefix check rather than as + ``"podcast:transcript"``, because the prefix is frequently absent by the + time we look. lxml in recover mode drops an *undeclared* namespace prefix, + so a feed emitting the tag without ``xmlns:podcast`` is invisible to a + prefixed selector -- and, more importantly here, ``Tag.prettify()`` drops + the declaration too, so re-parsing a stored ``PodcastEpisode.rss`` fragment + always yields an unprefixed ````. The existing + ``find("itunes:duration")`` calls get away with the prefixed form only + because they run against the live feed, where the namespace is declared. + + Args: + item: the episode's ```` soup element, or a soup containing it + + Returns: + ``{"url", "type", "language", "rel"}`` dicts in feed order + """ + return [ + { + "url": tag.get("url"), + "type": tag.get("type") or "", + "language": tag.get("language"), + "rel": tag.get("rel"), + } + for tag in item.find_all("transcript") + if tag.prefix in (None, "podcast") and tag.get("url") + ] + + +def transcript_tags_from_rss(rss: str | None) -> list[dict]: + """ + Read the transcript tags out of a stored ``PodcastEpisode.rss`` fragment. + + The ETL already saves each episode's ```` XML verbatim, so the tags + need no column of their own -- re-reading them here also means the + references can never drift from what the feed actually published. + + Args: + rss: the stored prettified ```` XML + + Returns: + ``{"url", "type", "language", "rel"}`` dicts in feed order + """ + if not rss: + return [] + return parse_transcript_tags(bs(rss, "xml")) + + +def _language_is_usable(language: object) -> bool: + """ + Check whether a transcript tag's language is one we can present. + + An absent language means the feed's own ````, which for every + podcast Learn ingests is English. A declared non-English language is + skipped rather than ranked last: showing it under a "Transcript" tab and + indexing it with the english analyzer would both be wrong. + + Args: + language: the tag's ``language`` attribute, if any + + Returns: + True if the transcript should be considered + """ + if language is None or language == "": + return True + return isinstance(language, str) and language.lower().startswith("en") + + +def _media_type_from_url(url: object) -> str | None: + """ + Infer a transcript's media type from its url extension. + + Args: + url: the transcript url + + Returns: + a type from ``TRANSCRIPT_TYPE_PREFERENCE``, or None if the extension + identifies nothing we can parse + """ + if not isinstance(url, str): + return None + try: + path = urlparse(url).path + except ValueError: + return None + return TRANSCRIPT_TYPE_BY_EXTENSION.get(PurePosixPath(path).suffix.lower()) + + +def _resolved_media_type(entry: dict) -> str | None: + """ + Work out which parser a transcript tag needs. + + A declared ``type`` we recognize is authoritative. Otherwise the url + extension is tried, so a tag that omits ``type`` or spells it unusually is + not discarded outright -- the response's own Content-Type is still checked + against the result in ``_served_type_matches``. + + Args: + entry: a ``{"url", "type", "language", "rel"}`` dict + + Returns: + a type from ``TRANSCRIPT_TYPE_PREFERENCE``, or None if none applies + """ + declared = (entry.get("type") or "").split(";")[0].strip().lower() + if declared in TRANSCRIPT_TYPE_PREFERENCE: + return declared + return _media_type_from_url(entry.get("url")) + + +class TranscriptCandidate(NamedTuple): + """A transcript tag worth fetching, with its media type resolved.""" + + url: str + media_type: str + is_captions: bool + + +def rank_transcript_candidates(entries: list[dict]) -> list[TranscriptCandidate]: + """ + Rank one feed item's transcript tags into the order they should be tried. + + Every usable tag is returned, not only the best one. The formats a feed + lists are almost always the same transcript, so a 404, an oversized body or + an HTML error page on the preferred format should fall through to the next + rather than cost the episode a transcript the feed also published as SRT or + JSON. + + Args: + entries: ``{"url", "type", "language", "rel"}`` dicts, in feed order + + Returns: + candidates in fetch order, at most ``MAX_TRANSCRIPT_CANDIDATES`` + """ + ranked = [] + for index, entry in enumerate(entries or []): + url = entry.get("url") + media_type = _resolved_media_type(entry) + if not url or media_type is None: + continue + if not _language_is_usable(entry.get("language")): + continue + # Feed order is the final tiebreak, so the order is deterministic when + # a feed repeats a format. + ranked.append( + ( + TRANSCRIPT_TYPE_PREFERENCE.index(media_type), + index, + TranscriptCandidate( + url=url, + media_type=media_type, + is_captions=entry.get("rel") == "captions", + ), + ) + ) + ranked.sort(key=lambda item: item[:2]) + if len(ranked) > MAX_TRANSCRIPT_CANDIDATES: + log.warning( + "Item lists %d usable transcripts, trying only the best %d", + len(ranked), + MAX_TRANSCRIPT_CANDIDATES, + ) + return [candidate for *_, candidate in ranked[:MAX_TRANSCRIPT_CANDIDATES]] + + +def _is_public_ip(address: str) -> bool: + """ + Check that an address is routable on the public internet. + + Args: + address: a textual IPv4 or IPv6 address + + Returns: + True if the address is outside every special-use range + """ + try: + ip = ipaddress.ip_address(address) + except ValueError: + return False + # Several IPv6 forms tunnel a v4 address that is itself special-use -- + # ::ffff:169.254.169.254, 2002:a9fe:a9fe::1 (6to4) and Teredo all reach the + # cloud metadata endpoint -- and ipaddress only classifies the embedded + # address, so unwrap before testing. + if isinstance(ip, ipaddress.IPv6Address): + embedded = ( + ip.ipv4_mapped or ip.sixtofour or (ip.teredo[1] if ip.teredo else None) + ) + if embedded is not None: + ip = embedded + # `is_global` alone would allow 64:ff9b::/96 (NAT64, which translates to + # arbitrary v4 including link-local); `is_reserved` catches that. Testing + # the individual negatives instead of `is_global` would in turn allow + # 100.64.0.0/10 (carrier NAT). Both checks are needed. + return ip.is_global and not ip.is_reserved + + +def _parsed_https_url(url: object) -> ParseResult | None: + """ + Parse a url, rejecting anything that is not an unambiguous https url. + + A backslash or userinfo in the netloc is refused outright: urllib reads + ``https://evil.io\\@ok.example`` as a url on ok.example while browsers read + it as a url on evil.io, so the hostname we check would not be the host that + is contacted. This is the same guard as ``ovs.parse_media_url``. + + Args: + url: the url to check + + Returns: + the parsed url, or None if its shape is not trustworthy + """ + if not isinstance(url, str) or not url: + return None + try: + parsed = urlparse(url) + except ValueError: + return None + if parsed.scheme != "https" or "\\" in parsed.netloc or "@" in parsed.netloc: + return None + if not parsed.hostname: + return None + return parsed + + +def _resolves_publicly(parsed: ParseResult) -> bool: + """ + Check that a url's host resolves only to publicly routable addresses. + + Every answer must be public, not merely one of them: we cannot control + which record urllib3 connects to, so a host with an A record on the + internet and an AAAA record on ``::1`` is disqualifying. + + Resolution happens before the request, so a name that changes answers + between this check and the connection leaves a narrow rebinding window. It + is accepted here because closing it means pinning the address and losing + SNI, and because the value at risk is a single unauthenticated GET whose + body is only ever parsed as text. + + Args: + parsed: an already shape-checked https url + + Returns: + True if the host may be contacted + """ + try: + resolved = socket.getaddrinfo( + parsed.hostname, parsed.port or 443, proto=socket.IPPROTO_TCP + ) + except (OSError, UnicodeError): + return False + return bool(resolved) and all(_is_public_ip(info[4][0]) for info in resolved) + + +def _parsed_public_url(url: object) -> ParseResult | None: + """ + Parse an https url and confirm its host resolves only to public addresses. + + Args: + url: the url to check + + Returns: + the parsed url, or None if it must not be fetched + """ + parsed = _parsed_https_url(url) + if parsed is None or not _resolves_publicly(parsed): + return None + return parsed + + +def _declared_encoding(response: requests.Response) -> str | None: + """ + Return the charset the server actually declared, if any. + + ``response.encoding`` cannot be used for this: requests follows RFC 2616 + and reports ISO-8859-1 for any ``text/*`` response that omits the charset + parameter, which most transcript hosts do. Decoding a UTF-8 transcript that + way turns every smart quote and non-ASCII name into mojibake. + + Args: + response: the transcript response + + Returns: + the declared charset, or None if the server declared none + """ + message = Message() + message["Content-Type"] = response.headers.get("Content-Type", "") + charset = message.get_param("charset") + # get_param returns a (charset, language, value) tuple for an RFC 2231 + # parameter, which a charset is never legitimately encoded as. + return charset if isinstance(charset, str) else None + + +def _read_capped(response: requests.Response) -> str | None: + """ + Read a response body, abandoning it if it exceeds ``MAX_SOURCE_BYTES``. + + Args: + response: a streamed response + + Returns: + the decoded body, or None if it is too large + """ + chunks = [] + total = 0 + for chunk in response.iter_content(chunk_size=64 * 1024): + total += len(chunk) + if total > MAX_SOURCE_BYTES: + log.warning( + "Transcript at %s exceeds %d bytes, skipping", + response.url, + MAX_SOURCE_BYTES, + ) + return None + chunks.append(chunk) + body = b"".join(chunks) + encoding = _declared_encoding(response) or "utf-8" + try: + text = body.decode(encoding, errors="replace") + except LookupError: + # An unrecognized charset name is not worth discarding the body over. + log.warning( + "Transcript at %s declared unknown charset %s", response.url, encoding + ) + text = body.decode("utf-8", errors="replace") + # A leading BOM would otherwise survive into the stored transcript: + # "WEBVTT" fails the WEBVTT header match, and a BOM before a cue + # number fails the digit check that strips it. + return text.lstrip("") + + +def _checked_response(url: str) -> requests.Response | None: + """ + GET a transcript url, revalidating the target at every redirect hop. + + ``requests`` would follow redirects itself, but then only the first url + would have been checked and a redirect into 169.254.169.254 would be + followed. Each hop is resolved and range-checked instead. + + Args: + url: the transcript url + + Returns: + the final response, or None if any hop is disallowed + """ + current = url + for _ in range(MAX_REDIRECTS + 1): + parsed = _parsed_public_url(current) + if parsed is None: + log.warning("Refusing to fetch transcript from %s", current) + return None + response = requests.get( + current, + headers=BROWSER_UA_HEADERS, + timeout=REQUEST_TIMEOUT, + stream=True, + allow_redirects=False, + ) + if response.is_redirect or response.is_permanent_redirect: + location = response.headers.get("Location") + response.close() + if not location: + return None + current = urljoin(current, location) + continue + try: + response.raise_for_status() + except requests.HTTPError: + # stream=True keeps the connection checked out of the pool until + # the body is consumed or the response closed. + response.close() + raise + return response + log.warning("Too many redirects fetching transcript from %s", url) + return None + + +def _cue_blocks(text: str) -> list[list[str]]: + """ + Split a VTT or SRT body into cue blocks of payload lines. + + Cue identifiers, timing lines, and VTT ``NOTE``/``STYLE``/``WEBVTT`` blocks + are dropped, leaving only the spoken text of each cue. + + Args: + text: the raw caption file body + + Returns: + a list of cues, each a list of payload lines + """ + blocks = [] + for raw_block in re.split(r"\n\s*\n", text.replace("\r\n", "\n")): + lines = [line.strip() for line in raw_block.split("\n") if line.strip()] + if not lines: + continue + if lines[0].upper().startswith(("WEBVTT", "NOTE", "STYLE", "REGION")): + continue + payload = [ + line + for line in lines + if not CUE_TIMING_RE.match(line) and not STAMP_ONLY_RE.match(line) + ] + # A bare numeric first line is the cue identifier, not speech. + if payload and payload[0].isdigit(): + payload = payload[1:] + if payload: + blocks.append(payload) + return blocks + + +def _split_speaker(text: str) -> tuple[str | None, str]: + """ + Pull a leading ``Speaker: `` label off a cue's text. + + Args: + text: one cue's text + + Returns: + the speaker name (or None) and the remaining text + """ + match = INLINE_SPEAKER_RE.match(text) + if not match: + return None, text + return match.group(1).strip(), text[match.end() :] + + +def _cue_speaker_and_text( + payload: list[str], *, decode_entities: bool = True +) -> tuple[str | None, str, str, bool]: + """ + Reduce one cue's payload lines to a speaker and its text. + + Both the label-free and the verbatim text are returned because whether an + inline ``Name: `` prefix is a speaker label is not decidable from one cue -- + it depends on how often labels recur across the whole file, which only + ``_turns_to_text`` can see. Returning just the stripped text would silently + delete the prefix of any cue that merely contains a colon, turning + "The bottom line: we need funding" into "we need funding". + + Args: + payload: the cue's payload lines + decode_entities: whether the payload still carries HTML character + references, as a caption file read off the wire does + + Returns: + the speaker name (or None), the text with any label removed, the text + verbatim, and whether the speaker came from a VTT voice tag rather + than a guessed inline label + """ + joined = " ".join(payload) + voice = VTT_VOICE_RE.search(joined) + speaker = voice.group(1).strip() if voice else None + explicit = speaker is not None + verbatim = VTT_TAG_RE.sub("", joined).strip() + if decode_entities: + # WebVTT requires "&" to be written "&", so cue text arrives with + # character references intact and would otherwise be stored literally. + # Decoding after the tags are stripped keeps a literal "<i>" as + # text rather than turning it into a tag to remove. + verbatim = html.unescape(verbatim) + if speaker: + speaker = html.unescape(speaker) + label_free = verbatim + if speaker is None: + speaker, label_free = _split_speaker(verbatim) + + def normalize(text: str) -> str: + return re.sub(r"\s+", " ", text).strip() + + return speaker, normalize(label_free), normalize(verbatim), explicit + + +def _inferred_labels_are_meaningful( + turns: list[tuple[str | None, str, str, bool]], +) -> bool: + """ + Decide whether a cue file's *guessed* speaker labels describe real turns. + + A label read from a VTT voice tag or a podcastindex ``speaker`` field is + explicit -- the format says outright who is speaking -- and is trusted + unconditionally by ``_turns_to_text`` regardless of what this returns. A + label guessed from an inline ``Name: `` prefix is not: it is + indistinguishable from a cue that merely contains a colon, so it is only + trusted once it recurs often enough to look like real turn-taking. A + single stray inferred label (Captivate's SRT has one across 861 cues) + would otherwise attribute the whole episode to one person. + + Args: + turns: ``(speaker, label_free_text, verbatim_text, explicit)`` tuples + in order + + Returns: + True if inferred speaker labels should be applied + """ + inferred = sum(1 for speaker, _, _, explicit in turns if speaker and not explicit) + return ( + inferred >= MIN_SPEAKER_LABELS + and inferred * MAX_CUES_PER_SPEAKER_LABEL >= len(turns) + ) + + +def _turns_to_text(turns: list[tuple[str | None, str, str, bool]]) -> str: + """ + Render cues or segments as prose paragraphs. + + A paragraph ends when the speaker changes, or -- so that a long monologue + or an unlabeled caption file does not become one unbroken block -- at the + first sentence end past ``PARAGRAPH_TARGET_CHARS``. Only the first + paragraph of a turn carries the speaker label; continuations are unlabeled, + which is how transcripts are conventionally set. + + An explicit label (a VTT voice tag, a podcastindex ``speaker`` field) is + always applied. An inferred label is applied only once + ``_inferred_labels_are_meaningful`` finds it recurs; otherwise the verbatim + text is used so that a cue which merely contains a colon keeps its opening + words. + + Consecutive turns are deduplicated on speaker *and* text together, not + text alone: rolling captions repeat a cue verbatim as the window scrolls, + but two different speakers who happen to both say "Yes." are not a repeat. + + Args: + turns: ``(speaker, label_free_text, verbatim_text, explicit)`` tuples + in order + + Returns: + paragraphs separated by a blank line + """ + trust_inferred = _inferred_labels_are_meaningful(turns) + paragraphs: list[str] = [] + current_speaker: str | None = None + label_pending = False + previous_text: str | None = None + previous_speaker: str | None = None + buffer: list[str] = [] + + def flush() -> None: + nonlocal label_pending + if not buffer: + return + body = " ".join(buffer).strip() + if body: + prefix = f"{current_speaker}: " if label_pending and current_speaker else "" + paragraphs.append(f"{prefix}{body}") + label_pending = False + buffer.clear() + + for speaker, label_free, verbatim, explicit in turns: + use_label = explicit or trust_inferred + text = label_free if use_label else verbatim + # A continuation cue (speaker is None) inherits whoever is currently + # speaking, so the dedup check below compares against that speaker + # rather than None. + effective_speaker = ( + (speaker if speaker is not None else current_speaker) if use_label else None + ) + if not text or ( + text == previous_text and effective_speaker == previous_speaker + ): + continue + previous_text = text + previous_speaker = effective_speaker + if use_label and speaker is not None and speaker != current_speaker: + flush() + current_speaker = speaker + label_pending = True + buffer.append(text) + if sum(len(part) for part in buffer) >= PARAGRAPH_TARGET_CHARS and ( + SENTENCE_END_RE.search(text) + ): + flush() + flush() + return "\n\n".join(paragraphs) + + +def parse_cue_format(text: str, *, decode_entities: bool = True) -> str: + """ + Normalize a VTT or SRT body to prose. + + Both formats share the same block structure and differ only in their + timestamp separator, which is discarded either way. + + Args: + text: the raw caption file body + decode_entities: whether cue text still carries HTML character + references. Pass False for text an HTML parser has already decoded. + + Returns: + normalized transcript text + """ + return _turns_to_text( + [ + _cue_speaker_and_text(payload, decode_entities=decode_entities) + for payload in _cue_blocks(text) + ] + ) + + +def parse_podcast_index_json(text: str) -> str: + """ + Normalize a podcastindex JSON transcript to prose. + + ``body`` is fragmented mid-sentence by at least one host, so segments are + merged by speaker rather than emitted one per paragraph. + + Args: + text: the raw JSON body + + Returns: + normalized transcript text, or "" if the payload is not the expected shape + """ + try: + payload = json.loads(text) + except ValueError: + log.warning("Transcript JSON could not be parsed") + return "" + segments = payload.get("segments") if isinstance(payload, dict) else None + if not isinstance(segments, list): + return "" + turns = [] + for segment in segments: + if not isinstance(segment, dict): + continue + body = str(segment.get("body") or "").strip() + speaker = segment.get("speaker") + speaker_name = str(speaker).strip() if speaker else None + # The speaker is a real field here, so there is no inline label to + # strip: label-free and verbatim text are the same. A speaker read + # from this field is explicit, the same as a VTT voice tag. + turns.append((speaker_name, body, body, speaker_name is not None)) + return _turns_to_text(turns) + + +def _preceding_cite_and_time(p) -> tuple[object, object]: + """ + Find a ``

``'s immediately preceding ```` and/or ``

``'s own ````/``

`` tag + + Returns: + the preceding ```` tag and ``

``'s ```` label and ``

`` they annotate; Captivate instead emits them as the + ``

``'s preceding siblings. Left alone, either shape falls apart under + the block-tag pass below: ````, ``

`` each force + their own paragraph break, so "Kevin: Hello." comes out as three + fragments -- "Kevin", the raw timestamp, and "Hello." -- with the + timestamp surviving into the stored transcript. This folds both shapes + into one ``

`` so the later pass sees a single paragraph with the label + restored and the timestamp gone. + + Args: + soup: the parsed HTML transcript, mutated in place + """ + for p in soup.find_all("p"): + cite = p.find("cite") + time_tag = p.find("time") + if cite is None or time_tag is None: + preceding_cite, preceding_time = _preceding_cite_and_time(p) + cite = cite or preceding_cite + time_tag = time_tag or preceding_time + if cite is None and time_tag is None: + continue + if time_tag is not None: + time_tag.decompose() + if cite is not None: + speaker = cite.get_text().strip().rstrip(":").strip() + cite.decompose() + if speaker: + p.insert(0, f"{speaker}: ") + + +def parse_html(text: str) -> str: + """ + Normalize an HTML transcript to prose. + + Some hosts serve a real HTML document; Captivate serves the first cue as + markup and the rest of the SRT file inside ``

`` tags. Both end up as + text here, and a body that still contains cue timings is handed to the cue + parser so the timecodes do not survive into the stored transcript. + + Args: + text: the raw HTML body + + Returns: + normalized transcript text + """ + soup = bs(text, "html.parser") + for tag in soup(["script", "style", "nav", "header", "footer"]): + tag.decompose() + _fold_cite_and_time(soup) + # Mark block boundaries before extracting rather than passing a separator + # to get_text: a separator would also land between inline elements and + # break sentences apart, while no separator at all would run consecutive + # paragraphs together. + for tag in soup.find_all(BLOCK_LEVEL_TAGS): + tag.append("\n\n") + extracted = soup.get_text() + if "-->" in extracted: + # get_text() has already resolved character references; decoding them a + # second time would collapse a doubly-escaped "&amp;" to a bare "&". + return parse_cue_format(extracted, decode_entities=False) + return parse_plain(extracted) + + +def parse_plain(text: str) -> str: + """ + Normalize a plain-text transcript, preserving its paragraph breaks. + + Args: + text: the raw text body + + Returns: + normalized transcript text + """ + paragraphs = [ + re.sub(r"\s+", " ", block).strip() + for block in re.split(r"\n\s*\n", text.replace("\r\n", "\n")) + ] + return "\n\n".join(block for block in paragraphs if block) + + +PARSERS = { + "text/plain": parse_plain, + "text/vtt": parse_cue_format, + "application/x-subrip": parse_cue_format, + "application/srt": parse_cue_format, + "application/json": parse_podcast_index_json, + "text/html": parse_html, +} + + +def _served_type_matches(served_type: str, declared_type: str) -> bool: + """ + Check that a response's Content-Type is consistent with the declared type. + + Hosts are loose about this in both directions — Captivate serves its + ``application/srt`` file as ``text/plain`` and doctorpodcasting serves its + ``text/vtt`` file as ``text/plain`` — so ``text/plain`` is accepted for any + declared type. The point of the check is only to catch a host answering a + caption request with an HTML error page, which would otherwise be stored as + prose. + + Args: + served_type: the response's Content-Type, without parameters + declared_type: the type the feed's transcript tag claimed + + Returns: + True if the body should be parsed + """ + if not served_type or served_type == declared_type: + return True + if served_type == "text/plain": + return True + # The two spellings of SRT are the same format. + srt_types = {"application/srt", "application/x-subrip"} + return served_type in srt_types and declared_type in srt_types + + +def _fetch_candidate(candidate: TranscriptCandidate) -> str: + """ + Fetch and normalize one transcript candidate. + + Args: + candidate: the transcript to fetch + + Returns: + the normalized transcript text, or "" if this candidate yielded nothing + """ + url = candidate.url + # One `except` around the request *and* the streamed read: a + # ChunkedEncodingError or ReadTimeout part-way through the body is a + # RequestException raised by iter_content, not by get(), and letting it + # escape would abort the caller's loop and skip every remaining episode. + try: + response = _checked_response(url) + if response is None: + return "" + try: + served_type = ( + response.headers.get("Content-Type", "").split(";")[0].strip().lower() + ) + if not _served_type_matches(served_type, candidate.media_type): + log.warning( + "Transcript at %s declared %s but served %s, skipping", + url, + candidate.media_type, + served_type, + ) + return "" + body = _read_capped(response) + finally: + response.close() + except requests.RequestException as exc: + # Not fatal and not exceptional: fetch_transcript falls through to the + # next format, so this is logged as a warning rather than a traceback. + log.warning("Failed to fetch transcript from %s: %s", url, exc) + return "" + if not body: + return "" + parser = PARSERS[candidate.media_type] + if candidate.is_captions and candidate.media_type == "text/plain": + # rel="captions" says the file carries cue numbers/timestamps even + # though it is typed text/plain, so parse_plain -- which does not + # know how to strip them -- would leak raw timecodes into the + # stored transcript. + parser = parse_cue_format + return parser(body)[:MAX_TRANSCRIPT_CHARS] + + +def fetch_transcript(entries: list[dict]) -> str: + """ + Fetch and normalize the best usable transcript among a feed item's tags. + + Candidates are tried in preference order until one yields text: a broken + url in the preferred format must not cost the episode a transcript the feed + also published in another one. + + Args: + entries: ``{"url", "type", "language", "rel"}`` dicts, in feed order + + Returns: + the normalized transcript text, or "" if none could be fetched + """ + for candidate in rank_transcript_candidates(entries): + transcript = _fetch_candidate(candidate) + if transcript: + return transcript + return "" diff --git a/learning_resources/etl/podcast_transcript_test.py b/learning_resources/etl/podcast_transcript_test.py new file mode 100644 index 0000000000..1f2c00b433 --- /dev/null +++ b/learning_resources/etl/podcast_transcript_test.py @@ -0,0 +1,963 @@ +"""Tests for podcast transcript selection, fetching and parsing""" + +import pytest +from bs4 import BeautifulSoup as bs # noqa: N813 + +from learning_resources.etl.podcast_transcript import ( + MAX_TRANSCRIPT_CANDIDATES, + TRANSCRIPT_TYPE_PREFERENCE, + _is_public_ip, + _parsed_public_url, + fetch_transcript, + parse_cue_format, + parse_html, + parse_plain, + parse_podcast_index_json, + parse_transcript_tags, + rank_transcript_candidates, + transcript_tags_from_rss, +) + +ITEM_TEMPLATE = """ + + + + An episode + {tags} + + + +""" + +PODCAST_NS = 'xmlns:podcast="https://podcastindex.org/namespace/1.0"' + + +def _item(tags, *, namespace=PODCAST_NS): + """Build an soup element carrying the given transcript tags""" + return bs(ITEM_TEMPLATE.format(namespace=namespace, tags=tags), "xml").find("item") + + +@pytest.mark.parametrize("namespace", [PODCAST_NS, ""]) +def test_parse_transcript_tags_tolerates_undeclared_namespace(namespace): + """ + Tags are found whether or not the feed declares xmlns:podcast. + + lxml in recover mode drops an undeclared prefix, so a + find_all("podcast:transcript") selector would silently return nothing for + the feeds that get the declaration wrong. + """ + item = _item( + '', + namespace=namespace, + ) + assert parse_transcript_tags(item) == [ + { + "url": "https://x/t.vtt", + "type": "text/vtt", + "language": "en", + "rel": "captions", + } + ] + + +def test_parse_transcript_tags_empty_when_absent(): + """An item with no transcript tag yields an empty list""" + assert parse_transcript_tags(_item("")) == [] + + +def test_transcript_tags_survive_the_rss_round_trip(): + """ + Tags are recoverable from a stored PodcastEpisode.rss fragment. + + This is what lets the feature work without a new column: the ETL already + saves `item.prettify()`. That drops the xmlns:podcast declaration, so the + re-parsed tags come back unprefixed -- which is precisely why + parse_transcript_tags matches on local name instead of "podcast:transcript". + """ + item = _item( + '' + '' + ) + stored = item.prettify() + + # The declaration is gone, but the tag text -- and so the queryset filter + # get_podcast_episodes_for_transcripts_job uses -- survives. + assert "xmlns:podcast" not in stored + assert "', + namespace="", + ).prettify() + + assert "No tags"]) +def test_transcript_tags_from_rss_empty(rss): + """A missing or tagless rss fragment yields no entries""" + assert transcript_tags_from_rss(rss) == [] + + +def test_parse_transcript_tags_skips_tags_with_no_url(): + """A tag missing the required url attribute is unusable""" + assert parse_transcript_tags(_item('')) == [] + + +def test_rank_transcript_candidates_prefers_better_formats(): + """ + Formats are ranked by the quality of the files feeds actually serve. + + Captivate publishes a malformed text/html alongside a clean SRT, so html + must lose to every other format. + """ + entries = [ + {"url": "h", "type": "text/html", "language": None}, + {"url": "j", "type": "application/json", "language": None}, + {"url": "s", "type": "application/srt", "language": None}, + {"url": "v", "type": "text/vtt", "language": None}, + {"url": "p", "type": "text/plain", "language": None}, + ] + # html ranks last and so falls outside MAX_TRANSCRIPT_CANDIDATES here; on + # its own it is still a usable candidate. + assert [c.url for c in rank_transcript_candidates(entries)] == ["p", "v", "s", "j"] + assert [c.url for c in rank_transcript_candidates(entries[:1])] == ["h"] + + +def test_rank_transcript_candidates_accepts_both_srt_spellings(): + """application/srt is non-standard but three MIT feeds send it""" + for mime_type in ("application/srt", "application/x-subrip"): + entry = {"url": "s", "type": mime_type, "language": None} + assert [c.media_type for c in rank_transcript_candidates([entry])] == [ + mime_type + ] + + +@pytest.mark.parametrize( + ("language", "expected"), + [ + (None, True), + ("", True), + ("en", True), + ("en-US", True), + ("EN-GB", True), + ("es", False), + ("de-DE", False), + ], +) +def test_rank_transcript_candidates_language_filter(language, expected): + """ + Non-English transcripts are dropped, not merely deranked. + + The stored text is analyzed with the english analyzer and shown under a + Transcript tab, so a Spanish translation of an English show is worse than + nothing. + """ + entry = {"url": "v", "type": "text/vtt", "language": language} + assert bool(rank_transcript_candidates([entry])) is expected + + +def test_rank_transcript_candidates_ignores_language_in_ordering(): + """ + Language decides inclusion, never rank. + + An explicit ``en`` tag is no better than an absent one -- an absent + language means the feed's own, which is English for every podcast Learn + ingests -- so format preference still decides. + """ + unlabeled = {"url": "a", "type": "text/plain", "language": None} + english = {"url": "b", "type": "text/vtt", "language": "en"} + assert [c.url for c in rank_transcript_candidates([english, unlabeled])] == [ + "a", + "b", + ] + + +def test_rank_transcript_candidates_handles_empty(): + """No tags means no candidates, and a missing list is not an error""" + assert rank_transcript_candidates([]) == [] + assert rank_transcript_candidates(None) == [] + + +def test_rank_transcript_candidates_keeps_every_usable_tag(): + """ + Ranking returns the losers too, so fetch_transcript can fall through. + + A feed lists the same transcript in several formats, so a broken url in the + preferred one must not cost the episode the transcript entirely. + """ + entries = [ + {"url": "h", "type": "text/html", "language": None}, + {"url": "p", "type": "text/plain", "language": None}, + {"url": "v", "type": "text/vtt", "language": None}, + ] + assert [c.url for c in rank_transcript_candidates(entries)] == ["p", "v", "h"] + + +def test_rank_transcript_candidates_drops_unusable_tags(): + """Unparseable types and non-English languages never become candidates""" + entries = [ + {"url": "x", "type": "application/pdf", "language": None}, + {"url": "es", "type": "text/vtt", "language": "es"}, + {"url": "v", "type": "text/vtt", "language": "en"}, + ] + assert [c.url for c in rank_transcript_candidates(entries)] == ["v"] + + +def test_rank_transcript_candidates_caps_the_list(caplog): + """One item cannot provoke unbounded requests, and the cap is not silent""" + entries = [ + {"url": f"u{index}", "type": mime_type, "language": None} + for index, mime_type in enumerate(TRANSCRIPT_TYPE_PREFERENCE) + ] + candidates = rank_transcript_candidates(entries) + assert [c.media_type for c in candidates] == list( + TRANSCRIPT_TYPE_PREFERENCE[:MAX_TRANSCRIPT_CANDIDATES] + ) + assert "trying only the best" in caplog.text + + +@pytest.mark.parametrize( + ("url", "declared_type", "expected"), + [ + # The spec makes `type` required; some feeds omit it anyway. + ("https://h.example/t.srt", None, "application/x-subrip"), + ("https://h.example/t.srt", "", "application/x-subrip"), + # A spelling outside TRANSCRIPT_TYPE_PREFERENCE. + ("https://h.example/t.srt", "text/srt", "application/x-subrip"), + ("https://h.example/t.vtt", None, "text/vtt"), + ("https://h.example/t.json", None, "application/json"), + ("https://h.example/t.txt", None, "text/plain"), + ("https://h.example/t.htm", None, "text/html"), + # A query string must not defeat the extension read. + ("https://h.example/t.vtt?token=abc", None, "text/vtt"), + # Nothing to go on: no usable type, no extension we can parse. + ("https://h.example/transcript", None, None), + ("https://h.example/t.pdf", None, None), + ], +) +def test_media_type_falls_back_to_the_url_extension(url, declared_type, expected): + """ + A tag with no usable `type` is identified by its url extension instead. + + Dropping it would make the episode look transcript-less. The response's own + Content-Type is still checked before the body is parsed. + """ + candidates = rank_transcript_candidates( + [{"url": url, "type": declared_type, "language": None}] + ) + assert (candidates[0].media_type if candidates else None) == expected + + +def test_declared_type_wins_over_the_url_extension(): + """A recognised `type` is authoritative even when the extension disagrees""" + entries = [{"url": "https://h.example/t.html", "type": "text/vtt"}] + assert rank_transcript_candidates(entries)[0].media_type == "text/vtt" + + +def test_rank_transcript_candidates_flags_captions_rel(): + """The rel="captions" hint survives onto the candidate""" + entries = [ + {"url": "p", "type": "text/plain", "language": None, "rel": "captions"}, + {"url": "v", "type": "text/vtt", "language": None, "rel": None}, + ] + candidates = {c.url: c for c in rank_transcript_candidates(entries)} + assert candidates["p"].is_captions is True + assert candidates["v"].is_captions is False + + +VTT = """WEBVTT + +NOTE +This file was generated by a tool + +00:00:02.939 --> 00:00:05.400 +Welcome to the show, + +00:00:05.400 --> 00:00:10.199 +a show about big questions. + +00:00:10.199 --> 00:00:11.519 +What do we value and why? +""" + +SRT = """1 +00:00:00,567 --> 00:00:03,570 +Sebastian Lourido: And we come back and I have +kind of swollen lymph nodes. + +2 +00:00:03,603 --> 00:00:06,606 +I don't feel terrible, but + +3 +00:00:07,607 --> 00:00:09,542 +it's notable. +""" + + +def test_parse_cue_format_vtt_uses_voice_tags(): + """VTT voice tags become speaker labels and start new paragraphs""" + assert parse_cue_format(VTT) == ( + "Susan Silbey: Welcome to the show, a show about big questions." + "\n\nEmily Pollock: What do we value and why?" + ) + + +def test_parse_cue_format_drops_headers_and_timings(): + """WEBVTT, NOTE blocks, cue numbers and timing lines never survive""" + output = parse_cue_format(VTT) + for artifact in ("WEBVTT", "NOTE", "-->", "00:00:02", " 00:00:02,000 +Alice: Good morning. + +2 +00:00:02,000 --> 00:00:04,000 +Bob: Morning, Alice. +""" + assert parse_cue_format(srt) == "Alice: Good morning.\n\nBob: Morning, Alice." + + +def test_parse_cue_format_keeps_colons_that_are_not_speakers(): + """ + A cue containing a colon mid-sentence keeps its opening words. + + The inline-label pattern matches "The bottom line: " the same way it + matches "Alice: ", so the prefix is only removed once the file as a whole + shows recurring labels. Otherwise the text is used verbatim -- dropping it + silently deleted content. + """ + srt = """1 +00:00:00,000 --> 00:00:03,000 +The bottom line: we need more funding. + +2 +00:00:03,000 --> 00:00:06,000 +And that is the whole story here. +""" + output = parse_cue_format(srt) + assert output.startswith("The bottom line: we need more funding.") + + +def test_parse_cue_format_drops_repeated_cues(): + """Rolling captions repeat a cue verbatim as the window scrolls""" + srt = """1 +00:00:00,000 --> 00:00:02,000 +the same line + +2 +00:00:02,000 --> 00:00:04,000 +the same line +""" + assert parse_cue_format(srt) == "the same line" + + +def test_parse_cue_format_keeps_identical_text_from_different_speakers(): + """ + Deduping on text alone would drop a different speaker's identical line. + + Two speakers each saying "Yes." back to back is real dialogue, not a + rolling caption repeating itself. + """ + vtt = ( + "WEBVTT\n\n" + "00:00:00.000 --> 00:00:01.000\nYes.\n\n" + "00:00:01.000 --> 00:00:02.000\nYes.\n\n" + "00:00:02.000 --> 00:00:03.000\nReally?\n" + ) + assert parse_cue_format(vtt) == "Alice: Yes.\n\nBob: Yes.\n\nAlice: Really?" + + +def test_parse_cue_format_trusts_a_single_explicit_voice_tag(): + """ + An explicit VTT voice tag is trusted even in a single-cue file. + + MIN_SPEAKER_LABELS exists to keep a single *guessed* inline label from + attributing a whole episode to one person -- it should not also strip a + label the format states outright. + """ + vtt = ( + "WEBVTT\n\n" + "00:00:00.000 --> 00:00:01.000\n" + "Welcome to a very short bonus episode.\n" + ) + assert parse_cue_format(vtt) == "Host: Welcome to a very short bonus episode." + + +def test_parse_cue_format_empty(): + """An empty body yields an empty transcript""" + assert parse_cue_format("") == "" + + +def test_parse_cue_format_decodes_character_references(): + """ + WebVTT requires "&" to be written "&", so cue text arrives escaped. + + Left encoded, the stored transcript shows the literal entity and React + escapes the ampersand a second time on render. Decoding happens after the + tags are stripped so an escaped "<i>" stays text. + """ + vtt = ( + "WEBVTT\n\n" + "00:00:01.000 --> 00:00:03.000\n" + "R&D at <MIT> matters.\n\n" + "00:00:03.000 --> 00:00:05.000\n" + "Fifty percent.\n" + ) + assert parse_cue_format(vtt) == ( + "Ada & Grace: R&D at matters.\n\nBob: Fifty percent." + ) + + +def test_parse_podcast_index_json_merges_fragmented_segments(): + """ + The podcastindex format fragments body mid-sentence. + + Adjacent segments by the same speaker are merged back into one paragraph. + """ + payload = ( + '{"version":"1.0.0","segments":[' + '{"speaker":"Susan","startTime":1,"body":"Welcome"},' + '{"speaker":"Susan","startTime":2,"body":"to the show."},' + '{"speaker":"Emily","startTime":3,"body":"What do we value?"}]}' + ) + assert parse_podcast_index_json(payload) == ( + "Susan: Welcome to the show.\n\nEmily: What do we value?" + ) + + +@pytest.mark.parametrize( + "payload", ["not json at all", "[]", '{"version":"1.0.0"}', '{"segments":"nope"}'] +) +def test_parse_podcast_index_json_bad_payloads(payload): + """Malformed or unexpected JSON yields no transcript rather than raising""" + assert parse_podcast_index_json(payload) == "" + + +def test_parse_podcast_index_json_trusts_a_single_explicit_speaker(): + """A single segment's `speaker` field is a real field, not a guess""" + payload = ( + '{"version":"1.0.0","segments":' + '[{"speaker":"Host","body":"Welcome to a short one."}]}' + ) + assert parse_podcast_index_json(payload) == "Host: Welcome to a short one." + + +def test_parse_html_recovers_captivate_srt_dump(): + """ + Captivate marks up only the first cue and dumps raw SRT after it. + + The extracted text still contains cue timings, so it is routed through the + cue parser and the timecodes do not reach the stored transcript. + """ + html = ( + "Sebastian Lourido:" + "

And we come back.

" + "

2

00:00:03,603 --> 00:00:06,606

" + "

I don't feel terrible, but

" + "

3

00:00:07,607 --> 00:00:09,542

" + "

it's notable.

" + ) + output = parse_html(html) + for artifact in ("-->", "00:00:00", "00:00:03", "

", ""): + assert artifact not in output + assert "And we come back." in output + assert "it's notable." in output + + +def test_parse_html_groups_nested_cite_time_labels(): + """ + The namespace's own HTML sample nests /

. + + Left ungrouped, ,

would each force their own + paragraph break, splitting one labeled line into three fragments. + """ + html = ( + "

HostWelcome to the show.

" + "

GuestThanks for having me.

" + ) + assert parse_html(html) == ( + "Host: Welcome to the show.\n\nGuest: Thanks for having me." + ) + + +def test_parse_html_groups_sibling_cite_time_labels(): + """ + Captivate emits /