diff --git a/src/lib/stageRemoteFile.js b/src/lib/stageRemoteFile.js new file mode 100644 index 00000000..2886ce3f --- /dev/null +++ b/src/lib/stageRemoteFile.js @@ -0,0 +1,53 @@ +'use strict'; + +const fs = require('fs-extra'); +const path = require('path'); +const { HAXCMS } = require('./HAXCMS.js'); +// reached through the module object so the network boundary can be stubbed in tests +const safeFetchLib = require('./safeFetch.js'); + +// Bulk-import staging shared by the site importers and createSite (#3060). +// HAXCMSFile.isValidBulkImportStagedPath only accepts real files under +// /tmp/imports, so a remote file is downloaded here first and +// the staged path then takes the same bulk-import save as any other file. + +// Directory under the HAXCMS config tree where bulk-import files are staged, +// created if it does not exist yet. +function getBulkImportStagingRoot() { + const root = path.join(HAXCMS.configDirectory, 'tmp', 'imports'); + try { + fs.ensureDirSync(root); + } catch (e) {} + return root; +} + +// Fetch a remote file via safeFetch (SSRF-guarded: every resolved address and +// redirect hop is validated and the connection is pinned to the checked +// address) and stage it under the bulk-import root so createSite can move it +// into the site tree. Returns the absolute staged path, or null on any +// fetch/write failure or empty body (the file is simply skipped, matching how +// page fetch failures are handled). idx keeps staged filenames unique across +// the import. +async function stageRemoteFile(url, stagingRoot, idx, relPath) { + try { + const response = await safeFetchLib.safeFetch(url); + if (!response || !response.ok) { + return null; + } + const buf = Buffer.from(await response.arrayBuffer()); + if (!buf || buf.length === 0) { + return null; + } + const ext = path.extname(relPath || ''); + const stagedPath = path.join( + stagingRoot, + 'haximp-' + Date.now() + '-' + idx + '-' + Math.floor(Math.random() * 1000000) + ext + ); + fs.writeFileSync(stagedPath, buf); + return stagedPath; + } catch (e) { + return null; + } +} + +module.exports = { getBulkImportStagingRoot, stageRemoteFile }; diff --git a/src/openapi/system-spec.yaml b/src/openapi/system-spec.yaml index e44740f4..5b8646dd 100644 --- a/src/openapi/system-spec.yaml +++ b/src/openapi/system-spec.yaml @@ -2574,6 +2574,53 @@ paths: application/json: schema: $ref: "#/components/schemas/ErrorEnvelope" + /system/api/v1/site/import/openstax: + post: + tags: + - actions + operationId: openstaxToSite + summary: Import an OpenStax textbook as a new site + security: + - bearerAuth: [] + userTokenHeader: [] + requestBody: + required: true + content: + application/json: + schema: + type: object + properties: + repoUrl: + type: string + description: >- + OpenStax book URL, either + https://openstax.org/details/books/ or + https://openstax.org/books//pages/ + required: + - repoUrl + responses: + "200": + description: Imported site schema items + content: + application/json: + schema: + type: object + properties: + status: + type: integer + data: + $ref: "#/components/schemas/ImportData" + additionalProperties: true + "400": + $ref: "#/components/responses/BadRequest" + "403": + $ref: "#/components/responses/Forbidden" + "422": + description: Unprocessable entity + content: + application/json: + schema: + $ref: "#/components/schemas/ErrorEnvelope" /system/api/v1/site/import/{platform}: post: tags: @@ -2913,7 +2960,7 @@ components: files: type: object additionalProperties: true - description: Map of file download locations for bulk-imported assets. + description: Map of site file paths (files/...) to their source, either an http(s) URL fetched server-side through the SSRF guard or a file staged under /tmp/imports. A URL that cannot be fetched is skipped. skeletonMachineName: type: string description: Machine name of a trusted skeleton to build from (structure=from-skeleton). diff --git a/src/systemRoutes/v1/routes/createSite.js b/src/systemRoutes/v1/routes/createSite.js index 3e5b0df6..5b390c6c 100644 --- a/src/systemRoutes/v1/routes/createSite.js +++ b/src/systemRoutes/v1/routes/createSite.js @@ -5,6 +5,10 @@ const HAXCMSFile = require('../../../lib/HAXCMSFile.js'); const fs = require('fs-extra'); const path = require('path'); const { safeFetch } = require('../../../lib/safeFetch.js'); +const { getBulkImportStagingRoot, stageRemoteFile } = require('../../../lib/stageRemoteFile.js'); +const EntityRegistry = require('../../../lib/EntityRegistry.js'); +const FileStorage = require('../../../lib/FileStorage.js'); +const FileContentScanner = require('../../../lib/FileContentScanner.js'); const SAFE_BULK_IMPORT_EXTENSION_REGEX = /\.(jpg|jpeg|png|gif|webm|webp|mp4|mp3|mov|csv|ppt|pptx|xlsx|doc|xls|docx|pdf|rtf|txt|vtt|html|md|xml)$/i; const DEFAULT_CREATE_SITE_THEME_ICON = 'icons:record-voice-over'; @@ -870,14 +874,9 @@ async function createSite(req, res) { await site.manifest.save(false); // walk through files if any came across and save each of them if (filesToDownload && typeof filesToDownload === 'object') { + let fileIndex = 0; for (var locationName in filesToDownload) { - let downloadLocation = filesToDownload[locationName]; - const normalizedImportName = normalizeBulkImportName(locationName); - if ( - !normalizedImportName || - !SAFE_BULK_IMPORT_EXTENSION_REGEX.test(normalizedImportName) || - !HAXCMSFile.isValidBulkImportStagedPath(downloadLocation) - ) { + if (!(await importBuildFile(site, locationName, filesToDownload[locationName], fileIndex++))) { return res.status(400).json({ status: 400, data: { @@ -885,14 +884,9 @@ async function createSite(req, res) { } }); } - let file = new HAXCMSFile(); - // check for a file upload; we block a few formats by design - await file.save({ - "name": normalizedImportName, - "tmp_name": downloadLocation, - "path": downloadLocation, - "bulk-import": true - }, site); + } + if (Object.keys(filesToDownload).length > 0) { + await linkImportedPageFiles(site); } } // download user-customized theme and custom files (imported from another instance) @@ -957,4 +951,86 @@ async function createSite(req, res) { res.status(403).json({ status: 403, data: { message: 'Authentication required' } }); } } -module.exports = createSite; \ No newline at end of file +/** + * #3060: bring one build.files entry into the site. Importers hand over + * remote files as http(s) URLs, so a URL is fetched through safeFetch (the + * SSRF baseline siteFiles uses) into the bulk-import staging root. From there + * every entry takes the same path: the staged-path check, then a bulk-import + * HAXCMSFile.save that validates the content and records the file entity in + * files.json. A URL that cannot be fetched is skipped rather than failing the + * site. Returns false for an invalid entry, which createSite answers with 400. + */ +async function importBuildFile(site, locationName, downloadLocation, index) { + const normalizedImportName = normalizeBulkImportName(locationName); + if (!normalizedImportName || !SAFE_BULK_IMPORT_EXTENSION_REGEX.test(normalizedImportName)) { + return false; + } + let downloaded = null; + if (typeof downloadLocation === 'string' && /^https?:\/\//i.test(downloadLocation)) { + downloaded = await stageRemoteFile(downloadLocation, getBulkImportStagingRoot(), index, normalizedImportName); + if (!downloaded) { + return true; + } + downloadLocation = downloaded; + } + const valid = HAXCMSFile.isValidBulkImportStagedPath(downloadLocation); + if (valid) { + const upload = { + "name": normalizedImportName, + "tmp_name": downloadLocation, + "path": downloadLocation, + "bulk-import": true + }; + if (downloaded) { + // a download carries its size so the site's maxUploadSizeMb applies (HAX-SEC-004) + upload.size = fs.statSync(downloaded).size; + } + // check for a file upload; we block a few formats by design + await new HAXCMSFile().save(upload, site); + } + // save moves an accepted file into the site; a download it rejected is dropped + if (downloaded) { + fs.removeSync(downloaded); + } + return valid; +} + +/** + * #3043: point each page at the file entities its content references, once + * the imported files exist. createSite writes the pages before it ingests + * build.files, so page.metadata.files cannot be set as each page is written. + * Identity comes from files.json through the Entity API, as it does for the + * docx import and for page saves; the FileStorage is created after the ingest + * so it reads the records the ingest just wrote. Returns the pages linked. + */ +async function linkImportedPageFiles(site) { + const fileStorage = FileStorage.registerOn(new EntityRegistry(site)); + let linked = 0; + for (const page of site.manifest.items) { + const content = await site.getPageContent(page); + const uuids = []; + for (const reference of FileContentScanner.extractFileReferences(content)) { + const uuid = await fileStorage.getDataStore().resolveUuidByPath(reference); + const entity = uuid ? fileStorage.load(uuid) : null; + if (entity && uuids.indexOf(entity.getUuid()) === -1) { + uuids.push(entity.getUuid()); + } + } + if (uuids.length > 0) { + if (!page.metadata || typeof page.metadata !== 'object') { + page.metadata = {}; + } + page.metadata.files = uuids; + linked++; + } + } + if (linked > 0) { + await site.manifest.save(false); + } + return linked; +} + +module.exports = createSite; +// exported for direct unit testing +module.exports.importBuildFile = importBuildFile; +module.exports.linkImportedPageFiles = linkImportedPageFiles; \ No newline at end of file diff --git a/src/systemRoutes/v1/routes/imports/convertHaxcmsToSite.js b/src/systemRoutes/v1/routes/imports/convertHaxcmsToSite.js index b6ef0cd7..3545eddb 100644 --- a/src/systemRoutes/v1/routes/imports/convertHaxcmsToSite.js +++ b/src/systemRoutes/v1/routes/imports/convertHaxcmsToSite.js @@ -5,49 +5,8 @@ const SITE_FILES_TO_IMPORT = [ ] const BOILERPLATE_CUSTOM_ES6 = '// custom comment script here' const { safeFetch } = require('../../../../lib/safeFetch.js') -const { HAXCMS } = require('../../../../lib/HAXCMS.js') -const fs = require('fs-extra') -const path = require('path') - -// Directory under the HAXCMS config tree where bulk-import files are staged -// before createSite's isValidBulkImportStagedPath validator accepts them. -// createSite's build.files contract is "local staged paths only" (it rejects -// URL schemes by design), so the converter downloads each referenced file -// here and hands createSite the staged path instead of the remote URL. -function getBulkImportStagingRoot() { - const root = path.join(HAXCMS.configDirectory, 'tmp', 'imports') - try { - fs.ensureDirSync(root) - } catch (e) {} - return root -} - -// Fetch a remote file via safeFetch (SSRF-guarded, no redirects) and stage it -// under the bulk-import root so createSite can move it into the site tree. -// Returns the absolute staged path, or null on any fetch/write failure or -// empty body (the file is simply skipped, matching how page fetch failures -// are handled). idx keeps staged filenames unique across the import. -async function stageRemoteFile(url, stagingRoot, idx, relPath) { - try { - const response = await safeFetch(url) - if (!response || !response.ok) { - return null - } - const buf = Buffer.from(await response.arrayBuffer()) - if (!buf || buf.length === 0) { - return null - } - const ext = path.extname(relPath || '') - const stagedPath = path.join( - stagingRoot, - 'haximp-' + Date.now() + '-' + idx + '-' + Math.floor(Math.random() * 1000000) + ext - ) - fs.writeFileSync(stagedPath, buf) - return stagedPath - } catch (e) { - return null - } -} +// files are staged locally so createSite can ingest them as file entities +const { getBulkImportStagingRoot, stageRemoteFile } = require('../../../../lib/stageRemoteFile.js') /** * POST /system/api/v1/site/import/:platform diff --git a/src/systemRoutes/v1/routes/imports/convertOpenstaxToSite.js b/src/systemRoutes/v1/routes/imports/convertOpenstaxToSite.js new file mode 100644 index 00000000..d35d86ef --- /dev/null +++ b/src/systemRoutes/v1/routes/imports/convertOpenstaxToSite.js @@ -0,0 +1,607 @@ +const path = require('path') +const crypto = require('crypto') +const fs = require('fs-extra') +const { parse } = require('node-html-parser') +const JSONOutlineSchemaItem = require('../../../../lib/JSONOutlineSchemaItem.js') +const { HAXCMS } = require('../../../../lib/HAXCMS.js') +const { escapeHTMLAttribute } = require('../../../../lib/sanitizeContent.js') +// reached through the module object so the network boundary can be stubbed in tests +const safeFetchLib = require('../../../../lib/safeFetch.js') + +const OPENSTAX_ORIGIN = 'https://openstax.org' +const RELEASE_URL = `${OPENSTAX_ORIGIN}/rex/release.json` +const BOOK_LOOKUP_URL = `${OPENSTAX_ORIGIN}/apps/cms/api/v2/pages/?type=books.Book&fields=title,cnx_id&slug=` +// Safety valves: a book runs to hundreds of pages and thousands of images. +// A 53 page book with 81 images took ~160s and 20MB, so the budget is set to +// carry the largest books (~260 pages) rather than truncate them. Exported so +// tests can exercise the limits without downloading a book. +const LIMITS = { + maxPages: 500, + maxImages: 1000, + maxImageBytes: 250 * 1024 * 1024, + fetchBudgetSeconds: 900, + requestDelayMs: 200, +} +// image types HAXCMSFile accepts, keyed by the content type the archive serves +const EXTENSION_BY_CONTENT_TYPE = { + 'image/png': 'png', + 'image/jpeg': 'jpg', + 'image/gif': 'gif', + 'image/webp': 'webp', +} +// license codes site.license understands (mirrors convertPressbooksToSite) +const SUPPORTED_SITE_LICENSES = ['by', 'by-sa', 'by-nd', 'by-nc', 'by-nc-sa', 'by-nc-nd'] +// attributes worth keeping, by tag. Everything else — os-* classes, fs-id ids, +// data-type hooks, inline styles — is dropped so HAX receives clean markup +const KEEP_ATTRIBUTES = { + a: ['href'], + img: ['src', 'alt', 'width', 'height'], + td: ['colspan', 'rowspan', 'headers'], + th: ['colspan', 'rowspan', 'scope'], + ol: ['start', 'type'], + math: ['display'], + 'media-image': ['source', 'alt'], +} +// dropped outright: presentation and script material with no content value +const DROP_ELEMENTS = ['style', 'script', 'link', 'meta', 'noscript', 'head'] + +/** + * POST /system/api/v1/site/import/openstax + * Convert an OpenStax textbook into a HAXcms site schema. + * + * Expects JSON body with a `repoUrl` param: any OpenStax book URL, either + * https://openstax.org/details/books/ or + * https://openstax.org/books//pages/. + * + * Reads the book through OpenStax's public archive API (the web reader is a + * client-rendered app whose table of contents is not in the served HTML): + * /rex/release.json -> archive version + book versions + * /apps/cms/api/v2/pages/?slug= -> book slug to content id + * /contents/@.json -> title, license, page tree + * /contents/@:.json -> one page of XHTML + * + * Returns { status: 200, data: { items: [...], filename: string, files: {...}, site: {...} } }. + */ +async function convertOpenstaxToSite(req, res) { + const body = readRequestBody(req) + const repoUrl = body && typeof body.repoUrl === 'string' ? body.repoUrl.trim() : '' + if (repoUrl === '') { + return sendError(res, 400, 'missing `repoUrl` param') + } + let bookSlug = '' + try { + bookSlug = bookSlugFromUrl(repoUrl) + } + catch (error) { + return sendError(res, 400, error.message) + } + try { + const book = await resolveBook(bookSlug) + const imported = await importBook(book) + return res.json({ + status: 200, + data: { + items: imported.items, + filename: bookSlug, + files: imported.files, + site: { license: book.licenseCode }, + // true when the page cap or time budget stopped the import early; + // those pages carry a link to their source instead of the body + truncated: imported.truncated, + }, + }) + } + catch (error) { + const status = error && error.status ? error.status : 422 + const message = + error && error.message ? error.message : 'Unable to import this OpenStax book' + return sendError(res, status, message) + } +} + +/** Accept a parsed body or a JSON string, as the sibling importers do. */ +function readRequestBody(req) { + if (req && req.body && typeof req.body === 'object') { + return req.body + } + if (req && req.body && typeof req.body === 'string') { + try { + return JSON.parse(req.body.trim()) + } + catch (e) { + return {} + } + } + return {} +} + +function sendError(res, status, message) { + return res.status(status).json({ + status: status, + data: { error: message, items: [], filename: null, files: {} }, + }) +} + +/** An error carrying the HTTP status the route should answer with. */ +function importError(status, message) { + const error = new Error(message) + error.status = status + return error +} + +/** Pull the book slug out of a details or reader URL, rejecting other hosts. */ +function bookSlugFromUrl(sourceUrl) { + let parsed = null + try { + parsed = new URL(sourceUrl) + } + catch (e) { + throw importError(400, `\`repoUrl\` is not a valid URL: ${sourceUrl}`) + } + const host = parsed.hostname.toLowerCase() + if (host !== 'openstax.org' && host !== 'www.openstax.org') { + throw importError(400, `\`repoUrl\` must be an openstax.org URL, got: ${parsed.hostname}`) + } + const segments = parsed.pathname.split('/').filter((segment) => segment !== '') + // /details/books/ and /books//pages/ + const booksIndex = segments.indexOf('books') + if (booksIndex === -1 || !segments[booksIndex + 1]) { + throw importError( + 400, + 'unable to read a book slug from `repoUrl`; expected /details/books/ or /books//pages/', + ) + } + return segments[booksIndex + 1] +} + +/** GET JSON through safeFetch, which applies the SSRF guard. */ +async function fetchJson(url, description) { + let response = null + try { + response = await safeFetchLib.safeFetch(url) + } + catch (e) { + throw importError(422, `unable to reach OpenStax for ${description}: ${e.message}`) + } + if (!response.ok) { + throw importError(422, `OpenStax returned ${response.status} for ${description}`) + } + try { + return JSON.parse(await response.text()) + } + catch (e) { + throw importError(422, `OpenStax returned unreadable data for ${description}`) + } +} + +/** + * Resolve a book slug to everything needed to read it: content id, version, + * archive base, title, license and the page tree. + */ +async function resolveBook(bookSlug) { + const release = await fetchJson(RELEASE_URL, 'the OpenStax release manifest') + const archivePath = release && release.archiveUrl ? String(release.archiveUrl) : '' + if (archivePath === '') { + throw importError(422, 'the OpenStax release manifest did not name an archive') + } + const lookup = await fetchJson( + BOOK_LOOKUP_URL + encodeURIComponent(bookSlug), + `the book "${bookSlug}"`, + ) + const entries = lookup && Array.isArray(lookup.items) ? lookup.items : [] + if (entries.length === 0 || !entries[0].cnx_id) { + throw importError(422, `OpenStax has no book with the slug "${bookSlug}"`) + } + const contentId = String(entries[0].cnx_id) + const books = release && release.books ? release.books : {} + const bookRelease = books[contentId] + if (!bookRelease || !bookRelease.defaultVersion) { + throw importError(422, `OpenStax is not currently publishing "${bookSlug}"`) + } + const archiveUrl = `${OPENSTAX_ORIGIN}${archivePath}` + const version = String(bookRelease.defaultVersion) + const contents = await fetchJson( + `${archiveUrl}/contents/${contentId}@${version}.json`, + `the contents of "${bookSlug}"`, + ) + if (!contents || !contents.tree || !Array.isArray(contents.tree.contents)) { + throw importError(422, `OpenStax returned no table of contents for "${bookSlug}"`) + } + const license = contents.license && typeof contents.license === 'object' ? contents.license : {} + return { + slug: bookSlug, + contentId: contentId, + version: version, + archiveUrl: archiveUrl, + title: plainText(contents.title || bookSlug), + license: license, + licenseCode: licenseCode(license), + tree: contents.tree, + } +} + +/** Map an OpenStax license record onto a site.license code, defaulting to by. */ +function licenseCode(license) { + const value = `${license.url || ''} ${license.name || ''}`.toLowerCase() + for (let i = 0; i < SUPPORTED_SITE_LICENSES.length; i++) { + const code = SUPPORTED_SITE_LICENSES[i] + if (value.indexOf(`/licenses/${code}/`) !== -1) { + return code + } + } + const compact = value.replace(/[^a-z]/g, '') + const nonCommercial = compact.indexOf('noncommercial') !== -1 + const noDerivatives = compact.indexOf('noderivatives') !== -1 + const shareAlike = compact.indexOf('sharealike') !== -1 + if (nonCommercial && noDerivatives) { + return 'by-nc-nd' + } + if (nonCommercial && shareAlike) { + return 'by-nc-sa' + } + if (nonCommercial) { + return 'by-nc' + } + if (noDerivatives) { + return 'by-nd' + } + if (shareAlike) { + return 'by-sa' + } + return 'by' +} + +/** + * Slug for one page title. Section titles end in punctuation often enough + * ("What Is Finance?") that cleanTitle leaves a trailing separator, so trim + * the edges; keep cleanTitle's output if trimming would empty the slug. + */ +function titleSlug(title) { + const cleaned = HAXCMS.cleanTitle(title) + const trimmed = cleaned.replace(/^-+|-+$/g, '') + return trimmed === '' ? cleaned : trimmed +} + +/** Strip markup and collapse whitespace; tree titles carry os-* span markup. */ +function plainText(value) { + const raw = String(value === null || value === undefined ? '' : value) + // tags become spaces first: OpenStax titles are adjacent spans ("1.1", + // "What Is Finance?") that would otherwise run together + return parse(raw.replace(/<[^>]+>/g, ' ')) + .text.replace(/\s+/g, ' ') + .trim() +} + +/** + * Walk the tree into a flat, ordered list of nodes. Units and chapters keep + * their children, so the JOS parent/indent relationship follows the book. + */ +function flattenTree(tree) { + const nodes = [] + const walk = (branch, depth, parentIndex) => { + const children = branch && Array.isArray(branch.contents) ? branch.contents : [] + let order = 0 + for (let i = 0; i < children.length; i++) { + const child = children[i] + const title = plainText(child.title) + if (title === '') { + continue + } + const hasChildren = Array.isArray(child.contents) && child.contents.length > 0 + const index = nodes.length + nodes.push({ + title: title, + depth: depth, + order: order, + parentIndex: parentIndex, + // leaf nodes are pages; "@" needs the version trimmed + pageId: hasChildren ? null : String(child.id || '').split('@')[0], + sourceSlug: child.slug ? String(child.slug) : '', + }) + order++ + if (hasChildren) { + walk(child, depth + 1, index) + } + } + } + walk(tree, 0, null) + return nodes +} + +/** Fetch and convert every page in the tree into JOS items plus staged files. */ +async function importBook(book) { + const nodes = flattenTree(book.tree) + if (nodes.length === 0) { + throw importError(422, `OpenStax returned an empty table of contents for "${book.slug}"`) + } + const context = { + book: book, + accessed: new Date().toISOString(), + // slugs by page id, so in-book links can point at the imported pages + slugByPageId: {}, + // staged image downloads, keyed by the archive resource path + imagesByResource: {}, + files: {}, + imageCount: 0, + imageBytes: 0, + stagingDirectory: null, + // keeps this import's staged names apart from any other import's + importId: crypto.randomUUID(), + startedAt: Date.now(), + truncated: false, + } + const items = [] + // first pass: items and slugs, so page bodies can link to siblings by slug + for (let i = 0; i < nodes.length; i++) { + const node = nodes[i] + const item = new JSONOutlineSchemaItem() + item.title = node.title + item.indent = node.depth + item.order = node.order + const parentItem = node.parentIndex === null ? null : items[node.parentIndex] + item.parent = parentItem ? parentItem.id : null + // nested slugs carry the parent path, matching convertPressbooksToSite + item.slug = parentItem + ? `${parentItem.slug}/${titleSlug(node.title)}` + : titleSlug(node.title) + item.metadata = { + sourceType: 'openstax', + openstax: { + bookSlug: book.slug, + bookTitle: book.title, + pageSlug: node.sourceSlug, + license: book.license, + publisher: 'OpenStax / Rice University', + accessed: context.accessed, + }, + } + if (node.pageId) { + item.metadata.source = `${OPENSTAX_ORIGIN}/books/${book.slug}/pages/${node.sourceSlug}` + context.slugByPageId[node.pageId] = item.slug + } + items.push(item) + node.item = item + } + // second pass: page bodies, bounded by the page cap and the time budget + let fetched = 0 + for (let i = 0; i < nodes.length; i++) { + const node = nodes[i] + if (!node.pageId) { + // a unit or chapter heading: a landing page with no body of its own + node.item.contents = '

' + continue + } + if (fetched >= LIMITS.maxPages || budgetExhausted(context)) { + context.truncated = true + node.item.contents = sourceFallback(node.item.metadata.source) + continue + } + if (fetched > 0) { + await delay(LIMITS.requestDelayMs) + } + const page = await fetchJson( + `${book.archiveUrl}/contents/${book.contentId}@${book.version}:${node.pageId}.json`, + `the page "${node.title}"`, + ) + fetched++ + node.item.contents = await cleanPage(page && page.content ? page.content : '', context) + if (page && page.abstract) { + node.item.description = plainText(page.abstract).slice(0, 280) + } + } + return { items: items, files: context.files, truncated: context.truncated } +} + +function budgetExhausted(context) { + return (Date.now() - context.startedAt) / 1000 > LIMITS.fetchBudgetSeconds +} + +function delay(milliseconds) { + return new Promise((resolve) => { + setTimeout(resolve, milliseconds) + }) +} + +/** Pages that could not be fetched still link out to their source. */ +function sourceFallback(source) { + if (!source) { + return '

' + } + return `

Read this page on OpenStax.

` +} + +/** + * Turn one page of OpenStax XHTML into clean semantic HTML: drop the styling + * layer, keep the structure, stage images into the site and point in-book + * links at the imported pages. + */ +async function cleanPage(content, context) { + if (typeof content !== 'string' || content.trim() === '') { + return '

' + } + const root = parse(content) + // the body of the page, minus the xhtml wrapper the archive serves + const container = root.querySelector('[data-type="page"]') || root.querySelector('body') || root + const dropSelector = DROP_ELEMENTS.join(',') + const dropped = container.querySelectorAll(dropSelector) + for (let i = 0; i < dropped.length; i++) { + dropped[i].remove() + } + // the page title becomes the item title, so it does not repeat in the body + const documentTitle = container.querySelector('[data-type="document-title"]') + if (documentTitle) { + documentTitle.remove() + } + await stageImages(container, context) + rewriteLinks(container, context) + stripPresentationAttributes(container) + // collapse only the archive's pretty-printing between tags: a blanket + // whitespace collapse would rewrite the inside of pre/code blocks + const html = container.innerHTML.replace(/>\s*\n\s*<').trim() + return html === '' ? '

' : html +} + +/** + * Download each image into the bulk-import staging directory and render it as + * media-image at its site path, the same markup the docx import produces. + * createSite only accepts staged local paths in build.files (see + * haxtheweb/issues#3060), ingests them as file entities, and then links each + * page to their uuids. Images that cannot be staged keep an absolute source. + */ +async function stageImages(container, context) { + const images = container.querySelectorAll('img') + for (let i = 0; i < images.length; i++) { + const image = images[i] + const source = image.getAttribute('src') + if (!source) { + continue + } + const resource = resourcePath(source) + if (resource === '') { + continue + } + const absolute = `${context.book.archiveUrl}/${resource}` + if (!Object.prototype.hasOwnProperty.call(context.imagesByResource, resource)) { + context.imagesByResource[resource] = await downloadImage(absolute, resource, context) + } + const staged = context.imagesByResource[resource] + if (staged) { + const alt = escapeHTMLAttribute(image.getAttribute('alt') || '') + image.replaceWith(``) + } + else { + image.setAttribute('src', absolute) + } + } +} + +/** Normalize an archive-relative image src ("../resources/") to a path. */ +function resourcePath(source) { + const value = String(source).trim() + if (value === '' || /^(https?:)?\/\//i.test(value) || value.indexOf('data:') === 0) { + return '' + } + const match = value.match(/resources\/([A-Za-z0-9._-]+)$/) + return match ? `resources/${match[1]}` : '' +} + +/** + * Fetch one image into /tmp/imports and record it + * in the files map. Returns null when it cannot be stored, so the caller can + * fall back to the source URL. + */ +async function downloadImage(url, resource, context) { + if (context.imageCount >= LIMITS.maxImages) { + return null + } + let response = null + try { + response = await safeFetchLib.safeFetch(url) + } + catch (e) { + return null + } + if (!response.ok) { + return null + } + const contentType = String(response.headers.get('content-type') || '') + .split(';')[0] + .trim() + .toLowerCase() + const extension = EXTENSION_BY_CONTENT_TYPE[contentType] + // archive resources are extension-less hashes, so the type names the file + if (!extension) { + return null + } + try { + const buffer = Buffer.from(await response.arrayBuffer()) + // stop before writing anything that would push the import past the budget + if (context.imageBytes + buffer.length > LIMITS.maxImageBytes) { + return null + } + // staged flat in the bulk-import root, like convertHaxcmsToSite, so the + // moves createSite makes on ingest leave nothing behind + if (!context.stagingDirectory) { + context.stagingDirectory = path.join(HAXCMS.configDirectory, 'tmp', 'imports') + fs.ensureDirSync(context.stagingDirectory) + } + const name = `${path.basename(resource)}.${extension}` + const stagedPath = path.join(context.stagingDirectory, `openstax-${context.importId}-${name}`) + fs.writeFileSync(stagedPath, buffer) + const sitePath = `files/${name}` + context.files[sitePath] = stagedPath + context.imageCount++ + context.imageBytes += buffer.length + return { sitePath: sitePath, stagedPath: stagedPath } + } + catch (e) { + return null + } +} + +/** + * Point in-book links at the imported pages and make every other relative + * link absolute, so nothing resolves against the new site by accident. + */ +function rewriteLinks(container, context) { + const links = container.querySelectorAll('a') + for (let i = 0; i < links.length; i++) { + const link = links[i] + const href = link.getAttribute('href') + if (!href) { + continue + } + const value = href.trim() + if (value === '' || value.indexOf('#') === 0 || /^[a-z][a-z0-9+.-]*:/i.test(value)) { + continue + } + // in-book links carry the target page id, with an optional fragment + const target = value.match(/\/contents\/([0-9a-f-]{36})/i) + const pageSlug = target ? context.slugByPageId[target[1].toLowerCase()] : null + if (pageSlug) { + const fragment = value.indexOf('#') === -1 ? '' : value.slice(value.indexOf('#')) + link.setAttribute('href', `${pageSlug}${fragment}`) + continue + } + try { + link.setAttribute('href', new URL(value, `${context.book.archiveUrl}/`).href) + } + catch (e) { + link.removeAttribute('href') + } + } +} + +/** + * Drop the OpenStax styling layer (os-* classes, fs-id ids, data-type hooks, + * inline styles) while keeping the attributes that carry meaning. MathML is + * left untouched: its attributes are part of the notation. + */ +function stripPresentationAttributes(container) { + const elements = container.querySelectorAll('*') + for (let i = 0; i < elements.length; i++) { + const element = elements[i] + const tag = String(element.rawTagName || '').toLowerCase() + if (tag === '' || insideMath(element)) { + continue + } + const keep = KEEP_ATTRIBUTES[tag] || [] + const names = Object.keys(element.attributes || {}) + for (let j = 0; j < names.length; j++) { + if (keep.indexOf(names[j].toLowerCase()) === -1) { + element.removeAttribute(names[j]) + } + } + } +} + +function insideMath(node) { + for (let parent = node.parentNode; parent; parent = parent.parentNode) { + if (String(parent.rawTagName || '').toLowerCase() === 'math') { + return true + } + } + return false +} + +module.exports = { convertOpenstaxToSite, LIMITS } diff --git a/src/systemRoutes/v1/routes/siteImport.js b/src/systemRoutes/v1/routes/siteImport.js index f1b62b02..c538957c 100644 --- a/src/systemRoutes/v1/routes/siteImport.js +++ b/src/systemRoutes/v1/routes/siteImport.js @@ -7,13 +7,14 @@ const { convertWordpressToSite } = require('./imports/convertWordpressToSite.js' const { convertElmslnToSite } = require('./imports/convertElmslnToSite.js') const { convertDrupalBookToSite } = require('./imports/convertDrupalBookToSite.js') const { convertPloneToSite } = require('./imports/convertPloneToSite.js') +const { convertOpenstaxToSite } = require('./imports/convertOpenstaxToSite.js') /** * POST /system/api/v1/site/import/:platform * Dispatcher that routes platform import requests to the correct converter. * * Supported platforms: haxcms, html, pressbooks, gitbook, notion, wordpress, - * elmsln, drupal-book, plone. + * elmsln, drupal-book, plone, openstax. * Returns { status: 200, data: { items: [...], filename: string, ... } }. */ async function siteImport(req, res) { @@ -41,6 +42,8 @@ async function siteImport(req, res) { return convertDrupalBookToSite(req, res) case 'plone': return convertPloneToSite(req, res) + case 'openstax': + return convertOpenstaxToSite(req, res) default: return res.status(400).json({ status: 400, diff --git a/test/api-conformance/ssrf.conformance.test.cjs b/test/api-conformance/ssrf.conformance.test.cjs index 4ef77770..5b5a0470 100644 --- a/test/api-conformance/ssrf.conformance.test.cjs +++ b/test/api-conformance/ssrf.conformance.test.cjs @@ -410,6 +410,138 @@ test('createSite siteFiles SSRF + extension guards', async (t) => { }) }) +test('createSite build.files SSRF guards (#3060)', async (t) => { + // build.files may carry http(s) URLs, which createSite downloads through + // safeFetch. As with siteFiles, the runtime's own loopback origin serves + // real content, so without the SSRF guard an .html entry pointing at it + // would be saved into the site; with it the entry is skipped and the site + // is still created. Values that are neither http(s) URLs nor staged files + // keep failing the request, as they have since GHSA-q862-gcgq-5m6g. + async function createWithFiles(siteName, build) { + const result = await sendHttpRequest({ + method: 'POST', + url: `${runtime.baseUrl}/system/api/v1/sites`, + headers: authHeaders(runtime.jwt), + data: JSON.stringify({ site: { name: siteName }, build: build }), + }) + let siteDir = null + if (result.status === 200) { + const body = JSON.parse(result.bodyText) + const createdName = + body && + body.data && + body.data.metadata && + body.data.metadata.site && + body.data.metadata.site.name + ? body.data.metadata.site.name + : siteName + siteDir = path.join(runtime.runtimeRoot, SITE_DIRECTORY_NAME, createdName) + } + return { result: result, siteDir: siteDir } + } + + await t.test('loopback download URL is skipped (file not written, site still created)', async () => { + const created = await createWithFiles(`files-loop-${runtime.testStartTimestamp}`, { + structure: 'website', + files: { 'files/loop.html': `${runtime.baseUrl}/` }, + }) + assert.equal(created.result.status, 200, `createSite failed: ${created.result.status}: ${created.result.bodyText}`) + assert.ok(fs.pathExistsSync(created.siteDir), 'expected the created site directory to exist') + assert.equal( + fs.pathExistsSync(path.join(created.siteDir, 'files', 'loop.html')), + false, + 'loopback build.files download must NOT be written to the site directory (SSRF guard)', + ) + }) + + await t.test('cloud metadata IP download URL is skipped (file not written)', async () => { + const created = await createWithFiles(`files-meta-${runtime.testStartTimestamp}`, { + structure: 'website', + files: { 'files/meta.txt': 'http://169.254.169.254/latest/meta-data/iam/security-credentials/' }, + }) + assert.equal(created.result.status, 200, `createSite failed: ${created.result.status}: ${created.result.bodyText}`) + assert.equal( + fs.pathExistsSync(path.join(created.siteDir, 'files', 'meta.txt')), + false, + 'metadata-IP build.files download must NOT be written (SSRF guard rejects 169.254.* before fetch)', + ) + }) + + await t.test('file URLs, unstaged paths and the advisory payload are still rejected', async () => { + const payloads = [ + { 'files/passwd.txt': 'file:///etc/passwd' }, + { 'files/passwd.txt': '/etc/passwd' }, + { 'poc.txt': { tmp_name: 'http://169.254.169.254/latest/meta-data/iam/security-credentials/' } }, + ] + for (let i = 0; i < payloads.length; i++) { + const created = await createWithFiles(`files-reject-${i}-${runtime.testStartTimestamp}`, { + structure: 'website', + files: payloads[i], + }) + assert.equal( + created.result.status, + 400, + `expected 400 for ${JSON.stringify(payloads[i])}: ${created.result.bodyText}`, + ) + } + }) + + await t.test('disallowed extension is rejected before any fetch (CWE-434)', async () => { + const created = await createWithFiles(`files-ext-${runtime.testStartTimestamp}`, { + structure: 'website', + files: { 'files/x.php': `${runtime.baseUrl}/` }, + }) + assert.equal(created.result.status, 400, `expected 400: ${created.result.bodyText}`) + }) + + await t.test('a staged file beside a blocked URL is saved and linked to its page', async () => { + const stagingRoot = path.join(runtime.runtimeConfigRoot, 'tmp', 'imports') + fs.ensureDirSync(stagingRoot) + const staged = path.join(stagingRoot, `conformance-${runtime.testStartTimestamp}.png`) + fs.writeFileSync( + staged, + Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=', + 'base64', + ), + ) + const created = await createWithFiles(`files-mixed-${runtime.testStartTimestamp}`, { + structure: 'import', + type: 'import', + items: [ + { + id: 'item-files', + title: 'Files', + slug: 'files-page', + indent: 0, + order: 0, + parent: null, + metadata: {}, + contents: '

staged

', + }, + ], + files: { + 'files/staged.png': staged, + 'files/meta.png': 'http://169.254.169.254/latest/meta-data/meta.png', + }, + }) + assert.equal(created.result.status, 200, `createSite failed: ${created.result.status}: ${created.result.bodyText}`) + assert.ok(fs.pathExistsSync(path.join(created.siteDir, 'files', 'staged.png')), 'the staged file is saved') + assert.equal( + fs.pathExistsSync(path.join(created.siteDir, 'files', 'meta.png')), + false, + 'the blocked URL beside it is skipped', + ) + const filesJson = fs.readJsonSync(path.join(created.siteDir, 'files', 'files.json')) + const record = filesJson.data.files.find((entry) => entry.path === 'files/staged.png') + assert.ok(record && record.uuid, 'files.json records the staged file') + const manifest = fs.readJsonSync(path.join(created.siteDir, 'site.json')) + const page = manifest.items.find((item) => item.id === 'item-files') + assert.ok(page, 'the imported page exists') + assert.deepEqual(page.metadata.files, [record.uuid], 'the page is linked to the file entity') + }) +}) + test('site/import converters reject private/loopback repoUrl', async (t) => { await t.test('import/html returns 400 for loopback repoUrl', async () => { // Without the SSRF guard, fetch(runtime.baseUrl) returns 200 (dashboard) diff --git a/test/unit/convertOpenstaxToSite.test.cjs b/test/unit/convertOpenstaxToSite.test.cjs new file mode 100644 index 00000000..7eaec02d --- /dev/null +++ b/test/unit/convertOpenstaxToSite.test.cjs @@ -0,0 +1,367 @@ +'use strict' + +// Unit tests for the OpenStax importer (#2912). +// +// The importer reads a book through OpenStax's archive API: the release +// manifest, a slug lookup, the book tree, then one JSON document per page. +// Every request goes through safeFetch, which these tests stub, so the suite +// never touches the network. +// +// Constraints honored: CommonJS (.cjs), require(), NO optional chaining, +// node:test + node:assert/strict. + +const { test, describe, beforeEach, afterEach } = require('node:test') +const assert = require('node:assert/strict') +const path = require('path') +const fs = require('fs-extra') + +const safeFetchLib = require('../../src/lib/safeFetch.js') +const { HAXCMS } = require('../../src/lib/HAXCMS.js') +const { convertOpenstaxToSite, LIMITS } = require('../../src/systemRoutes/v1/routes/imports/convertOpenstaxToSite.js') + +const BOOK_ID = '052b8372-0c5e-4ff6-8fd3-326377e9e91f' +const BOOK_VERSION = '56af1c4' +const ARCHIVE = '/apps/archive/20260604.144757' +const PAGE_ONE = '11111111-1111-1111-1111-111111111111' +const PAGE_TWO = '22222222-2222-2222-2222-222222222222' +const PREFACE = '33333333-3333-3333-3333-333333333333' +// a 1x1 PNG, standing in for an archive image resource +const PNG = Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=', + 'base64', +) + +// OpenStax page XHTML, with the styling layer and structure it really ships +const PAGE_ONE_CONTENT = ` +
+

1.1What Is Finance?

+
By the end of this section you will be able to explain finance.
+

Finance is the study of money.

+
A bar chart of returns
Figure 1.1Returns over time.
+

Link to Learning

+

Worked Example

Compute 2+22+2.

+
Year
2024
+
` + +const PAGE_TWO_CONTENT = `

1.2 The Role of Finance

Second section body.

` +const PREFACE_CONTENT = `

Preface

Welcome to the book.

` + +function treeNode(id, title, slug, contents) { + const node = { id: id + '@' + BOOK_VERSION, title: title, slug: slug } + if (contents) { + node.contents = contents + } + return node +} + +function defaultRoutes() { + return { + 'https://openstax.org/rex/release.json': { + json: { archiveUrl: ARCHIVE, books: { [BOOK_ID]: { defaultVersion: BOOK_VERSION } } }, + }, + 'https://openstax.org/apps/cms/api/v2/pages/?type=books.Book&fields=title,cnx_id&slug=principles-finance': { + json: { items: [{ title: 'Principles of Finance', cnx_id: BOOK_ID }] }, + }, + [`https://openstax.org${ARCHIVE}/contents/${BOOK_ID}@${BOOK_VERSION}.json`]: { + json: { + title: 'Principles of Finance', + license: { + name: 'Creative Commons Attribution-NonCommercial-ShareAlike License', + url: 'https://creativecommons.org/licenses/by-nc-sa/4.0/', + }, + tree: { + contents: [ + treeNode(PREFACE, 'Preface', 'preface'), + treeNode(BOOK_ID, 'Chapter 1Introduction to Finance', '1-introduction-to-finance', [ + treeNode(PAGE_ONE, '1.1What Is Finance?', '1-1-what-is-finance'), + treeNode(PAGE_TWO, '1.2 The Role of Finance', '1-2-the-role-of-finance'), + ]), + ], + }, + }, + }, + [`https://openstax.org${ARCHIVE}/contents/${BOOK_ID}@${BOOK_VERSION}:${PREFACE}.json`]: { + json: { slug: 'preface', title: 'Preface', content: PREFACE_CONTENT }, + }, + [`https://openstax.org${ARCHIVE}/contents/${BOOK_ID}@${BOOK_VERSION}:${PAGE_ONE}.json`]: { + json: { slug: '1-1-what-is-finance', title: 'What Is Finance?', abstract: 'Explain finance.', content: PAGE_ONE_CONTENT }, + }, + [`https://openstax.org${ARCHIVE}/contents/${BOOK_ID}@${BOOK_VERSION}:${PAGE_TWO}.json`]: { + json: { slug: '1-2-the-role-of-finance', title: 'The Role of Finance', content: PAGE_TWO_CONTENT }, + }, + [`https://openstax.org${ARCHIVE}/resources/abc123def456`]: { + buffer: PNG, + contentType: 'image/png', + }, + } +} + +function buildResponse(entry) { + const body = entry.buffer ? entry.buffer : Buffer.from(JSON.stringify(entry.json), 'utf8') + return { + ok: entry.ok === false ? false : true, + status: entry.status ? entry.status : 200, + headers: { + get: function (name) { + if (String(name).toLowerCase() === 'content-type') { + return entry.contentType ? entry.contentType : 'application/json' + } + return null + }, + }, + text: function () { + return Promise.resolve(body.toString('utf8')) + }, + arrayBuffer: function () { + return Promise.resolve(body) + }, + } +} + +function makeResponse() { + const res = { + statusCode: 200, + body: null, + status: function (code) { + this.statusCode = code + return this + }, + json: function (payload) { + this.body = payload + return this + }, + } + return res +} + +describe('convertOpenstaxToSite — #2912', () => { + let routes + let requested + let originalSafeFetch + let stagedFiles = [] + let originalLimits + + beforeEach(() => { + routes = defaultRoutes() + requested = [] + originalSafeFetch = safeFetchLib.safeFetch + originalLimits = Object.assign({}, LIMITS) + safeFetchLib.safeFetch = function (url) { + requested.push(url) + const entry = routes[url] + if (!entry) { + return Promise.reject(new Error('unexpected request: ' + url)) + } + return Promise.resolve(buildResponse(entry)) + } + }) + + afterEach(() => { + safeFetchLib.safeFetch = originalSafeFetch + Object.assign(LIMITS, originalLimits) + // staged files sit in the shared import root; drop only the ones this + // test created so nothing is left on disk + for (let i = 0; i < stagedFiles.length; i++) { + try { + fs.removeSync(stagedFiles[i]) + } catch (e) {} + } + stagedFiles = [] + }) + + // record every staged file the importer reports, for cleanup + function trackStaged(res) { + const files = res && res.body && res.body.data && res.body.data.files ? res.body.data.files : {} + const names = Object.keys(files) + for (let i = 0; i < names.length; i++) { + stagedFiles.push(files[names[i]]) + } + } + + async function importBook(repoUrl) { + const res = makeResponse() + await convertOpenstaxToSite({ body: { repoUrl: repoUrl || 'https://openstax.org/details/books/principles-finance' } }, res) + trackStaged(res) + return res + } + + test('rejects a missing, off-site or slug-less repoUrl', async () => { + const cases = [ + [undefined, 'missing'], + ['https://example.org/details/books/principles-finance', 'openstax.org'], + ['https://openstax.org/subjects/business', 'book slug'], + ['not a url', 'valid URL'], + ] + for (const entry of cases) { + const res = makeResponse() + await convertOpenstaxToSite({ body: entry[0] ? { repoUrl: entry[0] } : {} }, res) + assert.equal(res.statusCode, 400, JSON.stringify(res.body)) + assert.ok( + String(res.body.data.error).indexOf(entry[1]) !== -1, + `expected "${entry[1]}" in: ${res.body.data.error}`, + ) + } + assert.deepEqual(requested, [], 'nothing is fetched for a bad request') + }) + + test('reports a book OpenStax does not publish as 422', async () => { + routes['https://openstax.org/apps/cms/api/v2/pages/?type=books.Book&fields=title,cnx_id&slug=principles-finance'].json = { items: [] } + const res = await importBook() + assert.equal(res.statusCode, 422) + assert.ok(String(res.body.data.error).indexOf('no book with the slug') !== -1, res.body.data.error) + }) + + test('reports an unreadable table of contents as 422', async () => { + routes[`https://openstax.org${ARCHIVE}/contents/${BOOK_ID}@${BOOK_VERSION}.json`].json = { title: 'Principles of Finance' } + const res = await importBook() + assert.equal(res.statusCode, 422) + assert.ok(String(res.body.data.error).indexOf('no table of contents') !== -1, res.body.data.error) + }) + + test('builds the chapter and section hierarchy as JOS items', async () => { + const res = await importBook() + assert.equal(res.statusCode, 200, JSON.stringify(res.body)) + const items = res.body.data.items + assert.equal(res.body.data.filename, 'principles-finance') + assert.equal(items.length, 4) + const [preface, chapter, sectionOne, sectionTwo] = items + assert.equal(preface.title, 'Preface') + assert.equal(preface.indent, 0) + assert.equal(preface.parent, null) + assert.equal(preface.slug, 'preface') + assert.equal(chapter.title, 'Chapter 1 Introduction to Finance') + assert.equal(chapter.indent, 0) + assert.equal(chapter.order, 1) + assert.equal(sectionOne.indent, 1) + assert.equal(sectionOne.parent, chapter.id, 'sections hang off their chapter') + assert.equal(sectionOne.order, 0) + assert.equal(sectionTwo.order, 1) + assert.equal(sectionOne.slug, `${chapter.slug}/1-1-what-is-finance`, 'nested slugs carry the chapter path') + assert.equal(chapter.contents, '

', 'a chapter heading is a landing page') + }) + + test('carries the book license, source and attribution', async () => { + const res = await importBook() + assert.equal(res.body.data.site.license, 'by-nc-sa', 'the license is read per book, not assumed') + const section = res.body.data.items[2] + assert.equal(section.metadata.sourceType, 'openstax') + assert.equal(section.metadata.source, 'https://openstax.org/books/principles-finance/pages/1-1-what-is-finance') + assert.equal(section.metadata.openstax.bookTitle, 'Principles of Finance') + assert.equal(section.metadata.openstax.publisher, 'OpenStax / Rice University') + assert.equal(section.metadata.openstax.license.url, 'https://creativecommons.org/licenses/by-nc-sa/4.0/') + assert.ok(section.metadata.openstax.accessed, 'access date recorded') + }) + + test('strips the OpenStax styling layer but keeps the structure', async () => { + const res = await importBook() + const html = res.body.data.items[2].contents + for (const junk of ['os-para', 'os-figure', 'fs-id', 'data-type=', 'STYLING_FOR_DEVS', '') !== -1, 'figures survive') + assert.ok(html.indexOf('') !== -1 && html.indexOf('
') !== -1, 'tables keep headers') + assert.ok(html.indexOf('') !== -1, 'cell spans survive') + assert.ok(html.indexOf('Returns over time.') !== -1, 'captions survive as text') + assert.ok(html.indexOf('study of money') !== -1, 'inline markup survives') + assert.ok(html.indexOf('By the end of this section') !== -1, 'learning objectives survive') + }) + + test('keeps MathML, including the attributes that carry notation', async () => { + const res = await importBook() + const html = res.body.data.items[2].contents + assert.ok(html.indexOf('') !== -1, 'math element and display attribute kept') + assert.ok(html.indexOf('+') !== -1, 'attributes inside math are left alone') + assert.ok(html.indexOf('') !== -1, 'the TeX annotation is carried along') + }) + + test('stages images into files and renders them as media-image', async () => { + const res = await importBook() + const files = res.body.data.files + const names = Object.keys(files) + assert.deepEqual(names, ['files/abc123def456.png'], 'named from the resource with the served type') + assert.deepEqual(fs.readFileSync(files[names[0]]), PNG, 'the staged bytes are the image') + assert.ok( + files[names[0]].indexOf(path.join(HAXCMS.configDirectory, 'tmp', 'imports')) === 0, + 'staged where createSite accepts bulk imports', + ) + const html = res.body.data.items[2].contents + assert.ok( + html.indexOf('') !== -1, + 'rendered as media-image, the markup the docx import produces: ' + html.slice(0, 300), + ) + assert.equal(html.indexOf(' { + routes[`https://openstax.org${ARCHIVE}/resources/abc123def456`] = { buffer: Buffer.from(''), contentType: 'image/svg+xml' } + const res = await importBook() + assert.deepEqual(res.body.data.files, {}, 'nothing staged for an unsupported type') + const html = res.body.data.items[2].contents + assert.ok( + html.indexOf(` { + const res = await importBook() + const html = res.body.data.items[2].contents + const sectionTwo = res.body.data.items[3] + assert.ok(html.indexOf(``) !== -1, `in-book link rewritten: ${html}`) + assert.ok(html.indexOf('') !== -1, 'external links are untouched') + }) + + test('fetches each page once, in reading order', async () => { + const res = await importBook() + const pageIds = [] + for (let i = 0; i < requested.length; i++) { + const match = requested[i].match(/contents\/[0-9a-f-]+@[^:]+:([0-9a-f-]+)\.json$/) + if (match) { + pageIds.push(match[1]) + } + } + assert.deepEqual(pageIds, [PREFACE, PAGE_ONE, PAGE_TWO], 'reading order, no repeats') + }) + + test('stops at the page cap and links unimported pages to their source', async () => { + LIMITS.maxPages = 1 + const res = await importBook() + assert.equal(res.statusCode, 200, JSON.stringify(res.body)) + assert.equal(res.body.data.truncated, true, 'the caller is told the import was cut short') + const items = res.body.data.items + assert.ok(items[0].contents.indexOf('Welcome to the book') !== -1, 'the first page is imported') + assert.equal( + items[2].contents, + '

Read this page on OpenStax.

', + 'pages past the cap link out instead of arriving empty', + ) + }) + + test('reports an untruncated import as complete', async () => { + const res = await importBook() + assert.equal(res.body.data.truncated, false) + }) + + test('stops staging images at the image cap', async () => { + LIMITS.maxImages = 0 + const res = await importBook() + assert.deepEqual(res.body.data.files, {}, 'no image is staged once the cap is reached') + assert.ok( + res.body.data.items[2].contents.indexOf(` { + LIMITS.maxImageBytes = 1 + const res = await importBook() + assert.deepEqual(res.body.data.files, {}) + }) + + test('accepts a reader URL as well as a details URL', async () => { + const res = await importBook('https://openstax.org/books/principles-finance/pages/1-1-what-is-finance') + assert.equal(res.statusCode, 200, JSON.stringify(res.body)) + assert.equal(res.body.data.filename, 'principles-finance') + }) +}) diff --git a/test/unit/createSiteBuildFiles.test.cjs b/test/unit/createSiteBuildFiles.test.cjs new file mode 100644 index 00000000..c2a8d910 --- /dev/null +++ b/test/unit/createSiteBuildFiles.test.cjs @@ -0,0 +1,274 @@ +'use strict' + +// Unit tests for createSite's importBuildFile (#3060). +// +// Importers hand createSite remote files as http(s) URLs in build.files. Each +// one is fetched through safeFetch into the bulk-import staging root and then +// takes the same bulk-import HAXCMSFile.save as a staged file, so it becomes a +// file entity in files.json; linkImportedPageFiles then links pages to it. +// The GHSA-q862-gcgq-5m6g protections still hold: other schemes and paths +// outside the staging root are rejected, and private, loopback and metadata +// addresses are never contacted. +// +// The network is stubbed at safeFetch except where the real SSRF guard is +// under test, and HAXCMS.configDirectory points at a temp dir so staging and +// media settings stay isolated. +// +// Constraints honored: CommonJS (.cjs), require(), NO optional chaining, +// node:test + node:assert/strict. + +const { test, describe, beforeEach, afterEach } = require('node:test') +const assert = require('node:assert/strict') +const crypto = require('crypto') +const http = require('http') +const path = require('path') +const fs = require('fs-extra') +const os = require('os') +const sharp = require('sharp') + +const safeFetchMod = require('../../src/lib/safeFetch.js') +const { HAXCMS, HAXCMSSite } = require('../../src/lib/HAXCMS.js') +const createSite = require('../../src/systemRoutes/v1/routes/createSite.js') +const EntityRegistry = require('../../src/lib/EntityRegistry.js') +const FileStorage = require('../../src/lib/FileStorage.js') + +const { importBuildFile, linkImportedPageFiles } = createSite + +const PNG = Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=', + 'base64', +) + +describe('createSite importBuildFile — #3060', () => { + const realSafeFetch = safeFetchMod.safeFetch + const realConfigDirectory = HAXCMS.configDirectory + let tmpRoot + let stagingRoot + let site + let fetched + + beforeEach(async () => { + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), 'hax-test-')) + HAXCMS.configDirectory = path.join(tmpRoot, 'config') + stagingRoot = path.join(HAXCMS.configDirectory, 'tmp', 'imports') + await fs.ensureDir(stagingRoot) + site = new HAXCMSSite() + site.name = 'testsite' + site.siteDirectory = path.join(tmpRoot, 'testsite') + await fs.ensureDir(path.join(site.siteDirectory, 'files')) + site.manifest = { + metadata: { site: { name: 'testsite' } }, + items: [], + save: async function () { + return true + }, + } + fetched = [] + }) + + afterEach(async () => { + safeFetchMod.safeFetch = realSafeFetch + HAXCMS.configDirectory = realConfigDirectory + try { + await fs.remove(tmpRoot) + } catch (e) {} + }) + + // answer each URL from a map of url -> { status, body } or an Error to + // throw; anything unmapped answers 404 + function stubNetwork(responses) { + safeFetchMod.safeFetch = async function (url) { + fetched.push(url) + const entry = responses[url] + if (entry instanceof Error) { + throw entry + } + const status = entry ? entry.status || 200 : 404 + const body = entry && entry.body ? entry.body : Buffer.alloc(0) + return { + ok: status >= 200 && status < 300, + status: status, + headers: { + get: function () { + return null + }, + }, + arrayBuffer: async function () { + return body + }, + } + } + } + + // the file entity files.json holds for a path, or null; read-only, so it + // only finds records the save itself wrote + function entityAt(relativePath) { + const fileStorage = FileStorage.registerOn(new EntityRegistry(site)) + const record = fileStorage.getDataStore().getByPath(relativePath) + return record ? fileStorage.load(record.uuid) : null + } + + function stagedFiles() { + return fs.readdirSync(stagingRoot) + } + + test('an http(s) URL is fetched into staging and saved as a file entity', async () => { + stubNetwork({ 'https://example.org/img/chart.png': { body: PNG } }) + assert.equal(await importBuildFile(site, 'files/chart.png', 'https://example.org/img/chart.png', 0), true) + assert.deepEqual(fetched, ['https://example.org/img/chart.png']) + const entity = entityAt('files/chart.png') + assert.ok(entity, 'files.json records the downloaded file') + assert.equal(entity.getPath(), 'files/chart.png') + assert.equal(entity.isImage(), true) + assert.ok(fs.existsSync(path.join(site.siteDirectory, 'files', 'chart.png'))) + assert.deepEqual(stagedFiles(), [], 'nothing is left in staging') + }) + + test('a staged file and a URL in the same payload are both saved', async () => { + stubNetwork({ 'https://example.org/remote.png': { body: PNG } }) + const staged = path.join(stagingRoot, 'local.png') + await fs.writeFile(staged, PNG) + assert.equal(await importBuildFile(site, 'files/local.png', staged, 0), true) + assert.equal(await importBuildFile(site, 'files/remote.png', 'https://example.org/remote.png', 1), true) + assert.ok(entityAt('files/local.png'), 'the staged file is an entity') + assert.ok(entityAt('files/remote.png'), 'the downloaded file is an entity') + assert.deepEqual(stagedFiles(), []) + }) + + test('an upper-case scheme is still treated as a URL', async () => { + stubNetwork({ 'HTTPS://example.org/upper.png': { body: PNG } }) + assert.equal(await importBuildFile(site, 'files/upper.png', 'HTTPS://example.org/upper.png', 0), true) + assert.deepEqual(fetched, ['HTTPS://example.org/upper.png']) + assert.ok(entityAt('files/upper.png')) + }) + + test('a key without the files/ prefix is saved under files/', async () => { + stubNetwork({ 'https://example.org/bare.png': { body: PNG } }) + assert.equal(await importBuildFile(site, 'bare.png', 'https://example.org/bare.png', 0), true) + assert.ok(entityAt('files/bare.png')) + }) + + test('a URL without an extension is saved under the name its key gives', async () => { + // Plone serves images from paths like .../@@images/image + stubNetwork({ 'https://example.org/site/photo/@@images/image': { body: PNG } }) + assert.equal(await importBuildFile(site, 'files/photo.png', 'https://example.org/site/photo/@@images/image', 0), true) + const entity = entityAt('files/photo.png') + assert.ok(entity) + assert.equal(entity.getPath(), 'files/photo.png') + assert.equal(entity.getMimetype(), 'image/png') + }) + + test('a URL that cannot be fetched is skipped without failing the site', async () => { + // missing.png is unmapped, so it answers 404 + stubNetwork({ + 'https://example.org/empty.png': { body: Buffer.alloc(0) }, + 'https://example.org/down.png': new Error('socket hang up'), + }) + const sources = ['missing', 'empty', 'down'] + for (let i = 0; i < sources.length; i++) { + const name = `files/${sources[i]}.png` + assert.equal(await importBuildFile(site, name, `https://example.org/${sources[i]}.png`, i), true, name) + assert.equal(entityAt(name), null, name) + } + assert.equal(fetched.length, 3) + assert.deepEqual(stagedFiles(), []) + }) + + test('private, loopback and metadata addresses are refused without being contacted', async () => { + // the real safeFetch: its SSRF guard must stop these before any connection + let hits = 0 + const server = http.createServer(function (req, res) { + hits++ + res.end(PNG) + }) + await new Promise(function (resolve) { + server.listen(0, '127.0.0.1', resolve) + }) + const port = server.address().port + try { + const targets = [ + `http://127.0.0.1:${port}/secret.png`, + `http://localhost:${port}/secret.png`, + 'http://169.254.169.254/latest/meta-data/secret.png', + 'http://10.0.0.1/secret.png', + ] + for (let i = 0; i < targets.length; i++) { + const name = `files/secret-${i}.png` + assert.equal(await importBuildFile(site, name, targets[i], i), true, targets[i]) + assert.equal(entityAt(name), null, targets[i]) + } + assert.equal(hits, 0, 'the local server was never contacted') + assert.deepEqual(stagedFiles(), []) + } finally { + await new Promise(function (resolve) { + server.close(resolve) + }) + } + }) + + test('other schemes and files outside the staging root are still rejected', async () => { + stubNetwork({}) + const outside = path.join(tmpRoot, 'outside.png') + await fs.writeFile(outside, PNG) + const sources = [ + 'file:///etc/passwd', + 'gopher://example.org/x.png', + 'ftp://example.org/x.png', + '/etc/passwd', + outside, + 'relative/x.png', + '', + // the advisory's proof-of-concept payload shape + { tmp_name: 'http://169.254.169.254/latest/meta-data/iam/security-credentials/' }, + ] + for (let i = 0; i < sources.length; i++) { + assert.equal(await importBuildFile(site, 'files/x.png', sources[i], i), false, JSON.stringify(sources[i])) + } + assert.equal(fetched.length, 0, 'nothing was fetched') + assert.equal(entityAt('files/x.png'), null) + }) + + test('an entry with an unsafe name is rejected before anything is fetched', async () => { + stubNetwork({ 'https://example.org/x.png': { body: PNG } }) + assert.equal(await importBuildFile(site, 'files/../escape.png', 'https://example.org/x.png', 0), false) + assert.equal(await importBuildFile(site, 'files/shell.php', 'https://example.org/x.png', 1), false) + assert.equal(fetched.length, 0) + }) + + test('a download whose content does not match its extension is dropped', async () => { + stubNetwork({ 'https://example.org/fake.png': { body: Buffer.from('not an image') } }) + assert.equal(await importBuildFile(site, 'files/fake.png', 'https://example.org/fake.png', 0), true) + assert.equal(entityAt('files/fake.png'), null) + assert.equal(fs.existsSync(path.join(site.siteDirectory, 'files', 'fake.png')), false) + assert.deepEqual(stagedFiles(), [], 'the rejected download is removed from staging') + }) + + test('a download over the site upload limit is dropped', async () => { + // a real PNG of about 2MB: random pixels do not compress + const big = await sharp(crypto.randomBytes(800 * 800 * 3), { raw: { width: 800, height: 800, channels: 3 } }) + .png({ compressionLevel: 0 }) + .toBuffer() + assert.ok(big.length > 1024 * 1024 && big.length < 3 * 1024 * 1024) + stubNetwork({ 'https://example.org/big.png': { body: big } }) + const mediaSettings = path.join(HAXCMS.configDirectory, 'settings', 'media.json') + await fs.outputJson(mediaSettings, { maxUploadSizeMb: 1 }) + assert.equal(await importBuildFile(site, 'files/big.png', 'https://example.org/big.png', 0), true) + assert.equal(entityAt('files/big.png'), null, 'over the 1MB limit') + assert.deepEqual(stagedFiles(), [], 'the rejected download is removed from staging') + // the same file under a higher limit is saved, so the limit is what stopped it + await fs.outputJson(mediaSettings, { maxUploadSizeMb: 3 }) + assert.equal(await importBuildFile(site, 'files/big.png', 'https://example.org/big.png', 1), true) + assert.ok(entityAt('files/big.png'), 'under the 3MB limit') + }) + + test('a page that references a downloaded file is linked to its entity', async () => { + stubNetwork({ 'https://example.org/img/chart.png': { body: PNG } }) + const location = 'pages/intro/index.html' + await fs.outputFile(path.join(site.siteDirectory, location), '

Chart

') + const page = { id: 'intro', title: 'Intro', location: location, metadata: { files: [] } } + site.manifest.items.push(page) + await importBuildFile(site, 'files/chart.png', 'https://example.org/img/chart.png', 0) + await linkImportedPageFiles(site) + assert.deepEqual(page.metadata.files, [entityAt('files/chart.png').getUuid()]) + }) +}) diff --git a/test/unit/createSiteLinkImportedFiles.test.cjs b/test/unit/createSiteLinkImportedFiles.test.cjs new file mode 100644 index 00000000..9cedab76 --- /dev/null +++ b/test/unit/createSiteLinkImportedFiles.test.cjs @@ -0,0 +1,155 @@ +'use strict' + +// Unit tests for createSite's linkImportedPageFiles (#2912, #3043). +// +// createSite writes the imported pages before it ingests build.files, so the +// pages cannot reference the files by uuid as they are written. Once the +// files exist, linkImportedPageFiles points each page at the file entities +// its content references, reading identity from files.json through the +// Entity API — the same source the docx import and page saves use. +// +// Constraints honored: CommonJS (.cjs), require(), NO optional chaining, +// node:test + node:assert/strict. + +const { test, describe, beforeEach, afterEach } = require('node:test') +const assert = require('node:assert/strict') +const path = require('path') +const fs = require('fs-extra') +const os = require('os') + +const createSite = require('../../src/systemRoutes/v1/routes/createSite.js') +const { HAXCMS, HAXCMSSite } = require('../../src/lib/HAXCMS.js') +const HAXCMSFile = require('../../src/lib/HAXCMSFile.js') +const EntityRegistry = require('../../src/lib/EntityRegistry.js') +const FileStorage = require('../../src/lib/FileStorage.js') +const FilesDataStore = require('../../src/lib/FilesDataStore.js') + +const { linkImportedPageFiles } = createSite + +const PNG = Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=', + 'base64', +) + +describe('createSite linkImportedPageFiles — #2912', () => { + let tmpRoot + let site + let saves + let stagingDirectory + + beforeEach(async () => { + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), 'hax-test-')) + // build.files must be staged under the bulk-import root, as in production + stagingDirectory = path.join(HAXCMS.configDirectory, 'tmp', 'imports', 'test-' + path.basename(tmpRoot)) + await fs.ensureDir(stagingDirectory) + site = new HAXCMSSite() + site.name = 'testsite' + site.siteDirectory = path.join(tmpRoot, 'testsite') + await fs.ensureDir(path.join(site.siteDirectory, 'files')) + saves = 0 + site.manifest = { + metadata: { site: { name: 'testsite' } }, + items: [], + save: async function () { + saves++ + return true + }, + } + }) + + afterEach(async () => { + try { + await fs.remove(tmpRoot) + await fs.remove(stagingDirectory) + } catch (e) {} + }) + + // a page as createSite leaves it: written, with no file references yet + async function addPage(id, html) { + const location = `pages/${id}/index.html` + await fs.outputFile(path.join(site.siteDirectory, location), html) + const page = { id: id, title: id, location: location, metadata: { files: [] } } + site.manifest.items.push(page) + return page + } + + // ingest a file the way createSite's build.files loop does + async function ingest(name) { + const staged = path.join(stagingDirectory, name) + await fs.writeFile(staged, PNG) + const result = await new HAXCMSFile().save( + { name: name, tmp_name: staged, path: staged, 'bulk-import': true }, + site, + ) + assert.equal(Number(result.status), 200, JSON.stringify(result)) + } + + async function entityUuid(relativePath) { + const fileStorage = FileStorage.registerOn(new EntityRegistry(site)) + const uuid = await fileStorage.getDataStore().resolveUuidByPath(relativePath) + const entity = fileStorage.load(uuid) + assert.ok(entity, 'files.json has an entity for ' + relativePath) + return entity.getUuid() + } + + test('links each page to the file entities its content references', async () => { + await ingest('chart.png') + await ingest('photo.png') + const first = await addPage('first', '

') + const second = await addPage('second', '

Photo

chart

') + const linked = await linkImportedPageFiles(site) + assert.equal(linked, 2) + assert.deepEqual(first.metadata.files, [await entityUuid('files/chart.png')]) + assert.deepEqual(second.metadata.files, [await entityUuid('files/photo.png'), await entityUuid('files/chart.png')]) + }) + + test('an image shared by pages is one entity referenced by each page', async () => { + await ingest('shared.png') + const a = await addPage('a', '') + const b = await addPage('b', '') + await linkImportedPageFiles(site) + const uuid = await entityUuid('files/shared.png') + assert.deepEqual(a.metadata.files, [uuid]) + assert.deepEqual(b.metadata.files, [uuid]) + assert.equal(new FilesDataStore(site).getRecords().length, 1, 'one file entity, not one per page') + }) + + test('a reference repeated on a page is recorded once', async () => { + await ingest('twice.png') + const page = await addPage('twice', '') + await linkImportedPageFiles(site) + assert.deepEqual(page.metadata.files, [await entityUuid('files/twice.png')]) + }) + + test('saves the manifest once, after linking every page', async () => { + await ingest('one.png') + await addPage('p1', '') + await addPage('p2', '') + await addPage('p3', '

no files

') + await linkImportedPageFiles(site) + assert.equal(saves, 1) + }) + + test('leaves pages without file references, and the manifest, alone', async () => { + const page = await addPage('plain', '

Just text

') + const linked = await linkImportedPageFiles(site) + assert.equal(linked, 0) + assert.deepEqual(page.metadata.files, []) + assert.equal(saves, 0, 'nothing to save') + }) + + test('ignores references to files that were never ingested', async () => { + await ingest('real.png') + const page = await addPage('mixed', '') + await linkImportedPageFiles(site) + assert.deepEqual(page.metadata.files, [await entityUuid('files/real.png')]) + }) + + test('links a page that arrived without metadata', async () => { + await ingest('bare.png') + const page = await addPage('bare', '') + delete page.metadata + await linkImportedPageFiles(site) + assert.deepEqual(page.metadata, { files: [await entityUuid('files/bare.png')] }) + }) +}) diff --git a/test/unit/stageRemoteFile.test.cjs b/test/unit/stageRemoteFile.test.cjs new file mode 100644 index 00000000..9467892f --- /dev/null +++ b/test/unit/stageRemoteFile.test.cjs @@ -0,0 +1,161 @@ +'use strict' + +// Unit tests for src/lib/stageRemoteFile.js (#3060): the bulk-import staging +// helpers shared by the site importers and createSite. +// +// The network is stubbed at safeFetch except where the real SSRF guard is +// under test, and HAXCMS.configDirectory points at a temp dir so nothing is +// staged in the real config tree. +// +// Constraints honored: CommonJS (.cjs), require(), NO optional chaining, +// node:test + node:assert/strict. + +const { test, describe, beforeEach, afterEach } = require('node:test') +const assert = require('node:assert/strict') +const http = require('http') +const path = require('path') +const fs = require('fs-extra') +const os = require('os') + +const safeFetchMod = require('../../src/lib/safeFetch.js') +const { HAXCMS } = require('../../src/lib/HAXCMS.js') +const { getBulkImportStagingRoot, stageRemoteFile } = require('../../src/lib/stageRemoteFile.js') + +describe('stageRemoteFile — #3060', () => { + const realSafeFetch = safeFetchMod.safeFetch + const realConfigDirectory = HAXCMS.configDirectory + let tmpRoot + let fetched + + beforeEach(async () => { + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), 'hax-test-')) + HAXCMS.configDirectory = path.join(tmpRoot, 'config') + fetched = [] + }) + + afterEach(async () => { + safeFetchMod.safeFetch = realSafeFetch + HAXCMS.configDirectory = realConfigDirectory + try { + await fs.remove(tmpRoot) + } catch (e) {} + }) + + // answer every request with { status, body }, or throw the given Error + function stubNetwork(answer) { + safeFetchMod.safeFetch = async function (url) { + fetched.push(url) + if (answer instanceof Error) { + throw answer + } + const status = answer.status || 200 + return { + ok: status >= 200 && status < 300, + status: status, + headers: { + get: function () { + return null + }, + }, + arrayBuffer: async function () { + return answer.body || Buffer.alloc(0) + }, + } + } + } + + test('getBulkImportStagingRoot creates the import directory under the config tree', () => { + const expected = path.join(HAXCMS.configDirectory, 'tmp', 'imports') + assert.equal(fs.existsSync(expected), false) + assert.equal(getBulkImportStagingRoot(), expected) + assert.ok(fs.statSync(expected).isDirectory()) + // and hands back the same directory once it exists + assert.equal(getBulkImportStagingRoot(), expected) + }) + + test('a fetched file is staged under the root with the extension of its key', async () => { + stubNetwork({ body: Buffer.from('PNGBYTES') }) + const root = getBulkImportStagingRoot() + const staged = await stageRemoteFile('https://example.org/img/chart.png?v=2', root, 7, 'files/chart.png') + assert.deepEqual(fetched, ['https://example.org/img/chart.png?v=2'], 'the URL is fetched as given') + assert.equal(path.dirname(staged), root) + assert.match(path.basename(staged), /^haximp-\d+-7-\d+\.png$/) + assert.equal(fs.readFileSync(staged, 'utf8'), 'PNGBYTES') + }) + + test('each download gets its own staged file', async () => { + stubNetwork({ body: Buffer.from('X') }) + const root = getBulkImportStagingRoot() + const first = await stageRemoteFile('https://example.org/a.png', root, 0, 'a.png') + const second = await stageRemoteFile('https://example.org/b.png', root, 1, 'b.png') + assert.notEqual(first, second) + assert.equal(fs.readdirSync(root).length, 2) + }) + + test('getBulkImportStagingRoot still names the directory when it cannot be created', async () => { + // a file where the config tree should be makes the directory impossible + await fs.writeFile(path.join(tmpRoot, 'blocker'), '') + HAXCMS.configDirectory = path.join(tmpRoot, 'blocker', 'config') + const root = getBulkImportStagingRoot() + assert.equal(root, path.join(tmpRoot, 'blocker', 'config', 'tmp', 'imports')) + assert.equal(fs.existsSync(root), false) + // and a download into it declines rather than throwing + stubNetwork({ body: Buffer.from('X') }) + assert.equal(await stageRemoteFile('https://example.org/x.png', root, 0, 'x.png'), null) + }) + + test('a key without an extension is staged without one', async () => { + stubNetwork({ body: Buffer.from('TEXT') }) + const root = getBulkImportStagingRoot() + assert.equal(path.extname(await stageRemoteFile('https://example.org/README', root, 0, 'files/README')), '') + assert.equal(path.extname(await stageRemoteFile('https://example.org/README', root, 1)), '', 'or no key at all') + }) + + test('declines error responses, empty bodies and network failures', async () => { + const root = getBulkImportStagingRoot() + const answers = [{ status: 404 }, { status: 500 }, { body: Buffer.alloc(0) }, new Error('socket hang up')] + for (let i = 0; i < answers.length; i++) { + stubNetwork(answers[i]) + assert.equal(await stageRemoteFile('https://example.org/x.png', root, i, 'x.png'), null, 'answer ' + i) + } + assert.equal(fetched.length, 4) + assert.deepEqual(fs.readdirSync(root), [], 'nothing was staged') + }) + + test('declines when the staging root cannot be written', async () => { + stubNetwork({ body: Buffer.from('X') }) + const missing = path.join(tmpRoot, 'no-such-dir') + assert.equal(await stageRemoteFile('https://example.org/x.png', missing, 0, 'x.png'), null) + assert.equal(fs.existsSync(missing), false) + }) + + test('private, loopback and metadata addresses are refused without being contacted', async () => { + // the real safeFetch: its SSRF guard must stop these before any connection + let hits = 0 + const server = http.createServer(function (req, res) { + hits++ + res.end('secret') + }) + await new Promise(function (resolve) { + server.listen(0, '127.0.0.1', resolve) + }) + const port = server.address().port + const root = getBulkImportStagingRoot() + try { + const targets = [ + `http://127.0.0.1:${port}/s.png`, + `http://localhost:${port}/s.png`, + 'http://169.254.169.254/latest/meta-data/s.png', + ] + for (let i = 0; i < targets.length; i++) { + assert.equal(await stageRemoteFile(targets[i], root, i, 's.png'), null, targets[i]) + } + assert.equal(hits, 0, 'the local server was never contacted') + assert.deepEqual(fs.readdirSync(root), []) + } finally { + await new Promise(function (resolve) { + server.close(resolve) + }) + } + }) +})