diff --git a/src/backend/internal/webdav/webdav.ts b/src/backend/internal/webdav/webdav.ts index 5322871c..c6642319 100644 --- a/src/backend/internal/webdav/webdav.ts +++ b/src/backend/internal/webdav/webdav.ts @@ -1,12 +1,19 @@ -import { generateWebDavXml } from "../../pkg/utils" +import { generateWebDavXml, type WebDavItem } from "../../pkg/xml" -export interface WebDavItem { - name: string - size: number - isFolder: boolean - modified: string -} +export type { WebDavItem } -export function buildWebDavPropfindResponse(reqPath: string, items: WebDavItem[]): string { - return generateWebDavXml(reqPath, items) -} +/** + * 构造 PROPFIND 的 207 Multi-Status 响应体。 + * + * @param requestPath 被请求资源的请求路径(`URL.pathname`,已百分号编码)。 + * RFC 4918 §8.3 要求 与请求路径一致(§8.3.1 的示例允许相对引用或 + * 完整 URI 两种形态,本实现用相对引用),因此这里传请求路径而不是挂载前缀 —— + * 挂在子路径下(如 /list/dav)时也自动正确。 + * @param items 直接子项 + */ +export function buildWebDavPropfindResponse( + requestPath: string, + items: WebDavItem[], +): string { + return generateWebDavXml(requestPath, items) +} \ No newline at end of file diff --git a/src/backend/pkg/xml.ts b/src/backend/pkg/xml.ts index 565617bd..f881f93f 100644 --- a/src/backend/pkg/xml.ts +++ b/src/backend/pkg/xml.ts @@ -2,21 +2,110 @@ * XML generation utilities for OpenList protocols (WebDAV, S3). */ +/** + * 把请求路径里【URI 合法但在 XML 文本中有特殊含义】的字符改成百分号编码。 + * + * 背景:自身 href 取自 `URL.pathname`,它已经是合法的 URI —— 但合法不等于能 + * 直接写进 XML 文本。RFC 4918 §8.3.1 明确提醒过这点:「a legal URI may still + * contain characters that need to be escaped within XML character data, such + * as the ampersand character.」裸 `&` 会让 变成非法 XML,客户端解析 + * 整个 multistatus 失败。 + * + * 为什么用百分号编码而不是 XML 实体(`&`):本仓库自己的 WebDAV 驱动 + * (src/backend/drivers/webdav/util.ts 的 parseMultistatusXml)用正则取 href + * 文本,**不做 XML 实体反转义**,输出 `&` 会被它当成字面量。百分号编码 + * 同时满足两边:既是合法 URI,又无需反转义,decodeURIComponent 后还能还原成 + * 原始路径。 + * + * 为什么只处理这两个字符:WHATWG URL 在 pathname 里已经编码掉了 `< > " ` ` + * 与空格,`#` `?` 也不会出现在 pathname 中;而裸 `%` 必然是既有转义序列的 + * 一部分(用户输入的 `%` 会被编码成 `%25`),碰它会造成双重编码。因此这里 + * 只补 `&` 与 `'`,不做整条重新编码。 + */ +function encodeXmlSensitiveChars(requestPath: string): string { + return requestPath.replace(/[&']/g, (ch) => + ch === "&" ? "%26" : "%27", + ) +} + +/** + * 逐段百分号编码单个 WebDAV 路径段。 + * + * 为什么不能用 encodeDownloadPath:它按 Go 的 EncodePath 保留 `$&+,:;=@`, + * 其中 `&` 放进 XML 文本必须转义。这里用 encodeURIComponent,它会把 + * `& < > " ? # %` 空格与非 ASCII 全部编码。 + * + * 为什么必须逐段而不是整条 encodeURIComponent:整条会把分隔符 `/` 编码成 + * %2F,路径层级就没了。 + * + * 非法 UTF-16(孤立代理项)会让 encodeURIComponent 抛 URIError,因此 try + * 兜底,绝不因编码失败把整个 PROPFIND 打成 500。 + */ +function encodeDavPathSegment(segment: string): string { + try { + return encodeURIComponent(segment) + } catch { + return segment + } +} + +/** + * 集合 href 必须以 / 结尾(RFC 4918 §8.3:Identifiers for collections + * SHOULD end in a '/' character)。 + */ +function asCollectionHref(requestPath: string): string { + return requestPath.endsWith("/") ? requestPath : requestPath + "/" +} + +/** + * 构造子项 href:<自身 collection href> + 逐段编码的名称,目录再补一个 /。 + */ +export function buildDavChildHref( + collectionHref: string, + name: string, + isCollection: boolean, +): string { + return ( + encodeXmlSensitiveChars(asCollectionHref(collectionHref)) + + encodeDavPathSegment(name) + + (isCollection ? "/" : "") + ) +} + +export interface WebDavItem { + name: string + size: number + isFolder: boolean + modified: string +} + +/** + * 生成 PROPFIND 的 207 Multi-Status 响应体。 + * + * @param requestPath 被请求资源的**请求路径**(`URL.pathname`,已百分号编码)。 + * RFC 4918 §8.3 允许两种形态二选一:相对引用(path-absolute,客户端按 + * Request-URI 解析)或完整 absolute-URI;§8.3.1 的示例里 `'/sample/'` 与 + * `'http://example.com/sample/'` 都合法。本实现选**相对引用**,理由: + * 1. 与请求路径天然一致,满足 §8.3「MUST NOT have prefixes that do not + * match the Request-URI」; + * 2. 挂载在子路径(如 /list/dav)时自动正确,不必硬编码挂载点; + * 3. 不回显 Host 头,避免 Host 注入把攻击者域名写进客户端的解析结果。 + * §8.3 要求「同一个 Multi-Status 响应内所有 href 格式必须一致」,因此这里 + * 自身与子项统一用相对引用。 + * @param items 直接子项 + */ export function generateWebDavXml( - path: string, - items: Array<{ - name: string - size: number - isFolder: boolean - modified: string - }>, + requestPath: string, + items: WebDavItem[], ): string { + const selfHref = encodeXmlSensitiveChars(asCollectionHref(requestPath)) + let xml = `\n` xml += `\n` - // Current folder description + // 被请求的资源本身:PROPFIND 在本实现中总是列目录,故按集合输出 xml += ` \n` - xml += ` ${path}\n` + xml += ` ${selfHref}\n` xml += ` \n` xml += ` \n` xml += ` \n` @@ -26,9 +115,9 @@ export function generateWebDavXml( xml += ` \n` xml += ` \n` - // Children + // 直接子项 for (const item of items) { - const itemHref = `${path}${path.endsWith("/") ? "" : "/"}${encodeURIComponent(item.name)}` + const itemHref = buildDavChildHref(requestPath, item.name, item.isFolder) xml += ` \n` xml += ` ${itemHref}\n` xml += ` \n` @@ -52,4 +141,4 @@ export function generateWebDavXml( xml += `` return xml -} +} \ No newline at end of file diff --git a/src/backend/pkg/xml_webdav.test.ts b/src/backend/pkg/xml_webdav.test.ts new file mode 100644 index 00000000..ea7cefc3 --- /dev/null +++ b/src/backend/pkg/xml_webdav.test.ts @@ -0,0 +1,207 @@ +import test from "node:test" +import assert from "node:assert/strict" +import { buildDavChildHref, generateWebDavXml } from "./xml" + +/** + * WebDAV PROPFIND 的 href 回归测试(Issue #110,修复 Issue #104)。 + * + * RFC 4918 §8.3 规定 href 有两种合法形态:相对引用(path-absolute,客户端按 + * Request-URI 解析)或完整 absolute-URI,§8.3.1 的示例里 + * `'/sample/'` 与 `'http://example.com/sample/'` 均合法;本实现选相对引用。 + * 同时要求:同一个 Multi-Status 内格式一致、href 前缀须与 Request-URI 一致、 + * 集合标识符应以 '/' 结尾。 + * + * 因此测试的基准一律是**真实请求路径**(`URL.pathname`,已百分号编码), + * 而不是硬编码的挂载前缀。 + */ + +const hrefs = (xml: string): string[] => + [...xml.matchAll(/([^<]*)<\/d:href>/g)].map((m) => m[1]) + +/** 取出 XML 解析后的真实 href(反转义),用于断言语义而非字面量 */ +function decodedHrefs(xml: string): string[] { + return hrefs(xml).map((h) => + h + .replace(/&/g, "&") + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, '"') + .replace(/'/g, "'"), + ) +} + +const collectionFlags = (xml: string): boolean[] => + xml + .split("") + .slice(1) + .map((b) => //.test(b)) + +test("根目录:href 与请求路径一致且以 / 结尾", () => { + const xml = generateWebDavXml("/dav/", []) + assert.deepEqual(hrefs(xml), ["/dav/"]) +}) + +test("子目录:自身与子项都带挂载前缀", () => { + // Issue #104 的真实场景:PROPFIND /dav/wewe,Depth: 1 + const xml = generateWebDavXml("/dav/wewe", [ + { name: "WESSSDQ", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "70rop", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "a.txt", size: 12, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + assert.deepEqual(hrefs(xml), [ + "/dav/wewe/", + "/dav/wewe/WESSSDQ/", + "/dav/wewe/70rop/", + "/dav/wewe/a.txt", + ]) +}) + +test("请求不带尾斜杠时,集合 href 补上斜杠", () => { + assert.deepEqual(hrefs(generateWebDavXml("/dav/wewe", [])), ["/dav/wewe/"]) + assert.deepEqual(hrefs(generateWebDavXml("/dav/wewe/", [])), ["/dav/wewe/"]) +}) + +test("目录带尾斜杠、文件不带", () => { + const xml = generateWebDavXml("/dav/dir", [ + { name: "sub", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "file.bin", size: 3, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + assert.deepEqual(collectionFlags(xml), [true, true, false]) + const [, dir, file] = hrefs(xml) + assert.equal(dir, "/dav/dir/sub/") + assert.equal(file, "/dav/dir/file.bin") + assert.ok(!file.endsWith("/"), "文件 href 不应有尾斜杠") +}) + +/** + * 挂在子路径下(如 https://host/list/dav)时 href 仍须与请求路径一致。 + * 这是把 href 基准从「硬编码挂载前缀」改成「真实请求路径」的原因: + * 前者会输出 /dav/list/dav/x,与 Request-URI 对不上,客户端直接丢弃记录。 + */ +test("子路径挂载时 href 仍与请求路径一致", () => { + for (const reqPath of ["/list/dav/wewe", "/openlist/dav/wewe", "/dav/wewe"]) { + const xml = generateWebDavXml(reqPath, [ + { name: "sub", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + ]) + for (const h of hrefs(xml)) { + assert.ok( + h === reqPath || h.startsWith(reqPath.endsWith("/") ? reqPath : reqPath + "/"), + `href 必须以请求路径 ${reqPath} 为前缀,实际: ${h}`, + ) + } + } +}) + +test("子项名按段百分号编码(来自解码后的原文)", () => { + const xml = generateWebDavXml("/dav/wewe", [ + { name: "来自:TOKEN订阅", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "a b&c.txt", size: 1, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + { name: "x#y?z.txt", size: 1, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + const [, dir, amp, hash] = hrefs(xml) + assert.equal(dir, "/dav/wewe/%E6%9D%A5%E8%87%AA%EF%BC%9ATOKEN%E8%AE%A2%E9%98%85/") + assert.equal(amp, "/dav/wewe/a%20b%26c.txt") + // # 与 ? 会被当成 fragment / query,必须编码 + assert.equal(hash, "/dav/wewe/x%23y%3Fz.txt") +}) + +/** + * 自身 href 来自 URL.pathname(已百分号编码),不能再整条编码,否则双重编码。 + * §8.3.1 提醒「合法 URI 里仍可能需要在 XML 字符数据里转义的字符(如 &)」—— + * 本实现对 & 与 ' 做百分号编码而非 XML 实体,原因见 encodeXmlSensitiveChars + * 的注释(本仓库自己的 parseMultistatusXml 不做实体反转义)。 + */ +test("自身 href 不双重编码,且裸 & 被百分号编码", () => { + // URL.pathname 已把 %20 / 非 ASCII 编码好 + const encodedPathname = new URL("https://h/dav/a%20b/%E4%B8%AD?x=1").pathname + const xml = generateWebDavXml(encodedPathname, []) + const h = hrefs(xml)[0] + assert.equal(h, "/dav/a%20b/%E4%B8%AD/") // 没有出现 %2520 + assert.ok(!h.includes("%25"), "不应出现双重编码") + + // & 是 URI 合法字符,但放进 XML 文本必须处理 + const ampPath = new URL("https://h/dav/R&D").pathname + assert.equal(ampPath, "/dav/R&D", "前提:& 在 pathname 中保持原样") + const xml2 = generateWebDavXml(ampPath, []) + assert.ok(!/&/.test(xml2), "输出中不应出现裸 &") + assert.ok(!/&(amp|lt|gt|quot|apos);/.test(xml2), "不应使用 XML 实体") + assert.equal(hrefs(xml2)[0], "/dav/R%26D/") + // decodeURIComponent 后应还原为原始路径 + assert.equal(decodeURIComponent(hrefs(xml2)[0].replace("/dav", "")), "/R&D/") +}) + +test("apostrophe 也被百分号编码(与 & 同类)", () => { + const p = new URL("https://h/dav/it's").pathname + assert.equal(p, "/dav/it's") + const xml = generateWebDavXml(p, []) + assert.ok(!/\x27|'/.test(hrefs(xml)[0]), "href 不应含裸 apostrophe") + assert.equal(decodeURIComponent(hrefs(xml)[0].replace("/dav", "")), "/it's/") +}) + +/** + * 互操作:本仓库自己的 WebDAV 驱动用正则取 href 文本且不做 XML 实体反转义, + * 因此输出必须不含任何实体,否则驱动侧会把 & 当成字面量名字。 + */ +test("输出不含任何 XML 实体,可被朴素正则解析器直接消费", () => { + const xml = generateWebDavXml(new URL("https://h/dav/R&D").pathname, [ + { name: "a&b", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "c'd", size: 0, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + assert.ok(!xml.includes("&"), "不得出现 &") + assert.ok(!xml.includes("<"), "不得出现 <") + assert.ok(!xml.includes("'"), "不得出现 '") + + // 模拟 parseMultistatusXml 的取名逻辑:strip 尾斜杠后取末段并 decodeURIComponent + const derived = decodedHrefs(xml).map((h) => { + const clean = h.replace(/\/+$/, "") + return decodeURIComponent(clean.split("/").pop() || "") + }) + assert.deepEqual(derived, ["R&D", "a&b", "c'd"]) +}) + +test("自身 href 中的 < 不会提前截断标签", () => { + // < 在 pathname 里会被编码成 %3C;但若上层传入未编码的 <,也不能破坏 XML + const xml = generateWebDavXml("/dav/a%3Cb", []) + assert.equal(hrefs(xml).length, 1, "href 不应被 < 截断") + assert.equal(hrefs(xml)[0], "/dav/a%3Cb/") +}) + +test("路径分隔符不被编码成 %2F", () => { + assert.equal(hrefs(generateWebDavXml("/dav/a/b/c", []))[0], "/dav/a/b/c/") + assert.equal( + buildDavChildHref("/dav/a/b/", "c", false), + "/dav/a/b/c", + ) +}) + +test("孤立代理项不抛错(encodeURIComponent 会抛 URIError)", () => { + // 损坏的存储元数据可能给出非法 UTF-16;不能因此把整个 PROPFIND 打成 500 + const xml = generateWebDavXml("/dav/\uD800", [ + { name: "\uDC00", size: 1, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + assert.ok(xml.includes(" { + const xml = generateWebDavXml("/dav/dir", [ + { name: "sub", size: 0, isFolder: true, modified: "2026-10-04T00:00:00Z" }, + { name: "f.txt", size: 1, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + // 全部是相对引用(以 / 开头),不得混入 absolute-URI(http://…) + for (const h of hrefs(xml)) { + assert.ok(h.startsWith("/"), `href 应为相对引用: ${h}`) + assert.ok(!/^https?:/i.test(h)) + } +}) + +test("XML 结构完整且标签配平", () => { + const xml = generateWebDavXml("/dav/x", [ + { name: "f", size: 5, isFolder: false, modified: "2026-10-04T00:00:00Z" }, + ]) + assert.ok(xml.startsWith('')) + assert.ok(xml.includes('')) + assert.ok(xml.trimEnd().endsWith("")) + assert.equal((xml.match(//g) || []).length, 2) + assert.equal((xml.match(/<\/d:response>/g) || []).length, 2) + assert.equal((xml.match(//g) || []).length, 2) +}) \ No newline at end of file diff --git a/src/backend/server/webdav.ts b/src/backend/server/webdav.ts index eefbcff8..3ef7d4d8 100644 --- a/src/backend/server/webdav.ts +++ b/src/backend/server/webdav.ts @@ -71,11 +71,32 @@ async function webdavAuth(c: any): Promise { return null } -/** 从 URL pathname 中剥离 /dav 前缀,得到虚拟文件路径 */ +/** WebDAV 挂载前缀。必须与 index.ts 里 `app.route("/dav", webdavRouter)` 一致。 */ +const DAV_MOUNT = "/dav" + +/** + * 取请求路径(已百分号编码),作为 href 的基准。 + * + * RFC 4918 §8.3:href 要么是相对引用(客户端按 Request-URI 解析),要么是完整 + * URI;§8.3.1 的示例里两者都合法。用相对引用时 href 必须与 Request-URI 前缀 + * 一致,因此基准取真实请求路径而非硬编码的挂载前缀 —— 挂在子路径下 + * (/list/dav)时也自动正确,且不回显 Host 头。 + */ +function davRequestPath(c: any): string { + return new URL(c.req.url).pathname +} + +/** 从 URL pathname 中剥离挂载前缀,得到虚拟文件路径(仅用于存储层寻址) */ function davPathOf(c: any): string { - const pathname = new URL(c.req.url).pathname - let p = pathname.replace(/^\/dav/, "") + const pathname = davRequestPath(c) + let p = pathname + // 只在前缀真正匹配时才剥离:不能像 slice(DAV_MOUNT.length) 那样无条件截断, + // 否则挂在子路径下会把 /list/dav/x 截成 /t/dav/x。 + if (p === DAV_MOUNT || p.startsWith(DAV_MOUNT + "/")) { + p = p.slice(DAV_MOUNT.length) + } if (!p) p = "/" + if (!p.startsWith("/")) p = "/" + p try { return decodeURIComponent(p) } catch { @@ -131,13 +152,10 @@ webdavRouter.all("/*", async (c) => { isFolder: !!it.is_dir, modified: it.modified || new Date().toISOString(), })) - const href = - davPath === "/" - ? "/" - : davPath.endsWith("/") - ? davPath - : davPath + "/" - const xml = buildWebDavPropfindResponse(href, items) + // davPathOf 已剥掉挂载前缀,那是给存储层寻址用的;href 必须反映真实请求 + // 路径。少了挂载前缀,rclone / Windows 资源管理器会因「返回的 href 与请求 + // 路径对不上」丢弃全部记录(能下载但列不出目录)。见 Issue #104。 + const xml = buildWebDavPropfindResponse(davRequestPath(c), items) return c.body(xml, depth === "0" ? 207 : 207, { "Content-Type": "application/xml; charset=utf-8", }) @@ -186,9 +204,13 @@ webdavRouter.all("/*", async (c) => { const destRaw = c.req.header("Destination") || "" let dest = destRaw try { - dest = decodeURIComponent( - new URL(destRaw, c.req.url).pathname, - ).replace(/^\/dav/, "") + // 与 davPathOf 同样的前缀守卫:Destination 未挂载在 DAV_MOUNT 下时 + // 原样保留,不做无意义的前缀改写 + dest = decodeURIComponent(new URL(destRaw, c.req.url).pathname) + if (dest === DAV_MOUNT || dest.startsWith(DAV_MOUNT + "/")) { + dest = dest.slice(DAV_MOUNT.length) || "/" + } + if (!dest.startsWith("/")) dest = "/" + dest } catch {} const src = splitPath(davPath) const dst = splitPath(dest) @@ -201,9 +223,13 @@ webdavRouter.all("/*", async (c) => { const destRaw = c.req.header("Destination") || "" let dest = destRaw try { - dest = decodeURIComponent( - new URL(destRaw, c.req.url).pathname, - ).replace(/^\/dav/, "") + // 与 davPathOf 同样的前缀守卫:Destination 未挂载在 DAV_MOUNT 下时 + // 原样保留,不做无意义的前缀改写 + dest = decodeURIComponent(new URL(destRaw, c.req.url).pathname) + if (dest === DAV_MOUNT || dest.startsWith(DAV_MOUNT + "/")) { + dest = dest.slice(DAV_MOUNT.length) || "/" + } + if (!dest.startsWith("/")) dest = "/" + dest } catch {} const src = splitPath(davPath) const dst = splitPath(dest)