diff --git a/CONVERSIONS.md b/CONVERSIONS.md
index e69de29b..357a27c6 100644
--- a/CONVERSIONS.md
+++ b/CONVERSIONS.md
@@ -0,0 +1,41 @@
+# macdoc Conversion Matrix
+
+Status legend:
+- **implemented** — merged / available in repo
+- **active** — issue open and implementation in flight
+- **planned** — queue item, no open issue yet
+- **research** — needs protocol / package design before implementation
+
+## Current matrix
+
+| Source | Target | Package | Status | Notes |
+|--------|--------|---------|--------|-------|
+| Word (.docx) | Markdown | `word-to-md-swift` | implemented | Layer 3 converter |
+| HTML | Markdown | `html-to-md-swift` | active | Issue #13, SwiftSoup-based streaming emitter |
+| PDF | LaTeX | `pdf-to-latex-swift` | implemented | Phase 1 + Phase 2 pipeline |
+| BibLaTeX (.bib) | APA HTML | `apa-bib-to-html-swift` | implemented | style-aware renderer |
+| BibLaTeX (.bib) | APA Markdown | `apa-bib-to-md-swift` | implemented | style-aware renderer |
+| BibLaTeX (.bib) | APA JSON | `apa-bib-to-json-swift` | implemented | pre-rendered HTML + anchors |
+| PDF | Markdown | `pdf-to-md-swift` | planned | direct path, avoid hub loss through LaTeX |
+| Markdown | HTML | `md-to-html-swift` | planned | reverse pair promoted after `html-to-md-swift` |
+| Word (.docx) | HTML | `word-to-html-swift` | planned | direct path preserves Word semantics better than hub conversion |
+| HTML | Word (.docx) | `html-to-word-swift` | planned | useful reverse path after `word-to-html-swift` |
+| Markdown | Word (.docx) | `md-to-word-swift` | research | binary target + protocol shape need design |
+
+## Priority Queue
+
+| Priority | Converter | Status | Why now |
+|---------:|-----------|--------|---------|
+| P0 | `html-to-md-swift` | active (#13) | explicit future converter in `docs/modular-architecture.md`; fits existing `DocumentConverter` + `StreamingOutput` shape cleanly |
+| P0 | `md-to-html-swift` | planned | reverse path auto-promoted after `html-to-md-swift` |
+| P1 | `pdf-to-md-swift` | planned | direct markdown export is a natural companion to existing PDF parsing stack |
+| P1 | `word-to-html-swift` | planned | direct conversion avoids Markdown hub loss for rich Word semantics |
+| P2 | `html-to-word-swift` | planned | reverse path once Word↔HTML design stabilizes |
+| P3 | `md-to-word-swift` | research | requires target-binary converter story beyond current text-streaming protocol |
+
+## Rules
+
+- Open **one issue per converter** before writing code.
+- New forward converter implies the reverse path is reconsidered immediately; if the reverse path is text-targeted and architecturally straightforward, promote it to **P0**.
+- Prefer direct source→target converters over hub-based routing.
+- Keep Layer 3 packages independent: source format + target format + `doc-converter-swift`, no converter-to-converter imports.
diff --git a/Package.swift b/Package.swift
index 4afece72..5daf14de 100644
--- a/Package.swift
+++ b/Package.swift
@@ -15,6 +15,7 @@ let package = Package(
.package(url: "https://github.com/PsychQuant/markdown-swift.git", from: "0.1.0"),
.package(url: "https://github.com/PsychQuant/marker-swift.git", from: "0.1.0"),
.package(name: "pdf-to-latex-swift", path: "packages/pdf-to-latex-swift"),
+ .package(name: "HTMLToMDSwift", path: "packages/html-to-md-swift"),
.package(name: "APABibToHTML", path: "packages/apa-bib-to-html-swift"),
.package(name: "APABibToJSON", path: "packages/apa-bib-to-json-swift"),
.package(name: "APABibToMD", path: "packages/apa-bib-to-md-swift"),
@@ -34,6 +35,7 @@ let package = Package(
dependencies: [
.product(name: "DocConverterSwift", package: "doc-converter-swift"),
.product(name: "WordToMDSwift", package: "word-to-md-swift"),
+ .product(name: "HTMLToMDSwift", package: "HTMLToMDSwift"),
"MarkerWordConverter",
.product(name: "PDFToLaTeXCore", package: "pdf-to-latex-swift"),
.product(name: "APABibToHTML", package: "APABibToHTML"),
diff --git a/Sources/MacDocCLI/MacDoc+HTML.swift b/Sources/MacDocCLI/MacDoc+HTML.swift
new file mode 100644
index 00000000..d1eb6cfd
--- /dev/null
+++ b/Sources/MacDocCLI/MacDoc+HTML.swift
@@ -0,0 +1,52 @@
+import ArgumentParser
+import Foundation
+import DocConverterSwift
+import HTMLToMDSwift
+
+// MARK: - HTML 子命令群
+extension MacDoc {
+ struct HTML: ParsableCommand {
+ static let configuration = CommandConfiguration(
+ commandName: "html",
+ abstract: "轉換 HTML 到 Markdown"
+ )
+
+ @Argument(help: "輸入 .html / .htm 檔案路徑")
+ var input: String
+
+ @Option(name: [.short, .long], help: "輸出 .md 檔案路徑(預設為 stdout)")
+ var output: String?
+
+ @Flag(name: .long, help: "包含 HTML
與來源檔名作為 YAML frontmatter")
+ var frontmatter: Bool = false
+
+ @Flag(name: .long, help: "將
轉為 Markdown hard break")
+ var hardBreaks: Bool = false
+
+ @Flag(name: .long, help: "保留 /// 為 raw HTML extension")
+ var htmlExtensions: Bool = false
+
+ mutating func run() throws {
+ let inputURL = URL(fileURLWithPath: input)
+ guard FileManager.default.fileExists(atPath: inputURL.path) else {
+ throw ValidationError("找不到輸入檔案: \(input)")
+ }
+
+ let options = ConversionOptions(
+ includeFrontmatter: frontmatter,
+ hardLineBreaks: hardBreaks,
+ tableStyle: .pipe,
+ headingStyle: .atx,
+ useHTMLExtensions: htmlExtensions
+ )
+
+ let converter = HTMLConverter()
+ if let outputPath = output {
+ let outputURL = URL(fileURLWithPath: outputPath)
+ try converter.convertToFile(input: inputURL, output: outputURL, options: options)
+ } else {
+ try converter.convertToStdout(input: inputURL, options: options)
+ }
+ }
+ }
+}
diff --git a/Sources/MacDocCLI/MacDoc.swift b/Sources/MacDocCLI/MacDoc.swift
index e397ff6d..00789872 100644
--- a/Sources/MacDocCLI/MacDoc.swift
+++ b/Sources/MacDocCLI/MacDoc.swift
@@ -11,7 +11,7 @@ struct MacDoc: AsyncParsableCommand {
commandName: "macdoc",
abstract: "原生 macOS 文件處理工具",
version: "0.3.0",
- subcommands: [Word.self, PDF.self, Bib.self, Config.self]
+ subcommands: [Word.self, HTML.self, PDF.self, Bib.self, Config.self]
)
}
diff --git a/packages/html-to-md-swift/Package.swift b/packages/html-to-md-swift/Package.swift
new file mode 100644
index 00000000..e8751bbd
--- /dev/null
+++ b/packages/html-to-md-swift/Package.swift
@@ -0,0 +1,33 @@
+// swift-tools-version: 5.9
+import PackageDescription
+
+let package = Package(
+ name: "HTMLToMDSwift",
+ platforms: [.macOS(.v14)],
+ products: [
+ .library(name: "HTMLToMDSwift", targets: ["HTMLToMDSwift"]),
+ ],
+ dependencies: [
+ .package(url: "https://github.com/PsychQuant/doc-converter-swift.git", from: "0.3.0"),
+ .package(url: "https://github.com/PsychQuant/markdown-swift.git", from: "0.1.0"),
+ .package(url: "https://github.com/scinfu/SwiftSoup.git", from: "2.7.4"),
+ ],
+ targets: [
+ .target(
+ name: "HTMLToMDSwift",
+ dependencies: [
+ .product(name: "DocConverterSwift", package: "doc-converter-swift"),
+ .product(name: "MarkdownSwift", package: "markdown-swift"),
+ .product(name: "SwiftSoup", package: "SwiftSoup"),
+ ]
+ ),
+ .executableTarget(
+ name: "HTMLToMDSwiftSelfTest",
+ dependencies: [
+ "HTMLToMDSwift",
+ .product(name: "DocConverterSwift", package: "doc-converter-swift"),
+ ],
+ path: "Tests/HTMLToMDSwiftSelfTest"
+ ),
+ ]
+)
diff --git a/packages/html-to-md-swift/README.md b/packages/html-to-md-swift/README.md
new file mode 100644
index 00000000..2e97cd2b
--- /dev/null
+++ b/packages/html-to-md-swift/README.md
@@ -0,0 +1,47 @@
+# HTMLToMDSwift
+
+Native macOS HTML → Markdown converter written in Swift.
+Streaming conversion, no Markdown AST.
+
+## Architecture
+
+- **Layer 3 converter** in the macdoc ecosystem
+- Implements `DocumentConverter` from `doc-converter-swift`
+- Uses `markdown-swift` for Markdown-safe formatting
+- Uses `SwiftSoup` for HTML parsing
+
+## Current coverage
+
+### Block elements
+- `h1` ... `h6`
+- `p`
+- `ul` / `ol` / `li`
+- `blockquote`
+- `pre > code`
+- `hr`
+- `table`
+- wrapper blocks like `div`, `section`, `article`
+
+### Inline elements
+- `strong`, `b`
+- `em`, `i`
+- `del`, `s`, `strike`
+- `code`
+- `a[href]`
+- `img[src]`
+- `br`
+- optional raw HTML preservation for `u`, `sup`, `sub`, `mark`
+
+## Usage
+
+```swift
+import HTMLToMDSwift
+import DocConverterSwift
+
+let converter = HTMLConverter()
+let markdown = try converter.convertToString(input: htmlURL)
+```
+
+## Design notes
+
+The parser necessarily builds an HTML DOM through SwiftSoup, but Markdown output is emitted in document order and streamed through `StreamingOutput`. The converter does not build an intermediate Markdown tree.
diff --git a/packages/html-to-md-swift/Sources/HTMLToMDSwift/HTMLConverter.swift b/packages/html-to-md-swift/Sources/HTMLToMDSwift/HTMLConverter.swift
new file mode 100644
index 00000000..94a889a6
--- /dev/null
+++ b/packages/html-to-md-swift/Sources/HTMLToMDSwift/HTMLConverter.swift
@@ -0,0 +1,413 @@
+import Foundation
+import DocConverterSwift
+import MarkdownSwift
+import SwiftSoup
+
+public struct HTMLConverter: DocumentConverter {
+ public static let sourceFormat = "html"
+
+ public init() {}
+
+ public func convert(
+ input: URL,
+ output: inout W,
+ options: ConversionOptions
+ ) throws {
+ let html = try String(contentsOf: input, encoding: .utf8)
+ let document = try SwiftSoup.parse(html, input.deletingLastPathComponent().absoluteString)
+
+ if options.includeFrontmatter {
+ try emitFrontmatter(document: document, source: input, output: &output)
+ }
+
+ if let body = document.body() {
+ try emitBlockNodes(body.getChildNodes(), options: options, output: &output)
+ } else {
+ try emitBlockNodes(document.getChildNodes(), options: options, output: &output)
+ }
+ }
+
+ private func emitFrontmatter(
+ document: Document,
+ source: URL,
+ output: inout W
+ ) throws {
+ try output.writeLine("---")
+ let title = try document.title().trimmingCharacters(in: .whitespacesAndNewlines)
+ if !title.isEmpty {
+ try output.writeLine("title: \"\(escapeYAML(title))\"")
+ }
+ try output.writeLine("source: \"\(escapeYAML(source.lastPathComponent))\"")
+ try output.writeLine("format: \"html\"")
+ try output.writeLine("---")
+ try output.writeBlankLine()
+ }
+
+ private func emitBlockNodes(
+ _ nodes: [Node],
+ options: ConversionOptions,
+ output: inout W
+ ) throws {
+ for node in nodes {
+ try emitBlock(node, options: options, output: &output)
+ }
+ }
+
+ private func emitBlock(
+ _ node: Node,
+ options: ConversionOptions,
+ output: inout W
+ ) throws {
+ if let textNode = node as? TextNode {
+ let text = normalizeInlineWhitespace(textNode.getWholeText())
+ guard !text.isEmpty else { return }
+ try output.writeLine(MarkdownEscaping.escape(text, context: .paragraph))
+ try output.writeBlankLine()
+ return
+ }
+
+ guard let element = node as? Element else { return }
+ let tag = element.tagName().lowercased()
+
+ if ignoredTags.contains(tag) {
+ return
+ }
+
+ switch tag {
+ case "h1", "h2", "h3", "h4", "h5", "h6":
+ let level = Int(String(tag.dropFirst())) ?? 1
+ let text = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options))
+ guard !text.isEmpty else { return }
+ try output.writeLine("\(String(repeating: "#", count: max(1, min(level, 6)))) \(text)")
+ try output.writeBlankLine()
+
+ case "p":
+ let text = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options))
+ guard !text.isEmpty else { return }
+ try output.writeLine(text)
+ try output.writeBlankLine()
+
+ case "ul":
+ try emitList(element, ordered: false, depth: 0, options: options, output: &output)
+ try output.writeBlankLine()
+
+ case "ol":
+ try emitList(element, ordered: true, depth: 0, options: options, output: &output)
+ try output.writeBlankLine()
+
+ case "blockquote":
+ try emitBlockquote(element, options: options, output: &output)
+
+ case "pre":
+ try emitCodeBlock(element, output: &output)
+
+ case "hr":
+ try output.writeLine("---")
+ try output.writeBlankLine()
+
+ case "table":
+ try emitTable(element, options: options, output: &output)
+
+ case "br":
+ try output.writeLine("")
+
+ default:
+ if containerTags.contains(tag) {
+ let childElements = element.children().array()
+ if childElements.contains(where: { blockTags.contains($0.tagName().lowercased()) }) {
+ try emitBlockNodes(element.getChildNodes(), options: options, output: &output)
+ } else {
+ let text = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options))
+ guard !text.isEmpty else { return }
+ try output.writeLine(text)
+ try output.writeBlankLine()
+ }
+ } else {
+ let text = trimMarkdownText(try renderInlineNode(element, options: options, preserveWhitespace: false))
+ guard !text.isEmpty else { return }
+ try output.writeLine(text)
+ try output.writeBlankLine()
+ }
+ }
+ }
+
+ private func emitList(
+ _ list: Element,
+ ordered: Bool,
+ depth: Int,
+ options: ConversionOptions,
+ output: inout W
+ ) throws {
+ let items = list.children().array().filter { $0.tagName().lowercased() == "li" }
+ for (index, item) in items.enumerated() {
+ let prefix = ordered ? "\(index + 1). " : "- "
+ let indent = String(repeating: " ", count: depth)
+ let text = trimMarkdownText(try renderListItemInline(item, options: options))
+ if !text.isEmpty {
+ try output.writeLine("\(indent)\(prefix)\(text)")
+ }
+
+ for child in item.children().array() {
+ let childTag = child.tagName().lowercased()
+ if childTag == "ul" {
+ try emitList(child, ordered: false, depth: depth + 1, options: options, output: &output)
+ } else if childTag == "ol" {
+ try emitList(child, ordered: true, depth: depth + 1, options: options, output: &output)
+ }
+ }
+ }
+ }
+
+ private func renderListItemInline(_ item: Element, options: ConversionOptions) throws -> String {
+ var chunks: [String] = []
+ for child in item.getChildNodes() {
+ if let element = child as? Element {
+ let tag = element.tagName().lowercased()
+ if tag == "ul" || tag == "ol" {
+ continue
+ }
+ if blockTags.contains(tag) && tag != "p" {
+ let rendered = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options))
+ if !rendered.isEmpty {
+ chunks.append(rendered)
+ }
+ continue
+ }
+ }
+
+ let rendered = try renderInlineNode(child, options: options, preserveWhitespace: false)
+ if !rendered.isEmpty {
+ chunks.append(rendered)
+ }
+ }
+ return trimMarkdownText(chunks.joined())
+ }
+
+ private func emitBlockquote(
+ _ element: Element,
+ options: ConversionOptions,
+ output: inout W
+ ) throws {
+ var nested = DocConverterSwift.StringOutput()
+ try emitBlockNodes(element.getChildNodes(), options: options, output: &nested)
+ let content = nested.content.trimmingCharacters(in: .whitespacesAndNewlines)
+ guard !content.isEmpty else { return }
+
+ for line in content.components(separatedBy: .newlines) {
+ if line.isEmpty {
+ try output.writeLine(">")
+ } else {
+ try output.writeLine("> \(line)")
+ }
+ }
+ try output.writeBlankLine()
+ }
+
+ private func emitCodeBlock(
+ _ pre: Element,
+ output: inout W
+ ) throws {
+ let codeElement = pre.children().array().first { $0.tagName().lowercased() == "code" }
+ let sourceElement = codeElement ?? pre
+ let code = trimTrailingNewlines(rawText(sourceElement, preserveWhitespace: true))
+ guard !code.isEmpty else { return }
+ let language = codeElement.flatMap { detectCodeLanguage(from: $0) } ?? detectCodeLanguage(from: pre)
+ if let language, !language.isEmpty {
+ try output.writeLine("```\(language)")
+ } else {
+ try output.writeLine("```")
+ }
+ try output.writeLine(code)
+ try output.writeLine("```")
+ try output.writeBlankLine()
+ }
+
+ private func emitTable(
+ _ table: Element,
+ options: ConversionOptions,
+ output: inout W
+ ) throws {
+ let rows = try table.select("tr").array()
+ guard !rows.isEmpty else { return }
+
+ let matrix: [[String]] = try rows.map { row in
+ let cells = row.children().array().filter {
+ let tag = $0.tagName().lowercased()
+ return tag == "th" || tag == "td"
+ }
+ return try cells.map { cell in
+ let rendered = try renderInlineNodes(cell.getChildNodes(), options: options)
+ return sanitizeTableCell(trimMarkdownText(rendered))
+ }
+ }.filter { !$0.isEmpty }
+
+ guard let header = matrix.first, !header.isEmpty else { return }
+ let columnCount = matrix.map(\.count).max() ?? header.count
+ let normalized = matrix.map { row in
+ row + Array(repeating: "", count: max(0, columnCount - row.count))
+ }
+
+ try output.writeLine("| \(normalized[0].joined(separator: " | ")) |")
+ try output.writeLine("|\(Array(repeating: "---", count: columnCount).joined(separator: "|"))|")
+ for row in normalized.dropFirst() {
+ try output.writeLine("| \(row.joined(separator: " | ")) |")
+ }
+ try output.writeBlankLine()
+ }
+
+ private func renderInline(_ element: Element, options: ConversionOptions) throws -> String {
+ try renderInlineNodes(element.getChildNodes(), options: options)
+ }
+
+ private func renderInlineNodes(
+ _ nodes: [Node],
+ options: ConversionOptions,
+ preserveWhitespace: Bool = false
+ ) throws -> String {
+ try nodes.map { try renderInlineNode($0, options: options, preserveWhitespace: preserveWhitespace) }.joined()
+ }
+
+ private func renderInlineNode(
+ _ node: Node,
+ options: ConversionOptions,
+ preserveWhitespace: Bool
+ ) throws -> String {
+ if let textNode = node as? TextNode {
+ let text = preserveWhitespace ? textNode.getWholeText() : normalizeInlineWhitespace(textNode.getWholeText())
+ return preserveWhitespace ? text : MarkdownEscaping.escape(text, context: .paragraph)
+ }
+
+ guard let element = node as? Element else { return "" }
+ let tag = element.tagName().lowercased()
+
+ switch tag {
+ case "strong", "b":
+ return MarkdownInline.bold(trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace)))
+
+ case "em", "i":
+ return MarkdownInline.italic(trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace)))
+
+ case "del", "s", "strike":
+ return MarkdownInline.strikethrough(trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace)))
+
+ case "code":
+ if let parent = element.parent(), parent.tagName().lowercased() == "pre" {
+ return rawText(element, preserveWhitespace: true)
+ }
+ return MarkdownInline.code(trimMarkdownText(rawText(element, preserveWhitespace: true)))
+
+ case "a":
+ let text = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace))
+ let href = try element.attr("href")
+ guard !href.isEmpty else { return text }
+ return MarkdownInline.link(text.isEmpty ? href : text, url: href)
+
+ case "img":
+ let src = try element.attr("src")
+ guard !src.isEmpty else { return "" }
+ let alt = try element.attr("alt")
+ let title = try element.attr("title")
+ return title.isEmpty
+ ? MarkdownInline.image(alt, url: src)
+ : MarkdownInline.image(alt, url: src, title: title)
+
+ case "br":
+ return options.hardLineBreaks ? MarkdownInline.hardBreak() : "\n"
+
+ case "u":
+ let inner = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace))
+ guard options.useHTMLExtensions else { return inner }
+ return MarkdownInline.rawHTML("\(inner)")
+
+ case "sup":
+ let inner = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace))
+ guard options.useHTMLExtensions else { return inner }
+ return MarkdownInline.rawHTML("\(inner)")
+
+ case "sub":
+ let inner = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace))
+ guard options.useHTMLExtensions else { return inner }
+ return MarkdownInline.rawHTML("\(inner)")
+
+ case "mark":
+ let inner = trimMarkdownText(try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace))
+ guard options.useHTMLExtensions else { return inner }
+ return MarkdownInline.rawHTML("\(inner)")
+
+ case "p", "span", "small", "big", "label", "div", "section", "article", "header", "footer", "main", "body":
+ return try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace)
+
+ case "script", "style", "noscript":
+ return ""
+
+ default:
+ return try renderInlineNodes(element.getChildNodes(), options: options, preserveWhitespace: preserveWhitespace)
+ }
+ }
+
+ private func rawText(_ node: Node, preserveWhitespace: Bool) -> String {
+ if let textNode = node as? TextNode {
+ return preserveWhitespace ? textNode.getWholeText() : normalizeInlineWhitespace(textNode.getWholeText())
+ }
+ guard let element = node as? Element else { return "" }
+ return element.getChildNodes().map { rawText($0, preserveWhitespace: preserveWhitespace) }.joined()
+ }
+
+ private func detectCodeLanguage(from element: Element) -> String? {
+ guard let className = try? element.className(), !className.isEmpty else {
+ return nil
+ }
+
+ for token in className.split(whereSeparator: \.isWhitespace) {
+ let value = String(token)
+ if value.hasPrefix("language-") {
+ return String(value.dropFirst("language-".count))
+ }
+ if value.hasPrefix("lang-") {
+ return String(value.dropFirst("lang-".count))
+ }
+ }
+ return nil
+ }
+
+ private func normalizeInlineWhitespace(_ text: String) -> String {
+ let collapsed = text.replacingOccurrences(of: #"\s+"#, with: " ", options: .regularExpression)
+ return collapsed
+ }
+
+ private func trimMarkdownText(_ text: String) -> String {
+ text.trimmingCharacters(in: .whitespacesAndNewlines)
+ }
+
+ private func sanitizeTableCell(_ text: String) -> String {
+ text
+ .replacingOccurrences(of: "|", with: "\\|")
+ .replacingOccurrences(of: "\n", with: " ")
+ }
+
+ private func trimTrailingNewlines(_ text: String) -> String {
+ var result = text
+ while result.hasSuffix("\n") || result.hasSuffix("\r") {
+ result.removeLast()
+ }
+ return result
+ }
+
+ private func escapeYAML(_ text: String) -> String {
+ text
+ .replacingOccurrences(of: "\\", with: "\\\\")
+ .replacingOccurrences(of: "\"", with: "\\\"")
+ }
+
+ private let ignoredTags: Set = ["script", "style", "noscript", "head"]
+
+ private let containerTags: Set = [
+ "body", "main", "article", "section", "div", "header", "footer", "aside", "nav"
+ ]
+
+ private let blockTags: Set = [
+ "address", "article", "aside", "blockquote", "details", "dialog", "div", "dl", "fieldset",
+ "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", "h6", "header",
+ "hr", "li", "main", "nav", "ol", "p", "pre", "section", "table", "ul"
+ ]
+}
diff --git a/packages/html-to-md-swift/Tests/HTMLToMDSwiftSelfTest/main.swift b/packages/html-to-md-swift/Tests/HTMLToMDSwiftSelfTest/main.swift
new file mode 100644
index 00000000..11a2a259
--- /dev/null
+++ b/packages/html-to-md-swift/Tests/HTMLToMDSwiftSelfTest/main.swift
@@ -0,0 +1,186 @@
+import Foundation
+import DocConverterSwift
+import HTMLToMDSwift
+
+struct Failure: Error, CustomStringConvertible {
+ let description: String
+}
+
+let converter = HTMLConverter()
+var passed = 0
+
+func convert(
+ _ html: String,
+ options: ConversionOptions = .default,
+ fileName: String = "fixture.html"
+) throws -> String {
+ let tmpDir = FileManager.default.temporaryDirectory
+ .appendingPathComponent("html-to-md-tests-\(UUID().uuidString)", isDirectory: true)
+ try FileManager.default.createDirectory(at: tmpDir, withIntermediateDirectories: true)
+ let input = tmpDir.appendingPathComponent(fileName)
+ defer { try? FileManager.default.removeItem(at: tmpDir) }
+ try html.write(to: input, atomically: true, encoding: .utf8)
+ return try converter.convertToString(input: input, options: options)
+}
+
+func expect(_ condition: @autoclosure () -> Bool, _ message: String) throws {
+ if !condition() {
+ throw Failure(description: message)
+ }
+}
+
+func test(_ name: String, _ body: () throws -> Void) throws {
+ do {
+ try body()
+ passed += 1
+ print("✓ \(name)")
+ fflush(stdout)
+ } catch {
+ print("✗ \(name)")
+ fflush(stdout)
+ throw error
+ }
+}
+
+@main
+struct Runner {
+ static func main() {
+ do {
+ try runAll()
+ print("\nAll self-tests passed: \(passed)")
+ fflush(stdout)
+ } catch {
+ fputs("Self-test failed: \(error)\n", stderr)
+ exit(1)
+ }
+ }
+
+ static func runAll() throws {
+ try test("headingsAndParagraphs") {
+ let html = """
+
+ Main Title
+ Hello world.
+
+ """
+ let md = try convert(html)
+ try expect(md.contains("# Main Title"), "Got: \(md)")
+ try expect(md.contains("Hello **world**."), "Got: \(md)")
+ }
+
+ try test("inlineFormattingAndLinks") {
+ let html = """
+ Italic, bold, gone, x = 1, link.
+ """
+ let md = try convert(html)
+ try expect(md.contains("_Italic_"), "Got: \(md)")
+ try expect(md.contains("**bold**"), "Got: \(md)")
+ try expect(md.contains("~~gone~~"), "Got: \(md)")
+ try expect(md.contains("`x = 1`"), "Got: \(md)")
+ try expect(md.contains("[link](https://example.com)"), "Got: \(md)")
+ }
+
+ try test("unorderedAndOrderedLists") {
+ let html = """
+
+
+
+ - First
+ - Second
+
+
+ """
+ let md = try convert(html)
+ try expect(md.contains("- One"), "Got: \(md)")
+ try expect(md.contains("- Two"), "Got: \(md)")
+ try expect(md.contains(" - Nested"), "Got: \(md)")
+ try expect(md.contains("1. First"), "Got: \(md)")
+ try expect(md.contains("2. Second"), "Got: \(md)")
+ }
+
+ try test("codeBlockPreservesWhitespaceAndLanguage") {
+ let html = """
+ let x = 1
+ print(x)
+
+ """
+ let md = try convert(html)
+ try expect(md.contains("```swift"), "Got: \(md)")
+ try expect(md.contains("let x = 1\n print(x)"), "Got: \(md)")
+ }
+
+ try test("blockquote") {
+ let html = """
+
+ Quoted text.
+ Second line.
+
+ """
+ let md = try convert(html)
+ try expect(md.contains("> Quoted **text**."), "Got: \(md)")
+ try expect(md.contains("> Second line."), "Got: \(md)")
+ }
+
+ try test("table") {
+ let html = """
+
+ """
+ let md = try convert(html)
+ try expect(md.contains("| Name | Value |"), "Got: \(md)")
+ try expect(md.contains("| A | 1 |"), "Got: \(md)")
+ try expect(md.contains("| B | 2 |"), "Got: \(md)")
+ }
+
+ try test("imagesAndHorizontalRule") {
+ let html = """
+
+
+
+
+ """
+ let md = try convert(html)
+ try expect(md.contains(""), "Got: \(md)")
+ try expect(md.contains("---"), "Got: \(md)")
+ }
+
+ try test("hardBreakOption") {
+ let html = "Line 1
Line 2
"
+ var options = ConversionOptions.default
+ options.hardLineBreaks = true
+ let md = try convert(html, options: options)
+ try expect(md.contains("Line 1 \nLine 2"), "Got: \(md)")
+ }
+
+ try test("htmlExtensionsOptional") {
+ let html = "u 2 n hi
"
+ let plain = try convert(html)
+ try expect(!plain.contains(""), "Got: \(plain)")
+ try expect(plain.contains("u 2 n hi"), "Got: \(plain)")
+
+ var options = ConversionOptions.default
+ options.useHTMLExtensions = true
+ let extended = try convert(html, options: options)
+ try expect(extended.contains("u"), "Got: \(extended)")
+ try expect(extended.contains("2"), "Got: \(extended)")
+ try expect(extended.contains("n"), "Got: \(extended)")
+ try expect(extended.contains("hi"), "Got: \(extended)")
+ }
+
+ try test("frontmatterIncludesTitleAndSource") {
+ let html = "FixtureHello
"
+ var options = ConversionOptions.default
+ options.includeFrontmatter = true
+ let md = try convert(html, options: options, fileName: "doc.html")
+ try expect(md.contains("title: \"Fixture\""), "Got: \(md)")
+ try expect(md.contains("source: \"doc.html\""), "Got: \(md)")
+ try expect(md.contains("format: \"html\""), "Got: \(md)")
+ }
+ }
+}