diff --git a/Directory.Packages.props b/Directory.Packages.props index 0f93475..d2833b9 100644 --- a/Directory.Packages.props +++ b/Directory.Packages.props @@ -42,6 +42,7 @@ + diff --git a/THIRD-PARTY-NOTICES.md b/THIRD-PARTY-NOTICES.md index 81eb342..abadc4d 100644 --- a/THIRD-PARTY-NOTICES.md +++ b/THIRD-PARTY-NOTICES.md @@ -19,7 +19,7 @@ Each component remains under its own license. | SkiaSharp | MIT | https://github.com/mono/SkiaSharp | | Svg.Skia (incl. Svg, ShimSkiaSharp, ExCSS; text shaping in `Filee.Engines/Vector/OutlinedTextPlayback.cs` adapted from it) | MIT | https://github.com/wieslawsoltes/Svg.Skia | | HarfBuzzSharp | MIT | https://github.com/mono/SkiaSharp | -| SharpCompress (zstd decompression for engine downloads) | MIT | https://github.com/adamhathcock/sharpcompress | +| SharpCompress (zstd decompression for engine downloads; RAR, 7z and TAR comics) | MIT | https://github.com/adamhathcock/sharpcompress | | Unhwp | MIT | https://github.com/iyulab/unhwp | | Markdig | BSD-2-Clause | https://github.com/xoofx/markdig | | ExcelNumberFormat | MIT | https://github.com/andersnm/ExcelNumberFormat | @@ -27,6 +27,7 @@ Each component remains under its own license. | PdfPig (PDF text, layout and pictures) | Apache-2.0 | https://github.com/UglyToad/PdfPig | | MimeKitLite (e-mail / MIME parsing) | MIT | https://github.com/jstedfast/MimeKit | | ACadSharp (DWG/DXF reading and writing; includes CSMath and CSUtilities) | MIT, Copyright (c) Albert Domenech | https://github.com/DomCR/ACadSharp | +| AngleSharp | MIT | https://github.com/AngleSharp/AngleSharp | | pypandoc-hwpx (Pandoc AST mapping and package layout in `Filee.Engines/Hwp/Hwpx`, incl. the `blank.hwpx` reference document) | MIT, Copyright (c) 2024 pypandoc-hwpx Contributors | https://github.com/msjang/pypandoc-hwpx | | Microsoft.Extensions.* | MIT | https://github.com/dotnet/runtime | | cu2qu (fontTools): the cubic-to-quadratic approach followed by `Filee.Engines/Fonts/Cff/CubicToQuadratic.cs` | Apache-2.0, Copyright 2016 Google Inc. | https://github.com/fonttools/fonttools | @@ -59,6 +60,7 @@ Downloaded from their official release pages when the user chooses to install th | Ghostscript (conda-forge build) | AGPL-3.0 | https://www.ghostscript.com, https://github.com/conda-forge/ghostscript-feedstock | | Microsoft Visual C++ Redistributable (for Ghostscript, conda-forge `vc14_runtime`) | Microsoft Visual C++ Redistributable license | https://github.com/conda-forge/vc-feedstock | | FFmpeg 9.0.2 (gyan.dev "full_build-shared" Windows build; ffmpeg.exe, ffprobe.exe and their libraries, incl. x264, x265, libvpx, LAME, Opus, Vorbis, Theora, OpenCORE AMR) | GPL-3.0-or-later (the build is configured with `--enable-gpl --enable-version3`); FFmpeg itself LGPL-2.1-or-later | https://ffmpeg.org, build: https://www.gyan.dev/ffmpeg/builds/ (source: https://github.com/GyanD/codexffmpeg) | +| calibre (ebook-convert) | GPL-3.0 | https://calibre-ebook.com | These programs run as separate processes. Their source code is available from the linked projects. diff --git a/build/fetch-engines.ps1 b/build/fetch-engines.ps1 index 387646f..b9f7f4a 100644 --- a/build/fetch-engines.ps1 +++ b/build/fetch-engines.ps1 @@ -16,6 +16,8 @@ ghostscript/ Ghostscript (AGPL-3.0, separate program) from conda-forge, EPS/PS <-> PDF, with the Microsoft C++ runtime DLLs (vcruntime/) copied next to gswin64c.exe ffmpeg/ FFmpeg (GPL-3.0 build, separate program), video and audio; bin/ffmpeg.exe + bin/ffprobe.exe + calibre/ calibre (GPL-3.0, separate program), extracted from the official MSI via an administrative + install: ebook-convert for LIT, LRF, PDB, ... and MOBI / AZW3 output Every download is pinned to a version and verified with SHA-256 (src/Filee.Engines/Infrastructure/engines.json, shared with the app). Downloads are cached in build/.cache. @@ -216,4 +218,20 @@ if ('ghostscript' -in $selected) { Copy-Item (Join-Path $Destination 'vcruntime\*.dll') $target -Force } +if ('calibre' -in $selected) { + $msi = Get-Engine 'calibre' + $tmp = Join-Path $cache 'calibre-admin' + Reset-Folder $tmp + Write-Host 'unpack calibre (administrative install, no system changes)' + $proc = Start-Process msiexec.exe -ArgumentList @('/a', "`"$msi`"", '/qn', "TARGETDIR=`"$tmp`"") -Wait -PassThru + if ($proc.ExitCode -ne 0) { throw "msiexec /a failed with exit code $($proc.ExitCode)." } + + $convert = Get-ChildItem $tmp -Recurse -Filter 'ebook-convert.exe' | Select-Object -First 1 + if (-not $convert) { throw 'ebook-convert.exe not found after extracting the MSI.' } + $target = Join-Path $Destination 'calibre' + Reset-Folder $target + Get-ChildItem $convert.DirectoryName -Force | Move-Item -Destination $target + Remove-Item $tmp -Recurse -Force +} + Write-Host "Engines ready in $Destination" diff --git a/docs/ENGINES.md b/docs/ENGINES.md index 9c62c11..90f4c88 100644 --- a/docs/ENGINES.md +++ b/docs/ENGINES.md @@ -15,10 +15,11 @@ Filee chooses engines automatically. Settings → *Engines* shows their status, | Spreadsheets (ExcelDataReader for XLS) | built in (library) | **XLSX, XLS, ODS, CSV, TSV → XLSX, ODS, CSV, TSV** (one CSV / TSV per sheet) | | Office Open XML | built in | **DOCM / DOTX / DOTM ↔ DOCX, XLSM / XLTX ↔ XLSX, PPTM / POTX / PPSX ↔ PPTX** (macros removed for macro-free types) | | Fonts | built in | **TTF, OTF, WOFF, WOFF2, EOT ↔ each other**; CFF (PostScript) outlines become TrueType for TTF and EOT | -| **HWPX writer** | built in | **DOCX (+ DOCM/DOTX/DOTM), XLSX (+ XLSM/XLTX), XLS, ODS, CSV, TSV, PPTX (+ PPTM/POTX/PPSX), PDF, TXT, Markdown → HWPX** (and with rhwp → PDF and images); HTML, ODT, RTF, reStructuredText, LaTeX → HWPX with Pandoc | +| **HWPX writer** | built in | **DOCX (+ DOCM/DOTX/DOTM), XLSX (+ XLSM/XLTX), XLS, ODS, CSV, TSV, PPTX (+ PPTM/POTX/PPSX), PDF, HTML, EPUB, MOBI/AZW3, FB2, HWPX, TXT, Markdown → HWPX** (and with rhwp → PDF and images); ODT, RTF, reStructuredText, LaTeX → HWPX with Pandoc | | **DOCX writer** | built in | **Markdown, TXT, XLSX, CSV, PPTX (+ variants), PDF → DOCX** | | PDF text | built in (PdfPig) | **PDF → TXT** (reading order, also two columns; no OCR) | | E-mail | built in (MimeKit) | **EML → HTML** (headers + body, inline pictures), **TXT**, **ZIP** (the attachments) | +| **E-books** | built in | **EPUB, MOBI/AZW/AZW3/PRC, FB2, HTMLZ, TXTZ → EPUB, HWPX (→ PDF), TXT, HTML, Markdown, FB2, HTMLZ, TXTZ**; HTML, Markdown, TXT, DOCX → EPUB/FB2/HTMLZ/TXTZ; HTML → TXT; comics CBZ/CBR/CB7/CBT/CBC → PDF, EPUB, CBZ; PDF → CBZ; AZW4 → PDF | | CAD (ACadSharp + built-in renderer) | built in (library) | **DWG ↔ DXF**; DWG/DXF → **PDF** (vector), **SVG**, PNG, JPG, WEBP, TIFF, BMP, GIF, ICO, AVIF | | rhwp | bundled with the installer (`engines/rhwp`) | HWP/HWPX → PDF, **HWP → HWPX, HWPX → HWP** | | Archives (7-Zip) | bundled with the installer (`engines/7zip`, ~2.5 MB) + built-in readers | ZIP, 7Z, RAR, TAR (+ GZ/BZ2/XZ/Z/7Z/LZ), CAB, ISO, DMG, … **→ folder, ZIP, 7Z, TAR, TAR.GZ/BZ2/XZ**; ALZ, EGG, lzip read in-process; "Compress into one archive" for any files | @@ -26,6 +27,7 @@ Filee chooses engines automatically. Settings → *Engines* shows their status, | Pandoc | **optional download** (~42 MB, 240 MB on disk) | Markdown ↔ DOCX/ODT/RTF, DOCX/ODT/HTML/RTF → Markdown, HTML ↔ DOCX/ODT, reStructuredText and LaTeX ↔ Markdown/HTML/DOCX/ODT (and → RTF, EPUB, TXT) | | Ghostscript | **optional download** (~20 MB, 31 MB on disk) | EPS/PS → PDF, PDF → EPS/PS, PostScript-based AI → PDF (images through PDF and PDFium) | | FFmpeg | **optional download** (~100 MB, 272 MB on disk) | **all video and audio**: video ↔ video, video → animated GIF or a still frame, audio extraction, audio ↔ audio, GIF → MP4/WEBM/MOV | +| Calibre | **optional download** (~216 MB, 660 MB on disk) | rare e-book formats: LIT, LRF, CHM, PDB, PML, RB, SNB, TCR, OEB → EPUB; EPUB → MOBI, AZW3, LIT, LRF, PDB, PML, RB, SNB, TCR | DOCX, XLSX, XLS, ODS and PPTX → PDF need neither Microsoft Office nor LibreOffice: they are read in-process, written as HWPX and rendered by rhwp. Anything the built-in readers understand (including PDF) is written as DOCX by the DOCX writer. @@ -51,8 +53,9 @@ they can be installed or removed later in Settings → *Engines*. A conversion t engine folder only when complete. Redirects to mirrors (even plain HTTP) are followed, because the hash decides. - Engines go to `%LOCALAPPDATA%\Filee\engines`: outside the app folder, so updates keep them, and inside Filee's install root, so uninstalling removes them. -- LibreOffice comes as an MSI and is unpacked with an administrative install (`msiexec /a`): files only, no - registry entries, no admin rights. Help, gallery and most dictionaries are removed afterwards (~500 MB). +- LibreOffice and calibre come as MSIs and are unpacked with an administrative install (`msiexec /a`): files only, + no registry entries, no admin rights. LibreOffice's help, gallery and most dictionaries are removed afterwards + (~500 MB). - Ghostscript comes from conda-forge (Artifex publishes only an NSIS installer): a `.conda` package is a zip with a zstd tarball, of which only `Library/bin` is unpacked (SharpCompress; `fetch-engines.ps1` uses Windows' `tar.exe`). Its fonts and resources are compiled into `gsdll64.dll`. The Microsoft C++ runtime it was built against @@ -117,9 +120,16 @@ Readers turn the source into a small document model (`Hwp/Hwpx/HwpxModel.cs`) an start numbers), task lists, quotes, tables with spans, links, images, footnotes and math. No Pandoc needed. - **PDF** is read with PdfPig (see *PDF reader* below); DOCM / DOTX / DOTM, XLSM / XLTX and PPTM / POTX / PPSX are read like DOCX, XLSX and PPTX. -- **HTML, ODT, RTF, reStructuredText, LaTeX** are parsed by Pandoc into its JSON AST (`PandocAstReader.cs`, following +- **HTML** is parsed with AngleSharp (`HtmlReader.cs`, `HtmlCss.cs`): headings, paragraphs, bold / italic / + underline / strike / sub / sup / code / mark, the basic inline CSS (colour, background, font weight, style and + size, text-align, text-indent, page breaks) and simple `tag` / `.class` rules of style sheets, nested lists with + start numbers and types, tables with spans / header rows / borders, pictures (local files and data: URIs; remote + pictures are skipped), links and anchors, quotes, `pre`, `hr` and figures. Scripts, styles, forms and `nav` are + skipped. The encoding comes from the BOM, the XML declaration or ``, else UTF-8 or the system + code page. No Pandoc needed. +- **ODT, RTF** are parsed by Pandoc into its JSON AST (`PandocAstReader.cs`, following [pypandoc-hwpx](https://github.com/msjang/pypandoc-hwpx)). -- Markdown and Pandoc keep structure only, so page setup comes from the built-in template (A4). +- Markdown, HTML and Pandoc keep structure only, so page setup comes from the built-in template (A4). Element order and attribute values follow files saved by 한글 where the schema and 한글 disagree, for example: @@ -343,6 +353,45 @@ OLE objects, charts, video, form controls, text art (its text is kept), arcs / p master pages, memos, character ratio, relative size and offset, paragraph borders, picture cropping and rotation. Password-protected (encrypted) HWPX and DRM-wrapped files stop with a clear error. +## E-books + +The built-in e-book engine (`src/Filee.Engines/Ebooks`) reads books into the same document model as the HWPX +writer, so every e-book also reaches HWPX, PDF and images. The document model is written back out by +`XhtmlWriter` (EPUB chapters, HTMLZ, single-page HTML), `PlainTextWriter`, `MarkdownWriter` and `Fb2Writer`. + +- **EPUB 2 / 3**: container.xml → OPF → spine; each chapter is read by the HTML reader and starts a new page, links + between chapters become links to bookmarks, pictures come from the package, title / authors / language / cover + from the metadata. +- **EPUB output** is EPUB 3 with an NCX for older readers: `mimetype` first and stored, a navigation document from + the headings, chapters split at level-1 headings and page breaks, one style sheet, pictures in formats every + reader shows (others become JPEG / PNG), a cover page for covers the content does not show. +- **MOBI / AZW / AZW3 / PRC** (`Ebooks/Mobi`, written from the MobileRead wiki's format description): PalmDB + records, MOBI header and EXTH metadata, PalmDOC (LZ77) and HUFF/CDIC compression, MOBI 6 `filepos` links and + `recindex` pictures, KF8 text rebuilt from the skeleton and fragment indexes with `kindle:pos` / `kindle:embed` / + `kindle:flow` references resolved, and plain PalmDOC (TEXtREAd) books. **AZW4** (Print Replica) gives back its PDF. +- **FB2**: nested sections become headings, poems / epigraphs / citations / tables are kept, note links become + footnotes, pictures come from the base64 binaries; the XML declaration's encoding (often windows-1251) is used. + FB2 output nests sections by heading level. +- **HTMLZ / TXTZ** (calibre's zipped formats) are read and written, with their `metadata.opf`. +- **Comics**: CBZ, CBR, CB7, CBT and CBC (a ZIP of CBZ files) are opened with SharpCompress; pages are sorted + naturally ("page2" before "page10"). → PDF has one page per picture sized like the picture (JPEGs are embedded + as they are), → EPUB is fixed-layout, → CBZ repacks. PDF → CBZ renders the pages with PDFium at the preset DPI. +- Books with DRM (Adobe, Apple, Kindle) are refused with a clear message; Filee does not remove DRM. KFX and + Topaz books are recognised and refused too. + +## Calibre + +calibre's `ebook-convert` (GPL-3.0) is an optional download for the formats Filee does not read or write itself: +LIT, LRF, CHM, PDB, PML, RB, SNB, TCR and OEB → EPUB, and EPUB → MOBI, AZW3, LIT, LRF, PDB, PML, RB, SNB and TCR. +EPUB is the hub, so for example FB2 → AZW3 is FB2 → EPUB (built in) → AZW3 (calibre). Its edges cost 20, so +built-in routes always win where they exist. + +- The pinned `calibre-64bit-.msi` from download.calibre-ebook.com is unpacked like LibreOffice; the folder + with `ebook-convert.exe` becomes `engines/calibre`. +- Each run gets its own configuration, cache and temp folders in the job's work directory + (`CALIBRE_CONFIG_DIRECTORY`, ...), so a calibre the user installed is never read or changed; messages are + English (`CALIBRE_OVERRIDE_LANG`) and progress comes from its "34% ..." lines. + ## HWP ↔ HWPX rhwp converts between the two 한글 formats without Hancom Office (`export-hwpx` and `convert`). Anything → HWP goes diff --git a/src/Filee.App/Assets/i18n/en.json b/src/Filee.App/Assets/i18n/en.json index 1bbbc44..c9a773a 100644 --- a/src/Filee.App/Assets/i18n/en.json +++ b/src/Filee.App/Assets/i18n/en.json @@ -322,6 +322,8 @@ "engines.package.ghostscript.description": "EPS, PS and older Illustrator files → PDF, PNG and other images, and PDF or SVG → EPS/PS. SVG and AI files saved with PDF compatibility convert without it.", "engines.package.ffmpeg.name": "FFmpeg (video and audio)", "engines.package.ffmpeg.description": "Video ↔ video (MP4, MOV, MKV, WEBM, AVI, WMV and more), video → animated GIF or a still image, sound from videos (→ MP3, M4A, WAV …), audio ↔ audio (MP3, M4A, FLAC, WAV, OGG, OPUS and more) and GIF → MP4, WEBM, MOV. Only needed for video and audio.", + "engines.package.calibre.name": "Calibre (Kindle and rare e-book formats)", + "engines.package.calibre.description": "Saves as MOBI and AZW3 for Kindle, and converts rare e-book formats: LIT, LRF, PDB, PML, RB, SNB and TCR, plus CHM and OEB input. EPUB, MOBI, AZW3, FB2 and comics (CBZ, CBR) are read without it.", "engines.package.size": "Download {0} · {1} on disk", "engines.progress_detail": "{0} of {1} · {2}/s · {3}", "engines.eta.estimating": "estimating time left…", diff --git a/src/Filee.App/Assets/i18n/ko.json b/src/Filee.App/Assets/i18n/ko.json index efff326..c0cd6bb 100644 --- a/src/Filee.App/Assets/i18n/ko.json +++ b/src/Filee.App/Assets/i18n/ko.json @@ -322,6 +322,8 @@ "engines.package.ghostscript.description": "EPS·PS 파일과 예전 일러스트레이터 파일을 PDF·PNG 등 이미지로, PDF·SVG를 EPS·PS로 변환해요. SVG와 PDF 호환으로 저장한 AI 파일은 없어도 변환돼요.", "engines.package.ffmpeg.name": "FFmpeg (동영상·오디오)", "engines.package.ffmpeg.description": "동영상 ↔ 동영상(MP4·MOV·MKV·WEBM·AVI·WMV 등), 동영상 → 움직이는 GIF·정지 이미지, 동영상에서 소리 추출(→ MP3·M4A·WAV 등), 오디오 ↔ 오디오(MP3·M4A·FLAC·WAV·OGG·OPUS 등), GIF → MP4·WEBM·MOV 변환을 추가해요. 동영상과 오디오에만 필요해요.", + "engines.package.calibre.name": "Calibre (킨들과 드문 전자책 형식)", + "engines.package.calibre.description": "킨들용 MOBI·AZW3로 저장하고, 드문 전자책 형식(LIT·LRF·PDB·PML·RB·SNB·TCR, CHM·OEB 읽기)을 변환해요. EPUB·MOBI·AZW3·FB2와 만화책(CBZ·CBR)은 없어도 읽을 수 있어요.", "engines.package.size": "다운로드 {0} · 설치 후 {1}", "engines.progress_detail": "{0} / {1} · {2}/s · {3}", "engines.eta.estimating": "남은 시간 계산 중…", diff --git a/src/Filee.App/Assets/i18n/zh-CN.json b/src/Filee.App/Assets/i18n/zh-CN.json index 19d91d1..974ea24 100644 --- a/src/Filee.App/Assets/i18n/zh-CN.json +++ b/src/Filee.App/Assets/i18n/zh-CN.json @@ -322,6 +322,8 @@ "engines.package.ghostscript.description": "EPS、PS 和旧版 Illustrator 文件 → PDF、PNG 等图片,以及 PDF 或 SVG → EPS/PS。SVG 和以 PDF 兼容方式保存的 AI 文件无需它即可转换。", "engines.package.ffmpeg.name": "FFmpeg(视频和音频)", "engines.package.ffmpeg.description": "支持视频 ↔ 视频(MP4、MOV、MKV、WEBM、AVI、WMV 等)、视频 → GIF 动图或静态图片、从视频提取声音(→ MP3、M4A、WAV 等)、音频 ↔ 音频(MP3、M4A、FLAC、WAV、OGG、OPUS 等)以及 GIF → MP4、WEBM、MOV。仅视频和音频转换需要它。", + "engines.package.calibre.name": "Calibre(Kindle 与少见电子书格式)", + "engines.package.calibre.description": "可保存为 Kindle 使用的 MOBI 和 AZW3,并转换少见的电子书格式:LIT、LRF、PDB、PML、RB、SNB、TCR,以及读取 CHM 和 OEB。EPUB、MOBI、AZW3、FB2 和漫画(CBZ、CBR)无需它即可读取。", "engines.package.size": "下载 {0} · 占用 {1}", "engines.progress_detail": "{0} / {1} · {2}/s · {3}", "engines.eta.estimating": "正在估算剩余时间…", diff --git a/src/Filee.App/ViewModels/EngineSetupViewModel.cs b/src/Filee.App/ViewModels/EngineSetupViewModel.cs index b30d627..453792b 100644 --- a/src/Filee.App/ViewModels/EngineSetupViewModel.cs +++ b/src/Filee.App/ViewModels/EngineSetupViewModel.cs @@ -21,8 +21,8 @@ public EngineSetupViewModel(EngineDownloadService downloads, ILocalizer loc) foreach (var package in Packages) { // Large downloads are offered, not pre-selected: LibreOffice is only needed for older formats (DOC, XLS, - // PPT, OpenDocument), FFmpeg (~100 MB) only for video and audio. Ghostscript is small but only needed for - // EPS / PostScript. + // PPT, OpenDocument), FFmpeg (~100 MB) only for video and audio, calibre (~230 MB) only for rare e-book + // formats and Kindle output. Ghostscript is small but only needed for EPS / PostScript. package.Selected = !package.IsInstalled && Filee.Engines.Infrastructure.EngineDownloads.DownloadSize(package.Package) < PreselectLimit && package.Package.Id != "ghostscript"; diff --git a/src/Filee.Engines/Cad/CadConverter.cs b/src/Filee.Engines/Cad/CadConverter.cs index d0a45ff..c784adb 100644 --- a/src/Filee.Engines/Cad/CadConverter.cs +++ b/src/Filee.Engines/Cad/CadConverter.cs @@ -2,6 +2,7 @@ // DXF and DWG; the drawing is rendered with SkiaSharp to PDF (vector), SVG and raster images. using Filee.Core.Conversion; +using Filee.Engines.Infrastructure; using Filee.Engines.Magick; using ImageMagick; using Microsoft.Extensions.Logging; @@ -41,7 +42,7 @@ from target in (string[])["pdf", "svg", .. ImageEncoder.Writable] select new ConversionEdge(source, target), ]; - public EngineStatus GetStatus() => EngineStatus.Available("ACadSharp: DWG R13–2018+, DXF"); + public EngineStatus GetStatus() => EngineStatus.Available("ACadSharp: DWG R13–2018+, DXF", $"{EngineVersions.BuiltIn} · {EngineVersions.Library("ACadSharp", typeof(ACadSharp.CadDocument))}"); public Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) => Task.Run>(() => diff --git a/src/Filee.Engines/Ebooks/Book.cs b/src/Filee.Engines/Ebooks/Book.cs new file mode 100644 index 0000000..18e1282 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Book.cs @@ -0,0 +1,160 @@ +// An e-book in memory: the HWPX document model (content) plus the metadata e-book formats carry (title, authors, +// language, cover). Every e-book reader produces a Book and every e-book writer consumes one. + +using System.IO.Compression; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +/// Title, authors, language and cover of a book. +internal sealed class BookMetadata +{ + public string? Title { get; set; } + public List Authors { get; } = []; + + /// BCP 47 language tag ("ko", "en-US"), if known. + public string? Language { get; set; } + public string? Description { get; set; } + public string? Publisher { get; set; } + + /// Local picture file of the cover, if the source names one. + public string? CoverImage { get; set; } +} + +/// An e-book: content and metadata. +internal sealed record Book(HDocument Document, BookMetadata Metadata) +{ + /// The title to show: metadata, the document's own title, the first heading, else the file name. + public string DisplayTitle(string sourcePath) + { + if (!string.IsNullOrWhiteSpace(Metadata.Title)) + return Metadata.Title.Trim(); + if (!string.IsNullOrWhiteSpace(Document.Title)) + return Document.Title.Trim(); + var heading = Document.Sections.SelectMany(s => s.Blocks).OfType().FirstOrDefault(p => p.HeadingLevel > 0); + if (heading is not null && HDocumentWalker.Text(heading.Inlines) is { Length: > 0 } text) + return text.Length > 200 ? text[..200] : text; + return Filee.Core.Formats.FormatRegistry.NameWithoutExtension(sourcePath); + } + + /// The language from the metadata, else guessed from the text: Korean, Chinese, Japanese or English. + public string DisplayLanguage() + { + if (!string.IsNullOrWhiteSpace(Metadata.Language)) + return Metadata.Language.Trim(); + int hangul = 0, han = 0, kana = 0, latin = 0; + foreach (var text in Document.Sections.SelectMany(s => HDocumentWalker.Inlines(s.Blocks)).OfType().Take(2000)) + { + foreach (var ch in text.Text) + { + if (ch is >= '가' and <= '힣') + hangul++; + else if (ch is >= '぀' and <= 'ヿ') + kana++; + else if (ch is >= '一' and <= '鿿') + han++; + else if (char.IsAsciiLetter(ch)) + latin++; + } + } + return hangul > 0 && hangul * 3 >= latin ? "ko" + : kana > 0 && kana * 3 >= latin ? "ja" + : han > 0 && han * 3 >= latin ? "zh" + : "en"; + } +} + +/// Helpers shared by the e-book readers and writers. +internal static class EbookFiles +{ + /// Message for books with DRM: Filee never removes copy protection. + public const string DrmMessage = + "This book is protected with DRM (copy protection), so it cannot be converted. Filee does not remove DRM; " + + "open it in the reading app it was bought for."; + + /// + /// Unpacks a ZIP into . Entries that would land outside it ("../", absolute paths) + /// are skipped, so a crafted archive cannot write anywhere else. + /// + public static void Extract(ZipArchive zip, string folder) + { + var root = Path.GetFullPath(folder) + Path.DirectorySeparatorChar; + Directory.CreateDirectory(root); + foreach (var entry in zip.Entries) + { + if (entry.FullName.EndsWith('/') || entry.FullName.EndsWith('\\')) + continue; + if (SafePath(root, entry.FullName) is not { } target) + continue; + Directory.CreateDirectory(Path.GetDirectoryName(target)!); + entry.ExtractToFile(target, overwrite: true); + } + } + + /// A path inside for an archive entry name, or null if it would escape. + public static string? SafePath(string root, string entryName) + { + var relative = entryName.Replace('\\', '/').TrimStart('/'); + if (relative.Length == 0 || relative.Contains(':')) + return null; + try + { + var full = Path.GetFullPath(Path.Combine(root, relative)); + var prefix = Path.GetFullPath(root).TrimEnd(Path.DirectorySeparatorChar) + Path.DirectorySeparatorChar; + return full.StartsWith(prefix, StringComparison.OrdinalIgnoreCase) ? full : null; + } + catch (Exception ex) when (ex is ArgumentException or NotSupportedException or PathTooLongException) + { + return null; + } + } + + /// A fresh folder inside the job's work directory. + public static string NewFolder(string workDirectory, string prefix) + { + var folder = Path.Combine(workDirectory, $"{prefix}-{Guid.NewGuid().ToString("N")[..8]}"); + Directory.CreateDirectory(folder); + return folder; + } + + /// + /// A picture in a format every reading system shows: JPEG and PNG (plus GIF and SVG when + /// ) are kept, others are converted into — to JPEG for + /// opaque photos (WEBP, AVIF, ... would grow a lot as PNG), else PNG. Null when the file cannot be read. + /// + public static (string File, string MediaType)? CommonImage(string path, string folder, bool gifAndSvg) + { + var mediaType = ImageMediaType(path); + if (mediaType is "image/jpeg" or "image/png" || gifAndSvg && mediaType is "image/gif" or "image/svg+xml") + return (path, mediaType); + try + { + using var image = new ImageMagick.MagickImage(path); + var jpeg = !image.HasAlpha && mediaType is "image/webp" or "image/avif" or "image/jxl" or "image/tiff"; + Directory.CreateDirectory(folder); + var file = Path.Combine(folder, $"{Guid.NewGuid():N}.{(jpeg ? "jpg" : "png")}"); + image.Quality = 90; + image.Write(file, jpeg ? ImageMagick.MagickFormat.Jpeg : ImageMagick.MagickFormat.Png); + return (file, jpeg ? "image/jpeg" : "image/png"); + } + catch (ImageMagick.MagickException) + { + return null; + } + } + + /// The media type of a picture file by extension, or null for files that are not pictures. + public static string? ImageMediaType(string path) => Path.GetExtension(path).ToLowerInvariant() switch + { + ".jpg" or ".jpeg" or ".jpe" or ".jfif" => "image/jpeg", + ".png" => "image/png", + ".gif" => "image/gif", + ".svg" => "image/svg+xml", + ".webp" => "image/webp", + ".bmp" => "image/bmp", + ".tif" or ".tiff" => "image/tiff", + ".avif" => "image/avif", + ".jxl" => "image/jxl", + _ => null, + }; +} diff --git a/src/Filee.Engines/Ebooks/CalibreConverter.cs b/src/Filee.Engines/Ebooks/CalibreConverter.cs new file mode 100644 index 0000000..5ea69a9 --- /dev/null +++ b/src/Filee.Engines/Ebooks/CalibreConverter.cs @@ -0,0 +1,107 @@ +// E-book formats Filee does not read or write itself, with calibre's ebook-convert (GPL-3.0, a separate program the +// user downloads in Settings → Engines): LIT, LRF, CHM, PDB, PML, RB, SNB, TCR and OEB → EPUB, and EPUB → MOBI, +// AZW3, LIT, LRF, PDB, PML, RB, SNB and TCR. EPUB is the hub: the built-in engine handles every other step, and +// calibre's edges cost more so built-in routes always win where they exist. +// +// Each run gets its own calibre configuration, cache and temp folders inside the job's work directory, so Filee +// never reads or changes the settings of a calibre the user may have installed. + +using System.Globalization; +using System.Text.RegularExpressions; +using Filee.Core.Conversion; +using Filee.Engines.Infrastructure; + +namespace Filee.Engines.Ebooks; + +/// Conversions through calibre's ebook-convert. +public sealed partial class CalibreConverter : IConverter +{ + private static readonly TimeSpan Timeout = TimeSpan.FromMinutes(10); + + /// Formats only calibre reads. + internal static readonly string[] Inputs = ["lit", "lrf", "chm", "pdb", "pml", "rb", "snb", "tcr", "oeb"]; + + /// Formats only calibre writes. + internal static readonly string[] Outputs = ["mobi", "azw3", "lit", "lrf", "pdb", "pml", "rb", "snb", "tcr"]; + + private string? _executable; + + public string Id => "calibre"; + public string DisplayName => "Calibre"; + + /// ebook-convert is a large Python process; two at a time keep the PC responsive. + public int MaxParallelism => 2; + + public IReadOnlyList Edges { get; } = + [ + .. Inputs.Select(from => new ConversionEdge(from, "epub", 20)), + .. Outputs.Select(to => new ConversionEdge("epub", to, 20)), + ]; + + public EngineStatus GetStatus() + { + _executable = Locate(); + return _executable is null ? EngineStatus.Unavailable("engine.reason.not_installed") : EngineStatus.Available(_executable, EngineVersions.Component("calibre")); + } + + public async Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) + { + var executable = _executable ?? Locate() ?? throw new InvalidOperationException("Calibre is not installed."); + var work = EbookFiles.NewFolder(step.WorkDirectory, "calibre"); + + // ebook-convert picks formats by file extension: give it names it knows. An .opf (OEB) stays where it is, + // next to the files it lists; a zipped .oeb is read by calibre's ZIP input. + var input = step.InputPath; + if (!input.EndsWith(".opf", StringComparison.OrdinalIgnoreCase)) + { + input = Path.Combine(work, "input." + (step.From == "oeb" ? "zip" : step.From)); + File.Copy(step.InputPath, input); + } + var temp = Path.Combine(work, "output." + step.To); + + var environment = new Dictionary + { + ["CALIBRE_CONFIG_DIRECTORY"] = Directory.CreateDirectory(Path.Combine(work, "config")).FullName, + ["CALIBRE_CACHE_DIRECTORY"] = Directory.CreateDirectory(Path.Combine(work, "cache")).FullName, + ["CALIBRE_TEMP_DIR"] = Directory.CreateDirectory(Path.Combine(work, "temp")).FullName, + ["CALIBRE_OVERRIDE_LANG"] = "en", // English messages in the error details, whatever the Windows language + ["PYTHONIOENCODING"] = "utf-8", + }; + progress?.Report(0.02); + var result = await ProcessRunner.RunAsync(executable, [input, temp], Timeout, cancellationToken, + workingDirectory: work, environment: environment, onOutput: line => ReportProgress(line, progress)); + if (result.ExitCode != 0 || !File.Exists(temp)) + throw new InvalidOperationException($"Calibre failed (exit {result.ExitCode}). {LastLines(result.StandardError.Trim().Length > 0 ? result.StandardError : result.StandardOutput)}"); + + var output = step.Output.Allocate(step.To); + if (output is null) + return []; + File.Move(temp, output, overwrite: true); + progress?.Report(1); + return [output]; + } + + /// ebook-convert prints "34% Running transforms on e-book..." while it works. + private static void ReportProgress(string line, IProgress? progress) + { + if (progress is not null && Percent().Match(line) is { Success: true } match + && int.TryParse(match.Groups[1].Value, NumberStyles.None, CultureInfo.InvariantCulture, out var percent)) + progress.Report(Math.Clamp(percent, 0, 100) / 100.0 * 0.95); + } + + /// The end of calibre's output, where its error message (or Python traceback) is. + private static string LastLines(string output) + { + var lines = output.Split('\n', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries); + return string.Join(" ", lines.TakeLast(3)); + } + + [GeneratedRegex(@"^\s*(\d{1,3})%\s")] + private static partial Regex Percent(); + + /// Filee's own calibre (engines/calibre, downloaded on demand), never one installed on the system. + internal static string? Locate() => + OperatingSystem.IsWindows() && EngineEnvironment.FindBundled("calibre") is { } folder + ? EngineEnvironment.FirstExisting(Path.Combine(folder, "ebook-convert.exe")) + : null; +} diff --git a/src/Filee.Engines/Ebooks/ChapterReader.cs b/src/Filee.Engines/Ebooks/ChapterReader.cs new file mode 100644 index 0000000..34230d2 --- /dev/null +++ b/src/Filee.Engines/Ebooks/ChapterReader.cs @@ -0,0 +1,91 @@ +// Reads the XHTML chapters of a book (EPUB spine, KF8 parts, HTMLZ page) into one HDocument: every chapter starts +// on a new page, and links between chapters ("ch2.xhtml#note3") become links to bookmarks inside the document. + +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal static class ChapterReader +{ + /// Reads chapter files in reading order. + /// Scratch folder for images embedded as data: URIs. + public static HDocument Read(IReadOnlyList chapters, string mediaFolder, CancellationToken cancellationToken = default) + { + var index = new Dictionary(StringComparer.OrdinalIgnoreCase); + for (var i = 0; i < chapters.Count; i++) + index.TryAdd(Path.GetFullPath(chapters[i]), i); + + var section = new HSection(); + var document = new HDocument(); + document.Sections.Add(section); + for (var i = 0; i < chapters.Count; i++) + { + cancellationToken.ThrowIfCancellationRequested(); + var chapter = Path.GetFullPath(chapters[i]); + if (!File.Exists(chapter)) + continue; // a spine item missing from the package: skip it like reading systems do + var number = i; + var content = HtmlReader.ReadFile(chapter, new HtmlReadOptions + { + BaseFolder = Path.GetDirectoryName(chapter)!, + MediaFolder = mediaFolder, + MapId = id => Anchor(number, id), + MapLink = href => Link(href, chapter, number, index), + }); + document.Title ??= content.Title; + if (!content.Blocks.Any(HasContent)) + continue; + + // The chapter starts on a new page and carries a bookmark for links to the chapter file itself. + if (content.Blocks[0] is not HParagraph first) + content.Blocks.Insert(0, first = new HParagraph()); + first.PageBreakBefore = section.Blocks.Count > 0; + first.Inlines.Insert(0, new HBookmark(Anchor(number, null))); + section.Blocks.AddRange(content.Blocks); + } + HtmlReader.PruneBookmarks(document); + return document; + } + + private static bool HasContent(HBlock block) => block is HTable + || HDocumentWalker.Inlines([block]).Any(i => i is HImage || i is HText { Text: var text } && !string.IsNullOrWhiteSpace(text)); + + /// Bookmark name of an element id (or the chapter start) in chapter . + private static string Anchor(int chapter, string? id) + { + if (id is null) + return $"c{chapter + 1}"; + var clean = new char[id.Length]; + for (var i = 0; i < id.Length; i++) + clean[i] = char.IsAsciiLetterOrDigit(id[i]) || id[i] is '-' or '_' ? id[i] : '_'; + return $"c{chapter + 1}-{new string(clean)}"; + } + + /// Links inside the book point at bookmarks; web and mail links stay; links to other files are dropped. + private static string? Link(string href, string chapter, int number, Dictionary index) + { + if (href.StartsWith('#')) + return href.Length > 1 ? "#" + Anchor(number, Uri.UnescapeDataString(href[1..])) : null; + var colon = href.IndexOf(':'); + if (colon > 1 && href[..colon].All(char.IsAsciiLetter)) + return href.StartsWith("javascript:", StringComparison.OrdinalIgnoreCase) ? null : href; + + var hash = href.IndexOf('#'); + var file = hash >= 0 ? href[..hash] : href; + var fragment = hash >= 0 ? Uri.UnescapeDataString(href[(hash + 1)..]) : null; + var query = file.IndexOf('?'); + if (query >= 0) + file = file[..query]; + try + { + var target = Path.GetFullPath(Path.Combine(Path.GetDirectoryName(chapter)!, Uri.UnescapeDataString(file))); + return index.TryGetValue(target, out var chapterIndex) + ? "#" + Anchor(chapterIndex, string.IsNullOrEmpty(fragment) ? null : fragment) + : null; + } + catch (Exception ex) when (ex is ArgumentException or NotSupportedException or PathTooLongException) + { + return null; + } + } +} diff --git a/src/Filee.Engines/Ebooks/ComicBook.cs b/src/Filee.Engines/Ebooks/ComicBook.cs new file mode 100644 index 0000000..e66eb83 --- /dev/null +++ b/src/Filee.Engines/Ebooks/ComicBook.cs @@ -0,0 +1,214 @@ +// Comic book archives: CBZ (ZIP), CBR (RAR), CB7 (7-Zip), CBT (TAR) and CBC (a ZIP of CBZ files) are pictures in +// an archive, read with SharpCompress (MIT). Pages are sorted by name the way people number them ("page2" before +// "page10"). Comics become PDF (one page per picture, sized like the picture, JPEGs embedded as they are), EPUB +// (fixed layout) or CBZ; PDF becomes CBZ by rendering its pages with PDFium. + +using System.IO.Compression; +using Filee.Core.Conversion; +using ImageMagick; +using PdfSharp.Drawing; +using PdfSharp.Pdf; +using PDFtoImage; +using SharpCompress.Archives; +using SkiaSharp; + +// PDFtoImage marks its API for the platforms PDFium ships for (Windows, macOS, Linux, mobile); Filee only targets +// desktop platforms that are all covered. +#pragma warning disable CA1416 + +namespace Filee.Engines.Ebooks; + +internal static class ComicBook +{ + private static readonly HashSet PictureExtensions = + [".jpg", ".jpeg", ".jpe", ".jfif", ".png", ".gif", ".webp", ".bmp", ".avif", ".jxl", ".tif", ".tiff", ".heic", ".heif"]; + + /// Extracts the pages of a comic archive into , in reading order. + public static List ExtractPages(string path, string format, string folder, CancellationToken cancellationToken) + { + if (format == "cbc") + { + // A collection: the CBZ files inside, in the order of comics.txt ("file.cbz:Title" per line) or by name. + var comics = ExtractArchive(path, EbookFiles.NewFolder(folder, "cbc"), f => Path.GetExtension(f).Equals(".cbz", StringComparison.OrdinalIgnoreCase) || Path.GetFileName(f).Equals("comics.txt", StringComparison.OrdinalIgnoreCase), cancellationToken); + var list = comics.FirstOrDefault(c => Path.GetFileName(c).Equals("comics.txt", StringComparison.OrdinalIgnoreCase)); + var books = comics.Where(c => c != list).ToList(); + if (list is not null) + { + var order = File.ReadAllLines(list).Select(l => l.Split(':')[0].Trim()).Where(l => l.Length > 0).ToList(); + books = [.. books.OrderBy(b => order.FindIndex(o => b.EndsWith(o.Replace('/', Path.DirectorySeparatorChar), StringComparison.OrdinalIgnoreCase)) is var i and >= 0 ? i : int.MaxValue)]; + } + return [.. books.SelectMany(book => ExtractPages(book, "cbz", EbookFiles.NewFolder(folder, "cbz"), cancellationToken))]; + } + var pages = ExtractArchive(path, folder, f => PictureExtensions.Contains(Path.GetExtension(f).ToLowerInvariant()), cancellationToken); + if (pages.Count == 0) + throw new InvalidDataException("The comic contains no pictures."); + return pages; + } + + /// Extracts the matching entries (skipping macOS metadata) and returns them sorted naturally by path. + private static List ExtractArchive(string path, string folder, Func wanted, CancellationToken cancellationToken) + { + var files = new List<(string Key, string File)>(); + IArchive archive; + try + { + archive = ArchiveFactory.Open(path, null); + } + catch (Exception ex) when (ex is InvalidOperationException or InvalidDataException or ArgumentException or NotSupportedException or SharpCompress.Common.ArchiveException) + { + throw new InvalidDataException("The comic archive cannot be opened (damaged, or not the format its extension says).", ex); + } + using (archive) + { + if (archive.Entries.Any(e => e.IsEncrypted)) + throw new InvalidOperationException("The comic archive is password-protected."); + if (archive.IsSolid || archive.Type == SharpCompress.Common.ArchiveType.SevenZip) + { + // Sequential extraction: random access would decompress a solid block again for every entry. + using var reader = archive.ExtractAllEntries(); + while (reader.MoveToNextEntry()) + { + if (Wanted(reader.Entry) is { } target) + { + using var output = File.Create(target); + reader.WriteEntryTo(output); + } + } + } + else + { + foreach (var entry in archive.Entries) + { + if (Wanted(entry) is not { } target) + continue; + using var input = entry.OpenEntryStream(); + using var output = File.Create(target); + input.CopyTo(output); + } + } + } + return [.. files.OrderBy(f => f.Key, NaturalComparer.Instance).Select(f => f.File)]; + + // The file to extract an entry to, or null for folders, macOS metadata and unwanted files. + string? Wanted(SharpCompress.Common.IEntry entry) + { + cancellationToken.ThrowIfCancellationRequested(); + var key = entry.Key ?? ""; + var name = Path.GetFileName(key.Replace('\\', '/')); + if (entry.IsDirectory || key.Contains("__MACOSX", StringComparison.OrdinalIgnoreCase) || name.StartsWith('.') || !wanted(name)) + return null; + if (entry.IsEncrypted) + throw new InvalidOperationException("The comic archive is password-protected."); + var target = Path.Combine(folder, $"{files.Count:00000}{Path.GetExtension(name).ToLowerInvariant()}"); + files.Add((key, target)); + return target; + } + } + + /// One PDF page per picture, each as large as its picture (at its DPI, else 96 dpi). + public static void ToPdf(IReadOnlyList pages, string output, string workFolder, IProgress? progress, CancellationToken cancellationToken) + { + var converted = EbookFiles.NewFolder(workFolder, "pdf-pages"); + using var document = new PdfDocument(); + for (var i = 0; i < pages.Count; i++) + { + cancellationToken.ThrowIfCancellationRequested(); + // JPEG and PNG go in as they are (PDFsharp embeds JPEG data without re-encoding); others become one of them. + if (EbookFiles.CommonImage(pages[i], converted, gifAndSvg: false) is not ({ } file, _)) + continue; + var info = new MagickImageInfo(file); + var dpi = info.Density is { X: >= 36 } density ? density.Units == DensityUnit.PixelsPerCentimeter ? density.X * 2.54 : density.X : 96; + using var image = XImage.FromFile(file); + var page = document.AddPage(); + page.Width = XUnit.FromPoint(info.Width / dpi * 72); + page.Height = XUnit.FromPoint(info.Height / dpi * 72); + using (var graphics = XGraphics.FromPdfPage(page)) + graphics.DrawImage(image, 0, 0, page.Width.Point, page.Height.Point); + progress?.Report((i + 1.0) / pages.Count); + } + if (document.PageCount == 0) + throw new InvalidDataException("The comic contains no readable pictures."); + document.Save(output); + } + + /// Writes pages into a CBZ (stored: pictures do not compress further), numbered in order. + public static void ToCbz(IReadOnlyList pages, string output, CancellationToken cancellationToken) + { + var temp = output + ".tmp"; + using (var zip = ZipFile.Open(temp, ZipArchiveMode.Create)) + { + var digits = Math.Max(3, pages.Count.ToString(System.Globalization.CultureInfo.InvariantCulture).Length); + for (var i = 0; i < pages.Count; i++) + { + cancellationToken.ThrowIfCancellationRequested(); + var name = (i + 1).ToString(System.Globalization.CultureInfo.InvariantCulture).PadLeft(digits, '0') + Path.GetExtension(pages[i]).ToLowerInvariant(); + zip.CreateEntryFromFile(pages[i], name, CompressionLevel.NoCompression); + } + } + File.Move(temp, output, overwrite: true); + } + + /// + /// Renders the PDF's pages (the preset's page range) as JPEG files at the preset's DPI and quality. PDFtoImage + /// serializes its calls into PDFium (not thread-safe), so this may run next to PdfiumConverter. + /// + public static List RenderPdf(string pdf, ConversionStep step, string folder, IProgress? progress, CancellationToken cancellationToken) + { + var bytes = File.ReadAllBytes(pdf); + var pageNumbers = PageRange.Parse(step.Preset.Pdf.PageRange, Conversion.GetPageCount(bytes, null)); + var dpi = Math.Clamp(step.Preset.Pdf.RenderDpi, 36, 600); + var quality = Math.Clamp(step.Preset.Image.Quality, 1, 100); + var options = new RenderOptions(Dpi: dpi, WithAnnotations: true, WithFormFill: true, BackgroundColor: SKColors.White); + var pages = new List(); + foreach (var bitmap in Conversion.ToImages(bytes, pageNumbers, null, options)) + { + cancellationToken.ThrowIfCancellationRequested(); + using (bitmap) + using (var data = bitmap.Encode(SKEncodedImageFormat.Jpeg, quality)) + { + var file = Path.Combine(folder, $"{pages.Count + 1:00000}.jpg"); + File.WriteAllBytes(file, data.ToArray()); + pages.Add(file); + } + progress?.Report(0.9 * pages.Count / Math.Max(1, pageNumbers.Count)); + } + return pages; + } +} + +/// Compares names the way people number pages: digit runs by value ("p2" < "p10"), case-insensitively. +internal sealed class NaturalComparer : IComparer +{ + public static readonly NaturalComparer Instance = new(); + + public int Compare(string? x, string? y) + { + if (x is null || y is null) + return string.CompareOrdinal(x, y); + int i = 0, j = 0; + while (i < x.Length && j < y.Length) + { + if (char.IsAsciiDigit(x[i]) && char.IsAsciiDigit(y[j])) + { + var startX = i; + var startY = j; + while (i < x.Length && char.IsAsciiDigit(x[i])) + i++; + while (j < y.Length && char.IsAsciiDigit(y[j])) + j++; + var numberX = x[startX..i].TrimStart('0'); + var numberY = y[startY..j].TrimStart('0'); + var byValue = numberX.Length != numberY.Length ? numberX.Length.CompareTo(numberY.Length) : string.CompareOrdinal(numberX, numberY); + if (byValue != 0) + return byValue; + continue; + } + var byChar = char.ToLowerInvariant(x[i]).CompareTo(char.ToLowerInvariant(y[j])); + if (byChar != 0) + return byChar; + i++; + j++; + } + return (x.Length - i).CompareTo(y.Length - j); + } +} diff --git a/src/Filee.Engines/Ebooks/EbookConverter.cs b/src/Filee.Engines/Ebooks/EbookConverter.cs new file mode 100644 index 0000000..3c420bb --- /dev/null +++ b/src/Filee.Engines/Ebooks/EbookConverter.cs @@ -0,0 +1,211 @@ +// Built-in e-book conversions, no external program: EPUB, MOBI / AZW / AZW3 / PRC, FB2, HTMLZ and TXTZ are read into +// the HWPX document model and written as EPUB, HWPX (→ PDF / images through rhwp and PDFium), TXT, HTML, Markdown, +// FB2, HTMLZ or TXTZ; HTML, Markdown, TXT and DOCX become e-books too. Comics (CBZ / CBR / CB7 / CBT / CBC) become +// PDF, fixed-layout EPUB or CBZ, PDF becomes CBZ, and AZW4 (Print Replica) gives back its PDF. +// Formats only calibre reads or writes (LIT, LRF, PDB, MOBI / AZW3 output, ...) are in CalibreConverter. + +using System.Text; +using Filee.Core.Conversion; +using Filee.Engines.Hwp.Hwpx; +using Filee.Engines.Hwp.Hwpx.Docx; +using Filee.Engines.Infrastructure; + +namespace Filee.Engines.Ebooks; + +/// E-book and comic conversions without external programs. +public sealed class EbookConverter : IConverter +{ + /// E-book formats read into the document model. + private static readonly string[] BookReaders = ["epub", "mobi", "azw3", "azw", "prc", "fb2", "htmlz", "txtz"]; + + /// Formats written from the document model. + private static readonly string[] BookWriters = ["epub", "hwpx", "txt", "html", "md", "fb2", "htmlz", "txtz"]; + + /// Documents that become e-books (their other conversions belong to the document engines). + private static readonly string[] DocumentReaders = ["html", "md", "txt", "docx"]; + + private static readonly string[] EbookWriters = ["epub", "fb2", "htmlz", "txtz"]; + + private static readonly string[] Comics = ["cbz", "cbr", "cb7", "cbt", "cbc"]; + + public string Id => "ebook"; + public string DisplayName => "E-books (built-in)"; + public int MaxParallelism => 0; + + public IReadOnlyList Edges { get; } = + [ + .. BookReaders.SelectMany(from => BookWriters.Where(to => to != from).Select(to => new ConversionEdge(from, to, to == "md" ? 12 : 10))), + .. DocumentReaders.SelectMany(from => EbookWriters.Select(to => new ConversionEdge(from, to))), + // HTML → TXT has no other built-in route; HTML → Markdown is a fallback for when Pandoc is not installed. + new("html", "txt"), + new("html", "md", 14), + .. Comics.SelectMany(from => new[] { new ConversionEdge(from, "pdf"), new ConversionEdge(from, "epub") }), + .. Comics.Where(from => from != "cbz").Select(from => new ConversionEdge(from, "cbz")), + // Pages as pictures lose the text: other routes to EPUB (through DOCX, TXT, ...) should win when they exist. + new("pdf", "cbz", 15), + new("azw4", "pdf"), + ]; + + public EngineStatus GetStatus() => EngineStatus.Available("EPUB, MOBI / AZW3, FB2, HTMLZ, TXTZ, CBZ / CBR / CB7 / CBT / CBC, AZW4", + $"{EngineVersions.BuiltIn} · {EngineVersions.Library("AngleSharp", typeof(AngleSharp.BrowsingContext))}"); + + public Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) => + Task.Run(() => Convert(step, progress, cancellationToken), cancellationToken); + + private static IReadOnlyList Convert(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) + { + progress?.Report(0.05); + var work = EbookFiles.NewFolder(step.WorkDirectory, "ebook"); + if (Comics.Contains(step.From) || step.From == "pdf") + return Comic(step, work, progress, cancellationToken); + if (step.From == "azw4") + { + var pdf = MobiReader.PrintReplicaPdf(step.InputPath); + var target = step.Output.Allocate("pdf"); + if (target is null) + return []; + File.WriteAllBytes(target, pdf); + progress?.Report(1); + return [target]; + } + + var book = Read(step, work, cancellationToken); + progress?.Report(0.5); + cancellationToken.ThrowIfCancellationRequested(); + var output = step.Output.Allocate(step.To); + if (output is null) + return []; + Write(book, step, output, work); + progress?.Report(1); + return [output]; + } + + private static Book Read(ConversionStep step, string work, CancellationToken cancellationToken) + { + var input = step.InputPath; + var media = Path.Combine(work, "media"); + switch (step.From) + { + case "epub": + return EpubReader.ReadBook(input, work, cancellationToken); + case "mobi" or "azw3" or "azw" or "prc": + return MobiReader.ReadBook(input, work, cancellationToken); + case "fb2": + return Fb2Reader.ReadBook(input, work); + case "htmlz": + return ZippedText.ReadHtmlzBook(input, work, cancellationToken); + case "txtz": + return ZippedText.ReadTxtzBook(input, work); + case "html": + { + var document = HtmlReader.Read(input, media); + return new Book(document, new BookMetadata { Title = document.Title }); + } + case "md": + { + var document = MarkdownReader.Read(HwpxConverter.DecodeText(File.ReadAllBytes(input)), Path.GetDirectoryName(Path.GetFullPath(input))!); + return new Book(document, new BookMetadata { Title = document.Title }); + } + case "txt": + return new Book(PlainText.ToDocument(HwpxConverter.DecodeText(File.ReadAllBytes(input))), new BookMetadata()); + case "docx": + { + var document = DocxReader.Read(input, media); + return new Book(document, new BookMetadata { Title = document.Title }); + } + default: + throw new NotSupportedException($"{step.From} → {step.To}"); + } + } + + private static void Write(Book book, ConversionStep step, string output, string work) + { + switch (step.To) + { + case "epub": + EpubWriter.Write(book, output, step.InputPath, work); + break; + case "hwpx": + book.Document.Title = book.DisplayTitle(step.InputPath); + HwpxWriter.Write(book.Document, output); + break; + case "txt": + File.WriteAllText(output, PlainTextWriter.Write(book.Document), new UTF8Encoding(encoderShouldEmitUTF8Identifier: true)); + break; + case "html": + { + // One self-contained page: pictures are embedded as data: URIs. + var images = EbookFiles.NewFolder(work, "html-images"); + var page = XhtmlWriter.Write(book.Document, new XhtmlOptions { ImageSource = path => DataUri(path, images) }).Single(); + File.WriteAllText(output, XhtmlWriter.Page(book.DisplayTitle(step.InputPath), page.Body, book.DisplayLanguage(), null, epub: false), new UTF8Encoding(false)); + break; + } + case "md": + File.WriteAllText(output, MarkdownWriter.Write(book.Document, MarkdownImages(output, work)), new UTF8Encoding(false)); + break; + case "fb2": + Fb2Writer.Write(book, output, step.InputPath, work); + break; + case "htmlz": + ZippedText.WriteHtmlz(book, output, step.InputPath, work); + break; + case "txtz": + ZippedText.WriteTxtz(book, output, step.InputPath, work); + break; + default: + throw new NotSupportedException($"{step.From} → {step.To}"); + } + } + + private static IReadOnlyList Comic(ConversionStep step, string work, IProgress? progress, CancellationToken cancellationToken) + { + var pages = step.From == "pdf" + ? ComicBook.RenderPdf(step.InputPath, step, work, progress, cancellationToken) + : ComicBook.ExtractPages(step.InputPath, step.From, work, cancellationToken); + progress?.Report(0.4); + var output = step.Output.Allocate(step.To); + if (output is null) + return []; + switch (step.To) + { + case "pdf": + ComicBook.ToPdf(pages, output, work, progress, cancellationToken); + break; + case "epub": + EpubWriter.WriteFixedLayout(Filee.Core.Formats.FormatRegistry.NameWithoutExtension(step.InputPath), pages, output, work, cancellationToken); + break; + case "cbz": + ComicBook.ToCbz(pages, output, cancellationToken); + break; + default: + throw new NotSupportedException($"{step.From} → {step.To}"); + } + progress?.Report(1); + return [output]; + } + + private static string? DataUri(string path, string convertFolder) => + EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is ({ } file, { } mediaType) + ? $"data:{mediaType};base64,{System.Convert.ToBase64String(File.ReadAllBytes(file))}" + : null; + + /// Pictures of a Markdown export go to "<name>_files" next to it, referenced relatively. + private static Func MarkdownImages(string output, string work) + { + var folderName = Path.GetFileNameWithoutExtension(output) + "_files"; + var folder = Path.Combine(Path.GetDirectoryName(output)!, folderName); + var convert = EbookFiles.NewFolder(work, "md-images"); + var names = new Dictionary(StringComparer.OrdinalIgnoreCase); + return path => + { + if (names.TryGetValue(path, out var known)) + return known; + if (EbookFiles.CommonImage(path, convert, gifAndSvg: true) is not ({ } file, _)) + return null; + Directory.CreateDirectory(folder); + var name = $"img{names.Count + 1:0000}{Path.GetExtension(file).ToLowerInvariant()}"; + File.Copy(file, Path.Combine(folder, name), overwrite: true); + return names[path] = $"{Uri.EscapeDataString(folderName)}/{name}"; + }; + } +} diff --git a/src/Filee.Engines/Ebooks/EpubReader.cs b/src/Filee.Engines/Ebooks/EpubReader.cs new file mode 100644 index 0000000..fac6764 --- /dev/null +++ b/src/Filee.Engines/Ebooks/EpubReader.cs @@ -0,0 +1,92 @@ +// EPUB 2 / 3 → Book: META-INF/container.xml → the OPF package → spine → XHTML chapters through HtmlReader (each +// chapter starts a new page), pictures from the package, title / authors / language / cover from the metadata. +// Books with DRM (Adobe ADEPT, Apple FairPlay, ...) are refused; font obfuscation is not DRM and is fine. + +using System.IO.Compression; +using System.Xml.Linq; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal static class EpubReader +{ + /// Algorithms that only obfuscate embedded fonts (IDPF and Adobe), which do not protect the content. + private static readonly HashSet FontObfuscation = + [ + "http://www.idpf.org/2008/embedding", + "http://ns.adobe.com/pdf/enc#RC", + ]; + + /// Reads an EPUB's content (for the reader registry of the HWPX writer). + /// Scratch folder: the book is unpacked into a sub folder. + public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document; + + public static Book ReadBook(string path, string workFolder, CancellationToken cancellationToken = default) + { + var folder = EbookFiles.NewFolder(workFolder, "epub"); + using (var zip = OpenZip(path)) + { + CheckDrm(zip); + EbookFiles.Extract(zip, folder); + } + + var package = OpfPackage.Load(OpfPackage.RootFile(folder)); + var chapters = new List(); + foreach (var item in package.Spine) + { + if (item.MediaType is "application/xhtml+xml" or "text/html" or "application/xml" || item.MediaType.Length == 0 && item.Path.EndsWith("html", StringComparison.OrdinalIgnoreCase)) + chapters.Add(item.Path); + else if (item.MediaType.StartsWith("image/", StringComparison.Ordinal)) + chapters.Add(ImagePage(item.Path)); // a picture in the spine (comics, covers) is a page of its own + } + if (chapters.Count == 0) + throw new InvalidDataException("The EPUB has no readable chapters."); + + var document = ChapterReader.Read(chapters, Path.Combine(folder, "~media"), cancellationToken); + document.Title = package.Metadata.Title ?? document.Title; + return new Book(document, package.Metadata); + } + + private static ZipArchive OpenZip(string path) + { + try + { + return ZipFile.OpenRead(path); + } + catch (InvalidDataException ex) + { + throw new InvalidDataException("This is not a valid EPUB file (it is not a ZIP package).", ex); + } + } + + /// Throws for encrypted content: anything in encryption.xml except font obfuscation, or Apple's sinf.xml. + internal static void CheckDrm(ZipArchive zip) + { + if (zip.GetEntry("META-INF/sinf.xml") is not null) + throw new InvalidOperationException(EbookFiles.DrmMessage); + if (zip.GetEntry("META-INF/encryption.xml") is not { } entry) + return; + XDocument encryption; + try + { + using var stream = entry.Open(); + encryption = XDocument.Load(stream); + } + catch (System.Xml.XmlException) + { + return; // unreadable: the chapters will tell whether they are readable + } + var algorithms = encryption.Descendants().Where(e => e.Name.LocalName == "EncryptionMethod") + .Select(e => (string?)e.Attribute("Algorithm") ?? ""); + if (algorithms.Any(a => !FontObfuscation.Contains(a))) + throw new InvalidOperationException(EbookFiles.DrmMessage); + } + + /// An XHTML page next to a picture that the spine lists directly. + private static string ImagePage(string image) + { + var page = Path.Combine(Path.GetDirectoryName(image)!, $"~{Path.GetFileNameWithoutExtension(image)}-{Guid.NewGuid():N}.xhtml"); + File.WriteAllText(page, $"

"); + return page; + } +} diff --git a/src/Filee.Engines/Ebooks/EpubWriter.cs b/src/Filee.Engines/Ebooks/EpubWriter.cs new file mode 100644 index 0000000..82781e9 --- /dev/null +++ b/src/Filee.Engines/Ebooks/EpubWriter.cs @@ -0,0 +1,265 @@ +// Book → EPUB 3 with an EPUB 2 NCX for older readers: "mimetype" first and stored, container.xml, an OPF with Dublin +// Core metadata, a navigation document built from the headings, one XHTML file per chapter (split at level-1 +// headings and page breaks by XhtmlWriter), one style sheet and the pictures. Pictures in formats reading systems +// do not all show (BMP, TIFF, WEBP, ...) are converted to PNG or JPEG. Fixed-layout books (comics) have one page +// per picture, sized like the picture. + +using System.IO.Compression; +using System.Text; +using Filee.Engines.Hwp.Hwpx; +using ImageMagick; + +namespace Filee.Engines.Ebooks; + +internal static class EpubWriter +{ + /// Writes a reflowable EPUB. + /// The file being converted, for a title when the book has none. + public static void Write(Book book, string outputPath, string sourcePath, string workFolder) + { + var title = book.DisplayTitle(sourcePath); + var language = book.DisplayLanguage(); + var package = new EpubPackage(title, language, book.Metadata); + var imageFolder = EbookFiles.NewFolder(workFolder, "epub-images"); + + var chapters = XhtmlWriter.Write(book.Document, new XhtmlOptions + { + SplitChapters = true, + ChapterFile = i => $"ch{i + 1:000}.xhtml", + ImageSource = path => package.AddImage(path, imageFolder), + Epub = true, + }); + + // A cover page for covers the content does not show already (MOBI and FB2 keep the cover apart). + var cover = book.Metadata.CoverImage is { } coverFile && File.Exists(coverFile) ? package.AddImage(coverFile, imageFolder, cover: true) : null; + if (cover is not null && !HDocumentWalker.Images(book.Document).Any(i => string.Equals(i.Path, book.Metadata.CoverImage, StringComparison.OrdinalIgnoreCase))) + { + var body = $"
\"{XhtmlWriter.Escape(title)}\"
\n"; + package.AddPage("cover.xhtml", XhtmlWriter.Page(title, body, language, "style.css", epub: true)); + } + + foreach (var chapter in chapters) + package.AddPage(chapter.FileName, XhtmlWriter.Page(chapter.Title ?? title, chapter.Body, language, "style.css", epub: true)); + package.Toc.AddRange(TableOfContents(chapters, title)); + package.Save(outputPath, XhtmlWriter.Css, fixedLayout: false); + } + + /// Writes a fixed-layout EPUB with one page per picture (comics). + /// Picture files in reading order. + public static void WriteFixedLayout(string title, IReadOnlyList pages, string outputPath, string workFolder, CancellationToken cancellationToken) + { + var package = new EpubPackage(title, "en", new BookMetadata()); + var imageFolder = EbookFiles.NewFolder(workFolder, "epub-images"); + var count = 0; + foreach (var picture in pages) + { + cancellationToken.ThrowIfCancellationRequested(); + if (package.AddImage(picture, imageFolder, cover: count == 0) is not { } source) + continue; // unreadable picture: leave the page out + var info = new MagickImageInfo(picture); + var page = $"p{++count:0000}.xhtml"; + var head = $"\n"; + var body = $"
\"\"
\n"; + package.AddPage(page, XhtmlWriter.Page(title, body, "en", "style.css", epub: true, head)); + } + if (count == 0) + throw new InvalidDataException("The comic contains no readable pictures."); + package.Save(outputPath, "body{margin:0;padding:0;}\n.page{margin:0;padding:0;}\nimg{display:block;margin:0;padding:0;}\n", fixedLayout: true); + } + + /// + /// Entries for the navigation document: the two highest heading levels used in the book; chapters without + /// headings appear by their first words when no chapter has a heading. + /// + private static List TableOfContents(IReadOnlyList chapters, string title) + { + var headings = chapters.SelectMany(c => c.Headings).Where(h => h.Text.Length > 0).ToList(); + if (headings.Count > 0) + { + var top = headings.Min(h => h.Level); + return headings.Where(h => h.Level <= top + 1).Select(h => h with { Level = h.Level - top + 1 }).ToList(); + } + return chapters.Select((c, i) => new XhtmlHeading(1, i == 0 ? title : c.PlainStart.Length > 0 ? c.PlainStart + "…" : $"{i + 1}", c.FileName)).ToList(); + } +} + +/// Collects the files of an EPUB and writes the package. +internal sealed class EpubPackage(string title, string language, BookMetadata metadata) +{ + private readonly List<(string Href, string Content)> _pages = []; + private readonly List<(string Href, string Source, string MediaType, bool Cover)> _images = []; + private readonly Dictionary _imageBySource = new(StringComparer.OrdinalIgnoreCase); + + /// Navigation entries; level 1 is the top. + public List Toc { get; } = []; + + public void AddPage(string href, string content) => _pages.Add((href, content)); + + /// + /// Adds a picture and returns its path in the package, or null when it cannot be read. JPEG, PNG, GIF and SVG + /// are kept; other formats become PNG (JPEG for opaque photos, which would grow a lot as PNG). + /// + public string? AddImage(string path, string convertFolder, bool cover = false) + { + if (_imageBySource.TryGetValue(path, out var known)) + { + if (cover) + MarkCover(known); + return known; + } + if (EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is not ({ } file, { } mediaType)) + return null; + var extension = mediaType switch + { + "image/jpeg" => "jpg", + "image/png" => "png", + "image/gif" => "gif", + _ => "svg", + }; + var href = $"images/img{_images.Count + 1:0000}.{extension}"; + _images.Add((href, file, mediaType!, cover)); + _imageBySource[path] = href; + return href; + } + + private void MarkCover(string href) + { + var index = _images.FindIndex(i => i.Href == href); + if (index >= 0) + _images[index] = _images[index] with { Cover = true }; + } + + public void Save(string outputPath, string css, bool fixedLayout) + { + var identifier = $"urn:uuid:{Guid.NewGuid()}"; + var temp = outputPath + ".tmp"; + using (var stream = File.Create(temp)) + using (var zip = new ZipArchive(stream, ZipArchiveMode.Create)) + { + // OCF: "mimetype" first, stored, without extra fields. + Text(zip, "mimetype", "application/epub+zip", CompressionLevel.NoCompression); + Text(zip, "META-INF/container.xml", + "\n\n" + + "\n\n"); + Text(zip, "OEBPS/content.opf", Opf(identifier, fixedLayout)); + Text(zip, "OEBPS/nav.xhtml", Nav()); + Text(zip, "OEBPS/toc.ncx", Ncx(identifier)); + Text(zip, "OEBPS/style.css", css); + foreach (var (href, content) in _pages) + Text(zip, "OEBPS/" + href, content); + foreach (var (href, source, _, _) in _images) + { + // Pictures are compressed already. + using var target = zip.CreateEntry("OEBPS/" + href, CompressionLevel.NoCompression).Open(); + using var input = File.OpenRead(source); + input.CopyTo(target); + } + } + File.Move(temp, outputPath, overwrite: true); + } + + private string Opf(string identifier, bool fixedLayout) + { + var sb = new StringBuilder(); + sb.Append("\n"); + sb.Append($"\n\n"); + sb.Append($"{identifier}\n"); + sb.Append($"{E(title)}\n"); + sb.Append($"{E(language)}\n"); + foreach (var author in metadata.Authors) + sb.Append($"{E(author)}\n"); + if (metadata.Publisher is { } publisher) + sb.Append($"{E(publisher)}\n"); + if (metadata.Description is { } description) + sb.Append($"{E(description)}\n"); + sb.Append($"{DateTime.UtcNow:yyyy-MM-ddTHH:mm:ssZ}\n"); + if (_images.FindIndex(i => i.Cover) is var cover and >= 0) + sb.Append($"\n"); // EPUB 2 readers (and Kindle converters) + if (fixedLayout) + sb.Append("pre-paginated\nauto\n"); + sb.Append("\n\n"); + sb.Append("\n"); + sb.Append("\n"); + sb.Append("\n"); + for (var i = 0; i < _pages.Count; i++) + sb.Append($"\n"); + for (var i = 0; i < _images.Count; i++) + sb.Append($"\n"); + sb.Append("\n\n"); + for (var i = 0; i < _pages.Count; i++) + sb.Append($"\n"); + sb.Append("\n\n"); + return sb.ToString(); + } + + /// The EPUB 3 navigation document: a nested list of the table of contents. + private string Nav() + { + var sb = new StringBuilder(); + sb.Append("\n"); + return XhtmlWriter.Page(title, sb.ToString(), language, "style.css", epub: true); + } + + private string Ncx(string identifier) + { + var sb = new StringBuilder(); + sb.Append("\n\n\n"); + sb.Append($"\n e.Level))}\"/>\n"); + sb.Append("\n\n\n"); + sb.Append($"{E(title)}\n\n"); + var entries = Entries(); + var depth = 0; + for (var i = 0; i < entries.Count; i++) + { + var level = Math.Min(entries[i].Level, depth + 1); + for (; depth >= level; depth--) + sb.Append("\n"); + depth = level; + sb.Append($"{E(entries[i].Text)}\n"); + } + for (; depth > 0; depth--) + sb.Append("\n"); + sb.Append("\n\n"); + return sb.ToString(); + } + + /// Navigation entries; at least one (the first page), as EPUB requires a non-empty table of contents. + private List Entries() => + Toc.Count > 0 ? Toc : [new XhtmlHeading(1, title, _pages.Count > 0 ? _pages[0].Href : "nav.xhtml")]; + + private string ContentsLabel() => language.Split('-')[0].ToLowerInvariant() switch + { + "ko" => "목차", + "zh" => "目录", + "ja" => "目次", + _ => "Contents", + }; + + private static string E(string text) => XhtmlWriter.Escape(text); + + private static void Text(ZipArchive zip, string name, string content, CompressionLevel level = CompressionLevel.Optimal) + { + using var writer = new StreamWriter(zip.CreateEntry(name, level).Open(), new UTF8Encoding(false)); + writer.Write(content); + } +} diff --git a/src/Filee.Engines/Ebooks/Fb2Reader.cs b/src/Filee.Engines/Ebooks/Fb2Reader.cs new file mode 100644 index 0000000..1fd30bd --- /dev/null +++ b/src/Filee.Engines/Ebooks/Fb2Reader.cs @@ -0,0 +1,422 @@ +// FictionBook 2 (FB2) → Book. FB2 is XML: description/title-info holds the metadata (title, authors, language, +// cover), bodies hold nested sections (their titles become headings by depth), binary elements hold the pictures +// as base64. Poems, epigraphs, citations, subtitles and tables are kept; note links (type="note") become real +// footnotes with the text of the notes body. Every top-level section starts a new page. + +using System.Text; +using System.Xml; +using System.Xml.Linq; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal sealed class Fb2Reader +{ + private const int Indent = 2000; + private readonly Dictionary _binaries = new(StringComparer.Ordinal); + private readonly Dictionary _notes = new(StringComparer.Ordinal); + + private Fb2Reader() + { + } + + /// Reads an FB2 book's content (for the reader registry of the HWPX writer). + public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document; + + public static Book ReadBook(string path, string workFolder) + { + var root = Load(path).Root ?? throw new InvalidDataException("The FB2 file is empty."); + if (root.Name.LocalName != "FictionBook") + throw new InvalidDataException("This is not a FictionBook (FB2) file."); + var reader = new Fb2Reader(); + var folder = EbookFiles.NewFolder(workFolder, "fb2"); + reader.SaveBinaries(root, folder); + + var bodies = Children(root, "body").ToList(); + foreach (var notes in bodies.Where(b => (string?)b.Attribute("name") is "notes" or "comments")) + foreach (var section in notes.Descendants().Where(e => e.Name.LocalName == "section" && e.Attribute("id") is not null)) + reader._notes.TryAdd((string)section.Attribute("id")!, section); + + var metadata = reader.Metadata(root); + var sectionBlocks = new HSection(); + foreach (var body in bodies.Where(b => (string?)b.Attribute("name") is not ("notes" or "comments"))) + reader.Body(body, sectionBlocks.Blocks); + var document = new HDocument { Title = metadata.Title }; + document.Sections.Add(sectionBlocks); + HtmlReader.PruneBookmarks(document); + return new Book(document, metadata); + } + + private static XDocument Load(string path) + { + // FB2 files are often windows-1251; the XML declaration names the encoding. + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + var settings = new XmlReaderSettings { DtdProcessing = DtdProcessing.Ignore, XmlResolver = null }; + using var reader = XmlReader.Create(path, settings); + // Spaces between inline elements ("a b") are text. + return XDocument.Load(reader, LoadOptions.PreserveWhitespace); + } + + private static IEnumerable Children(XElement? parent, string name) => + parent?.Elements().Where(e => e.Name.LocalName == name) ?? []; + + private static XElement? Child(XElement? parent, string name) => Children(parent, name).FirstOrDefault(); + + /// The xlink:href of an image or link (namespace prefixes vary: l:, xlink:, none). + private static string? Href(XElement element) => + element.Attributes().FirstOrDefault(a => a.Name.LocalName == "href")?.Value; + + private BookMetadata Metadata(XElement root) + { + var info = Child(Child(root, "description"), "title-info"); + var metadata = new BookMetadata + { + Title = Child(info, "book-title")?.Value.Trim(), + Language = Child(info, "lang")?.Value.Trim() is { Length: > 0 } lang ? lang : null, + Description = Child(info, "annotation")?.Value.Trim() is { Length: > 0 } annotation ? annotation : null, + Publisher = Child(Child(Child(root, "description"), "publish-info"), "publisher")?.Value.Trim(), + }; + foreach (var author in Children(info, "author")) + { + var parts = new[] { "first-name", "middle-name", "last-name" }.Select(n => Child(author, n)?.Value.Trim()).Where(p => !string.IsNullOrEmpty(p)); + var name = string.Join(" ", parts); + if (name.Length == 0) + name = Child(author, "nickname")?.Value.Trim() ?? ""; + if (name.Length > 0) + metadata.Authors.Add(name); + } + if (Child(Child(info, "coverpage"), "image") is { } cover && Href(cover) is { } href && _binaries.TryGetValue(href.TrimStart('#'), out var file)) + metadata.CoverImage = file; + return metadata; + } + + private void SaveBinaries(XElement root, string folder) + { + foreach (var binary in Children(root, "binary")) + { + var id = (string?)binary.Attribute("id"); + if (string.IsNullOrEmpty(id)) + continue; + byte[] data; + try + { + data = Convert.FromBase64String(binary.Value.Trim()); + } + catch (FormatException) + { + continue; + } + var extension = ((string?)binary.Attribute("content-type"))?.ToLowerInvariant() switch + { + "image/png" => "png", + "image/gif" => "gif", + "image/jpeg" or "image/jpg" => "jpg", + _ => data is [0x89, (byte)'P', ..] ? "png" : data is [(byte)'G', (byte)'I', (byte)'F', ..] ? "gif" : "jpg", + }; + var file = Path.Combine(folder, $"binary{_binaries.Count + 1:0000}.{extension}"); + File.WriteAllBytes(file, data); + _binaries[id] = file; + } + } + + // ───────────────────────── Blocks ───────────────────────── + + private void Body(XElement body, List output) + { + foreach (var element in body.Elements()) + { + switch (element.Name.LocalName) + { + case "title": + Title(element, output, 1, center: true); + break; + case "epigraph": + Epigraph(element, output); + break; + case "image": + Image(element, output, 0); + break; + case "section": + Section(element, output, 1); + break; + } + } + } + + private void Section(XElement section, List output, int depth) + { + var start = output.Count; + foreach (var element in section.Elements()) + Block(element, output, depth, 0); + if (output.Count == start) + return; + // Every top-level section (a chapter) starts on a new page; bookmarks let links reach the section. + if (output[start] is not HParagraph first) + output.Insert(start, first = new HParagraph()); + first.PageBreakBefore = depth == 1 && start > 0; + if ((string?)section.Attribute("id") is { Length: > 0 } id) + first.Inlines.Insert(0, new HBookmark(id)); + } + + private void Block(XElement element, List output, int depth, int indent) + { + switch (element.Name.LocalName) + { + case "title": + Title(element, output, depth, center: false); + break; + case "section": + Section(element, output, depth + 1); + break; + case "p": + output.Add(Paragraph(element, new HParaFormat(Left: indent > 0 ? indent : null), default)); + break; + case "subtitle": + output.Add(Paragraph(element, new HParaFormat(Align: HAlign.Center), new HCharFormat(Bold: true))); + break; + case "empty-line": + output.Add(new HParagraph()); + break; + case "epigraph": + Epigraph(element, output); + break; + case "annotation" or "cite": + foreach (var child in element.Elements()) + Block(child, output, depth, indent + Indent); + break; + case "text-author": + output.Add(Paragraph(element, new HParaFormat(Align: HAlign.Right, Left: indent > 0 ? indent : null), new HCharFormat(Italic: true))); + break; + case "poem": + Poem(element, output, depth, indent + Indent); + break; + case "image": + Image(element, output, indent); + break; + case "table": + output.Add(Table(element)); + break; + } + } + + private void Title(XElement title, List output, int depth, bool center) + { + // A title may have several paragraphs; they form one heading, one line each. + var heading = new HParagraph + { + HeadingLevel = Math.Clamp(depth, 1, 6), + Format = center ? new HParaFormat(Align: HAlign.Center) : default, + }; + foreach (var paragraph in Children(title, "p")) + { + if (heading.Inlines.Count > 0) + heading.Inlines.Add(new HLineBreak(default)); + heading.Inlines.AddRange(Line(paragraph, default)); + } + if (heading.Inlines.Count > 0) + output.Add(heading); + } + + private void Epigraph(XElement epigraph, List output) + { + foreach (var child in epigraph.Elements()) + { + if (child.Name.LocalName == "p") + output.Add(Paragraph(child, new HParaFormat(Align: HAlign.Right), new HCharFormat(Italic: true))); + else + Block(child, output, 0, Indent); + } + } + + private void Poem(XElement poem, List output, int depth, int indent) + { + foreach (var child in poem.Elements()) + { + switch (child.Name.LocalName) + { + case "stanza": + { + // One paragraph per stanza, one line per verse. + var stanza = new HParagraph { Format = new HParaFormat(Left: indent) }; + foreach (var line in child.Elements()) + { + if (line.Name.LocalName is "title" or "subtitle") + { + output.Add(Paragraph(line.Name.LocalName == "title" ? Child(line, "p") ?? line : line, new HParaFormat(Left: indent), new HCharFormat(Bold: true))); + continue; + } + if (stanza.Inlines.Count > 0) + stanza.Inlines.Add(new HLineBreak(default)); + stanza.Inlines.AddRange(Line(line, default)); + } + if (stanza.Inlines.Count > 0) + output.Add(stanza); + break; + } + case "title": + Title(child, output, depth + 1, center: false); + break; + default: + Block(child, output, depth, indent); + break; + } + } + } + + private void Image(XElement image, List output, int indent) + { + if (Href(image) is not { } href || !_binaries.TryGetValue(href.TrimStart('#'), out var file)) + return; + var paragraph = new HParagraph { Format = new HParaFormat(Align: HAlign.Center, Left: indent > 0 ? indent : null) }; + paragraph.Inlines.Add(new HImage(file)); + output.Add(paragraph); + } + + private HTable Table(XElement table) + { + var hTable = new HTable(); + foreach (var row in Children(table, "tr")) + { + var cells = row.Elements().Where(c => c.Name.LocalName is "td" or "th").ToList(); + var hRow = new HRow { Header = cells.Count > 0 && cells.All(c => c.Name.LocalName == "th") }; + foreach (var cell in cells) + { + var hCell = new HCell + { + ColSpan = Math.Max(1, (int?)cell.Attribute("colspan") ?? 1), + RowSpan = Math.Max(1, (int?)cell.Attribute("rowspan") ?? 1), + }; + var align = ((string?)cell.Attribute("align"))?.ToLowerInvariant() switch + { + "center" => HAlign.Center, + "right" => HAlign.Right, + _ => (HAlign?)null, + }; + hCell.Blocks.Add(Paragraph(cell, new HParaFormat(Align: align), cell.Name.LocalName == "th" ? new HCharFormat(Bold: true) : default)); + hRow.Cells.Add(hCell); + } + if (hRow.Cells.Count > 0) + hTable.Rows.Add(hRow); + } + hTable.ColumnCount = Math.Max(1, hTable.Rows.Select(r => r.Cells.Sum(c => c.ColSpan)).DefaultIfEmpty(1).Max()); + return hTable; + } + + // ───────────────────────── Inlines ───────────────────────── + + private HParagraph Paragraph(XElement element, HParaFormat format, HCharFormat charFormat) + { + var paragraph = new HParagraph { Format = format }; + if ((string?)element.Attribute("id") is { Length: > 0 } id) + paragraph.Inlines.Add(new HBookmark(id)); + Inlines(element, paragraph.Inlines, charFormat); + HtmlReader.MergeRuns(paragraph.Inlines); + TrimEdges(paragraph.Inlines); + return paragraph; + } + + /// The inlines of one line (a title paragraph, a verse), edges trimmed. + private List Line(XElement element, HCharFormat format) + { + var inlines = new List(); + Inlines(element, inlines, format); + HtmlReader.MergeRuns(inlines); + TrimEdges(inlines); + return inlines; + } + + private static string CollapseWhitespace(string text) + { + var sb = new StringBuilder(text.Length); + foreach (var ch in text) + { + if (ch is not (' ' or '\n' or '\r' or '\t')) + sb.Append(ch); + else if (sb.Length == 0 || sb[^1] != ' ') + sb.Append(' '); + } + return sb.ToString(); + } + + /// Pretty-printed files indent paragraph text: spaces at the paragraph edges are dropped. + private static void TrimEdges(List inlines) + { + if (inlines.FindIndex(i => i is HText) is var first and >= 0 && inlines[first] is HText head) + inlines[first] = new HText(head.Text.TrimStart(), head.Format); + if (inlines.Count > 0 && inlines[^1] is HText tail) + inlines[^1] = new HText(tail.Text.TrimEnd(), tail.Format); + inlines.RemoveAll(i => i is HText { Text.Length: 0 }); + } + + private void Inlines(XElement element, List output, HCharFormat format) + { + foreach (var node in element.Nodes()) + { + if (node is XText text) + { + var value = CollapseWhitespace(text.Value); + if (value.Length > 0) + output.Add(new HText(value, format)); + continue; + } + if (node is not XElement child) + continue; + switch (child.Name.LocalName) + { + case "strong": + Inlines(child, output, format with { Bold = true }); + break; + case "emphasis": + Inlines(child, output, format with { Italic = true }); + break; + case "strikethrough": + Inlines(child, output, format with { Strike = true }); + break; + case "sub": + Inlines(child, output, format with { Subscript = true }); + break; + case "sup": + Inlines(child, output, format with { Superscript = true }); + break; + case "code": + Inlines(child, output, format with { Shade = HtmlReader.CodeShade }); + break; + case "image": + if (Href(child) is { } href && _binaries.TryGetValue(href.TrimStart('#'), out var file)) + output.Add(new HImage(file)); + break; + case "a": + Link(child, output, format); + break; + default: + Inlines(child, output, format); // style and unknown inline elements: their text + break; + } + } + } + + private void Link(XElement link, List output, HCharFormat format) + { + var href = Href(link) ?? ""; + if (href.StartsWith('#') && _notes.TryGetValue(href[1..], out var noteSection)) + { + // A note reference becomes a footnote with the note's text (its title is only the number). + var note = new HNote(endnote: false); + foreach (var child in noteSection.Elements().Where(e => e.Name.LocalName != "title")) + Block(child, note.Blocks, 0, 0); + if (note.Blocks.Count > 0) + { + output.Add(note); + return; + } + } + if (href.Length == 0) + { + Inlines(link, output, format); + return; + } + var hLink = new HLink(href); + Inlines(link, hLink.Content, format); + output.Add(hLink); + } +} diff --git a/src/Filee.Engines/Ebooks/Fb2Writer.cs b/src/Filee.Engines/Ebooks/Fb2Writer.cs new file mode 100644 index 0000000..54e0761 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Fb2Writer.cs @@ -0,0 +1,396 @@ +// Book → FictionBook 2 (FB2): headings become nested sections with titles, paragraphs keep bold / italic / +// strikethrough / sub / sup / code and links, pictures become base64 binaries (JPEG or PNG), tables stay tables, +// footnotes go to a "notes" body. FB2 has no lists or line breaks: list markers are written out and a line break +// starts a new paragraph. + +using System.Text; +using System.Xml; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal sealed class Fb2Writer +{ + private const string Fb2 = "http://www.gribuser.ru/xml/fictionbook/2.0"; + private const string XLink = "http://www.w3.org/1999/xlink"; + + private readonly string _imageFolder; + private readonly Dictionary _binaries = new(StringComparer.OrdinalIgnoreCase); + private readonly List> _notes = []; + private readonly ListNumbers _numbers = new(); + + private Fb2Writer(string imageFolder) => _imageFolder = imageFolder; + + /// A section of the output: its title and content (paragraphs, then sub-sections). + private sealed class Section(HParagraph? title) + { + public HParagraph? Title { get; } = title; + public int Level { get; init; } + public List Blocks { get; } = []; + public List
Children { get; } = []; + } + + public static void Write(Book book, string outputPath, string sourcePath, string workFolder) + { + var writer = new Fb2Writer(EbookFiles.NewFolder(workFolder, "fb2-images")); + var root = writer.Tree(book.Document); + var settings = new XmlWriterSettings { Encoding = new UTF8Encoding(false), Indent = true, IndentChars = " " }; + var temp = outputPath + ".tmp"; + using (var xml = XmlWriter.Create(temp, settings)) + { + xml.WriteStartDocument(); + xml.WriteStartElement("FictionBook", Fb2); + xml.WriteAttributeString("xmlns", "l", null, XLink); + writer.Description(xml, book, sourcePath); + + xml.WriteStartElement("body", Fb2); + foreach (var section in root.Children) + writer.WriteSection(xml, section); + if (root.Children.Count == 0) + { + xml.WriteStartElement("section", Fb2); + xml.WriteElementString("empty-line", Fb2, null); + xml.WriteEndElement(); + } + xml.WriteEndElement(); + + if (writer._notes.Count > 0) + { + xml.WriteStartElement("body", Fb2); + xml.WriteAttributeString("name", "notes"); + for (var i = 0; i < writer._notes.Count; i++) + { + xml.WriteStartElement("section", Fb2); + xml.WriteAttributeString("id", $"n{i + 1}"); + xml.WriteStartElement("title", Fb2); + xml.WriteElementString("p", Fb2, $"{i + 1}"); + xml.WriteEndElement(); + writer.Blocks(xml, writer._notes[i]); + xml.WriteEndElement(); + } + xml.WriteEndElement(); + } + + foreach (var (id, file, mediaType) in writer._binaries.Values) + { + xml.WriteStartElement("binary", Fb2); + xml.WriteAttributeString("id", id); + xml.WriteAttributeString("content-type", mediaType); + xml.WriteString(Convert.ToBase64String(File.ReadAllBytes(file), Base64FormattingOptions.InsertLineBreaks)); + xml.WriteEndElement(); + } + xml.WriteEndElement(); + xml.WriteEndDocument(); + } + File.Move(temp, outputPath, overwrite: true); + } + + /// Nests the blocks into sections by heading level (a level-2 heading opens a section in the level-1 one). + private Section Tree(HDocument document) + { + var root = new Section(null) { Level = 0 }; + var stack = new List
{ root }; + foreach (var block in document.Sections.SelectMany(s => s.Blocks)) + { + if (block is HParagraph { HeadingLevel: > 0 } heading) + { + while (stack[^1].Level >= heading.HeadingLevel) + stack.RemoveAt(stack.Count - 1); + var section = new Section(heading) { Level = heading.HeadingLevel }; + stack[^1].Children.Add(section); + stack.Add(section); + continue; + } + // Content before the first heading (or between a heading and its first sub-heading) of the root goes + // into an untitled section. + if (stack.Count == 1) + { + var untitled = new Section(null) { Level = 1 }; + root.Children.Add(untitled); + stack.Add(untitled); + } + stack[^1].Blocks.Add(block); + } + return root; + } + + private void Description(XmlWriter xml, Book book, string sourcePath) + { + xml.WriteStartElement("description", Fb2); + xml.WriteStartElement("title-info", Fb2); + xml.WriteElementString("genre", Fb2, "antique"); // required; calibre uses the same neutral default + var authors = book.Metadata.Authors.Count > 0 ? book.Metadata.Authors : [""]; + foreach (var author in authors) + { + xml.WriteStartElement("author", Fb2); + var parts = author.Split(' ', StringSplitOptions.RemoveEmptyEntries); + if (parts.Length >= 2) + { + xml.WriteElementString("first-name", Fb2, string.Join(" ", parts[..^1])); + xml.WriteElementString("last-name", Fb2, parts[^1]); + } + else + { + xml.WriteElementString("nickname", Fb2, parts.Length == 1 ? parts[0] : "Unknown"); + } + xml.WriteEndElement(); + } + xml.WriteElementString("book-title", Fb2, book.DisplayTitle(sourcePath)); + if (book.Metadata.Description is { } description) + { + xml.WriteStartElement("annotation", Fb2); + xml.WriteElementString("p", Fb2, Clean(description)); + xml.WriteEndElement(); + } + xml.WriteElementString("lang", Fb2, book.DisplayLanguage().Split('-')[0]); + if (book.Metadata.CoverImage is { } cover && Binary(cover) is { } coverId) + { + xml.WriteStartElement("coverpage", Fb2); + ImageElement(xml, coverId); + xml.WriteEndElement(); + } + xml.WriteEndElement(); + + xml.WriteStartElement("document-info", Fb2); + xml.WriteStartElement("author", Fb2); + xml.WriteElementString("nickname", Fb2, "Filee"); + xml.WriteEndElement(); + xml.WriteElementString("program-used", Fb2, "Filee"); + xml.WriteStartElement("date", Fb2); + xml.WriteAttributeString("value", DateTime.Now.ToString("yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture)); + xml.WriteString(DateTime.Now.ToString("yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture)); + xml.WriteEndElement(); + xml.WriteElementString("id", Fb2, Guid.NewGuid().ToString()); + xml.WriteElementString("version", Fb2, "1.0"); + xml.WriteEndElement(); + xml.WriteEndElement(); + } + + private void WriteSection(XmlWriter xml, Section section) + { + xml.WriteStartElement("section", Fb2); + if (section.Title is { } title && Bookmark(title.Inlines) is { } id) + xml.WriteAttributeString("id", id); + if (section.Title is not null) + { + xml.WriteStartElement("title", Fb2); + Paragraphs(xml, section.Title.Inlines.Where(i => i is not HBookmark).ToList(), "p"); // the id is on the section + xml.WriteEndElement(); + } + // FB2 sections hold either content or sub-sections: leading content moves into an untitled section. + if (section.Children.Count > 0 && section.Blocks.Count > 0) + { + xml.WriteStartElement("section", Fb2); + Blocks(xml, section.Blocks); + xml.WriteEndElement(); + } + else if (section.Blocks.Count > 0) + { + Blocks(xml, section.Blocks); + } + foreach (var child in section.Children) + WriteSection(xml, child); + if (section.Blocks.Count == 0 && section.Children.Count == 0) + xml.WriteElementString("empty-line", Fb2, null); + xml.WriteEndElement(); + } + + private void Blocks(XmlWriter xml, List blocks) + { + foreach (var block in blocks) + { + switch (block) + { + case HParagraph paragraph: + Paragraph(xml, paragraph); + break; + case HTable table: + Table(xml, table); + break; + } + } + } + + private void Paragraph(XmlWriter xml, HParagraph paragraph) + { + var visible = HDocumentWalker.InlinesOf(paragraph.Inlines).Any(i => i is HImage or HNote || i is HText { Text: var t } && t.Trim().Length > 0); + if (!visible) + { + if (paragraph.Inlines.Any(i => i is HShape { Kind: HShapeKind.Line })) + xml.WriteElementString("subtitle", Fb2, "* * *"); + else + xml.WriteElementString("empty-line", Fb2, null); + return; + } + // A picture on its own line is a block image. + if (paragraph.Inlines.Where(i => i is not HBookmark).ToList() is [HImage only] && Binary(only.Path) is { } imageId) + { + ImageElement(xml, imageId); + return; + } + var inlines = paragraph.Inlines; + if (paragraph.List is { } list) + { + var marker = list.Numbered ? _numbers.Next(list) + " " : ""; + inlines = [new HText(new string(' ', list.Level * 4) + marker, default), .. inlines]; + } + var element = paragraph.Format.Align == HAlign.Center && paragraph.Inlines.OfType().All(t => t.Format.Bold is true) ? "subtitle" : "p"; + Paragraphs(xml, inlines, element); + } + + /// Writes inlines as one element per line (FB2 paragraphs have no line breaks). + private void Paragraphs(XmlWriter xml, List inlines, string element) + { + var lines = new List> { new() }; + foreach (var inline in inlines) + { + if (inline is HLineBreak) + lines.Add([]); + else + lines[^1].Add(inline); + } + foreach (var line in lines) + { + xml.WriteStartElement(element, Fb2); + if (Bookmark(line) is { } id) + xml.WriteAttributeString("id", id); + Inlines(xml, line); + xml.WriteEndElement(); + } + } + + private static string? Bookmark(IEnumerable inlines) => + inlines.OfType().Select(b => Id(b.Name)).FirstOrDefault(); + + private static string Id(string name) + { + var id = new string(name.Select(c => char.IsAsciiLetterOrDigit(c) || c is '-' or '_' or '.' ? c : '_').ToArray()); + return id.Length > 0 && char.IsAsciiLetter(id[0]) ? id : "id-" + id; + } + + private void Inlines(XmlWriter xml, List inlines) + { + foreach (var inline in inlines) + { + switch (inline) + { + case HText text: + Run(xml, text.Text, text.Format); + break; + case HTab: + xml.WriteString(" "); + break; + case HLink link: + { + var target = link.Target.StartsWith('#') ? "#" + Id(link.Target[1..]) : link.Target; + xml.WriteStartElement("a", Fb2); + xml.WriteAttributeString("href", XLink, target); + Inlines(xml, link.Content.Where(i => i is not HLineBreak).ToList()); + xml.WriteEndElement(); + break; + } + case HImage image: + if (Binary(image.Path) is { } id) + ImageElement(xml, id); + break; + case HNote note: + _notes.Add(note.Blocks); + xml.WriteStartElement("a", Fb2); + xml.WriteAttributeString("href", XLink, $"#n{_notes.Count}"); + xml.WriteAttributeString("type", "note"); + xml.WriteString($"[{_notes.Count}]"); + xml.WriteEndElement(); + break; + case HTextBox box: + Inlines(xml, box.Blocks.OfType().SelectMany(p => p.Inlines.Prepend(new HText(" ", default))).Where(i => i is not HLineBreak).ToList()); + break; + } + } + } + + private static void Run(XmlWriter xml, string text, HCharFormat format) + { + var open = 0; + void Open(string name) + { + xml.WriteStartElement(name, Fb2); + open++; + } + if (format.Shade == HtmlReader.CodeShade) + Open("code"); + if (format.Bold is true) + Open("strong"); + if (format.Italic is true) + Open("emphasis"); + if (format.Strike is true) + Open("strikethrough"); + if (format.Superscript is true) + Open("sup"); + else if (format.Subscript is true) + Open("sub"); + xml.WriteString(Clean(text)); + for (; open > 0; open--) + xml.WriteEndElement(); + } + + private void Table(XmlWriter xml, HTable table) + { + xml.WriteStartElement("table", Fb2); + foreach (var row in table.Rows) + { + xml.WriteStartElement("tr", Fb2); + foreach (var cell in row.Cells) + { + xml.WriteStartElement(row.Header ? "th" : "td", Fb2); + if (cell.ColSpan > 1) + xml.WriteAttributeString("colspan", $"{cell.ColSpan}"); + if (cell.RowSpan > 1) + xml.WriteAttributeString("rowspan", $"{cell.RowSpan}"); + var paragraphs = cell.Blocks.OfType().ToList(); + for (var i = 0; i < paragraphs.Count; i++) + { + if (i > 0) + xml.WriteString(" "); + Inlines(xml, paragraphs[i].Inlines.Where(x => x is not HLineBreak and not HImage).ToList()); + } + xml.WriteEndElement(); + } + xml.WriteEndElement(); + } + xml.WriteEndElement(); + } + + private static void ImageElement(XmlWriter xml, string id) + { + xml.WriteStartElement("image", Fb2); + xml.WriteAttributeString("href", XLink, "#" + id); + xml.WriteEndElement(); + } + + /// The binary id of a picture (JPEG or PNG, converted when needed), or null when it cannot be read. + private string? Binary(string path) + { + if (_binaries.TryGetValue(path, out var known)) + return known.Id; + if (EbookFiles.CommonImage(path, _imageFolder, gifAndSvg: false) is not ({ } file, { } mediaType)) + return null; + var id = $"img{_binaries.Count + 1}.{(mediaType == "image/png" ? "png" : "jpg")}"; + _binaries[path] = (id, file, mediaType); + return id; + } + + /// Removes characters XML 1.0 does not allow (e.g. vertical tabs from Word). + private static string Clean(string text) + { + var sb = new StringBuilder(text.Length); + for (var i = 0; i < text.Length; i++) + { + var c = text[i]; + if (char.IsHighSurrogate(c) && i + 1 < text.Length && char.IsLowSurrogate(text[i + 1])) + sb.Append(c).Append(text[++i]); + else if (!char.IsSurrogate(c) && (c is '\t' or '\n' or '\r' || (c >= 0x20 && c != 0xFFFE && c != 0xFFFF))) + sb.Append(c); + } + return sb.ToString(); + } +} diff --git a/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs b/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs new file mode 100644 index 0000000..d80ca56 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs @@ -0,0 +1,128 @@ +// HUFF/CDIC compression of MOBI text records (Mobipocket "Huffman dictionary" compression), written from the +// format description on the MobileRead wiki: a canonical Huffman code (HUFF record) selects phrases from the +// dictionary (CDIC records); a phrase may itself be compressed and is unpacked on first use. + +using System.Buffers.Binary; + +namespace Filee.Engines.Ebooks; + +internal sealed class HuffCdic +{ + /// Per leading byte of a code: length, whether that length is final, and the adjusted max code. + private readonly (int Length, bool Terminal, ulong MaxCode)[] _table1 = new (int, bool, ulong)[256]; + private readonly ulong[] _minCode = new ulong[33]; + private readonly ulong[] _maxCode = new ulong[33]; + private readonly List _phrases = []; + + private sealed class Phrase(byte[] data, bool unpacked) + { + public byte[] Data { get; set; } = data; + public bool Unpacked { get; set; } = unpacked; + public bool Busy { get; set; } + } + + /// The HUFF record. + /// The CDIC records that follow it. + public HuffCdic(ReadOnlySpan huff, IEnumerable cdics) + { + if (huff.Length < 24 || !huff[..4].SequenceEqual("HUFF"u8)) + throw new InvalidDataException("The book's HUFF compression table is damaged."); + var table1 = (int)BinaryPrimitives.ReadUInt32BigEndian(huff[8..]); + var table2 = (int)BinaryPrimitives.ReadUInt32BigEndian(huff[12..]); + if (table1 + 256 * 4 > huff.Length || table2 + 64 * 4 > huff.Length) + throw new InvalidDataException("The book's HUFF compression table is damaged."); + + for (var i = 0; i < 256; i++) + { + var value = BinaryPrimitives.ReadUInt32BigEndian(huff[(table1 + i * 4)..]); + var length = (int)(value & 0x1F); + if (length == 0) + throw new InvalidDataException("The book's HUFF compression table is damaged."); + var maxCode = (((ulong)(value >> 8) + 1) << (32 - length)) - 1; + _table1[i] = (length, (value & 0x80) != 0, maxCode); + } + // Code lengths 1..32: the smallest and largest code of each length, left-aligned in 32 bits. + for (var length = 1; length <= 32; length++) + { + var min = BinaryPrimitives.ReadUInt32BigEndian(huff[(table2 + (length - 1) * 8)..]); + var max = BinaryPrimitives.ReadUInt32BigEndian(huff[(table2 + (length - 1) * 8 + 4)..]); + _minCode[length] = (ulong)min << (32 - length); + _maxCode[length] = (((ulong)max + 1) << (32 - length)) - 1; + } + + foreach (var cdic in cdics) + { + if (cdic.Length < 16 || !cdic.AsSpan(0, 4).SequenceEqual("CDIC"u8)) + throw new InvalidDataException("The book's CDIC dictionary is damaged."); + var total = (int)BinaryPrimitives.ReadUInt32BigEndian(cdic.AsSpan(8)); + var bits = (int)BinaryPrimitives.ReadUInt32BigEndian(cdic.AsSpan(12)); + var count = Math.Min(1 << Math.Clamp(bits, 0, 20), total - _phrases.Count); + for (var i = 0; i < count; i++) + { + var offset = 16 + PalmDatabase.U16(cdic, 16 + i * 2); + var header = PalmDatabase.U16(cdic, offset); + var length = Math.Min(header & 0x7FFF, Math.Max(0, cdic.Length - offset - 2)); + _phrases.Add(new Phrase(cdic.AsSpan(offset + 2, length).ToArray(), (header & 0x8000) != 0)); + } + } + } + + /// Decompresses one text record. + public byte[] Decompress(ReadOnlySpan data) + { + var output = new List(data.Length * 3); + Unpack(data, output, 0); + return [.. output]; + } + + private void Unpack(ReadOnlySpan data, List output, int depth) + { + if (depth > 32) + throw new InvalidDataException("The book's CDIC dictionary refers to itself."); + // Read 64 bits at a time; "n" counts the bits of the current 32-bit window not yet consumed. + Span padded = new byte[data.Length + 8]; + data.CopyTo(padded); + long bitsLeft = data.Length * 8L; + var position = 0; + var window = BinaryPrimitives.ReadUInt64BigEndian(padded); + var n = 32; + while (true) + { + if (n <= 0) + { + position += 4; + window = position + 8 <= padded.Length ? BinaryPrimitives.ReadUInt64BigEndian(padded[position..]) : 0; + n += 32; + } + var code = (window >> n) & 0xFFFFFFFF; + var (length, terminal, maxCode) = _table1[code >> 24]; + if (!terminal) + { + while (length < 32 && code < _minCode[length]) + length++; + maxCode = _maxCode[length]; + } + n -= length; + bitsLeft -= length; + if (bitsLeft < 0) + break; + + var index = (int)((maxCode - code) >> (32 - length)); + if (index < 0 || index >= _phrases.Count) + throw new InvalidDataException("The book's compressed text is damaged."); + var phrase = _phrases[index]; + if (!phrase.Unpacked) + { + if (phrase.Busy) + throw new InvalidDataException("The book's CDIC dictionary refers to itself."); + phrase.Busy = true; + var unpacked = new List(); + Unpack(phrase.Data, unpacked, depth + 1); + phrase.Data = [.. unpacked]; + phrase.Unpacked = true; + phrase.Busy = false; + } + output.AddRange(phrase.Data); + } + } +} diff --git a/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs b/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs new file mode 100644 index 0000000..442ba28 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs @@ -0,0 +1,180 @@ +// INDX records of Kindle books: the skeleton and fragment tables of KF8 (AZW3) text are stored as indexes. Written +// from the format description on the MobileRead wiki: a primary INDX record with the tag table (TAGX), then INDX +// records whose entries (located through the IDXT offset table) are a key followed by control bytes and +// variable-width tag values; strings live in CNCX records. + +using System.Text; + +namespace Filee.Engines.Ebooks; + +/// One index entry: its key and tag values. +internal sealed record MobiIndexEntry(string Key, IReadOnlyDictionary> Tags) +{ + public long Tag(int tag, int position = 0) => + Tags.TryGetValue(tag, out var values) && position < values.Count ? values[position] : -1; +} + +internal static class MobiIndex +{ + private readonly record struct TagDefinition(int Tag, int ValuesPerEntry, int Mask, bool EndFlag); + + /// Reads the index starting at record ; CNCX strings go to . + public static List Read(PalmDatabase database, int first, Dictionary strings) + { + var entries = new List(); + var primary = database.Record(first); + if (primary.Length < 56 || !primary[..4].SequenceEqual("INDX"u8)) + throw new InvalidDataException("The book's index is damaged."); + var headerLength = (int)PalmDatabase.U32(primary, 4); + var recordCount = (int)PalmDatabase.U32(primary, 24); + var cncxCount = (int)Math.Max(0, PalmDatabase.U32(primary, 52)); + var (controlBytes, tags) = TagTable(primary, headerLength); + + // CNCX records follow the index records; offsets in tag values address them as record * 0x10000 + offset. + for (var c = 0; c < cncxCount; c++) + { + var cncx = database.Record(first + recordCount + 1 + c); + var offset = 0; + while (offset < cncx.Length && cncx[offset] != 0) + { + var start = offset; + var (length, consumed) = ForwardVarint(cncx, offset); + offset += consumed; + length = Math.Min(length, cncx.Length - offset); + strings[c * 0x10000L + start] = Encoding.UTF8.GetString(cncx.Slice(offset, (int)length)); + offset += (int)length; + } + } + + for (var r = 1; r <= recordCount; r++) + { + var record = database.Record(first + r); + if (record.Length < 28 || !record[..4].SequenceEqual("INDX"u8)) + continue; + var idxt = (int)PalmDatabase.U32(record, 20); + var count = (int)PalmDatabase.U32(record, 24); + if (idxt < 0 || idxt + 4 + count * 2 > record.Length) + continue; + var offsets = new int[count + 1]; + for (var i = 0; i < count; i++) + offsets[i] = PalmDatabase.U16(record, idxt + 4 + i * 2); + offsets[count] = idxt; + for (var i = 0; i < count; i++) + { + var start = offsets[i]; + if (start >= record.Length) + continue; + var keyLength = record[start]; + var key = Encoding.Latin1.GetString(record.Slice(start + 1, Math.Min(keyLength, record.Length - start - 1))); + var values = TagValues(record, start + 1 + keyLength, offsets[i + 1], controlBytes, tags); + entries.Add(new MobiIndexEntry(key, values)); + } + } + return entries; + } + + private static (int ControlBytes, List Tags) TagTable(ReadOnlySpan record, int start) + { + var tags = new List(); + if (start < 0 || start + 12 > record.Length || !record.Slice(start, 4).SequenceEqual("TAGX"u8)) + throw new InvalidDataException("The book's index has no tag table."); + var length = (int)PalmDatabase.U32(record, start + 4); + var controlBytes = (int)PalmDatabase.U32(record, start + 8); + for (var i = 12; i + 4 <= length && start + i + 4 <= record.Length; i += 4) + { + var p = start + i; + tags.Add(new TagDefinition(record[p], record[p + 1], record[p + 2], record[p + 3] == 1)); + } + return (controlBytes, tags); + } + + /// + /// Tag values of one entry. Each tag's mask selects bits of a control byte: a partial value is the number of + /// value groups; all bits set with a multi-bit mask means a byte count follows; a one-bit mask means one group. + /// + private static Dictionary> TagValues(ReadOnlySpan record, int start, int end, int controlBytes, List tags) + { + var result = new Dictionary>(); + var controlIndex = 0; + var position = start + controlBytes; + var headers = new List<(int Tag, int? Count, int? Bytes, int PerEntry)>(); + foreach (var tag in tags) + { + if (tag.EndFlag) + { + controlIndex++; + continue; + } + if (start + controlIndex >= record.Length) + break; + var value = record[start + controlIndex] & tag.Mask; + if (value == 0) + continue; + if (value == tag.Mask) + { + if (System.Numerics.BitOperations.PopCount((uint)tag.Mask) > 1) + { + var (bytes, consumed) = ForwardVarint(record, position); + position += consumed; + headers.Add((tag.Tag, null, (int)bytes, tag.ValuesPerEntry)); + } + else + { + headers.Add((tag.Tag, 1, null, tag.ValuesPerEntry)); + } + } + else + { + var mask = tag.Mask; + while ((mask & 1) == 0) + { + mask >>= 1; + value >>= 1; + } + headers.Add((tag.Tag, value, null, tag.ValuesPerEntry)); + } + } + + foreach (var (tag, count, bytes, perEntry) in headers) + { + var values = new List(); + if (count is { } groups) + { + for (var i = 0; i < groups * perEntry && position < end; i++) + { + var (value, consumed) = ForwardVarint(record, position); + position += consumed; + values.Add(value); + } + } + else + { + var read = 0; + while (read < bytes && position < end) + { + var (value, consumed) = ForwardVarint(record, position); + position += consumed; + read += consumed; + values.Add(value); + } + } + result[tag] = values; + } + return result; + } + + /// A forward variable-width integer: 7 bits per byte, the last byte has the high bit set. + internal static (long Value, int Consumed) ForwardVarint(ReadOnlySpan data, int offset) + { + long value = 0; + var consumed = 0; + while (offset + consumed < data.Length && consumed < 9) + { + var b = data[offset + consumed++]; + value = (value << 7) | (uint)(b & 0x7F); + if ((b & 0x80) != 0) + break; + } + return (value, Math.Max(1, consumed)); + } +} diff --git a/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs b/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs new file mode 100644 index 0000000..0807e32 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs @@ -0,0 +1,616 @@ +// MOBI / AZW / AZW3 / PRC → Book, and the PDF inside AZW4 (Print Replica). Written from the MOBI format description +// on the MobileRead wiki (no code from KindleUnpack or calibre): +// * PalmDB records; record 0 holds the PalmDOC header, the MOBI header and EXTH metadata (title, authors, cover); +// * text records are PalmDOC (LZ77) or HUFF/CDIC compressed, with trailing entries stripped first; +// * MOBI 6 (KF7): one HTML document; filepos links become anchors, recindex pictures come from image records; +// * KF8 (AZW3, also the KF8 half of joint MOBI files): the text is cut into files by the skeleton (SKEL) and +// fragment (FRAG) indexes; kindle:pos, kindle:embed and kindle:flow references are resolved; +// * plain PalmDOC books (TEXtREAd) are text. +// Books with DRM are refused with a clear message; KFX and Topaz books are recognised and refused too. + +using System.Buffers.Binary; +using System.Text; +using System.Text.RegularExpressions; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal static partial class MobiReader +{ + /// Reads a Kindle book's content (for the reader registry of the HWPX writer). + public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document; + + public static Book ReadBook(string path, string workFolder, CancellationToken cancellationToken = default) + { + var database = Open(path); + var folder = EbookFiles.NewFolder(workFolder, "mobi"); + if (database.TypeCreator == "TEXtREAd") + return new Book(PlainText.ToDocument(DecodeText(PalmDocText(database))), new BookMetadata { Title = database.Name }); + + var header = MobiHeader.Parse(database, 0); + if (header.Encryption != 0) + throw new InvalidOperationException(EbookFiles.DrmMessage); + var metadata = header.Metadata(database.Name); + + HDocument document; + Resources resources; + var kf8Start = header.Version >= 8 ? 0 : header.Kf8Boundary; + if (kf8Start is { } start && start >= 0 && start < database.Count) + { + var kf8 = start == 0 ? header : MobiHeader.Parse(database, start); + if (kf8.Encryption != 0) + throw new InvalidOperationException(EbookFiles.DrmMessage); + // Joint files keep the pictures once, before the KF8 half, numbered from the MOBI 6 header. + resources = new Resources(database, start == 0 ? kf8.FirstImage : header.FirstImage, folder); + document = Kf8(database, kf8, start, resources, folder, cancellationToken); + } + else + { + resources = new Resources(database, header.FirstImage, folder); + document = Mobi6(database, header, resources, folder, cancellationToken); + } + + if (header.CoverOffset is { } cover && resources.Save(cover) is { } coverFile) + metadata.CoverImage = coverFile; + document.Title = metadata.Title; + return new Book(document, metadata); + } + + /// The first PDF of an AZW4 (Print Replica) book. + public static byte[] PrintReplicaPdf(string path) + { + var database = Open(path); + var header = MobiHeader.Parse(database, 0); + if (header.Encryption != 0) + throw new InvalidOperationException(EbookFiles.DrmMessage); + var text = Text(database, header, 0); + if (text.AsSpan().StartsWith("%MOP"u8) && text.Length >= 12) + { + // "%MOP", table count, section count per table, then (offset, length) per section; the first is the PDF. + var tables = BinaryPrimitives.ReadUInt32BigEndian(text.AsSpan(4)); + var index = 8 + 4 * (int)Math.Min(tables, 1000); + var offset = (int)PalmDatabase.U32(text, index); + var length = (int)PalmDatabase.U32(text, index + 4); + if (offset > 0 && length > 0 && offset + length <= text.Length) + return text.AsSpan(offset, length).ToArray(); + } + var pdf = text.AsSpan().IndexOf("%PDF-"u8); + if (pdf < 0) + throw new InvalidDataException("This book is not a Print Replica (AZW4) book: it contains no PDF."); + return text.AsSpan(pdf).ToArray(); + } + + private static PalmDatabase Open(string path) + { + var data = File.ReadAllBytes(path); + if (data is [(byte)'C', (byte)'O', (byte)'N', (byte)'T', ..] or [0xEA, (byte)'D', (byte)'R', (byte)'M', (byte)'I', (byte)'O', (byte)'N', ..]) + throw new InvalidDataException("This is a KFX book (Kindle's newest format), which cannot be converted. Download it from Amazon as an AZW3 or MOBI file (\"Download & transfer via USB\")."); + if (data.AsSpan().StartsWith("TPZ"u8)) + throw new InvalidDataException("This is a Topaz book, which cannot be converted."); + var database = PalmDatabase.Open(data); + if (database.TypeCreator is not ("BOOKMOBI" or "TEXtREAd")) + throw new InvalidDataException($"This is not a Mobipocket or Kindle book (type \"{database.TypeCreator}\")."); + return database; + } + + // ───────────────────────── Text records ───────────────────────── + + /// The decompressed text of the book whose header is at record . + private static byte[] Text(PalmDatabase database, MobiHeader header, int start) + { + Func, byte[]> decompress = header.Compression switch + { + 1 => data => data.ToArray(), + 2 => PalmDocCompression.Decompress, + 17480 => HuffCdicReader(database, header, start).Decompress, + var other => throw new InvalidDataException($"The book uses an unknown compression ({other})."), + }; + using var text = new MemoryStream(); + for (var i = 1; i <= header.TextRecords; i++) + { + var record = database.Record(start + i); + var size = record.Length - TrailingSize(record, header.ExtraFlags); + text.Write(decompress(record[..Math.Max(0, size)])); + } + var bytes = text.ToArray(); + return header.TextLength > 0 && header.TextLength < bytes.Length ? bytes[..(int)header.TextLength] : bytes; + } + + private static HuffCdic HuffCdicReader(PalmDatabase database, MobiHeader header, int start) + { + var huff = start + header.HuffRecord; + return new HuffCdic(database.Record(huff), Enumerable.Range(huff + 1, Math.Max(0, header.HuffCount - 1)).Select(database.RecordArray)); + } + + /// + /// Size of the extra data at the end of a text record (multibyte overlap and indexing entries), described by the + /// header's extra flags: each set bit above bit 0 adds an entry whose size is a backward varint at the end. + /// + internal static int TrailingSize(ReadOnlySpan record, int flags) + { + var size = 0; + for (var bits = flags >> 1; bits != 0; bits >>= 1) + { + if ((bits & 1) == 0) + continue; + // Backward varint: 7 bits per byte read from the end, the first byte (from the start) has the high bit set. + var value = 0; + var shift = 0; + for (var p = record.Length - size - 1; p >= 0; p--) + { + var b = record[p]; + value |= (b & 0x7F) << shift; + shift += 7; + if ((b & 0x80) != 0 || shift >= 28 || p == 0) + break; + } + size += value; + } + if ((flags & 1) != 0 && record.Length - size - 1 >= 0) + size += (record[record.Length - size - 1] & 0x3) + 1; + return Math.Clamp(size, 0, record.Length); + } + + private static byte[] PalmDocText(PalmDatabase database) + { + var header = database.Record(0); + var compression = PalmDatabase.U16(header, 0); + var records = PalmDatabase.U16(header, 8); + using var text = new MemoryStream(); + for (var i = 1; i <= records && i < database.Count; i++) + text.Write(compression == 2 ? PalmDocCompression.Decompress(database.Record(i)) : database.Record(i)); + return text.ToArray(); + } + + /// UTF-8 when valid, else Windows-1252 (the Mobipocket default). + private static string DecodeText(byte[] bytes) + { + try + { + return new UTF8Encoding(false, true).GetString(bytes); + } + catch (DecoderFallbackException) + { + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + return Encoding.GetEncoding(1252).GetString(bytes); + } + } + + // ───────────────────────── MOBI 6 ───────────────────────── + + private static HDocument Mobi6(PalmDatabase database, MobiHeader header, Resources resources, string folder, CancellationToken cancellationToken) + { + // Latin-1 keeps one character per byte, so filepos values (byte offsets) index the string directly. + var text = Encoding.Latin1.GetString(Text(database, header, 0)); + cancellationToken.ThrowIfCancellationRequested(); + + var positions = FilePos().Matches(text).Select(m => long.Parse(m.Groups[1].Value)).Where(p => p <= text.Length).Distinct().OrderDescending(); + var sb = new StringBuilder(text); + foreach (var position in positions) + { + // A position inside a tag moves to the start of that tag. + var at = (int)position; + var open = at > 0 ? text.LastIndexOf('<', at - 1) : -1; + var close = at > 0 ? text.LastIndexOf('>', at - 1) : -1; + if (open > close) + at = open; + sb.Insert(at, $""); + } + var html = FilePos().Replace(sb.ToString(), m => $"href=\"#filepos{long.Parse(m.Groups[1].Value)}\""); + html = RecIndex().Replace(html, m => resources.Save(int.Parse(m.Groups[1].Value) - 1) is { } file ? $"src=\"{Path.GetFileName(file)}\"" : ""); + + var encoding = header.Codepage == 65001 ? Encoding.UTF8 : CodePage(header.Codepage); + var page = Path.Combine(folder, "book.html"); + File.WriteAllText(page, encoding.GetString(Encoding.Latin1.GetBytes(html)), new UTF8Encoding(true)); + return ChapterReader.Read([page], Path.Combine(folder, "~media"), cancellationToken); + } + + private static Encoding CodePage(int codepage) + { + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + try + { + return Encoding.GetEncoding(codepage is 0 ? 1252 : codepage); + } + catch (Exception ex) when (ex is ArgumentException or NotSupportedException) + { + return Encoding.GetEncoding(1252); + } + } + + // Digit counts are capped so the numbers always fit (filepos values are 10 digits, recindex 5). + [GeneratedRegex(@"\bfilepos\s*=\s*['""]?0*(\d{1,12})['""]?", RegexOptions.IgnoreCase)] + private static partial Regex FilePos(); + + [GeneratedRegex(@"\b(?:hi|lo)?recindex\s*=\s*['""]?(\d{1,9})['""]?", RegexOptions.IgnoreCase)] + private static partial Regex RecIndex(); + + // ───────────────────────── KF8 ───────────────────────── + + private static HDocument Kf8(PalmDatabase database, MobiHeader header, int start, Resources resources, string folder, CancellationToken cancellationToken) + { + var text = Text(database, header, start); + cancellationToken.ThrowIfCancellationRequested(); + + // Flows: flow 0 is the XHTML text, the others are style sheets and SVG images. + var flows = new List(); + if (header.Fdst >= 0 && database.Record(start + (int)header.Fdst) is var fdst && fdst.Length >= 12 && fdst[..4].SequenceEqual("FDST"u8)) + { + var tableOffset = (int)PalmDatabase.U32(fdst, 4); + var count = (int)PalmDatabase.U32(fdst, 8); + for (var i = 0; i < count; i++) + { + var from = (int)Math.Clamp(PalmDatabase.U32(fdst, tableOffset + i * 8), 0, text.Length); + var to = (int)Math.Clamp(PalmDatabase.U32(fdst, tableOffset + i * 8 + 4), from, text.Length); + flows.Add(text[from..to]); + } + } + if (flows.Count == 0) + flows.Add(text); + var markup = flows[0]; + + var strings = new Dictionary(); + var skeletons = header.Skeleton >= 0 ? MobiIndex.Read(database, start + (int)header.Skeleton, strings) : []; + var fragments = header.Fragment >= 0 ? MobiIndex.Read(database, start + (int)header.Fragment, strings) : []; + var parts = Assemble(markup, skeletons, fragments, strings); + + // Work on Latin-1 strings: positions in kindle:pos links are byte offsets. + var texts = parts.Select(p => Encoding.Latin1.GetString(p.Text)).ToList(); + var insertPositions = fragments.Select(f => long.TryParse(f.Key, out var p) ? p : 0).ToList(); + var linkedAids = new HashSet(StringComparer.Ordinal); + string Resolve(Match m) + { + var fragment = (int)Base32(m.Groups[1].Value); + var position = (fragment < insertPositions.Count ? insertPositions[fragment] : 0) + Base32(m.Groups[2].Value); + var part = parts.FindIndex(p => position >= p.Start && position < p.Start + p.Text.Length); + if (part < 0) + return PartName(0); + var id = IdBefore(texts[part], (int)(position - parts[part].Start), linkedAids); + return id.Length > 0 ? $"{PartName(part)}#{id}" : PartName(part); + } + + var files = new List(); + for (var i = 0; i < texts.Count; i++) + { + var xhtml = KindlePos().Replace(texts[i], Resolve); + xhtml = KindleEmbed().Replace(xhtml, m => resources.Save((int)Base32(m.Groups[1].Value) - 1) is { } file ? Path.GetFileName(file) : "missing"); + xhtml = KindleFlow().Replace(xhtml, m => Flow(flows, (int)Base32(m.Groups[1].Value), m.Groups[2].Value, folder)); + texts[i] = xhtml; + } + for (var i = 0; i < texts.Count; i++) + { + // Links that point at an element by its Amazon "aid" get an id to land on. + var xhtml = linkedAids.Count == 0 ? texts[i] : Aid().Replace(texts[i], m => linkedAids.Contains(m.Groups[1].Value) ? $"{m.Value} id=\"aid-{m.Groups[1].Value}\"" : m.Value); + var file = Path.Combine(folder, PartName(i)); + File.WriteAllBytes(file, [.. Encoding.UTF8.Preamble, .. Encoding.Latin1.GetBytes(xhtml)]); + files.Add(file); + } + return ChapterReader.Read(files, Path.Combine(folder, "~media"), cancellationToken); + } + + private static string PartName(int index) => $"part{index:0000}.xhtml"; + + /// + /// Rebuilds the XHTML files: each skeleton is followed in the text by its fragments, which are inserted into it + /// at their insert positions (positions in the rebuilt file, so the skeleton's own start is subtracted). + /// + private static List<(byte[] Text, long Start)> Assemble(byte[] markup, List skeletons, List fragments, Dictionary strings) + { + if (skeletons.Count == 0) + return [(markup, 0)]; + var parts = new List<(byte[], long)>(); + var next = 0; + foreach (var skeleton in skeletons) + { + var skeletonStart = Math.Clamp(skeleton.Tag(6, 0), 0, markup.Length); + var skeletonLength = Math.Clamp(skeleton.Tag(6, 1), 0, markup.Length - skeletonStart); + var part = new List(markup.AsSpan((int)skeletonStart, (int)skeletonLength).ToArray()); + var position = skeletonStart + skeletonLength; + var count = Math.Max(0, skeleton.Tag(1)); + for (var i = 0; i < count && next < fragments.Count; i++, next++) + { + var fragment = fragments[next]; + var length = (int)Math.Clamp(fragment.Tag(6, 1), 0, markup.Length - position); + var slice = markup.AsSpan((int)position, length).ToArray(); + position += length; + var insert = (int)Math.Clamp((long.TryParse(fragment.Key, out var p) ? p : 0) - skeletonStart, 0, part.Count); + if (InsideTag(part, insert)) + insert = AfterAidTag(part, strings.GetValueOrDefault(fragment.Tag(2), "")) ?? NextTagEnd(part, insert); + part.InsertRange(insert, slice); + } + parts.Add(([.. part], skeletonStart)); + } + return parts; + } + + private static bool InsideTag(List text, int position) + { + for (var i = position - 1; i >= 0; i--) + { + if (text[i] == '>') + return false; + if (text[i] == '<') + return true; + } + return false; + } + + private static int NextTagEnd(List text, int position) + { + var end = text.IndexOf((byte)'>', position); + return end < 0 ? text.Count : end + 1; + } + + /// Insert position after the tag named by a fragment selector such as P-//*[@aid='3']. + private static int? AfterAidTag(List text, string selector) + { + var match = SelectorAid().Match(selector); + if (!match.Success) + return null; + var needle = Encoding.ASCII.GetBytes($"aid=\"{match.Groups[1].Value}\""); + var at = text.ToArray().AsSpan().IndexOf(needle); + return at < 0 ? null : NextTagEnd(text, at); + } + + /// + /// The id a link to lands on: the nearest id or name attribute of a tag before it + /// (an "aid" is noted so an id can be added), or "" for the top of the file. + /// + private static string IdBefore(string text, int position, HashSet linkedAids) + { + position = Math.Clamp(position, 0, text.Length); + var open = text.IndexOf('<', position); + var close = text.IndexOf('>', position); + if (close >= 0 && (open == position || open < 0 || close < open)) + position = close + 1; // inside a tag (or at its start): that tag counts + var end = position; + while (end > 0) + { + var tagEnd = text.LastIndexOf('>', end - 1); + if (tagEnd < 0) + break; + var tagStart = text.LastIndexOf('<', tagEnd); + if (tagStart < 0) + break; + var tag = text[tagStart..(tagEnd + 1)]; + end = tagStart; + if (tag.StartsWith(" flows, int index, string mime, string folder) + { + if (index <= 0 || index >= flows.Count) + return "missing"; + var extension = mime.Contains("svg", StringComparison.OrdinalIgnoreCase) ? "svg" : mime.Contains("css", StringComparison.OrdinalIgnoreCase) ? "css" : "txt"; + var name = $"flow{index:0000}.{extension}"; + var file = Path.Combine(folder, name); + if (!File.Exists(file)) + File.WriteAllBytes(file, flows[index]); + return name; + } + + /// Kindle's base-32 numbers: digits 0-9 then A-V. + internal static long Base32(string text) + { + long value = 0; + foreach (var ch in text.ToUpperInvariant()) + value = value * 32 + (ch <= '9' ? ch - '0' : ch - 'A' + 10); + return value; + } + + [GeneratedRegex(@"kindle:pos:fid:([0-9A-Va-v]{4}):off:([0-9A-Va-v]{10})")] + private static partial Regex KindlePos(); + + [GeneratedRegex(@"kindle:embed:([0-9A-Va-v]{4})(?:\?mime=[^'""\)\s]*)?")] + private static partial Regex KindleEmbed(); + + [GeneratedRegex(@"kindle:flow:([0-9A-Va-v]{4})(?:\?mime=([^'""\)\s]*))?")] + private static partial Regex KindleFlow(); + + [GeneratedRegex(@"\said\s*=\s*['""]([^'""]+)['""]")] + private static partial Regex Aid(); + + [GeneratedRegex(@"^<[^>]*\s(?:id|name)\s*=\s*['""]([^'""]*)['""]", RegexOptions.IgnoreCase)] + private static partial Regex IdAttribute(); + + [GeneratedRegex(@"^<[^>]+\said\s*=\s*['""]([^'""]+)['""]", RegexOptions.IgnoreCase)] + private static partial Regex AidAttribute(); + + [GeneratedRegex(@"@aid='([^']+)'")] + private static partial Regex SelectorAid(); + + // ───────────────────────── Pictures ───────────────────────── + + /// Picture records, numbered from the first image record; written to files on first use. + private sealed class Resources(PalmDatabase database, long firstImage, string folder) + { + private readonly Dictionary _saved = []; + + /// Saves resource (0-based) and returns its file, or null if it is no picture. + public string? Save(int index) + { + if (firstImage < 0 || index < 0) + return null; + if (_saved.TryGetValue(index, out var known)) + return known; + var record = database.Record((int)firstImage + index); + var extension = record switch + { + [0xFF, 0xD8, 0xFF, ..] => "jpg", + [0x89, (byte)'P', (byte)'N', (byte)'G', ..] => "png", + [(byte)'G', (byte)'I', (byte)'F', (byte)'8', ..] => "gif", + [(byte)'B', (byte)'M', ..] => "bmp", + _ => null, + }; + string? file = null; + if (extension is not null) + { + file = Path.Combine(folder, $"image{index + 1:00000}.{extension}"); + File.WriteAllBytes(file, record.ToArray()); + } + _saved[index] = file; + return file; + } + } +} + +/// The MOBI header of record 0 (or of the KF8 half of a joint file) and its EXTH metadata. +internal sealed class MobiHeader +{ + public int Compression { get; private init; } + public long TextLength { get; private init; } + public int TextRecords { get; private init; } + public int Encryption { get; private init; } + public int Codepage { get; private init; } = 1252; + public long Version { get; private init; } + public long FirstImage { get; private init; } = -1; + public int HuffRecord { get; private init; } + public int HuffCount { get; private init; } + public int ExtraFlags { get; private init; } + public long Fdst { get; private init; } = -1; + public long Fragment { get; private init; } = -1; + public long Skeleton { get; private init; } = -1; + + /// Record of the KF8 header in a joint MOBI 6 + KF8 file (EXTH 121). + public int? Kf8Boundary { get; private set; } + + /// Cover picture as an offset from the first image record (EXTH 201). + public int? CoverOffset { get; private set; } + + private string? _fullName; + private int _locale; + private readonly Dictionary> _exth = []; + + public static MobiHeader Parse(PalmDatabase database, int record) + { + var data = database.Record(record); + if (data.Length < 16) + throw new InvalidDataException("The book's header is damaged."); + var hasMobi = data.Length >= 24 && data[16..20].SequenceEqual("MOBI"u8); + var length = hasMobi ? (int)PalmDatabase.U32(data, 20) : 0; + var header = new MobiHeader + { + Compression = PalmDatabase.U16(data, 0), + TextLength = PalmDatabase.U32(data, 4), + TextRecords = PalmDatabase.U16(data, 8), + Encryption = PalmDatabase.U16(data, 12), + Codepage = hasMobi ? (int)PalmDatabase.U32(data, 28) : 1252, + Version = hasMobi ? PalmDatabase.U32(data, 36) : 0, + FirstImage = hasMobi ? PalmDatabase.U32(data, 108) : -1, + HuffRecord = hasMobi ? (int)Math.Max(0, PalmDatabase.U32(data, 112)) : 0, + HuffCount = hasMobi ? (int)Math.Max(0, PalmDatabase.U32(data, 116)) : 0, + ExtraFlags = hasMobi && length >= 0xE4 ? PalmDatabase.U16(data, 0xF2) : 0, + Fdst = hasMobi && length >= 0xE4 ? PalmDatabase.U32(data, 0xC0) : -1, + Fragment = hasMobi && length >= 0xE8 ? PalmDatabase.U32(data, 0xF8) : -1, + Skeleton = hasMobi && length >= 0xEC ? PalmDatabase.U32(data, 0xFC) : -1, + }; + if (!hasMobi) + return header; + // MOBI 6 files reuse 0xC0 for "first content record": only KF8 headers have an FDST index there. + if (header.Version < 8) + header = header.WithoutKf8Fields(); + + var nameOffset = (int)PalmDatabase.U32(data, 84); + var nameLength = (int)PalmDatabase.U32(data, 88); + header._locale = (int)Math.Max(0, PalmDatabase.U32(data, 92)); + if (nameOffset > 0 && nameLength > 0 && nameOffset + nameLength <= data.Length) + header._fullName = header.Decode(data.Slice(nameOffset, nameLength).ToArray()); + + if ((PalmDatabase.U32(data, 128) & 0x40) != 0 && 16 + length + 12 <= data.Length && data.Slice(16 + length, 4).SequenceEqual("EXTH"u8)) + { + var exth = 16 + length; + var count = (int)PalmDatabase.U32(data, exth + 8); + var position = exth + 12; + for (var i = 0; i < count && position + 8 <= data.Length; i++) + { + var type = (int)PalmDatabase.U32(data, position); + var size = (int)PalmDatabase.U32(data, position + 4); + if (size < 8 || position + size > data.Length) + break; + if (!header._exth.TryGetValue(type, out var values)) + header._exth[type] = values = []; + values.Add(data.Slice(position + 8, size - 8).ToArray()); + position += size; + } + header.Kf8Boundary = header.Number(121) is { } boundary and > 0 ? boundary : null; + header.CoverOffset = header.Number(201); + } + return header; + } + + private MobiHeader WithoutKf8Fields() => new() + { + Compression = Compression, + TextLength = TextLength, + TextRecords = TextRecords, + Encryption = Encryption, + Codepage = Codepage, + Version = Version, + FirstImage = FirstImage, + HuffRecord = HuffRecord, + HuffCount = HuffCount, + ExtraFlags = ExtraFlags, + }; + + /// Title (EXTH 503, else the full name, else the database name), authors, publisher, language. + public BookMetadata Metadata(string databaseName) + { + var metadata = new BookMetadata + { + Title = Strings(503).FirstOrDefault() ?? _fullName ?? databaseName.Replace('_', ' '), + Publisher = Strings(101).FirstOrDefault(), + Description = Strings(103).FirstOrDefault(), + Language = Strings(524).FirstOrDefault() ?? Language(_locale), + }; + foreach (var author in Strings(100)) + foreach (var name in author.Split('&', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)) + if (!metadata.Authors.Contains(name)) + metadata.Authors.Add(name); + return metadata; + } + + private IEnumerable Strings(int type) => + _exth.TryGetValue(type, out var values) ? values.Select(Decode).Select(s => s.Trim('\0', ' ')).Where(s => s.Length > 0) : []; + + private int? Number(int type) => + _exth.TryGetValue(type, out var values) && values[0].Length == 4 && BinaryPrimitives.ReadUInt32BigEndian(values[0]) is var value && value != uint.MaxValue + ? (int)Math.Min(value, int.MaxValue) + : null; + + private string Decode(byte[] bytes) + { + if (Codepage == 65001) + return Encoding.UTF8.GetString(bytes); + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + return Encoding.GetEncoding(1252).GetString(bytes); + } + + /// Windows language id (low byte of the MOBI locale) → language tag, for the common languages. + private static string? Language(int locale) => (locale & 0xFF) switch + { + 0x04 => "zh", + 0x07 => "de", + 0x09 => "en", + 0x0A => "es", + 0x0C => "fr", + 0x10 => "it", + 0x11 => "ja", + 0x12 => "ko", + 0x13 => "nl", + 0x16 => "pt", + 0x19 => "ru", + _ => null, + }; +} diff --git a/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs b/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs new file mode 100644 index 0000000..ae665c7 --- /dev/null +++ b/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs @@ -0,0 +1,116 @@ +// The Palm database (PDB) container of MOBI, AZW, AZW3, AZW4 and PRC books: a 78-byte header with the type and +// creator ("BOOKMOBI", "TEXtREAd"), then a table of record offsets. Written from the PalmOS database format +// description; all numbers are big-endian. + +using System.Buffers.Binary; +using System.Text; + +namespace Filee.Engines.Ebooks; + +internal sealed class PalmDatabase +{ + private readonly byte[] _data; + private readonly int[] _offsets; + + private PalmDatabase(byte[] data, string name, string typeCreator, int[] offsets) + { + _data = data; + Name = name; + TypeCreator = typeCreator; + _offsets = offsets; + } + + /// Database name from the header (a short title, often truncated). + public string Name { get; } + + /// Type and creator, e.g. "BOOKMOBI" (Mobipocket / Kindle) or "TEXtREAd" (PalmDOC). + public string TypeCreator { get; } + + public int Count => _offsets.Length; + + public static PalmDatabase Open(byte[] data) + { + if (data.Length < 78) + throw new InvalidDataException("The file is too short to be an e-book."); + var count = BinaryPrimitives.ReadUInt16BigEndian(data.AsSpan(76)); + if (count == 0 || 78 + count * 8 > data.Length) + throw new InvalidDataException("The e-book's record table is damaged."); + var offsets = new int[count]; + for (var i = 0; i < count; i++) + { + var offset = BinaryPrimitives.ReadUInt32BigEndian(data.AsSpan(78 + i * 8)); + if (offset > data.Length || (i > 0 && offset < offsets[i - 1])) + throw new InvalidDataException("The e-book's record table is damaged."); + offsets[i] = (int)offset; + } + var name = Encoding.Latin1.GetString(data, 0, 32).TrimEnd('\0'); + return new PalmDatabase(data, name, Encoding.ASCII.GetString(data, 60, 8), offsets); + } + + /// Record (empty when out of range). + public ReadOnlySpan Record(int index) + { + if (index < 0 || index >= _offsets.Length) + return []; + var end = index + 1 < _offsets.Length ? _offsets[index + 1] : _data.Length; + return _data.AsSpan(_offsets[index], Math.Max(0, end - _offsets[index])); + } + + public byte[] RecordArray(int index) => Record(index).ToArray(); + + // Big-endian helpers with bounds checks: damaged books must fail with a clear error, not an exception deep inside. + + public static int U16(ReadOnlySpan data, int offset) => + offset >= 0 && offset + 2 <= data.Length ? BinaryPrimitives.ReadUInt16BigEndian(data[offset..]) : 0; + + /// A 32-bit field; 0xFFFFFFFF ("none" in MOBI headers) becomes -1. + public static long U32(ReadOnlySpan data, int offset) + { + if (offset < 0 || offset + 4 > data.Length) + return -1; + var value = BinaryPrimitives.ReadUInt32BigEndian(data[offset..]); + return value == uint.MaxValue ? -1 : value; + } +} + +/// PalmDOC compression (a simple LZ77 variant used by MOBI and PalmDOC text records). +internal static class PalmDocCompression +{ + /// Decompresses one text record. + public static byte[] Decompress(ReadOnlySpan input) + { + var output = new List(input.Length * 2); + for (var i = 0; i < input.Length;) + { + var c = input[i++]; + if (c is >= 1 and <= 8) + { + // 1..8: that many literal bytes follow. + for (var n = 0; n < c && i < input.Length; n++) + output.Add(input[i++]); + } + else if (c < 0x80) + { + output.Add(c); // 0 and 9..0x7F: the byte itself + } + else if (c >= 0xC0) + { + output.Add((byte)' '); // a space followed by an ASCII character + output.Add((byte)(c ^ 0x80)); + } + else if (i < input.Length) + { + // 0x80..0xBF: two bytes = 11-bit distance back and a length of 3..10. + var pair = (c << 8) | input[i++]; + var distance = (pair >> 3) & 0x7FF; + var length = (pair & 7) + 3; + if (distance == 0 || distance > output.Count) + continue; // damaged: skip rather than fail the whole book + var start = output.Count - distance; + for (var n = 0; n < length; n++) + output.Add(output[start + n]); + } + } + return [.. output]; + } +} diff --git a/src/Filee.Engines/Ebooks/Opf.cs b/src/Filee.Engines/Ebooks/Opf.cs new file mode 100644 index 0000000..398c0da --- /dev/null +++ b/src/Filee.Engines/Ebooks/Opf.cs @@ -0,0 +1,127 @@ +// OPF package documents (EPUB 2 and 3, also the metadata.opf of HTMLZ and TXTZ): metadata, manifest and spine. +// Element names are matched by local name so that files from sloppy generators (missing or odd prefixes) still read. + +using System.Xml; +using System.Xml.Linq; + +namespace Filee.Engines.Ebooks; + +/// A manifest item: a file of the book. +/// Full path of the extracted file. +internal sealed record OpfItem(string Id, string Path, string MediaType, string Properties); + +/// A parsed OPF package document. +internal sealed class OpfPackage +{ + public BookMetadata Metadata { get; } = new(); + public Dictionary Manifest { get; } = new(StringComparer.Ordinal); + + /// Reading order (linear and non-linear items, as calibre does). + public List Spine { get; } = []; + + /// Reads an OPF file whose manifest paths are relative to its folder. + public static OpfPackage Load(string opfPath) + { + var package = new OpfPackage(); + var document = LoadXml(opfPath); + var root = document.Root ?? throw new InvalidDataException("The package document is empty."); + var folder = System.IO.Path.GetDirectoryName(System.IO.Path.GetFullPath(opfPath))!; + + foreach (var item in Elements(root, "manifest").SelectMany(m => Elements(m, "item"))) + { + var id = (string?)item.Attribute("id"); + var href = (string?)item.Attribute("href"); + if (id is null || href is null) + continue; + var path = EbookFiles.SafePath(folder, Uri.UnescapeDataString(href.Split('#')[0])); + if (path is not null) + package.Manifest.TryAdd(id, new OpfItem(id, path, ((string?)item.Attribute("media-type") ?? "").Trim().ToLowerInvariant(), (string?)item.Attribute("properties") ?? "")); + } + foreach (var itemRef in Elements(root, "spine").SelectMany(s => Elements(s, "itemref"))) + { + if ((string?)itemRef.Attribute("idref") is { } idRef && package.Manifest.TryGetValue(idRef, out var item)) + package.Spine.Add(item); + } + + var metadata = Elements(root, "metadata").FirstOrDefault(); + if (metadata is not null) + ReadMetadata(metadata, package); + return package; + } + + private static void ReadMetadata(XElement metadata, OpfPackage package) + { + // EPUB 2 puts dc: elements in ; search descendants. + string? First(string name) => metadata.Descendants().FirstOrDefault(e => e.Name.LocalName == name && e.Value.Trim().Length > 0)?.Value.Trim(); + var meta = package.Metadata; + meta.Title = First("title"); + meta.Language = First("language"); + meta.Description = First("description"); + meta.Publisher = First("publisher"); + foreach (var creator in metadata.Descendants().Where(e => e.Name.LocalName == "creator")) + { + var name = creator.Value.Trim(); + var role = creator.Attributes().FirstOrDefault(a => a.Name.LocalName == "role")?.Value; + if (name.Length > 0 && (role is null or "aut") && !meta.Authors.Contains(name)) + meta.Authors.Add(name); + } + + // Cover: EPUB 3 "cover-image" property, else EPUB 2 . + var cover = package.Manifest.Values.FirstOrDefault(i => i.Properties.Split(' ').Contains("cover-image")); + if (cover is null + && metadata.Descendants().FirstOrDefault(e => e.Name.LocalName == "meta" && (string?)e.Attribute("name") == "cover") is { } coverMeta + && (string?)coverMeta.Attribute("content") is { } coverId) + package.Manifest.TryGetValue(coverId, out cover); + if (cover is not null && cover.MediaType.StartsWith("image/", StringComparison.Ordinal) && File.Exists(cover.Path)) + meta.CoverImage = cover.Path; + } + + /// The first rootfile named by META-INF/container.xml. + public static string RootFile(string bookFolder) + { + var container = System.IO.Path.Combine(bookFolder, "META-INF", "container.xml"); + if (File.Exists(container)) + { + var rootFile = LoadXml(container).Descendants().FirstOrDefault(e => e.Name.LocalName == "rootfile" + && ((string?)e.Attribute("media-type") ?? "application/oebps-package+xml") == "application/oebps-package+xml"); + if ((string?)rootFile?.Attribute("full-path") is { } fullPath && EbookFiles.SafePath(bookFolder, Uri.UnescapeDataString(fullPath)) is { } path && File.Exists(path)) + return path; + } + // Broken container: use the first package document in the book. + return Directory.EnumerateFiles(bookFolder, "*.opf", SearchOption.AllDirectories).FirstOrDefault() + ?? throw new InvalidDataException("This is not a valid EPUB file: it has no package document (content.opf)."); + } + + /// + /// Loads XML without resolving external DTDs (no network access, no XXE); entity declarations inside the file + /// are still allowed because some generators declare &nbsp; and friends. + /// + public static XDocument LoadXml(string path) + { + var settings = new XmlReaderSettings { DtdProcessing = DtdProcessing.Ignore, XmlResolver = null }; + using var reader = XmlReader.Create(path, settings); + return XDocument.Load(reader); + } + + private static IEnumerable Elements(XElement parent, string localName) => + parent.Elements().Where(e => e.Name.LocalName == localName); + + /// A minimal OPF with Dublin Core metadata (HTMLZ and TXTZ keep their metadata this way). + public static string MetadataOnly(BookMetadata metadata, string title, string language) + { + XNamespace opf = "http://www.idpf.org/2007/opf"; + XNamespace dc = "http://purl.org/dc/elements/1.1/"; + var meta = new XElement(opf + "metadata", new XAttribute(XNamespace.Xmlns + "dc", dc.NamespaceName), + new XElement(dc + "title", title), + new XElement(dc + "language", language), + new XElement(dc + "identifier", new XAttribute("id", "uuid_id"), $"urn:uuid:{Guid.NewGuid()}")); + foreach (var author in metadata.Authors) + meta.Add(new XElement(dc + "creator", new XAttribute(opf + "role", "aut"), author)); + if (metadata.Publisher is { } publisher) + meta.Add(new XElement(dc + "publisher", publisher)); + if (metadata.Description is { } description) + meta.Add(new XElement(dc + "description", description)); + var package = new XElement(opf + "package", new XAttribute("version", "2.0"), new XAttribute("unique-identifier", "uuid_id"), meta); + return new XDeclaration("1.0", "utf-8", null) + "\n" + package; + } +} diff --git a/src/Filee.Engines/Ebooks/PlainText.cs b/src/Filee.Engines/Ebooks/PlainText.cs new file mode 100644 index 0000000..de7deb4 --- /dev/null +++ b/src/Filee.Engines/Ebooks/PlainText.cs @@ -0,0 +1,58 @@ +// Plain text → HDocument for e-books: paragraphs are separated by empty lines, and hard-wrapped lines inside a +// paragraph (Project Gutenberg style, ~70 characters) are joined. Text without empty lines keeps one paragraph per +// line. (TXT → HWPX keeps every line as it is, see HwpxConverter.TextDocument.) + +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal static class PlainText +{ + public static HDocument ToDocument(string text) + { + var lines = text.Replace("\r\n", "\n").Replace('\r', '\n').Split('\n'); + var paragraphs = new List>(); + var current = new List(); + var hasEmptyLines = lines.Any(l => l.Trim().Length == 0); + foreach (var line in lines) + { + if (line.Trim().Length == 0 || !hasEmptyLines) + { + if (current.Count > 0) + paragraphs.Add(current); + current = []; + if (line.Trim().Length == 0) + continue; + } + current.Add(line.TrimEnd()); + } + if (current.Count > 0) + paragraphs.Add(current); + + // Hard-wrapped text: most lines of multi-line paragraphs are long and of similar length. + var multiLine = paragraphs.Where(p => p.Count > 1).SelectMany(p => p.Take(p.Count - 1)).ToList(); + var wrapped = multiLine.Count > 0 && multiLine.Count(l => l.Length is >= 50 and <= 100) > multiLine.Count * 0.7; + + var section = new HSection(); + foreach (var paragraph in paragraphs) + { + var hParagraph = new HParagraph(); + for (var i = 0; i < paragraph.Count; i++) + { + if (i > 0) + { + if (wrapped) + hParagraph.Inlines.Add(new HText(" ", default)); + else + hParagraph.Inlines.Add(new HLineBreak(default)); + } + hParagraph.Inlines.Add(new HText(i > 0 && wrapped ? paragraph[i].TrimStart() : paragraph[i], default)); + } + HtmlReader.MergeRuns(hParagraph.Inlines); + section.Blocks.Add(hParagraph); + } + var document = new HDocument(); + document.Sections.Add(section); + return document; + } +} diff --git a/src/Filee.Engines/Ebooks/ZippedText.cs b/src/Filee.Engines/Ebooks/ZippedText.cs new file mode 100644 index 0000000..c06ec1f --- /dev/null +++ b/src/Filee.Engines/Ebooks/ZippedText.cs @@ -0,0 +1,143 @@ +// HTMLZ and TXTZ, calibre's zipped single-file formats: HTMLZ is index.html + style.css + images + metadata.opf, +// TXTZ is index.txt (Markdown) + images + metadata.opf. Both are read and written here. + +using System.IO.Compression; +using System.Text; +using Filee.Engines.Hwp.Hwpx; + +namespace Filee.Engines.Ebooks; + +internal static class ZippedText +{ + /// Reads an HTMLZ book (for the reader registry of the HWPX writer). + public static HDocument ReadHtmlz(string path, string workFolder) => ReadHtmlzBook(path, workFolder).Document; + + /// Reads a TXTZ book (for the reader registry of the HWPX writer). + public static HDocument ReadTxtz(string path, string workFolder) => ReadTxtzBook(path, workFolder).Document; + + public static Book ReadHtmlzBook(string path, string workFolder, CancellationToken cancellationToken = default) + { + var folder = Extract(path, workFolder, "htmlz"); + var page = MainFile(folder, "index.html", ".html", ".htm", ".xhtml") + ?? throw new InvalidDataException("The HTMLZ file contains no HTML page."); + var document = ChapterReader.Read([page], Path.Combine(folder, "~media"), cancellationToken); + var metadata = Metadata(folder); + document.Title = metadata.Title ?? document.Title; + return new Book(document, metadata); + } + + public static Book ReadTxtzBook(string path, string workFolder) + { + var folder = Extract(path, workFolder, "txtz"); + var text = MainFile(folder, "index.txt", ".txt", ".md", ".markdown", ".text") + ?? throw new InvalidDataException("The TXTZ file contains no text file."); + // calibre writes TXTZ text as plain text, Markdown or Textile; Markdown reads all of them sensibly. + var document = MarkdownReader.Read(HwpxConverter.DecodeText(File.ReadAllBytes(text)), Path.GetDirectoryName(text)!); + var metadata = Metadata(folder); + document.Title = metadata.Title ?? document.Title; + return new Book(document, metadata); + } + + public static void WriteHtmlz(Book book, string outputPath, string sourcePath, string workFolder) + { + var title = book.DisplayTitle(sourcePath); + var language = book.DisplayLanguage(); + var images = new ImageSet(EbookFiles.NewFolder(workFolder, "htmlz-images")); + var page = XhtmlWriter.Write(book.Document, new XhtmlOptions { ImageSource = images.Add }).Single(); + Save(outputPath, [ + ("index.html", XhtmlWriter.Page(title, page.Body, language, "style.css", epub: false)), + ("style.css", XhtmlWriter.Css), + ("metadata.opf", OpfPackage.MetadataOnly(book.Metadata, title, language)), + ], images); + } + + public static void WriteTxtz(Book book, string outputPath, string sourcePath, string workFolder) + { + var title = book.DisplayTitle(sourcePath); + var images = new ImageSet(EbookFiles.NewFolder(workFolder, "txtz-images")); + var text = MarkdownWriter.Write(book.Document, images.Add); + Save(outputPath, [ + ("index.txt", text), + ("metadata.opf", OpfPackage.MetadataOnly(book.Metadata, title, book.DisplayLanguage())), + ], images); + } + + private static string Extract(string path, string workFolder, string prefix) + { + var folder = EbookFiles.NewFolder(workFolder, prefix); + try + { + using var zip = ZipFile.OpenRead(path); + EbookFiles.Extract(zip, folder); + } + catch (InvalidDataException ex) + { + throw new InvalidDataException($"This is not a valid {prefix.ToUpperInvariant()} file (it is not a ZIP archive).", ex); + } + return folder; + } + + /// The preferred file name at the top level, else the first file with one of the extensions (shallowest first). + private static string? MainFile(string folder, string preferred, params string[] extensions) + { + var top = Path.Combine(folder, preferred); + if (File.Exists(top)) + return top; + return Directory.EnumerateFiles(folder, "*", SearchOption.AllDirectories) + .Where(f => extensions.Contains(Path.GetExtension(f).ToLowerInvariant())) + .OrderBy(f => f.Count(c => c == Path.DirectorySeparatorChar)) + .ThenBy(f => f, StringComparer.OrdinalIgnoreCase) + .FirstOrDefault(); + } + + private static BookMetadata Metadata(string folder) + { + var opf = Directory.EnumerateFiles(folder, "*.opf", SearchOption.AllDirectories).FirstOrDefault(); + if (opf is null) + return new BookMetadata(); + try + { + return OpfPackage.Load(opf).Metadata; + } + catch (System.Xml.XmlException) + { + return new BookMetadata(); // broken metadata: the content still converts + } + } + + private static void Save(string outputPath, IEnumerable<(string Name, string Text)> texts, ImageSet images) + { + var temp = outputPath + ".tmp"; + using (var zip = ZipFile.Open(temp, ZipArchiveMode.Create)) + { + foreach (var (name, text) in texts) + { + using var writer = new StreamWriter(zip.CreateEntry(name).Open(), new UTF8Encoding(false)); + writer.Write(text); + } + foreach (var (name, file) in images.Files) + zip.CreateEntryFromFile(file, name, CompressionLevel.NoCompression); + } + File.Move(temp, outputPath, overwrite: true); + } + + /// Pictures of a zipped book: images/imgNNNN.ext, converted to JPEG / PNG / GIF where needed. + private sealed class ImageSet(string convertFolder) + { + private readonly Dictionary _names = new(StringComparer.OrdinalIgnoreCase); + + public List<(string Name, string File)> Files { get; } = []; + + public string? Add(string path) + { + if (_names.TryGetValue(path, out var known)) + return known; + if (EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is not ({ } file, _)) + return null; + var name = $"images/img{Files.Count + 1:0000}{Path.GetExtension(file).ToLowerInvariant()}"; + Files.Add((name, file)); + _names[path] = name; + return name; + } + } +} diff --git a/src/Filee.Engines/EngineRegistry.cs b/src/Filee.Engines/EngineRegistry.cs index e57508d..3e0e9d0 100644 --- a/src/Filee.Engines/EngineRegistry.cs +++ b/src/Filee.Engines/EngineRegistry.cs @@ -3,6 +3,7 @@ using Filee.Core.Conversion; using Filee.Engines.Archives; using Filee.Engines.Cad; +using Filee.Engines.Ebooks; using Filee.Engines.Email; using Filee.Engines.Fonts; using Filee.Engines.Hwp; @@ -46,8 +47,10 @@ public static IReadOnlyList CreateAll(EngineEnvironment env) => new EmlConverter(), new FfmpegConverter(), new CadConverter(), + new EbookConverter(), new PandocConverter(), new GhostscriptConverter(), + new CalibreConverter(), new LibreOfficeConverter(env), ]; diff --git a/src/Filee.Engines/Filee.Engines.csproj b/src/Filee.Engines/Filee.Engines.csproj index c8a21e6..16dd1af 100644 --- a/src/Filee.Engines/Filee.Engines.csproj +++ b/src/Filee.Engines/Filee.Engines.csproj @@ -24,6 +24,7 @@ + diff --git a/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs b/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs index 5c9a7fd..7534e5d 100644 --- a/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs +++ b/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs @@ -4,6 +4,7 @@ using System.Text; using System.Text.Json.Nodes; +using Filee.Engines.Ebooks; using Filee.Engines.Hwp.Hwpx.Docx; using Filee.Engines.Hwp.Hwpx.Pptx; using Filee.Engines.Infrastructure; @@ -44,6 +45,15 @@ internal static class DocumentReaders ["pdf"] = PdfDocumentReader.Read, // HWP comes here as HWPX, converted by rhwp first (the route planner adds that step). ["hwpx"] = HwpxReader.Read, + ["html"] = HtmlReader.Read, + ["epub"] = EpubReader.Read, + ["mobi"] = MobiReader.Read, + ["azw3"] = MobiReader.Read, + ["azw"] = MobiReader.Read, + ["prc"] = MobiReader.Read, + ["fb2"] = Fb2Reader.Read, + ["htmlz"] = ZippedText.ReadHtmlz, + ["txtz"] = ZippedText.ReadTxtz, }; /// diff --git a/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs b/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs new file mode 100644 index 0000000..b224590 --- /dev/null +++ b/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs @@ -0,0 +1,87 @@ +// Walks the HDocument model: every inline of a block list, including those inside tables, notes, text boxes and +// links. Used by readers (bookmark pruning) and the writers that are not HWPX (XHTML, plain text, Markdown, FB2). + +using System.Text; + +namespace Filee.Engines.Hwp.Hwpx; + +internal static class HDocumentWalker +{ + /// Every inline, depth first: the content of links, notes and text boxes follows them. + public static IEnumerable Inlines(IEnumerable blocks) => + InlineLists(blocks).SelectMany(list => list); + + /// Every inline list (paragraph, link content, note and text box paragraphs, table cells). + public static IEnumerable> InlineLists(IEnumerable blocks) + { + foreach (var block in blocks) + { + switch (block) + { + case HParagraph paragraph: + foreach (var list in InlineLists(paragraph.Inlines)) + yield return list; + break; + case HTable table: + foreach (var list in InlineLists(table.Caption)) + yield return list; + foreach (var cell in table.Rows.SelectMany(r => r.Cells)) + foreach (var list in InlineLists(cell.Blocks)) + yield return list; + break; + } + } + } + + private static IEnumerable> InlineLists(List inlines) + { + yield return inlines; + foreach (var inline in inlines.ToList()) + { + var nested = inline switch + { + HLink link => InlineLists(link.Content), + HNote note => InlineLists(note.Blocks), + HTextBox box => InlineLists(box.Blocks), + _ => [], + }; + foreach (var list in nested) + yield return list; + } + } + + /// The inlines of one paragraph with the content of its links (not notes or text boxes). + public static IEnumerable InlinesOf(IEnumerable inlines) + { + foreach (var inline in inlines) + { + yield return inline; + if (inline is HLink link) + foreach (var child in InlinesOf(link.Content)) + yield return child; + } + } + + /// Every picture of the document, in reading order. + public static IEnumerable Images(HDocument document) => + document.Sections.SelectMany(s => Inlines(s.Blocks)).OfType(); + + /// The visible text of a paragraph's inlines (links included, notes left out), line breaks as spaces. + public static string Text(IEnumerable inlines) + { + var sb = new StringBuilder(); + foreach (var inline in InlinesOf(inlines)) + { + switch (inline) + { + case HText text: + sb.Append(text.Text); + break; + case HLineBreak or HTab: + sb.Append(' '); + break; + } + } + return sb.ToString().Trim(); + } +} diff --git a/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs b/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs new file mode 100644 index 0000000..809ee83 --- /dev/null +++ b/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs @@ -0,0 +1,291 @@ +// The small part of CSS the HTML reader understands: inline style="" declarations, simple stylesheet rules (tag, +// .class, tag.class), colours and lengths. Enough for the formatting e-books and saved web pages rely on (bold and +// italic classes, centred text, page breaks, first-line indents) without a full CSS engine. + +using System.Globalization; +using System.Text.RegularExpressions; + +namespace Filee.Engines.Hwp.Hwpx; + +/// Property → value pairs of one declaration block, lower-case property names. +internal sealed class CssDeclarations : Dictionary +{ + public CssDeclarations() : base(StringComparer.OrdinalIgnoreCase) + { + } + + /// Parses "color: red; font-weight: bold" (later declarations win, !important is ignored). + public static CssDeclarations Parse(string? text) + { + var result = new CssDeclarations(); + if (string.IsNullOrWhiteSpace(text)) + return result; + foreach (var declaration in text.Split(';')) + { + var colon = declaration.IndexOf(':'); + if (colon <= 0) + continue; + var name = declaration[..colon].Trim(); + var value = declaration[(colon + 1)..].Replace("!important", "", StringComparison.OrdinalIgnoreCase).Trim(); + if (name.Length > 0 && value.Length > 0) + result[name] = value; + } + return result; + } + + public string? Get(string name) => TryGetValue(name, out var value) ? value : null; +} + +/// Rules of the document's style sheets that the reader can match without a full selector engine. +internal sealed partial class CssStyleSheet +{ + /// Properties taken from style sheets. Colours and sizes are not: whole-book rules (body colour, base + /// font size) would otherwise be copied onto every run. + private static readonly HashSet SheetProperties = new(StringComparer.OrdinalIgnoreCase) + { + "font-weight", "font-style", "text-decoration", "text-decoration-line", "vertical-align", "text-align", + "display", "page-break-before", "page-break-after", "break-before", "break-after", "white-space", + "text-indent", "list-style", "list-style-type", "font-variant", + }; + + private readonly List _rules = []; + + /// Element name, or null for any element. + /// Classes the element must have (all of them). + /// Classes count more than the tag name. + private sealed record Rule(string? Tag, string[] Classes, int Specificity, int Order, CssDeclarations Declarations); + + public bool IsEmpty => _rules.Count == 0; + + /// Adds the rules of a style sheet. @media, @font-face and other at-rules are skipped. + public void Add(string css) + { + css = Comment().Replace(css, ""); + var position = 0; + while (position < css.Length) + { + var open = css.IndexOf('{', position); + if (open < 0) + break; + var selectors = css[position..open].Trim(); + var close = MatchingBrace(css, open); + var body = css[(open + 1)..Math.Max(open + 1, close)]; + position = close < 0 ? css.Length : close + 1; + if (selectors.StartsWith('@') || body.Contains('{')) + continue; // at-rules (nested blocks): not applied + + var declarations = CssDeclarations.Parse(body); + foreach (var key in declarations.Keys.Where(k => !SheetProperties.Contains(k)).ToList()) + declarations.Remove(key); + if (declarations.Count == 0) + continue; + foreach (var selector in selectors.Split(',')) + { + if (ParseSelector(selector.Trim()) is { } rule) + _rules.Add(rule with { Order = _rules.Count, Declarations = declarations }); + } + } + } + + /// Declarations that apply to an element, in cascade order (the last one wins). + public CssDeclarations Match(string tag, IReadOnlyCollection classes) + { + var result = new CssDeclarations(); + if (_rules.Count == 0) + return result; + foreach (var rule in _rules.Where(r => (r.Tag is null || r.Tag == tag) && r.Classes.All(classes.Contains)) + .OrderBy(r => r.Specificity).ThenBy(r => r.Order)) + { + foreach (var (name, value) in rule.Declarations) + result[name] = value; + } + return result; + } + + /// "p", ".note", "span.bold", "p.a.b"; anything else (descendants, ids, pseudo classes) is skipped. + private static Rule? ParseSelector(string selector) + { + if (!SimpleSelector().IsMatch(selector)) + return null; + var parts = selector.Split('.'); + var tag = parts[0].Length == 0 || parts[0] == "*" ? null : parts[0].ToLowerInvariant(); + var classes = parts.Skip(1).ToArray(); + return new Rule(tag, classes, classes.Length * 10 + (tag is null ? 0 : 1), 0, []); + } + + private static int MatchingBrace(string css, int open) + { + var depth = 0; + for (var i = open; i < css.Length; i++) + { + if (css[i] == '{') + depth++; + else if (css[i] == '}' && --depth == 0) + return i; + } + return -1; + } + + [GeneratedRegex(@"/\*.*?\*/", RegexOptions.Singleline)] + private static partial Regex Comment(); + + [GeneratedRegex(@"^(\*|[A-Za-z][\w-]*)?(\.[\w-]+)*$")] + private static partial Regex SimpleSelector(); +} + +/// CSS values: colours and lengths. +internal static class CssValues +{ + private static readonly Dictionary NamedColors = new(StringComparer.OrdinalIgnoreCase) + { + ["black"] = "#000000", + ["white"] = "#FFFFFF", + ["red"] = "#FF0000", + ["green"] = "#008000", + ["blue"] = "#0000FF", + ["yellow"] = "#FFFF00", + ["gray"] = "#808080", + ["grey"] = "#808080", + ["silver"] = "#C0C0C0", + ["maroon"] = "#800000", + ["purple"] = "#800080", + ["fuchsia"] = "#FF00FF", + ["magenta"] = "#FF00FF", + ["lime"] = "#00FF00", + ["olive"] = "#808000", + ["navy"] = "#000080", + ["teal"] = "#008080", + ["aqua"] = "#00FFFF", + ["cyan"] = "#00FFFF", + ["orange"] = "#FFA500", + ["brown"] = "#A52A2A", + ["pink"] = "#FFC0CB", + ["gold"] = "#FFD700", + ["darkred"] = "#8B0000", + ["darkblue"] = "#00008B", + ["darkgreen"] = "#006400", + ["darkgray"] = "#A9A9A9", + ["darkgrey"] = "#A9A9A9", + ["lightgray"] = "#D3D3D3", + ["lightgrey"] = "#D3D3D3", + ["dimgray"] = "#696969", + ["dimgrey"] = "#696969", + ["crimson"] = "#DC143C", + ["indigo"] = "#4B0082", + ["violet"] = "#EE82EE", + ["coral"] = "#FF7F50", + ["tomato"] = "#FF6347", + ["steelblue"] = "#4682B4", + ["royalblue"] = "#4169E1", + ["lightblue"] = "#ADD8E6", + ["lightyellow"] = "#FFFFE0", + ["lightgreen"] = "#90EE90", + ["whitesmoke"] = "#F5F5F5", + ["beige"] = "#F5F5DC", + ["ivory"] = "#FFFFF0", + }; + + /// "#RRGGBB" for #rgb, #rrggbb, rgb()/rgba() and common colour names; null for anything else + /// (transparent, inherit, gradients, ...). + public static string? Color(string? value) + { + if (string.IsNullOrWhiteSpace(value)) + return null; + var text = value.Trim(); + if (NamedColors.TryGetValue(text, out var named)) + return named; + if (text.StartsWith('#')) + { + var hex = text[1..]; + if (hex.Length is 3 or 4 && hex.All(Uri.IsHexDigit)) + return "#" + string.Concat(hex.Take(3).Select(c => $"{c}{c}")).ToUpperInvariant(); + if (hex.Length is 6 or 8 && hex.All(Uri.IsHexDigit)) + return "#" + hex[..6].ToUpperInvariant(); + return null; + } + if (text.StartsWith("rgb", StringComparison.OrdinalIgnoreCase) && text.IndexOf('(') is var open and > 0 && text.IndexOf(')') is var close && close > open) + { + var parts = text[(open + 1)..close].Split([',', ' ', '/'], StringSplitOptions.RemoveEmptyEntries); + if (parts.Length < 3) + return null; + var channels = new int[3]; + for (var i = 0; i < 3; i++) + { + var part = parts[i]; + var percent = part.EndsWith('%'); + if (!double.TryParse(part.TrimEnd('%'), NumberStyles.Float, CultureInfo.InvariantCulture, out var number)) + return null; + channels[i] = (int)Math.Clamp(Math.Round(percent ? number * 2.55 : number), 0, 255); + } + return $"#{channels[0]:X2}{channels[1]:X2}{channels[2]:X2}"; + } + return null; + } + + /// The first colour inside a shorthand such as "background: #eee url(x.png)". + public static string? FirstColor(string? value) + { + if (Color(value) is { } direct) + return direct; + foreach (var token in (value ?? "").Split(' ', StringSplitOptions.RemoveEmptyEntries)) + if (Color(token) is { } color) + return color; + return null; + } + + /// + /// A length in HWPUNIT. is the size of 1em (and the base of %) in HWPUNIT; null when the + /// value is not a length (auto, inherit, ...). + /// + public static int? Length(string? value, double em) + { + if (string.IsNullOrWhiteSpace(value)) + return null; + var text = value.Trim().ToLowerInvariant(); + var split = 0; + while (split < text.Length && (char.IsDigit(text[split]) || text[split] is '.' or '-' or '+')) + split++; + if (split == 0 || !double.TryParse(text[..split], NumberStyles.Float, CultureInfo.InvariantCulture, out var number)) + return null; + var unit = text[split..].Trim(); + double? result = unit switch + { + "em" or "rem" => number * em, + "ex" or "ch" => number * em / 2, + "%" => number * em / 100, + "pt" => number * 100, + "px" or "" => number * HwpxUnits.PerPixel, + "in" => number * 7200, + "cm" => number * HwpxUnits.PerMm * 10, + "mm" => number * HwpxUnits.PerMm, + "pc" => number * 1200, + _ => null, + }; + return result is null ? null : (int)Math.Round(result.Value); + } + + /// A font size in 1/100 pt, relative sizes based on (1/100 pt). + public static int? FontSize(string? value, int parent) + { + if (string.IsNullOrWhiteSpace(value)) + return null; + double? factor = value.Trim().ToLowerInvariant() switch + { + "xx-small" => 0.6, + "x-small" => 0.75, + "small" => 0.89, + "medium" => 1.0, + "large" => 1.2, + "x-large" => 1.5, + "xx-large" => 2.0, + "xxx-large" => 3.0, + "smaller" => 0.83, + "larger" => 1.2, + _ => null, + }; + if (factor is { } f) + return (int)Math.Round(parent * f); + // HWPUNIT and 1/100 pt are the same scale (1 pt = 100 HWPUNIT), so Length() fits font sizes directly. + return Length(value, parent) is { } size && size > 0 ? Math.Clamp(size, 100, 20000) : null; + } +} diff --git a/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs b/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs new file mode 100644 index 0000000..7e2383a --- /dev/null +++ b/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs @@ -0,0 +1,1142 @@ +// HTML / XHTML → HDocument with AngleSharp (MIT), so HTML → HWPX / PDF needs no Pandoc and e-books (EPUB, MOBI, +// HTMLZ) reuse the same reader for their chapters. +// +// Headings, paragraphs, inline formatting (b / strong / i / em / u / s / del / sub / sup / code / mark / small / +// font and the basic inline CSS: colour, background, font-weight, font-style, font-size, text-decoration, +// vertical-align, text-align), simple class and tag rules from \n" + : $"\n"); + sb.Append("\n\n").Append(body).Append("\n\n"); + return sb.ToString(); + } + + /// XML-escaped text without characters XML 1.0 forbids. + internal static string Escape(string text) => HwpxWriter.Escape(text); + + // ───────────────────────── Chapters ───────────────────────── + + /// Blocks per chapter: a new one at a section start, a page break or a level-1 heading. + private static List> Split(HDocument document, bool split) + { + var chapters = new List> { new() }; + var length = 0; + foreach (var section in document.Sections) + { + var sectionStart = true; + foreach (var block in section.Blocks) + { + var current = chapters[^1]; + if (split && HasContent(current)) + { + var paragraph = block as HParagraph; + var cut = sectionStart + || paragraph is { PageBreakBefore: true } or { HeadingLevel: 1 } + || (length > ChapterLimit && paragraph is { List: null, HeadingLevel: 0 }); + if (cut) + { + chapters.Add(current = []); + length = 0; + } + } + sectionStart = false; + current.Add(block); + length += HDocumentWalker.Inlines([block]).OfType().Sum(t => t.Text.Length); + } + } + return chapters; + } + + private static bool HasContent(List blocks) => + blocks.Any(b => b is HTable || HDocumentWalker.Inlines([b]).Any(i => i is HImage || i is HText { Text: var text } && !string.IsNullOrWhiteSpace(text))); + + /// The most common text size, written as 1em (other sizes become relative to it). + private static int BodySize(HDocument document) + { + var sizes = document.Sections.SelectMany(s => HDocumentWalker.Inlines(s.Blocks)).OfType() + .Where(t => t.Format.Size is not null) + .GroupBy(t => t.Format.Size!.Value) + .Select(g => (Size: g.Key, Characters: g.Sum(t => t.Text.Length))) + .OrderByDescending(g => g.Characters) + .FirstOrDefault(); + return sizes.Size > 0 ? sizes.Size : 1000; + } + + // ───────────────────────── Blocks ───────────────────────── + + private void Blocks(List blocks, StringBuilder sb) + { + var i = 0; + while (i < blocks.Count) + { + if (blocks[i] is HParagraph { List: not null }) + i = List(blocks, i, sb); + else if (IsCodeLine(blocks[i])) + i = Code(blocks, i, sb); + else + Block(blocks[i++], sb); + } + } + + private void Block(HBlock block, StringBuilder sb) + { + switch (block) + { + case HParagraph paragraph: + Paragraph(paragraph, sb); + break; + case HTable table: + Table(table, sb); + break; + } + } + + private void Paragraph(HParagraph paragraph, StringBuilder sb) + { + var breakClass = paragraph.PageBreakBefore && !_options.SplitChapters ? " class=\"page-break\"" : ""; + if (paragraph.Inlines.Any(i => i is HShape { Kind: HShapeKind.Line }) && !HasVisible(paragraph.Inlines)) + { + sb.Append(Anchors(paragraph.Inlines)).Append("
\n"); + return; + } + var content = Inlines(paragraph.Inlines, paragraph.HeadingLevel > 0); + if (!HasVisible(paragraph.Inlines)) + { + // Empty paragraphs are spacing in word processors; reading systems space paragraphs themselves. + if (content.Length > 0) + sb.Append("').Append(content).Append("\n"); + return; + } + + var style = ParagraphStyle(paragraph.Format); + if (paragraph.HeadingLevel > 0) + { + var level = Math.Clamp(paragraph.HeadingLevel, 1, 6); + var id = $"_h{++_headingCount}"; + _usedIds.Add(id); + _headings.Add(new XhtmlHeading(paragraph.HeadingLevel, HDocumentWalker.Text(paragraph.Inlines), $"{_options.ChapterFile(_chapter)}#{id}")); + sb.Append($"').Append(content).Append($"\n"); + return; + } + sb.Append("').Append(content).Append("

\n"); + } + + private static string ParagraphStyle(HParaFormat format) + { + var styles = new List(); + switch (format.Align) + { + case HAlign.Center: + styles.Add("text-align:center"); + break; + case HAlign.Right: + styles.Add("text-align:right"); + break; + case HAlign.Justify or HAlign.Distribute: + styles.Add("text-align:justify"); + break; + } + // HWPUNIT → em of a 10 pt body text (1000 HWPUNIT). + if (format.Left is > 0 and var left) + styles.Add($"margin-left:{Em(left)}"); + if (format.FirstLine is { } first && first != 0) + styles.Add($"text-indent:{Em(first)}"); + return styles.Count == 0 ? "" : $" style=\"{string.Join(';', styles)}\""; + } + + private static string Em(int hwpUnits) => (hwpUnits / 1000.0).ToString("0.##", CultureInfo.InvariantCulture) + "em"; + + /// A paragraph that is all code (shaded like the readers mark code blocks). + private static bool IsCodeLine(HBlock block) => + block is HParagraph { List: null, HeadingLevel: 0 } paragraph + && paragraph.Inlines.Count > 0 + && paragraph.Inlines.All(i => i is HText { Format.Shade: HtmlReader.CodeShade }); + + private static int Code(List blocks, int start, StringBuilder sb) + { + var lines = new List(); + var i = start; + while (i < blocks.Count && IsCodeLine(blocks[i])) + lines.Add(string.Concat(((HParagraph)blocks[i++]).Inlines.OfType().Select(t => t.Text))); + sb.Append("
").Append(Escape(string.Join("\n", lines))).Append("
\n"); + return i; + } + + private sealed class OpenList(HNumbering numbering, int level, bool ordered) + { + public HNumbering Numbering { get; } = numbering; + public int Level { get; } = level; + public bool Ordered { get; } = ordered; + } + + /// Consecutive list paragraphs as nested ul / ol; returns the index after the list. + private int List(List blocks, int start, StringBuilder sb) + { + var stack = new List(); + var i = start; + for (; i < blocks.Count && blocks[i] is HParagraph { List: { } item } paragraph; i++) + { + while (stack.Count > 0 && stack[^1].Level > item.Level) + Close(stack, sb); + if (!item.Numbered) + { + if (stack.Count == 0) + Paragraph(paragraph, sb); + else + sb.Append("').Append(Inlines(paragraph.Inlines, false)).Append("

"); + continue; + } + if (stack.Count > 0 && stack[^1].Level == item.Level && !ReferenceEquals(stack[^1].Numbering, item.Numbering)) + Close(stack, sb); + if (stack.Count > 0 && stack[^1].Level == item.Level) + { + sb.Append("\n"); + } + else + { + var level = item.Numbering.Levels.Count == 0 ? null : item.Numbering.Levels[Math.Min(item.Level, item.Numbering.Levels.Count - 1)]; + var ordered = level is { Bullet: false }; + stack.Add(new OpenList(item.Numbering, item.Level, ordered)); + if (!ordered) + { + sb.Append("
    \n"); + } + else + { + var type = level!.Format switch + { + "LATIN_SMALL" => " type=\"a\"", + "LATIN_CAPITAL" => " type=\"A\"", + "ROMAN_SMALL" => " type=\"i\"", + "ROMAN_CAPITAL" => " type=\"I\"", + _ => "", + }; + sb.Append("\n"); + } + } + sb.Append("') + .Append(Inlines(paragraph.Inlines, false)); + } + while (stack.Count > 0) + Close(stack, sb); + return i; + } + + private static void Close(List stack, StringBuilder sb) + { + sb.Append("\n").Append(stack[^1].Ordered ? "\n" : "
\n"); + stack.RemoveAt(stack.Count - 1); + } + + private void Table(HTable table, StringBuilder sb) + { + sb.Append(table.Borders == HBorders.Empty ? " 0) + sb.Append($" style=\"margin-left:{Em(table.Indent)}\""); + sb.Append(">\n"); + if (table.Caption.Count > 0) + sb.Append("\n"); + var header = table.Rows.TakeWhile(r => r.Header).Count(); + if (header > 0 && header < table.Rows.Count) + sb.Append("\n"); + for (var r = 0; r < table.Rows.Count; r++) + { + if (r == header && header > 0 && header < table.Rows.Count) + sb.Append("\n\n"); + var row = table.Rows[r]; + sb.Append(""); + foreach (var cell in row.Cells) + { + var tag = row.Header ? "th" : "td"; + sb.Append('<').Append(tag); + if (cell.RowSpan > 1) + sb.Append($" rowspan=\"{cell.RowSpan}\""); + if (cell.ColSpan > 1) + sb.Append($" colspan=\"{cell.ColSpan}\""); + var styles = new List(); + if (cell.Fill is { } fill) + styles.Add($"background-color:{fill}"); + if (cell.VerticalAlign != HVerticalAlign.Top) + styles.Add(cell.VerticalAlign == HVerticalAlign.Center ? "vertical-align:middle" : "vertical-align:bottom"); + if (styles.Count > 0) + sb.Append($" style=\"{string.Join(';', styles)}\""); + sb.Append('>'); + if (cell.Blocks is [HParagraph { List: null, HeadingLevel: 0 } only]) + sb.Append(Inlines(only.Inlines, false)); // a single paragraph needs no

+ else + Blocks(cell.Blocks, sb); + sb.Append("'); + } + sb.Append("

\n"); + } + if (header > 0 && header < table.Rows.Count) + sb.Append("\n"); + sb.Append("
").Append(string.Join(" ", table.Caption.OfType().Select(p => Inlines(p.Inlines, false)))).Append("
\n"); + } + + // ───────────────────────── Inlines ───────────────────────── + + private static bool HasVisible(List inlines) => HDocumentWalker.InlinesOf(inlines).Any(i => i switch + { + HText text => !string.IsNullOrWhiteSpace(text.Text) || text.Text.Contains(' '), + HImage or HNote or HTextBox or HTab => true, + _ => false, + }); + + private string Anchors(IEnumerable inlines) => + string.Concat(inlines.OfType().Select(b => Id(b.Name)).OfType().Select(id => $"")); + + private string Inlines(List inlines, bool heading) + { + var sb = new StringBuilder(); + foreach (var inline in inlines) + { + switch (inline) + { + case HText text: + Run(text.Text, text.Format, heading, sb); + break; + case HLineBreak: + sb.Append("
"); + break; + case HTab: + sb.Append(' '); + break; + case HBookmark bookmark: + if (Id(bookmark.Name) is { } id) + sb.Append($""); + break; + case HLink link: + Link(link, heading, sb); + break; + case HImage image: + Image(image, sb); + break; + case HNote note: + Note(note, sb); + break; + case HTextBox box: + { + var parts = box.Blocks.OfType().Select(p => Inlines(p.Inlines, false)).Where(p => p.Length > 0).ToList(); + if (parts.Count > 0) + sb.Append("").Append(string.Join("
", parts)).Append("
"); + break; + } + } + } + return sb.ToString(); + } + + private void Run(string text, HCharFormat format, bool heading, StringBuilder sb) + { + if (text.Length == 0) + return; + var close = new Stack(); + void Open(string tag, string attributes = "") + { + sb.Append('<').Append(tag).Append(attributes).Append('>'); + close.Push($""); + } + + var styles = new List(); + if (format.Color is { } color && color != "#000000") + styles.Add($"color:{color}"); + if (format.Shade is { } shade && shade is not (HtmlReader.CodeShade or HtmlReader.MarkShade)) + styles.Add($"background-color:{shade}"); + if (format.Size is { } size && Math.Abs(size - _bodySize) > _bodySize / 20 && format.Superscript is not true && format.Subscript is not true) + styles.Add($"font-size:{(size / (double)_bodySize).ToString("0.##", CultureInfo.InvariantCulture)}em"); + if (heading && format.Bold is false) + styles.Add("font-weight:normal"); + if (styles.Count > 0) + Open("span", $" style=\"{string.Join(';', styles)}\""); + if (format.Shade == HtmlReader.CodeShade) + Open("code"); + if (format.Shade == HtmlReader.MarkShade) + Open("mark"); + if (format.Bold is true && !heading) + Open("b"); + if (format.Italic is true) + Open("i"); + if (format.Underline is true) + Open("u"); + if (format.Strike is true) + Open("s"); + if (format.Superscript is true) + Open("sup"); + else if (format.Subscript is true) + Open("sub"); + sb.Append(Escape(text)); + while (close.Count > 0) + sb.Append(close.Pop()); + } + + private void Link(HLink link, bool heading, StringBuilder sb) + { + var content = Inlines(link.Content, heading); + var href = Href(link.Target); + if (href is null) + sb.Append(content); + else + sb.Append($"").Append(content).Append(""); + } + + /// The href of a link: bookmarks point into the right chapter; only absolute URIs leave the book. + private string? Href(string target) + { + if (target.StartsWith('#')) + { + var name = target[1..]; + if (!_bookmarkChapter.TryGetValue(name, out var chapter) || Id(name) is not { } id) + return null; // no such bookmark: keep the text only + return !_options.SplitChapters || chapter == _chapter ? "#" + id : $"{_options.ChapterFile(chapter)}#{id}"; + } + var colon = target.IndexOf(':'); + return colon > 1 && target[..colon].All(char.IsAsciiLetter) ? target : null; + } + + /// A valid, unique XML id for a bookmark name (the same name always gets the same id). + private string? Id(string name) + { + if (string.IsNullOrWhiteSpace(name)) + return null; + if (_ids.TryGetValue(name, out var known)) + return known; + var sb = new StringBuilder(); + foreach (var ch in name) + sb.Append(char.IsAsciiLetterOrDigit(ch) || ch is '-' or '_' or '.' ? ch : '_'); + if (!char.IsAsciiLetter(sb[0]) && sb[0] != '_') + sb.Insert(0, "id-"); + var id = sb.ToString(); + for (var n = 2; !_usedIds.Add(id); n++) + id = $"{sb}-{n}"; + _ids[name] = id; + return id; + } + + private void Image(HImage image, StringBuilder sb) + { + if (_options.ImageSource(image.Path) is not { } source) + return; + sb.Append($"\"\""); 0 and var width) + sb.Append($" style=\"width:{Math.Max(1, (int)Math.Round(width / HwpxUnits.PerPixel))}px\""); + sb.Append("/>"); + } + + private void Note(HNote note, StringBuilder sb) + { + var number = ++_noteCount; + var epub = _options.Epub ? " epub:type=\"noteref\"" : ""; + sb.Append($"{number}"); + + var content = new StringBuilder(); + Blocks(note.Blocks, content); + var type = _options.Epub ? " epub:type=\"footnote\"" : ""; + _notes.Append($"\n"); + } +} diff --git a/src/Filee.Engines/Infrastructure/EngineDownloads.cs b/src/Filee.Engines/Infrastructure/EngineDownloads.cs index ab8d7d7..45fbd3a 100644 --- a/src/Filee.Engines/Infrastructure/EngineDownloads.cs +++ b/src/Filee.Engines/Infrastructure/EngineDownloads.cs @@ -7,8 +7,8 @@ namespace Filee.Engines.Infrastructure; /// One download from engines.json. /// -/// "zip" (extracted), "msi" (administrative install, no system changes), "oxt" (LibreOffice extension) or "conda" -/// (conda-forge package: its Windows binaries are extracted). +/// "zip" (extracted), "msi" (administrative install, no system changes; LibreOffice, calibre), "oxt" (LibreOffice +/// extension) or "conda" (conda-forge package: its Windows binaries are extracted). /// /// Download size in bytes. public sealed record EngineComponent(string Id, string Version, string Url, string Sha256, long Size, string Kind); @@ -34,6 +34,7 @@ public static class EngineDownloads // Ghostscript from conda-forge plus the Microsoft C++ runtime it was built against (copied next to it). new("ghostscript", ["ghostscript", "vcruntime"], 31_000_000, ["ghostscript"]), new("ffmpeg", ["ffmpeg"], 272_000_000, ["ffmpeg"]), + new("calibre", ["calibre"], 663_000_000, ["calibre"]), ]; /// Total download size of a package in bytes. diff --git a/src/Filee.Engines/Infrastructure/EngineInstaller.cs b/src/Filee.Engines/Infrastructure/EngineInstaller.cs index 1935901..8a17741 100644 --- a/src/Filee.Engines/Infrastructure/EngineInstaller.cs +++ b/src/Filee.Engines/Infrastructure/EngineInstaller.cs @@ -3,8 +3,8 @@ // * Every download is checked against the SHA-256 pinned in engines.json before anything is unpacked. // * Unpacking happens in a staging folder that replaces the engine folder only when complete, so a failed or // cancelled install never leaves a half-installed engine behind. -// * The LibreOffice MSI is unpacked with an administrative install (msiexec /a): files only, no registry entries, -// no admin rights. Parts headless conversion never uses (help, gallery, most dictionaries) are removed. +// * MSIs (LibreOffice, calibre) are unpacked with an administrative install (msiexec /a): files only, no registry +// entries, no admin rights. Parts headless LibreOffice never uses (help, gallery, most dictionaries) are removed. // * conda-forge packages (Ghostscript, the Microsoft C++ runtime) are zip files holding zstd-compressed tarballs; // only their Windows binaries (Library/bin) are unpacked, with a managed decompressor (no 7-Zip, no installer). // * A marker file with the component's hash records what is installed, so an engines.json update is detected. @@ -182,12 +182,8 @@ private async Task UnpackAsync(EngineComponent component, string file, Cancellat Replace(component.Id, SingleRoot(staging)); break; case "msi": - await UnpackMsiAsync(file, staging, Path.Combine(Path.GetTempPath(), "filee-msiexec.log"), cancellationToken); - var soffice = Directory.EnumerateFiles(staging, "soffice.exe", SearchOption.AllDirectories).FirstOrDefault() - ?? throw new InvalidDataException("soffice.exe not found in the LibreOffice package."); - var libreOffice = Path.GetDirectoryName(Path.GetDirectoryName(soffice)!)!; // folder with program/, share/ - TrimLibreOffice(libreOffice); - Replace(component.Id, libreOffice); + await UnpackMsiAsync(component.Id, file, staging, Path.Combine(Path.GetTempPath(), "filee-msiexec.log"), cancellationToken); + Replace(component.Id, MsiProgramFolder(component.Id, staging)); break; case "conda": UnpackConda(file, staging, cancellationToken); @@ -216,10 +212,29 @@ private async Task UnpackAsync(EngineComponent component, string file, Cancellat } } - private static async Task UnpackMsiAsync(string msi, string target, string log, CancellationToken cancellationToken) + /// + /// The folder to keep from an administratively installed MSI, which lays files out as they would be installed + /// (PFiles64\...): LibreOffice's folder with program/ and share/ (trimmed), calibre's folder with ebook-convert. + /// + private static string MsiProgramFolder(string id, string staging) + { + if (id == "calibre") + { + var convert = Directory.EnumerateFiles(staging, "ebook-convert.exe", SearchOption.AllDirectories).FirstOrDefault() + ?? throw new InvalidDataException("ebook-convert.exe not found in the calibre package."); + return Path.GetDirectoryName(convert)!; + } + var soffice = Directory.EnumerateFiles(staging, "soffice.exe", SearchOption.AllDirectories).FirstOrDefault() + ?? throw new InvalidDataException("soffice.exe not found in the LibreOffice package."); + var libreOffice = Path.GetDirectoryName(Path.GetDirectoryName(soffice)!)!; // folder with program/, share/ + TrimLibreOffice(libreOffice); + return libreOffice; + } + + private static async Task UnpackMsiAsync(string id, string msi, string target, string log, CancellationToken cancellationToken) { if (!OperatingSystem.IsWindows()) - throw new PlatformNotSupportedException("The LibreOffice package is a Windows installer."); + throw new PlatformNotSupportedException($"The {id} package is a Windows installer."); Directory.CreateDirectory(target); // msiexec parses its own command line: the property value must be quoted as TARGETDIR="...", so the // argument list escaping of ProcessRunner cannot be used here. @@ -244,7 +259,7 @@ private static async Task UnpackMsiAsync(string msi, string target, string log, throw new InvalidOperationException(process.ExitCode switch { 1618 => "Another installation is running. Try again when it has finished.", - _ => $"Unpacking LibreOffice failed (msiexec exit code {process.ExitCode}, log: {log}).", + _ => $"Unpacking {id} failed (msiexec exit code {process.ExitCode}, log: {log}).", }); } } diff --git a/src/Filee.Engines/Infrastructure/ProcessRunner.cs b/src/Filee.Engines/Infrastructure/ProcessRunner.cs index 7590192..521b4a3 100644 --- a/src/Filee.Engines/Infrastructure/ProcessRunner.cs +++ b/src/Filee.Engines/Infrastructure/ProcessRunner.cs @@ -15,13 +15,15 @@ public static class ProcessRunner /// Starts with the given arguments (no shell, arguments are escaped), /// waits for it to exit and kills the whole process tree on timeout or cancellation. ///
+ /// Called with every line of standard output and error as it arrives (e.g. to parse progress). public static async Task RunAsync( string executable, IEnumerable arguments, TimeSpan timeout, CancellationToken cancellationToken, string? workingDirectory = null, - IDictionary? environment = null) + IDictionary? environment = null, + Action? onOutput = null) { var psi = new ProcessStartInfo(executable) { @@ -42,8 +44,8 @@ public static async Task RunAsync( using var process = new Process { StartInfo = psi }; var stdout = new StringBuilder(); var stderr = new StringBuilder(); - process.OutputDataReceived += (_, e) => { if (e.Data is not null) lock (stdout) stdout.AppendLine(e.Data); }; - process.ErrorDataReceived += (_, e) => { if (e.Data is not null) lock (stderr) stderr.AppendLine(e.Data); }; + process.OutputDataReceived += (_, e) => { if (e.Data is not null) { lock (stdout) stdout.AppendLine(e.Data); onOutput?.Invoke(e.Data); } }; + process.ErrorDataReceived += (_, e) => { if (e.Data is not null) { lock (stderr) stderr.AppendLine(e.Data); onOutput?.Invoke(e.Data); } }; if (!process.Start()) throw new InvalidOperationException($"Could not start {executable}."); diff --git a/src/Filee.Engines/Infrastructure/engines.json b/src/Filee.Engines/Infrastructure/engines.json index d7759e2..3dfc93b 100644 --- a/src/Filee.Engines/Infrastructure/engines.json +++ b/src/Filee.Engines/Infrastructure/engines.json @@ -63,6 +63,13 @@ "sha256": "8D31E162F1616E37AAB3FA2DB991B97E1B8DBEB1C8465FD81A81BE16FF91B328", "size": 99899672, "kind": "zip" + }, + "calibre": { + "version": "9.15.0", + "url": "https://download.calibre-ebook.com/9.15.0/calibre-64bit-9.15.0.msi", + "sha256": "0F96AE06165C2419607C1C66726C091A78E89E08EA69F077CF3E2C884E800853", + "size": 226549760, + "kind": "msi" } } } diff --git a/src/Filee.Engines/Media/FfmpegConverter.cs b/src/Filee.Engines/Media/FfmpegConverter.cs index 10cedc0..6e9eccb 100644 --- a/src/Filee.Engines/Media/FfmpegConverter.cs +++ b/src/Filee.Engines/Media/FfmpegConverter.cs @@ -57,7 +57,7 @@ public EngineStatus GetStatus() return EngineStatus.Unavailable("engine.reason.not_installed"); try { - return EngineStatus.Available($"{tools.Ffmpeg} (FFmpeg {VersionOf(tools.Ffmpeg)})"); + return EngineStatus.Available(tools.Ffmpeg, $"FFmpeg {VersionOf(tools.Ffmpeg)}"); } catch (Exception ex) when (ex is InvalidOperationException or System.ComponentModel.Win32Exception or IOException) { diff --git a/tests/Filee.App.Tests/RenderTests.cs b/tests/Filee.App.Tests/RenderTests.cs index f49db3e..ed6d66c 100644 --- a/tests/Filee.App.Tests/RenderTests.cs +++ b/tests/Filee.App.Tests/RenderTests.cs @@ -103,12 +103,12 @@ public void First_run_engine_setup_renders(string language) window.Show(); Pump(); - // Everything not installed is offered; small downloads are pre-selected, the large LibreOffice and FFmpeg and the + // Everything not installed is offered; small downloads are pre-selected, the large LibreOffice, FFmpeg and calibre and the // EPS-only Ghostscript not. Assert.All(vm.Packages.Where(p => !p.IsInstalled), p => Assert.Equal( Filee.Engines.Infrastructure.EngineDownloads.DownloadSize(p.Package) < EngineSetupViewModel.PreselectLimit && p.Package.Id != "ghostscript", p.Selected)); - Assert.DoesNotContain(vm.Packages, p => p.Package.Id is "libreoffice" or "ffmpeg" or "ghostscript" && p.Selected); + Assert.DoesNotContain(vm.Packages, p => p.Package.Id is "libreoffice" or "ffmpeg" or "calibre" or "ghostscript" && p.Selected); Assert.Equal(vm.Packages.Any(p => p.Selected && p.CanInstall), vm.InstallCommand.CanExecute(null)); Save(window, $"engine-setup-{language}.png"); window.Close(); diff --git a/tests/Filee.Engines.Tests/EbookBuilders.cs b/tests/Filee.Engines.Tests/EbookBuilders.cs new file mode 100644 index 0000000..130905d --- /dev/null +++ b/tests/Filee.Engines.Tests/EbookBuilders.cs @@ -0,0 +1,444 @@ +// Generates e-book test inputs in code: EPUB 3 packages, FB2 (windows-1251), MOBI 6 files (PalmDOC or HUFF/CDIC +// compressed, with EXTH metadata, trailing entries and picture records), AZW4 Print Replica wrappers and comic +// archives. The MOBI writer here is test-only: it follows the MobileRead format description. + +using System.Buffers.Binary; +using System.Formats.Tar; +using System.IO.Compression; +using System.Text; +using ImageMagick; + +namespace Filee.Engines.Tests; + +internal static class EbookBuilders +{ + /// A PNG of the given size and colour. + public static byte[] Png(uint width, uint height, MagickColor color) + { + using var image = new MagickImage(color, width, height); + return image.ToByteArray(MagickFormat.Png); + } + + // ───────────────────────── EPUB ───────────────────────── + + /// + /// An EPUB 3 with three chapters in OEBPS/Text (bold / italic, a class rule, a link from chapter 1 to an anchor in + /// chapter 3, a picture in OEBPS/Images, a table), a cover image, a style sheet, a nav document and an NCX. + /// + public static string Epub(string folder, string name = "소설.epub") + { + var path = Path.Combine(folder, name); + using var zip = ZipFile.Open(path, ZipArchiveMode.Create); + Entry(zip, "mimetype", "application/epub+zip", CompressionLevel.NoCompression); + Entry(zip, "META-INF/container.xml", """ + + + + + """); + Entry(zip, "OEBPS/content.opf", """ + + + + urn:uuid:12345678-1234-1234-1234-123456789012 + 별빛 이야기 + 김작가 + ko + 2026-01-01T00:00:00Z + + + + + + + + + + + + + + + + + + """); + Entry(zip, "OEBPS/nav.xhtml", """ + + 목차 + + """); + Entry(zip, "OEBPS/toc.ncx", """ + + 별빛 + 1장 + """); + Entry(zip, "OEBPS/Styles/book.css", ".loud { font-weight: bold } p.centre { text-align: center } body { color: #333 }"); + Entry(zip, "OEBPS/Text/ch1.xhtml", """ + + + <link rel="stylesheet" type="text/css" href="../Styles/book.css"/></head> + <body> + <h1>첫 번째 장</h1> + <p>밤하늘에 <b>별</b>이 <i>반짝</i>였다. <span class="loud">크게</span> 외쳤다.</p> + <p class="centre">가운데 문단</p> + <p>주석은 <a href="ch%203.xhtml#note1">여기</a>를 보세요.</p> + <div id="empty"/> + <p>빈 div 다음 문단</p> + </body></html> + """); + Entry(zip, "OEBPS/Text/ch2.xhtml", """ + <?xml version="1.0" encoding="utf-8"?> + <html xmlns="http://www.w3.org/1999/xhtml"><head><title>둘 + +

두 번째 장

+

그림이 있는 장입니다.

+

그림

+
이름값
별42
+ + """); + Entry(zip, "OEBPS/Text/ch 3.xhtml", """ + + 셋 + +

세 번째 장

+

주석 내용입니다.

+
  • 목록 하나
  • 목록 둘
+ + """); + Entry(zip, "OEBPS/Images/pic.png", Png(160, 90, MagickColors.Teal)); + Entry(zip, "OEBPS/Images/cover.png", Png(300, 450, MagickColors.Navy)); + return path; + } + + public static void Entry(ZipArchive zip, string name, string text, CompressionLevel level = CompressionLevel.Optimal) + { + using var writer = new StreamWriter(zip.CreateEntry(name, level).Open(), new UTF8Encoding(false)); + writer.Write(text); + } + + public static void Entry(ZipArchive zip, string name, byte[] data) + { + using var stream = zip.CreateEntry(name, CompressionLevel.NoCompression).Open(); + stream.Write(data); + } + + // ───────────────────────── FB2 ───────────────────────── + + /// A windows-1251 FB2 book with two chapters, a nested section, a poem, a note and a cover picture. + public static string Fb2(string folder) + { + var cover = Convert.ToBase64String(Png(120, 180, MagickColors.DarkRed)); + var xml = $""" + + + + + prose_classic + ЛевТолстой + Война и мир + ru + + + + + <p>Война и мир</p> +
+ <p>Глава первая</p> +

Первый абзац и курсив.[1]

+ + Строка стиха одинСтрока стиха два +
+ <p>Подглава</p> +

Текст подглавы.

+
+
+
+ <p>Глава вторая</p> +

Цитата из книги.

+
AB
12
+
+ + +
<p>1</p>

Текст примечания.

+ + {cover} +
+ """; + var path = Path.Combine(folder, "voina.fb2"); + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + File.WriteAllBytes(path, Encoding.GetEncoding(1251).GetBytes(xml)); + return path; + } + + // ───────────────────────── MOBI ───────────────────────── + + /// Options of a generated MOBI file. + /// 1 none, 2 PalmDOC, 17480 HUFF/CDIC. + /// 0 none; 2 marks the book as DRM-protected (the text stays plain). + /// EXTH 201: index of the cover among the pictures. + public sealed record MobiOptions(int Compression = 2, int Encryption = 0, int? Cover = null, string Title = "Mobi Title", string Author = "Mobi Author"); + + /// A MOBI 6 book from HTML (UTF-8) and pictures (referenced as recindex="00001", ...). + public static byte[] Mobi(string html, IReadOnlyList pictures, MobiOptions options) => + Mobi(Encoding.UTF8.GetBytes(html), pictures, options); + + public static byte[] Mobi(byte[] text, IReadOnlyList pictures, MobiOptions options) + { + const int RecordSize = 4096; + var records = new List { Array.Empty() }; // record 0 is filled in last + for (var offset = 0; offset < text.Length; offset += RecordSize) + { + var chunk = text.AsSpan(offset, Math.Min(RecordSize, text.Length - offset)).ToArray(); + var data = options.Compression switch + { + 2 => PalmDocCompress(chunk), + _ => chunk, // 1, and HUFF/CDIC with the identity code table below + }; + // Trailing entries (extra flags 0b11): a multibyte byte, then a 2-byte indexing entry at the very end. + records.Add([.. data, 0x00, 0x00, 0x82]); + } + var textRecords = records.Count - 1; + var firstImage = pictures.Count > 0 ? records.Count : -1; + records.AddRange(pictures); + var huff = -1; + if (options.Compression == 17480) + { + huff = records.Count; + records.Add(IdentityHuff()); + records.Add(IdentityCdic()); + } + records.Add([0xE9, 0x8E, 0x0D, 0x0A]); // EOF record + + records[0] = Record0(text.Length, textRecords, firstImage, huff, options); + return PalmDb(options.Title, "BOOKMOBI", records); + } + + private static byte[] Record0(int textLength, int textRecords, int firstImage, int huff, MobiOptions options) + { + var exth = new List<(int Type, byte[] Data)> + { + (100, Encoding.UTF8.GetBytes(options.Author)), + (503, Encoding.UTF8.GetBytes(options.Title)), + (524, "en"u8.ToArray()), + }; + if (options.Cover is { } cover) + exth.Add((201, BigEndian(cover))); + var exthBytes = new List(); + exthBytes.AddRange("EXTH"u8.ToArray()); + var exthLength = 12 + exth.Sum(e => 8 + e.Data.Length); + var padding = (4 - exthLength % 4) % 4; + exthBytes.AddRange(BigEndian(exthLength + padding)); + exthBytes.AddRange(BigEndian(exth.Count)); + foreach (var (type, data) in exth) + { + exthBytes.AddRange(BigEndian(type)); + exthBytes.AddRange(BigEndian(8 + data.Length)); + exthBytes.AddRange(data); + } + exthBytes.AddRange(new byte[padding]); + + const int MobiLength = 0xE8; + var header = new byte[16 + MobiLength]; + var span = header.AsSpan(); + BinaryPrimitives.WriteUInt16BigEndian(span[0..], (ushort)options.Compression); + BinaryPrimitives.WriteUInt32BigEndian(span[4..], (uint)textLength); + BinaryPrimitives.WriteUInt16BigEndian(span[8..], (ushort)textRecords); + BinaryPrimitives.WriteUInt16BigEndian(span[10..], 4096); + BinaryPrimitives.WriteUInt16BigEndian(span[12..], (ushort)options.Encryption); + "MOBI"u8.CopyTo(span[16..]); + BinaryPrimitives.WriteUInt32BigEndian(span[0x14..], MobiLength); + BinaryPrimitives.WriteUInt32BigEndian(span[0x18..], 2); // book + BinaryPrimitives.WriteUInt32BigEndian(span[0x1C..], 65001); // UTF-8 + BinaryPrimitives.WriteUInt32BigEndian(span[0x20..], 0x1234); + BinaryPrimitives.WriteUInt32BigEndian(span[0x24..], 6); // MOBI 6 + for (var o = 0x28; o <= 0x4C; o += 4) + BinaryPrimitives.WriteUInt32BigEndian(span[o..], uint.MaxValue); + BinaryPrimitives.WriteUInt32BigEndian(span[0x50..], (uint)(textRecords + 1)); + var nameOffset = header.Length + exthBytes.Count; + var name = Encoding.UTF8.GetBytes(options.Title); + BinaryPrimitives.WriteUInt32BigEndian(span[0x54..], (uint)nameOffset); + BinaryPrimitives.WriteUInt32BigEndian(span[0x58..], (uint)name.Length); + BinaryPrimitives.WriteUInt32BigEndian(span[0x5C..], 0x09); // English + BinaryPrimitives.WriteUInt32BigEndian(span[0x68..], 6); + BinaryPrimitives.WriteUInt32BigEndian(span[0x6C..], firstImage < 0 ? uint.MaxValue : (uint)firstImage); + BinaryPrimitives.WriteUInt32BigEndian(span[0x70..], huff < 0 ? 0 : (uint)huff); + BinaryPrimitives.WriteUInt32BigEndian(span[0x74..], huff < 0 ? 0u : 2u); + BinaryPrimitives.WriteUInt32BigEndian(span[0x80..], 0x40); // EXTH present + BinaryPrimitives.WriteUInt32BigEndian(span[0xA4..], uint.MaxValue); + BinaryPrimitives.WriteUInt32BigEndian(span[0xA8..], uint.MaxValue); + BinaryPrimitives.WriteUInt16BigEndian(span[0xC0..], 1); + BinaryPrimitives.WriteUInt16BigEndian(span[0xC2..], (ushort)textRecords); + BinaryPrimitives.WriteUInt32BigEndian(span[0xC4..], 1); + BinaryPrimitives.WriteUInt32BigEndian(span[0xC8..], uint.MaxValue); + BinaryPrimitives.WriteUInt32BigEndian(span[0xD0..], uint.MaxValue); + BinaryPrimitives.WriteUInt16BigEndian(span[0xF2..], 0b11); // multibyte + one indexing entry + return [.. header, .. exthBytes, .. name, 0, 0, 0, 0]; + } + + /// A Palm database with the given type/creator and records. + public static byte[] PalmDb(string name, string typeCreator, List records) + { + var header = new byte[78 + records.Count * 8 + 2]; + var nameBytes = Encoding.ASCII.GetBytes(name.Replace(' ', '_')); + nameBytes.AsSpan(0, Math.Min(31, nameBytes.Length)).CopyTo(header); + Encoding.ASCII.GetBytes(typeCreator).CopyTo(header, 60); + BinaryPrimitives.WriteUInt16BigEndian(header.AsSpan(76), (ushort)records.Count); + var offset = header.Length; + for (var i = 0; i < records.Count; i++) + { + BinaryPrimitives.WriteUInt32BigEndian(header.AsSpan(78 + i * 8), (uint)offset); + BinaryPrimitives.WriteUInt32BigEndian(header.AsSpan(78 + i * 8 + 4), (uint)(i * 2) & 0x00FFFFFF); + offset += records[i].Length; + } + using var stream = new MemoryStream(); + stream.Write(header); + foreach (var record in records) + stream.Write(record); + return stream.ToArray(); + } + + /// PalmDOC compression using every code: literals, literal runs, back-references and space pairs. + public static byte[] PalmDocCompress(byte[] input) + { + var output = new List(); + var i = 0; + while (i < input.Length) + { + // Back-reference: the longest match of 3..10 bytes within the last 2047 bytes. + var (bestLength, bestDistance) = (0, 0); + for (var distance = 1; distance <= Math.Min(2047, i); distance++) + { + var length = 0; + while (length < 10 && i + length < input.Length && input[i + length] == input[i - distance + length]) + length++; + if (length > bestLength) + (bestLength, bestDistance) = (length, distance); + } + if (bestLength >= 3) + { + var pair = 0x8000 | (bestDistance << 3) | (bestLength - 3); + output.Add((byte)(pair >> 8)); + output.Add((byte)pair); + i += bestLength; + continue; + } + var c = input[i]; + if (c == ' ' && i + 1 < input.Length && input[i + 1] is >= 0x40 and <= 0x7F) + { + output.Add((byte)(input[i + 1] ^ 0x80)); + i += 2; + } + else if (c is 0 or (>= 0x09 and <= 0x7F)) + { + output.Add(c); + i++; + } + else + { + // Bytes that need escaping (1..8 and 0x80+) go into a literal run of up to 8 bytes. + var run = new List(); + while (i < input.Length && run.Count < 8 && input[i] is (>= 1 and <= 8) or >= 0x80) + run.Add(input[i++]); + output.Add((byte)run.Count); + output.AddRange(run); + } + } + return [.. output]; + } + + /// A HUFF table in which every byte is an 8-bit code for dictionary entry of the same number. + private static byte[] IdentityHuff() + { + var huff = new byte[24 + 256 * 4 + 64 * 4]; + "HUFF"u8.CopyTo(huff); + BinaryPrimitives.WriteUInt32BigEndian(huff.AsSpan(4), 24); + BinaryPrimitives.WriteUInt32BigEndian(huff.AsSpan(8), 24); + BinaryPrimitives.WriteUInt32BigEndian(huff.AsSpan(12), 24 + 256 * 4); + for (var b = 0; b < 256; b++) + { + // Code length 8, terminal; the entry index is (max code - code) >> 24 = raw - b, so raw = 2b gives entry b. + var value = ((uint)(2 * b) << 8) | 0x80 | 8; + BinaryPrimitives.WriteUInt32BigEndian(huff.AsSpan(24 + b * 4), value); + } + return huff; + } + + /// A CDIC with 256 one-byte literal phrases. + private static byte[] IdentityCdic() + { + var cdic = new byte[16 + 256 * 2 + 256 * 3]; + "CDIC"u8.CopyTo(cdic); + BinaryPrimitives.WriteUInt32BigEndian(cdic.AsSpan(4), 16); + BinaryPrimitives.WriteUInt32BigEndian(cdic.AsSpan(8), 256); + BinaryPrimitives.WriteUInt32BigEndian(cdic.AsSpan(12), 8); + for (var b = 0; b < 256; b++) + { + var offset = 256 * 2 + b * 3; + BinaryPrimitives.WriteUInt16BigEndian(cdic.AsSpan(16 + b * 2), (ushort)offset); + BinaryPrimitives.WriteUInt16BigEndian(cdic.AsSpan(16 + offset), 0x8001); // literal, 1 byte + cdic[16 + offset + 2] = (byte)b; + } + return cdic; + } + + /// An AZW4 (Print Replica) wrapper: "%MOP", one table with one section holding the PDF. + public static byte[] PrintReplica(byte[] pdf) + { + var text = new byte[20 + pdf.Length]; + "%MOP"u8.CopyTo(text); + BinaryPrimitives.WriteUInt32BigEndian(text.AsSpan(4), 1); // tables + BinaryPrimitives.WriteUInt32BigEndian(text.AsSpan(8), 1); // sections in table 1 + BinaryPrimitives.WriteUInt32BigEndian(text.AsSpan(12), 20); // offset + BinaryPrimitives.WriteUInt32BigEndian(text.AsSpan(16), (uint)pdf.Length); + pdf.CopyTo(text, 20); + return Mobi(text, [], new MobiOptions(Compression: 1, Title: "Replica")); + } + + private static byte[] BigEndian(int value) + { + var bytes = new byte[4]; + BinaryPrimitives.WriteUInt32BigEndian(bytes, (uint)value); + return bytes; + } + + // ───────────────────────── Comics ───────────────────────── + + /// Pages named so that only a natural sort gets the order right, each with its own size. + public static readonly (string Name, uint Width, uint Height)[] ComicPages = + [ + ("chapter/page1.png", 200, 300), + ("chapter/page2.jpg", 220, 330), + ("chapter/page10.png", 240, 360), + ]; + + public static string Cbz(string folder) + { + var path = Path.Combine(folder, "만화.cbz"); + using var zip = ZipFile.Open(path, ZipArchiveMode.Create); + foreach (var (name, width, height) in ComicPages) + Entry(zip, name, Picture(name, width, height)); + Entry(zip, "__MACOSX/chapter/._page1.png", [0, 1, 2]); + Entry(zip, "ComicInfo.xml", ""); + return path; + } + + public static string Cbt(string folder) + { + var path = Path.Combine(folder, "comic.cbt"); + using var stream = File.Create(path); + using var tar = new TarWriter(stream); + foreach (var (name, width, height) in ComicPages) + { + var entry = new PaxTarEntry(TarEntryType.RegularFile, name) { DataStream = new MemoryStream(Picture(name, width, height)) }; + tar.WriteEntry(entry); + } + return path; + } + + private static byte[] Picture(string name, uint width, uint height) + { + using var image = new MagickImage(MagickColors.Coral, width, height); + return image.ToByteArray(name.EndsWith(".jpg", StringComparison.Ordinal) ? MagickFormat.Jpeg : MagickFormat.Png); + } +} diff --git a/tests/Filee.Engines.Tests/EbookTests.cs b/tests/Filee.Engines.Tests/EbookTests.cs new file mode 100644 index 0000000..0dc08c4 --- /dev/null +++ b/tests/Filee.Engines.Tests/EbookTests.cs @@ -0,0 +1,538 @@ +// E-books without external programs (EPUB, MOBI, FB2, HTMLZ, TXTZ, comics) and with the optional calibre package. +// Inputs are generated by EbookBuilders; outputs are parsed back (EPUB structure, PDF page counts, text content). + +using System.IO.Compression; +using System.Text; +using System.Xml.Linq; +using Filee.Core.Conversion; +using Filee.Core.Presets; +using Filee.Engines.Ebooks; +using ImageMagick; +using PdfSharp.Pdf.IO; + +namespace Filee.Engines.Tests; + +public class EbookTests(EngineFixture fx) : IClassFixture +{ + private static readonly string[] OptionalEngines = ["pandoc", "libreoffice", "calibre"]; + + private RoutePlanner BuiltInPlanner(params string[] assumeInstalled) => + new ConverterCatalog(fx.Converters.Where(c => !OptionalEngines.Contains(c.Id))) { Priority = fx.Catalog.Priority } + .CreatePlanner(assumeInstalled); + + private async Task ConvertAsync(string input, string target, Preset? preset = null) + { + preset ??= new Preset(); + preset.TargetFormat = target; + var job = await fx.ConvertAsync([input], preset); + Assert.True(job.State == JobState.Completed, string.Join("; ", job.Files.Select(f => $"{f.ErrorKey} {f.ErrorDetail}"))); + return job.Outputs.Single(); + } + + private async Task FailureAsync(string input, string target) + { + var job = await fx.ConvertAsync([input], new Preset { TargetFormat = target }); + Assert.Equal(JobState.Failed, job.State); + return job.Files.Single().ErrorDetail ?? ""; + } + + private bool HasRhwp => fx.Catalog.StatusOf(fx.Converters.Single(c => c.Id == "rhwp")).IsAvailable; + + private bool HasCalibre => fx.Catalog.StatusOf(fx.Converters.Single(c => c.Id == "calibre")).IsAvailable; + + private static int PdfPages(string pdf) + { + using var document = PdfReader.Open(pdf, PdfDocumentOpenMode.Import); + return document.PageCount; + } + + /// The text of an EPUB read back with the built-in reader. + private string EpubText(string epub) => + Hwp.Hwpx.PlainTextWriter.Write(EpubReader.ReadBook(epub, fx.NewFolder(), TestContext.Current.CancellationToken).Document); + + // ───────────────────────── Routes ───────────────────────── + + [Fact] + public void E_book_routes_use_the_built_in_engine() + { + var planner = BuiltInPlanner("rhwp"); + + Assert.Equal(["ebook", "rhwp"], planner.Plan("epub", "pdf")!.Steps.Select(s => s.Converter.Id)); + Assert.Equal(["ebook", "rhwp", "pdfium"], planner.Plan("mobi", "png")!.Steps.Select(s => s.Converter.Id)); + foreach (var (from, to) in new[] { ("mobi", "epub"), ("azw3", "epub"), ("fb2", "epub"), ("epub", "fb2"), ("md", "epub"), ("docx", "epub"), ("epub", "txt"), ("cbr", "pdf"), ("pdf", "cbz"), ("azw4", "pdf"), ("htmlz", "txtz") }) + Assert.Equal("ebook", Assert.Single(planner.Plan(from, to)!.Steps).Converter.Id); + Assert.Null(planner.Plan("epub", "mobi")); // needs calibre + } + + [Fact] + public void Calibre_adds_kindle_output_and_rare_formats_but_never_wins_over_built_in_routes() + { + var planner = fx.Catalog.CreatePlanner(["calibre"]); + + Assert.Equal("calibre", Assert.Single(planner.Plan("epub", "azw3")!.Steps).Converter.Id); + Assert.Equal(["ebook", "calibre"], planner.Plan("fb2", "mobi")!.Steps.Select(s => s.Converter.Id)); + Assert.Equal(["calibre", "ebook"], planner.Plan("lit", "txt")!.Steps.Select(s => s.Converter.Id)); + Assert.Equal("ebook", Assert.Single(planner.Plan("mobi", "epub")!.Steps).Converter.Id); + Assert.Equal("ebook", Assert.Single(planner.Plan("epub", "fb2")!.Steps).Converter.Id); + } + + // ───────────────────────── EPUB ───────────────────────── + + [Fact] + public void Epub_reader_keeps_chapters_links_pictures_and_metadata() + { + var book = EpubReader.ReadBook(EbookBuilders.Epub(fx.NewFolder()), fx.NewFolder(), TestContext.Current.CancellationToken); + + Assert.Equal("별빛 이야기", book.Metadata.Title); + Assert.Equal(["김작가"], book.Metadata.Authors); + Assert.Equal("ko", book.Metadata.Language); + Assert.EndsWith("cover.png", book.Metadata.CoverImage); + var blocks = book.Document.Sections.Single().Blocks; + var headings = blocks.OfType().Where(p => p.HeadingLevel == 1).ToList(); + Assert.Equal(["첫 번째 장", "두 번째 장", "세 번째 장"], headings.Select(h => Hwp.Hwpx.HDocumentWalker.Text(h.Inlines))); + Assert.Equal([false, true, true], headings.Select(h => h.PageBreakBefore)); // every chapter on a new page + var inlines = blocks.SelectMany(b => Hwp.Hwpx.HDocumentWalker.Inlines([b])).ToList(); + var link = inlines.OfType().Single(); + Assert.Contains(inlines.OfType(), b => "#" + b.Name == link.Target); // chapter 1 → note in "ch 3.xhtml" + Assert.Single(inlines.OfType()); + Assert.Single(blocks.OfType()); + Assert.Contains(inlines.OfType(), t => t.Text == "크게" && t.Format.Bold is true); // class rule from the linked CSS + Assert.Contains(inlines.OfType(), t => t.Text == "빈 div 다음 문단"); //
did not swallow the page + } + + [Fact] + public async Task Epub_to_pdf_puts_every_chapter_on_its_own_page() + { + Assert.SkipUnless(HasRhwp, "rhwp not found (pwsh build/fetch-engines.ps1 -Only rhwp)"); + var pdf = await ConvertAsync(EbookBuilders.Epub(fx.NewFolder()), "pdf"); + + Assert.True(PdfPages(pdf) >= 3); + } + + [Fact] + public async Task Epub_to_txt_html_and_markdown() + { + var epub = EbookBuilders.Epub(fx.NewFolder()); + + var text = await File.ReadAllTextAsync(await ConvertAsync(epub, "txt"), TestContext.Current.CancellationToken); + foreach (var expected in new[] { "첫 번째 장", "밤하늘에 별이 반짝였다.", "주석 내용입니다.", "● 목록 하나", "이름\t값" }) + Assert.Contains(expected, text); + Assert.DoesNotContain("<", text); + + var html = await File.ReadAllTextAsync(await ConvertAsync(epub, "html"), TestContext.Current.CancellationToken); + Assert.Contains("

첫 번째 장

", html); + Assert.Contains("src=\"data:image/png;base64,", html); // one self-contained file + Assert.Contains("별", html); + Assert.Contains("별빛 이야기", html); + + var md = await ConvertAsync(epub, "md"); + var markdown = await File.ReadAllTextAsync(md, TestContext.Current.CancellationToken); + Assert.Contains("# 첫 번째 장", markdown); + Assert.Contains("**별**", markdown); + Assert.Contains("| 이름 | 값 |", markdown); + var picture = System.Text.RegularExpressions.Regex.Match(markdown, @"!\[\]\(<([^>]+)>\)").Groups[1].Value; + Assert.True(File.Exists(Path.Combine(Path.GetDirectoryName(md)!, Uri.UnescapeDataString(picture)))); + } + + [Fact] + public async Task Markdown_to_epub_is_a_valid_package_that_reads_back() + { + var dir = fx.NewFolder(); + File.WriteAllBytes(Path.Combine(dir, "그림.png"), EbookBuilders.Png(120, 80, MagickColors.Gold)); + var md = Path.Combine(dir, "책.md"); + await File.WriteAllTextAsync(md, """ + --- + title: 마크다운 책 + --- + + # 1장 시작 + + 첫 문단에 **굵게**와 [링크](https://example.com)가 있습니다.[^1] + + ![그림](그림.png) + + ## 1.1 절 + + - 하나 + - 둘 + + # 2장 끝 + + | A | B | + |---|---| + | 1 | 2 | + + [^1]: 각주입니다. + """, TestContext.Current.CancellationToken); + + var epub = await ConvertAsync(md, "epub"); + + var package = EpubAssert.Valid(epub); + Assert.Equal("마크다운 책", package.Title); + Assert.Equal("ko", package.Language); + Assert.Equal(2, package.Chapters); // split at the level-1 headings + Assert.Equal(1, package.Images); + Assert.Contains("1장 시작", package.Toc); + Assert.Contains("1.1 절", package.Toc); + var text = EpubText(epub); + foreach (var expected in new[] { "1장 시작", "굵게", "하나", "2장 끝", "각주입니다." }) + Assert.Contains(expected, text); + } + + [Fact] + public async Task Text_html_and_docx_become_epub() + { + var dir = fx.NewFolder(); + var txt = Path.Combine(dir, "메모.txt"); + await File.WriteAllTextAsync(txt, "첫 문단의 첫 줄\n같은 문단\n\n둘째 문단", TestContext.Current.CancellationToken); + var html = Path.Combine(dir, "page.html"); + await File.WriteAllTextAsync(html, "웹 페이지

제목

본문

", TestContext.Current.CancellationToken); + var docx = new DocxBuilder().Paragraph(DocxBuilder.P(DocxBuilder.R("워드 문서 본문"))).Save(Path.Combine(dir, "워드.docx")); + + EpubAssert.Valid(await ConvertAsync(txt, "epub")); + Assert.Contains("둘째 문단", EpubText(await ConvertAsync(txt, "epub"))); + Assert.Equal("웹 페이지", EpubAssert.Valid(await ConvertAsync(html, "epub")).Title); + Assert.Contains("워드 문서 본문", EpubText(await ConvertAsync(docx, "epub"))); + } + + [Fact] + public async Task Drm_protected_epub_is_refused_with_a_clear_message() + { + var dir = fx.NewFolder(); + var epub = EbookBuilders.Epub(dir); + using (var zip = ZipFile.Open(epub, ZipArchiveMode.Update)) + EbookBuilders.Entry(zip, "META-INF/encryption.xml", """ + + + + + """); + + Assert.Contains("DRM", await FailureAsync(epub, "txt")); + } + + // ───────────────────────── MOBI ───────────────────────── + + private const string MobiHtml = """ + +

Chapter One

+

It was a dark and stormy night; the rain fell in torrents — except at occasional intervals.

+

See the note below. Ünïcödé text and 한글 too.

+

+ +

Chapter Two

+

The note is here.

+ + """; + + /// MOBI HTML with a filepos link pointing at "The note is here." (a byte offset in the text). + private static string MobiSource() + { + var placeholder = MobiHtml.Replace("FILEPOS", "0000000000", StringComparison.Ordinal); + var target = Encoding.UTF8.GetBytes(placeholder[..placeholder.IndexOf("

The note", StringComparison.Ordinal)]).Length; + return MobiHtml.Replace("FILEPOS", target.ToString("0000000000", System.Globalization.CultureInfo.InvariantCulture), StringComparison.Ordinal); + } + + [Theory] + [InlineData(2)] // PalmDOC + [InlineData(17480)] // HUFF/CDIC + [InlineData(1)] // uncompressed + public async Task Mobi_to_epub_keeps_text_pictures_links_and_metadata(int compression) + { + var dir = fx.NewFolder(); + var mobi = Path.Combine(dir, "book.mobi"); + var cover = EbookBuilders.Png(100, 150, MagickColors.DarkBlue); + await File.WriteAllBytesAsync(mobi, EbookBuilders.Mobi(MobiSource(), [EbookBuilders.Png(64, 32, MagickColors.Orange), cover], + new EbookBuilders.MobiOptions(Compression: compression, Cover: 1, Title: "Stormy Night", Author: "E. Bulwer-Lytton")), TestContext.Current.CancellationToken); + + var epub = await ConvertAsync(mobi, "epub"); + + var package = EpubAssert.Valid(epub); + Assert.Equal("Stormy Night", package.Title); + Assert.Contains("E. Bulwer-Lytton", package.Authors); + Assert.Equal(2, package.Images); // the picture and the cover + Assert.True(package.HasCover); + var book = EpubReader.ReadBook(epub, fx.NewFolder(), TestContext.Current.CancellationToken); + var text = Hwp.Hwpx.PlainTextWriter.Write(book.Document); + foreach (var expected in new[] { "Chapter One", "dark and stormy night", "Ünïcödé text and 한글 too", "Chapter Two", "The note is here." }) + Assert.Contains(expected, text); + var inlines = book.Document.Sections.SelectMany(s => Hwp.Hwpx.HDocumentWalker.Inlines(s.Blocks)).ToList(); + var link = inlines.OfType().Single(); + Assert.Contains(inlines.OfType(), b => "#" + b.Name == link.Target); // filepos link survives + Assert.True(inlines.OfType().Single(t => t.Text == "dark").Format.Bold); + } + + [Fact] + public async Task Drm_protected_mobi_is_refused_with_a_clear_message() + { + var mobi = Path.Combine(fx.NewFolder(), "locked.azw"); + await File.WriteAllBytesAsync(mobi, EbookBuilders.Mobi(MobiSource(), [], new EbookBuilders.MobiOptions(Encryption: 2)), TestContext.Current.CancellationToken); + + Assert.Contains("DRM", await FailureAsync(mobi, "epub")); + } + + [Fact] + public async Task Palmdoc_text_books_are_read_as_text() + { + var text = "Plain PalmDOC text.\n\nSecond paragraph of the book."u8.ToArray(); + var record0 = new byte[16]; + record0[1] = 2; // PalmDOC compression + System.Buffers.Binary.BinaryPrimitives.WriteUInt32BigEndian(record0.AsSpan(4), (uint)text.Length); + record0[9] = 1; // one text record + var prc = Path.Combine(fx.NewFolder(), "old.prc"); + await File.WriteAllBytesAsync(prc, EbookBuilders.PalmDb("Old Book", "TEXtREAd", [record0, EbookBuilders.PalmDocCompress(text)]), TestContext.Current.CancellationToken); + + var txt = await File.ReadAllTextAsync(await ConvertAsync(prc, "txt"), TestContext.Current.CancellationToken); + + Assert.Contains("Plain PalmDOC text.", txt); + Assert.Contains("Second paragraph of the book.", txt); + } + + [Fact] + public async Task Print_replica_gives_back_its_pdf() + { + var dir = fx.NewFolder(); + var pdf = await ConvertAsync(EbookBuilders.Cbz(dir), "pdf"); + var azw4 = Path.Combine(dir, "replica.azw4"); + await File.WriteAllBytesAsync(azw4, EbookBuilders.PrintReplica(await File.ReadAllBytesAsync(pdf, TestContext.Current.CancellationToken)), TestContext.Current.CancellationToken); + + var extracted = await ConvertAsync(azw4, "pdf"); + + Assert.Equal(await File.ReadAllBytesAsync(pdf, TestContext.Current.CancellationToken), await File.ReadAllBytesAsync(extracted, TestContext.Current.CancellationToken)); + } + + // ───────────────────────── FB2 ───────────────────────── + + [Fact] + public async Task Fb2_to_epub_keeps_sections_notes_cover_and_the_legacy_encoding() + { + var epub = await ConvertAsync(EbookBuilders.Fb2(fx.NewFolder()), "epub"); + + var package = EpubAssert.Valid(epub); + Assert.Equal("Война и мир", package.Title); + Assert.Equal(["Лев Толстой"], package.Authors); + Assert.Equal("ru", package.Language); + Assert.True(package.HasCover); + Assert.Contains("Глава первая", package.Toc); + Assert.Contains("epub:type=\"footnote\"", package.AllText); // the note became a footnote + var text = EpubText(epub); + foreach (var expected in new[] { "Первый абзац и курсив.", "Строка стиха один", "Текст подглавы.", "Цитата из книги.", "Текст примечания." }) + Assert.Contains(expected, text); + } + + [Fact] + public async Task Epub_to_fb2_round_trips() + { + var fb2 = await ConvertAsync(EbookBuilders.Epub(fx.NewFolder()), "fb2"); + + var xml = XDocument.Load(fb2); + XNamespace ns = "http://www.gribuser.ru/xml/fictionbook/2.0"; + Assert.Equal("별빛 이야기", xml.Descendants(ns + "book-title").Single().Value); + Assert.Equal(3, xml.Root!.Element(ns + "body")!.Elements(ns + "section").Count()); + Assert.Equal(2, xml.Descendants(ns + "binary").Count()); // picture + cover + var book = Fb2Reader.ReadBook(fb2, fx.NewFolder()); + var text = Hwp.Hwpx.PlainTextWriter.Write(book.Document); + foreach (var expected in new[] { "첫 번째 장", "밤하늘에 별이 반짝였다.", "주석 내용입니다.", "목록 하나" }) + Assert.Contains(expected, text); + Assert.Contains(book.Document.Sections.SelectMany(s => Hwp.Hwpx.HDocumentWalker.Inlines(s.Blocks)), i => i is Hwp.Hwpx.HText { Text: "별", Format.Bold: true }); + } + + // ───────────────────────── HTMLZ / TXTZ ───────────────────────── + + [Fact] + public async Task Htmlz_and_txtz_are_written_and_read() + { + var epub = EbookBuilders.Epub(fx.NewFolder()); + + var htmlz = await ConvertAsync(epub, "htmlz"); + using (var zip = ZipFile.OpenRead(htmlz)) + { + Assert.NotNull(zip.GetEntry("index.html")); + Assert.NotNull(zip.GetEntry("metadata.opf")); + Assert.Contains(zip.Entries, e => e.FullName.StartsWith("images/", StringComparison.Ordinal)); + } + var fromHtmlz = await File.ReadAllTextAsync(await ConvertAsync(htmlz, "txt"), TestContext.Current.CancellationToken); + Assert.Contains("주석 내용입니다.", fromHtmlz); + + var txtz = await ConvertAsync(epub, "txtz"); + var book = ZippedText.ReadTxtzBook(txtz, fx.NewFolder()); + Assert.Equal("별빛 이야기", book.Metadata.Title); + Assert.Single(Hwp.Hwpx.HDocumentWalker.Images(book.Document)); + Assert.Contains("밤하늘에", Hwp.Hwpx.PlainTextWriter.Write(book.Document)); + } + + // ───────────────────────── Comics ───────────────────────── + + [Fact] + public async Task Comics_become_pdf_with_one_page_per_picture_in_natural_order() + { + var dir = fx.NewFolder(); + foreach (var comic in new[] { EbookBuilders.Cbz(dir), EbookBuilders.Cbt(dir) }) + { + var pdf = await ConvertAsync(comic, "pdf"); + + using var document = PdfReader.Open(pdf, PdfDocumentOpenMode.Import); + Assert.Equal(EbookBuilders.ComicPages.Length, document.PageCount); + // Pages are as large as their pictures (96 dpi), in the order page1, page2, page10. + for (var i = 0; i < document.PageCount; i++) + Assert.Equal(EbookBuilders.ComicPages[i].Width * 72.0 / 96, document.Pages[i].Width.Point, 1); + } + } + + [Fact] + public async Task Comics_become_fixed_layout_epub_and_pdf_becomes_cbz() + { + var dir = fx.NewFolder(); + var cbz = EbookBuilders.Cbz(dir); + + var epub = await ConvertAsync(cbz, "epub"); + var package = EpubAssert.Valid(epub); + Assert.True(package.FixedLayout); + Assert.Equal(EbookBuilders.ComicPages.Length, package.Chapters); + Assert.Equal(EbookBuilders.ComicPages.Length, package.Images); + + var pdf = await ConvertAsync(cbz, "pdf"); + var back = await ConvertAsync(pdf, "cbz", new Preset { Pdf = { RenderDpi = 96 } }); + using var zip = ZipFile.OpenRead(back); + Assert.Equal(["001.jpg", "002.jpg", "003.jpg"], zip.Entries.Select(e => e.FullName)); + using var first = new MagickImage(zip.Entries[0].Open()); + Assert.Equal(EbookBuilders.ComicPages[0].Width, first.Width, 2u); + + var fromCbt = await ConvertAsync(EbookBuilders.Cbt(dir), "cbz"); + using var repacked = ZipFile.OpenRead(fromCbt); + Assert.Equal(["001.png", "002.jpg", "003.png"], repacked.Entries.Select(e => e.FullName)); + } + + // ───────────────────────── Calibre ───────────────────────── + + [Fact] + public async Task Calibre_writes_kindle_books_that_the_built_in_reader_reads_back() + { + Assert.SkipUnless(HasCalibre, "calibre not found (pwsh build/fetch-engines.ps1 -Only calibre)"); + var epub = EbookBuilders.Epub(fx.NewFolder()); + + foreach (var format in new[] { "azw3", "mobi" }) + { + var kindle = await ConvertAsync(epub, format); + Assert.True(new FileInfo(kindle).Length > 1000); + + // AZW3 is KF8 (skeleton / fragment indexes), MOBI is MOBI 6 (+ KF8 in calibre's joint files). + var book = MobiReader.ReadBook(kindle, fx.NewFolder(), TestContext.Current.CancellationToken); + Assert.Equal("별빛 이야기", book.Metadata.Title); + Assert.Contains("김작가", book.Metadata.Authors); + var text = Hwp.Hwpx.PlainTextWriter.Write(book.Document); + foreach (var expected in new[] { "첫 번째 장", "밤하늘에 별이 반짝였다.", "두 번째 장", "주석 내용입니다.", "목록 둘" }) + Assert.Contains(expected, text); + Assert.NotEmpty(Hwp.Hwpx.HDocumentWalker.Images(book.Document)); + var inlines = book.Document.Sections.SelectMany(s => Hwp.Hwpx.HDocumentWalker.Inlines(s.Blocks)).ToList(); + var link = inlines.OfType().FirstOrDefault(l => l.Target.StartsWith('#')); + Assert.NotNull(link); + Assert.Contains(inlines.OfType(), b => "#" + b.Name == link.Target); + } + } + + [Fact] + public async Task Calibre_reads_and_writes_rare_formats() + { + Assert.SkipUnless(HasCalibre, "calibre not found (pwsh build/fetch-engines.ps1 -Only calibre)"); + var epub = EbookBuilders.Epub(fx.NewFolder()); + + var lit = await ConvertAsync(epub, "lit"); + var text = await File.ReadAllTextAsync(await ConvertAsync(lit, "txt"), TestContext.Current.CancellationToken); // calibre → EPUB → built-in → TXT + + Assert.Contains("밤하늘에", text); + Assert.Contains("주석 내용입니다.", text); + } +} + +///

EPUB structure checks (the OCF / OPF rules epubcheck applies first). +internal static class EpubAssert +{ + /// A package path relative to a folder of the package ("OEBPS/Text" + "../Images/a.png"). + private static string Combine(string folder, string relative) + { + var parts = folder.Split('/', StringSplitOptions.RemoveEmptyEntries).ToList(); + foreach (var part in relative.Split('/')) + { + if (part == "..") + { + if (parts.Count > 0) + parts.RemoveAt(parts.Count - 1); + } + else if (part is not ("." or "")) + { + parts.Add(part); + } + } + return string.Join('/', parts); + } + + public sealed record Package(string Title, string Language, List Authors, int Chapters, int Images, bool HasCover, bool FixedLayout, string Toc, string AllText); + + public static Package Valid(string epub) + { + // "mimetype" is the first entry, stored, without extra field, so its content is at byte 38. + var bytes = File.ReadAllBytes(epub); + Assert.Equal("PK\u0003\u0004", Encoding.ASCII.GetString(bytes, 0, 4)); + Assert.Equal(0, BitConverter.ToUInt16(bytes, 8)); // stored + Assert.Equal(0, BitConverter.ToUInt16(bytes, 28)); // no extra field + Assert.Equal("mimetypeapplication/epub+zip", Encoding.ASCII.GetString(bytes, 30, 28)); + + using var zip = ZipFile.OpenRead(epub); + var container = XDocument.Load(zip.GetEntry("META-INF/container.xml")!.Open()); + var opfPath = container.Descendants().Single(e => e.Name.LocalName == "rootfile").Attribute("full-path")!.Value; + var opf = XDocument.Load(zip.GetEntry(opfPath)!.Open()); + var root = Path.GetDirectoryName(opfPath)!.Replace('\\', '/'); + string Resolve(string href) => (root.Length > 0 ? root + "/" : "") + Uri.UnescapeDataString(href); + + XNamespace opfNs = "http://www.idpf.org/2007/opf"; + XNamespace dc = "http://purl.org/dc/elements/1.1/"; + Assert.Equal("3.0", opf.Root!.Attribute("version")!.Value); + var items = opf.Descendants(opfNs + "item").ToList(); + foreach (var item in items) + Assert.NotNull(zip.GetEntry(Resolve(item.Attribute("href")!.Value))); // every manifest item exists + Assert.Equal(items.Count, items.Select(i => i.Attribute("id")!.Value).Distinct().Count()); + var ids = items.ToDictionary(i => i.Attribute("id")!.Value); + var spine = opf.Descendants(opfNs + "itemref").Select(r => ids[r.Attribute("idref")!.Value]).ToList(); + Assert.NotEmpty(spine); + Assert.NotNull(opf.Descendants(dc + "identifier").SingleOrDefault()); + Assert.NotNull(opf.Descendants(opfNs + "meta").SingleOrDefault(m => m.Attribute("property")?.Value == "dcterms:modified")); + + // Navigation: the nav document (with a non-empty toc) and the NCX, both well-formed. + var nav = items.Single(i => (i.Attribute("properties")?.Value ?? "").Split(' ').Contains("nav")); + var navXml = XDocument.Load(zip.GetEntry(Resolve(nav.Attribute("href")!.Value))!.Open()); + Assert.Contains(navXml.Descendants(), e => e.Name.LocalName == "li"); + var ncx = items.Single(i => i.Attribute("media-type")!.Value == "application/x-dtbncx+xml"); + var ncxXml = XDocument.Load(zip.GetEntry(Resolve(ncx.Attribute("href")!.Value))!.Open()); + Assert.Contains(ncxXml.Descendants(), e => e.Name.LocalName == "navPoint"); + + // Every XHTML file is well-formed and its links and pictures point at files in the package. + var allText = new StringBuilder(); + foreach (var item in items.Where(i => i.Attribute("media-type")!.Value == "application/xhtml+xml")) + { + var path = Resolve(item.Attribute("href")!.Value); + var xhtml = XDocument.Load(zip.GetEntry(path)!.Open()); + allText.Append(xhtml.ToString()); + var folder = Path.GetDirectoryName(path)!.Replace('\\', '/'); + foreach (var reference in xhtml.Descendants().SelectMany(e => e.Attributes().Where(a => a.Name.LocalName is "src" or "href"))) + { + var value = reference.Value.Split('#')[0]; + if (value.Length == 0 || value.Contains(':', StringComparison.Ordinal)) + continue; + var target = Combine(folder, Uri.UnescapeDataString(value)); + Assert.True(zip.GetEntry(target) is not null, $"{path} refers to {value}, which is not in the package."); + } + } + + return new Package( + opf.Descendants(dc + "title").First().Value, + opf.Descendants(dc + "language").First().Value, + [.. opf.Descendants(dc + "creator").Select(c => c.Value)], + spine.Count(s => !s.Attribute("href")!.Value.StartsWith("cover", StringComparison.Ordinal)), + items.Count(i => i.Attribute("media-type")!.Value.StartsWith("image/", StringComparison.Ordinal)), + items.Any(i => (i.Attribute("properties")?.Value ?? "").Contains("cover-image", StringComparison.Ordinal)), + opf.Descendants(opfNs + "meta").Any(m => m.Attribute("property")?.Value == "rendition:layout" && m.Value == "pre-paginated"), + navXml.ToString(), + allText.ToString()); + } +} diff --git a/tests/Filee.Engines.Tests/HtmlReaderTests.cs b/tests/Filee.Engines.Tests/HtmlReaderTests.cs new file mode 100644 index 0000000..a570698 --- /dev/null +++ b/tests/Filee.Engines.Tests/HtmlReaderTests.cs @@ -0,0 +1,295 @@ +// The built-in HTML reader (AngleSharp) and XHTML writer: structure, formatting, lists, tables, pictures, links, +// encodings and XHTML quirks, and HTML → HWPX / PDF without Pandoc. + +using System.Text; +using System.Xml.Linq; +using Filee.Core.Conversion; +using Filee.Core.Presets; +using Filee.Engines.Hwp.Hwpx; +using ImageMagick; +using static Filee.Engines.Tests.HwpxAssert; + +namespace Filee.Engines.Tests; + +public class HtmlReaderTests(EngineFixture fx) : IClassFixture +{ + /// Only engines that ship with the app (or run in-process). + private static readonly string[] OptionalEngines = ["pandoc", "libreoffice", "calibre"]; + + private HtmlContent Parse(string html, string? folder = null) + { + folder ??= fx.NewFolder(); + return HtmlReader.Parse(html, new HtmlReadOptions { BaseFolder = folder, MediaFolder = Path.Combine(folder, "media") }); + } + + private static List Paragraphs(HtmlContent content) => content.Blocks.OfType().ToList(); + + private static string Text(HParagraph paragraph) => HDocumentWalker.Text(paragraph.Inlines); + + private static HText Run(HtmlContent content, string text) => + content.Blocks.SelectMany(b => HDocumentWalker.Inlines([b])).OfType().First(t => t.Text.Contains(text, StringComparison.Ordinal)); + + [Fact] + public void Headings_paragraphs_and_inline_formatting() + { + var content = Parse(""" + 제목 + + +

큰 제목

+

작은 제목

+

일반 굵게 기울임 밑줄 취소 H2O x2 code 표시

+

빨강 클래스

+

가운데

+

오른쪽

+
숨김
+ + """); + + Assert.Equal("제목", content.Title); + var paragraphs = Paragraphs(content); + Assert.Equal((1, "큰 제목"), (paragraphs[0].HeadingLevel, Text(paragraphs[0]))); + Assert.Equal(3, paragraphs[1].HeadingLevel); + Assert.True(Run(content, "굵게").Format.Bold); + Assert.True(Run(content, "기울임").Format.Italic); + Assert.True(Run(content, "밑줄").Format.Underline); + Assert.True(Run(content, "취소").Format.Strike); + Assert.True(Run(content, "2").Format.Subscript); + Assert.Equal(HtmlReader.CodeShade, Run(content, "code").Format.Shade); + Assert.Equal(HtmlReader.MarkShade, Run(content, "표시").Format.Shade); + var red = Run(content, "빨강").Format; + Assert.Equal(("#FF0000", 2000, true), (red.Color, red.Size, red.Italic)); + Assert.True(Run(content, "클래스").Format.Bold); // style sheet class rule + Assert.Null(Run(content, "일반").Format.Color); // body colour from the style sheet is not copied onto runs + Assert.Equal(HAlign.Center, paragraphs.Single(p => Text(p) == "가운데").Format.Align); + Assert.Equal(HAlign.Right, paragraphs.Single(p => Text(p) == "오른쪽").Format.Align); + var all = string.Join("\n", paragraphs.Select(Text)); + Assert.DoesNotContain("메뉴", all); + Assert.DoesNotContain("숨김", all); + Assert.DoesNotContain("document.write", all); + } + + [Fact] + public void Whitespace_collapses_and_line_breaks_stay() + { + var paragraphs = Paragraphs(Parse("

하나 \n 둘
셋


\n
  들여쓴\n\n코드
")); + + Assert.Equal(["하나 둘", "셋"], paragraphs[0].Inlines.OfType().Select(t => t.Text)); + Assert.Single(paragraphs[0].Inlines.OfType()); // the trailing
is not shown + Assert.Empty(paragraphs[1].Inlines); //


: an intended empty line + var code = paragraphs.Skip(2).ToList(); + Assert.Equal([" 들여쓴", "", "코드"], code.Select(c => string.Concat(c.Inlines.OfType().Select(t => t.Text)))); + Assert.All(code.SelectMany(c => c.Inlines.OfType()), t => Assert.Equal(HtmlReader.CodeShade, t.Format.Shade)); + } + + [Fact] + public void Nested_lists_keep_levels_start_numbers_and_types() + { + var paragraphs = Paragraphs(Parse(""" +
    +
  1. 셋째
    • 안쪽
    • 안쪽 둘
    이어지는 글
  2. +
  3. 넷째

    둘째 문단

  4. +
+
  • 표시 없음
+ """)); + + var third = paragraphs.Single(p => Text(p) == "셋째"); + Assert.Equal((0, true), (third.List!.Level, third.List.Numbered)); + Assert.Equal(("LATIN_SMALL", 3, false), (third.List.Numbering.Levels[0].Format, third.List.Numbering.Levels[0].Start, third.List.Numbering.Levels[0].Bullet)); + var inner = paragraphs.Single(p => Text(p) == "안쪽"); + Assert.Equal((1, true), (inner.List!.Level, inner.List.Numbering.Levels[1].Bullet)); + Assert.Same(inner.List.Numbering, paragraphs.Single(p => Text(p) == "안쪽 둘").List!.Numbering); + var continued = paragraphs.Single(p => Text(p) == "이어지는 글"); + Assert.Equal((0, false), (continued.List!.Level, continued.List.Numbered)); + Assert.Same(third.List.Numbering, continued.List.Numbering); + Assert.True(paragraphs.Single(p => Text(p) == "넷째").List!.Numbered); + Assert.False(paragraphs.Single(p => Text(p) == "둘째 문단").List!.Numbered); + var unmarked = paragraphs.Single(p => Text(p) == "표시 없음"); + Assert.Null(unmarked.List); + Assert.True(unmarked.Format.Left > 0); + } + + [Fact] + public void Tables_keep_spans_header_rows_and_borders() + { + var content = Parse(""" + + + + + + + +
표 제목
이름값
병합12
34
+
ab
+ """); + + var tables = content.Blocks.OfType().ToList(); + var table = tables[0]; + Assert.Equal(3, table.ColumnCount); + Assert.Equal([true, false, false], table.Rows.Select(r => r.Header)); + Assert.True(((HText)((HParagraph)table.Rows[0].Cells[0].Blocks[0]).Inlines[0]).Format.Bold); + Assert.Equal(2, table.Rows[0].Cells[1].ColSpan); + Assert.Equal((2, "#EEEEEE"), (table.Rows[1].Cells[0].RowSpan, table.Rows[1].Cells[0].Fill)); + Assert.Equal(HBorders.Empty, table.Borders); + Assert.Equal("표 제목", Text((HParagraph)table.Caption[0])); + Assert.Null(tables[1].Borders); // default: thin grid + Assert.Equal([0.25, 0.75], tables[1].RelativeWidths!); + } + + [Fact] + public void Pictures_from_files_and_data_uris_links_and_anchors() + { + var folder = fx.NewFolder(); + Directory.CreateDirectory(Path.Combine(folder, "img")); + File.WriteAllBytes(Path.Combine(folder, "img", "로고 1.png"), EbookBuilders.Png(40, 20, MagickColors.Teal)); + var data = Convert.ToBase64String(EbookBuilders.Png(10, 10, MagickColors.Red)); + var content = Parse($""" +

둘째로 웹 아무도

+

+

둘째 부분

+ """, folder); + var document = new HDocument(); + document.Sections.Add(new HSection()); + document.Sections[0].Blocks.AddRange(content.Blocks); + HtmlReader.PruneBookmarks(document); + + var inlines = document.Sections[0].Blocks.SelectMany(b => HDocumentWalker.Inlines([b])).ToList(); + var links = inlines.OfType().ToList(); + Assert.Equal(["#part2", "https://example.com/x?a=1"], links.Select(l => l.Target)); + Assert.Equal(["part2"], inlines.OfType().Select(b => b.Name)); // "unused" is pruned + var images = inlines.OfType().ToList(); + Assert.Equal(2, images.Count); // the remote picture is skipped + Assert.Equal(Path.Combine(folder, "img", "로고 1.png"), images[0].Path); + Assert.Equal((int)(200 * HwpxUnits.PerPixel), images[0].Width); + Assert.True(File.Exists(images[1].Path)); + Assert.StartsWith(Path.Combine(folder, "media"), images[1].Path); + } + + [Fact] + public void Css_page_breaks_start_new_pages() + { + var paragraphs = Paragraphs(Parse(""" +

첫

+

둘

+

셋

+

넷

+ +

다섯

+ """)); + + Assert.Equal([false, false, false, true, true], paragraphs.Select(p => p.PageBreakBefore)); // none before the first paragraph + } + + [Fact] + public void Encodings_come_from_the_bom_the_xml_declaration_or_meta_charset() + { + Encoding.RegisterProvider(CodePagesEncodingProvider.Instance); + const string Korean = "한글 문서"; + const string Russian = "Русский текст"; + + Assert.Contains(Korean, HtmlReader.Decode(Encoding.GetEncoding(949).GetBytes($"{Korean}"))); + Assert.Contains(Korean, HtmlReader.Decode(Encoding.GetEncoding(949).GetBytes($"

{Korean}

"))); + Assert.Contains(Russian, HtmlReader.Decode(Encoding.GetEncoding(1251).GetBytes($"{Russian}"))); + Assert.Contains(Korean, HtmlReader.Decode([.. Encoding.Unicode.GetPreamble(), .. Encoding.Unicode.GetBytes($"

{Korean}

")])); + Assert.Contains(Korean, HtmlReader.Decode(Encoding.UTF8.GetBytes($"

{Korean}

"))); // undeclared UTF-8 + } + + [Fact] + public void Xhtml_self_closed_elements_do_not_swallow_the_page() + { + var content = HtmlReader.Parse(""" + + </head> + <body><div id="a"/><p>첫 문단</p><a id="b"/><p>둘째 문단</p><br/><hr/></body></html> + """, new HtmlReadOptions { BaseFolder = fx.NewFolder(), MediaFolder = fx.NewFolder() }, xhtml: true); + + var paragraphs = Paragraphs(content); + Assert.Equal("첫 문단", Text(paragraphs[0])); + Assert.Equal("둘째 문단", Text(paragraphs[1])); + Assert.DoesNotContain(paragraphs[1].Inlines, i => i is HLink); + Assert.IsType<HShape>(paragraphs[^1].Inlines.Single()); // <hr/> is a short line + } + + [Fact] + public void Absurdly_deep_html_is_read_as_text_without_exhausting_the_stack() + { + var html = string.Concat(Enumerable.Repeat("<div><span>", 5000)) + "깊은 글" + string.Concat(Enumerable.Repeat("</span></div>", 5000)); + + Assert.Contains(Paragraphs(Parse(html)), p => Text(p) == "깊은 글"); + } + + [Fact] + public void Xhtml_writer_output_is_well_formed_and_links_across_chapters() + { + var content = Parse(""" + <h1>하나</h1><p>본문 <a href="#target">링크</a> <b>굵게</b> & <기호></p> + <ol start="2"><li>둘<ul><li>안</li></ul></li><li>셋</li></ol> + <table><tr><th>A</th><th>B</th></tr><tr><td colspan="2">합침</td></tr></table> + <h1>둘</h1><p id="target">목표</p><pre>code line</pre> + """); + var document = new HDocument(); + document.Sections.Add(new HSection()); + document.Sections[0].Blocks.AddRange(content.Blocks); + ((HParagraph)document.Sections[0].Blocks[1]).Inlines.Add(new HNote(false) { Blocks = { new HParagraph { Inlines = { new HText("각주", default) } } } }); + + var chapters = XhtmlWriter.Write(document, new XhtmlOptions { SplitChapters = true, ImageSource = _ => null, Epub = true }); + + Assert.Equal(2, chapters.Count); + Assert.Equal(["하나", "둘"], chapters.Select(c => c.Title)); + foreach (var chapter in chapters) + XDocument.Parse(XhtmlWriter.Page(chapter.Title!, chapter.Body, "ko", "style.css", epub: true)); // throws if malformed + Assert.Contains("href=\"ch002.xhtml#target\"", chapters[0].Body); + Assert.Contains("id=\"target\"", chapters[1].Body); + Assert.Contains("<ol start=\"2\">", chapters[0].Body); + Assert.Matches("<li>둘<ul>\\s*<li>안</li>\\s*</ul>\\s*</li>", chapters[0].Body); + Assert.Contains("colspan=\"2\"", chapters[0].Body); + Assert.Contains("epub:type=\"footnote\"", chapters[0].Body); + Assert.Contains("<pre><code>code line</code></pre>", chapters[1].Body); + } + + [Fact] + public void Html_to_pdf_needs_no_optional_engine() + { + var planner = new ConverterCatalog(fx.Converters.Where(c => !OptionalEngines.Contains(c.Id))) { Priority = fx.Catalog.Priority }.CreatePlanner(["rhwp"]); + + Assert.Equal("hwpx-writer", Assert.Single(planner.Plan("html", "hwpx")!.Steps).Converter.Id); + Assert.Equal(["hwpx-writer", "rhwp"], planner.Plan("html", "pdf")!.Steps.Select(s => s.Converter.Id)); + Assert.Equal("ebook", Assert.Single(planner.Plan("html", "txt")!.Steps).Converter.Id); + Assert.Equal("ebook", Assert.Single(planner.Plan("html", "epub")!.Steps).Converter.Id); + } + + [Fact] + public async Task Html_becomes_hwpx_and_pdf_with_built_in_engines() + { + var dir = fx.NewFolder(); + File.WriteAllBytes(Path.Combine(dir, "chart.png"), EbookBuilders.Png(300, 150, MagickColors.MediumPurple)); + var html = Path.Combine(dir, "보고서.html"); + await File.WriteAllTextAsync(html, """ + <!DOCTYPE html><html><head><meta charset="utf-8"><title>보고서 +

분기 보고서

+

매출이 12% 늘었습니다.

+
  • 서울
  • 부산
+
지역매출
서울100
+

차트

+

둘째 쪽

+ + """, TestContext.Current.CancellationToken); + + var job = await fx.ConvertAsync([html], new Preset { TargetFormat = "hwpx" }); + Assert.True(job.State == JobState.Completed, string.Join("; ", job.Files.Select(f => $"{f.ErrorKey} {f.ErrorDetail}"))); + var hwpx = job.Outputs.Single(); + ValidPackage(hwpx, expectImages: 1); + using (var document = Unhwp.UnhwpDocument.ParseFile(hwpx)) + { + var text = document.ToText(); + foreach (var expected in new[] { "분기 보고서", "12%", "서울", "부산", "매출", "둘째 쪽" }) + Assert.Contains(expected, text); + } + Assert.Single(Xml(hwpx).Descendants(Hp + "tbl")); + + var pages = RenderWithRhwp(fx, hwpx, "hwpx-from-html"); + if (pages is not null) + Assert.Equal(2, pages); + } +} diff --git a/tests/Filee.Engines.Tests/HwpxWriterTests.cs b/tests/Filee.Engines.Tests/HwpxWriterTests.cs index 28a10fe..5d80d46 100644 --- a/tests/Filee.Engines.Tests/HwpxWriterTests.cs +++ b/tests/Filee.Engines.Tests/HwpxWriterTests.cs @@ -1,4 +1,4 @@ -// HWPX from Markdown (Markdig, built in), HTML (Pandoc AST → OWPML) and plain text, plus the writer's small helpers. +// HWPX from Markdown (Markdig), HTML (AngleSharp) and plain text, all built in, plus the writer's small helpers. // Output is checked structurally, read back with Unhwp and rendered with rhwp. DOCX has its own tests // (DocxToHwpxTests) because it is read without Pandoc. @@ -7,7 +7,6 @@ using Filee.Core.Presets; using Filee.Core.Settings; using Filee.Engines.Hwp.Hwpx; -using Filee.Engines.Office; using ImageMagick; using Unhwp; using static Filee.Engines.Tests.HwpxAssert; @@ -50,10 +49,6 @@ public class HwpxWriterTests(EngineFixture fx) : IClassFixture [^1]: 각주 내용입니다. """; - private void SkipUnlessPandoc() => - Assert.SkipUnless(PandocConverter.Locate() is not null, - "Pandoc not found (pwsh build/fetch-engines.ps1 -Only pandoc)"); - private async Task ConvertAsync(string input, string target = "hwpx") { var job = await fx.ConvertAsync([input], new Preset { TargetFormat = target }); @@ -89,10 +84,9 @@ public async Task Markdown_becomes_a_valid_hwpx_that_other_readers_understand() [Fact] public async Task Html_tables_with_merged_cells_keep_a_consistent_grid() { - SkipUnlessPandoc(); var dir = fx.NewFolder(); var html = Path.Combine(dir, "table.html"); - // Spans stay inside the body: Pandoc truncates a rowspan that would cross from a header row into the body. + // Row and column spans inside the body: every grid position must be covered exactly once. await File.WriteAllTextAsync(html, """
세로 병합가로 병합