diff --git a/Directory.Packages.props b/Directory.Packages.props
index 0f93475..d2833b9 100644
--- a/Directory.Packages.props
+++ b/Directory.Packages.props
@@ -42,6 +42,7 @@
+
diff --git a/THIRD-PARTY-NOTICES.md b/THIRD-PARTY-NOTICES.md
index 81eb342..abadc4d 100644
--- a/THIRD-PARTY-NOTICES.md
+++ b/THIRD-PARTY-NOTICES.md
@@ -19,7 +19,7 @@ Each component remains under its own license.
| SkiaSharp | MIT | https://github.com/mono/SkiaSharp |
| Svg.Skia (incl. Svg, ShimSkiaSharp, ExCSS; text shaping in `Filee.Engines/Vector/OutlinedTextPlayback.cs` adapted from it) | MIT | https://github.com/wieslawsoltes/Svg.Skia |
| HarfBuzzSharp | MIT | https://github.com/mono/SkiaSharp |
-| SharpCompress (zstd decompression for engine downloads) | MIT | https://github.com/adamhathcock/sharpcompress |
+| SharpCompress (zstd decompression for engine downloads; RAR, 7z and TAR comics) | MIT | https://github.com/adamhathcock/sharpcompress |
| Unhwp | MIT | https://github.com/iyulab/unhwp |
| Markdig | BSD-2-Clause | https://github.com/xoofx/markdig |
| ExcelNumberFormat | MIT | https://github.com/andersnm/ExcelNumberFormat |
@@ -27,6 +27,7 @@ Each component remains under its own license.
| PdfPig (PDF text, layout and pictures) | Apache-2.0 | https://github.com/UglyToad/PdfPig |
| MimeKitLite (e-mail / MIME parsing) | MIT | https://github.com/jstedfast/MimeKit |
| ACadSharp (DWG/DXF reading and writing; includes CSMath and CSUtilities) | MIT, Copyright (c) Albert Domenech | https://github.com/DomCR/ACadSharp |
+| AngleSharp | MIT | https://github.com/AngleSharp/AngleSharp |
| pypandoc-hwpx (Pandoc AST mapping and package layout in `Filee.Engines/Hwp/Hwpx`, incl. the `blank.hwpx` reference document) | MIT, Copyright (c) 2024 pypandoc-hwpx Contributors | https://github.com/msjang/pypandoc-hwpx |
| Microsoft.Extensions.* | MIT | https://github.com/dotnet/runtime |
| cu2qu (fontTools): the cubic-to-quadratic approach followed by `Filee.Engines/Fonts/Cff/CubicToQuadratic.cs` | Apache-2.0, Copyright 2016 Google Inc. | https://github.com/fonttools/fonttools |
@@ -59,6 +60,7 @@ Downloaded from their official release pages when the user chooses to install th
| Ghostscript (conda-forge build) | AGPL-3.0 | https://www.ghostscript.com, https://github.com/conda-forge/ghostscript-feedstock |
| Microsoft Visual C++ Redistributable (for Ghostscript, conda-forge `vc14_runtime`) | Microsoft Visual C++ Redistributable license | https://github.com/conda-forge/vc-feedstock |
| FFmpeg 9.0.2 (gyan.dev "full_build-shared" Windows build; ffmpeg.exe, ffprobe.exe and their libraries, incl. x264, x265, libvpx, LAME, Opus, Vorbis, Theora, OpenCORE AMR) | GPL-3.0-or-later (the build is configured with `--enable-gpl --enable-version3`); FFmpeg itself LGPL-2.1-or-later | https://ffmpeg.org, build: https://www.gyan.dev/ffmpeg/builds/ (source: https://github.com/GyanD/codexffmpeg) |
+| calibre (ebook-convert) | GPL-3.0 | https://calibre-ebook.com |
These programs run as separate processes. Their source code is available from the linked projects.
diff --git a/build/fetch-engines.ps1 b/build/fetch-engines.ps1
index 387646f..b9f7f4a 100644
--- a/build/fetch-engines.ps1
+++ b/build/fetch-engines.ps1
@@ -16,6 +16,8 @@
ghostscript/ Ghostscript (AGPL-3.0, separate program) from conda-forge, EPS/PS <-> PDF, with the Microsoft
C++ runtime DLLs (vcruntime/) copied next to gswin64c.exe
ffmpeg/ FFmpeg (GPL-3.0 build, separate program), video and audio; bin/ffmpeg.exe + bin/ffprobe.exe
+ calibre/ calibre (GPL-3.0, separate program), extracted from the official MSI via an administrative
+ install: ebook-convert for LIT, LRF, PDB, ... and MOBI / AZW3 output
Every download is pinned to a version and verified with SHA-256 (src/Filee.Engines/Infrastructure/engines.json,
shared with the app). Downloads are cached in build/.cache.
@@ -216,4 +218,20 @@ if ('ghostscript' -in $selected) {
Copy-Item (Join-Path $Destination 'vcruntime\*.dll') $target -Force
}
+if ('calibre' -in $selected) {
+ $msi = Get-Engine 'calibre'
+ $tmp = Join-Path $cache 'calibre-admin'
+ Reset-Folder $tmp
+ Write-Host 'unpack calibre (administrative install, no system changes)'
+ $proc = Start-Process msiexec.exe -ArgumentList @('/a', "`"$msi`"", '/qn', "TARGETDIR=`"$tmp`"") -Wait -PassThru
+ if ($proc.ExitCode -ne 0) { throw "msiexec /a failed with exit code $($proc.ExitCode)." }
+
+ $convert = Get-ChildItem $tmp -Recurse -Filter 'ebook-convert.exe' | Select-Object -First 1
+ if (-not $convert) { throw 'ebook-convert.exe not found after extracting the MSI.' }
+ $target = Join-Path $Destination 'calibre'
+ Reset-Folder $target
+ Get-ChildItem $convert.DirectoryName -Force | Move-Item -Destination $target
+ Remove-Item $tmp -Recurse -Force
+}
+
Write-Host "Engines ready in $Destination"
diff --git a/docs/ENGINES.md b/docs/ENGINES.md
index 9c62c11..90f4c88 100644
--- a/docs/ENGINES.md
+++ b/docs/ENGINES.md
@@ -15,10 +15,11 @@ Filee chooses engines automatically. Settings → *Engines* shows their status,
| Spreadsheets (ExcelDataReader for XLS) | built in (library) | **XLSX, XLS, ODS, CSV, TSV → XLSX, ODS, CSV, TSV** (one CSV / TSV per sheet) |
| Office Open XML | built in | **DOCM / DOTX / DOTM ↔ DOCX, XLSM / XLTX ↔ XLSX, PPTM / POTX / PPSX ↔ PPTX** (macros removed for macro-free types) |
| Fonts | built in | **TTF, OTF, WOFF, WOFF2, EOT ↔ each other**; CFF (PostScript) outlines become TrueType for TTF and EOT |
-| **HWPX writer** | built in | **DOCX (+ DOCM/DOTX/DOTM), XLSX (+ XLSM/XLTX), XLS, ODS, CSV, TSV, PPTX (+ PPTM/POTX/PPSX), PDF, TXT, Markdown → HWPX** (and with rhwp → PDF and images); HTML, ODT, RTF, reStructuredText, LaTeX → HWPX with Pandoc |
+| **HWPX writer** | built in | **DOCX (+ DOCM/DOTX/DOTM), XLSX (+ XLSM/XLTX), XLS, ODS, CSV, TSV, PPTX (+ PPTM/POTX/PPSX), PDF, HTML, EPUB, MOBI/AZW3, FB2, HWPX, TXT, Markdown → HWPX** (and with rhwp → PDF and images); ODT, RTF, reStructuredText, LaTeX → HWPX with Pandoc |
| **DOCX writer** | built in | **Markdown, TXT, XLSX, CSV, PPTX (+ variants), PDF → DOCX** |
| PDF text | built in (PdfPig) | **PDF → TXT** (reading order, also two columns; no OCR) |
| E-mail | built in (MimeKit) | **EML → HTML** (headers + body, inline pictures), **TXT**, **ZIP** (the attachments) |
+| **E-books** | built in | **EPUB, MOBI/AZW/AZW3/PRC, FB2, HTMLZ, TXTZ → EPUB, HWPX (→ PDF), TXT, HTML, Markdown, FB2, HTMLZ, TXTZ**; HTML, Markdown, TXT, DOCX → EPUB/FB2/HTMLZ/TXTZ; HTML → TXT; comics CBZ/CBR/CB7/CBT/CBC → PDF, EPUB, CBZ; PDF → CBZ; AZW4 → PDF |
| CAD (ACadSharp + built-in renderer) | built in (library) | **DWG ↔ DXF**; DWG/DXF → **PDF** (vector), **SVG**, PNG, JPG, WEBP, TIFF, BMP, GIF, ICO, AVIF |
| rhwp | bundled with the installer (`engines/rhwp`) | HWP/HWPX → PDF, **HWP → HWPX, HWPX → HWP** |
| Archives (7-Zip) | bundled with the installer (`engines/7zip`, ~2.5 MB) + built-in readers | ZIP, 7Z, RAR, TAR (+ GZ/BZ2/XZ/Z/7Z/LZ), CAB, ISO, DMG, … **→ folder, ZIP, 7Z, TAR, TAR.GZ/BZ2/XZ**; ALZ, EGG, lzip read in-process; "Compress into one archive" for any files |
@@ -26,6 +27,7 @@ Filee chooses engines automatically. Settings → *Engines* shows their status,
| Pandoc | **optional download** (~42 MB, 240 MB on disk) | Markdown ↔ DOCX/ODT/RTF, DOCX/ODT/HTML/RTF → Markdown, HTML ↔ DOCX/ODT, reStructuredText and LaTeX ↔ Markdown/HTML/DOCX/ODT (and → RTF, EPUB, TXT) |
| Ghostscript | **optional download** (~20 MB, 31 MB on disk) | EPS/PS → PDF, PDF → EPS/PS, PostScript-based AI → PDF (images through PDF and PDFium) |
| FFmpeg | **optional download** (~100 MB, 272 MB on disk) | **all video and audio**: video ↔ video, video → animated GIF or a still frame, audio extraction, audio ↔ audio, GIF → MP4/WEBM/MOV |
+| Calibre | **optional download** (~216 MB, 660 MB on disk) | rare e-book formats: LIT, LRF, CHM, PDB, PML, RB, SNB, TCR, OEB → EPUB; EPUB → MOBI, AZW3, LIT, LRF, PDB, PML, RB, SNB, TCR |
DOCX, XLSX, XLS, ODS and PPTX → PDF need neither Microsoft Office nor LibreOffice: they are read in-process, written as HWPX
and rendered by rhwp. Anything the built-in readers understand (including PDF) is written as DOCX by the DOCX writer.
@@ -51,8 +53,9 @@ they can be installed or removed later in Settings → *Engines*. A conversion t
engine folder only when complete. Redirects to mirrors (even plain HTTP) are followed, because the hash decides.
- Engines go to `%LOCALAPPDATA%\Filee\engines`: outside the app folder, so updates keep them, and inside Filee's
install root, so uninstalling removes them.
-- LibreOffice comes as an MSI and is unpacked with an administrative install (`msiexec /a`): files only, no
- registry entries, no admin rights. Help, gallery and most dictionaries are removed afterwards (~500 MB).
+- LibreOffice and calibre come as MSIs and are unpacked with an administrative install (`msiexec /a`): files only,
+ no registry entries, no admin rights. LibreOffice's help, gallery and most dictionaries are removed afterwards
+ (~500 MB).
- Ghostscript comes from conda-forge (Artifex publishes only an NSIS installer): a `.conda` package is a zip with a
zstd tarball, of which only `Library/bin` is unpacked (SharpCompress; `fetch-engines.ps1` uses Windows' `tar.exe`).
Its fonts and resources are compiled into `gsdll64.dll`. The Microsoft C++ runtime it was built against
@@ -117,9 +120,16 @@ Readers turn the source into a small document model (`Hwp/Hwpx/HwpxModel.cs`) an
start numbers), task lists, quotes, tables with spans, links, images, footnotes and math. No Pandoc needed.
- **PDF** is read with PdfPig (see *PDF reader* below); DOCM / DOTX / DOTM, XLSM / XLTX and PPTM / POTX / PPSX
are read like DOCX, XLSX and PPTX.
-- **HTML, ODT, RTF, reStructuredText, LaTeX** are parsed by Pandoc into its JSON AST (`PandocAstReader.cs`, following
+- **HTML** is parsed with AngleSharp (`HtmlReader.cs`, `HtmlCss.cs`): headings, paragraphs, bold / italic /
+ underline / strike / sub / sup / code / mark, the basic inline CSS (colour, background, font weight, style and
+ size, text-align, text-indent, page breaks) and simple `tag` / `.class` rules of style sheets, nested lists with
+ start numbers and types, tables with spans / header rows / borders, pictures (local files and data: URIs; remote
+ pictures are skipped), links and anchors, quotes, `pre`, `hr` and figures. Scripts, styles, forms and `nav` are
+ skipped. The encoding comes from the BOM, the XML declaration or ``, else UTF-8 or the system
+ code page. No Pandoc needed.
+- **ODT, RTF** are parsed by Pandoc into its JSON AST (`PandocAstReader.cs`, following
[pypandoc-hwpx](https://github.com/msjang/pypandoc-hwpx)).
-- Markdown and Pandoc keep structure only, so page setup comes from the built-in template (A4).
+- Markdown, HTML and Pandoc keep structure only, so page setup comes from the built-in template (A4).
Element order and attribute values follow files saved by 한글 where the schema and 한글 disagree, for example:
@@ -343,6 +353,45 @@ OLE objects, charts, video, form controls, text art (its text is kept), arcs / p
master pages, memos, character ratio, relative size and offset, paragraph borders, picture cropping and rotation.
Password-protected (encrypted) HWPX and DRM-wrapped files stop with a clear error.
+## E-books
+
+The built-in e-book engine (`src/Filee.Engines/Ebooks`) reads books into the same document model as the HWPX
+writer, so every e-book also reaches HWPX, PDF and images. The document model is written back out by
+`XhtmlWriter` (EPUB chapters, HTMLZ, single-page HTML), `PlainTextWriter`, `MarkdownWriter` and `Fb2Writer`.
+
+- **EPUB 2 / 3**: container.xml → OPF → spine; each chapter is read by the HTML reader and starts a new page, links
+ between chapters become links to bookmarks, pictures come from the package, title / authors / language / cover
+ from the metadata.
+- **EPUB output** is EPUB 3 with an NCX for older readers: `mimetype` first and stored, a navigation document from
+ the headings, chapters split at level-1 headings and page breaks, one style sheet, pictures in formats every
+ reader shows (others become JPEG / PNG), a cover page for covers the content does not show.
+- **MOBI / AZW / AZW3 / PRC** (`Ebooks/Mobi`, written from the MobileRead wiki's format description): PalmDB
+ records, MOBI header and EXTH metadata, PalmDOC (LZ77) and HUFF/CDIC compression, MOBI 6 `filepos` links and
+ `recindex` pictures, KF8 text rebuilt from the skeleton and fragment indexes with `kindle:pos` / `kindle:embed` /
+ `kindle:flow` references resolved, and plain PalmDOC (TEXtREAd) books. **AZW4** (Print Replica) gives back its PDF.
+- **FB2**: nested sections become headings, poems / epigraphs / citations / tables are kept, note links become
+ footnotes, pictures come from the base64 binaries; the XML declaration's encoding (often windows-1251) is used.
+ FB2 output nests sections by heading level.
+- **HTMLZ / TXTZ** (calibre's zipped formats) are read and written, with their `metadata.opf`.
+- **Comics**: CBZ, CBR, CB7, CBT and CBC (a ZIP of CBZ files) are opened with SharpCompress; pages are sorted
+ naturally ("page2" before "page10"). → PDF has one page per picture sized like the picture (JPEGs are embedded
+ as they are), → EPUB is fixed-layout, → CBZ repacks. PDF → CBZ renders the pages with PDFium at the preset DPI.
+- Books with DRM (Adobe, Apple, Kindle) are refused with a clear message; Filee does not remove DRM. KFX and
+ Topaz books are recognised and refused too.
+
+## Calibre
+
+calibre's `ebook-convert` (GPL-3.0) is an optional download for the formats Filee does not read or write itself:
+LIT, LRF, CHM, PDB, PML, RB, SNB, TCR and OEB → EPUB, and EPUB → MOBI, AZW3, LIT, LRF, PDB, PML, RB, SNB and TCR.
+EPUB is the hub, so for example FB2 → AZW3 is FB2 → EPUB (built in) → AZW3 (calibre). Its edges cost 20, so
+built-in routes always win where they exist.
+
+- The pinned `calibre-64bit-.msi` from download.calibre-ebook.com is unpacked like LibreOffice; the folder
+ with `ebook-convert.exe` becomes `engines/calibre`.
+- Each run gets its own configuration, cache and temp folders in the job's work directory
+ (`CALIBRE_CONFIG_DIRECTORY`, ...), so a calibre the user installed is never read or changed; messages are
+ English (`CALIBRE_OVERRIDE_LANG`) and progress comes from its "34% ..." lines.
+
## HWP ↔ HWPX
rhwp converts between the two 한글 formats without Hancom Office (`export-hwpx` and `convert`). Anything → HWP goes
diff --git a/src/Filee.App/Assets/i18n/en.json b/src/Filee.App/Assets/i18n/en.json
index 1bbbc44..c9a773a 100644
--- a/src/Filee.App/Assets/i18n/en.json
+++ b/src/Filee.App/Assets/i18n/en.json
@@ -322,6 +322,8 @@
"engines.package.ghostscript.description": "EPS, PS and older Illustrator files → PDF, PNG and other images, and PDF or SVG → EPS/PS. SVG and AI files saved with PDF compatibility convert without it.",
"engines.package.ffmpeg.name": "FFmpeg (video and audio)",
"engines.package.ffmpeg.description": "Video ↔ video (MP4, MOV, MKV, WEBM, AVI, WMV and more), video → animated GIF or a still image, sound from videos (→ MP3, M4A, WAV …), audio ↔ audio (MP3, M4A, FLAC, WAV, OGG, OPUS and more) and GIF → MP4, WEBM, MOV. Only needed for video and audio.",
+ "engines.package.calibre.name": "Calibre (Kindle and rare e-book formats)",
+ "engines.package.calibre.description": "Saves as MOBI and AZW3 for Kindle, and converts rare e-book formats: LIT, LRF, PDB, PML, RB, SNB and TCR, plus CHM and OEB input. EPUB, MOBI, AZW3, FB2 and comics (CBZ, CBR) are read without it.",
"engines.package.size": "Download {0} · {1} on disk",
"engines.progress_detail": "{0} of {1} · {2}/s · {3}",
"engines.eta.estimating": "estimating time left…",
diff --git a/src/Filee.App/Assets/i18n/ko.json b/src/Filee.App/Assets/i18n/ko.json
index efff326..c0cd6bb 100644
--- a/src/Filee.App/Assets/i18n/ko.json
+++ b/src/Filee.App/Assets/i18n/ko.json
@@ -322,6 +322,8 @@
"engines.package.ghostscript.description": "EPS·PS 파일과 예전 일러스트레이터 파일을 PDF·PNG 등 이미지로, PDF·SVG를 EPS·PS로 변환해요. SVG와 PDF 호환으로 저장한 AI 파일은 없어도 변환돼요.",
"engines.package.ffmpeg.name": "FFmpeg (동영상·오디오)",
"engines.package.ffmpeg.description": "동영상 ↔ 동영상(MP4·MOV·MKV·WEBM·AVI·WMV 등), 동영상 → 움직이는 GIF·정지 이미지, 동영상에서 소리 추출(→ MP3·M4A·WAV 등), 오디오 ↔ 오디오(MP3·M4A·FLAC·WAV·OGG·OPUS 등), GIF → MP4·WEBM·MOV 변환을 추가해요. 동영상과 오디오에만 필요해요.",
+ "engines.package.calibre.name": "Calibre (킨들과 드문 전자책 형식)",
+ "engines.package.calibre.description": "킨들용 MOBI·AZW3로 저장하고, 드문 전자책 형식(LIT·LRF·PDB·PML·RB·SNB·TCR, CHM·OEB 읽기)을 변환해요. EPUB·MOBI·AZW3·FB2와 만화책(CBZ·CBR)은 없어도 읽을 수 있어요.",
"engines.package.size": "다운로드 {0} · 설치 후 {1}",
"engines.progress_detail": "{0} / {1} · {2}/s · {3}",
"engines.eta.estimating": "남은 시간 계산 중…",
diff --git a/src/Filee.App/Assets/i18n/zh-CN.json b/src/Filee.App/Assets/i18n/zh-CN.json
index 19d91d1..974ea24 100644
--- a/src/Filee.App/Assets/i18n/zh-CN.json
+++ b/src/Filee.App/Assets/i18n/zh-CN.json
@@ -322,6 +322,8 @@
"engines.package.ghostscript.description": "EPS、PS 和旧版 Illustrator 文件 → PDF、PNG 等图片,以及 PDF 或 SVG → EPS/PS。SVG 和以 PDF 兼容方式保存的 AI 文件无需它即可转换。",
"engines.package.ffmpeg.name": "FFmpeg(视频和音频)",
"engines.package.ffmpeg.description": "支持视频 ↔ 视频(MP4、MOV、MKV、WEBM、AVI、WMV 等)、视频 → GIF 动图或静态图片、从视频提取声音(→ MP3、M4A、WAV 等)、音频 ↔ 音频(MP3、M4A、FLAC、WAV、OGG、OPUS 等)以及 GIF → MP4、WEBM、MOV。仅视频和音频转换需要它。",
+ "engines.package.calibre.name": "Calibre(Kindle 与少见电子书格式)",
+ "engines.package.calibre.description": "可保存为 Kindle 使用的 MOBI 和 AZW3,并转换少见的电子书格式:LIT、LRF、PDB、PML、RB、SNB、TCR,以及读取 CHM 和 OEB。EPUB、MOBI、AZW3、FB2 和漫画(CBZ、CBR)无需它即可读取。",
"engines.package.size": "下载 {0} · 占用 {1}",
"engines.progress_detail": "{0} / {1} · {2}/s · {3}",
"engines.eta.estimating": "正在估算剩余时间…",
diff --git a/src/Filee.App/ViewModels/EngineSetupViewModel.cs b/src/Filee.App/ViewModels/EngineSetupViewModel.cs
index b30d627..453792b 100644
--- a/src/Filee.App/ViewModels/EngineSetupViewModel.cs
+++ b/src/Filee.App/ViewModels/EngineSetupViewModel.cs
@@ -21,8 +21,8 @@ public EngineSetupViewModel(EngineDownloadService downloads, ILocalizer loc)
foreach (var package in Packages)
{
// Large downloads are offered, not pre-selected: LibreOffice is only needed for older formats (DOC, XLS,
- // PPT, OpenDocument), FFmpeg (~100 MB) only for video and audio. Ghostscript is small but only needed for
- // EPS / PostScript.
+ // PPT, OpenDocument), FFmpeg (~100 MB) only for video and audio, calibre (~230 MB) only for rare e-book
+ // formats and Kindle output. Ghostscript is small but only needed for EPS / PostScript.
package.Selected = !package.IsInstalled
&& Filee.Engines.Infrastructure.EngineDownloads.DownloadSize(package.Package) < PreselectLimit
&& package.Package.Id != "ghostscript";
diff --git a/src/Filee.Engines/Cad/CadConverter.cs b/src/Filee.Engines/Cad/CadConverter.cs
index d0a45ff..c784adb 100644
--- a/src/Filee.Engines/Cad/CadConverter.cs
+++ b/src/Filee.Engines/Cad/CadConverter.cs
@@ -2,6 +2,7 @@
// DXF and DWG; the drawing is rendered with SkiaSharp to PDF (vector), SVG and raster images.
using Filee.Core.Conversion;
+using Filee.Engines.Infrastructure;
using Filee.Engines.Magick;
using ImageMagick;
using Microsoft.Extensions.Logging;
@@ -41,7 +42,7 @@ from target in (string[])["pdf", "svg", .. ImageEncoder.Writable]
select new ConversionEdge(source, target),
];
- public EngineStatus GetStatus() => EngineStatus.Available("ACadSharp: DWG R13–2018+, DXF");
+ public EngineStatus GetStatus() => EngineStatus.Available("ACadSharp: DWG R13–2018+, DXF", $"{EngineVersions.BuiltIn} · {EngineVersions.Library("ACadSharp", typeof(ACadSharp.CadDocument))}");
public Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) =>
Task.Run>(() =>
diff --git a/src/Filee.Engines/Ebooks/Book.cs b/src/Filee.Engines/Ebooks/Book.cs
new file mode 100644
index 0000000..18e1282
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Book.cs
@@ -0,0 +1,160 @@
+// An e-book in memory: the HWPX document model (content) plus the metadata e-book formats carry (title, authors,
+// language, cover). Every e-book reader produces a Book and every e-book writer consumes one.
+
+using System.IO.Compression;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+/// Title, authors, language and cover of a book.
+internal sealed class BookMetadata
+{
+ public string? Title { get; set; }
+ public List Authors { get; } = [];
+
+ /// BCP 47 language tag ("ko", "en-US"), if known.
+ public string? Language { get; set; }
+ public string? Description { get; set; }
+ public string? Publisher { get; set; }
+
+ /// Local picture file of the cover, if the source names one.
+ public string? CoverImage { get; set; }
+}
+
+/// An e-book: content and metadata.
+internal sealed record Book(HDocument Document, BookMetadata Metadata)
+{
+ /// The title to show: metadata, the document's own title, the first heading, else the file name.
+ public string DisplayTitle(string sourcePath)
+ {
+ if (!string.IsNullOrWhiteSpace(Metadata.Title))
+ return Metadata.Title.Trim();
+ if (!string.IsNullOrWhiteSpace(Document.Title))
+ return Document.Title.Trim();
+ var heading = Document.Sections.SelectMany(s => s.Blocks).OfType().FirstOrDefault(p => p.HeadingLevel > 0);
+ if (heading is not null && HDocumentWalker.Text(heading.Inlines) is { Length: > 0 } text)
+ return text.Length > 200 ? text[..200] : text;
+ return Filee.Core.Formats.FormatRegistry.NameWithoutExtension(sourcePath);
+ }
+
+ /// The language from the metadata, else guessed from the text: Korean, Chinese, Japanese or English.
+ public string DisplayLanguage()
+ {
+ if (!string.IsNullOrWhiteSpace(Metadata.Language))
+ return Metadata.Language.Trim();
+ int hangul = 0, han = 0, kana = 0, latin = 0;
+ foreach (var text in Document.Sections.SelectMany(s => HDocumentWalker.Inlines(s.Blocks)).OfType().Take(2000))
+ {
+ foreach (var ch in text.Text)
+ {
+ if (ch is >= '가' and <= '힣')
+ hangul++;
+ else if (ch is >= '' and <= 'ヿ')
+ kana++;
+ else if (ch is >= '一' and <= '鿿')
+ han++;
+ else if (char.IsAsciiLetter(ch))
+ latin++;
+ }
+ }
+ return hangul > 0 && hangul * 3 >= latin ? "ko"
+ : kana > 0 && kana * 3 >= latin ? "ja"
+ : han > 0 && han * 3 >= latin ? "zh"
+ : "en";
+ }
+}
+
+/// Helpers shared by the e-book readers and writers.
+internal static class EbookFiles
+{
+ /// Message for books with DRM: Filee never removes copy protection.
+ public const string DrmMessage =
+ "This book is protected with DRM (copy protection), so it cannot be converted. Filee does not remove DRM; " +
+ "open it in the reading app it was bought for.";
+
+ ///
+ /// Unpacks a ZIP into . Entries that would land outside it ("../", absolute paths)
+ /// are skipped, so a crafted archive cannot write anywhere else.
+ ///
+ public static void Extract(ZipArchive zip, string folder)
+ {
+ var root = Path.GetFullPath(folder) + Path.DirectorySeparatorChar;
+ Directory.CreateDirectory(root);
+ foreach (var entry in zip.Entries)
+ {
+ if (entry.FullName.EndsWith('/') || entry.FullName.EndsWith('\\'))
+ continue;
+ if (SafePath(root, entry.FullName) is not { } target)
+ continue;
+ Directory.CreateDirectory(Path.GetDirectoryName(target)!);
+ entry.ExtractToFile(target, overwrite: true);
+ }
+ }
+
+ /// A path inside for an archive entry name, or null if it would escape.
+ public static string? SafePath(string root, string entryName)
+ {
+ var relative = entryName.Replace('\\', '/').TrimStart('/');
+ if (relative.Length == 0 || relative.Contains(':'))
+ return null;
+ try
+ {
+ var full = Path.GetFullPath(Path.Combine(root, relative));
+ var prefix = Path.GetFullPath(root).TrimEnd(Path.DirectorySeparatorChar) + Path.DirectorySeparatorChar;
+ return full.StartsWith(prefix, StringComparison.OrdinalIgnoreCase) ? full : null;
+ }
+ catch (Exception ex) when (ex is ArgumentException or NotSupportedException or PathTooLongException)
+ {
+ return null;
+ }
+ }
+
+ /// A fresh folder inside the job's work directory.
+ public static string NewFolder(string workDirectory, string prefix)
+ {
+ var folder = Path.Combine(workDirectory, $"{prefix}-{Guid.NewGuid().ToString("N")[..8]}");
+ Directory.CreateDirectory(folder);
+ return folder;
+ }
+
+ ///
+ /// A picture in a format every reading system shows: JPEG and PNG (plus GIF and SVG when
+ /// ) are kept, others are converted into — to JPEG for
+ /// opaque photos (WEBP, AVIF, ... would grow a lot as PNG), else PNG. Null when the file cannot be read.
+ ///
+ public static (string File, string MediaType)? CommonImage(string path, string folder, bool gifAndSvg)
+ {
+ var mediaType = ImageMediaType(path);
+ if (mediaType is "image/jpeg" or "image/png" || gifAndSvg && mediaType is "image/gif" or "image/svg+xml")
+ return (path, mediaType);
+ try
+ {
+ using var image = new ImageMagick.MagickImage(path);
+ var jpeg = !image.HasAlpha && mediaType is "image/webp" or "image/avif" or "image/jxl" or "image/tiff";
+ Directory.CreateDirectory(folder);
+ var file = Path.Combine(folder, $"{Guid.NewGuid():N}.{(jpeg ? "jpg" : "png")}");
+ image.Quality = 90;
+ image.Write(file, jpeg ? ImageMagick.MagickFormat.Jpeg : ImageMagick.MagickFormat.Png);
+ return (file, jpeg ? "image/jpeg" : "image/png");
+ }
+ catch (ImageMagick.MagickException)
+ {
+ return null;
+ }
+ }
+
+ /// The media type of a picture file by extension, or null for files that are not pictures.
+ public static string? ImageMediaType(string path) => Path.GetExtension(path).ToLowerInvariant() switch
+ {
+ ".jpg" or ".jpeg" or ".jpe" or ".jfif" => "image/jpeg",
+ ".png" => "image/png",
+ ".gif" => "image/gif",
+ ".svg" => "image/svg+xml",
+ ".webp" => "image/webp",
+ ".bmp" => "image/bmp",
+ ".tif" or ".tiff" => "image/tiff",
+ ".avif" => "image/avif",
+ ".jxl" => "image/jxl",
+ _ => null,
+ };
+}
diff --git a/src/Filee.Engines/Ebooks/CalibreConverter.cs b/src/Filee.Engines/Ebooks/CalibreConverter.cs
new file mode 100644
index 0000000..5ea69a9
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/CalibreConverter.cs
@@ -0,0 +1,107 @@
+// E-book formats Filee does not read or write itself, with calibre's ebook-convert (GPL-3.0, a separate program the
+// user downloads in Settings → Engines): LIT, LRF, CHM, PDB, PML, RB, SNB, TCR and OEB → EPUB, and EPUB → MOBI,
+// AZW3, LIT, LRF, PDB, PML, RB, SNB and TCR. EPUB is the hub: the built-in engine handles every other step, and
+// calibre's edges cost more so built-in routes always win where they exist.
+//
+// Each run gets its own calibre configuration, cache and temp folders inside the job's work directory, so Filee
+// never reads or changes the settings of a calibre the user may have installed.
+
+using System.Globalization;
+using System.Text.RegularExpressions;
+using Filee.Core.Conversion;
+using Filee.Engines.Infrastructure;
+
+namespace Filee.Engines.Ebooks;
+
+/// Conversions through calibre's ebook-convert.
+public sealed partial class CalibreConverter : IConverter
+{
+ private static readonly TimeSpan Timeout = TimeSpan.FromMinutes(10);
+
+ /// Formats only calibre reads.
+ internal static readonly string[] Inputs = ["lit", "lrf", "chm", "pdb", "pml", "rb", "snb", "tcr", "oeb"];
+
+ /// Formats only calibre writes.
+ internal static readonly string[] Outputs = ["mobi", "azw3", "lit", "lrf", "pdb", "pml", "rb", "snb", "tcr"];
+
+ private string? _executable;
+
+ public string Id => "calibre";
+ public string DisplayName => "Calibre";
+
+ /// ebook-convert is a large Python process; two at a time keep the PC responsive.
+ public int MaxParallelism => 2;
+
+ public IReadOnlyList Edges { get; } =
+ [
+ .. Inputs.Select(from => new ConversionEdge(from, "epub", 20)),
+ .. Outputs.Select(to => new ConversionEdge("epub", to, 20)),
+ ];
+
+ public EngineStatus GetStatus()
+ {
+ _executable = Locate();
+ return _executable is null ? EngineStatus.Unavailable("engine.reason.not_installed") : EngineStatus.Available(_executable, EngineVersions.Component("calibre"));
+ }
+
+ public async Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken)
+ {
+ var executable = _executable ?? Locate() ?? throw new InvalidOperationException("Calibre is not installed.");
+ var work = EbookFiles.NewFolder(step.WorkDirectory, "calibre");
+
+ // ebook-convert picks formats by file extension: give it names it knows. An .opf (OEB) stays where it is,
+ // next to the files it lists; a zipped .oeb is read by calibre's ZIP input.
+ var input = step.InputPath;
+ if (!input.EndsWith(".opf", StringComparison.OrdinalIgnoreCase))
+ {
+ input = Path.Combine(work, "input." + (step.From == "oeb" ? "zip" : step.From));
+ File.Copy(step.InputPath, input);
+ }
+ var temp = Path.Combine(work, "output." + step.To);
+
+ var environment = new Dictionary
+ {
+ ["CALIBRE_CONFIG_DIRECTORY"] = Directory.CreateDirectory(Path.Combine(work, "config")).FullName,
+ ["CALIBRE_CACHE_DIRECTORY"] = Directory.CreateDirectory(Path.Combine(work, "cache")).FullName,
+ ["CALIBRE_TEMP_DIR"] = Directory.CreateDirectory(Path.Combine(work, "temp")).FullName,
+ ["CALIBRE_OVERRIDE_LANG"] = "en", // English messages in the error details, whatever the Windows language
+ ["PYTHONIOENCODING"] = "utf-8",
+ };
+ progress?.Report(0.02);
+ var result = await ProcessRunner.RunAsync(executable, [input, temp], Timeout, cancellationToken,
+ workingDirectory: work, environment: environment, onOutput: line => ReportProgress(line, progress));
+ if (result.ExitCode != 0 || !File.Exists(temp))
+ throw new InvalidOperationException($"Calibre failed (exit {result.ExitCode}). {LastLines(result.StandardError.Trim().Length > 0 ? result.StandardError : result.StandardOutput)}");
+
+ var output = step.Output.Allocate(step.To);
+ if (output is null)
+ return [];
+ File.Move(temp, output, overwrite: true);
+ progress?.Report(1);
+ return [output];
+ }
+
+ /// ebook-convert prints "34% Running transforms on e-book..." while it works.
+ private static void ReportProgress(string line, IProgress? progress)
+ {
+ if (progress is not null && Percent().Match(line) is { Success: true } match
+ && int.TryParse(match.Groups[1].Value, NumberStyles.None, CultureInfo.InvariantCulture, out var percent))
+ progress.Report(Math.Clamp(percent, 0, 100) / 100.0 * 0.95);
+ }
+
+ /// The end of calibre's output, where its error message (or Python traceback) is.
+ private static string LastLines(string output)
+ {
+ var lines = output.Split('\n', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries);
+ return string.Join(" ", lines.TakeLast(3));
+ }
+
+ [GeneratedRegex(@"^\s*(\d{1,3})%\s")]
+ private static partial Regex Percent();
+
+ /// Filee's own calibre (engines/calibre, downloaded on demand), never one installed on the system.
+ internal static string? Locate() =>
+ OperatingSystem.IsWindows() && EngineEnvironment.FindBundled("calibre") is { } folder
+ ? EngineEnvironment.FirstExisting(Path.Combine(folder, "ebook-convert.exe"))
+ : null;
+}
diff --git a/src/Filee.Engines/Ebooks/ChapterReader.cs b/src/Filee.Engines/Ebooks/ChapterReader.cs
new file mode 100644
index 0000000..34230d2
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/ChapterReader.cs
@@ -0,0 +1,91 @@
+// Reads the XHTML chapters of a book (EPUB spine, KF8 parts, HTMLZ page) into one HDocument: every chapter starts
+// on a new page, and links between chapters ("ch2.xhtml#note3") become links to bookmarks inside the document.
+
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal static class ChapterReader
+{
+ /// Reads chapter files in reading order.
+ /// Scratch folder for images embedded as data: URIs.
+ public static HDocument Read(IReadOnlyList chapters, string mediaFolder, CancellationToken cancellationToken = default)
+ {
+ var index = new Dictionary(StringComparer.OrdinalIgnoreCase);
+ for (var i = 0; i < chapters.Count; i++)
+ index.TryAdd(Path.GetFullPath(chapters[i]), i);
+
+ var section = new HSection();
+ var document = new HDocument();
+ document.Sections.Add(section);
+ for (var i = 0; i < chapters.Count; i++)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ var chapter = Path.GetFullPath(chapters[i]);
+ if (!File.Exists(chapter))
+ continue; // a spine item missing from the package: skip it like reading systems do
+ var number = i;
+ var content = HtmlReader.ReadFile(chapter, new HtmlReadOptions
+ {
+ BaseFolder = Path.GetDirectoryName(chapter)!,
+ MediaFolder = mediaFolder,
+ MapId = id => Anchor(number, id),
+ MapLink = href => Link(href, chapter, number, index),
+ });
+ document.Title ??= content.Title;
+ if (!content.Blocks.Any(HasContent))
+ continue;
+
+ // The chapter starts on a new page and carries a bookmark for links to the chapter file itself.
+ if (content.Blocks[0] is not HParagraph first)
+ content.Blocks.Insert(0, first = new HParagraph());
+ first.PageBreakBefore = section.Blocks.Count > 0;
+ first.Inlines.Insert(0, new HBookmark(Anchor(number, null)));
+ section.Blocks.AddRange(content.Blocks);
+ }
+ HtmlReader.PruneBookmarks(document);
+ return document;
+ }
+
+ private static bool HasContent(HBlock block) => block is HTable
+ || HDocumentWalker.Inlines([block]).Any(i => i is HImage || i is HText { Text: var text } && !string.IsNullOrWhiteSpace(text));
+
+ /// Bookmark name of an element id (or the chapter start) in chapter .
+ private static string Anchor(int chapter, string? id)
+ {
+ if (id is null)
+ return $"c{chapter + 1}";
+ var clean = new char[id.Length];
+ for (var i = 0; i < id.Length; i++)
+ clean[i] = char.IsAsciiLetterOrDigit(id[i]) || id[i] is '-' or '_' ? id[i] : '_';
+ return $"c{chapter + 1}-{new string(clean)}";
+ }
+
+ /// Links inside the book point at bookmarks; web and mail links stay; links to other files are dropped.
+ private static string? Link(string href, string chapter, int number, Dictionary index)
+ {
+ if (href.StartsWith('#'))
+ return href.Length > 1 ? "#" + Anchor(number, Uri.UnescapeDataString(href[1..])) : null;
+ var colon = href.IndexOf(':');
+ if (colon > 1 && href[..colon].All(char.IsAsciiLetter))
+ return href.StartsWith("javascript:", StringComparison.OrdinalIgnoreCase) ? null : href;
+
+ var hash = href.IndexOf('#');
+ var file = hash >= 0 ? href[..hash] : href;
+ var fragment = hash >= 0 ? Uri.UnescapeDataString(href[(hash + 1)..]) : null;
+ var query = file.IndexOf('?');
+ if (query >= 0)
+ file = file[..query];
+ try
+ {
+ var target = Path.GetFullPath(Path.Combine(Path.GetDirectoryName(chapter)!, Uri.UnescapeDataString(file)));
+ return index.TryGetValue(target, out var chapterIndex)
+ ? "#" + Anchor(chapterIndex, string.IsNullOrEmpty(fragment) ? null : fragment)
+ : null;
+ }
+ catch (Exception ex) when (ex is ArgumentException or NotSupportedException or PathTooLongException)
+ {
+ return null;
+ }
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/ComicBook.cs b/src/Filee.Engines/Ebooks/ComicBook.cs
new file mode 100644
index 0000000..e66eb83
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/ComicBook.cs
@@ -0,0 +1,214 @@
+// Comic book archives: CBZ (ZIP), CBR (RAR), CB7 (7-Zip), CBT (TAR) and CBC (a ZIP of CBZ files) are pictures in
+// an archive, read with SharpCompress (MIT). Pages are sorted by name the way people number them ("page2" before
+// "page10"). Comics become PDF (one page per picture, sized like the picture, JPEGs embedded as they are), EPUB
+// (fixed layout) or CBZ; PDF becomes CBZ by rendering its pages with PDFium.
+
+using System.IO.Compression;
+using Filee.Core.Conversion;
+using ImageMagick;
+using PdfSharp.Drawing;
+using PdfSharp.Pdf;
+using PDFtoImage;
+using SharpCompress.Archives;
+using SkiaSharp;
+
+// PDFtoImage marks its API for the platforms PDFium ships for (Windows, macOS, Linux, mobile); Filee only targets
+// desktop platforms that are all covered.
+#pragma warning disable CA1416
+
+namespace Filee.Engines.Ebooks;
+
+internal static class ComicBook
+{
+ private static readonly HashSet PictureExtensions =
+ [".jpg", ".jpeg", ".jpe", ".jfif", ".png", ".gif", ".webp", ".bmp", ".avif", ".jxl", ".tif", ".tiff", ".heic", ".heif"];
+
+ /// Extracts the pages of a comic archive into , in reading order.
+ public static List ExtractPages(string path, string format, string folder, CancellationToken cancellationToken)
+ {
+ if (format == "cbc")
+ {
+ // A collection: the CBZ files inside, in the order of comics.txt ("file.cbz:Title" per line) or by name.
+ var comics = ExtractArchive(path, EbookFiles.NewFolder(folder, "cbc"), f => Path.GetExtension(f).Equals(".cbz", StringComparison.OrdinalIgnoreCase) || Path.GetFileName(f).Equals("comics.txt", StringComparison.OrdinalIgnoreCase), cancellationToken);
+ var list = comics.FirstOrDefault(c => Path.GetFileName(c).Equals("comics.txt", StringComparison.OrdinalIgnoreCase));
+ var books = comics.Where(c => c != list).ToList();
+ if (list is not null)
+ {
+ var order = File.ReadAllLines(list).Select(l => l.Split(':')[0].Trim()).Where(l => l.Length > 0).ToList();
+ books = [.. books.OrderBy(b => order.FindIndex(o => b.EndsWith(o.Replace('/', Path.DirectorySeparatorChar), StringComparison.OrdinalIgnoreCase)) is var i and >= 0 ? i : int.MaxValue)];
+ }
+ return [.. books.SelectMany(book => ExtractPages(book, "cbz", EbookFiles.NewFolder(folder, "cbz"), cancellationToken))];
+ }
+ var pages = ExtractArchive(path, folder, f => PictureExtensions.Contains(Path.GetExtension(f).ToLowerInvariant()), cancellationToken);
+ if (pages.Count == 0)
+ throw new InvalidDataException("The comic contains no pictures.");
+ return pages;
+ }
+
+ /// Extracts the matching entries (skipping macOS metadata) and returns them sorted naturally by path.
+ private static List ExtractArchive(string path, string folder, Func wanted, CancellationToken cancellationToken)
+ {
+ var files = new List<(string Key, string File)>();
+ IArchive archive;
+ try
+ {
+ archive = ArchiveFactory.Open(path, null);
+ }
+ catch (Exception ex) when (ex is InvalidOperationException or InvalidDataException or ArgumentException or NotSupportedException or SharpCompress.Common.ArchiveException)
+ {
+ throw new InvalidDataException("The comic archive cannot be opened (damaged, or not the format its extension says).", ex);
+ }
+ using (archive)
+ {
+ if (archive.Entries.Any(e => e.IsEncrypted))
+ throw new InvalidOperationException("The comic archive is password-protected.");
+ if (archive.IsSolid || archive.Type == SharpCompress.Common.ArchiveType.SevenZip)
+ {
+ // Sequential extraction: random access would decompress a solid block again for every entry.
+ using var reader = archive.ExtractAllEntries();
+ while (reader.MoveToNextEntry())
+ {
+ if (Wanted(reader.Entry) is { } target)
+ {
+ using var output = File.Create(target);
+ reader.WriteEntryTo(output);
+ }
+ }
+ }
+ else
+ {
+ foreach (var entry in archive.Entries)
+ {
+ if (Wanted(entry) is not { } target)
+ continue;
+ using var input = entry.OpenEntryStream();
+ using var output = File.Create(target);
+ input.CopyTo(output);
+ }
+ }
+ }
+ return [.. files.OrderBy(f => f.Key, NaturalComparer.Instance).Select(f => f.File)];
+
+ // The file to extract an entry to, or null for folders, macOS metadata and unwanted files.
+ string? Wanted(SharpCompress.Common.IEntry entry)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ var key = entry.Key ?? "";
+ var name = Path.GetFileName(key.Replace('\\', '/'));
+ if (entry.IsDirectory || key.Contains("__MACOSX", StringComparison.OrdinalIgnoreCase) || name.StartsWith('.') || !wanted(name))
+ return null;
+ if (entry.IsEncrypted)
+ throw new InvalidOperationException("The comic archive is password-protected.");
+ var target = Path.Combine(folder, $"{files.Count:00000}{Path.GetExtension(name).ToLowerInvariant()}");
+ files.Add((key, target));
+ return target;
+ }
+ }
+
+ /// One PDF page per picture, each as large as its picture (at its DPI, else 96 dpi).
+ public static void ToPdf(IReadOnlyList pages, string output, string workFolder, IProgress? progress, CancellationToken cancellationToken)
+ {
+ var converted = EbookFiles.NewFolder(workFolder, "pdf-pages");
+ using var document = new PdfDocument();
+ for (var i = 0; i < pages.Count; i++)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ // JPEG and PNG go in as they are (PDFsharp embeds JPEG data without re-encoding); others become one of them.
+ if (EbookFiles.CommonImage(pages[i], converted, gifAndSvg: false) is not ({ } file, _))
+ continue;
+ var info = new MagickImageInfo(file);
+ var dpi = info.Density is { X: >= 36 } density ? density.Units == DensityUnit.PixelsPerCentimeter ? density.X * 2.54 : density.X : 96;
+ using var image = XImage.FromFile(file);
+ var page = document.AddPage();
+ page.Width = XUnit.FromPoint(info.Width / dpi * 72);
+ page.Height = XUnit.FromPoint(info.Height / dpi * 72);
+ using (var graphics = XGraphics.FromPdfPage(page))
+ graphics.DrawImage(image, 0, 0, page.Width.Point, page.Height.Point);
+ progress?.Report((i + 1.0) / pages.Count);
+ }
+ if (document.PageCount == 0)
+ throw new InvalidDataException("The comic contains no readable pictures.");
+ document.Save(output);
+ }
+
+ /// Writes pages into a CBZ (stored: pictures do not compress further), numbered in order.
+ public static void ToCbz(IReadOnlyList pages, string output, CancellationToken cancellationToken)
+ {
+ var temp = output + ".tmp";
+ using (var zip = ZipFile.Open(temp, ZipArchiveMode.Create))
+ {
+ var digits = Math.Max(3, pages.Count.ToString(System.Globalization.CultureInfo.InvariantCulture).Length);
+ for (var i = 0; i < pages.Count; i++)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ var name = (i + 1).ToString(System.Globalization.CultureInfo.InvariantCulture).PadLeft(digits, '0') + Path.GetExtension(pages[i]).ToLowerInvariant();
+ zip.CreateEntryFromFile(pages[i], name, CompressionLevel.NoCompression);
+ }
+ }
+ File.Move(temp, output, overwrite: true);
+ }
+
+ ///
+ /// Renders the PDF's pages (the preset's page range) as JPEG files at the preset's DPI and quality. PDFtoImage
+ /// serializes its calls into PDFium (not thread-safe), so this may run next to PdfiumConverter.
+ ///
+ public static List RenderPdf(string pdf, ConversionStep step, string folder, IProgress? progress, CancellationToken cancellationToken)
+ {
+ var bytes = File.ReadAllBytes(pdf);
+ var pageNumbers = PageRange.Parse(step.Preset.Pdf.PageRange, Conversion.GetPageCount(bytes, null));
+ var dpi = Math.Clamp(step.Preset.Pdf.RenderDpi, 36, 600);
+ var quality = Math.Clamp(step.Preset.Image.Quality, 1, 100);
+ var options = new RenderOptions(Dpi: dpi, WithAnnotations: true, WithFormFill: true, BackgroundColor: SKColors.White);
+ var pages = new List();
+ foreach (var bitmap in Conversion.ToImages(bytes, pageNumbers, null, options))
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ using (bitmap)
+ using (var data = bitmap.Encode(SKEncodedImageFormat.Jpeg, quality))
+ {
+ var file = Path.Combine(folder, $"{pages.Count + 1:00000}.jpg");
+ File.WriteAllBytes(file, data.ToArray());
+ pages.Add(file);
+ }
+ progress?.Report(0.9 * pages.Count / Math.Max(1, pageNumbers.Count));
+ }
+ return pages;
+ }
+}
+
+/// Compares names the way people number pages: digit runs by value ("p2" < "p10"), case-insensitively.
+internal sealed class NaturalComparer : IComparer
+{
+ public static readonly NaturalComparer Instance = new();
+
+ public int Compare(string? x, string? y)
+ {
+ if (x is null || y is null)
+ return string.CompareOrdinal(x, y);
+ int i = 0, j = 0;
+ while (i < x.Length && j < y.Length)
+ {
+ if (char.IsAsciiDigit(x[i]) && char.IsAsciiDigit(y[j]))
+ {
+ var startX = i;
+ var startY = j;
+ while (i < x.Length && char.IsAsciiDigit(x[i]))
+ i++;
+ while (j < y.Length && char.IsAsciiDigit(y[j]))
+ j++;
+ var numberX = x[startX..i].TrimStart('0');
+ var numberY = y[startY..j].TrimStart('0');
+ var byValue = numberX.Length != numberY.Length ? numberX.Length.CompareTo(numberY.Length) : string.CompareOrdinal(numberX, numberY);
+ if (byValue != 0)
+ return byValue;
+ continue;
+ }
+ var byChar = char.ToLowerInvariant(x[i]).CompareTo(char.ToLowerInvariant(y[j]));
+ if (byChar != 0)
+ return byChar;
+ i++;
+ j++;
+ }
+ return (x.Length - i).CompareTo(y.Length - j);
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/EbookConverter.cs b/src/Filee.Engines/Ebooks/EbookConverter.cs
new file mode 100644
index 0000000..3c420bb
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/EbookConverter.cs
@@ -0,0 +1,211 @@
+// Built-in e-book conversions, no external program: EPUB, MOBI / AZW / AZW3 / PRC, FB2, HTMLZ and TXTZ are read into
+// the HWPX document model and written as EPUB, HWPX (→ PDF / images through rhwp and PDFium), TXT, HTML, Markdown,
+// FB2, HTMLZ or TXTZ; HTML, Markdown, TXT and DOCX become e-books too. Comics (CBZ / CBR / CB7 / CBT / CBC) become
+// PDF, fixed-layout EPUB or CBZ, PDF becomes CBZ, and AZW4 (Print Replica) gives back its PDF.
+// Formats only calibre reads or writes (LIT, LRF, PDB, MOBI / AZW3 output, ...) are in CalibreConverter.
+
+using System.Text;
+using Filee.Core.Conversion;
+using Filee.Engines.Hwp.Hwpx;
+using Filee.Engines.Hwp.Hwpx.Docx;
+using Filee.Engines.Infrastructure;
+
+namespace Filee.Engines.Ebooks;
+
+/// E-book and comic conversions without external programs.
+public sealed class EbookConverter : IConverter
+{
+ /// E-book formats read into the document model.
+ private static readonly string[] BookReaders = ["epub", "mobi", "azw3", "azw", "prc", "fb2", "htmlz", "txtz"];
+
+ /// Formats written from the document model.
+ private static readonly string[] BookWriters = ["epub", "hwpx", "txt", "html", "md", "fb2", "htmlz", "txtz"];
+
+ /// Documents that become e-books (their other conversions belong to the document engines).
+ private static readonly string[] DocumentReaders = ["html", "md", "txt", "docx"];
+
+ private static readonly string[] EbookWriters = ["epub", "fb2", "htmlz", "txtz"];
+
+ private static readonly string[] Comics = ["cbz", "cbr", "cb7", "cbt", "cbc"];
+
+ public string Id => "ebook";
+ public string DisplayName => "E-books (built-in)";
+ public int MaxParallelism => 0;
+
+ public IReadOnlyList Edges { get; } =
+ [
+ .. BookReaders.SelectMany(from => BookWriters.Where(to => to != from).Select(to => new ConversionEdge(from, to, to == "md" ? 12 : 10))),
+ .. DocumentReaders.SelectMany(from => EbookWriters.Select(to => new ConversionEdge(from, to))),
+ // HTML → TXT has no other built-in route; HTML → Markdown is a fallback for when Pandoc is not installed.
+ new("html", "txt"),
+ new("html", "md", 14),
+ .. Comics.SelectMany(from => new[] { new ConversionEdge(from, "pdf"), new ConversionEdge(from, "epub") }),
+ .. Comics.Where(from => from != "cbz").Select(from => new ConversionEdge(from, "cbz")),
+ // Pages as pictures lose the text: other routes to EPUB (through DOCX, TXT, ...) should win when they exist.
+ new("pdf", "cbz", 15),
+ new("azw4", "pdf"),
+ ];
+
+ public EngineStatus GetStatus() => EngineStatus.Available("EPUB, MOBI / AZW3, FB2, HTMLZ, TXTZ, CBZ / CBR / CB7 / CBT / CBC, AZW4",
+ $"{EngineVersions.BuiltIn} · {EngineVersions.Library("AngleSharp", typeof(AngleSharp.BrowsingContext))}");
+
+ public Task> ConvertAsync(ConversionStep step, IProgress? progress, CancellationToken cancellationToken) =>
+ Task.Run(() => Convert(step, progress, cancellationToken), cancellationToken);
+
+ private static IReadOnlyList Convert(ConversionStep step, IProgress? progress, CancellationToken cancellationToken)
+ {
+ progress?.Report(0.05);
+ var work = EbookFiles.NewFolder(step.WorkDirectory, "ebook");
+ if (Comics.Contains(step.From) || step.From == "pdf")
+ return Comic(step, work, progress, cancellationToken);
+ if (step.From == "azw4")
+ {
+ var pdf = MobiReader.PrintReplicaPdf(step.InputPath);
+ var target = step.Output.Allocate("pdf");
+ if (target is null)
+ return [];
+ File.WriteAllBytes(target, pdf);
+ progress?.Report(1);
+ return [target];
+ }
+
+ var book = Read(step, work, cancellationToken);
+ progress?.Report(0.5);
+ cancellationToken.ThrowIfCancellationRequested();
+ var output = step.Output.Allocate(step.To);
+ if (output is null)
+ return [];
+ Write(book, step, output, work);
+ progress?.Report(1);
+ return [output];
+ }
+
+ private static Book Read(ConversionStep step, string work, CancellationToken cancellationToken)
+ {
+ var input = step.InputPath;
+ var media = Path.Combine(work, "media");
+ switch (step.From)
+ {
+ case "epub":
+ return EpubReader.ReadBook(input, work, cancellationToken);
+ case "mobi" or "azw3" or "azw" or "prc":
+ return MobiReader.ReadBook(input, work, cancellationToken);
+ case "fb2":
+ return Fb2Reader.ReadBook(input, work);
+ case "htmlz":
+ return ZippedText.ReadHtmlzBook(input, work, cancellationToken);
+ case "txtz":
+ return ZippedText.ReadTxtzBook(input, work);
+ case "html":
+ {
+ var document = HtmlReader.Read(input, media);
+ return new Book(document, new BookMetadata { Title = document.Title });
+ }
+ case "md":
+ {
+ var document = MarkdownReader.Read(HwpxConverter.DecodeText(File.ReadAllBytes(input)), Path.GetDirectoryName(Path.GetFullPath(input))!);
+ return new Book(document, new BookMetadata { Title = document.Title });
+ }
+ case "txt":
+ return new Book(PlainText.ToDocument(HwpxConverter.DecodeText(File.ReadAllBytes(input))), new BookMetadata());
+ case "docx":
+ {
+ var document = DocxReader.Read(input, media);
+ return new Book(document, new BookMetadata { Title = document.Title });
+ }
+ default:
+ throw new NotSupportedException($"{step.From} → {step.To}");
+ }
+ }
+
+ private static void Write(Book book, ConversionStep step, string output, string work)
+ {
+ switch (step.To)
+ {
+ case "epub":
+ EpubWriter.Write(book, output, step.InputPath, work);
+ break;
+ case "hwpx":
+ book.Document.Title = book.DisplayTitle(step.InputPath);
+ HwpxWriter.Write(book.Document, output);
+ break;
+ case "txt":
+ File.WriteAllText(output, PlainTextWriter.Write(book.Document), new UTF8Encoding(encoderShouldEmitUTF8Identifier: true));
+ break;
+ case "html":
+ {
+ // One self-contained page: pictures are embedded as data: URIs.
+ var images = EbookFiles.NewFolder(work, "html-images");
+ var page = XhtmlWriter.Write(book.Document, new XhtmlOptions { ImageSource = path => DataUri(path, images) }).Single();
+ File.WriteAllText(output, XhtmlWriter.Page(book.DisplayTitle(step.InputPath), page.Body, book.DisplayLanguage(), null, epub: false), new UTF8Encoding(false));
+ break;
+ }
+ case "md":
+ File.WriteAllText(output, MarkdownWriter.Write(book.Document, MarkdownImages(output, work)), new UTF8Encoding(false));
+ break;
+ case "fb2":
+ Fb2Writer.Write(book, output, step.InputPath, work);
+ break;
+ case "htmlz":
+ ZippedText.WriteHtmlz(book, output, step.InputPath, work);
+ break;
+ case "txtz":
+ ZippedText.WriteTxtz(book, output, step.InputPath, work);
+ break;
+ default:
+ throw new NotSupportedException($"{step.From} → {step.To}");
+ }
+ }
+
+ private static IReadOnlyList Comic(ConversionStep step, string work, IProgress? progress, CancellationToken cancellationToken)
+ {
+ var pages = step.From == "pdf"
+ ? ComicBook.RenderPdf(step.InputPath, step, work, progress, cancellationToken)
+ : ComicBook.ExtractPages(step.InputPath, step.From, work, cancellationToken);
+ progress?.Report(0.4);
+ var output = step.Output.Allocate(step.To);
+ if (output is null)
+ return [];
+ switch (step.To)
+ {
+ case "pdf":
+ ComicBook.ToPdf(pages, output, work, progress, cancellationToken);
+ break;
+ case "epub":
+ EpubWriter.WriteFixedLayout(Filee.Core.Formats.FormatRegistry.NameWithoutExtension(step.InputPath), pages, output, work, cancellationToken);
+ break;
+ case "cbz":
+ ComicBook.ToCbz(pages, output, cancellationToken);
+ break;
+ default:
+ throw new NotSupportedException($"{step.From} → {step.To}");
+ }
+ progress?.Report(1);
+ return [output];
+ }
+
+ private static string? DataUri(string path, string convertFolder) =>
+ EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is ({ } file, { } mediaType)
+ ? $"data:{mediaType};base64,{System.Convert.ToBase64String(File.ReadAllBytes(file))}"
+ : null;
+
+ /// Pictures of a Markdown export go to "<name>_files" next to it, referenced relatively.
+ private static Func MarkdownImages(string output, string work)
+ {
+ var folderName = Path.GetFileNameWithoutExtension(output) + "_files";
+ var folder = Path.Combine(Path.GetDirectoryName(output)!, folderName);
+ var convert = EbookFiles.NewFolder(work, "md-images");
+ var names = new Dictionary(StringComparer.OrdinalIgnoreCase);
+ return path =>
+ {
+ if (names.TryGetValue(path, out var known))
+ return known;
+ if (EbookFiles.CommonImage(path, convert, gifAndSvg: true) is not ({ } file, _))
+ return null;
+ Directory.CreateDirectory(folder);
+ var name = $"img{names.Count + 1:0000}{Path.GetExtension(file).ToLowerInvariant()}";
+ File.Copy(file, Path.Combine(folder, name), overwrite: true);
+ return names[path] = $"{Uri.EscapeDataString(folderName)}/{name}";
+ };
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/EpubReader.cs b/src/Filee.Engines/Ebooks/EpubReader.cs
new file mode 100644
index 0000000..fac6764
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/EpubReader.cs
@@ -0,0 +1,92 @@
+// EPUB 2 / 3 → Book: META-INF/container.xml → the OPF package → spine → XHTML chapters through HtmlReader (each
+// chapter starts a new page), pictures from the package, title / authors / language / cover from the metadata.
+// Books with DRM (Adobe ADEPT, Apple FairPlay, ...) are refused; font obfuscation is not DRM and is fine.
+
+using System.IO.Compression;
+using System.Xml.Linq;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal static class EpubReader
+{
+ /// Algorithms that only obfuscate embedded fonts (IDPF and Adobe), which do not protect the content.
+ private static readonly HashSet FontObfuscation =
+ [
+ "http://www.idpf.org/2008/embedding",
+ "http://ns.adobe.com/pdf/enc#RC",
+ ];
+
+ /// Reads an EPUB's content (for the reader registry of the HWPX writer).
+ /// Scratch folder: the book is unpacked into a sub folder.
+ public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document;
+
+ public static Book ReadBook(string path, string workFolder, CancellationToken cancellationToken = default)
+ {
+ var folder = EbookFiles.NewFolder(workFolder, "epub");
+ using (var zip = OpenZip(path))
+ {
+ CheckDrm(zip);
+ EbookFiles.Extract(zip, folder);
+ }
+
+ var package = OpfPackage.Load(OpfPackage.RootFile(folder));
+ var chapters = new List();
+ foreach (var item in package.Spine)
+ {
+ if (item.MediaType is "application/xhtml+xml" or "text/html" or "application/xml" || item.MediaType.Length == 0 && item.Path.EndsWith("html", StringComparison.OrdinalIgnoreCase))
+ chapters.Add(item.Path);
+ else if (item.MediaType.StartsWith("image/", StringComparison.Ordinal))
+ chapters.Add(ImagePage(item.Path)); // a picture in the spine (comics, covers) is a page of its own
+ }
+ if (chapters.Count == 0)
+ throw new InvalidDataException("The EPUB has no readable chapters.");
+
+ var document = ChapterReader.Read(chapters, Path.Combine(folder, "~media"), cancellationToken);
+ document.Title = package.Metadata.Title ?? document.Title;
+ return new Book(document, package.Metadata);
+ }
+
+ private static ZipArchive OpenZip(string path)
+ {
+ try
+ {
+ return ZipFile.OpenRead(path);
+ }
+ catch (InvalidDataException ex)
+ {
+ throw new InvalidDataException("This is not a valid EPUB file (it is not a ZIP package).", ex);
+ }
+ }
+
+ /// Throws for encrypted content: anything in encryption.xml except font obfuscation, or Apple's sinf.xml.
+ internal static void CheckDrm(ZipArchive zip)
+ {
+ if (zip.GetEntry("META-INF/sinf.xml") is not null)
+ throw new InvalidOperationException(EbookFiles.DrmMessage);
+ if (zip.GetEntry("META-INF/encryption.xml") is not { } entry)
+ return;
+ XDocument encryption;
+ try
+ {
+ using var stream = entry.Open();
+ encryption = XDocument.Load(stream);
+ }
+ catch (System.Xml.XmlException)
+ {
+ return; // unreadable: the chapters will tell whether they are readable
+ }
+ var algorithms = encryption.Descendants().Where(e => e.Name.LocalName == "EncryptionMethod")
+ .Select(e => (string?)e.Attribute("Algorithm") ?? "");
+ if (algorithms.Any(a => !FontObfuscation.Contains(a)))
+ throw new InvalidOperationException(EbookFiles.DrmMessage);
+ }
+
+ /// An XHTML page next to a picture that the spine lists directly.
+ private static string ImagePage(string image)
+ {
+ var page = Path.Combine(Path.GetDirectoryName(image)!, $"~{Path.GetFileNameWithoutExtension(image)}-{Guid.NewGuid():N}.xhtml");
+ File.WriteAllText(page, $"
");
+ return page;
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/EpubWriter.cs b/src/Filee.Engines/Ebooks/EpubWriter.cs
new file mode 100644
index 0000000..82781e9
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/EpubWriter.cs
@@ -0,0 +1,265 @@
+// Book → EPUB 3 with an EPUB 2 NCX for older readers: "mimetype" first and stored, container.xml, an OPF with Dublin
+// Core metadata, a navigation document built from the headings, one XHTML file per chapter (split at level-1
+// headings and page breaks by XhtmlWriter), one style sheet and the pictures. Pictures in formats reading systems
+// do not all show (BMP, TIFF, WEBP, ...) are converted to PNG or JPEG. Fixed-layout books (comics) have one page
+// per picture, sized like the picture.
+
+using System.IO.Compression;
+using System.Text;
+using Filee.Engines.Hwp.Hwpx;
+using ImageMagick;
+
+namespace Filee.Engines.Ebooks;
+
+internal static class EpubWriter
+{
+ /// Writes a reflowable EPUB.
+ /// The file being converted, for a title when the book has none.
+ public static void Write(Book book, string outputPath, string sourcePath, string workFolder)
+ {
+ var title = book.DisplayTitle(sourcePath);
+ var language = book.DisplayLanguage();
+ var package = new EpubPackage(title, language, book.Metadata);
+ var imageFolder = EbookFiles.NewFolder(workFolder, "epub-images");
+
+ var chapters = XhtmlWriter.Write(book.Document, new XhtmlOptions
+ {
+ SplitChapters = true,
+ ChapterFile = i => $"ch{i + 1:000}.xhtml",
+ ImageSource = path => package.AddImage(path, imageFolder),
+ Epub = true,
+ });
+
+ // A cover page for covers the content does not show already (MOBI and FB2 keep the cover apart).
+ var cover = book.Metadata.CoverImage is { } coverFile && File.Exists(coverFile) ? package.AddImage(coverFile, imageFolder, cover: true) : null;
+ if (cover is not null && !HDocumentWalker.Images(book.Document).Any(i => string.Equals(i.Path, book.Metadata.CoverImage, StringComparison.OrdinalIgnoreCase)))
+ {
+ var body = $"
\n";
+ package.AddPage("cover.xhtml", XhtmlWriter.Page(title, body, language, "style.css", epub: true));
+ }
+
+ foreach (var chapter in chapters)
+ package.AddPage(chapter.FileName, XhtmlWriter.Page(chapter.Title ?? title, chapter.Body, language, "style.css", epub: true));
+ package.Toc.AddRange(TableOfContents(chapters, title));
+ package.Save(outputPath, XhtmlWriter.Css, fixedLayout: false);
+ }
+
+ /// Writes a fixed-layout EPUB with one page per picture (comics).
+ /// Picture files in reading order.
+ public static void WriteFixedLayout(string title, IReadOnlyList pages, string outputPath, string workFolder, CancellationToken cancellationToken)
+ {
+ var package = new EpubPackage(title, "en", new BookMetadata());
+ var imageFolder = EbookFiles.NewFolder(workFolder, "epub-images");
+ var count = 0;
+ foreach (var picture in pages)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ if (package.AddImage(picture, imageFolder, cover: count == 0) is not { } source)
+ continue; // unreadable picture: leave the page out
+ var info = new MagickImageInfo(picture);
+ var page = $"p{++count:0000}.xhtml";
+ var head = $"\n";
+ var body = $"
\n";
+ package.AddPage(page, XhtmlWriter.Page(title, body, "en", "style.css", epub: true, head));
+ }
+ if (count == 0)
+ throw new InvalidDataException("The comic contains no readable pictures.");
+ package.Save(outputPath, "body{margin:0;padding:0;}\n.page{margin:0;padding:0;}\nimg{display:block;margin:0;padding:0;}\n", fixedLayout: true);
+ }
+
+ ///
+ /// Entries for the navigation document: the two highest heading levels used in the book; chapters without
+ /// headings appear by their first words when no chapter has a heading.
+ ///
+ private static List TableOfContents(IReadOnlyList chapters, string title)
+ {
+ var headings = chapters.SelectMany(c => c.Headings).Where(h => h.Text.Length > 0).ToList();
+ if (headings.Count > 0)
+ {
+ var top = headings.Min(h => h.Level);
+ return headings.Where(h => h.Level <= top + 1).Select(h => h with { Level = h.Level - top + 1 }).ToList();
+ }
+ return chapters.Select((c, i) => new XhtmlHeading(1, i == 0 ? title : c.PlainStart.Length > 0 ? c.PlainStart + "…" : $"{i + 1}", c.FileName)).ToList();
+ }
+}
+
+/// Collects the files of an EPUB and writes the package.
+internal sealed class EpubPackage(string title, string language, BookMetadata metadata)
+{
+ private readonly List<(string Href, string Content)> _pages = [];
+ private readonly List<(string Href, string Source, string MediaType, bool Cover)> _images = [];
+ private readonly Dictionary _imageBySource = new(StringComparer.OrdinalIgnoreCase);
+
+ /// Navigation entries; level 1 is the top.
+ public List Toc { get; } = [];
+
+ public void AddPage(string href, string content) => _pages.Add((href, content));
+
+ ///
+ /// Adds a picture and returns its path in the package, or null when it cannot be read. JPEG, PNG, GIF and SVG
+ /// are kept; other formats become PNG (JPEG for opaque photos, which would grow a lot as PNG).
+ ///
+ public string? AddImage(string path, string convertFolder, bool cover = false)
+ {
+ if (_imageBySource.TryGetValue(path, out var known))
+ {
+ if (cover)
+ MarkCover(known);
+ return known;
+ }
+ if (EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is not ({ } file, { } mediaType))
+ return null;
+ var extension = mediaType switch
+ {
+ "image/jpeg" => "jpg",
+ "image/png" => "png",
+ "image/gif" => "gif",
+ _ => "svg",
+ };
+ var href = $"images/img{_images.Count + 1:0000}.{extension}";
+ _images.Add((href, file, mediaType!, cover));
+ _imageBySource[path] = href;
+ return href;
+ }
+
+ private void MarkCover(string href)
+ {
+ var index = _images.FindIndex(i => i.Href == href);
+ if (index >= 0)
+ _images[index] = _images[index] with { Cover = true };
+ }
+
+ public void Save(string outputPath, string css, bool fixedLayout)
+ {
+ var identifier = $"urn:uuid:{Guid.NewGuid()}";
+ var temp = outputPath + ".tmp";
+ using (var stream = File.Create(temp))
+ using (var zip = new ZipArchive(stream, ZipArchiveMode.Create))
+ {
+ // OCF: "mimetype" first, stored, without extra fields.
+ Text(zip, "mimetype", "application/epub+zip", CompressionLevel.NoCompression);
+ Text(zip, "META-INF/container.xml",
+ "\n\n" +
+ "\n\n");
+ Text(zip, "OEBPS/content.opf", Opf(identifier, fixedLayout));
+ Text(zip, "OEBPS/nav.xhtml", Nav());
+ Text(zip, "OEBPS/toc.ncx", Ncx(identifier));
+ Text(zip, "OEBPS/style.css", css);
+ foreach (var (href, content) in _pages)
+ Text(zip, "OEBPS/" + href, content);
+ foreach (var (href, source, _, _) in _images)
+ {
+ // Pictures are compressed already.
+ using var target = zip.CreateEntry("OEBPS/" + href, CompressionLevel.NoCompression).Open();
+ using var input = File.OpenRead(source);
+ input.CopyTo(target);
+ }
+ }
+ File.Move(temp, outputPath, overwrite: true);
+ }
+
+ private string Opf(string identifier, bool fixedLayout)
+ {
+ var sb = new StringBuilder();
+ sb.Append("\n");
+ sb.Append($"\n\n");
+ sb.Append($"{identifier}\n");
+ sb.Append($"{E(title)}\n");
+ sb.Append($"{E(language)}\n");
+ foreach (var author in metadata.Authors)
+ sb.Append($"{E(author)}\n");
+ if (metadata.Publisher is { } publisher)
+ sb.Append($"{E(publisher)}\n");
+ if (metadata.Description is { } description)
+ sb.Append($"{E(description)}\n");
+ sb.Append($"{DateTime.UtcNow:yyyy-MM-ddTHH:mm:ssZ}\n");
+ if (_images.FindIndex(i => i.Cover) is var cover and >= 0)
+ sb.Append($"\n"); // EPUB 2 readers (and Kindle converters)
+ if (fixedLayout)
+ sb.Append("pre-paginated\nauto\n");
+ sb.Append("\n\n");
+ sb.Append("\n");
+ sb.Append("\n");
+ sb.Append("\n");
+ for (var i = 0; i < _pages.Count; i++)
+ sb.Append($"\n");
+ for (var i = 0; i < _images.Count; i++)
+ sb.Append($"\n");
+ sb.Append("\n\n");
+ for (var i = 0; i < _pages.Count; i++)
+ sb.Append($"\n");
+ sb.Append("\n\n");
+ return sb.ToString();
+ }
+
+ /// The EPUB 3 navigation document: a nested list of the table of contents.
+ private string Nav()
+ {
+ var sb = new StringBuilder();
+ sb.Append("\n");
+ return XhtmlWriter.Page(title, sb.ToString(), language, "style.css", epub: true);
+ }
+
+ private string Ncx(string identifier)
+ {
+ var sb = new StringBuilder();
+ sb.Append("\n\n\n");
+ sb.Append($"\n e.Level))}\"/>\n");
+ sb.Append("\n\n\n");
+ sb.Append($"{E(title)}\n\n");
+ var entries = Entries();
+ var depth = 0;
+ for (var i = 0; i < entries.Count; i++)
+ {
+ var level = Math.Min(entries[i].Level, depth + 1);
+ for (; depth >= level; depth--)
+ sb.Append("\n");
+ depth = level;
+ sb.Append($"{E(entries[i].Text)}\n");
+ }
+ for (; depth > 0; depth--)
+ sb.Append("\n");
+ sb.Append("\n\n");
+ return sb.ToString();
+ }
+
+ /// Navigation entries; at least one (the first page), as EPUB requires a non-empty table of contents.
+ private List Entries() =>
+ Toc.Count > 0 ? Toc : [new XhtmlHeading(1, title, _pages.Count > 0 ? _pages[0].Href : "nav.xhtml")];
+
+ private string ContentsLabel() => language.Split('-')[0].ToLowerInvariant() switch
+ {
+ "ko" => "목차",
+ "zh" => "目录",
+ "ja" => "目次",
+ _ => "Contents",
+ };
+
+ private static string E(string text) => XhtmlWriter.Escape(text);
+
+ private static void Text(ZipArchive zip, string name, string content, CompressionLevel level = CompressionLevel.Optimal)
+ {
+ using var writer = new StreamWriter(zip.CreateEntry(name, level).Open(), new UTF8Encoding(false));
+ writer.Write(content);
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Fb2Reader.cs b/src/Filee.Engines/Ebooks/Fb2Reader.cs
new file mode 100644
index 0000000..1fd30bd
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Fb2Reader.cs
@@ -0,0 +1,422 @@
+// FictionBook 2 (FB2) → Book. FB2 is XML: description/title-info holds the metadata (title, authors, language,
+// cover), bodies hold nested sections (their titles become headings by depth), binary elements hold the pictures
+// as base64. Poems, epigraphs, citations, subtitles and tables are kept; note links (type="note") become real
+// footnotes with the text of the notes body. Every top-level section starts a new page.
+
+using System.Text;
+using System.Xml;
+using System.Xml.Linq;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal sealed class Fb2Reader
+{
+ private const int Indent = 2000;
+ private readonly Dictionary _binaries = new(StringComparer.Ordinal);
+ private readonly Dictionary _notes = new(StringComparer.Ordinal);
+
+ private Fb2Reader()
+ {
+ }
+
+ /// Reads an FB2 book's content (for the reader registry of the HWPX writer).
+ public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document;
+
+ public static Book ReadBook(string path, string workFolder)
+ {
+ var root = Load(path).Root ?? throw new InvalidDataException("The FB2 file is empty.");
+ if (root.Name.LocalName != "FictionBook")
+ throw new InvalidDataException("This is not a FictionBook (FB2) file.");
+ var reader = new Fb2Reader();
+ var folder = EbookFiles.NewFolder(workFolder, "fb2");
+ reader.SaveBinaries(root, folder);
+
+ var bodies = Children(root, "body").ToList();
+ foreach (var notes in bodies.Where(b => (string?)b.Attribute("name") is "notes" or "comments"))
+ foreach (var section in notes.Descendants().Where(e => e.Name.LocalName == "section" && e.Attribute("id") is not null))
+ reader._notes.TryAdd((string)section.Attribute("id")!, section);
+
+ var metadata = reader.Metadata(root);
+ var sectionBlocks = new HSection();
+ foreach (var body in bodies.Where(b => (string?)b.Attribute("name") is not ("notes" or "comments")))
+ reader.Body(body, sectionBlocks.Blocks);
+ var document = new HDocument { Title = metadata.Title };
+ document.Sections.Add(sectionBlocks);
+ HtmlReader.PruneBookmarks(document);
+ return new Book(document, metadata);
+ }
+
+ private static XDocument Load(string path)
+ {
+ // FB2 files are often windows-1251; the XML declaration names the encoding.
+ Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
+ var settings = new XmlReaderSettings { DtdProcessing = DtdProcessing.Ignore, XmlResolver = null };
+ using var reader = XmlReader.Create(path, settings);
+ // Spaces between inline elements ("ab") are text.
+ return XDocument.Load(reader, LoadOptions.PreserveWhitespace);
+ }
+
+ private static IEnumerable Children(XElement? parent, string name) =>
+ parent?.Elements().Where(e => e.Name.LocalName == name) ?? [];
+
+ private static XElement? Child(XElement? parent, string name) => Children(parent, name).FirstOrDefault();
+
+ /// The xlink:href of an image or link (namespace prefixes vary: l:, xlink:, none).
+ private static string? Href(XElement element) =>
+ element.Attributes().FirstOrDefault(a => a.Name.LocalName == "href")?.Value;
+
+ private BookMetadata Metadata(XElement root)
+ {
+ var info = Child(Child(root, "description"), "title-info");
+ var metadata = new BookMetadata
+ {
+ Title = Child(info, "book-title")?.Value.Trim(),
+ Language = Child(info, "lang")?.Value.Trim() is { Length: > 0 } lang ? lang : null,
+ Description = Child(info, "annotation")?.Value.Trim() is { Length: > 0 } annotation ? annotation : null,
+ Publisher = Child(Child(Child(root, "description"), "publish-info"), "publisher")?.Value.Trim(),
+ };
+ foreach (var author in Children(info, "author"))
+ {
+ var parts = new[] { "first-name", "middle-name", "last-name" }.Select(n => Child(author, n)?.Value.Trim()).Where(p => !string.IsNullOrEmpty(p));
+ var name = string.Join(" ", parts);
+ if (name.Length == 0)
+ name = Child(author, "nickname")?.Value.Trim() ?? "";
+ if (name.Length > 0)
+ metadata.Authors.Add(name);
+ }
+ if (Child(Child(info, "coverpage"), "image") is { } cover && Href(cover) is { } href && _binaries.TryGetValue(href.TrimStart('#'), out var file))
+ metadata.CoverImage = file;
+ return metadata;
+ }
+
+ private void SaveBinaries(XElement root, string folder)
+ {
+ foreach (var binary in Children(root, "binary"))
+ {
+ var id = (string?)binary.Attribute("id");
+ if (string.IsNullOrEmpty(id))
+ continue;
+ byte[] data;
+ try
+ {
+ data = Convert.FromBase64String(binary.Value.Trim());
+ }
+ catch (FormatException)
+ {
+ continue;
+ }
+ var extension = ((string?)binary.Attribute("content-type"))?.ToLowerInvariant() switch
+ {
+ "image/png" => "png",
+ "image/gif" => "gif",
+ "image/jpeg" or "image/jpg" => "jpg",
+ _ => data is [0x89, (byte)'P', ..] ? "png" : data is [(byte)'G', (byte)'I', (byte)'F', ..] ? "gif" : "jpg",
+ };
+ var file = Path.Combine(folder, $"binary{_binaries.Count + 1:0000}.{extension}");
+ File.WriteAllBytes(file, data);
+ _binaries[id] = file;
+ }
+ }
+
+ // ───────────────────────── Blocks ─────────────────────────
+
+ private void Body(XElement body, List output)
+ {
+ foreach (var element in body.Elements())
+ {
+ switch (element.Name.LocalName)
+ {
+ case "title":
+ Title(element, output, 1, center: true);
+ break;
+ case "epigraph":
+ Epigraph(element, output);
+ break;
+ case "image":
+ Image(element, output, 0);
+ break;
+ case "section":
+ Section(element, output, 1);
+ break;
+ }
+ }
+ }
+
+ private void Section(XElement section, List output, int depth)
+ {
+ var start = output.Count;
+ foreach (var element in section.Elements())
+ Block(element, output, depth, 0);
+ if (output.Count == start)
+ return;
+ // Every top-level section (a chapter) starts on a new page; bookmarks let links reach the section.
+ if (output[start] is not HParagraph first)
+ output.Insert(start, first = new HParagraph());
+ first.PageBreakBefore = depth == 1 && start > 0;
+ if ((string?)section.Attribute("id") is { Length: > 0 } id)
+ first.Inlines.Insert(0, new HBookmark(id));
+ }
+
+ private void Block(XElement element, List output, int depth, int indent)
+ {
+ switch (element.Name.LocalName)
+ {
+ case "title":
+ Title(element, output, depth, center: false);
+ break;
+ case "section":
+ Section(element, output, depth + 1);
+ break;
+ case "p":
+ output.Add(Paragraph(element, new HParaFormat(Left: indent > 0 ? indent : null), default));
+ break;
+ case "subtitle":
+ output.Add(Paragraph(element, new HParaFormat(Align: HAlign.Center), new HCharFormat(Bold: true)));
+ break;
+ case "empty-line":
+ output.Add(new HParagraph());
+ break;
+ case "epigraph":
+ Epigraph(element, output);
+ break;
+ case "annotation" or "cite":
+ foreach (var child in element.Elements())
+ Block(child, output, depth, indent + Indent);
+ break;
+ case "text-author":
+ output.Add(Paragraph(element, new HParaFormat(Align: HAlign.Right, Left: indent > 0 ? indent : null), new HCharFormat(Italic: true)));
+ break;
+ case "poem":
+ Poem(element, output, depth, indent + Indent);
+ break;
+ case "image":
+ Image(element, output, indent);
+ break;
+ case "table":
+ output.Add(Table(element));
+ break;
+ }
+ }
+
+ private void Title(XElement title, List output, int depth, bool center)
+ {
+ // A title may have several paragraphs; they form one heading, one line each.
+ var heading = new HParagraph
+ {
+ HeadingLevel = Math.Clamp(depth, 1, 6),
+ Format = center ? new HParaFormat(Align: HAlign.Center) : default,
+ };
+ foreach (var paragraph in Children(title, "p"))
+ {
+ if (heading.Inlines.Count > 0)
+ heading.Inlines.Add(new HLineBreak(default));
+ heading.Inlines.AddRange(Line(paragraph, default));
+ }
+ if (heading.Inlines.Count > 0)
+ output.Add(heading);
+ }
+
+ private void Epigraph(XElement epigraph, List output)
+ {
+ foreach (var child in epigraph.Elements())
+ {
+ if (child.Name.LocalName == "p")
+ output.Add(Paragraph(child, new HParaFormat(Align: HAlign.Right), new HCharFormat(Italic: true)));
+ else
+ Block(child, output, 0, Indent);
+ }
+ }
+
+ private void Poem(XElement poem, List output, int depth, int indent)
+ {
+ foreach (var child in poem.Elements())
+ {
+ switch (child.Name.LocalName)
+ {
+ case "stanza":
+ {
+ // One paragraph per stanza, one line per verse.
+ var stanza = new HParagraph { Format = new HParaFormat(Left: indent) };
+ foreach (var line in child.Elements())
+ {
+ if (line.Name.LocalName is "title" or "subtitle")
+ {
+ output.Add(Paragraph(line.Name.LocalName == "title" ? Child(line, "p") ?? line : line, new HParaFormat(Left: indent), new HCharFormat(Bold: true)));
+ continue;
+ }
+ if (stanza.Inlines.Count > 0)
+ stanza.Inlines.Add(new HLineBreak(default));
+ stanza.Inlines.AddRange(Line(line, default));
+ }
+ if (stanza.Inlines.Count > 0)
+ output.Add(stanza);
+ break;
+ }
+ case "title":
+ Title(child, output, depth + 1, center: false);
+ break;
+ default:
+ Block(child, output, depth, indent);
+ break;
+ }
+ }
+ }
+
+ private void Image(XElement image, List output, int indent)
+ {
+ if (Href(image) is not { } href || !_binaries.TryGetValue(href.TrimStart('#'), out var file))
+ return;
+ var paragraph = new HParagraph { Format = new HParaFormat(Align: HAlign.Center, Left: indent > 0 ? indent : null) };
+ paragraph.Inlines.Add(new HImage(file));
+ output.Add(paragraph);
+ }
+
+ private HTable Table(XElement table)
+ {
+ var hTable = new HTable();
+ foreach (var row in Children(table, "tr"))
+ {
+ var cells = row.Elements().Where(c => c.Name.LocalName is "td" or "th").ToList();
+ var hRow = new HRow { Header = cells.Count > 0 && cells.All(c => c.Name.LocalName == "th") };
+ foreach (var cell in cells)
+ {
+ var hCell = new HCell
+ {
+ ColSpan = Math.Max(1, (int?)cell.Attribute("colspan") ?? 1),
+ RowSpan = Math.Max(1, (int?)cell.Attribute("rowspan") ?? 1),
+ };
+ var align = ((string?)cell.Attribute("align"))?.ToLowerInvariant() switch
+ {
+ "center" => HAlign.Center,
+ "right" => HAlign.Right,
+ _ => (HAlign?)null,
+ };
+ hCell.Blocks.Add(Paragraph(cell, new HParaFormat(Align: align), cell.Name.LocalName == "th" ? new HCharFormat(Bold: true) : default));
+ hRow.Cells.Add(hCell);
+ }
+ if (hRow.Cells.Count > 0)
+ hTable.Rows.Add(hRow);
+ }
+ hTable.ColumnCount = Math.Max(1, hTable.Rows.Select(r => r.Cells.Sum(c => c.ColSpan)).DefaultIfEmpty(1).Max());
+ return hTable;
+ }
+
+ // ───────────────────────── Inlines ─────────────────────────
+
+ private HParagraph Paragraph(XElement element, HParaFormat format, HCharFormat charFormat)
+ {
+ var paragraph = new HParagraph { Format = format };
+ if ((string?)element.Attribute("id") is { Length: > 0 } id)
+ paragraph.Inlines.Add(new HBookmark(id));
+ Inlines(element, paragraph.Inlines, charFormat);
+ HtmlReader.MergeRuns(paragraph.Inlines);
+ TrimEdges(paragraph.Inlines);
+ return paragraph;
+ }
+
+ /// The inlines of one line (a title paragraph, a verse), edges trimmed.
+ private List Line(XElement element, HCharFormat format)
+ {
+ var inlines = new List();
+ Inlines(element, inlines, format);
+ HtmlReader.MergeRuns(inlines);
+ TrimEdges(inlines);
+ return inlines;
+ }
+
+ private static string CollapseWhitespace(string text)
+ {
+ var sb = new StringBuilder(text.Length);
+ foreach (var ch in text)
+ {
+ if (ch is not (' ' or '\n' or '\r' or '\t'))
+ sb.Append(ch);
+ else if (sb.Length == 0 || sb[^1] != ' ')
+ sb.Append(' ');
+ }
+ return sb.ToString();
+ }
+
+ /// Pretty-printed files indent paragraph text: spaces at the paragraph edges are dropped.
+ private static void TrimEdges(List inlines)
+ {
+ if (inlines.FindIndex(i => i is HText) is var first and >= 0 && inlines[first] is HText head)
+ inlines[first] = new HText(head.Text.TrimStart(), head.Format);
+ if (inlines.Count > 0 && inlines[^1] is HText tail)
+ inlines[^1] = new HText(tail.Text.TrimEnd(), tail.Format);
+ inlines.RemoveAll(i => i is HText { Text.Length: 0 });
+ }
+
+ private void Inlines(XElement element, List output, HCharFormat format)
+ {
+ foreach (var node in element.Nodes())
+ {
+ if (node is XText text)
+ {
+ var value = CollapseWhitespace(text.Value);
+ if (value.Length > 0)
+ output.Add(new HText(value, format));
+ continue;
+ }
+ if (node is not XElement child)
+ continue;
+ switch (child.Name.LocalName)
+ {
+ case "strong":
+ Inlines(child, output, format with { Bold = true });
+ break;
+ case "emphasis":
+ Inlines(child, output, format with { Italic = true });
+ break;
+ case "strikethrough":
+ Inlines(child, output, format with { Strike = true });
+ break;
+ case "sub":
+ Inlines(child, output, format with { Subscript = true });
+ break;
+ case "sup":
+ Inlines(child, output, format with { Superscript = true });
+ break;
+ case "code":
+ Inlines(child, output, format with { Shade = HtmlReader.CodeShade });
+ break;
+ case "image":
+ if (Href(child) is { } href && _binaries.TryGetValue(href.TrimStart('#'), out var file))
+ output.Add(new HImage(file));
+ break;
+ case "a":
+ Link(child, output, format);
+ break;
+ default:
+ Inlines(child, output, format); // style and unknown inline elements: their text
+ break;
+ }
+ }
+ }
+
+ private void Link(XElement link, List output, HCharFormat format)
+ {
+ var href = Href(link) ?? "";
+ if (href.StartsWith('#') && _notes.TryGetValue(href[1..], out var noteSection))
+ {
+ // A note reference becomes a footnote with the note's text (its title is only the number).
+ var note = new HNote(endnote: false);
+ foreach (var child in noteSection.Elements().Where(e => e.Name.LocalName != "title"))
+ Block(child, note.Blocks, 0, 0);
+ if (note.Blocks.Count > 0)
+ {
+ output.Add(note);
+ return;
+ }
+ }
+ if (href.Length == 0)
+ {
+ Inlines(link, output, format);
+ return;
+ }
+ var hLink = new HLink(href);
+ Inlines(link, hLink.Content, format);
+ output.Add(hLink);
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Fb2Writer.cs b/src/Filee.Engines/Ebooks/Fb2Writer.cs
new file mode 100644
index 0000000..54e0761
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Fb2Writer.cs
@@ -0,0 +1,396 @@
+// Book → FictionBook 2 (FB2): headings become nested sections with titles, paragraphs keep bold / italic /
+// strikethrough / sub / sup / code and links, pictures become base64 binaries (JPEG or PNG), tables stay tables,
+// footnotes go to a "notes" body. FB2 has no lists or line breaks: list markers are written out and a line break
+// starts a new paragraph.
+
+using System.Text;
+using System.Xml;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal sealed class Fb2Writer
+{
+ private const string Fb2 = "http://www.gribuser.ru/xml/fictionbook/2.0";
+ private const string XLink = "http://www.w3.org/1999/xlink";
+
+ private readonly string _imageFolder;
+ private readonly Dictionary _binaries = new(StringComparer.OrdinalIgnoreCase);
+ private readonly List> _notes = [];
+ private readonly ListNumbers _numbers = new();
+
+ private Fb2Writer(string imageFolder) => _imageFolder = imageFolder;
+
+ /// A section of the output: its title and content (paragraphs, then sub-sections).
+ private sealed class Section(HParagraph? title)
+ {
+ public HParagraph? Title { get; } = title;
+ public int Level { get; init; }
+ public List Blocks { get; } = [];
+ public List Children { get; } = [];
+ }
+
+ public static void Write(Book book, string outputPath, string sourcePath, string workFolder)
+ {
+ var writer = new Fb2Writer(EbookFiles.NewFolder(workFolder, "fb2-images"));
+ var root = writer.Tree(book.Document);
+ var settings = new XmlWriterSettings { Encoding = new UTF8Encoding(false), Indent = true, IndentChars = " " };
+ var temp = outputPath + ".tmp";
+ using (var xml = XmlWriter.Create(temp, settings))
+ {
+ xml.WriteStartDocument();
+ xml.WriteStartElement("FictionBook", Fb2);
+ xml.WriteAttributeString("xmlns", "l", null, XLink);
+ writer.Description(xml, book, sourcePath);
+
+ xml.WriteStartElement("body", Fb2);
+ foreach (var section in root.Children)
+ writer.WriteSection(xml, section);
+ if (root.Children.Count == 0)
+ {
+ xml.WriteStartElement("section", Fb2);
+ xml.WriteElementString("empty-line", Fb2, null);
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+
+ if (writer._notes.Count > 0)
+ {
+ xml.WriteStartElement("body", Fb2);
+ xml.WriteAttributeString("name", "notes");
+ for (var i = 0; i < writer._notes.Count; i++)
+ {
+ xml.WriteStartElement("section", Fb2);
+ xml.WriteAttributeString("id", $"n{i + 1}");
+ xml.WriteStartElement("title", Fb2);
+ xml.WriteElementString("p", Fb2, $"{i + 1}");
+ xml.WriteEndElement();
+ writer.Blocks(xml, writer._notes[i]);
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+ }
+
+ foreach (var (id, file, mediaType) in writer._binaries.Values)
+ {
+ xml.WriteStartElement("binary", Fb2);
+ xml.WriteAttributeString("id", id);
+ xml.WriteAttributeString("content-type", mediaType);
+ xml.WriteString(Convert.ToBase64String(File.ReadAllBytes(file), Base64FormattingOptions.InsertLineBreaks));
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+ xml.WriteEndDocument();
+ }
+ File.Move(temp, outputPath, overwrite: true);
+ }
+
+ /// Nests the blocks into sections by heading level (a level-2 heading opens a section in the level-1 one).
+ private Section Tree(HDocument document)
+ {
+ var root = new Section(null) { Level = 0 };
+ var stack = new List { root };
+ foreach (var block in document.Sections.SelectMany(s => s.Blocks))
+ {
+ if (block is HParagraph { HeadingLevel: > 0 } heading)
+ {
+ while (stack[^1].Level >= heading.HeadingLevel)
+ stack.RemoveAt(stack.Count - 1);
+ var section = new Section(heading) { Level = heading.HeadingLevel };
+ stack[^1].Children.Add(section);
+ stack.Add(section);
+ continue;
+ }
+ // Content before the first heading (or between a heading and its first sub-heading) of the root goes
+ // into an untitled section.
+ if (stack.Count == 1)
+ {
+ var untitled = new Section(null) { Level = 1 };
+ root.Children.Add(untitled);
+ stack.Add(untitled);
+ }
+ stack[^1].Blocks.Add(block);
+ }
+ return root;
+ }
+
+ private void Description(XmlWriter xml, Book book, string sourcePath)
+ {
+ xml.WriteStartElement("description", Fb2);
+ xml.WriteStartElement("title-info", Fb2);
+ xml.WriteElementString("genre", Fb2, "antique"); // required; calibre uses the same neutral default
+ var authors = book.Metadata.Authors.Count > 0 ? book.Metadata.Authors : [""];
+ foreach (var author in authors)
+ {
+ xml.WriteStartElement("author", Fb2);
+ var parts = author.Split(' ', StringSplitOptions.RemoveEmptyEntries);
+ if (parts.Length >= 2)
+ {
+ xml.WriteElementString("first-name", Fb2, string.Join(" ", parts[..^1]));
+ xml.WriteElementString("last-name", Fb2, parts[^1]);
+ }
+ else
+ {
+ xml.WriteElementString("nickname", Fb2, parts.Length == 1 ? parts[0] : "Unknown");
+ }
+ xml.WriteEndElement();
+ }
+ xml.WriteElementString("book-title", Fb2, book.DisplayTitle(sourcePath));
+ if (book.Metadata.Description is { } description)
+ {
+ xml.WriteStartElement("annotation", Fb2);
+ xml.WriteElementString("p", Fb2, Clean(description));
+ xml.WriteEndElement();
+ }
+ xml.WriteElementString("lang", Fb2, book.DisplayLanguage().Split('-')[0]);
+ if (book.Metadata.CoverImage is { } cover && Binary(cover) is { } coverId)
+ {
+ xml.WriteStartElement("coverpage", Fb2);
+ ImageElement(xml, coverId);
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+
+ xml.WriteStartElement("document-info", Fb2);
+ xml.WriteStartElement("author", Fb2);
+ xml.WriteElementString("nickname", Fb2, "Filee");
+ xml.WriteEndElement();
+ xml.WriteElementString("program-used", Fb2, "Filee");
+ xml.WriteStartElement("date", Fb2);
+ xml.WriteAttributeString("value", DateTime.Now.ToString("yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture));
+ xml.WriteString(DateTime.Now.ToString("yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture));
+ xml.WriteEndElement();
+ xml.WriteElementString("id", Fb2, Guid.NewGuid().ToString());
+ xml.WriteElementString("version", Fb2, "1.0");
+ xml.WriteEndElement();
+ xml.WriteEndElement();
+ }
+
+ private void WriteSection(XmlWriter xml, Section section)
+ {
+ xml.WriteStartElement("section", Fb2);
+ if (section.Title is { } title && Bookmark(title.Inlines) is { } id)
+ xml.WriteAttributeString("id", id);
+ if (section.Title is not null)
+ {
+ xml.WriteStartElement("title", Fb2);
+ Paragraphs(xml, section.Title.Inlines.Where(i => i is not HBookmark).ToList(), "p"); // the id is on the section
+ xml.WriteEndElement();
+ }
+ // FB2 sections hold either content or sub-sections: leading content moves into an untitled section.
+ if (section.Children.Count > 0 && section.Blocks.Count > 0)
+ {
+ xml.WriteStartElement("section", Fb2);
+ Blocks(xml, section.Blocks);
+ xml.WriteEndElement();
+ }
+ else if (section.Blocks.Count > 0)
+ {
+ Blocks(xml, section.Blocks);
+ }
+ foreach (var child in section.Children)
+ WriteSection(xml, child);
+ if (section.Blocks.Count == 0 && section.Children.Count == 0)
+ xml.WriteElementString("empty-line", Fb2, null);
+ xml.WriteEndElement();
+ }
+
+ private void Blocks(XmlWriter xml, List blocks)
+ {
+ foreach (var block in blocks)
+ {
+ switch (block)
+ {
+ case HParagraph paragraph:
+ Paragraph(xml, paragraph);
+ break;
+ case HTable table:
+ Table(xml, table);
+ break;
+ }
+ }
+ }
+
+ private void Paragraph(XmlWriter xml, HParagraph paragraph)
+ {
+ var visible = HDocumentWalker.InlinesOf(paragraph.Inlines).Any(i => i is HImage or HNote || i is HText { Text: var t } && t.Trim().Length > 0);
+ if (!visible)
+ {
+ if (paragraph.Inlines.Any(i => i is HShape { Kind: HShapeKind.Line }))
+ xml.WriteElementString("subtitle", Fb2, "* * *");
+ else
+ xml.WriteElementString("empty-line", Fb2, null);
+ return;
+ }
+ // A picture on its own line is a block image.
+ if (paragraph.Inlines.Where(i => i is not HBookmark).ToList() is [HImage only] && Binary(only.Path) is { } imageId)
+ {
+ ImageElement(xml, imageId);
+ return;
+ }
+ var inlines = paragraph.Inlines;
+ if (paragraph.List is { } list)
+ {
+ var marker = list.Numbered ? _numbers.Next(list) + " " : "";
+ inlines = [new HText(new string(' ', list.Level * 4) + marker, default), .. inlines];
+ }
+ var element = paragraph.Format.Align == HAlign.Center && paragraph.Inlines.OfType().All(t => t.Format.Bold is true) ? "subtitle" : "p";
+ Paragraphs(xml, inlines, element);
+ }
+
+ /// Writes inlines as one element per line (FB2 paragraphs have no line breaks).
+ private void Paragraphs(XmlWriter xml, List inlines, string element)
+ {
+ var lines = new List> { new() };
+ foreach (var inline in inlines)
+ {
+ if (inline is HLineBreak)
+ lines.Add([]);
+ else
+ lines[^1].Add(inline);
+ }
+ foreach (var line in lines)
+ {
+ xml.WriteStartElement(element, Fb2);
+ if (Bookmark(line) is { } id)
+ xml.WriteAttributeString("id", id);
+ Inlines(xml, line);
+ xml.WriteEndElement();
+ }
+ }
+
+ private static string? Bookmark(IEnumerable inlines) =>
+ inlines.OfType().Select(b => Id(b.Name)).FirstOrDefault();
+
+ private static string Id(string name)
+ {
+ var id = new string(name.Select(c => char.IsAsciiLetterOrDigit(c) || c is '-' or '_' or '.' ? c : '_').ToArray());
+ return id.Length > 0 && char.IsAsciiLetter(id[0]) ? id : "id-" + id;
+ }
+
+ private void Inlines(XmlWriter xml, List inlines)
+ {
+ foreach (var inline in inlines)
+ {
+ switch (inline)
+ {
+ case HText text:
+ Run(xml, text.Text, text.Format);
+ break;
+ case HTab:
+ xml.WriteString(" ");
+ break;
+ case HLink link:
+ {
+ var target = link.Target.StartsWith('#') ? "#" + Id(link.Target[1..]) : link.Target;
+ xml.WriteStartElement("a", Fb2);
+ xml.WriteAttributeString("href", XLink, target);
+ Inlines(xml, link.Content.Where(i => i is not HLineBreak).ToList());
+ xml.WriteEndElement();
+ break;
+ }
+ case HImage image:
+ if (Binary(image.Path) is { } id)
+ ImageElement(xml, id);
+ break;
+ case HNote note:
+ _notes.Add(note.Blocks);
+ xml.WriteStartElement("a", Fb2);
+ xml.WriteAttributeString("href", XLink, $"#n{_notes.Count}");
+ xml.WriteAttributeString("type", "note");
+ xml.WriteString($"[{_notes.Count}]");
+ xml.WriteEndElement();
+ break;
+ case HTextBox box:
+ Inlines(xml, box.Blocks.OfType().SelectMany(p => p.Inlines.Prepend(new HText(" ", default))).Where(i => i is not HLineBreak).ToList());
+ break;
+ }
+ }
+ }
+
+ private static void Run(XmlWriter xml, string text, HCharFormat format)
+ {
+ var open = 0;
+ void Open(string name)
+ {
+ xml.WriteStartElement(name, Fb2);
+ open++;
+ }
+ if (format.Shade == HtmlReader.CodeShade)
+ Open("code");
+ if (format.Bold is true)
+ Open("strong");
+ if (format.Italic is true)
+ Open("emphasis");
+ if (format.Strike is true)
+ Open("strikethrough");
+ if (format.Superscript is true)
+ Open("sup");
+ else if (format.Subscript is true)
+ Open("sub");
+ xml.WriteString(Clean(text));
+ for (; open > 0; open--)
+ xml.WriteEndElement();
+ }
+
+ private void Table(XmlWriter xml, HTable table)
+ {
+ xml.WriteStartElement("table", Fb2);
+ foreach (var row in table.Rows)
+ {
+ xml.WriteStartElement("tr", Fb2);
+ foreach (var cell in row.Cells)
+ {
+ xml.WriteStartElement(row.Header ? "th" : "td", Fb2);
+ if (cell.ColSpan > 1)
+ xml.WriteAttributeString("colspan", $"{cell.ColSpan}");
+ if (cell.RowSpan > 1)
+ xml.WriteAttributeString("rowspan", $"{cell.RowSpan}");
+ var paragraphs = cell.Blocks.OfType().ToList();
+ for (var i = 0; i < paragraphs.Count; i++)
+ {
+ if (i > 0)
+ xml.WriteString(" ");
+ Inlines(xml, paragraphs[i].Inlines.Where(x => x is not HLineBreak and not HImage).ToList());
+ }
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+ }
+ xml.WriteEndElement();
+ }
+
+ private static void ImageElement(XmlWriter xml, string id)
+ {
+ xml.WriteStartElement("image", Fb2);
+ xml.WriteAttributeString("href", XLink, "#" + id);
+ xml.WriteEndElement();
+ }
+
+ /// The binary id of a picture (JPEG or PNG, converted when needed), or null when it cannot be read.
+ private string? Binary(string path)
+ {
+ if (_binaries.TryGetValue(path, out var known))
+ return known.Id;
+ if (EbookFiles.CommonImage(path, _imageFolder, gifAndSvg: false) is not ({ } file, { } mediaType))
+ return null;
+ var id = $"img{_binaries.Count + 1}.{(mediaType == "image/png" ? "png" : "jpg")}";
+ _binaries[path] = (id, file, mediaType);
+ return id;
+ }
+
+ /// Removes characters XML 1.0 does not allow (e.g. vertical tabs from Word).
+ private static string Clean(string text)
+ {
+ var sb = new StringBuilder(text.Length);
+ for (var i = 0; i < text.Length; i++)
+ {
+ var c = text[i];
+ if (char.IsHighSurrogate(c) && i + 1 < text.Length && char.IsLowSurrogate(text[i + 1]))
+ sb.Append(c).Append(text[++i]);
+ else if (!char.IsSurrogate(c) && (c is '\t' or '\n' or '\r' || (c >= 0x20 && c != 0xFFFE && c != 0xFFFF)))
+ sb.Append(c);
+ }
+ return sb.ToString();
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs b/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs
new file mode 100644
index 0000000..d80ca56
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Mobi/HuffCdic.cs
@@ -0,0 +1,128 @@
+// HUFF/CDIC compression of MOBI text records (Mobipocket "Huffman dictionary" compression), written from the
+// format description on the MobileRead wiki: a canonical Huffman code (HUFF record) selects phrases from the
+// dictionary (CDIC records); a phrase may itself be compressed and is unpacked on first use.
+
+using System.Buffers.Binary;
+
+namespace Filee.Engines.Ebooks;
+
+internal sealed class HuffCdic
+{
+ /// Per leading byte of a code: length, whether that length is final, and the adjusted max code.
+ private readonly (int Length, bool Terminal, ulong MaxCode)[] _table1 = new (int, bool, ulong)[256];
+ private readonly ulong[] _minCode = new ulong[33];
+ private readonly ulong[] _maxCode = new ulong[33];
+ private readonly List _phrases = [];
+
+ private sealed class Phrase(byte[] data, bool unpacked)
+ {
+ public byte[] Data { get; set; } = data;
+ public bool Unpacked { get; set; } = unpacked;
+ public bool Busy { get; set; }
+ }
+
+ /// The HUFF record.
+ /// The CDIC records that follow it.
+ public HuffCdic(ReadOnlySpan huff, IEnumerable cdics)
+ {
+ if (huff.Length < 24 || !huff[..4].SequenceEqual("HUFF"u8))
+ throw new InvalidDataException("The book's HUFF compression table is damaged.");
+ var table1 = (int)BinaryPrimitives.ReadUInt32BigEndian(huff[8..]);
+ var table2 = (int)BinaryPrimitives.ReadUInt32BigEndian(huff[12..]);
+ if (table1 + 256 * 4 > huff.Length || table2 + 64 * 4 > huff.Length)
+ throw new InvalidDataException("The book's HUFF compression table is damaged.");
+
+ for (var i = 0; i < 256; i++)
+ {
+ var value = BinaryPrimitives.ReadUInt32BigEndian(huff[(table1 + i * 4)..]);
+ var length = (int)(value & 0x1F);
+ if (length == 0)
+ throw new InvalidDataException("The book's HUFF compression table is damaged.");
+ var maxCode = (((ulong)(value >> 8) + 1) << (32 - length)) - 1;
+ _table1[i] = (length, (value & 0x80) != 0, maxCode);
+ }
+ // Code lengths 1..32: the smallest and largest code of each length, left-aligned in 32 bits.
+ for (var length = 1; length <= 32; length++)
+ {
+ var min = BinaryPrimitives.ReadUInt32BigEndian(huff[(table2 + (length - 1) * 8)..]);
+ var max = BinaryPrimitives.ReadUInt32BigEndian(huff[(table2 + (length - 1) * 8 + 4)..]);
+ _minCode[length] = (ulong)min << (32 - length);
+ _maxCode[length] = (((ulong)max + 1) << (32 - length)) - 1;
+ }
+
+ foreach (var cdic in cdics)
+ {
+ if (cdic.Length < 16 || !cdic.AsSpan(0, 4).SequenceEqual("CDIC"u8))
+ throw new InvalidDataException("The book's CDIC dictionary is damaged.");
+ var total = (int)BinaryPrimitives.ReadUInt32BigEndian(cdic.AsSpan(8));
+ var bits = (int)BinaryPrimitives.ReadUInt32BigEndian(cdic.AsSpan(12));
+ var count = Math.Min(1 << Math.Clamp(bits, 0, 20), total - _phrases.Count);
+ for (var i = 0; i < count; i++)
+ {
+ var offset = 16 + PalmDatabase.U16(cdic, 16 + i * 2);
+ var header = PalmDatabase.U16(cdic, offset);
+ var length = Math.Min(header & 0x7FFF, Math.Max(0, cdic.Length - offset - 2));
+ _phrases.Add(new Phrase(cdic.AsSpan(offset + 2, length).ToArray(), (header & 0x8000) != 0));
+ }
+ }
+ }
+
+ /// Decompresses one text record.
+ public byte[] Decompress(ReadOnlySpan data)
+ {
+ var output = new List(data.Length * 3);
+ Unpack(data, output, 0);
+ return [.. output];
+ }
+
+ private void Unpack(ReadOnlySpan data, List output, int depth)
+ {
+ if (depth > 32)
+ throw new InvalidDataException("The book's CDIC dictionary refers to itself.");
+ // Read 64 bits at a time; "n" counts the bits of the current 32-bit window not yet consumed.
+ Span padded = new byte[data.Length + 8];
+ data.CopyTo(padded);
+ long bitsLeft = data.Length * 8L;
+ var position = 0;
+ var window = BinaryPrimitives.ReadUInt64BigEndian(padded);
+ var n = 32;
+ while (true)
+ {
+ if (n <= 0)
+ {
+ position += 4;
+ window = position + 8 <= padded.Length ? BinaryPrimitives.ReadUInt64BigEndian(padded[position..]) : 0;
+ n += 32;
+ }
+ var code = (window >> n) & 0xFFFFFFFF;
+ var (length, terminal, maxCode) = _table1[code >> 24];
+ if (!terminal)
+ {
+ while (length < 32 && code < _minCode[length])
+ length++;
+ maxCode = _maxCode[length];
+ }
+ n -= length;
+ bitsLeft -= length;
+ if (bitsLeft < 0)
+ break;
+
+ var index = (int)((maxCode - code) >> (32 - length));
+ if (index < 0 || index >= _phrases.Count)
+ throw new InvalidDataException("The book's compressed text is damaged.");
+ var phrase = _phrases[index];
+ if (!phrase.Unpacked)
+ {
+ if (phrase.Busy)
+ throw new InvalidDataException("The book's CDIC dictionary refers to itself.");
+ phrase.Busy = true;
+ var unpacked = new List();
+ Unpack(phrase.Data, unpacked, depth + 1);
+ phrase.Data = [.. unpacked];
+ phrase.Unpacked = true;
+ phrase.Busy = false;
+ }
+ output.AddRange(phrase.Data);
+ }
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs b/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs
new file mode 100644
index 0000000..442ba28
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Mobi/MobiIndex.cs
@@ -0,0 +1,180 @@
+// INDX records of Kindle books: the skeleton and fragment tables of KF8 (AZW3) text are stored as indexes. Written
+// from the format description on the MobileRead wiki: a primary INDX record with the tag table (TAGX), then INDX
+// records whose entries (located through the IDXT offset table) are a key followed by control bytes and
+// variable-width tag values; strings live in CNCX records.
+
+using System.Text;
+
+namespace Filee.Engines.Ebooks;
+
+/// One index entry: its key and tag values.
+internal sealed record MobiIndexEntry(string Key, IReadOnlyDictionary> Tags)
+{
+ public long Tag(int tag, int position = 0) =>
+ Tags.TryGetValue(tag, out var values) && position < values.Count ? values[position] : -1;
+}
+
+internal static class MobiIndex
+{
+ private readonly record struct TagDefinition(int Tag, int ValuesPerEntry, int Mask, bool EndFlag);
+
+ /// Reads the index starting at record ; CNCX strings go to .
+ public static List Read(PalmDatabase database, int first, Dictionary strings)
+ {
+ var entries = new List();
+ var primary = database.Record(first);
+ if (primary.Length < 56 || !primary[..4].SequenceEqual("INDX"u8))
+ throw new InvalidDataException("The book's index is damaged.");
+ var headerLength = (int)PalmDatabase.U32(primary, 4);
+ var recordCount = (int)PalmDatabase.U32(primary, 24);
+ var cncxCount = (int)Math.Max(0, PalmDatabase.U32(primary, 52));
+ var (controlBytes, tags) = TagTable(primary, headerLength);
+
+ // CNCX records follow the index records; offsets in tag values address them as record * 0x10000 + offset.
+ for (var c = 0; c < cncxCount; c++)
+ {
+ var cncx = database.Record(first + recordCount + 1 + c);
+ var offset = 0;
+ while (offset < cncx.Length && cncx[offset] != 0)
+ {
+ var start = offset;
+ var (length, consumed) = ForwardVarint(cncx, offset);
+ offset += consumed;
+ length = Math.Min(length, cncx.Length - offset);
+ strings[c * 0x10000L + start] = Encoding.UTF8.GetString(cncx.Slice(offset, (int)length));
+ offset += (int)length;
+ }
+ }
+
+ for (var r = 1; r <= recordCount; r++)
+ {
+ var record = database.Record(first + r);
+ if (record.Length < 28 || !record[..4].SequenceEqual("INDX"u8))
+ continue;
+ var idxt = (int)PalmDatabase.U32(record, 20);
+ var count = (int)PalmDatabase.U32(record, 24);
+ if (idxt < 0 || idxt + 4 + count * 2 > record.Length)
+ continue;
+ var offsets = new int[count + 1];
+ for (var i = 0; i < count; i++)
+ offsets[i] = PalmDatabase.U16(record, idxt + 4 + i * 2);
+ offsets[count] = idxt;
+ for (var i = 0; i < count; i++)
+ {
+ var start = offsets[i];
+ if (start >= record.Length)
+ continue;
+ var keyLength = record[start];
+ var key = Encoding.Latin1.GetString(record.Slice(start + 1, Math.Min(keyLength, record.Length - start - 1)));
+ var values = TagValues(record, start + 1 + keyLength, offsets[i + 1], controlBytes, tags);
+ entries.Add(new MobiIndexEntry(key, values));
+ }
+ }
+ return entries;
+ }
+
+ private static (int ControlBytes, List Tags) TagTable(ReadOnlySpan record, int start)
+ {
+ var tags = new List();
+ if (start < 0 || start + 12 > record.Length || !record.Slice(start, 4).SequenceEqual("TAGX"u8))
+ throw new InvalidDataException("The book's index has no tag table.");
+ var length = (int)PalmDatabase.U32(record, start + 4);
+ var controlBytes = (int)PalmDatabase.U32(record, start + 8);
+ for (var i = 12; i + 4 <= length && start + i + 4 <= record.Length; i += 4)
+ {
+ var p = start + i;
+ tags.Add(new TagDefinition(record[p], record[p + 1], record[p + 2], record[p + 3] == 1));
+ }
+ return (controlBytes, tags);
+ }
+
+ ///
+ /// Tag values of one entry. Each tag's mask selects bits of a control byte: a partial value is the number of
+ /// value groups; all bits set with a multi-bit mask means a byte count follows; a one-bit mask means one group.
+ ///
+ private static Dictionary> TagValues(ReadOnlySpan record, int start, int end, int controlBytes, List tags)
+ {
+ var result = new Dictionary>();
+ var controlIndex = 0;
+ var position = start + controlBytes;
+ var headers = new List<(int Tag, int? Count, int? Bytes, int PerEntry)>();
+ foreach (var tag in tags)
+ {
+ if (tag.EndFlag)
+ {
+ controlIndex++;
+ continue;
+ }
+ if (start + controlIndex >= record.Length)
+ break;
+ var value = record[start + controlIndex] & tag.Mask;
+ if (value == 0)
+ continue;
+ if (value == tag.Mask)
+ {
+ if (System.Numerics.BitOperations.PopCount((uint)tag.Mask) > 1)
+ {
+ var (bytes, consumed) = ForwardVarint(record, position);
+ position += consumed;
+ headers.Add((tag.Tag, null, (int)bytes, tag.ValuesPerEntry));
+ }
+ else
+ {
+ headers.Add((tag.Tag, 1, null, tag.ValuesPerEntry));
+ }
+ }
+ else
+ {
+ var mask = tag.Mask;
+ while ((mask & 1) == 0)
+ {
+ mask >>= 1;
+ value >>= 1;
+ }
+ headers.Add((tag.Tag, value, null, tag.ValuesPerEntry));
+ }
+ }
+
+ foreach (var (tag, count, bytes, perEntry) in headers)
+ {
+ var values = new List();
+ if (count is { } groups)
+ {
+ for (var i = 0; i < groups * perEntry && position < end; i++)
+ {
+ var (value, consumed) = ForwardVarint(record, position);
+ position += consumed;
+ values.Add(value);
+ }
+ }
+ else
+ {
+ var read = 0;
+ while (read < bytes && position < end)
+ {
+ var (value, consumed) = ForwardVarint(record, position);
+ position += consumed;
+ read += consumed;
+ values.Add(value);
+ }
+ }
+ result[tag] = values;
+ }
+ return result;
+ }
+
+ /// A forward variable-width integer: 7 bits per byte, the last byte has the high bit set.
+ internal static (long Value, int Consumed) ForwardVarint(ReadOnlySpan data, int offset)
+ {
+ long value = 0;
+ var consumed = 0;
+ while (offset + consumed < data.Length && consumed < 9)
+ {
+ var b = data[offset + consumed++];
+ value = (value << 7) | (uint)(b & 0x7F);
+ if ((b & 0x80) != 0)
+ break;
+ }
+ return (value, Math.Max(1, consumed));
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs b/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs
new file mode 100644
index 0000000..0807e32
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Mobi/MobiReader.cs
@@ -0,0 +1,616 @@
+// MOBI / AZW / AZW3 / PRC → Book, and the PDF inside AZW4 (Print Replica). Written from the MOBI format description
+// on the MobileRead wiki (no code from KindleUnpack or calibre):
+// * PalmDB records; record 0 holds the PalmDOC header, the MOBI header and EXTH metadata (title, authors, cover);
+// * text records are PalmDOC (LZ77) or HUFF/CDIC compressed, with trailing entries stripped first;
+// * MOBI 6 (KF7): one HTML document; filepos links become anchors, recindex pictures come from image records;
+// * KF8 (AZW3, also the KF8 half of joint MOBI files): the text is cut into files by the skeleton (SKEL) and
+// fragment (FRAG) indexes; kindle:pos, kindle:embed and kindle:flow references are resolved;
+// * plain PalmDOC books (TEXtREAd) are text.
+// Books with DRM are refused with a clear message; KFX and Topaz books are recognised and refused too.
+
+using System.Buffers.Binary;
+using System.Text;
+using System.Text.RegularExpressions;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal static partial class MobiReader
+{
+ /// Reads a Kindle book's content (for the reader registry of the HWPX writer).
+ public static HDocument Read(string path, string workFolder) => ReadBook(path, workFolder).Document;
+
+ public static Book ReadBook(string path, string workFolder, CancellationToken cancellationToken = default)
+ {
+ var database = Open(path);
+ var folder = EbookFiles.NewFolder(workFolder, "mobi");
+ if (database.TypeCreator == "TEXtREAd")
+ return new Book(PlainText.ToDocument(DecodeText(PalmDocText(database))), new BookMetadata { Title = database.Name });
+
+ var header = MobiHeader.Parse(database, 0);
+ if (header.Encryption != 0)
+ throw new InvalidOperationException(EbookFiles.DrmMessage);
+ var metadata = header.Metadata(database.Name);
+
+ HDocument document;
+ Resources resources;
+ var kf8Start = header.Version >= 8 ? 0 : header.Kf8Boundary;
+ if (kf8Start is { } start && start >= 0 && start < database.Count)
+ {
+ var kf8 = start == 0 ? header : MobiHeader.Parse(database, start);
+ if (kf8.Encryption != 0)
+ throw new InvalidOperationException(EbookFiles.DrmMessage);
+ // Joint files keep the pictures once, before the KF8 half, numbered from the MOBI 6 header.
+ resources = new Resources(database, start == 0 ? kf8.FirstImage : header.FirstImage, folder);
+ document = Kf8(database, kf8, start, resources, folder, cancellationToken);
+ }
+ else
+ {
+ resources = new Resources(database, header.FirstImage, folder);
+ document = Mobi6(database, header, resources, folder, cancellationToken);
+ }
+
+ if (header.CoverOffset is { } cover && resources.Save(cover) is { } coverFile)
+ metadata.CoverImage = coverFile;
+ document.Title = metadata.Title;
+ return new Book(document, metadata);
+ }
+
+ /// The first PDF of an AZW4 (Print Replica) book.
+ public static byte[] PrintReplicaPdf(string path)
+ {
+ var database = Open(path);
+ var header = MobiHeader.Parse(database, 0);
+ if (header.Encryption != 0)
+ throw new InvalidOperationException(EbookFiles.DrmMessage);
+ var text = Text(database, header, 0);
+ if (text.AsSpan().StartsWith("%MOP"u8) && text.Length >= 12)
+ {
+ // "%MOP", table count, section count per table, then (offset, length) per section; the first is the PDF.
+ var tables = BinaryPrimitives.ReadUInt32BigEndian(text.AsSpan(4));
+ var index = 8 + 4 * (int)Math.Min(tables, 1000);
+ var offset = (int)PalmDatabase.U32(text, index);
+ var length = (int)PalmDatabase.U32(text, index + 4);
+ if (offset > 0 && length > 0 && offset + length <= text.Length)
+ return text.AsSpan(offset, length).ToArray();
+ }
+ var pdf = text.AsSpan().IndexOf("%PDF-"u8);
+ if (pdf < 0)
+ throw new InvalidDataException("This book is not a Print Replica (AZW4) book: it contains no PDF.");
+ return text.AsSpan(pdf).ToArray();
+ }
+
+ private static PalmDatabase Open(string path)
+ {
+ var data = File.ReadAllBytes(path);
+ if (data is [(byte)'C', (byte)'O', (byte)'N', (byte)'T', ..] or [0xEA, (byte)'D', (byte)'R', (byte)'M', (byte)'I', (byte)'O', (byte)'N', ..])
+ throw new InvalidDataException("This is a KFX book (Kindle's newest format), which cannot be converted. Download it from Amazon as an AZW3 or MOBI file (\"Download & transfer via USB\").");
+ if (data.AsSpan().StartsWith("TPZ"u8))
+ throw new InvalidDataException("This is a Topaz book, which cannot be converted.");
+ var database = PalmDatabase.Open(data);
+ if (database.TypeCreator is not ("BOOKMOBI" or "TEXtREAd"))
+ throw new InvalidDataException($"This is not a Mobipocket or Kindle book (type \"{database.TypeCreator}\").");
+ return database;
+ }
+
+ // ───────────────────────── Text records ─────────────────────────
+
+ /// The decompressed text of the book whose header is at record .
+ private static byte[] Text(PalmDatabase database, MobiHeader header, int start)
+ {
+ Func, byte[]> decompress = header.Compression switch
+ {
+ 1 => data => data.ToArray(),
+ 2 => PalmDocCompression.Decompress,
+ 17480 => HuffCdicReader(database, header, start).Decompress,
+ var other => throw new InvalidDataException($"The book uses an unknown compression ({other})."),
+ };
+ using var text = new MemoryStream();
+ for (var i = 1; i <= header.TextRecords; i++)
+ {
+ var record = database.Record(start + i);
+ var size = record.Length - TrailingSize(record, header.ExtraFlags);
+ text.Write(decompress(record[..Math.Max(0, size)]));
+ }
+ var bytes = text.ToArray();
+ return header.TextLength > 0 && header.TextLength < bytes.Length ? bytes[..(int)header.TextLength] : bytes;
+ }
+
+ private static HuffCdic HuffCdicReader(PalmDatabase database, MobiHeader header, int start)
+ {
+ var huff = start + header.HuffRecord;
+ return new HuffCdic(database.Record(huff), Enumerable.Range(huff + 1, Math.Max(0, header.HuffCount - 1)).Select(database.RecordArray));
+ }
+
+ ///
+ /// Size of the extra data at the end of a text record (multibyte overlap and indexing entries), described by the
+ /// header's extra flags: each set bit above bit 0 adds an entry whose size is a backward varint at the end.
+ ///
+ internal static int TrailingSize(ReadOnlySpan record, int flags)
+ {
+ var size = 0;
+ for (var bits = flags >> 1; bits != 0; bits >>= 1)
+ {
+ if ((bits & 1) == 0)
+ continue;
+ // Backward varint: 7 bits per byte read from the end, the first byte (from the start) has the high bit set.
+ var value = 0;
+ var shift = 0;
+ for (var p = record.Length - size - 1; p >= 0; p--)
+ {
+ var b = record[p];
+ value |= (b & 0x7F) << shift;
+ shift += 7;
+ if ((b & 0x80) != 0 || shift >= 28 || p == 0)
+ break;
+ }
+ size += value;
+ }
+ if ((flags & 1) != 0 && record.Length - size - 1 >= 0)
+ size += (record[record.Length - size - 1] & 0x3) + 1;
+ return Math.Clamp(size, 0, record.Length);
+ }
+
+ private static byte[] PalmDocText(PalmDatabase database)
+ {
+ var header = database.Record(0);
+ var compression = PalmDatabase.U16(header, 0);
+ var records = PalmDatabase.U16(header, 8);
+ using var text = new MemoryStream();
+ for (var i = 1; i <= records && i < database.Count; i++)
+ text.Write(compression == 2 ? PalmDocCompression.Decompress(database.Record(i)) : database.Record(i));
+ return text.ToArray();
+ }
+
+ /// UTF-8 when valid, else Windows-1252 (the Mobipocket default).
+ private static string DecodeText(byte[] bytes)
+ {
+ try
+ {
+ return new UTF8Encoding(false, true).GetString(bytes);
+ }
+ catch (DecoderFallbackException)
+ {
+ Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
+ return Encoding.GetEncoding(1252).GetString(bytes);
+ }
+ }
+
+ // ───────────────────────── MOBI 6 ─────────────────────────
+
+ private static HDocument Mobi6(PalmDatabase database, MobiHeader header, Resources resources, string folder, CancellationToken cancellationToken)
+ {
+ // Latin-1 keeps one character per byte, so filepos values (byte offsets) index the string directly.
+ var text = Encoding.Latin1.GetString(Text(database, header, 0));
+ cancellationToken.ThrowIfCancellationRequested();
+
+ var positions = FilePos().Matches(text).Select(m => long.Parse(m.Groups[1].Value)).Where(p => p <= text.Length).Distinct().OrderDescending();
+ var sb = new StringBuilder(text);
+ foreach (var position in positions)
+ {
+ // A position inside a tag moves to the start of that tag.
+ var at = (int)position;
+ var open = at > 0 ? text.LastIndexOf('<', at - 1) : -1;
+ var close = at > 0 ? text.LastIndexOf('>', at - 1) : -1;
+ if (open > close)
+ at = open;
+ sb.Insert(at, $"");
+ }
+ var html = FilePos().Replace(sb.ToString(), m => $"href=\"#filepos{long.Parse(m.Groups[1].Value)}\"");
+ html = RecIndex().Replace(html, m => resources.Save(int.Parse(m.Groups[1].Value) - 1) is { } file ? $"src=\"{Path.GetFileName(file)}\"" : "");
+
+ var encoding = header.Codepage == 65001 ? Encoding.UTF8 : CodePage(header.Codepage);
+ var page = Path.Combine(folder, "book.html");
+ File.WriteAllText(page, encoding.GetString(Encoding.Latin1.GetBytes(html)), new UTF8Encoding(true));
+ return ChapterReader.Read([page], Path.Combine(folder, "~media"), cancellationToken);
+ }
+
+ private static Encoding CodePage(int codepage)
+ {
+ Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
+ try
+ {
+ return Encoding.GetEncoding(codepage is 0 ? 1252 : codepage);
+ }
+ catch (Exception ex) when (ex is ArgumentException or NotSupportedException)
+ {
+ return Encoding.GetEncoding(1252);
+ }
+ }
+
+ // Digit counts are capped so the numbers always fit (filepos values are 10 digits, recindex 5).
+ [GeneratedRegex(@"\bfilepos\s*=\s*['""]?0*(\d{1,12})['""]?", RegexOptions.IgnoreCase)]
+ private static partial Regex FilePos();
+
+ [GeneratedRegex(@"\b(?:hi|lo)?recindex\s*=\s*['""]?(\d{1,9})['""]?", RegexOptions.IgnoreCase)]
+ private static partial Regex RecIndex();
+
+ // ───────────────────────── KF8 ─────────────────────────
+
+ private static HDocument Kf8(PalmDatabase database, MobiHeader header, int start, Resources resources, string folder, CancellationToken cancellationToken)
+ {
+ var text = Text(database, header, start);
+ cancellationToken.ThrowIfCancellationRequested();
+
+ // Flows: flow 0 is the XHTML text, the others are style sheets and SVG images.
+ var flows = new List();
+ if (header.Fdst >= 0 && database.Record(start + (int)header.Fdst) is var fdst && fdst.Length >= 12 && fdst[..4].SequenceEqual("FDST"u8))
+ {
+ var tableOffset = (int)PalmDatabase.U32(fdst, 4);
+ var count = (int)PalmDatabase.U32(fdst, 8);
+ for (var i = 0; i < count; i++)
+ {
+ var from = (int)Math.Clamp(PalmDatabase.U32(fdst, tableOffset + i * 8), 0, text.Length);
+ var to = (int)Math.Clamp(PalmDatabase.U32(fdst, tableOffset + i * 8 + 4), from, text.Length);
+ flows.Add(text[from..to]);
+ }
+ }
+ if (flows.Count == 0)
+ flows.Add(text);
+ var markup = flows[0];
+
+ var strings = new Dictionary();
+ var skeletons = header.Skeleton >= 0 ? MobiIndex.Read(database, start + (int)header.Skeleton, strings) : [];
+ var fragments = header.Fragment >= 0 ? MobiIndex.Read(database, start + (int)header.Fragment, strings) : [];
+ var parts = Assemble(markup, skeletons, fragments, strings);
+
+ // Work on Latin-1 strings: positions in kindle:pos links are byte offsets.
+ var texts = parts.Select(p => Encoding.Latin1.GetString(p.Text)).ToList();
+ var insertPositions = fragments.Select(f => long.TryParse(f.Key, out var p) ? p : 0).ToList();
+ var linkedAids = new HashSet(StringComparer.Ordinal);
+ string Resolve(Match m)
+ {
+ var fragment = (int)Base32(m.Groups[1].Value);
+ var position = (fragment < insertPositions.Count ? insertPositions[fragment] : 0) + Base32(m.Groups[2].Value);
+ var part = parts.FindIndex(p => position >= p.Start && position < p.Start + p.Text.Length);
+ if (part < 0)
+ return PartName(0);
+ var id = IdBefore(texts[part], (int)(position - parts[part].Start), linkedAids);
+ return id.Length > 0 ? $"{PartName(part)}#{id}" : PartName(part);
+ }
+
+ var files = new List();
+ for (var i = 0; i < texts.Count; i++)
+ {
+ var xhtml = KindlePos().Replace(texts[i], Resolve);
+ xhtml = KindleEmbed().Replace(xhtml, m => resources.Save((int)Base32(m.Groups[1].Value) - 1) is { } file ? Path.GetFileName(file) : "missing");
+ xhtml = KindleFlow().Replace(xhtml, m => Flow(flows, (int)Base32(m.Groups[1].Value), m.Groups[2].Value, folder));
+ texts[i] = xhtml;
+ }
+ for (var i = 0; i < texts.Count; i++)
+ {
+ // Links that point at an element by its Amazon "aid" get an id to land on.
+ var xhtml = linkedAids.Count == 0 ? texts[i] : Aid().Replace(texts[i], m => linkedAids.Contains(m.Groups[1].Value) ? $"{m.Value} id=\"aid-{m.Groups[1].Value}\"" : m.Value);
+ var file = Path.Combine(folder, PartName(i));
+ File.WriteAllBytes(file, [.. Encoding.UTF8.Preamble, .. Encoding.Latin1.GetBytes(xhtml)]);
+ files.Add(file);
+ }
+ return ChapterReader.Read(files, Path.Combine(folder, "~media"), cancellationToken);
+ }
+
+ private static string PartName(int index) => $"part{index:0000}.xhtml";
+
+ ///
+ /// Rebuilds the XHTML files: each skeleton is followed in the text by its fragments, which are inserted into it
+ /// at their insert positions (positions in the rebuilt file, so the skeleton's own start is subtracted).
+ ///
+ private static List<(byte[] Text, long Start)> Assemble(byte[] markup, List skeletons, List fragments, Dictionary strings)
+ {
+ if (skeletons.Count == 0)
+ return [(markup, 0)];
+ var parts = new List<(byte[], long)>();
+ var next = 0;
+ foreach (var skeleton in skeletons)
+ {
+ var skeletonStart = Math.Clamp(skeleton.Tag(6, 0), 0, markup.Length);
+ var skeletonLength = Math.Clamp(skeleton.Tag(6, 1), 0, markup.Length - skeletonStart);
+ var part = new List(markup.AsSpan((int)skeletonStart, (int)skeletonLength).ToArray());
+ var position = skeletonStart + skeletonLength;
+ var count = Math.Max(0, skeleton.Tag(1));
+ for (var i = 0; i < count && next < fragments.Count; i++, next++)
+ {
+ var fragment = fragments[next];
+ var length = (int)Math.Clamp(fragment.Tag(6, 1), 0, markup.Length - position);
+ var slice = markup.AsSpan((int)position, length).ToArray();
+ position += length;
+ var insert = (int)Math.Clamp((long.TryParse(fragment.Key, out var p) ? p : 0) - skeletonStart, 0, part.Count);
+ if (InsideTag(part, insert))
+ insert = AfterAidTag(part, strings.GetValueOrDefault(fragment.Tag(2), "")) ?? NextTagEnd(part, insert);
+ part.InsertRange(insert, slice);
+ }
+ parts.Add(([.. part], skeletonStart));
+ }
+ return parts;
+ }
+
+ private static bool InsideTag(List text, int position)
+ {
+ for (var i = position - 1; i >= 0; i--)
+ {
+ if (text[i] == '>')
+ return false;
+ if (text[i] == '<')
+ return true;
+ }
+ return false;
+ }
+
+ private static int NextTagEnd(List text, int position)
+ {
+ var end = text.IndexOf((byte)'>', position);
+ return end < 0 ? text.Count : end + 1;
+ }
+
+ /// Insert position after the tag named by a fragment selector such as P-//*[@aid='3'].
+ private static int? AfterAidTag(List text, string selector)
+ {
+ var match = SelectorAid().Match(selector);
+ if (!match.Success)
+ return null;
+ var needle = Encoding.ASCII.GetBytes($"aid=\"{match.Groups[1].Value}\"");
+ var at = text.ToArray().AsSpan().IndexOf(needle);
+ return at < 0 ? null : NextTagEnd(text, at);
+ }
+
+ ///
+ /// The id a link to lands on: the nearest id or name attribute of a tag before it
+ /// (an "aid" is noted so an id can be added), or "" for the top of the file.
+ ///
+ private static string IdBefore(string text, int position, HashSet linkedAids)
+ {
+ position = Math.Clamp(position, 0, text.Length);
+ var open = text.IndexOf('<', position);
+ var close = text.IndexOf('>', position);
+ if (close >= 0 && (open == position || open < 0 || close < open))
+ position = close + 1; // inside a tag (or at its start): that tag counts
+ var end = position;
+ while (end > 0)
+ {
+ var tagEnd = text.LastIndexOf('>', end - 1);
+ if (tagEnd < 0)
+ break;
+ var tagStart = text.LastIndexOf('<', tagEnd);
+ if (tagStart < 0)
+ break;
+ var tag = text[tagStart..(tagEnd + 1)];
+ end = tagStart;
+ if (tag.StartsWith(" flows, int index, string mime, string folder)
+ {
+ if (index <= 0 || index >= flows.Count)
+ return "missing";
+ var extension = mime.Contains("svg", StringComparison.OrdinalIgnoreCase) ? "svg" : mime.Contains("css", StringComparison.OrdinalIgnoreCase) ? "css" : "txt";
+ var name = $"flow{index:0000}.{extension}";
+ var file = Path.Combine(folder, name);
+ if (!File.Exists(file))
+ File.WriteAllBytes(file, flows[index]);
+ return name;
+ }
+
+ /// Kindle's base-32 numbers: digits 0-9 then A-V.
+ internal static long Base32(string text)
+ {
+ long value = 0;
+ foreach (var ch in text.ToUpperInvariant())
+ value = value * 32 + (ch <= '9' ? ch - '0' : ch - 'A' + 10);
+ return value;
+ }
+
+ [GeneratedRegex(@"kindle:pos:fid:([0-9A-Va-v]{4}):off:([0-9A-Va-v]{10})")]
+ private static partial Regex KindlePos();
+
+ [GeneratedRegex(@"kindle:embed:([0-9A-Va-v]{4})(?:\?mime=[^'""\)\s]*)?")]
+ private static partial Regex KindleEmbed();
+
+ [GeneratedRegex(@"kindle:flow:([0-9A-Va-v]{4})(?:\?mime=([^'""\)\s]*))?")]
+ private static partial Regex KindleFlow();
+
+ [GeneratedRegex(@"\said\s*=\s*['""]([^'""]+)['""]")]
+ private static partial Regex Aid();
+
+ [GeneratedRegex(@"^<[^>]*\s(?:id|name)\s*=\s*['""]([^'""]*)['""]", RegexOptions.IgnoreCase)]
+ private static partial Regex IdAttribute();
+
+ [GeneratedRegex(@"^<[^>]+\said\s*=\s*['""]([^'""]+)['""]", RegexOptions.IgnoreCase)]
+ private static partial Regex AidAttribute();
+
+ [GeneratedRegex(@"@aid='([^']+)'")]
+ private static partial Regex SelectorAid();
+
+ // ───────────────────────── Pictures ─────────────────────────
+
+ /// Picture records, numbered from the first image record; written to files on first use.
+ private sealed class Resources(PalmDatabase database, long firstImage, string folder)
+ {
+ private readonly Dictionary _saved = [];
+
+ /// Saves resource (0-based) and returns its file, or null if it is no picture.
+ public string? Save(int index)
+ {
+ if (firstImage < 0 || index < 0)
+ return null;
+ if (_saved.TryGetValue(index, out var known))
+ return known;
+ var record = database.Record((int)firstImage + index);
+ var extension = record switch
+ {
+ [0xFF, 0xD8, 0xFF, ..] => "jpg",
+ [0x89, (byte)'P', (byte)'N', (byte)'G', ..] => "png",
+ [(byte)'G', (byte)'I', (byte)'F', (byte)'8', ..] => "gif",
+ [(byte)'B', (byte)'M', ..] => "bmp",
+ _ => null,
+ };
+ string? file = null;
+ if (extension is not null)
+ {
+ file = Path.Combine(folder, $"image{index + 1:00000}.{extension}");
+ File.WriteAllBytes(file, record.ToArray());
+ }
+ _saved[index] = file;
+ return file;
+ }
+ }
+}
+
+/// The MOBI header of record 0 (or of the KF8 half of a joint file) and its EXTH metadata.
+internal sealed class MobiHeader
+{
+ public int Compression { get; private init; }
+ public long TextLength { get; private init; }
+ public int TextRecords { get; private init; }
+ public int Encryption { get; private init; }
+ public int Codepage { get; private init; } = 1252;
+ public long Version { get; private init; }
+ public long FirstImage { get; private init; } = -1;
+ public int HuffRecord { get; private init; }
+ public int HuffCount { get; private init; }
+ public int ExtraFlags { get; private init; }
+ public long Fdst { get; private init; } = -1;
+ public long Fragment { get; private init; } = -1;
+ public long Skeleton { get; private init; } = -1;
+
+ /// Record of the KF8 header in a joint MOBI 6 + KF8 file (EXTH 121).
+ public int? Kf8Boundary { get; private set; }
+
+ /// Cover picture as an offset from the first image record (EXTH 201).
+ public int? CoverOffset { get; private set; }
+
+ private string? _fullName;
+ private int _locale;
+ private readonly Dictionary> _exth = [];
+
+ public static MobiHeader Parse(PalmDatabase database, int record)
+ {
+ var data = database.Record(record);
+ if (data.Length < 16)
+ throw new InvalidDataException("The book's header is damaged.");
+ var hasMobi = data.Length >= 24 && data[16..20].SequenceEqual("MOBI"u8);
+ var length = hasMobi ? (int)PalmDatabase.U32(data, 20) : 0;
+ var header = new MobiHeader
+ {
+ Compression = PalmDatabase.U16(data, 0),
+ TextLength = PalmDatabase.U32(data, 4),
+ TextRecords = PalmDatabase.U16(data, 8),
+ Encryption = PalmDatabase.U16(data, 12),
+ Codepage = hasMobi ? (int)PalmDatabase.U32(data, 28) : 1252,
+ Version = hasMobi ? PalmDatabase.U32(data, 36) : 0,
+ FirstImage = hasMobi ? PalmDatabase.U32(data, 108) : -1,
+ HuffRecord = hasMobi ? (int)Math.Max(0, PalmDatabase.U32(data, 112)) : 0,
+ HuffCount = hasMobi ? (int)Math.Max(0, PalmDatabase.U32(data, 116)) : 0,
+ ExtraFlags = hasMobi && length >= 0xE4 ? PalmDatabase.U16(data, 0xF2) : 0,
+ Fdst = hasMobi && length >= 0xE4 ? PalmDatabase.U32(data, 0xC0) : -1,
+ Fragment = hasMobi && length >= 0xE8 ? PalmDatabase.U32(data, 0xF8) : -1,
+ Skeleton = hasMobi && length >= 0xEC ? PalmDatabase.U32(data, 0xFC) : -1,
+ };
+ if (!hasMobi)
+ return header;
+ // MOBI 6 files reuse 0xC0 for "first content record": only KF8 headers have an FDST index there.
+ if (header.Version < 8)
+ header = header.WithoutKf8Fields();
+
+ var nameOffset = (int)PalmDatabase.U32(data, 84);
+ var nameLength = (int)PalmDatabase.U32(data, 88);
+ header._locale = (int)Math.Max(0, PalmDatabase.U32(data, 92));
+ if (nameOffset > 0 && nameLength > 0 && nameOffset + nameLength <= data.Length)
+ header._fullName = header.Decode(data.Slice(nameOffset, nameLength).ToArray());
+
+ if ((PalmDatabase.U32(data, 128) & 0x40) != 0 && 16 + length + 12 <= data.Length && data.Slice(16 + length, 4).SequenceEqual("EXTH"u8))
+ {
+ var exth = 16 + length;
+ var count = (int)PalmDatabase.U32(data, exth + 8);
+ var position = exth + 12;
+ for (var i = 0; i < count && position + 8 <= data.Length; i++)
+ {
+ var type = (int)PalmDatabase.U32(data, position);
+ var size = (int)PalmDatabase.U32(data, position + 4);
+ if (size < 8 || position + size > data.Length)
+ break;
+ if (!header._exth.TryGetValue(type, out var values))
+ header._exth[type] = values = [];
+ values.Add(data.Slice(position + 8, size - 8).ToArray());
+ position += size;
+ }
+ header.Kf8Boundary = header.Number(121) is { } boundary and > 0 ? boundary : null;
+ header.CoverOffset = header.Number(201);
+ }
+ return header;
+ }
+
+ private MobiHeader WithoutKf8Fields() => new()
+ {
+ Compression = Compression,
+ TextLength = TextLength,
+ TextRecords = TextRecords,
+ Encryption = Encryption,
+ Codepage = Codepage,
+ Version = Version,
+ FirstImage = FirstImage,
+ HuffRecord = HuffRecord,
+ HuffCount = HuffCount,
+ ExtraFlags = ExtraFlags,
+ };
+
+ /// Title (EXTH 503, else the full name, else the database name), authors, publisher, language.
+ public BookMetadata Metadata(string databaseName)
+ {
+ var metadata = new BookMetadata
+ {
+ Title = Strings(503).FirstOrDefault() ?? _fullName ?? databaseName.Replace('_', ' '),
+ Publisher = Strings(101).FirstOrDefault(),
+ Description = Strings(103).FirstOrDefault(),
+ Language = Strings(524).FirstOrDefault() ?? Language(_locale),
+ };
+ foreach (var author in Strings(100))
+ foreach (var name in author.Split('&', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries))
+ if (!metadata.Authors.Contains(name))
+ metadata.Authors.Add(name);
+ return metadata;
+ }
+
+ private IEnumerable Strings(int type) =>
+ _exth.TryGetValue(type, out var values) ? values.Select(Decode).Select(s => s.Trim('\0', ' ')).Where(s => s.Length > 0) : [];
+
+ private int? Number(int type) =>
+ _exth.TryGetValue(type, out var values) && values[0].Length == 4 && BinaryPrimitives.ReadUInt32BigEndian(values[0]) is var value && value != uint.MaxValue
+ ? (int)Math.Min(value, int.MaxValue)
+ : null;
+
+ private string Decode(byte[] bytes)
+ {
+ if (Codepage == 65001)
+ return Encoding.UTF8.GetString(bytes);
+ Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
+ return Encoding.GetEncoding(1252).GetString(bytes);
+ }
+
+ /// Windows language id (low byte of the MOBI locale) → language tag, for the common languages.
+ private static string? Language(int locale) => (locale & 0xFF) switch
+ {
+ 0x04 => "zh",
+ 0x07 => "de",
+ 0x09 => "en",
+ 0x0A => "es",
+ 0x0C => "fr",
+ 0x10 => "it",
+ 0x11 => "ja",
+ 0x12 => "ko",
+ 0x13 => "nl",
+ 0x16 => "pt",
+ 0x19 => "ru",
+ _ => null,
+ };
+}
diff --git a/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs b/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs
new file mode 100644
index 0000000..ae665c7
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Mobi/PalmDatabase.cs
@@ -0,0 +1,116 @@
+// The Palm database (PDB) container of MOBI, AZW, AZW3, AZW4 and PRC books: a 78-byte header with the type and
+// creator ("BOOKMOBI", "TEXtREAd"), then a table of record offsets. Written from the PalmOS database format
+// description; all numbers are big-endian.
+
+using System.Buffers.Binary;
+using System.Text;
+
+namespace Filee.Engines.Ebooks;
+
+internal sealed class PalmDatabase
+{
+ private readonly byte[] _data;
+ private readonly int[] _offsets;
+
+ private PalmDatabase(byte[] data, string name, string typeCreator, int[] offsets)
+ {
+ _data = data;
+ Name = name;
+ TypeCreator = typeCreator;
+ _offsets = offsets;
+ }
+
+ /// Database name from the header (a short title, often truncated).
+ public string Name { get; }
+
+ /// Type and creator, e.g. "BOOKMOBI" (Mobipocket / Kindle) or "TEXtREAd" (PalmDOC).
+ public string TypeCreator { get; }
+
+ public int Count => _offsets.Length;
+
+ public static PalmDatabase Open(byte[] data)
+ {
+ if (data.Length < 78)
+ throw new InvalidDataException("The file is too short to be an e-book.");
+ var count = BinaryPrimitives.ReadUInt16BigEndian(data.AsSpan(76));
+ if (count == 0 || 78 + count * 8 > data.Length)
+ throw new InvalidDataException("The e-book's record table is damaged.");
+ var offsets = new int[count];
+ for (var i = 0; i < count; i++)
+ {
+ var offset = BinaryPrimitives.ReadUInt32BigEndian(data.AsSpan(78 + i * 8));
+ if (offset > data.Length || (i > 0 && offset < offsets[i - 1]))
+ throw new InvalidDataException("The e-book's record table is damaged.");
+ offsets[i] = (int)offset;
+ }
+ var name = Encoding.Latin1.GetString(data, 0, 32).TrimEnd('\0');
+ return new PalmDatabase(data, name, Encoding.ASCII.GetString(data, 60, 8), offsets);
+ }
+
+ /// Record (empty when out of range).
+ public ReadOnlySpan Record(int index)
+ {
+ if (index < 0 || index >= _offsets.Length)
+ return [];
+ var end = index + 1 < _offsets.Length ? _offsets[index + 1] : _data.Length;
+ return _data.AsSpan(_offsets[index], Math.Max(0, end - _offsets[index]));
+ }
+
+ public byte[] RecordArray(int index) => Record(index).ToArray();
+
+ // Big-endian helpers with bounds checks: damaged books must fail with a clear error, not an exception deep inside.
+
+ public static int U16(ReadOnlySpan data, int offset) =>
+ offset >= 0 && offset + 2 <= data.Length ? BinaryPrimitives.ReadUInt16BigEndian(data[offset..]) : 0;
+
+ /// A 32-bit field; 0xFFFFFFFF ("none" in MOBI headers) becomes -1.
+ public static long U32(ReadOnlySpan data, int offset)
+ {
+ if (offset < 0 || offset + 4 > data.Length)
+ return -1;
+ var value = BinaryPrimitives.ReadUInt32BigEndian(data[offset..]);
+ return value == uint.MaxValue ? -1 : value;
+ }
+}
+
+/// PalmDOC compression (a simple LZ77 variant used by MOBI and PalmDOC text records).
+internal static class PalmDocCompression
+{
+ /// Decompresses one text record.
+ public static byte[] Decompress(ReadOnlySpan input)
+ {
+ var output = new List(input.Length * 2);
+ for (var i = 0; i < input.Length;)
+ {
+ var c = input[i++];
+ if (c is >= 1 and <= 8)
+ {
+ // 1..8: that many literal bytes follow.
+ for (var n = 0; n < c && i < input.Length; n++)
+ output.Add(input[i++]);
+ }
+ else if (c < 0x80)
+ {
+ output.Add(c); // 0 and 9..0x7F: the byte itself
+ }
+ else if (c >= 0xC0)
+ {
+ output.Add((byte)' '); // a space followed by an ASCII character
+ output.Add((byte)(c ^ 0x80));
+ }
+ else if (i < input.Length)
+ {
+ // 0x80..0xBF: two bytes = 11-bit distance back and a length of 3..10.
+ var pair = (c << 8) | input[i++];
+ var distance = (pair >> 3) & 0x7FF;
+ var length = (pair & 7) + 3;
+ if (distance == 0 || distance > output.Count)
+ continue; // damaged: skip rather than fail the whole book
+ var start = output.Count - distance;
+ for (var n = 0; n < length; n++)
+ output.Add(output[start + n]);
+ }
+ }
+ return [.. output];
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/Opf.cs b/src/Filee.Engines/Ebooks/Opf.cs
new file mode 100644
index 0000000..398c0da
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/Opf.cs
@@ -0,0 +1,127 @@
+// OPF package documents (EPUB 2 and 3, also the metadata.opf of HTMLZ and TXTZ): metadata, manifest and spine.
+// Element names are matched by local name so that files from sloppy generators (missing or odd prefixes) still read.
+
+using System.Xml;
+using System.Xml.Linq;
+
+namespace Filee.Engines.Ebooks;
+
+/// A manifest item: a file of the book.
+/// Full path of the extracted file.
+internal sealed record OpfItem(string Id, string Path, string MediaType, string Properties);
+
+/// A parsed OPF package document.
+internal sealed class OpfPackage
+{
+ public BookMetadata Metadata { get; } = new();
+ public Dictionary Manifest { get; } = new(StringComparer.Ordinal);
+
+ /// Reading order (linear and non-linear items, as calibre does).
+ public List Spine { get; } = [];
+
+ /// Reads an OPF file whose manifest paths are relative to its folder.
+ public static OpfPackage Load(string opfPath)
+ {
+ var package = new OpfPackage();
+ var document = LoadXml(opfPath);
+ var root = document.Root ?? throw new InvalidDataException("The package document is empty.");
+ var folder = System.IO.Path.GetDirectoryName(System.IO.Path.GetFullPath(opfPath))!;
+
+ foreach (var item in Elements(root, "manifest").SelectMany(m => Elements(m, "item")))
+ {
+ var id = (string?)item.Attribute("id");
+ var href = (string?)item.Attribute("href");
+ if (id is null || href is null)
+ continue;
+ var path = EbookFiles.SafePath(folder, Uri.UnescapeDataString(href.Split('#')[0]));
+ if (path is not null)
+ package.Manifest.TryAdd(id, new OpfItem(id, path, ((string?)item.Attribute("media-type") ?? "").Trim().ToLowerInvariant(), (string?)item.Attribute("properties") ?? ""));
+ }
+ foreach (var itemRef in Elements(root, "spine").SelectMany(s => Elements(s, "itemref")))
+ {
+ if ((string?)itemRef.Attribute("idref") is { } idRef && package.Manifest.TryGetValue(idRef, out var item))
+ package.Spine.Add(item);
+ }
+
+ var metadata = Elements(root, "metadata").FirstOrDefault();
+ if (metadata is not null)
+ ReadMetadata(metadata, package);
+ return package;
+ }
+
+ private static void ReadMetadata(XElement metadata, OpfPackage package)
+ {
+ // EPUB 2 puts dc: elements in ; search descendants.
+ string? First(string name) => metadata.Descendants().FirstOrDefault(e => e.Name.LocalName == name && e.Value.Trim().Length > 0)?.Value.Trim();
+ var meta = package.Metadata;
+ meta.Title = First("title");
+ meta.Language = First("language");
+ meta.Description = First("description");
+ meta.Publisher = First("publisher");
+ foreach (var creator in metadata.Descendants().Where(e => e.Name.LocalName == "creator"))
+ {
+ var name = creator.Value.Trim();
+ var role = creator.Attributes().FirstOrDefault(a => a.Name.LocalName == "role")?.Value;
+ if (name.Length > 0 && (role is null or "aut") && !meta.Authors.Contains(name))
+ meta.Authors.Add(name);
+ }
+
+ // Cover: EPUB 3 "cover-image" property, else EPUB 2 .
+ var cover = package.Manifest.Values.FirstOrDefault(i => i.Properties.Split(' ').Contains("cover-image"));
+ if (cover is null
+ && metadata.Descendants().FirstOrDefault(e => e.Name.LocalName == "meta" && (string?)e.Attribute("name") == "cover") is { } coverMeta
+ && (string?)coverMeta.Attribute("content") is { } coverId)
+ package.Manifest.TryGetValue(coverId, out cover);
+ if (cover is not null && cover.MediaType.StartsWith("image/", StringComparison.Ordinal) && File.Exists(cover.Path))
+ meta.CoverImage = cover.Path;
+ }
+
+ /// The first rootfile named by META-INF/container.xml.
+ public static string RootFile(string bookFolder)
+ {
+ var container = System.IO.Path.Combine(bookFolder, "META-INF", "container.xml");
+ if (File.Exists(container))
+ {
+ var rootFile = LoadXml(container).Descendants().FirstOrDefault(e => e.Name.LocalName == "rootfile"
+ && ((string?)e.Attribute("media-type") ?? "application/oebps-package+xml") == "application/oebps-package+xml");
+ if ((string?)rootFile?.Attribute("full-path") is { } fullPath && EbookFiles.SafePath(bookFolder, Uri.UnescapeDataString(fullPath)) is { } path && File.Exists(path))
+ return path;
+ }
+ // Broken container: use the first package document in the book.
+ return Directory.EnumerateFiles(bookFolder, "*.opf", SearchOption.AllDirectories).FirstOrDefault()
+ ?? throw new InvalidDataException("This is not a valid EPUB file: it has no package document (content.opf).");
+ }
+
+ ///
+ /// Loads XML without resolving external DTDs (no network access, no XXE); entity declarations inside the file
+ /// are still allowed because some generators declare and friends.
+ ///
+ public static XDocument LoadXml(string path)
+ {
+ var settings = new XmlReaderSettings { DtdProcessing = DtdProcessing.Ignore, XmlResolver = null };
+ using var reader = XmlReader.Create(path, settings);
+ return XDocument.Load(reader);
+ }
+
+ private static IEnumerable Elements(XElement parent, string localName) =>
+ parent.Elements().Where(e => e.Name.LocalName == localName);
+
+ /// A minimal OPF with Dublin Core metadata (HTMLZ and TXTZ keep their metadata this way).
+ public static string MetadataOnly(BookMetadata metadata, string title, string language)
+ {
+ XNamespace opf = "http://www.idpf.org/2007/opf";
+ XNamespace dc = "http://purl.org/dc/elements/1.1/";
+ var meta = new XElement(opf + "metadata", new XAttribute(XNamespace.Xmlns + "dc", dc.NamespaceName),
+ new XElement(dc + "title", title),
+ new XElement(dc + "language", language),
+ new XElement(dc + "identifier", new XAttribute("id", "uuid_id"), $"urn:uuid:{Guid.NewGuid()}"));
+ foreach (var author in metadata.Authors)
+ meta.Add(new XElement(dc + "creator", new XAttribute(opf + "role", "aut"), author));
+ if (metadata.Publisher is { } publisher)
+ meta.Add(new XElement(dc + "publisher", publisher));
+ if (metadata.Description is { } description)
+ meta.Add(new XElement(dc + "description", description));
+ var package = new XElement(opf + "package", new XAttribute("version", "2.0"), new XAttribute("unique-identifier", "uuid_id"), meta);
+ return new XDeclaration("1.0", "utf-8", null) + "\n" + package;
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/PlainText.cs b/src/Filee.Engines/Ebooks/PlainText.cs
new file mode 100644
index 0000000..de7deb4
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/PlainText.cs
@@ -0,0 +1,58 @@
+// Plain text → HDocument for e-books: paragraphs are separated by empty lines, and hard-wrapped lines inside a
+// paragraph (Project Gutenberg style, ~70 characters) are joined. Text without empty lines keeps one paragraph per
+// line. (TXT → HWPX keeps every line as it is, see HwpxConverter.TextDocument.)
+
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal static class PlainText
+{
+ public static HDocument ToDocument(string text)
+ {
+ var lines = text.Replace("\r\n", "\n").Replace('\r', '\n').Split('\n');
+ var paragraphs = new List>();
+ var current = new List();
+ var hasEmptyLines = lines.Any(l => l.Trim().Length == 0);
+ foreach (var line in lines)
+ {
+ if (line.Trim().Length == 0 || !hasEmptyLines)
+ {
+ if (current.Count > 0)
+ paragraphs.Add(current);
+ current = [];
+ if (line.Trim().Length == 0)
+ continue;
+ }
+ current.Add(line.TrimEnd());
+ }
+ if (current.Count > 0)
+ paragraphs.Add(current);
+
+ // Hard-wrapped text: most lines of multi-line paragraphs are long and of similar length.
+ var multiLine = paragraphs.Where(p => p.Count > 1).SelectMany(p => p.Take(p.Count - 1)).ToList();
+ var wrapped = multiLine.Count > 0 && multiLine.Count(l => l.Length is >= 50 and <= 100) > multiLine.Count * 0.7;
+
+ var section = new HSection();
+ foreach (var paragraph in paragraphs)
+ {
+ var hParagraph = new HParagraph();
+ for (var i = 0; i < paragraph.Count; i++)
+ {
+ if (i > 0)
+ {
+ if (wrapped)
+ hParagraph.Inlines.Add(new HText(" ", default));
+ else
+ hParagraph.Inlines.Add(new HLineBreak(default));
+ }
+ hParagraph.Inlines.Add(new HText(i > 0 && wrapped ? paragraph[i].TrimStart() : paragraph[i], default));
+ }
+ HtmlReader.MergeRuns(hParagraph.Inlines);
+ section.Blocks.Add(hParagraph);
+ }
+ var document = new HDocument();
+ document.Sections.Add(section);
+ return document;
+ }
+}
diff --git a/src/Filee.Engines/Ebooks/ZippedText.cs b/src/Filee.Engines/Ebooks/ZippedText.cs
new file mode 100644
index 0000000..c06ec1f
--- /dev/null
+++ b/src/Filee.Engines/Ebooks/ZippedText.cs
@@ -0,0 +1,143 @@
+// HTMLZ and TXTZ, calibre's zipped single-file formats: HTMLZ is index.html + style.css + images + metadata.opf,
+// TXTZ is index.txt (Markdown) + images + metadata.opf. Both are read and written here.
+
+using System.IO.Compression;
+using System.Text;
+using Filee.Engines.Hwp.Hwpx;
+
+namespace Filee.Engines.Ebooks;
+
+internal static class ZippedText
+{
+ /// Reads an HTMLZ book (for the reader registry of the HWPX writer).
+ public static HDocument ReadHtmlz(string path, string workFolder) => ReadHtmlzBook(path, workFolder).Document;
+
+ /// Reads a TXTZ book (for the reader registry of the HWPX writer).
+ public static HDocument ReadTxtz(string path, string workFolder) => ReadTxtzBook(path, workFolder).Document;
+
+ public static Book ReadHtmlzBook(string path, string workFolder, CancellationToken cancellationToken = default)
+ {
+ var folder = Extract(path, workFolder, "htmlz");
+ var page = MainFile(folder, "index.html", ".html", ".htm", ".xhtml")
+ ?? throw new InvalidDataException("The HTMLZ file contains no HTML page.");
+ var document = ChapterReader.Read([page], Path.Combine(folder, "~media"), cancellationToken);
+ var metadata = Metadata(folder);
+ document.Title = metadata.Title ?? document.Title;
+ return new Book(document, metadata);
+ }
+
+ public static Book ReadTxtzBook(string path, string workFolder)
+ {
+ var folder = Extract(path, workFolder, "txtz");
+ var text = MainFile(folder, "index.txt", ".txt", ".md", ".markdown", ".text")
+ ?? throw new InvalidDataException("The TXTZ file contains no text file.");
+ // calibre writes TXTZ text as plain text, Markdown or Textile; Markdown reads all of them sensibly.
+ var document = MarkdownReader.Read(HwpxConverter.DecodeText(File.ReadAllBytes(text)), Path.GetDirectoryName(text)!);
+ var metadata = Metadata(folder);
+ document.Title = metadata.Title ?? document.Title;
+ return new Book(document, metadata);
+ }
+
+ public static void WriteHtmlz(Book book, string outputPath, string sourcePath, string workFolder)
+ {
+ var title = book.DisplayTitle(sourcePath);
+ var language = book.DisplayLanguage();
+ var images = new ImageSet(EbookFiles.NewFolder(workFolder, "htmlz-images"));
+ var page = XhtmlWriter.Write(book.Document, new XhtmlOptions { ImageSource = images.Add }).Single();
+ Save(outputPath, [
+ ("index.html", XhtmlWriter.Page(title, page.Body, language, "style.css", epub: false)),
+ ("style.css", XhtmlWriter.Css),
+ ("metadata.opf", OpfPackage.MetadataOnly(book.Metadata, title, language)),
+ ], images);
+ }
+
+ public static void WriteTxtz(Book book, string outputPath, string sourcePath, string workFolder)
+ {
+ var title = book.DisplayTitle(sourcePath);
+ var images = new ImageSet(EbookFiles.NewFolder(workFolder, "txtz-images"));
+ var text = MarkdownWriter.Write(book.Document, images.Add);
+ Save(outputPath, [
+ ("index.txt", text),
+ ("metadata.opf", OpfPackage.MetadataOnly(book.Metadata, title, book.DisplayLanguage())),
+ ], images);
+ }
+
+ private static string Extract(string path, string workFolder, string prefix)
+ {
+ var folder = EbookFiles.NewFolder(workFolder, prefix);
+ try
+ {
+ using var zip = ZipFile.OpenRead(path);
+ EbookFiles.Extract(zip, folder);
+ }
+ catch (InvalidDataException ex)
+ {
+ throw new InvalidDataException($"This is not a valid {prefix.ToUpperInvariant()} file (it is not a ZIP archive).", ex);
+ }
+ return folder;
+ }
+
+ /// The preferred file name at the top level, else the first file with one of the extensions (shallowest first).
+ private static string? MainFile(string folder, string preferred, params string[] extensions)
+ {
+ var top = Path.Combine(folder, preferred);
+ if (File.Exists(top))
+ return top;
+ return Directory.EnumerateFiles(folder, "*", SearchOption.AllDirectories)
+ .Where(f => extensions.Contains(Path.GetExtension(f).ToLowerInvariant()))
+ .OrderBy(f => f.Count(c => c == Path.DirectorySeparatorChar))
+ .ThenBy(f => f, StringComparer.OrdinalIgnoreCase)
+ .FirstOrDefault();
+ }
+
+ private static BookMetadata Metadata(string folder)
+ {
+ var opf = Directory.EnumerateFiles(folder, "*.opf", SearchOption.AllDirectories).FirstOrDefault();
+ if (opf is null)
+ return new BookMetadata();
+ try
+ {
+ return OpfPackage.Load(opf).Metadata;
+ }
+ catch (System.Xml.XmlException)
+ {
+ return new BookMetadata(); // broken metadata: the content still converts
+ }
+ }
+
+ private static void Save(string outputPath, IEnumerable<(string Name, string Text)> texts, ImageSet images)
+ {
+ var temp = outputPath + ".tmp";
+ using (var zip = ZipFile.Open(temp, ZipArchiveMode.Create))
+ {
+ foreach (var (name, text) in texts)
+ {
+ using var writer = new StreamWriter(zip.CreateEntry(name).Open(), new UTF8Encoding(false));
+ writer.Write(text);
+ }
+ foreach (var (name, file) in images.Files)
+ zip.CreateEntryFromFile(file, name, CompressionLevel.NoCompression);
+ }
+ File.Move(temp, outputPath, overwrite: true);
+ }
+
+ /// Pictures of a zipped book: images/imgNNNN.ext, converted to JPEG / PNG / GIF where needed.
+ private sealed class ImageSet(string convertFolder)
+ {
+ private readonly Dictionary _names = new(StringComparer.OrdinalIgnoreCase);
+
+ public List<(string Name, string File)> Files { get; } = [];
+
+ public string? Add(string path)
+ {
+ if (_names.TryGetValue(path, out var known))
+ return known;
+ if (EbookFiles.CommonImage(path, convertFolder, gifAndSvg: true) is not ({ } file, _))
+ return null;
+ var name = $"images/img{Files.Count + 1:0000}{Path.GetExtension(file).ToLowerInvariant()}";
+ Files.Add((name, file));
+ _names[path] = name;
+ return name;
+ }
+ }
+}
diff --git a/src/Filee.Engines/EngineRegistry.cs b/src/Filee.Engines/EngineRegistry.cs
index e57508d..3e0e9d0 100644
--- a/src/Filee.Engines/EngineRegistry.cs
+++ b/src/Filee.Engines/EngineRegistry.cs
@@ -3,6 +3,7 @@
using Filee.Core.Conversion;
using Filee.Engines.Archives;
using Filee.Engines.Cad;
+using Filee.Engines.Ebooks;
using Filee.Engines.Email;
using Filee.Engines.Fonts;
using Filee.Engines.Hwp;
@@ -46,8 +47,10 @@ public static IReadOnlyList CreateAll(EngineEnvironment env) =>
new EmlConverter(),
new FfmpegConverter(),
new CadConverter(),
+ new EbookConverter(),
new PandocConverter(),
new GhostscriptConverter(),
+ new CalibreConverter(),
new LibreOfficeConverter(env),
];
diff --git a/src/Filee.Engines/Filee.Engines.csproj b/src/Filee.Engines/Filee.Engines.csproj
index c8a21e6..16dd1af 100644
--- a/src/Filee.Engines/Filee.Engines.csproj
+++ b/src/Filee.Engines/Filee.Engines.csproj
@@ -24,6 +24,7 @@
+
diff --git a/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs b/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs
index 5c9a7fd..7534e5d 100644
--- a/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs
+++ b/src/Filee.Engines/Hwp/Hwpx/DocumentReaders.cs
@@ -4,6 +4,7 @@
using System.Text;
using System.Text.Json.Nodes;
+using Filee.Engines.Ebooks;
using Filee.Engines.Hwp.Hwpx.Docx;
using Filee.Engines.Hwp.Hwpx.Pptx;
using Filee.Engines.Infrastructure;
@@ -44,6 +45,15 @@ internal static class DocumentReaders
["pdf"] = PdfDocumentReader.Read,
// HWP comes here as HWPX, converted by rhwp first (the route planner adds that step).
["hwpx"] = HwpxReader.Read,
+ ["html"] = HtmlReader.Read,
+ ["epub"] = EpubReader.Read,
+ ["mobi"] = MobiReader.Read,
+ ["azw3"] = MobiReader.Read,
+ ["azw"] = MobiReader.Read,
+ ["prc"] = MobiReader.Read,
+ ["fb2"] = Fb2Reader.Read,
+ ["htmlz"] = ZippedText.ReadHtmlz,
+ ["txtz"] = ZippedText.ReadTxtz,
};
///
diff --git a/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs b/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs
new file mode 100644
index 0000000..b224590
--- /dev/null
+++ b/src/Filee.Engines/Hwp/Hwpx/HDocumentWalker.cs
@@ -0,0 +1,87 @@
+// Walks the HDocument model: every inline of a block list, including those inside tables, notes, text boxes and
+// links. Used by readers (bookmark pruning) and the writers that are not HWPX (XHTML, plain text, Markdown, FB2).
+
+using System.Text;
+
+namespace Filee.Engines.Hwp.Hwpx;
+
+internal static class HDocumentWalker
+{
+ /// Every inline, depth first: the content of links, notes and text boxes follows them.
+ public static IEnumerable Inlines(IEnumerable blocks) =>
+ InlineLists(blocks).SelectMany(list => list);
+
+ /// Every inline list (paragraph, link content, note and text box paragraphs, table cells).
+ public static IEnumerable> InlineLists(IEnumerable blocks)
+ {
+ foreach (var block in blocks)
+ {
+ switch (block)
+ {
+ case HParagraph paragraph:
+ foreach (var list in InlineLists(paragraph.Inlines))
+ yield return list;
+ break;
+ case HTable table:
+ foreach (var list in InlineLists(table.Caption))
+ yield return list;
+ foreach (var cell in table.Rows.SelectMany(r => r.Cells))
+ foreach (var list in InlineLists(cell.Blocks))
+ yield return list;
+ break;
+ }
+ }
+ }
+
+ private static IEnumerable> InlineLists(List inlines)
+ {
+ yield return inlines;
+ foreach (var inline in inlines.ToList())
+ {
+ var nested = inline switch
+ {
+ HLink link => InlineLists(link.Content),
+ HNote note => InlineLists(note.Blocks),
+ HTextBox box => InlineLists(box.Blocks),
+ _ => [],
+ };
+ foreach (var list in nested)
+ yield return list;
+ }
+ }
+
+ /// The inlines of one paragraph with the content of its links (not notes or text boxes).
+ public static IEnumerable InlinesOf(IEnumerable inlines)
+ {
+ foreach (var inline in inlines)
+ {
+ yield return inline;
+ if (inline is HLink link)
+ foreach (var child in InlinesOf(link.Content))
+ yield return child;
+ }
+ }
+
+ /// Every picture of the document, in reading order.
+ public static IEnumerable Images(HDocument document) =>
+ document.Sections.SelectMany(s => Inlines(s.Blocks)).OfType();
+
+ /// The visible text of a paragraph's inlines (links included, notes left out), line breaks as spaces.
+ public static string Text(IEnumerable inlines)
+ {
+ var sb = new StringBuilder();
+ foreach (var inline in InlinesOf(inlines))
+ {
+ switch (inline)
+ {
+ case HText text:
+ sb.Append(text.Text);
+ break;
+ case HLineBreak or HTab:
+ sb.Append(' ');
+ break;
+ }
+ }
+ return sb.ToString().Trim();
+ }
+}
diff --git a/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs b/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs
new file mode 100644
index 0000000..809ee83
--- /dev/null
+++ b/src/Filee.Engines/Hwp/Hwpx/HtmlCss.cs
@@ -0,0 +1,291 @@
+// The small part of CSS the HTML reader understands: inline style="" declarations, simple stylesheet rules (tag,
+// .class, tag.class), colours and lengths. Enough for the formatting e-books and saved web pages rely on (bold and
+// italic classes, centred text, page breaks, first-line indents) without a full CSS engine.
+
+using System.Globalization;
+using System.Text.RegularExpressions;
+
+namespace Filee.Engines.Hwp.Hwpx;
+
+/// Property → value pairs of one declaration block, lower-case property names.
+internal sealed class CssDeclarations : Dictionary
+{
+ public CssDeclarations() : base(StringComparer.OrdinalIgnoreCase)
+ {
+ }
+
+ /// Parses "color: red; font-weight: bold" (later declarations win, !important is ignored).
+ public static CssDeclarations Parse(string? text)
+ {
+ var result = new CssDeclarations();
+ if (string.IsNullOrWhiteSpace(text))
+ return result;
+ foreach (var declaration in text.Split(';'))
+ {
+ var colon = declaration.IndexOf(':');
+ if (colon <= 0)
+ continue;
+ var name = declaration[..colon].Trim();
+ var value = declaration[(colon + 1)..].Replace("!important", "", StringComparison.OrdinalIgnoreCase).Trim();
+ if (name.Length > 0 && value.Length > 0)
+ result[name] = value;
+ }
+ return result;
+ }
+
+ public string? Get(string name) => TryGetValue(name, out var value) ? value : null;
+}
+
+/// Rules of the document's style sheets that the reader can match without a full selector engine.
+internal sealed partial class CssStyleSheet
+{
+ /// Properties taken from style sheets. Colours and sizes are not: whole-book rules (body colour, base
+ /// font size) would otherwise be copied onto every run.
+ private static readonly HashSet SheetProperties = new(StringComparer.OrdinalIgnoreCase)
+ {
+ "font-weight", "font-style", "text-decoration", "text-decoration-line", "vertical-align", "text-align",
+ "display", "page-break-before", "page-break-after", "break-before", "break-after", "white-space",
+ "text-indent", "list-style", "list-style-type", "font-variant",
+ };
+
+ private readonly List _rules = [];
+
+ /// Element name, or null for any element.
+ /// Classes the element must have (all of them).
+ /// Classes count more than the tag name.
+ private sealed record Rule(string? Tag, string[] Classes, int Specificity, int Order, CssDeclarations Declarations);
+
+ public bool IsEmpty => _rules.Count == 0;
+
+ /// Adds the rules of a style sheet. @media, @font-face and other at-rules are skipped.
+ public void Add(string css)
+ {
+ css = Comment().Replace(css, "");
+ var position = 0;
+ while (position < css.Length)
+ {
+ var open = css.IndexOf('{', position);
+ if (open < 0)
+ break;
+ var selectors = css[position..open].Trim();
+ var close = MatchingBrace(css, open);
+ var body = css[(open + 1)..Math.Max(open + 1, close)];
+ position = close < 0 ? css.Length : close + 1;
+ if (selectors.StartsWith('@') || body.Contains('{'))
+ continue; // at-rules (nested blocks): not applied
+
+ var declarations = CssDeclarations.Parse(body);
+ foreach (var key in declarations.Keys.Where(k => !SheetProperties.Contains(k)).ToList())
+ declarations.Remove(key);
+ if (declarations.Count == 0)
+ continue;
+ foreach (var selector in selectors.Split(','))
+ {
+ if (ParseSelector(selector.Trim()) is { } rule)
+ _rules.Add(rule with { Order = _rules.Count, Declarations = declarations });
+ }
+ }
+ }
+
+ /// Declarations that apply to an element, in cascade order (the last one wins).
+ public CssDeclarations Match(string tag, IReadOnlyCollection classes)
+ {
+ var result = new CssDeclarations();
+ if (_rules.Count == 0)
+ return result;
+ foreach (var rule in _rules.Where(r => (r.Tag is null || r.Tag == tag) && r.Classes.All(classes.Contains))
+ .OrderBy(r => r.Specificity).ThenBy(r => r.Order))
+ {
+ foreach (var (name, value) in rule.Declarations)
+ result[name] = value;
+ }
+ return result;
+ }
+
+ /// "p", ".note", "span.bold", "p.a.b"; anything else (descendants, ids, pseudo classes) is skipped.
+ private static Rule? ParseSelector(string selector)
+ {
+ if (!SimpleSelector().IsMatch(selector))
+ return null;
+ var parts = selector.Split('.');
+ var tag = parts[0].Length == 0 || parts[0] == "*" ? null : parts[0].ToLowerInvariant();
+ var classes = parts.Skip(1).ToArray();
+ return new Rule(tag, classes, classes.Length * 10 + (tag is null ? 0 : 1), 0, []);
+ }
+
+ private static int MatchingBrace(string css, int open)
+ {
+ var depth = 0;
+ for (var i = open; i < css.Length; i++)
+ {
+ if (css[i] == '{')
+ depth++;
+ else if (css[i] == '}' && --depth == 0)
+ return i;
+ }
+ return -1;
+ }
+
+ [GeneratedRegex(@"/\*.*?\*/", RegexOptions.Singleline)]
+ private static partial Regex Comment();
+
+ [GeneratedRegex(@"^(\*|[A-Za-z][\w-]*)?(\.[\w-]+)*$")]
+ private static partial Regex SimpleSelector();
+}
+
+/// CSS values: colours and lengths.
+internal static class CssValues
+{
+ private static readonly Dictionary NamedColors = new(StringComparer.OrdinalIgnoreCase)
+ {
+ ["black"] = "#000000",
+ ["white"] = "#FFFFFF",
+ ["red"] = "#FF0000",
+ ["green"] = "#008000",
+ ["blue"] = "#0000FF",
+ ["yellow"] = "#FFFF00",
+ ["gray"] = "#808080",
+ ["grey"] = "#808080",
+ ["silver"] = "#C0C0C0",
+ ["maroon"] = "#800000",
+ ["purple"] = "#800080",
+ ["fuchsia"] = "#FF00FF",
+ ["magenta"] = "#FF00FF",
+ ["lime"] = "#00FF00",
+ ["olive"] = "#808000",
+ ["navy"] = "#000080",
+ ["teal"] = "#008080",
+ ["aqua"] = "#00FFFF",
+ ["cyan"] = "#00FFFF",
+ ["orange"] = "#FFA500",
+ ["brown"] = "#A52A2A",
+ ["pink"] = "#FFC0CB",
+ ["gold"] = "#FFD700",
+ ["darkred"] = "#8B0000",
+ ["darkblue"] = "#00008B",
+ ["darkgreen"] = "#006400",
+ ["darkgray"] = "#A9A9A9",
+ ["darkgrey"] = "#A9A9A9",
+ ["lightgray"] = "#D3D3D3",
+ ["lightgrey"] = "#D3D3D3",
+ ["dimgray"] = "#696969",
+ ["dimgrey"] = "#696969",
+ ["crimson"] = "#DC143C",
+ ["indigo"] = "#4B0082",
+ ["violet"] = "#EE82EE",
+ ["coral"] = "#FF7F50",
+ ["tomato"] = "#FF6347",
+ ["steelblue"] = "#4682B4",
+ ["royalblue"] = "#4169E1",
+ ["lightblue"] = "#ADD8E6",
+ ["lightyellow"] = "#FFFFE0",
+ ["lightgreen"] = "#90EE90",
+ ["whitesmoke"] = "#F5F5F5",
+ ["beige"] = "#F5F5DC",
+ ["ivory"] = "#FFFFF0",
+ };
+
+ /// "#RRGGBB" for #rgb, #rrggbb, rgb()/rgba() and common colour names; null for anything else
+ /// (transparent, inherit, gradients, ...).
+ public static string? Color(string? value)
+ {
+ if (string.IsNullOrWhiteSpace(value))
+ return null;
+ var text = value.Trim();
+ if (NamedColors.TryGetValue(text, out var named))
+ return named;
+ if (text.StartsWith('#'))
+ {
+ var hex = text[1..];
+ if (hex.Length is 3 or 4 && hex.All(Uri.IsHexDigit))
+ return "#" + string.Concat(hex.Take(3).Select(c => $"{c}{c}")).ToUpperInvariant();
+ if (hex.Length is 6 or 8 && hex.All(Uri.IsHexDigit))
+ return "#" + hex[..6].ToUpperInvariant();
+ return null;
+ }
+ if (text.StartsWith("rgb", StringComparison.OrdinalIgnoreCase) && text.IndexOf('(') is var open and > 0 && text.IndexOf(')') is var close && close > open)
+ {
+ var parts = text[(open + 1)..close].Split([',', ' ', '/'], StringSplitOptions.RemoveEmptyEntries);
+ if (parts.Length < 3)
+ return null;
+ var channels = new int[3];
+ for (var i = 0; i < 3; i++)
+ {
+ var part = parts[i];
+ var percent = part.EndsWith('%');
+ if (!double.TryParse(part.TrimEnd('%'), NumberStyles.Float, CultureInfo.InvariantCulture, out var number))
+ return null;
+ channels[i] = (int)Math.Clamp(Math.Round(percent ? number * 2.55 : number), 0, 255);
+ }
+ return $"#{channels[0]:X2}{channels[1]:X2}{channels[2]:X2}";
+ }
+ return null;
+ }
+
+ /// The first colour inside a shorthand such as "background: #eee url(x.png)".
+ public static string? FirstColor(string? value)
+ {
+ if (Color(value) is { } direct)
+ return direct;
+ foreach (var token in (value ?? "").Split(' ', StringSplitOptions.RemoveEmptyEntries))
+ if (Color(token) is { } color)
+ return color;
+ return null;
+ }
+
+ ///
+ /// A length in HWPUNIT. is the size of 1em (and the base of %) in HWPUNIT; null when the
+ /// value is not a length (auto, inherit, ...).
+ ///
+ public static int? Length(string? value, double em)
+ {
+ if (string.IsNullOrWhiteSpace(value))
+ return null;
+ var text = value.Trim().ToLowerInvariant();
+ var split = 0;
+ while (split < text.Length && (char.IsDigit(text[split]) || text[split] is '.' or '-' or '+'))
+ split++;
+ if (split == 0 || !double.TryParse(text[..split], NumberStyles.Float, CultureInfo.InvariantCulture, out var number))
+ return null;
+ var unit = text[split..].Trim();
+ double? result = unit switch
+ {
+ "em" or "rem" => number * em,
+ "ex" or "ch" => number * em / 2,
+ "%" => number * em / 100,
+ "pt" => number * 100,
+ "px" or "" => number * HwpxUnits.PerPixel,
+ "in" => number * 7200,
+ "cm" => number * HwpxUnits.PerMm * 10,
+ "mm" => number * HwpxUnits.PerMm,
+ "pc" => number * 1200,
+ _ => null,
+ };
+ return result is null ? null : (int)Math.Round(result.Value);
+ }
+
+ /// A font size in 1/100 pt, relative sizes based on (1/100 pt).
+ public static int? FontSize(string? value, int parent)
+ {
+ if (string.IsNullOrWhiteSpace(value))
+ return null;
+ double? factor = value.Trim().ToLowerInvariant() switch
+ {
+ "xx-small" => 0.6,
+ "x-small" => 0.75,
+ "small" => 0.89,
+ "medium" => 1.0,
+ "large" => 1.2,
+ "x-large" => 1.5,
+ "xx-large" => 2.0,
+ "xxx-large" => 3.0,
+ "smaller" => 0.83,
+ "larger" => 1.2,
+ _ => null,
+ };
+ if (factor is { } f)
+ return (int)Math.Round(parent * f);
+ // HWPUNIT and 1/100 pt are the same scale (1 pt = 100 HWPUNIT), so Length() fits font sizes directly.
+ return Length(value, parent) is { } size && size > 0 ? Math.Clamp(size, 100, 20000) : null;
+ }
+}
diff --git a/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs b/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs
new file mode 100644
index 0000000..7e2383a
--- /dev/null
+++ b/src/Filee.Engines/Hwp/Hwpx/HtmlReader.cs
@@ -0,0 +1,1142 @@
+// HTML / XHTML → HDocument with AngleSharp (MIT), so HTML → HWPX / PDF needs no Pandoc and e-books (EPUB, MOBI,
+// HTMLZ) reuse the same reader for their chapters.
+//
+// Headings, paragraphs, inline formatting (b / strong / i / em / u / s / del / sub / sup / code / mark / small /
+// font and the basic inline CSS: colour, background, font-weight, font-style, font-size, text-decoration,
+// vertical-align, text-align), simple class and tag rules from \n"
+ : $"\n");
+ sb.Append("\n\n").Append(body).Append("\n