theseus/bundled-addons/docx-editor/lib/read.js
Local Dev 56b3f2f206 feat(docx-editor): ship it in the build, updated over the first-party channel
Moving it out of the build left it with no way in. Settings can only install
from the community catalogue, so a first-party extension that isn't bundled
has a working update channel and no first copy for anyone to update — the
mechanism was all there and the front door was missing.

So it goes back beside screenshot, aegis and pdf-editor: seeded into every
profile by the build, listed under "Built into Theseus", and kept current
between releases by the operator-signed channel at
theseus.x/extensions/docx-editor/. That is the arrangement docs/ADDON-UPDATES.md
describes, and the one the signing script was written for.

About 400 KB compressed in the installer, most of it the vendored editor
libraries — next to the ~4 MB of pdf.js that pdf-editor already ships, the
weight argument for keeping it out didn't survive contact with the numbers.

The end-to-end driver goes back to checking that a fresh profile seeds it,
which is the property that actually matters now.
2026-09-21 22:47:13 +02:00

502 lines
19 KiB
JavaScript

// .docx -> editor model.
//
// mammoth does the hard part of reading OOXML: resolving style inheritance,
// numbering definitions, relationship targets, merged table cells. What it
// is designed to produce, though, is semantic HTML — and HTML has nowhere to
// put a run's colour or a paragraph's line spacing, so its converter throws
// them away.
//
// So we don't use its HTML at all. `transformDocument` hands us mammoth's
// parsed document model on the way past, and we walk THAT into ProseMirror
// JSON. Everything the model carries survives; see addon-build/docx-editor/
// patches.mjs for the handful of properties we taught it to carry.
(function (root, factory) {
const api = factory();
if (typeof module === "object" && module.exports) module.exports = api;
root.DocxEditor = Object.assign(root.DocxEditor || {}, { read: api });
})(typeof globalThis !== "undefined" ? globalThis : this, function () {
"use strict";
const V = () => globalThis.DOCXV;
const PAGE_CONTENT_PT = 468; // 6.5in of text between 1in margins
const DEFAULT_IMAGE_PT = 300;
// ------------------------------------------------------------- images ---
// Natural pixel size, straight out of the file header. Decoding through an
// <img> would work in the editor but not in the round-trip tests, and a
// size that depends on which half of the codebase is asking is a bug
// waiting to happen.
function imagePixelSize(bytes) {
const b = bytes;
const u16 = (i, le) => le ? b[i] | (b[i + 1] << 8) : (b[i] << 8) | b[i + 1];
const u32 = (i, le) => le
? (b[i] | (b[i + 1] << 8) | (b[i + 2] << 16) | (b[i + 3] << 24)) >>> 0
: ((b[i] << 24) | (b[i + 1] << 16) | (b[i + 2] << 8) | b[i + 3]) >>> 0;
if (b.length > 24 && b[0] === 0x89 && b[1] === 0x50) { // PNG
return { w: u32(16, false), h: u32(20, false) };
}
if (b.length > 10 && b[0] === 0x47 && b[1] === 0x49) { // GIF
return { w: u16(6, true), h: u16(8, true) };
}
if (b.length > 26 && b[0] === 0x42 && b[1] === 0x4d) { // BMP
return { w: u32(18, true), h: u32(22, true) };
}
if (b.length > 4 && b[0] === 0xff && b[1] === 0xd8) { // JPEG
let i = 2;
while (i + 9 < b.length) {
if (b[i] !== 0xff) { i++; continue; }
const marker = b[i + 1];
if (marker === 0xd8 || marker === 0x01 || (marker >= 0xd0 && marker <= 0xd7)) { i += 2; continue; }
const len = u16(i + 2, false);
// SOF0..SOF15, skipping the four that aren't start-of-frame.
if (marker >= 0xc0 && marker <= 0xcf && marker !== 0xc4 && marker !== 0xc8 && marker !== 0xcc) {
return { h: u16(i + 5, false), w: u16(i + 7, false) };
}
i += 2 + len;
}
}
return null;
}
function imageSizePt(bytes, declared) {
// What Word was drawing it at always wins — the author chose it.
if (declared && declared.widthPt) {
return { width: declared.widthPt, height: declared.heightPt || null };
}
const px = imagePixelSize(bytes);
if (!px || !px.w) return { width: DEFAULT_IMAGE_PT, height: null };
const ratio = px.h / px.w;
let w = px.w * 0.75; // 96dpi pixels to points
if (w > PAGE_CONTENT_PT) w = PAGE_CONTENT_PT;
return { width: Math.round(w * 100) / 100, height: Math.round(w * ratio * 100) / 100 };
}
function bytesToBase64(bytes) {
if (typeof Buffer !== "undefined") return Buffer.from(bytes).toString("base64");
let s = "";
for (let i = 0; i < bytes.length; i += 0x8000) {
s += String.fromCharCode.apply(null, bytes.subarray(i, i + 0x8000));
}
return btoa(s);
}
// --------------------------------------------------------------- marks ---
const HIGHLIGHT_NAMES = new Set([
"yellow", "green", "cyan", "magenta", "blue", "red", "darkBlue", "darkCyan",
"darkGreen", "darkMagenta", "darkRed", "darkYellow", "darkGray", "lightGray", "black",
]);
function runMarks(run, inherited) {
const marks = inherited ? inherited.slice() : [];
const add = (type, attrs) => marks.push(attrs ? { type, attrs } : { type });
if (run.isBold) add("strong");
if (run.isItalic) add("em");
if (run.isUnderline) add("underline");
if (run.isStrikethrough) add("strike");
if (run.isAllCaps) add("caps");
if (run.isSmallCaps) add("smallcaps");
if (run.verticalAlignment === "superscript") add("sup");
if (run.verticalAlignment === "subscript") add("sub");
if (run.font) add("font", { family: run.font });
if (run.fontSize) add("fsize", { pt: run.fontSize });
if (run.color && run.color !== "000000") add("color", { hex: run.color });
if (run.highlight && HIGHLIGHT_NAMES.has(run.highlight)) add("highlight", { name: run.highlight });
return marks;
}
// ---------------------------------------------------------- paragraphs ---
const HEADING_RE = /^heading\s*([1-6])$/i;
// Which block a paragraph becomes, from its Word style. styleName is what
// the user sees in the styles gallery; styleId is the internal one, and
// documents from non-Word producers often set only one of the two.
function classifyParagraph(p) {
const name = (p.styleName || "").trim();
const id = (p.styleId || "").trim();
const m = HEADING_RE.exec(name) || /^Heading([1-6])$/.exec(id);
if (m) return { kind: "heading", level: parseInt(m[1], 10) };
if (/^title$/i.test(name) || id === "Title") return { kind: "heading", level: 1 };
if (/^subtitle$/i.test(name) || id === "Subtitle") return { kind: "heading", level: 2 };
if (/quote$/i.test(name) || /Quote$/.test(id)) return { kind: "blockquote" };
if (/^(source code|html preformatted|code|plain text|preformatted text)$/i.test(name) ||
/^(SourceCode|HTMLPreformatted|PlainText)$/.test(id)) return { kind: "code_block" };
return { kind: "paragraph" };
}
function alignmentOf(p) {
const a = (p.alignment || "").toLowerCase();
if (a === "center") return "center";
if (a === "right" || a === "end") return "right";
if (a === "both" || a === "justify" || a === "distribute") return "justify";
if (a === "left" || a === "start") return "left";
return null;
}
function indentLevelOf(p) {
const twips = parseInt((p.indent && (p.indent.start)) || "0", 10);
if (!Number.isFinite(twips) || twips <= 0) return 0;
return Math.min(8, Math.round(twips / 720));
}
function spacingOf(p) {
const s = p.spacing;
if (!s) return { lineHeight: null, spaceBefore: null, spaceAfter: null };
// w:line is 240ths of a line under the "auto" rule; under exact/atLeast
// it's twips, which the editor has no control for, so it is left alone
// (and reported as lossy).
const lineHeight = s.line && (!s.lineRule || s.lineRule === "auto")
? Math.round((s.line / 240) * 100) / 100 : null;
const pt = (twips) => (twips == null ? null : Math.round((twips / 20) * 10) / 10);
return { lineHeight, spaceBefore: pt(s.before), spaceAfter: pt(s.after) };
}
function paragraphAttrs(p) {
return Object.assign({ align: alignmentOf(p), indent: indentLevelOf(p) }, spacingOf(p));
}
// --------------------------------------------------------------- walker ---
function Reader(options) {
this.warnings = [];
this.options = options || {};
}
Reader.prototype.warn = function (msg) {
if (!this.warnings.includes(msg)) this.warnings.push(msg);
};
// Inline children of a paragraph or table cell. Returns
// {inline: [...], breaks: [...]} — a page break inside a paragraph has to
// become a sibling block, so it is reported up rather than inlined.
Reader.prototype.inlineChildren = async function (children, marks) {
const out = [];
let sawPageBreak = false;
for (const child of children) {
switch (child.type) {
case "run": {
const sub = await this.inlineChildren(child.children, runMarks(child, marks));
out.push(...sub.inline);
sawPageBreak = sawPageBreak || sub.sawPageBreak;
break;
}
case "text": {
if (child.value) out.push({ type: "text", text: child.value, marks: marks.length ? marks : undefined });
break;
}
case "tab": {
out.push({ type: "text", text: "\t", marks: marks.length ? marks : undefined });
break;
}
case "checkbox": {
out.push({ type: "text", text: child.checked ? "☒" : "☐", marks: marks.length ? marks : undefined });
break;
}
case "break": {
if (child.breakType === "line") out.push({ type: "hard_break" });
else if (child.breakType === "page") sawPageBreak = true;
else if (child.breakType === "column") { sawPageBreak = true; this.warn("columns"); }
break;
}
case "hyperlink": {
const href = child.href || (child.anchor ? "#" + child.anchor : "");
const linkMark = { type: "link", attrs: { href, title: null, anchor: child.anchor || null } };
const sub = await this.inlineChildren(child.children, marks.concat([linkMark]));
out.push(...sub.inline);
sawPageBreak = sawPageBreak || sub.sawPageBreak;
break;
}
case "image": {
const node = await this.imageNode(child);
if (node) out.push(node);
break;
}
case "noteReference": {
out.push({
type: "note_ref",
attrs: {
noteType: child.noteType === "endnote" ? "endnote" : "footnote",
noteId: String(child.noteId),
label: child.noteType === "endnote" ? "†" : "*",
},
});
break;
}
case "commentReference":
this.warn("comments");
break;
case "bookmarkStart":
// Anchors for internal links. Dropped, but only worth mentioning
// when it isn't Word's own cursor-position bookmark.
if (child.name && !String(child.name).startsWith("_GoBack")) this.warn("bookmarks");
break;
default:
break;
}
}
return { inline: out, sawPageBreak };
};
Reader.prototype.imageNode = async function (image) {
let bytes;
try {
const buf = await image.readAsArrayBuffer();
bytes = new Uint8Array(buf);
} catch (e) {
this.warn("unreadable-image");
return null;
}
const size = imageSizePt(bytes, image);
const type = image.contentType || "image/png";
return {
type: "image",
attrs: {
src: `data:${type};base64,${bytesToBase64(bytes)}`,
alt: image.altText || null,
title: null,
width: size.width,
height: size.height,
},
};
};
// ----------------------------------------------------------- list stack ---
// Word has no list elements: every list item is a paragraph carrying a
// numbering id and a level. Rebuilding the nesting is on us.
function ListStack(out) {
this.out = out; // the block array lists get appended to
this.stack = []; // [{level, ordered, node}]
}
ListStack.prototype.flush = function () { this.stack.length = 0; };
ListStack.prototype.push = function (numbering, itemBlocks) {
const level = Math.max(0, Math.min(8, parseInt(numbering.level, 10) || 0));
const ordered = !!numbering.isOrdered;
const format = numbering.numFmt || (ordered ? "decimal" : null);
const numId = numbering.numId == null ? null : String(numbering.numId);
// Leaving a deeper level.
while (this.stack.length && this.stack[this.stack.length - 1].level > level) this.stack.pop();
let top = this.stack[this.stack.length - 1];
// Same level, but a different list. Word marks the boundary between two
// adjacent lists with a change of numbering id — it's the difference
// between "4. 5. 6." and a second list starting again at 1 — and a
// change of bullet-versus-number means the same thing.
if (top && top.level === level &&
(top.ordered !== ordered || (numId !== null && top.numId !== null && top.numId !== numId))) {
this.stack.pop();
top = this.stack[this.stack.length - 1];
}
if (!top || top.level < level) {
const node = ordered
? { type: "ordered_list", attrs: { order: 1, format: format || "decimal" }, content: [] }
: { type: "bullet_list", content: [] };
if (top) {
// Nested: the sub-list belongs inside the parent's last item.
let parentItems = top.node.content;
if (!parentItems.length) {
parentItems.push({ type: "list_item", content: [{ type: "paragraph", content: [] }] });
}
parentItems[parentItems.length - 1].content.push(node);
} else {
this.out.push(node);
}
this.stack.push({ level, ordered, numId, node });
top = this.stack[this.stack.length - 1];
}
top.node.content.push({ type: "list_item", content: itemBlocks });
};
// --------------------------------------------------------------- blocks ---
Reader.prototype.blocks = async function (children) {
const out = [];
const lists = new ListStack(out);
// Consecutive code-styled paragraphs read as one code block, the way
// they were almost certainly written.
let codeRun = null;
const closeCode = () => { codeRun = null; };
for (const child of children) {
if (child.type === "paragraph") {
const cls = classifyParagraph(child);
const { inline, sawPageBreak } = await this.inlineChildren(child.children, []);
if (cls.kind === "code_block") {
const text = inline.filter((n) => n.type === "text").map((n) => n.text).join("");
if (codeRun) codeRun.content.push({ type: "text", text: "\n" + text });
else {
codeRun = { type: "code_block", content: text ? [{ type: "text", text }] : [] };
lists.flush();
out.push(codeRun);
}
continue;
}
closeCode();
if (sawPageBreak) {
lists.flush();
out.push({ type: "page_break" });
// A paragraph that held nothing but the break IS the break; keeping
// the husk would grow the document by one blank line every save.
if (!inline.length) continue;
}
// An empty paragraph carrying only a bottom border is Word's
// horizontal rule (what AutoFormat makes from "---").
if (!inline.length && child.hasBottomBorder) {
lists.flush();
out.push({ type: "horizontal_rule" });
continue;
}
const attrs = paragraphAttrs(child);
let block;
if (cls.kind === "heading") {
block = { type: "heading", attrs: Object.assign({ level: cls.level }, attrs), content: inline };
} else {
block = { type: "paragraph", attrs, content: inline };
}
if (child.numbering) {
// List items don't carry their own indent — the list level owns it.
block.attrs = Object.assign({}, block.attrs, { indent: 0 });
lists.push(child.numbering, [block]);
continue;
}
lists.flush();
if (cls.kind === "blockquote") {
// The blockquote owns the indent; leaving it on the paragraph too
// would push the quote one level deeper on every round-trip.
block.attrs = Object.assign({}, block.attrs, { indent: 0 });
out.push({ type: "blockquote", content: [block] });
}
else out.push(block);
continue;
}
closeCode();
lists.flush();
if (child.type === "table") {
const table = await this.table(child);
if (table) out.push(table);
continue;
}
// Anything else at body level (bookmarks, stray runs) contributes no
// block of its own.
if (child.type === "bookmarkStart") continue;
}
return out;
};
Reader.prototype.table = async function (table) {
const rows = [];
for (const row of table.children) {
if (row.type !== "tableRow") continue;
const cells = [];
for (const cell of row.children) {
if (cell.type !== "tableCell") continue;
let content = await this.blocks(cell.children);
if (!content.length) content = [{ type: "paragraph", content: [] }];
// OOXML forbids a cell that ends with a table, so every document
// with a nested table carries an empty paragraph after it that the
// author never typed. Dropping it here keeps the cell stable across
// saves; the writer puts it back on the way out.
if (content.length > 1) {
const last = content[content.length - 1];
const prev = content[content.length - 2];
if (prev.type === "table" && last.type === "paragraph" && !(last.content || []).length) {
content.pop();
}
}
cells.push({
type: row.isHeader ? "table_header" : "table_cell",
attrs: {
colspan: cell.colSpan || 1,
rowspan: cell.rowSpan || 1,
colwidth: null,
background: null,
},
content,
});
}
// A row whose every column is covered by a merge from above has no
// cells of its own, and that's not an empty row to be thrown away —
// it's how both this model and HTML represent the middle of a
// vertical merge. Dropping it turns a 12-row merge into a 2-row one.
rows.push({ type: "table_row", content: cells });
}
if (!rows.length) return null;
return { type: "table", content: rows };
};
// ----------------------------------------------------------------- api ---
/**
* Read a .docx into an editor document.
*
* @param {ArrayBuffer|Uint8Array} bytes
* @param {object} schema the ProseMirror schema from schema.js
* @returns {Promise<{doc, report, meta, setup, warnings, messages}>}
*/
async function docxToDoc(bytes, schema) {
const { mammoth, pm } = V();
const { pkg } = globalThis.DocxEditor;
const arrayBuffer = bytes instanceof Uint8Array
? bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength)
: bytes;
const u8 = new Uint8Array(arrayBuffer);
const zip = await pkg.loadZip(u8);
const [report, setup, meta] = await Promise.all([
pkg.scan(zip), pkg.readSectionSetup(zip), pkg.readCoreProps(zip),
]);
let captured = null;
// mammoth takes {arrayBuffer} in the browser and {buffer} under node;
// the round-trip tests run under node against this same file.
const underNode = typeof process !== "undefined" && !!(process.versions && process.versions.node) &&
typeof Buffer !== "undefined";
const input = underNode ? { buffer: Buffer.from(u8) } : { arrayBuffer };
// The HTML this produces is thrown away — transformDocument is just the
// public seam that hands over the parsed model.
const result = await mammoth.convertToHtml(input, {
transformDocument: (document) => {
captured = document;
return document;
},
});
if (!captured) throw new Error("mammoth did not hand back a document model");
const reader = new Reader();
const blocks = await reader.blocks(captured.children);
if (!blocks.length) blocks.push({ type: "paragraph", content: [] });
const doc = pm.model.Node.fromJSON(schema, { type: "doc", content: blocks });
doc.check();
return {
doc,
report,
meta,
setup,
warnings: reader.warnings,
messages: (result.messages || []).map((m) => `${m.type}: ${m.message}`),
};
}
return { docxToDoc, imagePixelSize, imageSizePt, classifyParagraph, bytesToBase64 };
});