mirror of
https://github.com/mozilla/pdf.js.git
synced 2026-07-25 08:27:19 +02:00
Handle revisioned structure attributes
This commit is contained in:
parent
16290afa3b
commit
eb58ccc100
@ -709,12 +709,21 @@ class PDFEditor {
|
|||||||
}
|
}
|
||||||
for (let attr of attributes) {
|
for (let attr of attributes) {
|
||||||
attr = this.xrefWrapper.fetchIfRef(attr);
|
attr = this.xrefWrapper.fetchIfRef(attr);
|
||||||
|
if (!(attr instanceof Dict)) {
|
||||||
|
// An attribute array may interleave dictionaries and revision
|
||||||
|
// numbers (ISO 32000-2, 14.7.6.3).
|
||||||
|
continue;
|
||||||
|
}
|
||||||
if (isName(attr.get("O"), "Table") && attr.has("Headers")) {
|
if (isName(attr.get("O"), "Table") && attr.has("Headers")) {
|
||||||
const headers = this.xrefWrapper.fetchIfRef(attr.getRaw("Headers"));
|
const headers = this.xrefWrapper.fetchIfRef(attr.getRaw("Headers"));
|
||||||
if (Array.isArray(headers)) {
|
if (Array.isArray(headers)) {
|
||||||
for (let i = 0, ii = headers.length; i < ii; i++) {
|
for (let i = 0, ii = headers.length; i < ii; i++) {
|
||||||
|
const header = this.xrefWrapper.fetchIfRef(headers[i]);
|
||||||
|
if (typeof header !== "string") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
const newId = dedupIDs.get(
|
const newId = dedupIDs.get(
|
||||||
stringToPDFString(headers[i], /* keepEscapeSequence = */ false)
|
stringToPDFString(header, /* keepEscapeSequence = */ false)
|
||||||
);
|
);
|
||||||
if (newId) {
|
if (newId) {
|
||||||
headers[i] = newId;
|
headers[i] = newId;
|
||||||
|
|||||||
@ -6801,6 +6801,79 @@ small scripts as well as for`);
|
|||||||
await loadingTask.destroy();
|
await loadingTask.destroy();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves revisioned structure attributes", async function () {
|
||||||
|
const objects = [
|
||||||
|
"1 0 obj\n<< /Type /Catalog /Pages 2 0 R " +
|
||||||
|
"/StructTreeRoot 4 0 R >>\nendobj\n",
|
||||||
|
"2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n",
|
||||||
|
"3 0 obj\n<< /Type /Page /Parent 2 0 R /StructParents 0 " +
|
||||||
|
"/MediaBox [0 0 100 100] >>\nendobj\n",
|
||||||
|
"4 0 obj\n<< /Type /StructTreeRoot /K [5 0 R] " +
|
||||||
|
"/ParentTree 7 0 R /ParentTreeNextKey 1 >>\nendobj\n",
|
||||||
|
"5 0 obj\n<< /Type /StructElem /S /Document /P 4 0 R " +
|
||||||
|
"/K 6 0 R >>\nendobj\n",
|
||||||
|
"6 0 obj\n<< /Type /StructElem /S /P /P 5 0 R /Pg 3 0 R /K 0 " +
|
||||||
|
"/A [<< /O /Layout /Placement /Block >> 0] >>\nendobj\n",
|
||||||
|
"7 0 obj\n<< /Nums [0 [6 0 R]] >>\nendobj\n",
|
||||||
|
];
|
||||||
|
let loadingTask = getDocument({ data: assemblePdf(objects) });
|
||||||
|
let pdfDoc = await loadingTask.promise;
|
||||||
|
const data = await pdfDoc.extractPages([{ document: null }]);
|
||||||
|
await loadingTask.destroy();
|
||||||
|
|
||||||
|
const pdfString = bytesToString(data);
|
||||||
|
expect(pdfString).toContain("/O /Layout");
|
||||||
|
expect(pdfString).toContain("/Placement");
|
||||||
|
expect(pdfString).toMatch(
|
||||||
|
/\/A\s+\[<<\s+\/O\s+\/Layout\s+\/Placement\s+\/Block>>\s+0\]/
|
||||||
|
);
|
||||||
|
|
||||||
|
loadingTask = getDocument({ data });
|
||||||
|
pdfDoc = await loadingTask.promise;
|
||||||
|
const tree = await (await pdfDoc.getPage(1)).getStructTree();
|
||||||
|
expect(tree.children[0].role).toEqual("Document");
|
||||||
|
expect(tree.children[0].children[0].role).toEqual("P");
|
||||||
|
await loadingTask.destroy();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("remaps indirect table header ids when deduplicating struct trees", async function () {
|
||||||
|
const objects = [
|
||||||
|
"1 0 obj\n<< /Type /Catalog /Pages 2 0 R " +
|
||||||
|
"/StructTreeRoot 4 0 R >>\nendobj\n",
|
||||||
|
"2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n",
|
||||||
|
"3 0 obj\n<< /Type /Page /Parent 2 0 R /StructParents 0 " +
|
||||||
|
"/MediaBox [0 0 100 100] >>\nendobj\n",
|
||||||
|
"4 0 obj\n<< /Type /StructTreeRoot /K [5 0 R] " +
|
||||||
|
"/ParentTree 9 0 R /ParentTreeNextKey 1 /IDTree 10 0 R >>\nendobj\n",
|
||||||
|
"5 0 obj\n<< /Type /StructElem /S /Table /P 4 0 R /Pg 3 0 R " +
|
||||||
|
"/K [6 0 R 7 0 R] >>\nendobj\n",
|
||||||
|
"6 0 obj\n<< /Type /StructElem /S /TH /P 5 0 R /Pg 3 0 R " +
|
||||||
|
"/ID (h1) /K 0 >>\nendobj\n",
|
||||||
|
"7 0 obj\n<< /Type /StructElem /S /TD /P 5 0 R /Pg 3 0 R /K 1 " +
|
||||||
|
"/A << /O /Table /Headers [8 0 R] >> >>\nendobj\n",
|
||||||
|
"8 0 obj\n(h1)\nendobj\n",
|
||||||
|
"9 0 obj\n<< /Nums [0 [6 0 R 7 0 R]] >>\nendobj\n",
|
||||||
|
"10 0 obj\n<< /Names [(h1) 6 0 R] >>\nendobj\n",
|
||||||
|
];
|
||||||
|
|
||||||
|
let loadingTask = getDocument({ data: assemblePdf(objects) });
|
||||||
|
let pdfDoc = await loadingTask.promise;
|
||||||
|
const data = await pdfDoc.extractPages([
|
||||||
|
{ document: null },
|
||||||
|
{ document: null },
|
||||||
|
]);
|
||||||
|
await loadingTask.destroy();
|
||||||
|
|
||||||
|
const pdfString = bytesToString(data);
|
||||||
|
expect(pdfString).toContain("/ID (h1_1)");
|
||||||
|
expect(pdfString).toContain("/Headers [(h1_1)]");
|
||||||
|
|
||||||
|
loadingTask = getDocument({ data });
|
||||||
|
pdfDoc = await loadingTask.promise;
|
||||||
|
expect(pdfDoc.numPages).toEqual(2);
|
||||||
|
await loadingTask.destroy();
|
||||||
|
});
|
||||||
|
|
||||||
it("extract pages and merge struct trees", async function () {
|
it("extract pages and merge struct trees", async function () {
|
||||||
let loadingTask = getDocument(
|
let loadingTask = getDocument(
|
||||||
buildGetDocumentParams("two_paragraphs.pdf")
|
buildGetDocumentParams("two_paragraphs.pdf")
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user