Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions pdfbox/pom.xml
Original file line number Diff line number Diff line change
Expand Up @@ -1098,6 +1098,19 @@
<sha512>f3d0934c9d0babedd76431565bdb1cc1bf6d9f0a5ace7656a9ded53fb1e701dd0d12c784a58ae87098793f57ba5b98cbfb657e047e27d5464b26634511162264</sha512>
</configuration>
</execution>
<execution>
<id>PDFBOX-5953</id>
<phase>generate-test-resources</phase>
<goals>
<goal>wget</goal>
</goals>
<configuration>
<url>https://issues.apache.org/jira/secure/attachment/13074663/test.pdf</url>
<outputDirectory>${project.build.directory}/pdfs</outputDirectory>
<outputFileName>PDFBOX-5953-test.pdf</outputFileName>
<sha512>022755d4abeaa60ed99b3252441aaa1b65daa8d5414348f6c4ae5c8e3dfdb7d2be5ea0705117b8bfbb88f13e79d168dcb1f3b2ff9d9c7be21ad987965ccfc562</sha512>
</configuration>
</execution>
</executions>
</plugin>
</plugins>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -398,6 +398,20 @@ protected byte[] encode(int unicode, PDType0Font parent)
cid = 0;
}
}
else if (!parent.getCMap().getName().startsWith("Identity-") &&
parent.getCMap().getName().startsWith("Uni") &&
parent.getCMapUCS2() != null)
{
// PDFBOX-5953: predefined "Uni...-UCS2/UTF16" CMaps use the raw UTF-16BE
// Unicode value as their codespace/code by definition (see e.g. the
// UniGB-UTF16-H resource, whose begincidchar entries map codes that are
// themselves Unicode code points, e.g. <00a4> to a CID). Writing the
// substituted font's own glyph index here instead (as done below for the
// general case) produces a code that decodes to a wrong, essentially
// arbitrary character once the CID/Unicode round trip in
// PDCIDFontType2#codeToGID or PDType0Font#toUnicode is applied to it.
cid = unicode;
}
else
{
// a non-embedded font always has a cmap (otherwise it we wouldn't load it)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,84 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership.
* The ASF licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package org.apache.pdfbox.pdmodel.font;

import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assumptions.assumeTrue;

import java.io.File;
import java.io.IOException;

import org.apache.fontbox.ttf.CmapLookup;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;

import org.junit.jupiter.api.Test;

/**
* A non-embedded CIDFontType2 whose declared /Encoding is a predefined "Uni...-UCS2/UTF16" CMap
* must write a code that is valid under that encoding - the raw Unicode value - when generating
* new appearance content, not the substituted font's own glyph index. Before the fix, the
* substitute font's glyph index was written regardless of the declared encoding; since that
* index is unrelated to the Unicode-based codespace of a "Uni...-UTF16-H" CMap, a conforming
* reader (including PDFBox itself) decoded it back into a wrong, essentially arbitrary
* character. PDFBOX-5953.
*
* <p>The fixture is the AcroForm/DR "SimSun" font (Type0/CIDFontType2, /Encoding
* UniGB-UTF16-H, not embedded) of a real-world Chinese bank receipt form attached to the JIRA
* issue, whose filled-in field values (account numbers, amounts, company names) render as
* garbled or blank text because of this bug.
*/
class PDCIDFontType2SubstituteTest
{
@Test
void testEncodeNonEmbeddedPredefinedUnicodeEncoding() throws IOException
{
File file = new File("target/pdfs", "PDFBOX-5953-test.pdf");
try (PDDocument doc = Loader.loadPDF(file))
{
PDAcroForm acroForm = doc.getDocumentCatalog().getAcroForm();
PDType0Font font = (PDType0Font) acroForm.getDefaultResources()
.getFont(COSName.getPDFName("SimSun"));
PDCIDFontType2 cidFont = (PDCIDFontType2) font.getDescendantFont();
assumeTrue(!cidFont.isEmbedded(),
"font is embedded in this fixture, can't test substitution");
assumeTrue(font.getCMap().getName().startsWith("Uni"),
"fixture font's encoding changed, can't test");

CmapLookup substituteCmap = cidFont.getTrueTypeFont().getUnicodeCmapLookup(false);
// '百' (as in 百色, the city in the receipts): a common GB1 Hanzi, guaranteed to be
// resolvable through the Adobe-GB1 <-> Unicode predefined CMaps
int unicode = 0x767E;
int expectedGid = substituteCmap.getGlyphId(unicode);
assumeTrue(expectedGid != 0,
"no CJK glyph for U+767E in the substituted font, can't test");

byte[] bytes = cidFont.encode(unicode, font);
assertEquals(2, bytes.length);
int code = ((bytes[0] & 0xff) << 8) | (bytes[1] & 0xff);

// before the fix, 'code' was the substitute font's own glyph index; decoding it
// again (as any reader must, since the font's /Encoding is UniGB-UTF16-H, not
// Identity-H) produced a wrong, essentially arbitrary glyph instead of round
// tripping back to the glyph for U+767E
assertEquals(expectedGid, cidFont.codeToGID(code, font),
"encoded code must decode back to the substitute font's glyph for U+767E");
}
}
}