From 6341ea91f4e30aa02dae5237035f858b6d25ccbf Mon Sep 17 00:00:00 2001 From: Jason Harrop Date: Sat, 3 Oct 2026 07:24:16 +1000 Subject: [PATCH] FOP-3346: ToUnicode selectors no longer drift by one after a supplementary-plane character CIDSubset.getChars builds a char[] with StringBuilder.appendCodePoint, so a character outside the BMP occupies two slots. PDFToUnicodeCMap derived each character selector from the array position, so every selector after such a character was written one too high, and the text after an emoji or a mathematical letter extracted as its neighbour. Measured on this branch, DejaVu Math TeX Gyre, the text "A" U+1D400 "BZ". The content stream uses selectors 3 4 5 6; the CMap said before <0003> <0041> <0004> <0006> <0042> <0007> <005a> after <0003> <0041> <0004> <0005> <0042> <0006> <005a> Before, selector 5 (the B) had no entry and selector 6 (the Z) was published as B. The CMap is now built from one destination per selector, a String, so a surrogate pair is one entry of length two rather than two array positions. The char[] constructor converts (toDestinations) and callers are unchanged. The range logic needs no surrogate special cases any more: an entry may join a bfrange when it is one code point, and two entries are consecutive when their code points are and their selectors share a 256 block. A destination of several characters is written as a bfchar with a string, which the format allows and FOP-3345 uses. PDFToUnicodeCMapTestCase pinned the drift (surrogatePairTest expected the entry after the pair at 0x63 to be 0x65); its expectations change accordingly, in surrogatePairTest, surrogatePairRangeTest, surrogatePairsRangeTest and rangeSizeSurrogateTest, the last of which also ran its low surrogates past U+DFFF and now starts them at U+DC00. ToUnicodeCharacterisationTestCase records the writer's output for the common shapes, so a change to the range packing has to be deliberate. Co-Authored-By: Claude Fable 5.1 --- .../org/apache/fop/pdf/PDFToUnicodeCMap.java | 377 +++++++----------- .../fop/pdf/PDFToUnicodeCMapTestCase.java | 40 +- .../ToUnicodeCharacterisationTestCase.java | 151 +++++++ 3 files changed, 329 insertions(+), 239 deletions(-) create mode 100644 fop-core/src/test/java/org/apache/fop/pdf/ToUnicodeCharacterisationTestCase.java diff --git a/fop-core/src/main/java/org/apache/fop/pdf/PDFToUnicodeCMap.java b/fop-core/src/main/java/org/apache/fop/pdf/PDFToUnicodeCMap.java index ad41e03de82..ea17da69bc3 100644 --- a/fop-core/src/main/java/org/apache/fop/pdf/PDFToUnicodeCMap.java +++ b/fop-core/src/main/java/org/apache/fop/pdf/PDFToUnicodeCMap.java @@ -42,10 +42,13 @@ public class PDFToUnicodeCMap extends PDFCMap { /** - * The array of Unicode characters ordered by character code - * (maps from character code to Unicode code point). + * One destination per character selector, in selector order: the UTF-16 text the glyph + * stands for. Usually one code point; several for a ligature or other glyph produced from + * more than one character; empty for a glyph whose text is carried by a neighbouring glyph; + * a lone high surrogate for an unpaired one, which is reported and written with a zero + * low surrogate. */ - protected char[] unicodeCharMap; + protected String[] destinations; private boolean singleByte; @@ -54,26 +57,70 @@ public class PDFToUnicodeCMap extends PDFCMap { /** * Constructor. * - * @param unicodeCharMap An array of Unicode characters ordered by character code - * (maps from character code to Unicode code point) + * @param destinations One destination string per character selector, in selector order * @param name One of the registered names found in Table 5.14 in PDF * Reference, Second Edition. * @param sysInfo The attributes of the character collection of the CIDFont. * @param singleByte true for single-byte, false for double-byte * @param eventBroadcaster Event broadcaster. May be null. */ - public PDFToUnicodeCMap(char[] unicodeCharMap, String name, PDFCIDSystemInfo sysInfo, + public PDFToUnicodeCMap(String[] destinations, String name, PDFCIDSystemInfo sysInfo, boolean singleByte, EventBroadcaster eventBroadcaster) { super(name, sysInfo); - if (singleByte && unicodeCharMap.length > 256) { + if (singleByte && destinations.length > 256) { throw new IllegalArgumentException("unicodeCharMap may not contain more than" + " 256 characters for single-byte encodings"); } - this.unicodeCharMap = unicodeCharMap; + this.destinations = destinations; this.singleByte = singleByte; this.eventBroadcaster = eventBroadcaster; } + /** + * Constructor from a positional array of UTF-16 code units, where a surrogate pair + * occupies two slots and stands for one character selector. + * + * @param unicodeCharMap An array of Unicode characters ordered by character code + * (maps from character code to Unicode code point) + * @param name One of the registered names found in Table 5.14 in PDF + * Reference, Second Edition. + * @param sysInfo The attributes of the character collection of the CIDFont. + * @param singleByte true for single-byte, false for double-byte + * @param eventBroadcaster Event broadcaster. May be null. + */ + public PDFToUnicodeCMap(char[] unicodeCharMap, String name, PDFCIDSystemInfo sysInfo, + boolean singleByte, EventBroadcaster eventBroadcaster) { + this(toDestinations(unicodeCharMap), name, sysInfo, singleByte, eventBroadcaster); + } + + /** + * Turns a positional array of UTF-16 code units into one destination per character + * selector: a high surrogate takes the unit after it as its low surrogate, and a high + * surrogate at the end of the array stands alone. + * @param unicodeCharMap the positional array + * @return one destination per selector + */ + public static String[] toDestinations(char[] unicodeCharMap) { + int count = 0; + for (int i = 0; i < unicodeCharMap.length; i++) { + if (isHighSurrogate(unicodeCharMap[i]) && i + 1 < unicodeCharMap.length) { + i++; + } + count++; + } + String[] destinations = new String[count]; + int d = 0; + for (int i = 0; i < unicodeCharMap.length; i++) { + if (isHighSurrogate(unicodeCharMap[i]) && i + 1 < unicodeCharMap.length) { + destinations[d++] = new String(unicodeCharMap, i, 2); + i++; + } else { + destinations[d++] = String.valueOf(unicodeCharMap[i]); + } + } + return destinations; + } + /** {@inheritDoc} */ protected CMapBuilder createCMapBuilder(Writer writer) { return new ToUnicodeCMapBuilder(writer); @@ -103,104 +150,64 @@ public void writeCMap() throws IOException { * Writes the character mappings for this font. */ protected void writeBFEntries() throws IOException { - if (unicodeCharMap != null) { - writeBFCharEntries(unicodeCharMap); - writeBFRangeEntries(unicodeCharMap); + if (destinations != null) { + writeBFCharEntries(); + writeBFRangeEntries(); } } /** - * Writes the entries for single characters of a base font (only characters which cannot be - * expressed as part of a character range). - * @param charArray all the characters to map - * @throws IOException + * Writes the entries for single selectors (those which cannot be expressed as part of + * a range), in sections of at most 100. + * @throws IOException if an I/O error occurs */ - protected void writeBFCharEntries(char[] charArray) throws IOException { + protected void writeBFCharEntries() throws IOException { int totalEntries = 0; - int charIndex = 0; - if (charArray.length > 0) { - do { - if (!partOfRange(charArray, charIndex)) { - totalEntries++; - } - if (isHighSurrogate(charArray[charIndex])) { - charIndex++; - } - } while (++charIndex < charArray.length); + for (int i = 0; i < destinations.length; i++) { + if (!partOfRange(i)) { + totalEntries++; + } } if (totalEntries < 1) { return; } int remainingEntries = totalEntries; - charIndex = 0; + int index = 0; do { /* Limited to 100 entries in each section */ int entriesThisSection = Math.min(remainingEntries, 100); writer.write(entriesThisSection + " beginbfchar\n"); int sectionEntryCount = 0; do { - /* Go to the next char not in a range */ - while (partOfRange(charArray, charIndex)) { - if (isHighSurrogate(charArray[charIndex])) { - charIndex++; - } - charIndex++; + /* Go to the next selector not in a range */ + while (partOfRange(index)) { + index++; } - - writer.write("<" + padCharIndex(charIndex) + "> "); - - if (isHighSurrogate(charArray[charIndex])) { - char secondChar = 0; // Invalid low surrogate (valid: 0xDC00 - 0xDFFF) - if (charIndex + 1 < charArray.length) { - secondChar = charArray[charIndex + 1]; - } else { - if (eventBroadcaster != null) { - PDFEventProducer pdfEventProducer = PDFEventProducer.Provider.get(eventBroadcaster); - pdfEventProducer.unpairedSurrogate(this); - } - } - writer.write("<" + padHexString(Integer.toHexString(charArray[charIndex]), 4) - + padHexString(Integer.toHexString(secondChar), 4) + ">\n"); - charIndex++; - } else { - writer.write("<" + padHexString(Integer.toHexString(charArray[charIndex]), 4) - + ">\n"); - } - charIndex++; + writer.write("<" + padSelector(index) + "> "); + writer.write("<" + destinationHex(index) + ">\n"); + index++; } while (++sectionEntryCount < entriesThisSection); - remainingEntries -= entriesThisSection; writer.write("endbfchar\n"); } while (remainingEntries > 0); } - private String padCharIndex(int charIndex) { - return padHexString(Integer.toHexString(charIndex), (singleByte ? 2 : 4)); - } - /** - * Writes the entries for character ranges for a base font. - * @param charArray all the characters to map - * @throws IOException + * Writes the entries for selector ranges, in sections of at most 100. + * @throws IOException if an I/O error occurs */ - protected void writeBFRangeEntries(char[] charArray) throws IOException { + protected void writeBFRangeEntries() throws IOException { int totalEntries = 0; - int charIndex = 0; - if (charArray.length > 0) { - do { - if (startOfRange(charArray, charIndex)) { - totalEntries++; - } - if (isHighSurrogate(charArray[charIndex])) { - charIndex++; - } - } while (++charIndex < charArray.length); + for (int i = 0; i < destinations.length; i++) { + if (startOfRange(i)) { + totalEntries++; + } } if (totalEntries < 1) { return; } int remainingEntries = totalEntries; - charIndex = 0; + int index = 0; do { /* Limited to 100 entries in each section */ int entriesThisSection = Math.min(remainingEntries, 100); @@ -208,182 +215,106 @@ protected void writeBFRangeEntries(char[] charArray) throws IOException { int sectionEntryCount = 0; do { /* Go to the next start of a range */ - while (!startOfRange(charArray, charIndex)) { - if (isHighSurrogate(charArray[charIndex])) { - charIndex++; - } - charIndex++; + while (!startOfRange(index)) { + index++; } - writer.write("<" + padCharIndex(charIndex) + "> "); - writer.write("<" - + padCharIndex(endOfRange(charArray, charIndex)) - + "> "); - if (isHighSurrogate(charArray[charIndex])) { - char secondChar = 0; - if (charIndex + 1 < charArray.length) { - secondChar = charArray[charIndex + 1]; - } else { - if (eventBroadcaster != null) { - PDFEventProducer pdfEventProducer = PDFEventProducer.Provider.get(eventBroadcaster); - pdfEventProducer.unpairedSurrogate(this); - } - } - writer.write("<" + padHexString(Integer.toHexString(charArray[charIndex]), 4) - + padHexString(Integer.toHexString(secondChar), 4) - + ">\n"); - } else { - writer.write("<" + padHexString(Integer.toHexString(charArray[charIndex]), 4) - + ">\n"); - } - charIndex++; + writer.write("<" + padSelector(index) + "> "); + writer.write("<" + padSelector(endOfRange(index)) + "> "); + writer.write("<" + destinationHex(index) + ">\n"); + index++; } while (++sectionEntryCount < entriesThisSection); remainingEntries -= entriesThisSection; writer.write("endbfrange\n"); } while (remainingEntries > 0); } + private String padSelector(int index) { + return padHexString(Integer.toHexString(index), (singleByte ? 2 : 4)); + } + /** - * Find the end of the current range. - * @param charArray The array which is being tested. - * @param startOfRange The index to the array element that is the start of - * the range. - * @return The index to the element that is the end of the range. + * The destination of a selector as UTF-16BE hex, four lower-case digits per code unit. + * A lone high surrogate is reported and written with a zero low surrogate, as before. */ - private int endOfRange(char[] charArray, int startOfRange) { - int i = startOfRange; - if (isHighSurrogate(charArray[i])) { - while (i < charArray.length - 3 && sameRangeEntryAsNext(charArray, i)) { - i += 2; - } - } else { - while (i < charArray.length - 1 && sameRangeEntryAsNext(charArray, i)) { - i++; + private String destinationHex(int index) { + String d = destinations[index]; + StringBuilder hex = new StringBuilder(4 * Math.max(1, d.length())); + for (int i = 0; i < d.length(); i++) { + hex.append(padHexString(Integer.toHexString(d.charAt(i)), 4)); + } + if (d.length() == 1 && isHighSurrogate(d.charAt(0))) { + if (eventBroadcaster != null) { + PDFEventProducer pdfEventProducer = PDFEventProducer.Provider.get(eventBroadcaster); + pdfEventProducer.unpairedSurrogate(this); } + hex.append("0000"); } - return i; + return hex.toString(); } /** - * Determine whether this array element should be part of a bfchar entry or - * a bfrange entry. - * @param charArray The array to be tested. - * @param arrayIndex The index to the array element to be tested. - * @return True if this array element should be included in a range. + * The value a destination contributes to a range, or -1 if it can be in no range. Only + * a destination of exactly one code point can: a single non-surrogate code unit, or a + * high surrogate followed by one more unit. Two destinations are consecutive when their + * keys differ by one, which for a pair means the same high surrogate and the next low. */ - private boolean partOfRange(char[] charArray, int arrayIndex) { - int minBytesInRange = 2; - if (isHighSurrogate(charArray[arrayIndex])) { - minBytesInRange = 4; + private long rangeKey(int index) { + String d = destinations[index]; + if (d.length() == 1 && !Character.isSurrogate(d.charAt(0))) { + return d.charAt(0); + } else if (d.length() == 2 && isHighSurrogate(d.charAt(0))) { + return ((long) d.charAt(0) << 16) | d.charAt(1); + } else { + return -1; } - if (charArray.length < minBytesInRange) { + } + + /** + * Determine whether two consecutive selectors can be in the same bfrange entry: both + * destinations are one code point, the second is the next code point, and the two + * selectors are in the same block of 256, since only the low byte may vary in a range. + * @param index the first of the two selectors + * @return true if both are in the same range + */ + private boolean sameRangeEntryAsNext(int index) { + if (index < 0 || index >= destinations.length - 1) { return false; } - if (arrayIndex == 0) { - return sameRangeEntryAsNext(charArray, 0); - } - if (isHighSurrogate(charArray[arrayIndex])) { - if (arrayIndex == charArray.length - 2) { - return sameRangeEntryAsNext(charArray, arrayIndex - 2); - } - } - if (arrayIndex == charArray.length - 1) { - return sameRangeEntryAsNext(charArray, arrayIndex - 1); - } - if (isHighSurrogate(charArray[arrayIndex])) { - if (sameRangeEntryAsNext(charArray, arrayIndex - 2)) { - return true; - } - } - if (sameRangeEntryAsNext(charArray, arrayIndex - 1)) { - return true; - } - if (sameRangeEntryAsNext(charArray, arrayIndex)) { - return true; - } - return false; + long key = rangeKey(index); + return key >= 0 && rangeKey(index + 1) == key + 1 + && index / 256 == (index + 1) / 256; } /** - * Determine whether two code points can be included in the same bfrange entry. - * Range sizes are limited to a maximum of 256 (128 for surrogate pairs). - * @param charArray The array holding the code points to be tested. - * @param firstItem The first char of the first code point in the array to be tested. - * The first byte of the second code point is firstItem + n, where n is the number - * of chars in the firstItem code point. - * @return True if both: - * 1) the next code point in the array is sequential with this one, and - * 2) this code point and the next are both NOT surrogate pairs - * or - * this code point and the next are both surrogate pairs and - * the high-surrogates are the same, and - * 3) the resulting range cannot be greater than 256 in size. + * Determine whether this selector should be part of a bfrange entry rather than a + * bfchar entry. + * @param index the selector + * @return true if it is in a range */ - private boolean sameRangeEntryAsNext(char[] charArray, int firstItem) { - boolean retval = false; - do { - if (firstItem < 0 || firstItem >= charArray.length - 1) { - break; - } - if (isHighSurrogate(charArray[firstItem])) { - if (firstItem < charArray.length - 3) { - if (charArray[firstItem + 2] == charArray[firstItem]) { - if (charArray[firstItem + 3] == charArray[firstItem + 1] + 1) { - if (firstItem / 256 == (firstItem + 2) / 256) { - retval = true; - } - } - } - } - } else { - if (charArray[firstItem] + 1 == charArray[firstItem + 1]) { - if (firstItem / 256 == (firstItem + 1) / 256) { - retval = true; - } - } - } - } while (false); - return retval; + private boolean partOfRange(int index) { + return sameRangeEntryAsNext(index - 1) || sameRangeEntryAsNext(index); } /** - * Determine whether this array element should be the start of a bfrange - * entry. - * @param charArray The array to be tested. - * @param arrayIndex The index to the array element to be tested. - * @return True if this array element is the beginning of a range. + * Determine whether this selector starts a bfrange entry. + * @param index the selector + * @return true if it is the first of a range */ - private boolean startOfRange(char[] charArray, int arrayIndex) { - // Can't be the start of a range if not part of a range. - if (!partOfRange(charArray, arrayIndex)) { - return false; - } - // If part of a range and first element in the array, must be start of a range - if (arrayIndex == 0) { - return true; - } - // If last element in the array, cannot be start of a range - if (isHighSurrogate(charArray[arrayIndex])) { - if (arrayIndex == charArray.length - 2) { - return false; - } - } - if (arrayIndex == charArray.length - 1) { - return false; - } - /* - * If part of same range as the previous element is, cannot be start - * of range. - */ - if (isHighSurrogate(charArray[arrayIndex])) { - if (sameRangeEntryAsNext(charArray, arrayIndex - 2)) { - return false; - } - } - if (sameRangeEntryAsNext(charArray, arrayIndex - 1)) { - return false; + private boolean startOfRange(int index) { + return sameRangeEntryAsNext(index) && !sameRangeEntryAsNext(index - 1); + } + + /** + * Find the end of the range that starts at a selector. + * @param startOfRange the selector that starts the range + * @return the last selector of the range + */ + private int endOfRange(int startOfRange) { + int i = startOfRange; + while (sameRangeEntryAsNext(i)) { + i++; } - // Otherwise, this is start of a range. - return true; + return i; } /** diff --git a/fop-core/src/test/java/org/apache/fop/pdf/PDFToUnicodeCMapTestCase.java b/fop-core/src/test/java/org/apache/fop/pdf/PDFToUnicodeCMapTestCase.java index 70545d327c8..dd1e6f1eecf 100644 --- a/fop-core/src/test/java/org/apache/fop/pdf/PDFToUnicodeCMapTestCase.java +++ b/fop-core/src/test/java/org/apache/fop/pdf/PDFToUnicodeCMapTestCase.java @@ -179,6 +179,10 @@ public void rangeTest() throws IOException { /** * Checks that one surrogate pair is correctly handled, even when it crosses a section boundary. + * The pair is one character selector, so the selector after it is 0x64, not 0x65: the writer + * once numbered selectors by array position, which put every entry after a pair one selector + * too high (FOP-3346; measured on a rendered PDF, where the letter after a supplementary-plane + * character extracted as the letter after that). * @throws IOException */ @Test @@ -201,22 +205,23 @@ public void surrogatePairTest() throws IOException { + "<63> \n" + "endbfchar\n" + "56 beginbfchar\n" - + "<65> <00fc>\n" - + "<66> <00fe>"); + + "<64> <00fc>\n" + + "<65> <00fe>"); configPairs.put(false, "<0060> <00f2>\n" + "<0061> <00f4>\n" + "<0062> <00f6>\n" + "<0063> \n" + "endbfchar\n" + "56 beginbfchar\n" - + "<0065> <00fc>\n" - + "<0066> <00fe>"); + + "<0064> <00fc>\n" + + "<0065> <00fe>"); buildAndAssert(unicodeCharMap, configPairs); } /** - * Checks that a range of surrogate pairs is correctly handled. + * Checks that a range of surrogate pairs is correctly handled. Two pairs are two selectors, + * 9 and 10 (see surrogatePairTest). * @throws IOException */ @Test @@ -236,17 +241,18 @@ public void surrogatePairRangeTest() throws IOException { Map configPairs = new HashMap<>(); configPairs.put(true, "1 beginbfrange\n" - + "<09> <0b> \n" + + "<09> <0a> \n" + "endbfrange"); configPairs.put(false, "1 beginbfrange\n" - + "<0009> <000b> \n" + + "<0009> <000a> \n" + "endbfrange"); buildAndAssert(unicodeCharMap, configPairs); } /** - * Checks that CMap is correct, even when made up of just one range of surrogate pairs. + * Checks that CMap is correct, even when made up of just one range of surrogate pairs. Ten + * pairs are selectors 0 to 9 (see surrogatePairTest). * @throws IOException */ @Test @@ -264,10 +270,10 @@ public void surrogatePairsRangeTest() throws IOException { Map configPairs = new HashMap<>(); configPairs.put(true, "1 beginbfrange\n" - + "<00> <12> \n" + + "<00> <09> \n" + "endbfrange"); configPairs.put(false, "1 beginbfrange\n" - + "<0000> <0012> \n" + + "<0000> <0009> \n" + "endbfrange"); buildAndAssert(unicodeCharMap, configPairs); @@ -351,12 +357,14 @@ public void rangeSizeTest() throws IOException { } /** - * Checks that a range of surrogate pairs is limited in size. + * Checks that a range of surrogate pairs is limited in size: 256 selectors, the same as for + * any other range, since a pair is one selector (see surrogatePairTest). The low surrogates + * start at U+DC00 so that 300 of them stay valid. * @throws IOException */ @Test public void rangeSizeSurrogateTest() throws IOException { - final int charMapSize = 300; + final int charMapSize = 600; char[] unicodeCharMap = new char[charMapSize]; @@ -364,14 +372,14 @@ public void rangeSizeSurrogateTest() throws IOException { unicodeCharMap[i] = '\uD83C'; } for (int i = 0; i < charMapSize / 2; ++i) { - unicodeCharMap[1 + i * 2] = (char)('\uDF65' + i); + unicodeCharMap[1 + i * 2] = (char)('\uDC00' + i); } Map configPairs = new HashMap<>(); - // PDFToUnicodeCMap CTOR rejects unicodeCharMap with > 256 elements where singleByte is true. + // PDFToUnicodeCMap CTOR rejects a map of more than 256 selectors where singleByte is true. configPairs.put(false, "2 beginbfrange\n" - + "<0000> <00fe> \n" - + "<0100> <012a> \n" + + "<0000> <00ff> \n" + + "<0100> <012b> \n" + "endbfrange"); buildAndAssert(unicodeCharMap, configPairs); diff --git a/fop-core/src/test/java/org/apache/fop/pdf/ToUnicodeCharacterisationTestCase.java b/fop-core/src/test/java/org/apache/fop/pdf/ToUnicodeCharacterisationTestCase.java new file mode 100644 index 00000000000..4a144cc6b40 --- /dev/null +++ b/fop-core/src/test/java/org/apache/fop/pdf/ToUnicodeCharacterisationTestCase.java @@ -0,0 +1,151 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* $Id$ */ + +package org.apache.fop.pdf; + +import java.io.CharArrayWriter; +import java.io.IOException; + +import org.junit.Test; +import static org.junit.Assert.assertEquals; + +/** + * Pins the exact ToUnicode CMap this writer produces today. + * + *

These are characterisation tests, not specifications: they record current output so a + * change to the writer has to be deliberate. Every PDF FOP produces gets its text layer + * from here, so a silent change to the range packing would be invisible in rendering and + * visible only to search, copy and paste, and screen readers. FOP-3345 and FOP-3346 touch + * this class. + * + *

If one of these fails, decide whether the new output is correct before updating it. + * Do not update the expectation to make the build pass.

+ */ +public class ToUnicodeCharacterisationTestCase { + + private String cmapOf(char[] chars, boolean singleByte) throws IOException { + PDFToUnicodeCMap cmap = new PDFToUnicodeCMap(chars, PDFCMap.ENC_IDENTITY_H, + new PDFCIDSystemInfo("Adobe", "Identity", 0), singleByte, null); + CharArrayWriter writer = new CharArrayWriter(); + cmap.createCMapBuilder(writer).writeCMap(); + return writer.toString(); + } + + private String body(String cmap) { + int from = cmap.indexOf("endcodespacerange\n"); + int to = cmap.indexOf("endcmap"); + return cmap.substring(from + "endcodespacerange\n".length(), to); + } + + /** Contiguous code points pack into a single range. */ + @Test + public void testContiguousRunBecomesOneRange() throws IOException { + assertEquals("1 beginbfrange\n<0000> <0003> <0041>\nendbfrange\n", + body(cmapOf(new char[] {'A', 'B', 'C', 'D'}, false))); + } + + /** Isolated code points are written one by one. */ + @Test + public void testScatteredCodePointsBecomeChars() throws IOException { + assertEquals("3 beginbfchar\n<0000> <0041>\n<0001> <005a>\n<0002> <0072>\nendbfchar\n", + body(cmapOf(new char[] {'A', 'Z', 'r'}, false))); + } + + /** A surrogate pair is one code point across two array slots, and still ranges. */ + @Test + public void testSurrogatePairIsOneCodePoint() throws IOException { + assertEquals("1 beginbfchar\n<0000> \nendbfchar\n", + body(cmapOf(new char[] {'\uD800', '\uDF00'}, false))); + } + + /** + * The defect FOP-3345 addresses, pinned as it stands: a ligature glyph carries a + * private-use code point, and consecutive ones pack into a range, so the text layer + * says U+E000 upward rather than the letters. + */ + @Test + public void testPrivateUseLigaturesRangeTogetherToday() throws IOException { + assertEquals("1 beginbfrange\n<0000> <0002> \nendbfrange\n", + body(cmapOf(new char[] {'', '', ''}, false))); + } + + /** + * Destinations are written in lower-case hex while the code space is upper-case. Pinned + * because it is a byte-level property of every PDF FOP writes, and easy to change by + * accident when reworking the writer. + */ + @Test + public void testHexCaseIsMixedByDesign() throws IOException { + String cmap = cmapOf(new char[] {'\uABCD'}, false); + assertEquals("1 beginbfchar\n<0000> \nendbfchar\n", body(cmap)); + assertEquals(true, cmap.contains("<0000> ")); + } + + private String cmapOf(String[] destinations) throws IOException { + PDFToUnicodeCMap cmap = new PDFToUnicodeCMap(destinations, PDFCMap.ENC_IDENTITY_H, + new PDFCIDSystemInfo("Adobe", "Identity", 0), false, null); + CharArrayWriter writer = new CharArrayWriter(); + cmap.createCMapBuilder(writer).writeCMap(); + return writer.toString(); + } + + /** + * A glyph standing for several characters publishes them as a string, in a bfchar, + * never in a range; its single-character neighbours still range. + */ + @Test + public void testLigaturePublishesItsLetters() throws IOException { + assertEquals("1 beginbfchar\n<0002> <00660069>\nendbfchar\n" + + "1 beginbfrange\n<0000> <0001> <0066>\nendbfrange\n", + body(cmapOf(new String[] {"f", "g", "fi"}))); + } + + /** + * A surrogate pair is one selector, so the selectors after it do not drift by one as they + * did when the pair occupied two slots of a positional array; the positional constructor + * now gives the same CMap as the per-selector one. + */ + @Test + public void testSelectorsDoNotDriftAfterASurrogatePair() throws IOException { + String expected = "3 beginbfchar\n<0000> <0041>\n<0001> \n<0002> <0042>\nendbfchar\n"; + assertEquals(expected, body(cmapOf(new String[] {"A", "\uD835\uDC00", "B"}))); + assertEquals(expected, body(cmapOf(new char[] {'A', '\uD835', '\uDC00', 'B'}, false))); + } + + /** Consecutive supplementary-plane code points still pack into a range. */ + @Test + public void testSurrogatePairsStillRange() throws IOException { + assertEquals("1 beginbfrange\n<0000> <0001> \nendbfrange\n", + body(cmapOf(new String[] {"\uD835\uDC00", "\uD835\uDC01"}))); + } + + /** An empty destination is written as an empty string. */ + @Test + public void testEmptyDestination() throws IOException { + assertEquals("2 beginbfchar\n<0000> <0041>\n<0001> <>\nendbfchar\n", + body(cmapOf(new String[] {"A", ""}))); + } + + /** Single-byte code space, for the simple-font path. */ + @Test + public void testSingleByteCodeSpace() throws IOException { + assertEquals("1 beginbfrange\n<00> <03> <0041>\nendbfrange\n", + body(cmapOf(new char[] {'A', 'B', 'C', 'D'}, true))); + } +}