ImageSharp/SixLabors.Fonts/Tables/Cff/CffParserBase.cs
2026-08-03 22:31:27 +02:00

508 lines
19 KiB
C#
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System;
using System.Collections.Generic;
using System.Diagnostics.CodeAnalysis;
using System.Globalization;
using System.Text;
namespace SixLabors.Fonts.Tables.Cff {
/// <summary>
/// Base class for CFF1 and CFF2 parsers providing shared DICT parsing,
/// INDEX reading, FDSelect, and subroutine loading functionality.
/// </summary>
internal abstract class CffParserBase
{
private readonly StringBuilder pooledStringBuilder = new();
/// <summary>
/// Reads the FDSelect structure that maps glyph indices to Font DICT indices.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="offset">The absolute offset of the CFF table.</param>
/// <param name="cidFontInfo">The CIDFont information to populate with FDSelect data.</param>
protected static void ReadFdSelect(BigEndianBinaryReader reader, long offset, CidFontInfo cidFontInfo)
{
if (cidFontInfo.FDSelect is 0)
{
return;
}
reader.BaseStream.Position = offset + cidFontInfo.FDSelect;
switch (reader.ReadByte())
{
case 0:
{
cidFontInfo.FdSelectFormat = 0;
for (int i = 0; i < cidFontInfo.CIDFountCount; i++)
{
cidFontInfo.FdSelectMap[i] = reader.ReadByte();
}
break;
}
case 3:
{
cidFontInfo.FdSelectFormat = 3;
ushort nRanges = reader.ReadUInt16();
FDRange[] ranges = new FDRange[nRanges + 1];
cidFontInfo.FdSelectFormat = 3;
cidFontInfo.FdRanges = ranges;
for (int i = 0; i < nRanges; ++i)
{
ranges[i] = new FDRange(reader.ReadUInt16(), reader.ReadByte());
}
ranges[nRanges] = new FDRange(reader.ReadUInt16(), 0); // sentinel
break;
}
case 4:
{
cidFontInfo.FdSelectFormat = 4;
uint nRanges = reader.ReadUInt32();
FDRange[] ranges = new FDRange[nRanges + 1];
cidFontInfo.FdSelectFormat = 3;
cidFontInfo.FdRanges = ranges;
for (int i = 0; i < nRanges; ++i)
{
ranges[i] = new FDRange(reader.ReadUInt32(), reader.ReadUInt16());
}
ranges[nRanges] = new FDRange(reader.ReadUInt32(), 0); // sentinel
break;
}
default:
throw new NotSupportedException("Only FD Select format 0, 3 and 4 are supported");
}
}
/// <summary>
/// Reads the Font DICT Array (FDArray), which contains per-font-dictionary entries for CIDFonts.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="offset">The absolute offset of the CFF table.</param>
/// <param name="fdArrayOffset">The offset to the FDArray INDEX relative to the CFF table.</param>
/// <param name="cff2">Whether to use CFF2 INDEX format (32-bit count).</param>
/// <returns>An array of <see cref="FontDict"/> entries.</returns>
protected FontDict[] ReadFdArray(BigEndianBinaryReader reader, long offset, long fdArrayOffset, bool cff2 = false)
{
if (fdArrayOffset is 0)
{
return [];
}
reader.BaseStream.Position = offset + fdArrayOffset;
if (!TryReadIndexDataOffsets(reader, out CffIndexOffset[]? offsets, cff2))
{
return [];
}
FontDict[] fontDicts = new FontDict[offsets.Length];
for (int i = 0; i < fontDicts.Length; ++i)
{
// Read DICT data.
List<CffDataDicEntry> dic = this.ReadDictData(reader, offsets[i].Length);
// translate
int fontDictsOffset = 0;
int size = 0;
int name = 0;
foreach (CffDataDicEntry entry in dic)
{
switch (entry.Operator.Name)
{
default:
throw new NotSupportedException();
case "FontName":
name = (int)entry.Operands[0].RealNumValue;
break;
case "Private": // private dic
size = (int)entry.Operands[0].RealNumValue;
fontDictsOffset = (int)entry.Operands[1].RealNumValue;
break;
}
}
fontDicts[i] = new FontDict(name, size, fontDictsOffset);
}
foreach (FontDict fdict in fontDicts)
{
reader.BaseStream.Position = offset + fdict.PrivateDicOffset;
List<CffDataDicEntry> dicData = this.ReadDictData(reader, fdict.PrivateDicSize);
if (dicData.Count > 0)
{
// Interpret the values of private dict.
foreach (CffDataDicEntry dicEntry in dicData)
{
switch (dicEntry.Operator.Name)
{
case "Subrs":
int localSubrsOffset = (int)dicEntry.Operands[0].RealNumValue;
reader.BaseStream.Position = offset + fdict.PrivateDicOffset + localSubrsOffset;
fdict.LocalSubr = ReadSubrBuffer(reader, cff2);
break;
case "vsindex":
fdict.VsIndex = (int)dicEntry.Operands[0].RealNumValue;
break;
case "defaultWidthX":
case "nominalWidthX":
break;
}
}
}
}
return fontDicts;
}
/// <summary>
/// Reads a single DICT entry consisting of operands followed by an operator.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <returns>The parsed <see cref="CffDataDicEntry"/>.</returns>
protected CffDataDicEntry ReadEntry(BigEndianBinaryReader reader)
{
List<CffOperand> operands = new();
//-----------------------------
// An operator is preceded by the operand(s) that
// specify its value.
//--------------------------------
//-----------------------------
// Operators and operands may be distinguished by inspection of
// their first byte:
// 021 specify operators and
// 28, 29, 30, and 32254 specify operands(numbers).
// Byte values 2227, 31, and 255 are reserved.
// An operator may be preceded by up to a maximum of 48 operands
CFFOperator? @operator;
while (true)
{
byte b0 = reader.ReadUInt8();
if (b0 is >= 0 and <= 24)
{
// operators
@operator = ReadOperator(reader, b0);
break; // **break after found operator
}
else if (b0 is 28 or 29)
{
int num = ReadIntegerNumber(reader, b0);
operands.Add(new CffOperand(num, OperandKind.IntNumber));
}
else if (b0 == 30)
{
double num = this.ReadRealNumber(reader);
operands.Add(new CffOperand(num, OperandKind.RealNumber));
}
else if (b0 is >= 32 and <= 254)
{
int num = ReadIntegerNumber(reader, b0);
operands.Add(new CffOperand(num, OperandKind.IntNumber));
}
else
{
throw new NotSupportedException("invalid DICT data b0 byte: " + b0);
}
}
// I'm fairly confident that the operator can never be null.
return new CffDataDicEntry(@operator!, operands.ToArray());
}
/// <summary>
/// Attempts to read the offset array from a CFF INDEX structure.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="value">When this method returns, contains the parsed index offsets, or <see langword="null"/> if the INDEX is empty.</param>
/// <param name="cff2">Whether to use CFF2 INDEX format (32-bit count).</param>
/// <returns><see langword="true"/> if the INDEX contained at least one element; otherwise, <see langword="false"/>.</returns>
protected static bool TryReadIndexDataOffsets(BigEndianBinaryReader reader, [NotNullWhen(true)] out CffIndexOffset[]? value, bool cff2 = false)
{
// INDEX Data
// An INDEX is an array of variable-sized objects.It comprises a
// header, an offset array, and object data.
// The offset array specifies offsets within the object data.
// An object is retrieved by
// indexing the offset array and fetching the object at the
// specified offset.
// The objects length can be determined by subtracting its offset
// from the next offset in the offset array.
// An additional offset is added at the end of the offset array so the
// length of the last object may be determined.
// The INDEX format is shown in Table 7
// Table 7 INDEX Format
// Type Name Description
// Card16 count Number of objects stored in INDEX
// OffSize offSize Offset array element size
// Offset offset[count + 1] Offset array(from byte preceding object data)
// Card8 data[<varies>] Object data
// Offsets in the offset array are relative to the byte that precedes
// the object data. Therefore the first element of the offset array
// is always 1. (This ensures that every object has a corresponding
// offset which is always nonzero and permits the efficient
// implementation of dynamic object loading.)
// An empty INDEX is represented by a count field with a 0 value
// and no additional fields.Thus, the total size of an empty INDEX
// is 2 bytes.
// Note 2
// An INDEX may be skipped by jumping to the offset specified by the last
// element of the offset array
// CFF2 uses a 32-bit count; CFF1 uses 16-bit.
uint count = cff2 ? reader.ReadUInt32() : reader.ReadUInt16();
if (count == 0)
{
value = null;
return false;
}
int offSize = reader.ReadByte();
int[] offsets = new int[count + 1];
CffIndexOffset[] indexElems = new CffIndexOffset[count];
for (int i = 0; i <= count; ++i)
{
offsets[i] = reader.ReadOffset(offSize);
}
for (int i = 0; i < count; ++i)
{
indexElems[i] = new CffIndexOffset(offsets[i], offsets[i + 1] - offsets[i]);
}
value = indexElems;
return true;
}
/// <summary>
/// Reads a subroutine INDEX and returns the raw byte buffers for each subroutine.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="cff2">Whether to use CFF2 INDEX format (32-bit count).</param>
/// <returns>An array of byte arrays, each containing a subroutine charstring.</returns>
protected static byte[][] ReadSubrBuffer(BigEndianBinaryReader reader, bool cff2 = false)
{
if (!TryReadIndexDataOffsets(reader, out CffIndexOffset[]? offsets, cff2))
{
return [];
}
byte[][] rawBufferList = new byte[offsets.Length][];
for (int i = 0; i < rawBufferList.Length; ++i)
{
CffIndexOffset offset = offsets[i];
rawBufferList[i] = reader.ReadBytes(offset.Length);
}
return rawBufferList;
}
/// <summary>
/// Reads DICT data of the specified length, parsing all operator-operand entries.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="length">The length in bytes of the DICT data to read.</param>
/// <returns>A list of parsed <see cref="CffDataDicEntry"/> entries.</returns>
protected List<CffDataDicEntry> ReadDictData(BigEndianBinaryReader reader, int length)
{
// 4. DICT Data
// Font dictionary data comprising key-value pairs is represented
// in a compact tokenized format that is similar to that used to
// represent Type 1 charstrings.
// Dictionary keys are encoded as 1- or 2-byte operators and dictionary values are encoded as
// variable-size numeric operands that represent either integer or
// real values.
//-----------------------------
// A DICT is simply a sequence of
// operand(s)/operator bytes concatenated together.
int maxIndex = (int)(reader.BaseStream.Position + length);
List<CffDataDicEntry> dicData = new();
while (reader.BaseStream.Position < maxIndex)
{
CffDataDicEntry dicEntry = this.ReadEntry(reader);
dicData.Add(dicEntry);
}
return dicData;
}
/// <summary>
/// Reads a DICT operator (one or two bytes) from the reader.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="b0">The first byte of the operator.</param>
/// <returns>The resolved <see cref="CFFOperator"/>.</returns>
private static CFFOperator ReadOperator(BigEndianBinaryReader reader, byte b0)
{
// Read operator key.
byte b1 = 0;
if (b0 == 12)
{
// 2 bytes
b1 = reader.ReadUInt8();
}
// Get registered operator by its key.
return CFFOperator.GetOperatorByKey(b0, b1);
}
/// <summary>
/// Reads a real number operand encoded as a nibble-based BCD sequence.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <returns>The decoded real number value.</returns>
private double ReadRealNumber(BigEndianBinaryReader reader)
{
// from https://typekit.files.wordpress.com/2013/05/5176.cff.pdf
// A real number operand is provided in addition to integer
// operands.This operand begins with a byte value of 30 followed
// by a variable-length sequence of bytes.Each byte is composed
// of two 4 - bit nibbles asdefined in Table 5.
// The first nibble of a
// pair is stored in the most significant 4 bits of a byte and the
// second nibble of a pair is stored in the least significant 4 bits of a byte
StringBuilder sb = this.pooledStringBuilder;
sb.Clear(); // reset
bool done = false;
bool exponentMissing = false;
while (!done)
{
int b = reader.ReadByte();
int nb_0 = (b >> 4) & 0xf;
int nb_1 = b & 0xf;
for (int i = 0; !done && i < 2; ++i)
{
int nibble = (i == 0) ? nb_0 : nb_1;
switch (nibble)
{
case 0x0:
case 0x1:
case 0x2:
case 0x3:
case 0x4:
case 0x5:
case 0x6:
case 0x7:
case 0x8:
case 0x9:
sb.Append(nibble);
exponentMissing = false;
break;
case 0xa:
sb.Append('.');
break;
case 0xb:
sb.Append('E');
exponentMissing = true;
break;
case 0xc:
sb.Append("E-");
exponentMissing = true;
break;
case 0xd:
break;
case 0xe:
sb.Append('-');
break;
case 0xf:
done = true;
break;
default:
throw new FontException("Unable to read real number.");
}
}
}
if (exponentMissing)
{
// the exponent is missing, just append "0" to avoid an exception
// not sure if 0 is the correct value, but it seems to fit
// see PDFBOX-1522
sb.Append('0');
}
if (sb.Length == 0)
{
return 0d;
}
if (!double.TryParse(
sb.ToString(),
NumberStyles.Number | NumberStyles.AllowExponent,
CultureInfo.InvariantCulture,
out double value))
{
throw new NotSupportedException();
}
return value;
}
/// <summary>
/// Reads an integer number operand from the DICT data based on the initial byte.
/// </summary>
/// <param name="reader">The binary reader.</param>
/// <param name="b0">The initial byte that determines the encoding format.</param>
/// <returns>The decoded integer value.</returns>
private static int ReadIntegerNumber(BigEndianBinaryReader reader, byte b0)
{
if (b0 == 28)
{
return reader.ReadInt16();
}
if (b0 == 29)
{
return reader.ReadInt32();
}
if (b0 is >= 32 and <= 246)
{
return b0 - 139;
}
if (b0 is >= 247 and <= 250)
{
int b1 = reader.ReadByte();
return ((b0 - 247) * 256) + b1 + 108;
}
if (b0 is >= 251 and <= 254)
{
int b1 = reader.ReadByte();
return (-(b0 - 251) * 256) - b1 - 108;
}
throw new InvalidFontFileException("Invalid DICT data b0 byte: " + b0);
}
}
}