diff --git a/CHANGELOG.md b/CHANGELOG.md index 4be4f31d8..070225dee 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,9 @@ The release run heads these entries with the version and opens a fresh ## Unreleased +- CFF fonts bound INDEX, dictionary and glyph reads, and check offsets before + conversion. Real operands ignore the host locale. + - SFNT output uses format 12 for large character maps and U+FFFF mappings, writes Unicode font names correctly, and validates table and name sizes. Synthesized PostScript names now obey OpenType character and length limits. diff --git a/src/odr/internal/font/cff_font.cpp b/src/odr/internal/font/cff_font.cpp index a5dfbeb3d..72853b4b4 100644 --- a/src/odr/internal/font/cff_font.cpp +++ b/src/odr/internal/font/cff_font.cpp @@ -3,12 +3,15 @@ #include #include #include +#include #include #include #include #include +#include #include +#include #include #include #include @@ -81,21 +84,39 @@ namespace bs = util::byte_string; /// to an absolute offset @p p into the buffer. An out-of-range @p p yields an /// empty view, which `byte_string` reports as a (throwing) short read. [[nodiscard]] std::string_view at(const std::string_view d, - const std::uint32_t p) { + const std::size_t p) { return p <= d.size() ? d.substr(p) : std::string_view{}; } -[[nodiscard]] std::uint8_t u8(const std::string_view d, const std::uint32_t p) { +[[nodiscard]] std::uint8_t u8(const std::string_view d, const std::size_t p) { return bs::read_u8(at(d, p)); } /// Read a big-endian unsigned of @p size bytes (1..4) at @p p. [[nodiscard]] std::uint32_t read_be(const std::string_view d, - const std::uint32_t p, + const std::size_t p, const std::uint32_t size) { return bs::read_uint_be(at(d, p), size); } +std::string_view checked_slice(const std::string_view bytes, + const std::size_t offset, + const std::size_t length) { + if (offset > bytes.size() || length > bytes.size() - offset) { + throw std::runtime_error("cff: range exceeds font data"); + } + return bytes.substr(offset, length); +} + +std::uint32_t dict_offset(const double value) { + if (!std::isfinite(value) || value < 0 || + value > std::numeric_limits::max() || + std::trunc(value) != value) { + throw std::runtime_error("cff: invalid DICT offset or length"); + } + return static_cast(value); +} + /// A parsed DICT: operator -> operands. Reals are decoded to double; integers /// stay exact within double range (CFF integers fit). using Dict = std::map>; @@ -106,14 +127,12 @@ using Dict = std::map>; return static_cast(std::clamp(value, -32768.0, 32767.0)); } -/// Parse a CFF DICT occupying the byte range [begin, end) of @p d. -[[nodiscard]] Dict parse_dict(const std::string_view d, - const std::uint32_t begin, - const std::uint32_t end) { +/// Parse one bounded CFF DICT (Adobe TN #5176 section 4). +[[nodiscard]] Dict parse_dict(const std::string_view d) { Dict dict; std::vector operands; - std::uint32_t p = begin; - while (p < end) { + std::size_t p = 0; + while (p < d.size()) { const std::uint8_t b0 = u8(d, p); if (b0 <= 21) { // operator @@ -136,7 +155,7 @@ using Dict = std::map>; ++p; std::string number; bool done = false; - while (!done && p < end) { + while (!done && p < d.size()) { const std::uint8_t byte = u8(d, p++); for (const std::uint8_t nibble : {static_cast(byte >> 4), @@ -157,25 +176,26 @@ using Dict = std::map>; } } } - try { - operands.push_back(number.empty() ? 0.0 : std::stod(number)); - } catch (const std::exception &) { - operands.push_back(0.0); - } + // An unreadable real reads as zero, so it does not refuse the font. + const std::optional value = util::number::parse(number); + operands.push_back(value && std::isfinite(*value) ? *value : 0.0); } else if (b0 >= 32 && b0 <= 246) { - operands.push_back(static_cast(b0) - 139); + operands.push_back(static_cast(b0) - 139); ++p; } else if (b0 >= 247 && b0 <= 250) { - operands.push_back((static_cast(b0) - 247) * 256 + u8(d, p + 1) + - 108); + operands.push_back((static_cast(b0) - 247) * 256 + + u8(d, p + 1) + 108); p += 2; } else if (b0 >= 251 && b0 <= 254) { - operands.push_back(-(static_cast(b0) - 251) * 256 - u8(d, p + 1) - - 108); + operands.push_back(-(static_cast(b0) - 251) * 256 - + u8(d, p + 1) - 108); p += 2; } else { throw std::runtime_error("cff: invalid DICT byte"); } + if (operands.size() > 48) { + throw std::runtime_error("cff: too many DICT operands"); + } } return dict; } @@ -228,40 +248,38 @@ CffFont::CffFont(std::string data) : m_data{std::move(data)} { parse(); } std::vector CffFont::read_index(const std::uint32_t offset, std::uint32_t &end) const { - const std::string_view d{m_data}; - const std::uint16_t count = static_cast(read_be(d, offset, 2)); + const std::string_view d = at(m_data, offset); + const std::uint16_t count = bs::read_u16_be(d); if (count == 0) { end = offset + 2; return {}; } - const std::uint8_t off_size = u8(d, offset + 2); + const std::uint8_t off_size = u8(d, 2); if (off_size < 1 || off_size > 4) { throw std::runtime_error("cff: bad INDEX offSize"); } - const std::uint32_t offset_array = offset + 3; - const std::uint32_t data_base = offset_array + (count + 1) * off_size - 1; - + const std::size_t data_base = 3 + (std::size_t{count} + 1) * off_size; + (void)checked_slice(d, 0, data_base); + std::uint32_t prev = read_be(d, 3, off_size); std::vector members; members.reserve(count); - std::uint32_t prev = read_be(d, offset_array, off_size); - for (std::uint16_t i = 1; i <= count; ++i) { - const std::uint32_t next = - read_be(d, offset_array + i * off_size, off_size); - // The offset array is non-decreasing (Adobe TN #5176 ยง5); otherwise the - // member length would wrap. - if (next < prev) { - throw std::runtime_error("cff: non-monotonic INDEX offsets"); + for (std::size_t i = 1; i <= count; ++i) { + const std::uint32_t next = read_be(d, 3 + i * off_size, off_size); + if (next < prev || next - 1 > d.size() - data_base) { + throw std::runtime_error("cff: invalid INDEX member range"); } - members.push_back({data_base + prev, next - prev}); + members.push_back( + {static_cast(offset + data_base + prev - 1), + next - prev}); prev = next; } - end = data_base + prev; + end = static_cast(offset + data_base + prev - 1); return members; } void CffFont::parse() { const std::string_view d{m_data}; - if (!is_cff(d)) { + if (!is_cff(d) || !std::in_range(d.size())) { throw std::runtime_error("cff: not a CFF font"); } const std::uint8_t header_size = u8(d, 2); @@ -295,7 +313,7 @@ void CffFont::parse() { void CffFont::parse_top_dict(const Range top_dict) { const std::string_view d{m_data}; const Dict dict = - parse_dict(d, top_dict.offset, top_dict.offset + top_dict.length); + parse_dict(checked_slice(d, top_dict.offset, top_dict.length)); m_cid_keyed = dict.find(op_ros) != dict.end(); @@ -303,7 +321,7 @@ void CffFont::parse_top_dict(const Range top_dict) { it != dict.end() && it->second.size() == 6 && it->second[0] != 0.0) { // unitsPerEm = 1 / FontMatrix[0] (design units per em). const double upm = 1.0 / it->second[0]; - if (upm > 0 && upm < 65536) { + if (upm >= 0.5 && upm < 65535.5) { m_units_per_em = static_cast(std::lround(upm)); } } @@ -316,23 +334,22 @@ void CffFont::parse_top_dict(const Range top_dict) { if (const auto it = dict.find(op_char_strings); it != dict.end()) { std::uint32_t end = 0; - m_charstrings = - read_index(static_cast(it->second.at(0)), end); + m_charstrings = read_index(dict_offset(it->second.at(0)), end); } if (const auto it = dict.find(op_private); it != dict.end() && it->second.size() == 2) { - const auto size = static_cast(it->second[0]); - const auto offset = static_cast(it->second[1]); + const auto size = dict_offset(it->second[0]); + const auto offset = dict_offset(it->second[1]); m_widths = parse_private_dict({offset, size}); } // A CID-keyed font keeps its Private DICTs per FD, not in the Top DICT. if (const auto it = dict.find(op_fd_array); it != dict.end()) { - parse_fd_array(static_cast(it->second.at(0))); + parse_fd_array(dict_offset(it->second.at(0))); } if (const auto it = dict.find(op_fd_select); it != dict.end()) { - parse_fd_select(static_cast(it->second.at(0))); + parse_fd_select(dict_offset(it->second.at(0))); } // charset: an offset past the predefined ids (0/1/2) is a custom charset; @@ -341,7 +358,7 @@ void CffFont::parse_top_dict(const Range top_dict) { // the predefined name charsets only apply to name-keyed fonts. std::uint32_t charset = predefined_charset_iso_adobe; if (const auto it = dict.find(op_charset); it != dict.end()) { - charset = static_cast(it->second.at(0)); + charset = dict_offset(it->second.at(0)); } if (charset > predefined_charset_expert_subset) { parse_charset(charset); @@ -375,8 +392,8 @@ CffFont::Widths CffFont::parse_private_dict(const Range private_dict) const { return widths; } const std::string_view d{m_data}; - const Dict dict = parse_dict(d, private_dict.offset, - private_dict.offset + private_dict.length); + const Dict dict = + parse_dict(checked_slice(d, private_dict.offset, private_dict.length)); if (const auto it = dict.find(op_default_width_x); it != dict.end()) { widths.default_width = it->second.at(0); } @@ -391,26 +408,26 @@ void CffFont::parse_fd_array(const std::uint32_t offset) { std::uint32_t end = 0; for (const Range font_dict : read_index(offset, end)) { const Dict dict = - parse_dict(d, font_dict.offset, font_dict.offset + font_dict.length); + parse_dict(checked_slice(d, font_dict.offset, font_dict.length)); Widths widths; if (const auto it = dict.find(op_private); it != dict.end() && it->second.size() == 2) { - widths = parse_private_dict({static_cast(it->second[1]), - static_cast(it->second[0])}); + widths = parse_private_dict( + {dict_offset(it->second[1]), dict_offset(it->second[0])}); } m_fd_widths.push_back(widths); } } void CffFont::parse_fd_select(const std::uint32_t offset) { - const std::string_view d{m_data}; + const std::string_view d = at(m_data, offset); const std::uint16_t glyphs = glyph_count(); - const std::uint8_t format = u8(d, offset); + const std::uint8_t format = u8(d, 0); if (format == 0) { m_fd_select.reserve(glyphs); for (std::uint16_t gid = 0; gid < glyphs; ++gid) { - m_fd_select.push_back(u8(d, offset + 1 + gid)); + m_fd_select.push_back(u8(d, 1 + gid)); } return; } @@ -419,13 +436,14 @@ void CffFont::parse_fd_select(const std::uint32_t offset) { } // format 3: ranges of [first, next first) sharing one FD, then a sentinel - const auto ranges = static_cast(read_be(d, offset + 1, 2)); + const auto ranges = static_cast(read_be(d, 1, 2)); m_fd_select.assign(glyphs, 0); for (std::uint16_t i = 0; i < ranges; ++i) { - const std::uint32_t entry = offset + 3 + 3 * i; + const std::size_t entry = 3 + 3 * std::size_t{i}; const auto first = static_cast(read_be(d, entry, 2)); const std::uint8_t fd = u8(d, entry + 2); const auto next = static_cast(read_be(d, entry + 3, 2)); + // An FD index out of range falls back in `widths_for_glyph`. for (std::uint32_t gid = first; gid < next && gid < glyphs; ++gid) { m_fd_select[gid] = fd; } @@ -444,12 +462,12 @@ CffFont::widths_for_glyph(const std::uint16_t glyph) const { } void CffFont::parse_charset(const std::uint32_t offset) { - const std::string_view d{m_data}; + const std::string_view d = at(m_data, offset); const std::uint16_t glyphs = glyph_count(); m_charset.assign(glyphs, 0); // glyph 0 (.notdef) -> SID/CID 0 - const std::uint8_t format = u8(d, offset); - std::uint32_t p = offset + 1; + const std::uint8_t format = u8(d, 0); + std::size_t p = 1; std::uint16_t gid = 1; // glyph 0 is implicit if (format == 0) { for (; gid < glyphs; ++gid) { @@ -463,7 +481,9 @@ void CffFont::parse_charset(const std::uint32_t offset) { p += 2; const std::uint32_t n_left = read_be(d, p, n_left_size); p += n_left_size; - for (std::uint32_t i = 0; i <= n_left && gid < glyphs; ++i, ++gid) { + // A last range past the glyph count is clipped. + for (std::uint32_t i = 0; + i <= n_left && gid < glyphs && first + i <= 0xFFFF; ++i, ++gid) { m_charset[gid] = static_cast(first + i); } } @@ -513,17 +533,16 @@ CffFont::charstring_width(const std::uint16_t glyph) const { if (glyph >= m_charstrings.size()) { return std::nullopt; } - const std::string_view d{m_data}; const Range cs = m_charstrings[glyph]; - const std::uint32_t end = cs.offset + cs.length; - std::uint32_t p = cs.offset; + const std::string_view d = checked_slice(m_data, cs.offset, cs.length); + std::size_t p = 0; // Walk operands until the first width-bearing operator; the Type2 width, when // present, is the first operand and makes the operand count exceed the // operator's nominal arity (ISO/Adobe Type2 charstring spec, "width"). - std::int32_t count = 0; + std::size_t count = 0; double first = 0.0; - while (p < end) { + while (p < d.size()) { const std::uint8_t b0 = u8(d, p); if (b0 >= marker_int_min || b0 == marker_shortint) { // operand @@ -532,13 +551,15 @@ CffFont::charstring_width(const std::uint16_t glyph) const { value = static_cast(read_be(d, p + 1, 2)); p += 3; } else if (b0 <= 246) { - value = static_cast(b0) - 139; + value = static_cast(b0) - 139; p += 1; } else if (b0 <= 250) { - value = (static_cast(b0) - 247) * 256 + u8(d, p + 1) + 108; + value = + (static_cast(b0) - 247) * 256 + u8(d, p + 1) + 108; p += 2; } else if (b0 <= 254) { - value = -(static_cast(b0) - 251) * 256 - u8(d, p + 1) - 108; + value = + -(static_cast(b0) - 251) * 256 - u8(d, p + 1) - 108; p += 2; } else { // marker_fixed: 16.16 fixed value = static_cast(read_be(d, p + 1, 4)) / 65536.0; diff --git a/test/src/internal/font/cff_font.cpp b/test/src/internal/font/cff_font.cpp index bf1650b50..8bccc5ce8 100644 --- a/test/src/internal/font/cff_font.cpp +++ b/test/src/internal/font/cff_font.cpp @@ -9,6 +9,8 @@ #include +#include +#include #include #include #include @@ -318,6 +320,12 @@ std::string build_cid_keyed_cff() { return out; } +std::string cff_with_top_dict(const std::string &dict, + const std::vector &strings = {}) { + return std::string("\x01\0\x04\x01", 4) + build_index({"TestFont"}) + + build_index({dict}) + build_index(strings) + build_index({}); +} + } // namespace TEST(CffFontTest, ParsesFactsFromMinimalFont) { @@ -516,3 +524,99 @@ TEST(CffFontTest, WrapDropsExtraEntriesPastGlyphCount) { EXPECT_EQ(wrapped.glyph_for_code_point('B'), 0); // dropped: not mapped EXPECT_EQ(wrapped.glyph_for_code_point(pua_code_point(1)), 1); } + +TEST(CffFontTest, IndexCountAndRangesAreBounded) { + EXPECT_EQ( + CffFont(cff_with_top_dict({}, std::vector(65535))).name(), + "TestFont"); + const std::string valid = cff_with_top_dict({}); + for (const std::uint8_t offset : {0, 255}) { + std::string bytes = valid; + bytes[8] = static_cast(offset); + EXPECT_THROW(CffFont{bytes}, std::runtime_error); + } + std::string past_the_data = valid; + past_the_data[7] = static_cast(255); + EXPECT_THROW(CffFont{past_the_data}, std::runtime_error); + const std::string oversized( + "\x01\0\x04\x04\0\x01\x04\0\0\0\x01\xff\xff\xff\xff", 15); + EXPECT_THROW(CffFont{oversized}, std::runtime_error); +} + +TEST(CffFontTest, DictOperandsStayWithinTheirDeclaredRange) { + const std::array invalid{ + std::string("\x1d", 1), std::string("\x0c", 1), + std::string("\x1e\x1a\x5f\x11", 4), // fractional CharStrings offset + std::string("\x8a\x11", 2), // negative CharStrings offset + std::string(49, static_cast(139)) + static_cast(5)}; + for (const std::string &dict : invalid) { + EXPECT_THROW(CffFont(cff_with_top_dict(dict)), std::runtime_error); + } + + const std::vector glyphs{{".notdef", "\x1c"}, {"A", "\x0e"}}; + const CffFont font( + odr::internal::font::cff::build_cff("Short", glyphs, 0, 0, {})); + EXPECT_THROW((void)font.advance_width(0), std::runtime_error); +} + +TEST(CffFontTest, RealOperandsIgnoreNumericLocale) { + struct LocaleGuard final { + std::string previous{std::setlocale(LC_NUMERIC, nullptr)}; + ~LocaleGuard() { std::setlocale(LC_NUMERIC, previous.c_str()); } + } guard; + if (std::setlocale(LC_NUMERIC, "de_DE.UTF-8") == nullptr && + std::setlocale(LC_NUMERIC, "de_DE.utf8") == nullptr) { + GTEST_SKIP() << "German locale unavailable"; + } + const std::string scale("\x1e\x0a\x00\x05\xff", 5); // 0.0005 + const std::string zero(1, static_cast(139)); + const CffFont font(cff_with_top_dict(scale + zero + zero + scale + zero + + zero + std::string("\x0c\x07", 2))); + EXPECT_EQ(font.units_per_em(), 2000); +} + +TEST(CffFontTest, QuirkyDictOperandsDoNotRefuseTheFont) { + const std::string zero(1, static_cast(139)); + const auto units_per_em = [&](const std::string &scale) { + return CffFont(cff_with_top_dict(scale + zero + zero + scale + zero + zero + + std::string("\x0c\x07", 2))) + .units_per_em(); + }; + // A reserved nibble is skipped: 0.0005 gives 2000 units per em. + EXPECT_EQ(units_per_em(std::string("\x1e\x0a\x00\x0d\x5f", 5)), 2000); + // An overflowing real reads as zero, so the default stays. + EXPECT_EQ(units_per_em(std::string("\x1e\x1b\x99\x9f", 4)), 1000); + EXPECT_EQ(CffFont(cff_with_top_dict(zero)).name(), "TestFont"); +} + +TEST(CffFontTest, FontDictionarySelectionToleratesQuirkyRanges) { + const auto font_bytes = [](const std::string &selection) { + const std::string glyphs = build_index({"\x0e", "\x0e", "\x0e"}); + const std::string dictionaries = build_index({""}); + const auto dict = [&](const std::int32_t base) { + std::string out; + dict_int(out, base); + out += static_cast(17); + dict_int(out, base + static_cast(glyphs.size())); + out += std::string("\x0c\x24", 2); + dict_int(out, base + static_cast(glyphs.size() + + dictionaries.size())); + out += std::string("\x0c\x25", 2); + return out; + }; + const auto base = + static_cast(cff_with_top_dict(dict(0)).size()); + return cff_with_top_dict(dict(base)) + glyphs + dictionaries + selection; + }; + const std::string ranges("\x03\0\x01\0\0\0\0\x03", 8); + EXPECT_EQ(CffFont(font_bytes(ranges)).glyph_count(), 3); + EXPECT_EQ(CffFont(font_bytes(std::string(4, '\0'))).advance_width(2), 0); + // A range past the glyphs, a missing sentinel or an unknown FD is tolerated. + for (const std::size_t index : {2, 4, 5, 7}) { + std::string quirky = ranges; + quirky[index] = index == 2 ? 0 : 1; + EXPECT_EQ(CffFont(font_bytes(quirky)).advance_width(2), 0); + } + EXPECT_EQ(CffFont(font_bytes(std::string("\0\0\x01\0", 4))).advance_width(1), + 0); +}