Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,9 @@ The release run heads these entries with the version and opens a fresh

## Unreleased

- Legacy Word bounds font and piece tables by their declared lengths, checks
piece coverage and offsets, and reads PLC entries without unaligned access.

- Legacy Word validates FIB signatures and extension lengths. It stores the
shared header fields by value, removing unsafe deletion through base pointers.

Expand Down
15 changes: 9 additions & 6 deletions src/odr/internal/oldms/text/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -43,17 +43,20 @@ a future base `nFib` may still use the common prefix. No versioned object is
allocated or deleted through a non-virtual base.

**Piece table.** The `PlcPcd` is `n+1` ascending CP boundaries followed by `n`
`Pcd`s. `PlcPcdMap` is a zero-copy view over the raw bytes. Each `Pcd`'s
`FcCompressed`: `fCompressed == 0` is UTF-16 at `fc`, `== 1` is one byte per
CP at `fc/2`, with `0x82` to `0x9F` remapped by `uncompress_char` (搂2.9.73)
and every other byte `b` mapped to `U+00b`.
`Pcd`s. `PlcPcdMap` views the raw bytes and copies each checked entry to avoid
unaligned access. The Clx and font table are bounded by their declared lengths.
Each `Pcd`'s `FcCompressed`: `fCompressed == 0` is UTF-16 at `fc`, `== 1` is
one byte per CP at `fc/2`, with `0x82` to `0x9F` remapped by `uncompress_char`
(搂2.9.73) and every other byte `b` mapped to `U+00b`.

**Fail early.** Throw on: `nFib` below `nFib97` or an unknown `nFibNew`; a
`ccpText` with the sign bit set (搂2.5.5); a `csw` or `cslw` count too small
for an array the module reads; an unexpected Clx lead byte (not `0x01` or
`0x02`); non-ascending CP boundaries; a bad compressed byte; early EOF. Pass
`0x02`); non-contiguous CP boundaries; piece offsets beyond the 32-bit range;
missing body coverage; an FFN shorter than its fixed part; early EOF. Pass
through what is not modelled: text after the main body, `Prc` formatting
runs, and every control or field character `TextCleaner` drops.
runs, every control or field character `TextCleaner` drops, and a font name
with an odd byte, extra bytes or no NUL.

**Direct character formatting, resolved to styled spans.** Each span and each
paragraph stores a style index in the element registry, resolved through the
Expand Down
18 changes: 11 additions & 7 deletions src/odr/internal/oldms/text/doc_helper.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
#include <odr/internal/oldms/text/doc_io.hpp>
#include <odr/internal/oldms/text/doc_style.hpp>
#include <odr/internal/util/byte_stream_util.hpp>
#include <odr/internal/util/stream_util.hpp>

#include <array>
#include <cstring>
Expand All @@ -12,6 +11,7 @@
#include <string>
#include <string_view>
#include <unordered_map>
#include <utility>

namespace odr::internal::oldms::text {

Expand Down Expand Up @@ -62,11 +62,11 @@ text::CharacterIndex text::read_character_index(std::istream &in) {
throw std::runtime_error("Unexpected input: " + std::to_string(c));
}
const std::uint32_t lcb = util::byte_stream::read<std::uint32_t>(in);
std::string plcPcd = util::stream::read(in, lcb);
std::string plcPcd = util::byte_stream::read_u8s(in, lcb);
const PlcPcdMap plc_pcd_map(plcPcd.data(), plcPcd.size());

const std::uint32_t count = plc_pcd_map.n();
for (std::uint32_t i = 0; i < count; ++i) {
const std::size_t count = plc_pcd_map.n();
for (std::size_t i = 0; i < count; ++i) {
// aCp is strictly ascending ([MS-DOC] 2.8.35); otherwise the piece
// length below would wrap around.
if (plc_pcd_map.aCP(i + 1) <= plc_pcd_map.aCP(i)) {
Expand Down Expand Up @@ -97,7 +97,8 @@ text::read_character_runs(std::istream &document_stream,
const TextStyle default_style = styles.at(0);

table_stream.seekg(plcf_bte_chpx.fc);
std::string plc_bytes = util::stream::read(table_stream, plcf_bte_chpx.lcb);
std::string plc_bytes =
util::byte_stream::read_u8s(table_stream, plcf_bte_chpx.lcb);
const PlcBteChpxMap plc(plc_bytes.data(), plc_bytes.size());

// One resolved style per distinct Chpx byte sequence.
Expand All @@ -108,6 +109,9 @@ text::read_character_runs(std::istream &document_stream,
}
const auto [it, inserted] = style_cache.try_emplace(std::string(grpprl));
if (inserted) {
if (!std::in_range<std::uint32_t>(styles.size())) {
throw std::length_error("doc: too many character styles");
}
styles.push_back(
apply_character_sprms(default_style, grpprl, font_names));
it->second = static_cast<std::uint32_t>(styles.size() - 1);
Expand All @@ -117,8 +121,8 @@ text::read_character_runs(std::istream &document_stream,

// A ChpxFkp page ([MS-DOC] 2.9.33) is 512 bytes at pn * 512; rgb[j] * 2
// locates the run's Chpx within it, 0 meaning default properties.
const std::uint32_t page_count = plc.n();
for (std::uint32_t i = 0; i < page_count; ++i) {
const std::size_t page_count = plc.n();
for (std::size_t i = 0; i < page_count; ++i) {
std::array<char, 512> page;
document_stream.seekg(static_cast<std::streamoff>(plc.aData(i).pn) * 512);
document_stream.read(page.data(), page.size());
Expand Down
18 changes: 13 additions & 5 deletions src/odr/internal/oldms/text/doc_helper.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
#include <algorithm>
#include <cstdint>
#include <iosfwd>
#include <limits>
#include <span>
#include <stdexcept>
#include <string>
Expand All @@ -28,11 +29,18 @@ class CharacterIndex {

void append(const std::size_t start_cp, const std::size_t length_cp,
const std::size_t data_offset, const bool is_compressed) {
const std::size_t end_cp = start_cp + length_cp;
if (end_cp < last_cp()) {
if (start_cp != last_cp() || length_cp == 0 ||
length_cp > std::numeric_limits<std::size_t>::max() - start_cp) {
throw std::runtime_error(
"doc: character pieces must be contiguous and nonempty");
}
constexpr auto max_offset = std::numeric_limits<std::uint32_t>::max();
if (data_offset > max_offset ||
length_cp > (max_offset - data_offset) / (is_compressed ? 1 : 2)) {
throw std::runtime_error(
"append must be used in order of increasing start_cp");
"doc: character piece exceeds stream offset range");
}
const std::size_t end_cp = start_cp + length_cp;
m_entries.emplace_back(end_cp, data_offset, is_compressed);
}

Expand Down Expand Up @@ -74,7 +82,7 @@ class CharacterIndex {
}

bool operator==(const Iterator &other) const {
return m_index == other.m_index;
return m_parent == other.m_parent && m_index == other.m_index;
}

private:
Expand All @@ -93,7 +101,7 @@ class CharacterIndex {
[[nodiscard]] Iterator end() const { return {*this, m_entries.size()}; }
[[nodiscard]] Iterator find(const std::size_t cp) const {
const auto it =
std::ranges::lower_bound(m_entries, cp, {}, &InternalEntry::end_cp);
std::ranges::upper_bound(m_entries, cp, {}, &InternalEntry::end_cp);
if (it == m_entries.end()) {
return end();
}
Expand Down
16 changes: 14 additions & 2 deletions src/odr/internal/oldms/text/doc_parser.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,8 @@
#include <odr/internal/oldms/text/doc_io.hpp>
#include <odr/internal/oldms/text/doc_structs.hpp>
#include <odr/internal/oldms/text/doc_style.hpp>
#include <odr/internal/util/byte_stream_util.hpp>
#include <odr/internal/util/stream_util.hpp>

#include <algorithm>
#include <cstdint>
Expand Down Expand Up @@ -107,6 +109,9 @@ struct StyledRun {
std::vector<StyledRun> decode_styled_runs(
std::istream &document_stream, const text::CharacterIndex &character_index,
const text::CharacterRuns &character_runs, const std::size_t ccp_text) {
if (character_index.last_cp() < ccp_text) {
throw std::runtime_error("doc: piece table does not cover the body");
}
std::vector<StyledRun> runs;
std::size_t consumed_cp = 0;

Expand Down Expand Up @@ -158,7 +163,11 @@ ElementIdentifier text::parse_tree(ElementRegistry &registry,
const abstract::ReadableFilesystem &files) {
auto [root_id, _] = registry.create_element(ElementType::root);

const auto document_stream = files.open(AbsPath("/WordDocument"))->stream();
const auto document_file = files.open(AbsPath("/WordDocument"));
if (document_file == nullptr) {
throw std::runtime_error("doc: missing WordDocument stream");
}
const auto document_stream = document_file->stream();
ParsedFib fib;
read(*document_stream, fib);

Expand All @@ -181,7 +190,10 @@ ElementIdentifier text::parse_tree(ElementRegistry &registry,
style_registry = StyleRegistry(std::move(styles));

table_stream->seekg(fib.fibRgFcLcb.clx.fc);
const CharacterIndex character_index = read_character_index(*table_stream);
const std::string clx_bytes =
util::byte_stream::read_u8s(*table_stream, fib.fibRgFcLcb.clx.lcb);
util::stream::ViewStream clx_stream(clx_bytes);
const CharacterIndex character_index = read_character_index(clx_stream);

const auto ccp_text = static_cast<std::size_t>(fib.ccpText());
const std::vector<StyledRun> runs = decode_styled_runs(
Expand Down
25 changes: 19 additions & 6 deletions src/odr/internal/oldms/text/doc_structs.hpp
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
#pragma once

#include <odr/internal/util/byte_string.hpp>

#include <array>
#include <cstdint>
#include <optional>
Expand Down Expand Up @@ -273,20 +275,31 @@ template <typename Data> class PlcMap {
: m_data(data), m_cbPlc(cbPlc) {}

/// Number of data elements; throws unless cbPlc yields a whole number.
[[nodiscard]] std::uint32_t n() const {
[[nodiscard]] std::size_t n() const {
constexpr std::size_t stride = 4 + sizeof(Data);
if (m_cbPlc < 4 || (m_cbPlc - 4) % stride != 0) {
throw std::runtime_error("doc: malformed Plc size");
}
return static_cast<std::uint32_t>((m_cbPlc - 4) / stride);
return (m_cbPlc - 4) / stride;
}

[[nodiscard]] std::uint32_t aCP(const std::uint32_t i) const {
return reinterpret_cast<const std::uint32_t *>(m_data)[i];
[[nodiscard]] std::uint32_t aCP(const std::size_t i) const {
if (i > n()) {
throw std::out_of_range("doc: PLC boundary index out of range");
}
util::byte_string::Reader cursor(std::string_view(m_data, m_cbPlc));
cursor.skip(i * 4);
return cursor.read<std::uint32_t>();
}

[[nodiscard]] Data aData(const std::uint32_t i) const {
return reinterpret_cast<const Data *>(m_data + (n() + 1) * 4)[i];
[[nodiscard]] Data aData(const std::size_t i) const {
const std::size_t count = n();
if (i >= count) {
throw std::out_of_range("doc: PLC data index out of range");
}
util::byte_string::Reader cursor(std::string_view(m_data, m_cbPlc));
cursor.skip((count + 1) * 4 + i * sizeof(Data));
return cursor.read<Data>();
}

private:
Expand Down
42 changes: 24 additions & 18 deletions src/odr/internal/oldms/text/doc_style.cpp
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
#include <odr/internal/oldms/text/doc_style.hpp>

#include <odr/internal/oldms/text/doc_io.hpp>
#include <odr/internal/util/byte_stream_util.hpp>
#include <odr/internal/util/byte_string.hpp>
#include <odr/internal/util/byte_util.hpp>
#include <odr/internal/util/string_util.hpp>

Expand Down Expand Up @@ -173,35 +173,41 @@ std::vector<std::string> text::read_font_names(std::istream &table_stream,
}
table_stream.seekg(sttbf_ffn.fc);

const std::string body =
util::byte_stream::read_u8s(table_stream, sttbf_ffn.lcb);
util::byte_string::Reader cursor(body);

// SttbfFfn is a non-extended STTB ([MS-DOC] 2.2.4, 2.9.286): u16 cData,
// u16 cbExtra (0), then per entry a u8 byte count and an FFN.
const auto c_data = util::byte_stream::read<std::uint16_t>(table_stream);
const auto c_data = cursor.read<std::uint16_t>();
if (c_data == 0xFFFF) {
throw std::runtime_error("doc: unexpected extended SttbfFfn");
}
const auto cb_extra = util::byte_stream::read<std::uint16_t>(table_stream);
const auto cb_extra = cursor.read<std::uint16_t>();

// Inside the declared length, a quirk of one font name does not refuse the
// document: an odd byte after the name and any cbExtra bytes are skipped,
// and a name without its NUL takes all of its units.
result.reserve(c_data);
for (std::uint16_t i = 0; i < c_data; ++i) {
const auto cch_data = util::byte_stream::read<std::uint8_t>(table_stream);
for (std::size_t i = 0; i < c_data; ++i) {
const auto cch_data = cursor.read<std::uint8_t>();
if (cch_data < sizeof(FfnFixed)) {
throw std::runtime_error("doc: FFN too short");
}
const auto ffn = util::byte_stream::read<FfnFixed>(table_stream);
(void)ffn;

// xszFfn: null-terminated UTF-16; an alternative name may follow it.
const std::size_t name_units = (cch_data - sizeof(FfnFixed)) / 2;
std::u16string name = read_string_uncompressed(table_stream, name_units);
if ((cch_data - sizeof(FfnFixed)) % 2 != 0) {
table_stream.ignore(1);
}
if (const std::size_t nul = name.find(u'\0'); nul != std::u16string::npos) {
name.resize(nul);
cursor.skip(sizeof(FfnFixed));
// xszFfn: NUL-terminated UTF-16; an alternative name may follow it.
const std::size_t name_bytes = cch_data - sizeof(FfnFixed);
std::u16string name;
bool terminated = false;
for (std::size_t unit = 0; unit < name_bytes / 2; ++unit) {
const auto character = cursor.read<char16_t>();
terminated = terminated || character == 0;
if (!terminated) {
name.push_back(character);
}
}
cursor.skip(name_bytes % 2 + cb_extra);
result.push_back(util::string::u16string_to_string(name));

table_stream.ignore(cb_extra);
}

return result;
Expand Down
Loading
Loading