diff --git a/docs/mars-text.md b/docs/mars-text.md new file mode 100644 index 0000000..74254ab --- /dev/null +++ b/docs/mars-text.md @@ -0,0 +1,183 @@ +# `mars::text` — flat text readers + +`src/mars/text/` reimplements the three simple text formats the game reads +next to the brace-block catalogs: flat `KEY value` tuning tables (and their +positional-row cousins), the numbered id manifests, and the `#`/`//`-commented +CSVs. Library target: `mars_text` (static, `src/mars/text/CMakeLists.txt`). +Include as `mars/text/
.h`. C++17, no dependencies, no exceptions cross +the API; every reader is total and reports oddities as `Problem`s. + +The behaviour spec is the RE repo's proven Python readers +(`sots-re/verify/parsers/flat_kv.py`, `manifest.py`), which parse all 1,595 +shipped data files. This module is checked against them file by file (below). + +## API + +```cpp +namespace mars::text { + +// result.h +struct Problem { enum class Kind {...}; Kind kind; int line; long long id; std::string message; }; +template struct Result { T value; std::vector problems; bool ok() const; }; + +// value.h — lazily typed tokens +enum class ValueKind { Int, Float, Bool, Bareword, String /*quoted*/ }; +struct Token { std::string text; bool quoted; ValueKind kind() const; + std::optional as_int() const; std::optional as_float() const; + std::optional as_bool() const; }; +struct Color { double r, g, b, a; int components; }; +std::optional parse_color(std::string_view "r g b [a]"); + +// flat_kv.h +struct KvEntry { int line; std::string key /*original spelling*/; std::vector values; }; +class FlatKV { entries(); find(key) /*case-insensitive, last wins*/; has(key); + get_int/get_float/get_bool/get_string/get_color(key); duplicates(); }; +struct Row { int line; std::vector tokens; }; using Rows = std::vector; +Result parse_flat_kv(std::string_view); +Result parse_rows(std::string_view); +std::string_view strip_comment(std::string_view line); std::vector split_tokens(std::string_view line); + +// manifest.h +struct ManifestEntry { int line; int id; std::string name; }; +class Manifest { entries(); deleted(); name_of(id); id_of(name) /*case-insensitive*/; is_deleted(id); }; +Result parse_manifest(std::string_view); + +// csv.h +struct CsvRow { int line; std::vector cells; }; +struct Csv { std::optional> header; std::vector rows; }; +Result parse_csv(std::string_view); // comments dropped, cells stripped +Result> split_csv_records(std::string_view); // raw RFC-4180 split +} +``` + +Input is the file's raw bytes (`std::string_view`). The files are cp1252 with +mixed CRLF/LF; bytes ≥ 0x80 are passed through untouched and are never +whitespace or case-folded. Whitespace is ASCII only (space, `\t`, `\r`, `\n`, +`\v`, `\f`). + +## Behaviour + +### Flat `KEY value` tables (`parse_flat_kv`) and row tables (`parse_rows`) + +Per line: strip a trailing `//` comment (quote-aware: `//` inside `"..."` is +text, and an unclosed `"` disables comment stripping for the rest of the +line), trim, skip if empty, then tokenise: + +- `"..."` is one token (may be empty, may contain spaces/tabs/`//`); no escape + processing, so `"Data\\x.txt"` keeps both backslashes. +- Otherwise a run of non-whitespace is one bareword; a `"` that has no + closing quote, or sits mid-word, is an ordinary character. + +`parse_flat_kv` takes the first token as the key and the remaining tokens as +the value: none (key-only line), one (scalar) or several (a list — the +reference reader returns a Python list here; none of the 20 shipped tables +actually do this). `parse_rows` returns every line's token list. + +Typing is lazy: `Token::kind()` classifies a bareword exactly like the +reference `coerce()`: `[+-]?\d+` → Int, `[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?` +→ Float (`.5`, `-.8`, `1.`, `7e+8`), `true`/`false` any case → Bool, else +Bareword (`0x10`, `inf`, `nan`, `1,5` are barewords). A quoted token is always +a String — `"8"` never becomes a number. `as_float()` also accepts Int shapes; +`as_int()` refuses values that do not fit in 64 bits. + +Keys: lookup is ASCII case-insensitive (`find("mars_default_color")` finds +`MARS_DEFAULT_COLOR`); the entry keeps the original spelling and line. When a +key repeats, `find()` returns the **last** line — the same "later line wins" +default as the reference `parse_kv(on_dup='last')`. All entries stay in +`entries()`; `duplicates()` lists repeated (case-folded) keys with their line +numbers. The shipped tables contain no duplicates, exact or case-folded, so +the case-folding in `duplicates()` is a deliberate superset of the reference's +exact-key check. + +Colours are quoted `"r g b"` / `"r g b a"` strings (tabs inside are fine, as in +`globals.txt`); `get_color` / `parse_color` require 3 or 4 numeric components. + +Problems: `UnbalancedQuote` (odd number of `"` on a line) is reported but the +line is still parsed. None occur in shipped data. + +### Id manifests (`parse_manifest`) + +Per trimmed line: blank → skip; a `// DELETED - ` marker anywhere on the +line (word case-insensitive, whitespace optional) → tombstone in `deleted()`; +other `//...` → comment; ` ` and nothing else → entry. +Anything else is `Problem::Kind::Unrecognised` — including an entry followed +by a trailing comment, matching the reference regex `^\s*(\d+)\s+(\S+)\s*$`. +A repeated id is `DuplicateId` (the later entry is kept and wins lookups); an +id that is both a tombstone and an entry is `DeletedAndAssigned`. +`id_of()` folds case (`DEWar.SHIPSECTION` ↔ `dewar.shipsection`). + +### CSV (`parse_csv`, `split_csv_records`) + +`split_csv_records` reproduces the reference (Python `csv`, excel dialect, +non-strict) byte for byte on the shipped files: + +- `,` delimits; `\r\n`, `\n` and a lone `\r` each end a record; a blank line is + an empty record `[]`; a trailing `,` yields a trailing empty cell; the last + record needs no terminator. +- A cell is quoted only if `"` is its **first byte**. Inside quotes `""` is a + literal `"` and newline bytes are kept verbatim (`Strings.csv` has one such + cell). Text after the closing quote is appended as-is (`"tail"junk` → + `tailjunk`). +- A `"` anywhere else is an ordinary byte: `EVENTMSG_DERELICT_UNRESOLVED"` + keeps its quote, and ` "Effects/Blastoid1.effect"` (Asteroids.csv, quote + after leading spaces) keeps both quotes. +- End of input inside a quoted cell → `UnterminatedQuotedCell`, cell kept. + +`parse_csv` then drops comment rows (first cell, after leading whitespace, +starts with `#` or `//`), empty rows and rows whose cells are all blank +(`,,`), and strips every remaining cell. The header is the first `#` row with +more than one cell that appears before any data row, with `#` and any `<`/`>` +trimmed from each cell: `# ,` → `tech, human-pri`; +`"# species","event"` → `species, event`; `#Key+A955,String,Size,Notes` → +`Key+A955, ...`. A `#` row with a single cell is a plain comment and does not +end the search. + +## Oracle results + +`tests/mars_text/build_and_run.sh` compiles with `g++ -std=c++17 -Wall -Wextra +-Werror`, runs the unit tests (165 checks, hand-written samples for every +quirk above), the real-data facts test, and — when `SOTS_DATA_DIR` is set — +dumps every file `mars_text` owns through both `dump_json` (C++) and +`oracle/dump.py` (the Python readers) and compares them structurally, +type-aware (`compare.py`). + +Run 2026-09-07 against the extracted `sots.gob` + `sots_local_en.gob` tree: + +| kind | files | agree | +|---|---|---| +| kv (`Data/**/*.txt`) | 20 | 20 | +| rows (`_turrets.txt`, `_defaultweapons.txt`, `damfx*.txt`, `playercolors.txt`, `BadgeTable.txt`, `AvatarTable.txt`, `WeaponIconPlacements.txt`) | 8 | 8 | +| manifest (`_weapons.txt`, 7 × `_shipsections.txt`) | 8 | 8 | +| csv (AI tables, scenarios, `RealSpace.csv`, `SpriteTable.csv`, `Strings.csv`, `SpeechEvents.csv`, asteroids, music, sound_ui) | 28 | 28 | +| **total** | **64** | **64 (100%)** | + +Compared per file: the full key → value map (exact keys, last wins, typed +values), duplicate-key report, every row's typed tokens, manifest entries / +tombstones / problems (kind, line, id), CSV header and every stripped cell. +Real-data facts also asserted: `globals.txt` 364 keys, `MARS_DEFAULT_COLOR` = +(255,177,39); `_turrets.txt` 42 × 8 tokens; `_weapons.txt` 123 ids, deleted +36/58/59; Human `_shipsections.txt` 145 ids with `dewar.shipsection` → 98; +`Strings.csv` 5,722 raw records → 5,200 data rows → 5,196 distinct keys, +exactly one multi-line raw cell; `aitechpri.csv` 0 rows with a 7-name schema. + +## Deliberate divergences from the reference + +None affect shipped data (verified above); listed so nobody chases them: + +- Whitespace is ASCII-only. The Python readers operate on decoded `str`, so + they would also treat cp1252 `0xA0` (NBSP) and bytes `0x1C`–`0x1F` as + whitespace / line breaks. The relevant files contain no such bytes. +- `duplicates()` folds key case; the reference compares exact keys. +- Ints are 64-bit; the reference has unbounded ints. An overflowing bareword + classifies as `Int` but `as_int()` returns nullopt. +- The reference `csv` module rejects a NUL byte; this reader treats it as data. +- `Problem` messages are our own wording; the oracle comparison matches on + kind / line / id, not text. + +## Open questions + +- The engine's own tolerance is unknown for the cases where the reference is + strict (manifest entry with a trailing comment, a key that repeats). We copy + the reference; both are absent from shipped data. +- Multi-token `KEY a b c` values never occur in shipped tables; the API keeps + them as a token list rather than guessing a type. diff --git a/src/mars/text/CMakeLists.txt b/src/mars/text/CMakeLists.txt new file mode 100644 index 0000000..4ab5ca2 --- /dev/null +++ b/src/mars/text/CMakeLists.txt @@ -0,0 +1,11 @@ +# mars_text -- flat KEY/value tables, id manifests, commented CSVs. +# Include from the root with add_subdirectory(src/mars/text); link mars_text. +add_library(mars_text STATIC + value.cpp + flat_kv.cpp + manifest.cpp + csv.cpp +) +target_include_directories(mars_text PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../..) +target_compile_features(mars_text PUBLIC cxx_std_17) +target_compile_options(mars_text PRIVATE -Wall -Wextra) diff --git a/src/mars/text/csv.cpp b/src/mars/text/csv.cpp new file mode 100644 index 0000000..45c2686 --- /dev/null +++ b/src/mars/text/csv.cpp @@ -0,0 +1,134 @@ +#include "mars/text/csv.h" + +#include "mars/text/value.h" + +namespace mars::text { + +Result> split_csv_records(std::string_view text) { + enum class State { StartRecord, StartField, InField, InQuotedField, QuoteInQuotedField }; + + Result> r; + std::vector& records = r.value; + State state = State::StartRecord; + CsvRow row; + std::string field; + int lineno = 1; + + auto save_field = [&] { + row.cells.push_back(std::move(field)); + field.clear(); + }; + auto end_record = [&] { + records.push_back(std::move(row)); + row = CsvRow{}; + state = State::StartRecord; + }; + + const std::size_t n = text.size(); + for (std::size_t i = 0; i < n; ++i) { + char c = text[i]; + bool newline = (c == '\n' || c == '\r'); + // "\r\n" is a single terminator; `term` holds its bytes verbatim. + std::string_view term; + if (newline) { + std::size_t len = (c == '\r' && i + 1 < n && text[i + 1] == '\n') ? 2 : 1; + term = text.substr(i, len); + i += len - 1; + } + if (state == State::StartRecord) { + row.line = lineno; + if (newline) { + end_record(); // blank line -> empty record + ++lineno; + continue; + } + state = State::StartField; + } + switch (state) { + case State::StartField: + if (newline) { save_field(); end_record(); } + else if (c == '"') state = State::InQuotedField; + else if (c == ',') save_field(); + else { field.push_back(c); state = State::InField; } + break; + case State::InField: + if (newline) { save_field(); end_record(); } + else if (c == ',') { save_field(); state = State::StartField; } + else field.push_back(c); + break; + case State::InQuotedField: + if (c == '"') state = State::QuoteInQuotedField; + else if (newline) field.append(term); // multi-line cell: keep the bytes as written + else field.push_back(c); + break; + case State::QuoteInQuotedField: + if (c == '"') { field.push_back('"'); state = State::InQuotedField; } + else if (c == ',') { save_field(); state = State::StartField; } + else if (newline) { save_field(); end_record(); } + else { field.push_back(c); state = State::InField; } // lenient: text after a closing quote + break; + case State::StartRecord: + break; // unreachable + } + if (newline) ++lineno; + } + if (state != State::StartRecord) { + if (state == State::InQuotedField) + r.problems.push_back(Problem{Problem::Kind::UnterminatedQuotedCell, row.line, -1, + "line " + std::to_string(row.line) + ": end of input inside a quoted cell"}); + save_field(); + end_record(); + } + return r; +} + +namespace { + +bool starts_with(std::string_view s, std::string_view p) { return s.substr(0, p.size()) == p; } + +bool is_comment_row(const CsvRow& row) { + if (row.cells.empty()) return true; + std::string_view first = lstrip(row.cells.front()); + if (starts_with(first, "#") || starts_with(first, "//")) return true; + for (const std::string& c : row.cells) + if (!strip(c).empty()) return false; + return true; +} + +std::string header_name(std::string_view cell) { + cell = strip(cell); + while (!cell.empty() && cell.front() == '#') cell.remove_prefix(1); + cell = strip(cell); + while (!cell.empty() && (cell.front() == '<' || cell.front() == '>')) cell.remove_prefix(1); + while (!cell.empty() && (cell.back() == '<' || cell.back() == '>')) cell.remove_suffix(1); + return std::string(cell); +} + +} // namespace + +Result parse_csv(std::string_view text) { + Result> raw = split_csv_records(text); + Result r; + r.problems = std::move(raw.problems); + + bool header_settled = false; + for (CsvRow& row : raw.value) { + if (!header_settled && !row.cells.empty()) { + std::string_view first = lstrip(row.cells.front()); + if (starts_with(first, "#") && row.cells.size() > 1) { + std::vector names; + for (const std::string& c : row.cells) names.push_back(header_name(c)); + r.value.header = std::move(names); + header_settled = true; + } else if (!is_comment_row(row)) { + header_settled = true; // data before any schema row: no header + } + } + if (is_comment_row(row)) continue; + for (std::string& c : row.cells) c = std::string(strip(c)); + r.value.rows.push_back(std::move(row)); + } + return r; +} + +} // namespace mars::text diff --git a/src/mars/text/csv.h b/src/mars/text/csv.h new file mode 100644 index 0000000..e554698 --- /dev/null +++ b/src/mars/text/csv.h @@ -0,0 +1,44 @@ +// mars::text -- the '#' / '//'-commented CSVs (Data/Strategy/**/*.csv, +// Scenarios/*.csv, GUI/SpriteTable.csv, Locale/EN/Strings.csv ...). +// +// Records follow RFC 4180 as the game's files use it: ',' delimiter, '"' +// quotes a cell only when it is the cell's first byte, '""' inside quotes is +// a literal quote, a quoted cell may span lines (the newline bytes are kept +// verbatim), and "\r\n" / "\n" / "\r" all end a record. A '"' anywhere else +// is an ordinary byte, and text after a closing quote is appended as-is. +// +// parse_csv() drops comment rows (first cell starts with '#' or '//' after +// leading whitespace), empty rows and rows whose cells are all blank, and +// strips every remaining cell. The first '#' row with more than one cell, +// if it precedes the data, is returned as the header with '#' and '<>' +// trimmed ("# ," -> tech, human-pri). +#pragma once + +#include +#include +#include +#include + +#include "mars/text/result.h" + +namespace mars::text { + +struct CsvRow { + int line = 0; // line on which the record starts + std::vector cells; +}; + +struct Csv { + std::optional> header; + std::vector rows; + + std::size_t size() const { return rows.size(); } +}; + +Result parse_csv(std::string_view text); + +// The raw record split (no comment filtering, no stripping); exposed for +// tools and tests. +Result> split_csv_records(std::string_view text); + +} // namespace mars::text diff --git a/src/mars/text/flat_kv.cpp b/src/mars/text/flat_kv.cpp new file mode 100644 index 0000000..92238ff --- /dev/null +++ b/src/mars/text/flat_kv.cpp @@ -0,0 +1,149 @@ +#include "mars/text/flat_kv.h" + +#include "mars/text/lines.h" + +namespace mars::text { + +std::string_view strip_comment(std::string_view line) { + bool in_quote = false; + for (std::size_t i = 0; i < line.size(); ++i) { + char c = line[i]; + if (c == '"') { + in_quote = !in_quote; + } else if (c == '/' && !in_quote && i + 1 < line.size() && line[i + 1] == '/') { + return line.substr(0, i); + } + } + return line; +} + +std::vector split_tokens(std::string_view line) { + std::vector out; + std::size_t i = 0, n = line.size(); + while (i < n) { + if (is_space(line[i])) { ++i; continue; } + if (line[i] == '"') { + std::size_t close = line.find('"', i + 1); + if (close != std::string_view::npos) { + out.push_back(Token{std::string(line.substr(i + 1, close - i - 1)), true}); + i = close + 1; + continue; + } + // no closing quote: the '"' is an ordinary bareword character + } + std::size_t j = i; + while (j < n && !is_space(line[j])) ++j; + out.push_back(Token{std::string(line.substr(i, j - i)), false}); + i = j; + } + return out; +} + +namespace { + +bool has_unbalanced_quote(std::string_view line) { + std::size_t quotes = 0; + for (char c : line) + if (c == '"') ++quotes; + return quotes % 2 == 1; +} + +// Shared line pass: yields (line number, tokens) for every non-blank line. +template +void tokenised_lines(std::string_view text, std::vector& problems, Fn&& fn) { + for_each_line(text, [&](int lineno, std::string_view raw) { + std::string_view line = strip(strip_comment(raw)); + if (line.empty()) return; + if (has_unbalanced_quote(line)) + problems.push_back(Problem{Problem::Kind::UnbalancedQuote, lineno, -1, + "line " + std::to_string(lineno) + ": unbalanced quote"}); + fn(lineno, split_tokens(line)); + }); +} + +} // namespace + +// ---- FlatKV -------------------------------------------------------------- + +void FlatKV::add(KvEntry entry) { + last_[fold_case(entry.key)] = entries_.size(); + entries_.push_back(std::move(entry)); +} + +const KvEntry* FlatKV::find(std::string_view key) const { + auto it = last_.find(fold_case(key)); + return it == last_.end() ? nullptr : &entries_[it->second]; +} + +namespace { +const Token* first_token(const FlatKV& kv, std::string_view key) { + const KvEntry* e = kv.find(key); + return (e && !e->values.empty()) ? &e->values.front() : nullptr; +} +} // namespace + +std::optional FlatKV::get_int(std::string_view key) const { + const Token* t = first_token(*this, key); + return t ? t->as_int() : std::nullopt; +} + +std::optional FlatKV::get_float(std::string_view key) const { + const Token* t = first_token(*this, key); + return t ? t->as_float() : std::nullopt; +} + +std::optional FlatKV::get_bool(std::string_view key) const { + const Token* t = first_token(*this, key); + return t ? t->as_bool() : std::nullopt; +} + +std::optional FlatKV::get_string(std::string_view key) const { + const Token* t = first_token(*this, key); + if (!t) return std::nullopt; + return std::string_view(t->text); +} + +std::optional FlatKV::get_color(std::string_view key) const { + const Token* t = first_token(*this, key); + return t ? parse_color(t->text) : std::nullopt; +} + +std::vector>> FlatKV::duplicates() const { + std::unordered_map> lines; + std::vector order; + for (const KvEntry& e : entries_) { + std::string k = fold_case(e.key); + auto& v = lines[k]; + if (v.empty()) order.push_back(k); + v.push_back(e.line); + } + std::vector>> out; + for (const std::string& k : order) + if (lines[k].size() > 1) out.emplace_back(k, lines[k]); + return out; +} + +Result parse_flat_kv(std::string_view text) { + Result r; + tokenised_lines(text, r.problems, [&](int lineno, std::vector toks) { + KvEntry e; + e.line = lineno; + e.key = std::move(toks.front().text); + toks.erase(toks.begin()); + e.values = std::move(toks); + r.value.add(std::move(e)); + }); + return r; +} + +// ---- rows ---------------------------------------------------------------- + +Result parse_rows(std::string_view text) { + Result r; + tokenised_lines(text, r.problems, [&](int lineno, std::vector toks) { + r.value.push_back(Row{lineno, std::move(toks)}); + }); + return r; +} + +} // namespace mars::text diff --git a/src/mars/text/flat_kv.h b/src/mars/text/flat_kv.h new file mode 100644 index 0000000..d11eab0 --- /dev/null +++ b/src/mars/text/flat_kv.h @@ -0,0 +1,72 @@ +// mars::text -- the flat "KEY value" tuning tables and whitespace-positional +// row tables (Data/**/*.txt, Weapons/_turrets.txt, Badges/BadgeTable.txt ...). +// +// Line grammar (both shapes): +// line := token* ('//' comment)? -- '//' inside "..." is not a comment +// token := '"' [^"]* '"' | non-whitespace+ -- a quoted token may be empty ("") +// -- an unpaired '"' is just a bareword char +// KEY value tables take the first token of a line as the key and the rest as +// its value(s); a key alone is allowed (empty value). Keys are looked up +// case-insensitively; when a key repeats, the last line wins (earlier entries +// stay visible through entries() / duplicates()). +#pragma once + +#include +#include +#include +#include +#include +#include + +#include "mars/text/result.h" +#include "mars/text/value.h" + +namespace mars::text { + +struct KvEntry { + int line = 0; + std::string key; // original spelling + std::vector values; // 0 = key only, 1 = scalar, n = list +}; + +class FlatKV { +public: + const std::vector& entries() const { return entries_; } + std::size_t size() const { return entries_.size(); } + + // Case-insensitive lookup; the last entry with that key. + const KvEntry* find(std::string_view key) const; + bool has(std::string_view key) const { return find(key) != nullptr; } + + // Convenience accessors on the first value token of the entry. + // nullopt when the key is missing, has no value, or the token has the wrong shape. + std::optional get_int(std::string_view key) const; + std::optional get_float(std::string_view key) const; + std::optional get_bool(std::string_view key) const; + std::optional get_string(std::string_view key) const; // any token kind + std::optional get_color(std::string_view key) const; + + // Keys (case-folded) that occur more than once -> their line numbers. + std::vector>> duplicates() const; + + void add(KvEntry entry); + +private: + std::vector entries_; + std::unordered_map last_; // folded key -> index of last entry +}; + +struct Row { + int line = 0; + std::vector tokens; +}; +using Rows = std::vector; + +Result parse_flat_kv(std::string_view text); +Result parse_rows(std::string_view text); + +// Building blocks, exposed for tests and for one-off line readers. +std::string_view strip_comment(std::string_view line); // drop a trailing // comment (quote-aware) +std::vector split_tokens(std::string_view line); // tokenise per the grammar above + +} // namespace mars::text diff --git a/src/mars/text/lines.h b/src/mars/text/lines.h new file mode 100644 index 0000000..74f927e --- /dev/null +++ b/src/mars/text/lines.h @@ -0,0 +1,28 @@ +// mars::text -- internal: iterate the lines of a text buffer. +// Line terminators are "\r\n", "\n" and a lone "\r" (the shipped files mix +// CRLF and LF). A trailing terminator does not produce an extra empty line. +#pragma once + +#include + +namespace mars::text { + +template +void for_each_line(std::string_view text, Fn&& fn) { + int lineno = 1; + std::size_t start = 0; + const std::size_t n = text.size(); + while (start < n) { + std::size_t end = start; + while (end < n && text[end] != '\n' && text[end] != '\r') ++end; + fn(lineno, text.substr(start, end - start)); + if (end < n) { + if (text[end] == '\r' && end + 1 < n && text[end + 1] == '\n') ++end; + ++end; + } + start = end; + ++lineno; + } +} + +} // namespace mars::text diff --git a/src/mars/text/manifest.cpp b/src/mars/text/manifest.cpp new file mode 100644 index 0000000..7b61196 --- /dev/null +++ b/src/mars/text/manifest.cpp @@ -0,0 +1,123 @@ +#include "mars/text/manifest.h" + +#include +#include + +#include "mars/text/lines.h" +#include "mars/text/value.h" + +namespace mars::text { + +void Manifest::add_entry(ManifestEntry entry) { + by_id_[entry.id] = entries_.size(); + by_name_[fold_case(entry.name)] = entries_.size(); + entries_.push_back(std::move(entry)); +} + +std::optional Manifest::name_of(int id) const { + auto it = by_id_.find(id); + if (it == by_id_.end()) return std::nullopt; + return std::string_view(entries_[it->second].name); +} + +std::optional Manifest::id_of(std::string_view name) const { + auto it = by_name_.find(fold_case(name)); + if (it == by_name_.end()) return std::nullopt; + return entries_[it->second].id; +} + +bool Manifest::is_deleted(int id) const { + return std::find(deleted_.begin(), deleted_.end(), id) != deleted_.end(); +} + +namespace { + +bool is_digit(char c) { return c >= '0' && c <= '9'; } + +std::optional read_int(std::string_view digits) { + int v = 0; + auto r = std::from_chars(digits.data(), digits.data() + digits.size(), v); + if (r.ec != std::errc{} || r.ptr != digits.data() + digits.size()) return std::nullopt; + return v; +} + +std::size_t skip_ws(std::string_view s, std::size_t i) { + while (i < s.size() && is_space(s[i])) ++i; + return i; +} + +// "// DELETED - " anywhere on the line (word matched case-insensitively, +// whitespace optional around the dash). Returns the id. +std::optional deleted_marker(std::string_view line) { + for (std::size_t at = line.find("//"); at != std::string_view::npos; at = line.find("//", at + 1)) { + std::size_t i = skip_ws(line, at + 2); + static constexpr std::string_view word = "deleted"; + if (line.size() - i < word.size() || !equals_fold(line.substr(i, word.size()), word)) continue; + i = skip_ws(line, i + word.size()); + if (i >= line.size() || line[i] != '-') continue; + i = skip_ws(line, i + 1); + std::size_t j = i; + while (j < line.size() && is_digit(line[j])) ++j; + if (j == i) continue; + if (auto id = read_int(line.substr(i, j - i))) return id; + } + return std::nullopt; +} + +// " " and nothing else (line is already stripped). +std::optional parse_entry(std::string_view line) { + std::size_t i = 0; + while (i < line.size() && is_digit(line[i])) ++i; + if (i == 0) return std::nullopt; + std::string_view digits = line.substr(0, i); + std::size_t j = skip_ws(line, i); + if (j == i) return std::nullopt; // need whitespace between id and name + std::size_t k = j; + while (k < line.size() && !is_space(line[k])) ++k; + if (k == j || k != line.size()) return std::nullopt; + auto id = read_int(digits); + if (!id) return std::nullopt; + ManifestEntry e; + e.id = *id; + e.name = std::string(line.substr(j, k - j)); + return e; +} + +} // namespace + +Result parse_manifest(std::string_view text) { + Result r; + std::unordered_map first_line; + for_each_line(text, [&](int lineno, std::string_view raw) { + std::string_view line = strip(raw); + if (line.empty()) return; + if (auto id = deleted_marker(line)) { + r.value.add_deleted(*id); + return; + } + if (line.substr(0, 2) == "//") return; + auto entry = parse_entry(line); + if (!entry) { + r.problems.push_back(Problem{Problem::Kind::Unrecognised, lineno, -1, + "line " + std::to_string(lineno) + ": unrecognised '" + std::string(line) + "'"}); + return; + } + entry->line = lineno; + auto seen = first_line.find(entry->id); + if (seen != first_line.end()) { + r.problems.push_back(Problem{Problem::Kind::DuplicateId, lineno, entry->id, + "line " + std::to_string(lineno) + ": duplicate id " + std::to_string(entry->id) + + " (first at line " + std::to_string(seen->second) + ")"}); + } + first_line[entry->id] = lineno; + r.value.add_entry(std::move(*entry)); + }); + for (int id : r.value.deleted()) { + if (first_line.count(id)) + r.problems.push_back(Problem{Problem::Kind::DeletedAndAssigned, 0, id, + "id " + std::to_string(id) + " is both DELETED and assigned"}); + } + return r; +} + +} // namespace mars::text diff --git a/src/mars/text/manifest.h b/src/mars/text/manifest.h new file mode 100644 index 0000000..526dcbd --- /dev/null +++ b/src/mars/text/manifest.h @@ -0,0 +1,54 @@ +// mars::text -- numbered id manifests (Weapons/_weapons.txt, +// Species//sections/_shipsections.txt). +// +// one entry per line; ids are the persistent +// network/savegame ids and must be unique +// // DELETED - a retired id, kept as a tombstone +// // anything else comment +// +// Any other non-blank line is reported as Problem::Kind::Unrecognised (an +// entry followed by a trailing comment is *not* accepted, matching the +// reference reader). Names are matched case-insensitively: the shipped +// manifests spell 'DEWar.SHIPSECTION' against lower-case files. +#pragma once + +#include +#include +#include +#include +#include + +#include "mars/text/result.h" + +namespace mars::text { + +struct ManifestEntry { + int line = 0; + int id = 0; + std::string name; // as written +}; + +class Manifest { +public: + const std::vector& entries() const { return entries_; } + const std::vector& deleted() const { return deleted_; } + + // id -> name (last assignment wins if an id repeats) + std::optional name_of(int id) const; + // name -> id, case-insensitive (last assignment wins if a name repeats) + std::optional id_of(std::string_view name) const; + bool is_deleted(int id) const; + + void add_entry(ManifestEntry entry); + void add_deleted(int id) { deleted_.push_back(id); } + +private: + std::vector entries_; + std::vector deleted_; + std::unordered_map by_id_; + std::unordered_map by_name_; // folded name +}; + +Result parse_manifest(std::string_view text); + +} // namespace mars::text diff --git a/src/mars/text/result.h b/src/mars/text/result.h new file mode 100644 index 0000000..45d27f4 --- /dev/null +++ b/src/mars/text/result.h @@ -0,0 +1,34 @@ +// mars::text -- result/diagnostic types shared by the text readers. +// +// The readers are total: every input yields a value. Anything worth telling +// the caller about (a malformed manifest line, a quote that never closes) is +// reported as a Problem alongside the value instead of thrown. +#pragma once + +#include +#include + +namespace mars::text { + +struct Problem { + enum class Kind { + Unrecognised, // manifest: line is neither entry, comment nor tombstone + DuplicateId, // manifest: id assigned twice (id = the repeated id) + DeletedAndAssigned, // manifest: id is both a tombstone and an entry (line = 0) + UnbalancedQuote, // flat kv / rows: odd number of '"' on a line + UnterminatedQuotedCell // csv: end of input inside a quoted cell + }; + Kind kind; + int line = 0; // 1-based; 0 when the problem is not tied to a line + long long id = -1; // manifest id when relevant, else -1 + std::string message; // human-readable, for logs +}; + +template +struct Result { + T value{}; + std::vector problems; + bool ok() const { return problems.empty(); } +}; + +} // namespace mars::text diff --git a/src/mars/text/value.cpp b/src/mars/text/value.cpp new file mode 100644 index 0000000..05e645c --- /dev/null +++ b/src/mars/text/value.cpp @@ -0,0 +1,153 @@ +#include "mars/text/value.h" + +#include +#include +#include + +namespace mars::text { + +namespace { + +bool is_digit(char c) { return c >= '0' && c <= '9'; } + +char lower(char c) { return (c >= 'A' && c <= 'Z') ? static_cast(c - 'A' + 'a') : c; } + +std::string_view skip_sign(std::string_view s) { + if (!s.empty() && (s[0] == '+' || s[0] == '-')) s.remove_prefix(1); + return s; +} + +} // namespace + +bool looks_like_int(std::string_view s) { + s = skip_sign(s); + if (s.empty()) return false; + for (char c : s) + if (!is_digit(c)) return false; + return true; +} + +bool looks_like_float(std::string_view s) { + // [+-]? ( d+ '.' d* | '.' d+ | d+ ) ( [eE] [+-]? d+ )? + s = skip_sign(s); + std::size_t i = 0, n = s.size(); + std::size_t whole = 0; + while (i < n && is_digit(s[i])) { ++i; ++whole; } + if (i < n && s[i] == '.') { + ++i; + std::size_t frac = 0; + while (i < n && is_digit(s[i])) { ++i; ++frac; } + if (whole == 0 && frac == 0) return false; + } else if (whole == 0) { + return false; + } + if (i < n && (s[i] == 'e' || s[i] == 'E')) { + ++i; + if (i < n && (s[i] == '+' || s[i] == '-')) ++i; + std::size_t exp = 0; + while (i < n && is_digit(s[i])) { ++i; ++exp; } + if (exp == 0) return false; + } + return i == n; +} + +ValueKind classify(std::string_view s) { + if (looks_like_int(s)) return ValueKind::Int; + if (looks_like_float(s)) return ValueKind::Float; + if (parse_bool(s)) return ValueKind::Bool; + return ValueKind::Bareword; +} + +std::optional parse_int(std::string_view s) { + if (!looks_like_int(s)) return std::nullopt; + bool neg = !s.empty() && s[0] == '-'; + std::string_view digits = skip_sign(s); + std::uint64_t mag = 0; + auto r = std::from_chars(digits.data(), digits.data() + digits.size(), mag); + if (r.ec != std::errc{} || r.ptr != digits.data() + digits.size()) return std::nullopt; + const std::uint64_t limit = neg ? (static_cast(INT64_MAX) + 1u) + : static_cast(INT64_MAX); + if (mag > limit) return std::nullopt; + if (neg) return mag == limit ? INT64_MIN : -static_cast(mag); + return static_cast(mag); +} + +std::optional parse_float(std::string_view s) { + if (!looks_like_float(s)) return std::nullopt; // Int shapes pass this too + bool neg = !s.empty() && s[0] == '-'; + std::string_view body = skip_sign(s); + double v = 0; +#if defined(__cpp_lib_to_chars) && __cpp_lib_to_chars >= 201611L + auto r = std::from_chars(body.data(), body.data() + body.size(), v); + if (r.ec == std::errc::result_out_of_range) { + // Out-of-range magnitudes: let strtod saturate (inf / 0) like Python's float(). + std::string tmp(body); + v = std::strtod(tmp.c_str(), nullptr); + } else if (r.ec != std::errc{} || r.ptr != body.data() + body.size()) { + return std::nullopt; + } +#else + std::string tmp(body); + char* end = nullptr; + v = std::strtod(tmp.c_str(), &end); + if (end != tmp.c_str() + tmp.size()) return std::nullopt; +#endif + return neg ? -v : v; +} + +std::optional parse_bool(std::string_view s) { + if (equals_fold(s, "true")) return true; + if (equals_fold(s, "false")) return false; + return std::nullopt; +} + +std::optional parse_color(std::string_view s) { + std::vector parts; + std::size_t i = 0, n = s.size(); + while (i < n) { + while (i < n && is_space(s[i])) ++i; + if (i >= n) break; + std::size_t j = i; + while (j < n && !is_space(s[j])) ++j; + auto v = parse_float(s.substr(i, j - i)); + if (!v) return std::nullopt; + parts.push_back(*v); + i = j; + } + if (parts.size() != 3 && parts.size() != 4) return std::nullopt; + Color c; + c.r = parts[0]; + c.g = parts[1]; + c.b = parts[2]; + if (parts.size() == 4) c.a = parts[3]; + c.components = static_cast(parts.size()); + return c; +} + +std::string_view lstrip(std::string_view s) { + std::size_t i = 0; + while (i < s.size() && is_space(s[i])) ++i; + return s.substr(i); +} + +std::string_view strip(std::string_view s) { + s = lstrip(s); + std::size_t n = s.size(); + while (n > 0 && is_space(s[n - 1])) --n; + return s.substr(0, n); +} + +std::string fold_case(std::string_view s) { + std::string out(s); + for (char& c : out) c = lower(c); + return out; +} + +bool equals_fold(std::string_view a, std::string_view b) { + if (a.size() != b.size()) return false; + for (std::size_t i = 0; i < a.size(); ++i) + if (lower(a[i]) != lower(b[i])) return false; + return true; +} + +} // namespace mars::text diff --git a/src/mars/text/value.h b/src/mars/text/value.h new file mode 100644 index 0000000..697d327 --- /dev/null +++ b/src/mars/text/value.h @@ -0,0 +1,64 @@ +// mars::text -- lazily typed tokens for the flat tuning tables. +// +// A token is the raw bytes of one whitespace-delimited item (or the inside of +// a "quoted string"). Nothing is converted at parse time; kind() classifies a +// bareword the way the game data reads it (int / float / true|false / word) +// and the as_*() accessors convert on demand. Quoted tokens are always +// strings, even when they look numeric ("8" stays "8"). +#pragma once + +#include +#include +#include +#include + +namespace mars::text { + +enum class ValueKind { + Int, // [+-]?digits + Float, // [+-]?(d+.d* | .d+ | d+)([eE][+-]?d+)? (and not Int) + Bool, // true / false, any case + Bareword, // any other unquoted token + String // a "quoted" token +}; + +// Classification of an unquoted token. +ValueKind classify(std::string_view bareword); +bool looks_like_int(std::string_view s); +bool looks_like_float(std::string_view s); + +// Conversions; nullopt when the text does not have the requested shape +// (or, for parse_int, does not fit in 64 bits). +std::optional parse_int(std::string_view s); +std::optional parse_float(std::string_view s); // accepts Int or Float shapes +std::optional parse_bool(std::string_view s); + +struct Token { + std::string text; + bool quoted = false; + + ValueKind kind() const { return quoted ? ValueKind::String : classify(text); } + std::optional as_int() const { return quoted ? std::nullopt : parse_int(text); } + std::optional as_float() const { return quoted ? std::nullopt : parse_float(text); } + std::optional as_bool() const { return quoted ? std::nullopt : parse_bool(text); } +}; + +// A colour written as "r g b" or "r g b a" (any numeric components; the data +// uses 0-255 ints). a defaults to 1 when absent. `components` is 3 or 4. +struct Color { + double r = 0, g = 0, b = 0, a = 1; + int components = 0; +}; +std::optional parse_color(std::string_view s); + +// ASCII helpers shared by the readers (the data is cp1252; high bytes are +// never whitespace and never case-folded). +inline bool is_space(char c) { + return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\v' || c == '\f'; +} +std::string_view strip(std::string_view s); +std::string_view lstrip(std::string_view s); +std::string fold_case(std::string_view s); // ASCII lower-case copy +bool equals_fold(std::string_view a, std::string_view b); // ASCII case-insensitive + +} // namespace mars::text diff --git a/tests/mars_text/.gitignore b/tests/mars_text/.gitignore new file mode 100644 index 0000000..567609b --- /dev/null +++ b/tests/mars_text/.gitignore @@ -0,0 +1 @@ +build/ diff --git a/tests/mars_text/CMakeLists.txt b/tests/mars_text/CMakeLists.txt new file mode 100644 index 0000000..bf5ba66 --- /dev/null +++ b/tests/mars_text/CMakeLists.txt @@ -0,0 +1,13 @@ +# Optional CMake wiring for the mars_text tests; the canonical runner is +# build_and_run.sh (plain g++). Include from the root with +# add_subdirectory(tests/mars_text) after add_subdirectory(src/mars/text). +add_executable(mars_text_unit_tests unit_tests.cpp) +target_link_libraries(mars_text_unit_tests PRIVATE mars_text) +add_test(NAME mars_text_unit COMMAND mars_text_unit_tests) + +add_executable(mars_text_realdata_test realdata_test.cpp) +target_link_libraries(mars_text_realdata_test PRIVATE mars_text) +add_test(NAME mars_text_realdata COMMAND mars_text_realdata_test) # SKIPs without SOTS_DATA_DIR + +add_executable(mars_text_dump_json dump_json.cpp) +target_link_libraries(mars_text_dump_json PRIVATE mars_text) diff --git a/tests/mars_text/build_and_run.sh b/tests/mars_text/build_and_run.sh new file mode 100755 index 0000000..c9f354a --- /dev/null +++ b/tests/mars_text/build_and_run.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash +# Self-contained build + test for src/mars/text (no root CMake needed). +# tests/mars_text/build_and_run.sh unit tests (+ real-data tests if SOTS_DATA_DIR is set) +# Env: +# SOTS_DATA_DIR extracted game text tree; unset -> real-data tests SKIP +# SOTS_RE_PARSERS dir with the Python reference readers (default ~/sots-re/verify/parsers); +# missing -> oracle cross-check SKIP +# CXX compiler (default g++) +set -euo pipefail +here="$(cd "$(dirname "$0")" && pwd)" +root="$(cd "$here/../.." && pwd)" +src="$root/src/mars/text" +build="${BUILD_DIR:-$here/build}" +mkdir -p "$build" + +CXX="${CXX:-g++}" +flags=(-std=c++17 -Wall -Wextra -Werror -O1 -I"$root/src") +lib=("$src/value.cpp" "$src/flat_kv.cpp" "$src/manifest.cpp" "$src/csv.cpp") + +echo "== build ($CXX)" +"$CXX" "${flags[@]}" "${lib[@]}" "$here/unit_tests.cpp" -o "$build/unit_tests" +"$CXX" "${flags[@]}" "${lib[@]}" "$here/dump_json.cpp" -o "$build/dump_json" +"$CXX" "${flags[@]}" "${lib[@]}" "$here/realdata_test.cpp" -o "$build/realdata_test" + +echo "== unit tests" +"$build/unit_tests" + +echo "== real-data facts" +"$build/realdata_test" + +echo "== oracle cross-check" +if [ -z "${SOTS_DATA_DIR:-}" ]; then + echo "oracle: SKIP (SOTS_DATA_DIR not set)" + exit 0 +fi +parsers="${SOTS_RE_PARSERS:-$HOME/sots-re/verify/parsers}" +if [ ! -f "$parsers/flat_kv.py" ]; then + echo "oracle: SKIP (reference readers not found at $parsers; set SOTS_RE_PARSERS)" + exit 0 +fi +export SOTS_RE_PARSERS="$parsers" +py="${PYTHON:-/usr/bin/python3}" +out="$build/oracle" +rm -rf "$out"; mkdir -p "$out" +total=0; agree=0; failed=() +while IFS=$'\t' read -r kind rel; do + total=$((total + 1)) + safe="$(echo "$rel" | tr '/ ' '__')" + "$build/dump_json" "$kind" "$SOTS_DATA_DIR/$rel" > "$out/$safe.ours.json" + "$py" "$here/oracle/dump.py" "$kind" "$SOTS_DATA_DIR/$rel" > "$out/$safe.oracle.json" + if "$py" "$here/oracle/compare.py" "$out/$safe.ours.json" "$out/$safe.oracle.json" "$kind $rel"; then + agree=$((agree + 1)) + else + failed+=("$rel") + fi +done < <("$py" "$here/oracle/dump.py" list "$SOTS_DATA_DIR") +echo "oracle: $agree / $total files agree" +if [ "$agree" -ne "$total" ]; then + printf 'disagree: %s\n' "${failed[@]}" + exit 1 +fi diff --git a/tests/mars_text/dump_json.cpp b/tests/mars_text/dump_json.cpp new file mode 100644 index 0000000..a65956a --- /dev/null +++ b/tests/mars_text/dump_json.cpp @@ -0,0 +1,198 @@ +// dump_json +// Emits the parse result as JSON in the same shape as tests/mars_text/oracle/dump.py +// so the two can be compared structurally. Bytes >= 0x80 are written as +// \u00XX (one escape per byte) on both sides. +#include +#include +#include +#include +#include + +#include "mars/text/csv.h" +#include "mars/text/flat_kv.h" +#include "mars/text/manifest.h" + +using namespace mars::text; + +namespace { + +void json_string(std::string_view s, std::string& out) { + out.push_back('"'); + for (unsigned char c : s) { + switch (c) { + case '"': out += "\\\""; break; + case '\\': out += "\\\\"; break; + case '\n': out += "\\n"; break; + case '\r': out += "\\r"; break; + case '\t': out += "\\t"; break; + default: + if (c < 0x20 || c >= 0x7f) { + char buf[8]; + std::snprintf(buf, sizeof buf, "\\u%04x", c); + out += buf; + } else { + out.push_back(static_cast(c)); + } + } + } + out.push_back('"'); +} + +void json_token(const Token& t, std::string& out) { + switch (t.kind()) { + case ValueKind::Int: out += std::to_string(*t.as_int()); break; + case ValueKind::Float: { + char buf[64]; + std::snprintf(buf, sizeof buf, "%.17g", *t.as_float()); + std::string s = buf; + if (s.find_first_of(".eEn") == std::string::npos) s += ".0"; // keep it a JSON float + out += s; + break; + } + case ValueKind::Bool: out += *t.as_bool() ? "true" : "false"; break; + case ValueKind::Bareword: + case ValueKind::String: json_string(t.text, out); break; + } +} + +void json_tokens(const std::vector& toks, std::string& out) { + out.push_back('['); + for (std::size_t i = 0; i < toks.size(); ++i) { + if (i) out.push_back(','); + json_token(toks[i], out); + } + out.push_back(']'); +} + +// kv value: null / scalar / list, matching the reference reader's shape +void json_kv_value(const std::vector& toks, std::string& out) { + if (toks.empty()) out += "null"; + else if (toks.size() == 1) json_token(toks[0], out); + else json_tokens(toks, out); +} + +std::string dump_kv(const FlatKV& kv) { + std::map last; // exact key, last wins + for (const KvEntry& e : kv.entries()) last[e.key] = &e; + std::string out = "{\"values\":{"; + bool first = true; + for (const auto& [key, e] : last) { + if (!first) out.push_back(','); + first = false; + json_string(key, out); + out.push_back(':'); + json_kv_value(e->values, out); + } + out += "},\"duplicates\":{"; + first = true; + for (const auto& [key, lines] : kv.duplicates()) { + if (!first) out.push_back(','); + first = false; + json_string(key, out); + out += ":["; + for (std::size_t i = 0; i < lines.size(); ++i) out += (i ? "," : "") + std::to_string(lines[i]); + out.push_back(']'); + } + out += "}}"; + return out; +} + +std::string dump_rows(const Rows& rows) { + std::string out = "{\"rows\":["; + for (std::size_t i = 0; i < rows.size(); ++i) { + if (i) out.push_back(','); + json_tokens(rows[i].tokens, out); + } + out += "]}"; + return out; +} + +const char* problem_kind(Problem::Kind k) { + switch (k) { + case Problem::Kind::Unrecognised: return "unrecognised"; + case Problem::Kind::DuplicateId: return "duplicate_id"; + case Problem::Kind::DeletedAndAssigned: return "deleted_and_assigned"; + case Problem::Kind::UnbalancedQuote: return "unbalanced_quote"; + case Problem::Kind::UnterminatedQuotedCell: return "unterminated_quoted_cell"; + } + return "?"; +} + +std::string dump_manifest(const Result& r) { + std::string out = "{\"entries\":["; + bool first = true; + for (const ManifestEntry& e : r.value.entries()) { + if (!first) out.push_back(','); + first = false; + out += "[" + std::to_string(e.id) + ","; + json_string(e.name, out); + out.push_back(']'); + } + out += "],\"deleted\":["; + const auto& del = r.value.deleted(); + for (std::size_t i = 0; i < del.size(); ++i) out += (i ? "," : "") + std::to_string(del[i]); + out += "],\"problems\":["; + first = true; + for (const Problem& p : r.problems) { + if (!first) out.push_back(','); + first = false; + out += "{\"line\":" + (p.line ? std::to_string(p.line) : std::string("null")) + ",\"kind\":\"" + problem_kind(p.kind) + + "\",\"id\":" + (p.id >= 0 ? std::to_string(p.id) : std::string("null")) + "}"; + } + out += "]}"; + return out; +} + +std::string dump_csv(const Csv& csv) { + std::string out = "{\"header\":"; + if (csv.header) { + out.push_back('['); + for (std::size_t i = 0; i < csv.header->size(); ++i) { + if (i) out.push_back(','); + json_string((*csv.header)[i], out); + } + out.push_back(']'); + } else { + out += "null"; + } + out += ",\"rows\":["; + for (std::size_t i = 0; i < csv.rows.size(); ++i) { + if (i) out.push_back(','); + out.push_back('['); + for (std::size_t j = 0; j < csv.rows[i].cells.size(); ++j) { + if (j) out.push_back(','); + json_string(csv.rows[i].cells[j], out); + } + out.push_back(']'); + } + out += "]}"; + return out; +} + +} // namespace + +int main(int argc, char** argv) { + if (argc != 3) { + std::fprintf(stderr, "usage: dump_json \n"); + return 2; + } + std::ifstream in(argv[2], std::ios::binary); + if (!in) { + std::fprintf(stderr, "cannot read %s\n", argv[2]); + return 2; + } + std::string text((std::istreambuf_iterator(in)), std::istreambuf_iterator()); + std::string kind = argv[1]; + std::string out; + if (kind == "kv") out = dump_kv(parse_flat_kv(text).value); + else if (kind == "rows") out = dump_rows(parse_rows(text).value); + else if (kind == "manifest") out = dump_manifest(parse_manifest(text)); + else if (kind == "csv") out = dump_csv(parse_csv(text).value); + else { + std::fprintf(stderr, "unknown kind %s\n", argv[1]); + return 2; + } + std::fwrite(out.data(), 1, out.size(), stdout); + std::fputc('\n', stdout); + return 0; +} diff --git a/tests/mars_text/oracle/compare.py b/tests/mars_text/oracle/compare.py new file mode 100755 index 0000000..efcd1cb --- /dev/null +++ b/tests/mars_text/oracle/compare.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python3 +"""compare.py [label] + +Structural, type-aware comparison of two dump files (bool != int != float, +str != list). Prints OK or the first few differences; exit 1 on any +difference. +""" +import json +import sys + +MAX_DIFFS = 8 + + +def walk(a, b, path, diffs): + if len(diffs) >= MAX_DIFFS: + return + if type(a) is not type(b): + diffs.append(f"{path}: type {type(a).__name__} != {type(b).__name__} ({a!r} vs {b!r})") + return + if isinstance(a, dict): + for k in sorted(set(a) | set(b)): + if k not in a: + diffs.append(f"{path}.{k}: missing in ours") + elif k not in b: + diffs.append(f"{path}.{k}: extra in ours") + else: + walk(a[k], b[k], f"{path}.{k}", diffs) + elif isinstance(a, list): + if len(a) != len(b): + diffs.append(f"{path}: length {len(a)} != {len(b)}") + for i, (x, y) in enumerate(zip(a, b)): + walk(x, y, f"{path}[{i}]", diffs) + elif a != b: + diffs.append(f"{path}: {a!r} != {b!r}") + + +def main(argv): + ours = json.load(open(argv[1], encoding="ascii")) + oracle = json.load(open(argv[2], encoding="ascii")) + label = argv[3] if len(argv) > 3 else argv[1] + diffs = [] + walk(ours, oracle, "$", diffs) + if diffs: + print(f"DIFF {label}") + for d in diffs: + print(" ", d) + return 1 + print(f"OK {label}") + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/tests/mars_text/oracle/dump.py b/tests/mars_text/oracle/dump.py new file mode 100755 index 0000000..4c2c158 --- /dev/null +++ b/tests/mars_text/oracle/dump.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""Oracle dump for the mars_text cross-check. + + dump.py list -> "\t" per file handled by mars_text + dump.py -> JSON of the reference reader's result + +The reference readers live in the RE repo (sots-re/verify/parsers); point +SOTS_RE_PARSERS at that directory (default ~/sots-re/verify/parsers). +Strings are re-encoded to cp1252 bytes and emitted one \\u00XX escape per +byte so the JSON matches what dump_json.cpp writes for raw bytes. +""" +import json +import os +import re +import sys + +sys.path.insert(0, os.environ.get("SOTS_RE_PARSERS", os.path.expanduser("~/sots-re/verify/parsers"))) +import flat_kv # noqa: E402 +import manifest # noqa: E402 + +# Same classification as the RE repo's verify.py: everything textual that is +# not brace-block, effect, shader or prose. +BRACE_TXT = {"Data/tutorial.txt", "Data/credits.txt", "Data/Strategy/systemnames.txt", + "Data/Combat/ctechvars.txt", "Data/Combat/shipai.txt", + "Models/Skysphere/skydefs.txt", "Models/Skysphere/NodeSpace-skydefs.txt"} +ROWS_TXT = {"Weapons/_turrets.txt", "Weapons/_defaultweapons.txt", "Data/Combat/damfx.txt", + "Data/Combat/damfx_levels.txt", "Data/Strategy/playercolors.txt", + "Badges/BadgeTable.txt", "Avatars/AvatarTable.txt", "GUI/WeaponIconPlacements.txt"} +PROSE = {"Locale/EN/ChatTrans.txt"} + + +def kind_of(rel): + ext = rel.rsplit(".", 1)[-1].lower() + base = os.path.basename(rel) + if ext == "csv": + return "csv" + if ext == "txt": + if base in ("_weapons.txt", "_shipsections.txt"): + return "manifest" + if rel.startswith("Scenarios/") or rel in BRACE_TXT: + return None + if rel in ROWS_TXT: + return "rows" + if rel.startswith("Locale/EN/Desc") or rel in PROSE: + return None + return "kv" + return None + + +def list_files(root): + out = [] + for dirpath, _, names in os.walk(root): + for n in names: + rel = os.path.relpath(os.path.join(dirpath, n), root).replace(os.sep, "/") + k = kind_of(rel) + if k: + out.append((k, rel)) + return sorted(out) + + +def bytes_str(s): + """str decoded from cp1252 -> str whose code points are the original bytes.""" + return s.encode("cp1252").decode("latin-1") + + +def norm(v): + if isinstance(v, str): + return bytes_str(v) + if isinstance(v, list): + return [norm(x) for x in v] + if isinstance(v, tuple): + return [norm(x) for x in v] + return v + + +_PROBLEM_RES = [ + (re.compile(r"^line (\d+): unrecognised "), "unrecognised"), + (re.compile(r"^line (\d+): duplicate id (\d+) "), "duplicate_id"), + (re.compile(r"^id (\d+) is both DELETED and assigned"), "deleted_and_assigned"), +] + + +def problem_obj(msg): + for rx, kind in _PROBLEM_RES: + m = rx.match(msg) + if m: + if kind == "unrecognised": + return {"line": int(m.group(1)), "kind": kind, "id": None} + if kind == "duplicate_id": + return {"line": int(m.group(1)), "kind": kind, "id": int(m.group(2))} + return {"line": None, "kind": kind, "id": int(m.group(1))} + return {"line": None, "kind": "unknown:" + msg, "id": None} + + +def dump(kind, path): + text = manifest.read_text(path) + if kind == "kv": + d = flat_kv.parse_kv(text) + return {"values": {bytes_str(k): norm(v) for k, v in d.items()}, + "duplicates": {bytes_str(k): v for k, v in flat_kv.duplicates(text).items()}} + if kind == "rows": + return {"rows": norm(flat_kv.parse_rows(text))} + if kind == "manifest": + m = manifest.parse_manifest(text) + return {"entries": [[i, bytes_str(n)] for i, n in m.entries], + "deleted": list(m.deleted), + "problems": [problem_obj(p) for p in m.problems]} + if kind == "csv": + header, rows = manifest.parse_csv_with_header(text) + return {"header": norm(header) if header is not None else None, "rows": norm(rows)} + raise SystemExit(f"unknown kind {kind}") + + +def main(argv): + if len(argv) != 3: + raise SystemExit(__doc__) + if argv[1] == "list": + for k, rel in list_files(argv[2]): + print(f"{k}\t{rel}") + return + json.dump(dump(argv[1], argv[2]), sys.stdout, ensure_ascii=True) + print() + + +if __name__ == "__main__": + main(sys.argv) diff --git a/tests/mars_text/realdata_test.cpp b/tests/mars_text/realdata_test.cpp new file mode 100644 index 0000000..aad1f4a --- /dev/null +++ b/tests/mars_text/realdata_test.cpp @@ -0,0 +1,147 @@ +// Real-data facts test: reads the owner's extracted game text tree from +// $SOTS_DATA_DIR and checks a handful of known facts (counts and values +// established by the RE repo's verification). Skips cleanly when unset. +#include +#include +#include +#include +#include +#include + +#include "mars/text/csv.h" +#include "mars/text/flat_kv.h" +#include "mars/text/manifest.h" + +using namespace mars::text; + +static int g_failures = 0; +static int g_checks = 0; + +#define CHECK(cond) \ + do { \ + ++g_checks; \ + if (!(cond)) { \ + ++g_failures; \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + } \ + } while (0) + +static std::string read_file(const std::string& path) { + std::ifstream in(path, std::ios::binary); + if (!in) { + std::fprintf(stderr, "FAIL cannot read %s\n", path.c_str()); + ++g_failures; + return {}; + } + return std::string((std::istreambuf_iterator(in)), std::istreambuf_iterator()); +} + +int main() { + const char* root = std::getenv("SOTS_DATA_DIR"); + if (!root || !*root) { + std::printf("realdata_test: SKIP (SOTS_DATA_DIR not set)\n"); + return 0; + } + const std::string base = std::string(root) + "/"; + + // globals.txt: a colour and the comment-in-quotes quirk line + { + auto r = parse_flat_kv(read_file(base + "Data/globals.txt")); + CHECK(r.ok()); + CHECK(r.value.size() == 364); + CHECK(r.value.duplicates().empty()); + auto c = r.value.get_color("mars_default_color"); // case-insensitive + CHECK(c && c->r == 255 && c->g == 177 && c->b == 39); + CHECK(r.value.find("MARS_DEFAULT_COLOR")->key == "MARS_DEFAULT_COLOR"); + auto endgame = r.value.get_color("ENDGAME_FILL_COLOR"); + CHECK(endgame && endgame->components == 3); + } + // StrategyVars.txt: typed access + { + auto r = parse_flat_kv(read_file(base + "Data/Strategy/StrategyVars.txt")); + CHECK(r.ok()); + CHECK(r.value.size() == 97); + CHECK(r.value.get_float("SLAVES_DEATH_RATE").has_value()); + } + // _turrets.txt: 42 rows, 8 columns, quoted model in the last column + { + auto r = parse_rows(read_file(base + "Weapons/_turrets.txt")); + CHECK(r.ok()); + CHECK(r.value.size() == 42); + for (const Row& row : r.value) { + CHECK(row.tokens.size() == 8); + CHECK(row.tokens.back().quoted); + CHECK(row.tokens[3].as_int().has_value()); + CHECK(row.tokens[5].as_float().has_value()); + } + } + // _weapons.txt: 123 ids, tombstones 36/58/59, no problems + { + auto r = parse_manifest(read_file(base + "Weapons/_weapons.txt")); + CHECK(r.ok()); + CHECK(r.value.entries().size() == 123); + CHECK((r.value.deleted() == std::vector{36, 58, 59})); + CHECK(r.value.id_of("CAN_AM.WEAPON").has_value()); + CHECK(r.value.name_of(1).has_value()); + } + // Human _shipsections.txt: 145 ids; the misspelt-case DEWar entry resolves both ways + { + auto r = parse_manifest(read_file(base + "Species/Human/sections/_shipsections.txt")); + CHECK(r.ok()); + CHECK(r.value.entries().size() == 145); + CHECK(r.value.id_of("dewar.shipsection") == 98); + } + // Strings.csv: 5,722 raw records / 5,200 data rows, 4-column schema row, + // exactly one multi-line cell (its line break sits just before the closing + // quote, so cell stripping removes it), cp1252 bytes untouched + { + std::string text = read_file(base + "Locale/EN/Strings.csv"); + auto raw = split_csv_records(text); + CHECK(raw.ok()); + CHECK(raw.value.size() == 5722); + int multiline = 0; + for (const CsvRow& row : raw.value) + for (const std::string& c : row.cells) + if (c.find("\r\n") != std::string::npos) ++multiline; + CHECK(multiline == 1); + + auto r = parse_csv(text); + CHECK(r.ok()); + CHECK(r.value.size() == 5200); + CHECK(r.value.header && r.value.header->size() == 4 && (*r.value.header)[1] == "String"); + int highbytes = 0, embedded_newline = 0; + for (const CsvRow& row : r.value.rows) + for (const std::string& c : row.cells) { + if (c.find('\n') != std::string::npos) ++embedded_newline; + for (unsigned char ch : c) + if (ch >= 0x80) { ++highbytes; break; } + } + CHECK(embedded_newline == 0); + CHECK(highbytes > 0); + // the 4 keys duplicated via a trailing space collapse after stripping + std::size_t distinct = 0; + { + std::vector keys; + for (const CsvRow& row : r.value.rows) keys.push_back(row.cells.at(0)); + std::sort(keys.begin(), keys.end()); + distinct = static_cast(std::unique(keys.begin(), keys.end()) - keys.begin()); + } + CHECK(distinct == 5196); + } + // AI template CSV: comment-only, but the schema row is recovered + { + auto r = parse_csv(read_file(base + "Data/Strategy/AI/aitechpri.csv")); + CHECK(r.ok()); + CHECK(r.value.size() == 0); + CHECK(r.value.header && r.value.header->size() == 7 && (*r.value.header)[0] == "tech"); + } + // stock_diplomacy_messages.csv: every cell quoted, "# species" header, blank ",," rows dropped + { + auto r = parse_csv(read_file(base + "Data/Strategy/AI/stock_diplomacy_messages.csv")); + CHECK(r.value.size() == 755); + CHECK(r.value.header && (*r.value.header)[0] == "species"); + } + + std::printf("realdata_test: %d checks, %d failures\n", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; +} diff --git a/tests/mars_text/unit_tests.cpp b/tests/mars_text/unit_tests.cpp new file mode 100644 index 0000000..749247f --- /dev/null +++ b/tests/mars_text/unit_tests.cpp @@ -0,0 +1,338 @@ +// Unit tests for mars::text -- hand-written minimal samples covering every +// quirk the shipped data exercises (see docs/mars-text.md). No game data. +#include +#include + +#include "mars/text/csv.h" +#include "mars/text/flat_kv.h" +#include "mars/text/manifest.h" +#include "mars/text/value.h" + +using namespace mars::text; + +static int g_failures = 0; +static int g_checks = 0; + +#define CHECK(cond) \ + do { \ + ++g_checks; \ + if (!(cond)) { \ + ++g_failures; \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + } \ + } while (0) + +#define CHECK_EQ(a, b) CHECK((a) == (b)) + +// ---- value --------------------------------------------------------------- + +static void test_classify() { + CHECK(classify("12") == ValueKind::Int); + CHECK(classify("-7") == ValueKind::Int); + CHECK(classify("+3") == ValueKind::Int); + CHECK(classify("0.05") == ValueKind::Float); + CHECK(classify(".5") == ValueKind::Float); + CHECK(classify("-.8") == ValueKind::Float); + CHECK(classify("1.") == ValueKind::Float); + CHECK(classify("7e+8") == ValueKind::Float); + CHECK(classify("1E5") == ValueKind::Float); + CHECK(classify("1e") == ValueKind::Bareword); + CHECK(classify(".") == ValueKind::Bareword); + CHECK(classify("-") == ValueKind::Bareword); + CHECK(classify("0x10") == ValueKind::Bareword); + CHECK(classify("inf") == ValueKind::Bareword); + CHECK(classify("nan") == ValueKind::Bareword); + CHECK(classify("1,5") == ValueKind::Bareword); + CHECK(classify("true") == ValueKind::Bool); + CHECK(classify("FALSE") == ValueKind::Bool); + CHECK(classify("True") == ValueKind::Bool); + CHECK(classify("yes") == ValueKind::Bareword); + CHECK(classify("") == ValueKind::Bareword); + + CHECK_EQ(parse_int("42").value(), 42); + CHECK_EQ(parse_int("-42").value(), -42); + CHECK_EQ(parse_int("+9").value(), 9); + CHECK(!parse_int("4.0")); + CHECK(!parse_int("99999999999999999999")); // does not fit in 64 bits + CHECK_EQ(parse_int("-9223372036854775808").value(), INT64_MIN); + CHECK_EQ(parse_float("0.05").value(), 0.05); + CHECK_EQ(parse_float(".5").value(), 0.5); + CHECK_EQ(parse_float("-.8").value(), -0.8); + CHECK_EQ(parse_float("1.").value(), 1.0); + CHECK_EQ(parse_float("7e+8").value(), 7e8); + CHECK_EQ(parse_float("3").value(), 3.0); // int shape converts too + CHECK(!parse_float("inf")); + CHECK(!parse_float("1e")); + CHECK_EQ(parse_bool("TRUE").value(), true); + CHECK_EQ(parse_bool("false").value(), false); + CHECK(!parse_bool("0")); + + Token quoted{"8", true}; + CHECK(quoted.kind() == ValueKind::String); + CHECK(!quoted.as_int()); + Token bare{"8", false}; + CHECK(bare.kind() == ValueKind::Int); + CHECK_EQ(bare.as_int().value(), 8); + CHECK_EQ(bare.as_float().value(), 8.0); +} + +static void test_color() { + auto c = parse_color("255\t177\t\t39"); // tabs inside the quotes, as in globals.txt + CHECK(c && c->components == 3 && c->r == 255 && c->g == 177 && c->b == 39 && c->a == 1); + auto c4 = parse_color(" 0 0.5 1 0.25 "); + CHECK(c4 && c4->components == 4 && c4->g == 0.5 && c4->a == 0.25); + CHECK(!parse_color("1 2")); + CHECK(!parse_color("1 2 3 4 5")); + CHECK(!parse_color("red green blue")); + CHECK(!parse_color("")); +} + +// ---- flat kv ------------------------------------------------------------- + +static void test_strip_comment_and_tokens() { + CHECK_EQ(strip_comment("A 1 // c"), std::string_view("A 1 ")); + CHECK_EQ(strip_comment("A \"x // y\" // c"), std::string_view("A \"x // y\" ")); + CHECK_EQ(strip_comment("A \"unclosed // not a comment"), std::string_view("A \"unclosed // not a comment")); + CHECK_EQ(strip_comment("A / 1"), std::string_view("A / 1")); + CHECK_EQ(strip_comment("//"), std::string_view("")); + + auto t = split_tokens(" a\t\"b c\" \"\" d\"e f\" g"); + CHECK_EQ(t.size(), 6u); + CHECK(t[0].text == "a" && !t[0].quoted); + CHECK(t[1].text == "b c" && t[1].quoted); + CHECK(t[2].text == "" && t[2].quoted); // empty quoted token survives + CHECK(t[3].text == "d\"e" && !t[3].quoted); // quote mid-bareword is literal + CHECK(t[4].text == "f\"" && !t[4].quoted); + CHECK(t[5].text == "g" && !t[5].quoted); + auto u = split_tokens("\"open x"); + CHECK_EQ(u.size(), 2u); + CHECK(u[0].text == "\"open" && !u[0].quoted); // unpaired quote: bareword + CHECK(u[1].text == "x"); + CHECK(split_tokens(" ").empty()); +} + +static void test_flat_kv() { + const char* text = + "// header comment\r\n" + "\r\n" + "SLAVES_DEATH_RATE 0.05\r\n" + "Max_Ships\t\t12 // trailing comment\n" + "TITLE \"Sword // of the Stars\"\n" + "EMPTY \"\"\n" + "COLOR \"255\t177\t\t39\"\n" + "FLAG true\n" + "NAME_ONLY\n" + "LIST 1 2 three\n" + "path \"Data\\\\x.txt\"\n" + "SLAVES_DEATH_RATE 0.10\n" + "BADQ \"oops\n"; + auto r = parse_flat_kv(text); + const FlatKV& kv = r.value; + CHECK_EQ(kv.size(), 11u); + + // typed access, case-insensitive lookup, original spelling preserved + CHECK_EQ(kv.get_int("max_ships").value(), 12); + CHECK_EQ(kv.find("MAX_SHIPS")->key, std::string("Max_Ships")); + CHECK_EQ(kv.find("MAX_SHIPS")->line, 4); + CHECK(!kv.get_int("TITLE")); + CHECK_EQ(kv.get_string("TITLE").value(), std::string_view("Sword // of the Stars")); + CHECK(kv.find("TITLE")->values[0].quoted); + CHECK_EQ(kv.get_string("EMPTY").value(), std::string_view("")); + auto c = kv.get_color("COLOR"); + CHECK(c && c->r == 255 && c->g == 177 && c->b == 39); + CHECK_EQ(kv.get_bool("flag").value(), true); + CHECK(kv.has("NAME_ONLY")); + CHECK(kv.find("NAME_ONLY")->values.empty()); + CHECK(!kv.get_string("NAME_ONLY")); + CHECK_EQ(kv.find("LIST")->values.size(), 3u); + CHECK(kv.find("LIST")->values[2].kind() == ValueKind::Bareword); + CHECK_EQ(kv.get_int("LIST").value(), 1); // first token + CHECK_EQ(kv.get_string("path").value(), std::string_view("Data\\\\x.txt")); // no escape processing + CHECK(!kv.has("missing")); + CHECK(!kv.get_int("missing")); + + // duplicates: last wins for lookup, both entries kept + CHECK_EQ(kv.get_float("slaves_death_rate").value(), 0.10); + auto dups = kv.duplicates(); + CHECK_EQ(dups.size(), 1u); + CHECK(dups[0].first == "slaves_death_rate"); + CHECK((dups[0].second == std::vector{3, 12})); + + // unbalanced quote: line still parsed, warning reported + CHECK_EQ(r.problems.size(), 1u); + CHECK(r.problems[0].kind == Problem::Kind::UnbalancedQuote && r.problems[0].line == 13); + CHECK_EQ(kv.get_string("BADQ").value(), std::string_view("\"oops")); + + // key case only differs: a duplicate for us + auto r2 = parse_flat_kv("Foo 1\nFOO 2\n"); + CHECK_EQ(r2.value.get_int("foo").value(), 2); + CHECK_EQ(r2.value.duplicates().size(), 1u); + + // empty / comment-only input + CHECK_EQ(parse_flat_kv("").value.size(), 0u); + CHECK_EQ(parse_flat_kv("// nothing\n\n \n").value.size(), 0u); + // line numbers with mixed terminators + auto r3 = parse_flat_kv("A 1\rB 2\r\nC 3\nD 4"); + CHECK_EQ(r3.value.find("D")->line, 4); + // high bytes pass through untouched + auto r4 = parse_flat_kv("K \"caf\xe9 \x93q\x94\""); + CHECK_EQ(r4.value.get_string("K").value(), std::string_view("caf\xe9 \x93q\x94")); +} + +static void test_rows() { + const char* text = + "// size weapon-size class health \"model\"\r\n" + "\r\n" + "small\ttiny\tstandard 20 360 1.00 \"turret_pd.x\"\r\n" + "large small standard 800 120 1.00 \"turret with space.x\" // c\n" + "\tHiver\t\tBADGE_HIVER_GW100\t\ttrue\n"; + auto r = parse_rows(text); + CHECK(r.ok()); + CHECK_EQ(r.value.size(), 3u); + CHECK_EQ(r.value[0].line, 3); + CHECK_EQ(r.value[0].tokens.size(), 7u); + CHECK_EQ(r.value[0].tokens[3].as_int().value(), 20); + CHECK(r.value[0].tokens[5].kind() == ValueKind::Float); + CHECK(r.value[0].tokens[6].quoted && r.value[0].tokens[6].text == "turret_pd.x"); + CHECK_EQ(r.value[1].tokens[6].text, std::string("turret with space.x")); + CHECK_EQ(r.value[2].tokens.size(), 3u); + CHECK_EQ(r.value[2].tokens[2].as_bool().value(), true); +} + +// ---- manifest ------------------------------------------------------------ + +static void test_manifest() { + const char* text = + "// Only add to this list: always use new IDs.\r\n" + "\r\n" + "1 can_am.weapon\r\n" + "2\tDEWar.SHIPSECTION\r\n" + "// DELETED - 36\r\n" + "//deleted-58\n" + " // Deleted - 59 \n" + "3 x.weapon // trailing comment is not accepted\n" + "4 two words\n" + "not_an_id y.weapon\n" + "1 dup.weapon\n" + "7 z.weapon\n" + "// DELETED - 7\n"; + auto r = parse_manifest(text); + const Manifest& m = r.value; + CHECK_EQ(m.entries().size(), 4u); + CHECK((m.deleted() == std::vector{36, 58, 59, 7})); + CHECK_EQ(m.entries()[1].line, 4); + CHECK_EQ(m.name_of(2).value(), std::string_view("DEWar.SHIPSECTION")); + CHECK_EQ(m.id_of("dewar.shipsection").value(), 2); // case-folded lookup + CHECK_EQ(m.id_of("CAN_AM.WEAPON").value(), 1); + CHECK(!m.id_of("nope")); + CHECK(!m.name_of(99)); + CHECK(m.is_deleted(36) && !m.is_deleted(1)); + CHECK_EQ(m.name_of(1).value(), std::string_view("dup.weapon")); // last assignment wins + + CHECK_EQ(r.problems.size(), 5u); + CHECK(r.problems[0].kind == Problem::Kind::Unrecognised && r.problems[0].line == 8); + CHECK(r.problems[1].kind == Problem::Kind::Unrecognised && r.problems[1].line == 9); + CHECK(r.problems[2].kind == Problem::Kind::Unrecognised && r.problems[2].line == 10); + CHECK(r.problems[3].kind == Problem::Kind::DuplicateId && r.problems[3].line == 11 && r.problems[3].id == 1); + CHECK(r.problems[4].kind == Problem::Kind::DeletedAndAssigned && r.problems[4].id == 7); + + CHECK(parse_manifest("").ok()); + CHECK(parse_manifest("").value.entries().empty()); +} + +// ---- csv ----------------------------------------------------------------- + +static void test_csv_records() { + // raw split: quoting, escapes, terminators, multi-line cells + auto r = split_csv_records("a,b\r\n\r\n\"x,y\",\"say \"\"hi\"\"\"\n\"multi\r\nline\",end\r\nlast,"); + CHECK(r.ok()); + const auto& rows = r.value; + CHECK_EQ(rows.size(), 5u); + CHECK((rows[0].cells == std::vector{"a", "b"})); + CHECK(rows[1].cells.empty()); + CHECK((rows[2].cells == std::vector{"x,y", "say \"hi\""})); + CHECK((rows[3].cells == std::vector{"multi\r\nline", "end"})); + CHECK_EQ(rows[3].line, 4); + CHECK((rows[4].cells == std::vector{"last", ""})); // trailing comma, no final newline + CHECK_EQ(rows[4].line, 6); + + // quote is literal unless it is the first byte of the cell + auto q = split_csv_records("KEY\",text\n \"Effects/x.effect\",1\n\"tail\"junk,2\n"); + CHECK((q.value[0].cells == std::vector{"KEY\"", "text"})); + CHECK((q.value[1].cells == std::vector{" \"Effects/x.effect\"", "1"})); + CHECK((q.value[2].cells == std::vector{"tailjunk", "2"})); // lenient, like the reference + + // lone CR terminators, and CR LF sequences counted once + auto cr = split_csv_records("a\rb\r\nc"); + CHECK_EQ(cr.value.size(), 3u); + CHECK_EQ(cr.value[2].line, 3); + + // unterminated quoted cell: content kept, problem reported + auto u = split_csv_records("a,\"never closed\nmore"); + CHECK_EQ(u.value.size(), 1u); + CHECK((u.value[0].cells == std::vector{"a", "never closed\nmore"})); + CHECK_EQ(u.problems.size(), 1u); + CHECK(u.problems[0].kind == Problem::Kind::UnterminatedQuotedCell); + + CHECK(split_csv_records("").value.empty()); + CHECK_EQ(split_csv_records("\n").value.size(), 1u); +} + +static void test_csv() { + const char* text = + "# ,,\r\n" + "#\r\n" + "# tech: tech label\r\n" + "\r\n" + " \r\n" + ",,\r\n" + "// slash comment, with comma\r\n" + " WEP_Laser , 10, 20 \r\n" + " # indented comment,1,2\r\n" + "WEP_Gauss,\"1,5\",\"a \"\"b\"\"\"\r\n" + "\"multi\r\nline\",x,y\r\n"; + auto r = parse_csv(text); + CHECK(r.ok()); + const Csv& csv = r.value; + CHECK(csv.header.has_value()); + CHECK((csv.header.value() == std::vector{"tech", "human-pri", "hiver-pri"})); + CHECK_EQ(csv.size(), 3u); + CHECK((csv.rows[0].cells == std::vector{"WEP_Laser", "10", "20"})); // stripped + CHECK_EQ(csv.rows[0].line, 8); + CHECK((csv.rows[1].cells == std::vector{"WEP_Gauss", "1,5", "a \"b\""})); + CHECK((csv.rows[2].cells == std::vector{"multi\r\nline", "x", "y"})); + CHECK_EQ(csv.rows[2].line, 11); + + // header variants + auto h1 = parse_csv("\"# species\",\"event\",\"message\"\n,,\n\"Human\",\"E\",\"M\"\n"); + CHECK((h1.value.header.value() == std::vector{"species", "event", "message"})); + CHECK_EQ(h1.value.size(), 1u); + auto h2 = parse_csv("#Key+A955,String,Size,Notes\nK,v,,\n"); + CHECK((h2.value.header.value() == std::vector{"Key+A955", "String", "Size", "Notes"})); + CHECK((h2.value.rows[0].cells == std::vector{"K", "v", "", ""})); + auto h3 = parse_csv("# just a comment\nA,B\n# ,\n"); + CHECK(!h3.value.header.has_value()); // single-cell '#' rows are not schemas; data ends the search + CHECK_EQ(h3.value.size(), 1u); + auto h4 = parse_csv("A\nB\n"); + CHECK(!h4.value.header.has_value()); + CHECK_EQ(h4.value.size(), 2u); + CHECK_EQ(parse_csv("").value.size(), 0u); + CHECK_EQ(parse_csv("#a,b\n#c\n").value.size(), 0u); // comment-only template + + // bytes >= 0x80 pass through + auto hb = parse_csv("K,caf\xe9 \x85\n"); + CHECK_EQ(hb.value.rows[0].cells[1], std::string("caf\xe9 \x85")); +} + +int main() { + test_classify(); + test_color(); + test_strip_comment_and_tokens(); + test_flat_kv(); + test_rows(); + test_manifest(); + test_csv_records(); + test_csv(); + std::printf("mars_text unit tests: %d checks, %d failures\n", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; +}