Compare commits
No commits in common. "4f0a9dbd82da51cb7a78c3995ac9403527e23661" and "750ad59ff70ed45e2ca86fb69266d8817c51473c" have entirely different histories.
4f0a9dbd82
...
750ad59ff7
15 changed files with 487 additions and 963 deletions
|
|
@ -1,107 +1,85 @@
|
||||||
# `mars::parse` — the engine's data-file readers
|
# `mars::parse` — the engine's data-file readers
|
||||||
|
|
||||||
`src/mars/parse/` reimplements the two text readers the original engine uses
|
`src/mars/parse/` reimplements the two text readers the original engine uses
|
||||||
for its catalogs: the **brace-block** key/value format (weapons, ship
|
for its catalogs: the recursive **brace-block** key/value format (weapons, ship
|
||||||
sections, the tech tree, combat missions, badge/light defs, GUI scripts,
|
sections, the tech tree, combat missions, badge/light defs, GUI scripts,
|
||||||
scenario and other block-form `.txt`) and the separate **`.effect`** format.
|
scenario and other block-form `.txt`) and the separate **`.effect`** format.
|
||||||
C++17, no dependencies, no exceptions across the API, bytes passed through
|
C++17, no dependencies, no exceptions across the API, bytes passed through
|
||||||
untouched (the data is cp1252).
|
untouched (the data is cp1252).
|
||||||
|
|
||||||
Behaviour is specified by the original engine's `Mars::Script` tokenizer as
|
Behaviour is specified by the proven Python readers in the RE repo
|
||||||
recovered in the RE repo (`findings/subsystems/loader-prototypes.md` §M3) and
|
(`sots-re/verify/parsers/mars_data.py`, `effect_txt.py`) and the findings in
|
||||||
mirrored by the RE repo's Python readers (`sots-re/verify/parsers/mars_data.py`,
|
`findings/subsystems/data-parsers.md`. The C++ is written from that
|
||||||
`effect_txt.py`), which serve as the test oracle. The C++ is written from that
|
understanding, not transliterated; the Python readers serve as the test oracle.
|
||||||
understanding, not transliterated.
|
|
||||||
|
|
||||||
```
|
```
|
||||||
src/mars/parse/
|
src/mars/parse/
|
||||||
result.h Diagnostic {line, message}; Result<T>
|
result.h Diagnostic {line, message}; Result<T>
|
||||||
value.h/.cpp classify / as_int / as_double / as_bool / iequals
|
value.h/.cpp classify / as_int / as_double / as_bool / iequals
|
||||||
script.h header-only pull tokenizer with the engine's semantics
|
|
||||||
(read_token / next / skip_block); shared with mars::text
|
|
||||||
blocks.h/.cpp brace-block reader: parse_blocks(text, Options) -> Result<Document>
|
blocks.h/.cpp brace-block reader: parse_blocks(text, Options) -> Result<Document>
|
||||||
effect.h/.cpp .effect reader: parse_effect(text) -> Result<EffectFile>
|
effect.h/.cpp .effect reader: parse_effect(text) -> Result<EffectFile>
|
||||||
CMakeLists.txt static library `mars_parse` (include root: src/)
|
CMakeLists.txt static library `mars_parse` (include root: src/)
|
||||||
tests/mars_parse/
|
tests/mars_parse/
|
||||||
build_and_run.sh plain g++ build; unit tests; real-data oracle test
|
build_and_run.sh plain g++ build; unit tests; real-data oracle test
|
||||||
test_*.cpp hand-written samples, one test per rule
|
test_*.cpp hand-written samples, one test per quirk
|
||||||
canon.h/.cpp canonical JSON rendering (matches oracle/dump.py)
|
canon.h/.cpp canonical JSON rendering (matches oracle/dump.py)
|
||||||
oracle_check.cpp parses every real file, diffs against the oracle
|
oracle_check.cpp parses every real file, diffs against the oracle
|
||||||
oracle/dump.py runs the RE-repo readers, emits canonical JSON per file
|
oracle/dump.py runs the RE-repo readers, emits canonical JSON per file
|
||||||
```
|
```
|
||||||
|
|
||||||
## 1. The tokenizer — `Script` (script.h)
|
## 1. Brace-block format — grammar as implemented
|
||||||
|
|
||||||
The original has no tree. Every loader (`WeaponDef::ParseScript`,
|
```
|
||||||
`SectionDef::ParseScript`, `MasterTechTree::ParseTech`, the GUI script
|
body := (block | pair | item)*
|
||||||
callbacks, `GlobalConsts::LoadFile`) drives one pull tokenizer with three
|
block := NAME '{' body '}'
|
||||||
operations and interprets the steps itself:
|
pair := NAME value
|
||||||
|
item := QUOTED -- a quoted string with no key
|
||||||
|
| NAME -- a bareword directly before '}' or EOF
|
||||||
|
value := QUOTED | BAREWORD
|
||||||
|
comment := '//' ... end-of-line -- only at a token boundary
|
||||||
|
```
|
||||||
|
|
||||||
| operation | what it does |
|
Tokens (whitespace-driven, not line-based):
|
||||||
|
|
||||||
|
| token | rule |
|
||||||
|---|---|
|
|---|---|
|
||||||
| `read_token(Raw&)` | one token. `Ok` = read, input remains; `AtEnd` = read, the token touches the end of input; `NoInput` = nothing to read |
|
| whitespace | space, tab, CR, LF, VT, FF |
|
||||||
| `next(ScriptToken&)` | reads a key; if it is `}` the step is `Close`. Otherwise reads a value; `Open` if the value is `{`, else `Pair`. Returns the status of the **value** read |
|
| `{` `}` | single-character tokens; they also terminate a bareword (`weapon{` is two tokens) |
|
||||||
| `skip_block(depth)` | consumes tokens counting whole `{`/`}` tokens until depth 0 |
|
| QUOTED | `"` … `"`; no escapes at all — backslashes and `//` inside are literal; may span lines; an empty `""` is a valid value |
|
||||||
|
| BAREWORD | maximal run of anything that is not whitespace, `{`, `}` or `"` — so `abc"def"` is bareword `abc` followed by a quoted token, and `3//x` is one bareword |
|
||||||
|
| comment | `//` at the start of a token runs to end of line |
|
||||||
|
|
||||||
Token rules:
|
|
||||||
|
|
||||||
| rule | detail |
|
|
||||||
|---|---|
|
|
||||||
| whitespace | space, tab, CR, LF — nothing else (`\v`, `\f`, NBSP are token bytes) |
|
|
||||||
| bareword | a run of non-whitespace. **Braces are not delimiters**: `{` and `}` count only as whole tokens, so `weapon{` is one word and `abc"def"` is one word |
|
|
||||||
| quoted | a token starting with `"`, `'` or a backtick ends at the next occurrence of that same character; the quotes are stripped, no escape processing, whitespace and newlines inside are kept; an unterminated quote runs to end of input. Text glued after a closing quote starts the next token |
|
|
||||||
| comment | a token whose extracted text starts with `//` — bare or quoted — is a comment: the rest of the line is skipped and reading continues. `3//x` is one word; `"0 0 0"// junk` is the value `0 0 0` then a comment |
|
|
||||||
| length | text is capped at 1023 bytes (the whole token is still consumed) |
|
|
||||||
| brace test | exact compare on the stripped text, so a quoted `"{"` / `"}"` acts as a brace |
|
|
||||||
|
|
||||||
`Script` is public so a per-consumer loader can be written against the same
|
|
||||||
step stream when a catalog needs the original's own key handling.
|
|
||||||
|
|
||||||
## 2. Brace-block trees — `parse_blocks`
|
|
||||||
|
|
||||||
`parse_blocks` runs the loaders' `while (next() == Ok)` loop and records the
|
|
||||||
steps: `Pair` → `Entry::Kind::Pair` (the key may be quoted — `key_quoted`),
|
|
||||||
`Open` → `Entry::Kind::Block` with a recursive body, `Close` → the block ends.
|
|
||||||
The file's top level is itself a body: catalog files hold one or more named
|
The file's top level is itself a body: catalog files hold one or more named
|
||||||
blocks, scenario `.txt` files mix top-level pairs and `player{}` blocks.
|
blocks, scenario `.txt` files mix top-level pairs and `player{}` blocks.
|
||||||
|
|
||||||
There is no "item" concept: a quoted string in key position is a key
|
Disambiguation of a bareword `N`: look at the next token. `{` → a block named
|
||||||
(`systemnames.txt` lists become pairs of consecutive names), and a lone word
|
`N`; `}` or EOF → an item; anything else → a pair `N value` (the value may be
|
||||||
before `}` takes the `}` as its value. That is what the original sees.
|
bare or quoted). A quoted token in key position is always an item.
|
||||||
|
|
||||||
### End of input and other tolerances
|
### Leniency (what the shipped data needs)
|
||||||
|
|
||||||
Everything the original merely tolerates is recorded as a `Diagnostic` in
|
| condition | default (`Options::strict = false`) | `strict = true` |
|
||||||
`Document::warnings`; with `Options::strict` each one is an error instead.
|
|---|---|---|
|
||||||
|
| EOF while blocks are open | close every open block, one `Diagnostic` per open block in `Document::warnings` | error at the EOF line |
|
||||||
|
| `}` at top level | ignored, warning, parsing continues | error |
|
||||||
|
| unterminated `"` | error | error |
|
||||||
|
| `{` with no name before it | error | error |
|
||||||
|
|
||||||
| condition | what the original does (and this reader) |
|
The two recoveries are exactly what 12 shipped shipsections require (11 never
|
||||||
|---|---|
|
close their outer block, `Human/CrPropaganda.shipsection` has one `}` too many).
|
||||||
| input ends inside a block | the block simply ends (one warning per open block) |
|
They load in the game, so the engine performs the same repairs. Nothing in the
|
||||||
| `}` at top level | ignored |
|
shipped data hits the two hard errors.
|
||||||
| final `KEY value` whose value token touches the end of input (no trailing newline) | the step returns `AtEnd` and the pair is **dropped** |
|
|
||||||
| final KEY with no value; final `NAME {` touching the end | dropped |
|
|
||||||
| unterminated quote | runs to end of input (and, touching it, that pair is dropped) |
|
|
||||||
| `{` in key position | an ordinary key; kept |
|
|
||||||
| `}` that closes a block and touches the end of input | a plain close — no warning; the original stops there either way and the outcome is identical |
|
|
||||||
|
|
||||||
The shipped data exercises: 12 shipsections (11 unclosed outer blocks, Human
|
|
||||||
`CrPropaganda` with one `}` too many) and `Data/Strategy/systemnames.txt`,
|
|
||||||
whose `hiver` and `liir` lists have an odd number of names — the last name
|
|
||||||
pairs with the `}`, so `tarkas`/`liir` nest inside `hiver` and `morrigi`
|
|
||||||
inside `liir`, and two blocks are left open at the end. 27 catalog files end
|
|
||||||
with a `}` as their last byte (plain close). No brace file drops a pair.
|
|
||||||
|
|
||||||
### Node model
|
### Node model
|
||||||
|
|
||||||
```cpp
|
```cpp
|
||||||
struct Entry { // one step in a block, in file order
|
struct Entry { // one thing in a block, in file order
|
||||||
enum class Kind { Pair, Block } kind;
|
enum class Kind { Pair, Item, Block } kind;
|
||||||
std::string key; // as written (case preserved, quotes stripped)
|
std::string key; // Pair/Block: as written (case preserved)
|
||||||
std::string value; // Pair: value text
|
std::string value; // Pair: value text; Item: the item text
|
||||||
bool quoted; // value was quoted
|
bool quoted; // value/item was in double quotes
|
||||||
bool key_quoted; // key was quoted
|
|
||||||
std::unique_ptr<Node> block; // Block only
|
std::unique_ptr<Node> block; // Block only
|
||||||
int line; // 1-based, counts '\n'
|
int line; // 1-based
|
||||||
};
|
};
|
||||||
struct Node { // a named block; the root has name ""
|
struct Node { // a named block; the root has name ""
|
||||||
std::string name;
|
std::string name;
|
||||||
|
|
@ -110,7 +88,7 @@ struct Node { // a named block; the root has name ""
|
||||||
const Entry* first(key); std::vector<const Entry*> all(key); // pairs + blocks
|
const Entry* first(key); std::vector<const Entry*> all(key); // pairs + blocks
|
||||||
const std::string* first_value(key); std::vector<std::string_view> values(key);
|
const std::string* first_value(key); std::vector<std::string_view> values(key);
|
||||||
const Node* first_block(key); std::vector<const Node*> blocks(key);
|
const Node* first_block(key); std::vector<const Node*> blocks(key);
|
||||||
bool has(key);
|
std::vector<std::string_view> items(); bool has(key);
|
||||||
};
|
};
|
||||||
struct Document { Node root; std::vector<Diagnostic> warnings; };
|
struct Document { Node root; std::vector<Diagnostic> warnings; };
|
||||||
Result<Document> parse_blocks(std::string_view, Options = {});
|
Result<Document> parse_blocks(std::string_view, Options = {});
|
||||||
|
|
@ -118,9 +96,10 @@ Result<Document> parse_blocks(std::string_view, Options = {});
|
||||||
|
|
||||||
Design points, each tied to a data fact:
|
Design points, each tied to a data fact:
|
||||||
|
|
||||||
- **Keys are matched case-insensitively** (`first`/`all`/… use `iequals`, the
|
- **Keys are matched case-insensitively** (`first`/`all`/… use `iequals`), the
|
||||||
original uses `_stricmp`), the original spelling is kept. The data mixes
|
original spelling is kept. The data mixes `Requires`/`requires`,
|
||||||
`Requires`/`requires`, `badge`/`Badge`.
|
`badge`/`Badge`; identifiers in values are also case-mixed but that is the
|
||||||
|
consumer's concern (`iequals` is public for it).
|
||||||
- **Repeated keys are not merged**: `all("requires")` returns them in file
|
- **Repeated keys are not merged**: `all("requires")` returns them in file
|
||||||
order. 970 `requires`, 3,721 `bank`, and the 19 banks that repeat
|
order. 970 `requires`, 3,721 `bank`, and the 19 banks that repeat
|
||||||
`turretsize` all rely on order being kept.
|
`turretsize` all rely on order being kept.
|
||||||
|
|
@ -130,14 +109,16 @@ Design points, each tied to a data fact:
|
||||||
- **Values are raw text**. `50`, `.5`, `7e+8`, `TRUE`, `@WEAPON_X` and `"8"`
|
- **Values are raw text**. `50`, `.5`, `7e+8`, `TRUE`, `@WEAPON_X` and `"8"`
|
||||||
all stay as written, with `quoted` recording the difference. Conversion is
|
all stay as written, with `quoted` recording the difference. Conversion is
|
||||||
explicit via `value.h`: `classify()` reports Int / Float / Bool / Text using
|
explicit via `value.h`: `classify()` reports Int / Float / Bool / Text using
|
||||||
the C-float shapes the data uses; `as_int`/`as_double`/`as_bool` convert.
|
the C-float shapes the data uses (`[+-]?d+`, `[+-]?(d+.d*|.d+|d+)([eE][+-]?d+)?`,
|
||||||
(The original uses `atoi`/`atof` per key, so quoting never changes typing
|
`true`/`false` any case); `as_int`/`as_double`/`as_bool` convert. A quoted
|
||||||
there; the oracle's canonical form types only barewords, and consumers
|
string is never a number to the oracle, and consumers should honour `quoted`
|
||||||
wanting engine typing should convert regardless of `quoted`.)
|
the same way.
|
||||||
|
- **Items**: bare quoted strings (`systemnames.txt` lists) and a lone trailing
|
||||||
|
bareword are `Kind::Item`, reachable through `items()`.
|
||||||
|
|
||||||
## 3. `.effect` format — grammar as implemented
|
## 2. `.effect` format — grammar as implemented
|
||||||
|
|
||||||
Line-based, not brace-block (a different engine class, `Mars::TextFileStream`).
|
Line-based, not brace-block.
|
||||||
|
|
||||||
```
|
```
|
||||||
line 1: TXT magic (surrounding whitespace ignored)
|
line 1: TXT magic (surrounding whitespace ignored)
|
||||||
|
|
@ -175,41 +156,37 @@ struct EffectFile { EffectGroup root; };
|
||||||
Result<EffectFile> parse_effect(std::string_view);
|
Result<EffectFile> parse_effect(std::string_view);
|
||||||
```
|
```
|
||||||
|
|
||||||
## 4. Tests
|
## 3. Tests
|
||||||
|
|
||||||
`tests/mars_parse/build_and_run.sh` (plain `g++ -std=c++17 -Wall -Wextra
|
`tests/mars_parse/build_and_run.sh` (plain `g++ -std=c++17 -Wall -Wextra
|
||||||
-Wpedantic -Werror`):
|
-Wpedantic -Werror`):
|
||||||
|
|
||||||
1. **Unit tests** — 51 cases with hand-written samples: the `Script` status
|
1. **Unit tests** — 39 cases with hand-written samples: nesting, same-line
|
||||||
codes, `next()` step shapes, `skip_block`, the 1023-byte cap; nesting,
|
blocks and glued braces, quoted values with spaces/backslashes/`//`, empty
|
||||||
same-line blocks, braces glued to words, the three quote characters,
|
`""`, comments at token boundaries, repeated keys in order, case-insensitive
|
||||||
quotes inside words, quoted braces, the four-character whitespace set,
|
lookup, quoted items and trailing barewords, `option` scalar+block, `@TOKEN`
|
||||||
comments bare / glued / quoted / at end of input, repeated keys in order,
|
verbatim, EOF-closes-blocks and stray-`}` in lenient and strict mode,
|
||||||
case-insensitive lookup, quoted keys and odd-count lists, lone word before
|
unterminated string and nameless `{` errors, CRLF and line numbers,
|
||||||
`}`, `option` scalar+block, `@TOKEN` verbatim, EOF-ends-blocks, stray `}`,
|
multi-line strings, high bytes, and every effect error path.
|
||||||
`}` as last byte, dropped final pair / key / block header, unterminated
|
|
||||||
quote, `{` as key — each in lenient and strict mode where it applies —
|
|
||||||
CRLF and line numbers, multi-line strings, high bytes, and every effect
|
|
||||||
error path.
|
|
||||||
2. **Real-data oracle test** — runs when `SOTS_DATA_DIR` points at an
|
2. **Real-data oracle test** — runs when `SOTS_DATA_DIR` points at an
|
||||||
extracted `sots.gob` tree. `oracle/dump.py` runs the RE-repo readers
|
extracted `sots.gob` tree. `oracle/dump.py` runs the RE-repo readers
|
||||||
(`SOTS_RE_PARSERS`, default `~/sots-re/verify/parsers`; exit 3 → the C++
|
(`SOTS_RE_PARSERS`, default `~/sots-re/verify/parsers`; exit 3 → the C++
|
||||||
side runs count-only) and writes a canonical JSON per file; `oracle_check`
|
side runs count-only) and writes a canonical JSON per file; `oracle_check`
|
||||||
parses every file with `mars_parse`, renders the same canonical form
|
parses every file with `mars_parse`, renders the same canonical form
|
||||||
(`canon.cpp`) and requires byte equality, plus equal warning counts.
|
(`canon.cpp`) and requires byte equality, plus equal lenient-warning counts.
|
||||||
|
|
||||||
The canonical form is the oracle's own dict shape (keys lower-cased and
|
The canonical form is the oracle's own dict shape (keys lower-cased and
|
||||||
sorted, repeats → lists, bare pair values typed), with floats rendered by both
|
sorted, repeats → lists, items under `_items`, bare pair values typed), with
|
||||||
sides as `"\x01" + printf("%.17g")` so formatting is shared, and cp1252 bytes
|
floats rendered by both sides as `"\x01" + printf("%.17g")` so formatting is
|
||||||
rendered one `\u00xx` per byte. A one-character change in an oracle file is
|
shared, and cp1252 bytes rendered one `\u00xx` per byte. A one-character
|
||||||
detected (verified).
|
change in an oracle file is detected (verified).
|
||||||
|
|
||||||
Skips cleanly with `SOTS_DATA_DIR` unset. Nothing from the game is committed;
|
Skips cleanly with `SOTS_DATA_DIR` unset. Nothing from the game is committed;
|
||||||
`tests/mars_parse/build/` (which holds the oracle JSON) is git-ignored.
|
`tests/mars_parse/build/` (which holds the oracle JSON) is git-ignored.
|
||||||
|
|
||||||
### Oracle results (2026-09-07, engine-parity rules, owner's `gob-extract`)
|
### Oracle results (2026-09-07, owner's `gob-extract`)
|
||||||
|
|
||||||
| kind | files | parsed | oracle match | tolerances taken |
|
| kind | files | parsed | oracle match | need lenient recovery |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|
|
||||||
| shipsection | 875 | 875 | 875 | 12 |
|
| shipsection | 875 | 875 | 875 | 12 |
|
||||||
| weapon | 207 | 207 | 207 | 0 |
|
| weapon | 207 | 207 | 207 | 0 |
|
||||||
|
|
@ -218,25 +195,28 @@ Skips cleanly with `SOTS_DATA_DIR` unset. Nothing from the game is committed;
|
||||||
| combat | 3 | 3 | 3 | 0 |
|
| combat | 3 | 3 | 3 | 0 |
|
||||||
| script | 4 | 4 | 4 | 0 |
|
| script | 4 | 4 | 4 | 0 |
|
||||||
| def | 2 | 2 | 2 | 0 |
|
| def | 2 | 2 | 2 | 0 |
|
||||||
| block-form txt (Scenarios + 7 Data/Models files) | 24 | 24 | 24 | 1 (`systemnames.txt`) |
|
| block-form txt (Scenarios + 7 Data/Models files) | 24 | 24 | 24 | 0 |
|
||||||
| **total** | **1,531** | **1,531** | **1,531 (100%)** | 13 |
|
| **total** | **1,531** | **1,531** | **1,531 (100%)** | 12 |
|
||||||
|
|
||||||
Against the previous (tree-grammar) reading, exactly one file's canonical
|
The 12 strict-mode rejects are exactly the list in `data-parsers.md`
|
||||||
output changed: `Data/Strategy/systemnames.txt` (quoted names become pairs;
|
(Hiver 6, Liir 4, Morrigi 1 unclosed; Human `CrPropaganda` stray `}`), and the
|
||||||
odd-count lists nest the following blocks). Every weapon, shipsection, tech,
|
per-file warning counts equal the oracle's. The remaining 64 of the Python
|
||||||
combat, def, script and scenario file renders byte-identically, so nothing the
|
suite's 1,595 files are CSV / flat key-value / manifest / positional tables —
|
||||||
cross-links or catalogs consume moved. The remaining 64 of the Python suite's
|
different readers, outside this module.
|
||||||
1,595 files are CSV / flat key-value / manifest / positional tables — see
|
|
||||||
`docs/mars-text.md`.
|
|
||||||
|
|
||||||
## 5. Open questions
|
## 4. Open questions
|
||||||
|
|
||||||
- **`systemnames.txt` consumer.** The generic step stream pairs its names up;
|
- **Engine leniency beyond the two observed repairs.** Unterminated strings,
|
||||||
the original's system-name loader "handles that format itself" (RE notes)
|
a nameless `{`, and malformed `.effect` nesting are errors here because no
|
||||||
and its exact handling of the odd-count lists is not recovered. A dedicated
|
shipped file exercises them; the original may be more forgiving. Worth a
|
||||||
reader over `Script` will be needed for name generation parity.
|
shim-side probe before the loader is wired to user mods.
|
||||||
- **`skip_block` granularity.** Assumed to count raw `{`/`}` tokens (so a
|
- **`//` glued to a bareword** (`3//x`) is one token, matching the oracle.
|
||||||
`}` in value position inside a skipped block still closes it, unlike
|
Whether the engine's tokenizer cuts a comment mid-word is unverified; the
|
||||||
`next()`). No shipped file distinguishes the two.
|
data never depends on it.
|
||||||
- **Malformed `.effect` nesting** stays an error here because no shipped file
|
- **Whitespace set.** The C++ uses the C `isspace` set; the Python oracle's
|
||||||
exercises it; the original may be more forgiving.
|
`\s` also covers U+00A0 and a few control characters. The data contains
|
||||||
|
neither (checked: all 1,531 files are pure ASCII), so this cannot be
|
||||||
|
observed from data.
|
||||||
|
- **Typed values.** The oracle types barewords (`50` → int) at parse time;
|
||||||
|
this module keeps text and offers converters. If a consumer needs
|
||||||
|
oracle-identical typing it should use `classify()` and honour `quoted`.
|
||||||
|
|
|
||||||
|
|
@ -7,12 +7,9 @@ CSVs. Library target: `mars_text` (static, `src/mars/text/CMakeLists.txt`).
|
||||||
Include as `mars/text/<header>.h`. C++17, no dependencies, no exceptions cross
|
Include as `mars/text/<header>.h`. C++17, no dependencies, no exceptions cross
|
||||||
the API; every reader is total and reports oddities as `Problem`s.
|
the API; every reader is total and reports oddities as `Problem`s.
|
||||||
|
|
||||||
The behaviour spec for the `KEY value` tables is the original engine's
|
The behaviour spec is the RE repo's proven Python readers
|
||||||
`GlobalConsts` loader as recovered in the RE repo
|
(`sots-re/verify/parsers/flat_kv.py`, `manifest.py`), which parse all 1,595
|
||||||
(`findings/subsystems/loader-prototypes.md`, §M1 with the tokenizer of §M3);
|
shipped data files. This module is checked against them file by file (below).
|
||||||
the RE repo's Python readers (`sots-re/verify/parsers/flat_kv.py`,
|
|
||||||
`manifest.py`) implement the same rules and serve as the oracle this module
|
|
||||||
is checked against file by file (below).
|
|
||||||
|
|
||||||
## API
|
## API
|
||||||
|
|
||||||
|
|
@ -32,12 +29,12 @@ struct Color { double r, g, b, a; int components; };
|
||||||
std::optional<Color> parse_color(std::string_view "r g b [a]");
|
std::optional<Color> parse_color(std::string_view "r g b [a]");
|
||||||
|
|
||||||
// flat_kv.h
|
// flat_kv.h
|
||||||
struct KvEntry { int line; std::string key /*original spelling*/; Token value; };
|
struct KvEntry { int line; std::string key /*original spelling*/; std::vector<Token> values; };
|
||||||
class FlatKV { entries(); find(key) /*case-insensitive, FIRST wins*/; has(key);
|
class FlatKV { entries(); find(key) /*case-insensitive, last wins*/; has(key);
|
||||||
get_int/get_float/get_bool/get_string/get_color(key); duplicates(); };
|
get_int/get_float/get_bool/get_string/get_color(key); duplicates(); };
|
||||||
struct Row { int line; std::vector<Token> tokens; }; using Rows = std::vector<Row>;
|
struct Row { int line; std::vector<Token> tokens; }; using Rows = std::vector<Row>;
|
||||||
Result<FlatKV> parse_flat_kv(std::string_view); // engine loader loop (mars/parse/script.h)
|
Result<FlatKV> parse_flat_kv(std::string_view);
|
||||||
Result<Rows> parse_rows(std::string_view); // line-based
|
Result<Rows> parse_rows(std::string_view);
|
||||||
std::string_view strip_comment(std::string_view line); std::vector<Token> split_tokens(std::string_view line);
|
std::string_view strip_comment(std::string_view line); std::vector<Token> split_tokens(std::string_view line);
|
||||||
|
|
||||||
// manifest.h
|
// manifest.h
|
||||||
|
|
@ -55,74 +52,48 @@ Result<std::vector<CsvRow>> split_csv_records(std::string_view); // raw RFC-418
|
||||||
|
|
||||||
Input is the file's raw bytes (`std::string_view`). The files are cp1252 with
|
Input is the file's raw bytes (`std::string_view`). The files are cp1252 with
|
||||||
mixed CRLF/LF; bytes ≥ 0x80 are passed through untouched and are never
|
mixed CRLF/LF; bytes ≥ 0x80 are passed through untouched and are never
|
||||||
whitespace or case-folded.
|
whitespace or case-folded. Whitespace is ASCII only (space, `\t`, `\r`, `\n`,
|
||||||
|
`\v`, `\f`).
|
||||||
|
|
||||||
## Behaviour
|
## Behaviour
|
||||||
|
|
||||||
### Flat `KEY value` tables (`parse_flat_kv`)
|
### Flat `KEY value` tables (`parse_flat_kv`) and row tables (`parse_rows`)
|
||||||
|
|
||||||
This is the engine's `GlobalConsts::LoadFile` loop, verbatim in shape. The
|
Per line: strip a trailing `//` comment (quote-aware: `//` inside `"..."` is
|
||||||
file is **not** read by lines: it is stepped through with the same pull
|
text, and an unclosed `"` disables comment stripping for the rest of the
|
||||||
tokenizer the brace-block catalogs use (`mars/parse/script.h`,
|
line), trim, skip if empty, then tokenise:
|
||||||
`Script::next()`), and each step is one of
|
|
||||||
|
|
||||||
| step | what the loader does |
|
- `"..."` is one token (may be empty, may contain spaces/tabs/`//`); no escape
|
||||||
|---|---|
|
processing, so `"Data\\x.txt"` keeps both backslashes.
|
||||||
| `KEY value` | one token each. `value` is consumed by the key's registered parser; a colour must therefore be quoted (`"48 29 2"`), and `LIST 1 2 three` is the two pairs `LIST=1`, `2=three` |
|
- Otherwise a run of non-whitespace is one bareword; a `"` that has no
|
||||||
| `NAME {` | the block is skipped to its matching `}` (`Problem::Kind::SkippedBlock`) |
|
closing quote, or sits mid-word, is an ordinary character.
|
||||||
| `}` | ignored |
|
|
||||||
|
|
||||||
Tokenizer rules (script.h): whitespace is space/tab/CR/LF only; a bareword
|
`parse_flat_kv` takes the first token as the key and the remaining tokens as
|
||||||
runs to whitespace, so braces glued to a word are part of it; `"`, `'` and
|
the value: none (key-only line), one (scalar) or several (a list — the
|
||||||
backtick open a quoted token that ends at the same character, no escapes,
|
reference reader returns a Python list here; none of the 20 shipped tables
|
||||||
quotes stripped, an unterminated quote runs to end of input
|
actually do this). `parse_rows` returns every line's token list.
|
||||||
(`UnterminatedQuote`); a token whose text starts with `//` is a comment to
|
|
||||||
end of line (so `3//x` is one word, `"0 0 0"// junk` is the value `0 0 0`
|
|
||||||
followed by a comment); tokens are capped at 1023 bytes.
|
|
||||||
|
|
||||||
Keys are matched **case-insensitively** and the **first occurrence wins**:
|
|
||||||
the loader erases a key from its expected-set once consumed, so a later
|
|
||||||
duplicate is logged "multiply defined" and ignored. `find()` returns the first
|
|
||||||
entry; every pair stays in `entries()` in file order; the later line is
|
|
||||||
reported as `DuplicateKey`; `duplicates()` lists repeated (case-folded) keys
|
|
||||||
with their line numbers. Unknown keys are the caller's business (the engine
|
|
||||||
ignores them); a registered key absent from the file keeps its default.
|
|
||||||
|
|
||||||
**End of input**: the loader stops at the first step that does not complete.
|
|
||||||
A final `KEY value` whose value token touches the end of the file — no
|
|
||||||
trailing newline or space — is therefore **dropped** (`DroppedTrailingPair`),
|
|
||||||
as is a final key with no value. Two shipped files lose a key this way:
|
|
||||||
`Data/Strategy/StrategyVars.txt` (`CIVILIAN_BURDEN_RATIO 0.5`) and
|
|
||||||
`Data/encounters.txt` (`HERALD_SPEECH_MAX_INVERVAL 45`) — the game runs on
|
|
||||||
those keys' compiled-in defaults.
|
|
||||||
|
|
||||||
Typing is lazy: `Token::kind()` classifies a bareword exactly like the
|
Typing is lazy: `Token::kind()` classifies a bareword exactly like the
|
||||||
reference `coerce()`: `[+-]?\d+` → Int, `[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?`
|
reference `coerce()`: `[+-]?\d+` → Int, `[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?`
|
||||||
→ Float (`.5`, `-.8`, `1.`, `7e+8`), `true`/`false` any case → Bool, else
|
→ Float (`.5`, `-.8`, `1.`, `7e+8`), `true`/`false` any case → Bool, else
|
||||||
Bareword (`0x10`, `inf`, `nan`, `1,5` are barewords). A quoted token is always
|
Bareword (`0x10`, `inf`, `nan`, `1,5` are barewords). A quoted token is always
|
||||||
a String — `"8"` never becomes a number. `as_float()` also accepts Int shapes;
|
a String — `"8"` never becomes a number. `as_float()` also accepts Int shapes;
|
||||||
`as_int()` refuses values that do not fit in 64 bits. (The engine types by the
|
`as_int()` refuses values that do not fit in 64 bits.
|
||||||
key's registered parser — `sscanf` `%d`/`%f` — not by the text; see the RE
|
|
||||||
notes. `Token` keeps the text so a consumer can apply the registered parser.)
|
Keys: lookup is ASCII case-insensitive (`find("mars_default_color")` finds
|
||||||
|
`MARS_DEFAULT_COLOR`); the entry keeps the original spelling and line. When a
|
||||||
|
key repeats, `find()` returns the **last** line — the same "later line wins"
|
||||||
|
default as the reference `parse_kv(on_dup='last')`. All entries stay in
|
||||||
|
`entries()`; `duplicates()` lists repeated (case-folded) keys with their line
|
||||||
|
numbers. The shipped tables contain no duplicates, exact or case-folded, so
|
||||||
|
the case-folding in `duplicates()` is a deliberate superset of the reference's
|
||||||
|
exact-key check.
|
||||||
|
|
||||||
Colours are quoted `"r g b"` / `"r g b a"` strings (tabs inside are fine, as in
|
Colours are quoted `"r g b"` / `"r g b a"` strings (tabs inside are fine, as in
|
||||||
`globals.txt`); `get_color` / `parse_color` require 3 or 4 numeric components.
|
`globals.txt`); `get_color` / `parse_color` require 3 or 4 numeric components.
|
||||||
|
|
||||||
`KvEntry::line` counts `\n` only (the engine has no notion of lines; the
|
Problems: `UnbalancedQuote` (odd number of `"` on a line) is reported but the
|
||||||
number is for diagnostics).
|
line is still parsed. None occur in shipped data.
|
||||||
|
|
||||||
### Row tables (`parse_rows`)
|
|
||||||
|
|
||||||
Line-based (these files are read by other engine code whose reader has not
|
|
||||||
been recovered; the behaviour is the reference reader's). Per line: strip a
|
|
||||||
trailing `//` comment (quote-aware: `//` inside `"..."` is text, and an
|
|
||||||
unclosed `"` disables comment stripping for the rest of the line), trim, skip
|
|
||||||
if empty, then tokenise: `"..."` is one token (may be empty, may contain
|
|
||||||
spaces/tabs/`//`; no escapes), otherwise a run of non-whitespace is one
|
|
||||||
bareword; a `"` that has no closing quote, or sits mid-word, is an ordinary
|
|
||||||
character. Whitespace here is the C set (space, `\t`, `\r`, `\n`, `\v`, `\f`).
|
|
||||||
`UnbalancedQuote` (odd number of `"` on a line) is reported but the line is
|
|
||||||
still parsed. None occur in shipped data.
|
|
||||||
|
|
||||||
### Id manifests (`parse_manifest`)
|
### Id manifests (`parse_manifest`)
|
||||||
|
|
||||||
|
|
@ -164,14 +135,13 @@ end the search.
|
||||||
## Oracle results
|
## Oracle results
|
||||||
|
|
||||||
`tests/mars_text/build_and_run.sh` compiles with `g++ -std=c++17 -Wall -Wextra
|
`tests/mars_text/build_and_run.sh` compiles with `g++ -std=c++17 -Wall -Wextra
|
||||||
-Werror`, runs the unit tests (201 checks, hand-written samples for every
|
-Werror`, runs the unit tests (165 checks, hand-written samples for every
|
||||||
rule above, including a dedicated engine-step suite for the kv loader), the
|
quirk above), the real-data facts test, and — when `SOTS_DATA_DIR` is set —
|
||||||
real-data facts test, and — when `SOTS_DATA_DIR` is set — dumps every file
|
dumps every file `mars_text` owns through both `dump_json` (C++) and
|
||||||
`mars_text` owns through both `dump_json` (C++) and `oracle/dump.py` (the
|
`oracle/dump.py` (the Python readers) and compares them structurally,
|
||||||
Python readers) and compares them structurally, type-aware (`compare.py`).
|
type-aware (`compare.py`).
|
||||||
|
|
||||||
Run 2026-09-07 (engine-parity rules) against the extracted `sots.gob` +
|
Run 2026-09-07 against the extracted `sots.gob` + `sots_local_en.gob` tree:
|
||||||
`sots_local_en.gob` tree:
|
|
||||||
|
|
||||||
| kind | files | agree |
|
| kind | files | agree |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
|
|
@ -181,33 +151,23 @@ Run 2026-09-07 (engine-parity rules) against the extracted `sots.gob` +
|
||||||
| csv (AI tables, scenarios, `RealSpace.csv`, `SpriteTable.csv`, `Strings.csv`, `SpeechEvents.csv`, asteroids, music, sound_ui) | 28 | 28 |
|
| csv (AI tables, scenarios, `RealSpace.csv`, `SpriteTable.csv`, `Strings.csv`, `SpeechEvents.csv`, asteroids, music, sound_ui) | 28 | 28 |
|
||||||
| **total** | **64** | **64 (100%)** |
|
| **total** | **64** | **64 (100%)** |
|
||||||
|
|
||||||
Compared per file: the full key → value map (case-insensitive keys under
|
Compared per file: the full key → value map (exact keys, last wins, typed
|
||||||
their first spelling, first wins, typed values), duplicate-key report, every
|
values), duplicate-key report, every row's typed tokens, manifest entries /
|
||||||
row's typed tokens, manifest entries / tombstones / problems (kind, line, id),
|
tombstones / problems (kind, line, id), CSV header and every stripped cell.
|
||||||
CSV header and every stripped cell. Real-data facts also asserted:
|
Real-data facts also asserted: `globals.txt` 364 keys, `MARS_DEFAULT_COLOR` =
|
||||||
`globals.txt` 364 keys, no duplicates, `MARS_DEFAULT_COLOR` = (255,177,39);
|
(255,177,39); `_turrets.txt` 42 × 8 tokens; `_weapons.txt` 123 ids, deleted
|
||||||
`StrategyVars.txt` 96 keys with `CIVILIAN_BURDEN_RATIO` dropped and
|
36/58/59; Human `_shipsections.txt` 145 ids with `dewar.shipsection` → 98;
|
||||||
`encounters.txt` with `HERALD_SPEECH_MAX_INVERVAL` dropped (one
|
`Strings.csv` 5,722 raw records → 5,200 data rows → 5,196 distinct keys,
|
||||||
`DroppedTrailingPair` each); `_turrets.txt` 42 × 8 tokens; `_weapons.txt` 123
|
exactly one multi-line raw cell; `aitechpri.csv` 0 rows with a 7-name schema.
|
||||||
ids, deleted 36/58/59; Human `_shipsections.txt` 145 ids with
|
|
||||||
`dewar.shipsection` → 98; `Strings.csv` 5,722 raw records → 5,200 data rows →
|
|
||||||
5,196 distinct keys, exactly one multi-line raw cell; `aitechpri.csv` 0 rows
|
|
||||||
with a 7-name schema.
|
|
||||||
|
|
||||||
What changed when the kv reader moved from the earlier line-based, last-wins
|
|
||||||
reading to the engine's rules: no shipped kv file contains a duplicate key
|
|
||||||
(exact or case-folded), so first-wins flips no value; the only observable
|
|
||||||
difference is the two dropped trailing pairs above.
|
|
||||||
|
|
||||||
## Deliberate divergences from the reference
|
## Deliberate divergences from the reference
|
||||||
|
|
||||||
None affect shipped data (verified above); listed so nobody chases them:
|
None affect shipped data (verified above); listed so nobody chases them:
|
||||||
|
|
||||||
- Row/manifest/CSV whitespace is ASCII-only. The Python readers operate on
|
- Whitespace is ASCII-only. The Python readers operate on decoded `str`, so
|
||||||
decoded `str`, so they would also treat cp1252 `0xA0` (NBSP) and bytes
|
they would also treat cp1252 `0xA0` (NBSP) and bytes `0x1C`–`0x1F` as
|
||||||
`0x1C`–`0x1F` as whitespace / line breaks. The relevant files contain no
|
whitespace / line breaks. The relevant files contain no such bytes.
|
||||||
such bytes. (The kv reader and the Python `Script` both use the engine's
|
- `duplicates()` folds key case; the reference compares exact keys.
|
||||||
exact four-character set.)
|
|
||||||
- Ints are 64-bit; the reference has unbounded ints. An overflowing bareword
|
- Ints are 64-bit; the reference has unbounded ints. An overflowing bareword
|
||||||
classifies as `Int` but `as_int()` returns nullopt.
|
classifies as `Int` but `as_int()` returns nullopt.
|
||||||
- The reference `csv` module rejects a NUL byte; this reader treats it as data.
|
- The reference `csv` module rejects a NUL byte; this reader treats it as data.
|
||||||
|
|
@ -216,8 +176,8 @@ None affect shipped data (verified above); listed so nobody chases them:
|
||||||
|
|
||||||
## Open questions
|
## Open questions
|
||||||
|
|
||||||
- The `SkipBlock` used for `NAME {` inside a kv file is assumed to count raw
|
- The engine's own tolerance is unknown for the cases where the reference is
|
||||||
`{`/`}` tokens (`Script::skip_block`); no shipped kv file contains a block,
|
strict (manifest entry with a trailing comment, a key that repeats). We copy
|
||||||
so this cannot be observed from data.
|
the reference; both are absent from shipped data.
|
||||||
- The engine's own tolerance is unknown for a manifest entry with a trailing
|
- Multi-token `KEY a b c` values never occur in shipped tables; the API keeps
|
||||||
comment; we copy the reference (absent from shipped data).
|
them as a token list rather than guessing a type.
|
||||||
|
|
|
||||||
|
|
@ -93,7 +93,9 @@ Block snapshot(const mars::parse::Node& node) {
|
||||||
b.attrs.push_back(std::move(a));
|
b.attrs.push_back(std::move(a));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
// (bare quoted items are now Pair entries with key_quoted; engine-parity change)
|
case Entry::Kind::Item:
|
||||||
|
b.items.push_back(e.value);
|
||||||
|
break;
|
||||||
case Entry::Kind::Block: {
|
case Entry::Kind::Block: {
|
||||||
Block sub = snapshot(*e.block);
|
Block sub = snapshot(*e.block);
|
||||||
sub.name = e.key;
|
sub.name = e.key;
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,6 @@
|
||||||
|
|
||||||
#include <utility>
|
#include <utility>
|
||||||
|
|
||||||
#include "script.h"
|
|
||||||
#include "value.h"
|
#include "value.h"
|
||||||
|
|
||||||
namespace mars::parse {
|
namespace mars::parse {
|
||||||
|
|
@ -18,14 +17,14 @@ Entry::~Entry() = default;
|
||||||
|
|
||||||
const Entry* Node::first(std::string_view key) const {
|
const Entry* Node::first(std::string_view key) const {
|
||||||
for (const Entry& e : entries)
|
for (const Entry& e : entries)
|
||||||
if (iequals(e.key, key)) return &e;
|
if (!e.is_item() && iequals(e.key, key)) return &e;
|
||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<const Entry*> Node::all(std::string_view key) const {
|
std::vector<const Entry*> Node::all(std::string_view key) const {
|
||||||
std::vector<const Entry*> out;
|
std::vector<const Entry*> out;
|
||||||
for (const Entry& e : entries)
|
for (const Entry& e : entries)
|
||||||
if (iequals(e.key, key)) out.push_back(&e);
|
if (!e.is_item() && iequals(e.key, key)) out.push_back(&e);
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -55,26 +54,108 @@ std::vector<const Node*> Node::blocks(std::string_view key) const {
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- parser -----------------------------------------------------------------
|
std::vector<std::string_view> Node::items() const {
|
||||||
|
std::vector<std::string_view> out;
|
||||||
|
for (const Entry& e : entries)
|
||||||
|
if (e.is_item()) out.push_back(e.value);
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- tokenizer --------------------------------------------------------------
|
||||||
|
|
||||||
namespace {
|
namespace {
|
||||||
|
|
||||||
|
enum class Tok { Open, Close, Quoted, Bare, End };
|
||||||
|
|
||||||
|
struct Token {
|
||||||
|
Tok kind = Tok::End;
|
||||||
|
std::string_view text; // Quoted: contents without the quotes
|
||||||
|
int line = 0;
|
||||||
|
};
|
||||||
|
|
||||||
|
bool is_space(unsigned char c) {
|
||||||
|
return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\v' || c == '\f';
|
||||||
|
}
|
||||||
|
|
||||||
|
class Lexer {
|
||||||
|
public:
|
||||||
|
explicit Lexer(std::string_view s) : s_(s) {}
|
||||||
|
|
||||||
|
// Returns false (with `err` set) only for an unterminated string.
|
||||||
|
bool next(Token& out, Diagnostic& err) {
|
||||||
|
skip_trivia();
|
||||||
|
out.line = line_;
|
||||||
|
if (pos_ >= s_.size()) {
|
||||||
|
out.kind = Tok::End;
|
||||||
|
out.text = {};
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
char c = s_[pos_];
|
||||||
|
if (c == '{') { ++pos_; out.kind = Tok::Open; out.text = s_.substr(pos_ - 1, 1); return true; }
|
||||||
|
if (c == '}') { ++pos_; out.kind = Tok::Close; out.text = s_.substr(pos_ - 1, 1); return true; }
|
||||||
|
if (c == '"') {
|
||||||
|
std::size_t close = s_.find('"', pos_ + 1);
|
||||||
|
if (close == std::string_view::npos) {
|
||||||
|
err = {line_, "unterminated string"};
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
out.kind = Tok::Quoted;
|
||||||
|
out.text = s_.substr(pos_ + 1, close - pos_ - 1);
|
||||||
|
for (char q : out.text) if (q == '\n') ++line_; // strings may span lines
|
||||||
|
pos_ = close + 1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
std::size_t start = pos_;
|
||||||
|
while (pos_ < s_.size()) {
|
||||||
|
unsigned char u = static_cast<unsigned char>(s_[pos_]);
|
||||||
|
if (is_space(u) || u == '{' || u == '}' || u == '"') break;
|
||||||
|
++pos_;
|
||||||
|
}
|
||||||
|
out.kind = Tok::Bare;
|
||||||
|
out.text = s_.substr(start, pos_ - start);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
void skip_trivia() {
|
||||||
|
while (pos_ < s_.size()) {
|
||||||
|
unsigned char c = static_cast<unsigned char>(s_[pos_]);
|
||||||
|
if (is_space(c)) {
|
||||||
|
if (c == '\n') ++line_;
|
||||||
|
++pos_;
|
||||||
|
} else if (c == '/' && pos_ + 1 < s_.size() && s_[pos_ + 1] == '/') {
|
||||||
|
while (pos_ < s_.size() && s_[pos_] != '\n') ++pos_;
|
||||||
|
} else {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string_view s_;
|
||||||
|
std::size_t pos_ = 0;
|
||||||
|
int line_ = 1;
|
||||||
|
};
|
||||||
|
|
||||||
|
// ---- parser -----------------------------------------------------------------
|
||||||
|
|
||||||
class Parser {
|
class Parser {
|
||||||
public:
|
public:
|
||||||
Parser(std::string_view text, Options opt) : script_(text), opt_(opt) {}
|
Parser(std::string_view text, Options opt) : lex_(text), opt_(opt) {}
|
||||||
|
|
||||||
Result<Document> run() {
|
Result<Document> run() {
|
||||||
Document doc;
|
Document doc;
|
||||||
if (body(doc.root, 0) == Step::Error) return err_;
|
if (!advance()) return err_;
|
||||||
|
if (!body(doc.root, 0)) return err_;
|
||||||
doc.warnings = std::move(warnings_);
|
doc.warnings = std::move(warnings_);
|
||||||
return doc;
|
return doc;
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
enum class Step { Closed, Eof, Error };
|
// Fills `cur_` with the next token. False on lexer error.
|
||||||
|
bool advance() { return lex_.next(cur_, err_); }
|
||||||
|
|
||||||
// Every tolerance the loaders exhibit is recorded here; strict rejects it.
|
// Lenient recovery point: warn, or fail under strict.
|
||||||
bool tolerate(int line, std::string message) {
|
bool recover(int line, std::string message) {
|
||||||
if (opt_.strict) {
|
if (opt_.strict) {
|
||||||
err_ = {line, std::move(message)};
|
err_ = {line, std::move(message)};
|
||||||
return false;
|
return false;
|
||||||
|
|
@ -83,101 +164,88 @@ private:
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
static std::string quote(std::string_view s) { return "'" + std::string(s) + "'"; }
|
bool fail(int line, std::string message) {
|
||||||
|
err_ = {line, std::move(message)};
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
// Runs the loaders' step loop for one block body. Closed: a `}` ended it.
|
// Parses entries into `node` until the matching '}' (depth > 0) or EOF.
|
||||||
// Eof: the input ran out (the caller, if inside a block, reports that).
|
// On return `cur_` is the token after the block.
|
||||||
Step body(Node& node, int depth) {
|
bool body(Node& node, int depth) {
|
||||||
for (;;) {
|
for (;;) {
|
||||||
ScriptToken t;
|
switch (cur_.kind) {
|
||||||
const ReadStatus rc = script_.next(t);
|
case Tok::End:
|
||||||
|
if (depth > 0 &&
|
||||||
|
!recover(cur_.line, "end of file inside block '" + node.name +
|
||||||
|
"' (depth " + std::to_string(depth) + ")"))
|
||||||
|
return false;
|
||||||
|
return true; // EOF closes every open block
|
||||||
|
|
||||||
if (t.key_unterminated && !tolerate(t.key_line, "unterminated quote runs to end of input"))
|
case Tok::Close:
|
||||||
return Step::Error;
|
|
||||||
if (t.value_unterminated && !tolerate(t.value_line, "unterminated quote runs to end of input"))
|
|
||||||
return Step::Error;
|
|
||||||
|
|
||||||
if (rc != ReadStatus::Ok) {
|
|
||||||
// The original returns here with the step incomplete; whatever
|
|
||||||
// was read is discarded. Report what that means for this file.
|
|
||||||
if (t.key_status == ReadStatus::AtEnd && t.key == "}") {
|
|
||||||
// A closing brace as the very last byte: the block ends
|
|
||||||
// exactly as it would have with a newline after it.
|
|
||||||
if (depth == 0) {
|
if (depth == 0) {
|
||||||
if (!tolerate(t.key_line, "stray '}' at top level")) return Step::Error;
|
if (!recover(cur_.line, "stray '}' at top level")) return false;
|
||||||
return Step::Eof;
|
if (!advance()) return false;
|
||||||
}
|
|
||||||
return Step::Closed;
|
|
||||||
}
|
|
||||||
if (t.key_status == ReadStatus::AtEnd) {
|
|
||||||
if (!tolerate(t.key_line, "trailing key " + quote(t.key) +
|
|
||||||
" touches end of input: dropped"))
|
|
||||||
return Step::Error;
|
|
||||||
} else if (t.key_status == ReadStatus::Ok) {
|
|
||||||
std::string what;
|
|
||||||
if (t.value_status == ReadStatus::NoInput)
|
|
||||||
what = "trailing key " + quote(t.key) + " has no value: dropped";
|
|
||||||
else if (t.type == ScriptToken::Type::Open)
|
|
||||||
what = "block header " + quote(t.key) + " { touches end of input: dropped";
|
|
||||||
else
|
|
||||||
what = "final pair " + quote(t.key) + " " + quote(t.value) +
|
|
||||||
" touches end of input (no trailing newline): dropped";
|
|
||||||
if (!tolerate(t.key_line, what)) return Step::Error;
|
|
||||||
}
|
|
||||||
return Step::Eof;
|
|
||||||
}
|
|
||||||
|
|
||||||
switch (t.type) {
|
|
||||||
case ScriptToken::Type::Close:
|
|
||||||
if (depth == 0) {
|
|
||||||
if (!tolerate(t.key_line, "stray '}' at top level")) return Step::Error;
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
return Step::Closed;
|
return advance();
|
||||||
|
|
||||||
case ScriptToken::Type::Open: {
|
case Tok::Open:
|
||||||
if (t.key == "{" && !tolerate(t.key_line, "'{' used as a block name")) return Step::Error;
|
return fail(cur_.line, "'{' without a block name");
|
||||||
|
|
||||||
|
case Tok::Quoted: {
|
||||||
|
Entry e;
|
||||||
|
e.kind = Entry::Kind::Item;
|
||||||
|
e.value = std::string(cur_.text);
|
||||||
|
e.quoted = true;
|
||||||
|
e.line = cur_.line;
|
||||||
|
node.entries.push_back(std::move(e));
|
||||||
|
if (!advance()) return false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
case Tok::Bare: {
|
||||||
|
Token name = cur_;
|
||||||
|
if (!advance()) return false;
|
||||||
|
if (cur_.kind == Tok::End || cur_.kind == Tok::Close) {
|
||||||
|
// A lone word right before the block closes: a flag item.
|
||||||
|
Entry e;
|
||||||
|
e.kind = Entry::Kind::Item;
|
||||||
|
e.value = std::string(name.text);
|
||||||
|
e.line = name.line;
|
||||||
|
node.entries.push_back(std::move(e));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (cur_.kind == Tok::Open) {
|
||||||
Entry e;
|
Entry e;
|
||||||
e.kind = Entry::Kind::Block;
|
e.kind = Entry::Kind::Block;
|
||||||
e.key = std::string(t.key);
|
e.key = std::string(name.text);
|
||||||
e.key_quoted = t.key_quoted;
|
e.line = name.line;
|
||||||
e.line = t.key_line;
|
|
||||||
e.block = std::make_unique<Node>();
|
e.block = std::make_unique<Node>();
|
||||||
e.block->name = e.key;
|
e.block->name = e.key;
|
||||||
e.block->line = t.value_line;
|
e.block->line = cur_.line;
|
||||||
const Step inner = body(*e.block, depth + 1);
|
if (!advance()) return false;
|
||||||
if (inner == Step::Error) return inner;
|
if (!body(*e.block, depth + 1)) return false;
|
||||||
node.entries.push_back(std::move(e));
|
node.entries.push_back(std::move(e));
|
||||||
if (inner == Step::Eof) {
|
|
||||||
if (!tolerate(script_.line(), "end of input inside block " + quote(t.key) +
|
|
||||||
" (depth " + std::to_string(depth + 1) + ")"))
|
|
||||||
return Step::Error;
|
|
||||||
return Step::Eof;
|
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
// Bare or Quoted: a key/value pair.
|
||||||
case ScriptToken::Type::Pair: {
|
|
||||||
if (t.key == "{" && !tolerate(t.key_line, "'{' used as a key")) return Step::Error;
|
|
||||||
Entry e;
|
Entry e;
|
||||||
e.kind = Entry::Kind::Pair;
|
e.kind = Entry::Kind::Pair;
|
||||||
e.key = std::string(t.key);
|
e.key = std::string(name.text);
|
||||||
e.value = std::string(t.value);
|
e.value = std::string(cur_.text);
|
||||||
e.quoted = t.value_quoted;
|
e.quoted = cur_.kind == Tok::Quoted;
|
||||||
e.key_quoted = t.key_quoted;
|
e.line = name.line;
|
||||||
e.line = t.key_line;
|
|
||||||
node.entries.push_back(std::move(e));
|
node.entries.push_back(std::move(e));
|
||||||
|
if (!advance()) return false;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
case ScriptToken::Type::None:
|
|
||||||
break; // not produced when rc == Ok
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Script script_;
|
Lexer lex_;
|
||||||
Options opt_;
|
Options opt_;
|
||||||
|
Token cur_;
|
||||||
Diagnostic err_;
|
Diagnostic err_;
|
||||||
std::vector<Diagnostic> warnings_;
|
std::vector<Diagnostic> warnings_;
|
||||||
};
|
};
|
||||||
|
|
|
||||||
|
|
@ -4,35 +4,23 @@
|
||||||
// *.shipsection, *.tech, *.combat, *.def, *.script and the block-form *.txt
|
// *.shipsection, *.tech, *.combat, *.def, *.script and the block-form *.txt
|
||||||
// files (scenarios, tutorial, credits, system names, sky defs, ...).
|
// files (scenarios, tutorial, credits, system names, sky defs, ...).
|
||||||
//
|
//
|
||||||
// The original engine has no tree: every loader pulls `Script::next()` steps
|
// body := (block | pair | item)*
|
||||||
// (see script.h) and interprets them itself. This module runs exactly that
|
// block := NAME '{' body '}'
|
||||||
// step loop and records what it yields, so the tree below is the step stream
|
// pair := NAME value
|
||||||
// made addressable, nothing more:
|
// item := QUOTED | NAME (a NAME directly before '}' or EOF)
|
||||||
|
// value := QUOTED | BAREWORD
|
||||||
|
// comment := '//' ... end of line
|
||||||
//
|
//
|
||||||
// step := KEY value -> Entry::Kind::Pair (key may itself be quoted)
|
// The tokenizer is whitespace-driven, not line-based: a block may open on the
|
||||||
// | NAME { -> Entry::Kind::Block (body until the matching `}`)
|
// same line as a preceding pair, a name may sit on the same line as its brace,
|
||||||
// | } -> closes the block
|
// and a quoted string may span lines. There is no escape syntax -- backslashes
|
||||||
|
// and '//' inside quotes are literal. Bytes pass through untouched (cp1252).
|
||||||
//
|
//
|
||||||
// Tokenizer: whitespace-only delimiting (`{`/`}` count only as whole tokens),
|
// Leniency (the shipped data needs both; the engine accepts them):
|
||||||
// `"` `'` backtick quotes without escapes, `//` comments, 1023-byte tokens --
|
// * end of file closes every open block -> Diagnostic in Document::warnings
|
||||||
// the rules are listed in script.h.
|
// * a stray '}' at top level is ignored -> Diagnostic in Document::warnings
|
||||||
//
|
// With Options::strict both become errors. Three things are always errors:
|
||||||
// End-of-input behaviour is what the loaders' `while (next() == Ok)` loops
|
// an unterminated quoted string, a '{' with no name before it, and nothing else.
|
||||||
// produce, and each case is reported in Document::warnings:
|
|
||||||
// * input ends inside a block -> the block simply ends
|
|
||||||
// * a `}` at top level -> ignored
|
|
||||||
// * a final `KEY value` whose value touches the end of input (no trailing
|
|
||||||
// newline) -> that pair is DROPPED
|
|
||||||
// * a final KEY with no value -> dropped
|
|
||||||
// * a final `NAME {` touching the end -> dropped
|
|
||||||
// * an unterminated quote -> runs to the end of input
|
|
||||||
// * `{` in key position -> an ordinary key (kept)
|
|
||||||
// A `}` that closes a block and touches the end of input is a plain close
|
|
||||||
// (the original stops there either way; the outcome is identical).
|
|
||||||
// With Options::strict every warning is an error instead.
|
|
||||||
//
|
|
||||||
// No escape syntax -- backslashes and `//` inside quotes are literal. Bytes
|
|
||||||
// pass through untouched (cp1252).
|
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include <memory>
|
#include <memory>
|
||||||
|
|
@ -46,17 +34,16 @@ namespace mars::parse {
|
||||||
|
|
||||||
struct Node;
|
struct Node;
|
||||||
|
|
||||||
// One step inside a block, in file order.
|
// One thing inside a block, in file order.
|
||||||
struct Entry {
|
struct Entry {
|
||||||
enum class Kind { Pair, Block };
|
enum class Kind { Pair, Item, Block };
|
||||||
|
|
||||||
Kind kind = Kind::Pair;
|
Kind kind = Kind::Pair;
|
||||||
std::string key; // the name as written (original case, quotes stripped)
|
std::string key; // Pair/Block: the name as written (original case). Item: empty.
|
||||||
std::string value; // Pair: the value text. Block: empty.
|
std::string value; // Pair: the value text; Item: the item text. Block: empty.
|
||||||
bool quoted = false; // value was written inside quotes
|
bool quoted = false; // value/item was written inside double quotes
|
||||||
bool key_quoted = false; // key was written inside quotes (systemnames.txt lists)
|
|
||||||
std::unique_ptr<Node> block; // Block only
|
std::unique_ptr<Node> block; // Block only
|
||||||
int line = 0; // 1-based line of the key
|
int line = 0; // 1-based line of the key (or of the item)
|
||||||
|
|
||||||
Entry();
|
Entry();
|
||||||
Entry(Entry&&) noexcept;
|
Entry(Entry&&) noexcept;
|
||||||
|
|
@ -66,6 +53,7 @@ struct Entry {
|
||||||
Entry& operator=(const Entry&) = delete;
|
Entry& operator=(const Entry&) = delete;
|
||||||
|
|
||||||
bool is_pair() const { return kind == Kind::Pair; }
|
bool is_pair() const { return kind == Kind::Pair; }
|
||||||
|
bool is_item() const { return kind == Kind::Item; }
|
||||||
bool is_block() const { return kind == Kind::Block; }
|
bool is_block() const { return kind == Kind::Block; }
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
@ -85,16 +73,17 @@ struct Node {
|
||||||
std::vector<std::string_view> values(std::string_view key) const;
|
std::vector<std::string_view> values(std::string_view key) const;
|
||||||
const Node* first_block(std::string_view key) const;
|
const Node* first_block(std::string_view key) const;
|
||||||
std::vector<const Node*> blocks(std::string_view key) const;
|
std::vector<const Node*> blocks(std::string_view key) const;
|
||||||
|
std::vector<std::string_view> items() const; // bare quoted / trailing barewords
|
||||||
bool has(std::string_view key) const { return first(key) != nullptr; }
|
bool has(std::string_view key) const { return first(key) != nullptr; }
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Document {
|
struct Document {
|
||||||
Node root; // the file's top level is itself a body
|
Node root; // the file's top level is itself a body
|
||||||
std::vector<Diagnostic> warnings; // end-of-input / stray-brace cases listed above
|
std::vector<Diagnostic> warnings; // lenient recoveries taken
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Options {
|
struct Options {
|
||||||
bool strict = false; // reject files the engine would tolerate (every warning -> error)
|
bool strict = false; // reject files the engine would repair
|
||||||
};
|
};
|
||||||
|
|
||||||
Result<Document> parse_blocks(std::string_view text, Options options = {});
|
Result<Document> parse_blocks(std::string_view text, Options options = {});
|
||||||
|
|
|
||||||
|
|
@ -1,182 +0,0 @@
|
||||||
// mars::parse::Script -- the engine's pull tokenizer for every brace-block
|
|
||||||
// and flat KEY/value text file (weapons, ship sections, tech tree, GUI
|
|
||||||
// scripts, scenario .txt, and the GlobalConsts tables under Data/).
|
|
||||||
//
|
|
||||||
// Header-only so both mars_parse (block trees) and mars_text (flat tables)
|
|
||||||
// share one implementation of the rules without linking each other.
|
|
||||||
//
|
|
||||||
// The original reads a file token by token; there is no line structure and no
|
|
||||||
// tree. Three operations exist and consumers drive them directly:
|
|
||||||
//
|
|
||||||
// read_token one whitespace-delimited token
|
|
||||||
// next the `KEY value` / `NAME {` / `}` step every loader loops on
|
|
||||||
// skip_block consume a block the caller does not understand
|
|
||||||
//
|
|
||||||
// Tokenizer rules (exact):
|
|
||||||
// * whitespace is ' ', '\t', '\r', '\n' -- nothing else;
|
|
||||||
// * a bareword runs until whitespace. Braces are NOT delimiters: `{` and `}`
|
|
||||||
// are only recognised when they are whole tokens, so `name{` is one word;
|
|
||||||
// * a token starting with `"`, `'` or a backtick is quoted; it ends at the
|
|
||||||
// next occurrence of the same character. No escape processing, the quotes
|
|
||||||
// are stripped, whitespace inside is kept, an unterminated quote runs to
|
|
||||||
// the end of the input (`Raw::unterminated` reports it);
|
|
||||||
// * a token whose text starts with `//` is a comment: the rest of that line
|
|
||||||
// is skipped and the next token is read. The check is made on the
|
|
||||||
// extracted text, so `3//x` is one word and a quoted token whose content
|
|
||||||
// starts with `//` is also a comment;
|
|
||||||
// * token text is silently truncated to kMaxTokenLength bytes (the scan
|
|
||||||
// still consumes the whole token);
|
|
||||||
// * no `/* */`, no `=`, no BOM handling.
|
|
||||||
//
|
|
||||||
// Status codes mirror the original: Ok = a token was read and input remains;
|
|
||||||
// AtEnd = a token was read and it touches the end of the input; NoInput = no
|
|
||||||
// token could be read. `next()` returns the status of its *value* read, which
|
|
||||||
// is why a final pair whose value touches the end of input is dropped by every
|
|
||||||
// consumer (they loop `while (next() == Ok)`).
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include <cstddef>
|
|
||||||
#include <string_view>
|
|
||||||
|
|
||||||
namespace mars::parse {
|
|
||||||
|
|
||||||
enum class ReadStatus { Ok = 0, NoInput = 1, AtEnd = 2 };
|
|
||||||
|
|
||||||
struct ScriptToken {
|
|
||||||
enum class Type { None = 0, Pair = 1, Open = 2, Close = 3 };
|
|
||||||
|
|
||||||
Type type = Type::None;
|
|
||||||
std::string_view key; // text with quotes stripped (Close: "}")
|
|
||||||
std::string_view value; // Pair: the value; Open: "{"; else empty
|
|
||||||
bool key_quoted = false;
|
|
||||||
bool value_quoted = false;
|
|
||||||
bool key_unterminated = false; // quoted token that ran to end of input
|
|
||||||
bool value_unterminated = false;
|
|
||||||
int key_line = 0; // 1-based line where each token started
|
|
||||||
int value_line = 0;
|
|
||||||
ReadStatus key_status = ReadStatus::NoInput; // status of each read;
|
|
||||||
ReadStatus value_status = ReadStatus::NoInput; // next() returns value_status
|
|
||||||
};
|
|
||||||
|
|
||||||
class Script {
|
|
||||||
public:
|
|
||||||
static constexpr std::size_t kMaxTokenLength = 1023;
|
|
||||||
|
|
||||||
struct Raw {
|
|
||||||
std::string_view text;
|
|
||||||
bool quoted = false;
|
|
||||||
bool unterminated = false;
|
|
||||||
int line = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
explicit Script(std::string_view text) : s_(text) {}
|
|
||||||
|
|
||||||
std::size_t cursor() const { return pos_; }
|
|
||||||
int line() const { return line_; }
|
|
||||||
bool at_end() const { return pos_ >= s_.size(); }
|
|
||||||
|
|
||||||
static bool is_space(char c) { return c == ' ' || c == '\t' || c == '\r' || c == '\n'; }
|
|
||||||
|
|
||||||
// One token. `out` is reset first; on NoInput it stays empty.
|
|
||||||
ReadStatus read_token(Raw& out) {
|
|
||||||
out = Raw{};
|
|
||||||
for (;;) {
|
|
||||||
skip_space();
|
|
||||||
if (pos_ >= s_.size()) return ReadStatus::NoInput;
|
|
||||||
|
|
||||||
const int line = line_;
|
|
||||||
const char c = s_[pos_];
|
|
||||||
std::string_view text;
|
|
||||||
bool quoted = false, unterminated = false;
|
|
||||||
if (c == '"' || c == '\'' || c == '`') {
|
|
||||||
quoted = true;
|
|
||||||
const std::size_t close = s_.find(c, pos_ + 1);
|
|
||||||
const std::size_t end = close == std::string_view::npos ? s_.size() : close;
|
|
||||||
text = s_.substr(pos_ + 1, end - pos_ - 1);
|
|
||||||
unterminated = close == std::string_view::npos;
|
|
||||||
advance_to(unterminated ? s_.size() : close + 1);
|
|
||||||
} else {
|
|
||||||
std::size_t end = pos_;
|
|
||||||
while (end < s_.size() && !is_space(s_[end])) ++end;
|
|
||||||
text = s_.substr(pos_, end - pos_);
|
|
||||||
pos_ = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (text.size() >= 2 && text[0] == '/' && text[1] == '/') {
|
|
||||||
skip_line(); // comment: to after the next \r/\n run, then read again
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (text.size() > kMaxTokenLength) text = text.substr(0, kMaxTokenLength);
|
|
||||||
out.text = text;
|
|
||||||
out.quoted = quoted;
|
|
||||||
out.unterminated = unterminated;
|
|
||||||
out.line = line;
|
|
||||||
return pos_ >= s_.size() ? ReadStatus::AtEnd : ReadStatus::Ok;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// The loaders' step: a key, then (unless the key is `}`) a value.
|
|
||||||
// Returns the status of the last read; the token is only complete when
|
|
||||||
// that is Ok. The text of whatever was read is filled in regardless.
|
|
||||||
ReadStatus next(ScriptToken& out) {
|
|
||||||
out = ScriptToken{};
|
|
||||||
Raw k;
|
|
||||||
out.key_status = read_token(k);
|
|
||||||
out.key = k.text;
|
|
||||||
out.key_quoted = k.quoted;
|
|
||||||
out.key_unterminated = k.unterminated;
|
|
||||||
out.key_line = k.line;
|
|
||||||
if (out.key_status != ReadStatus::Ok) return out.key_status;
|
|
||||||
if (k.text == "}") { // exact compare on the stripped text
|
|
||||||
out.type = ScriptToken::Type::Close;
|
|
||||||
return ReadStatus::Ok;
|
|
||||||
}
|
|
||||||
Raw v;
|
|
||||||
out.value_status = read_token(v);
|
|
||||||
out.value = v.text;
|
|
||||||
out.value_quoted = v.quoted;
|
|
||||||
out.value_unterminated = v.unterminated;
|
|
||||||
out.value_line = v.line;
|
|
||||||
out.type = v.text == "{" ? ScriptToken::Type::Open : ScriptToken::Type::Pair;
|
|
||||||
return out.value_status;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Consume tokens, counting `{` / `}`, until `depth` reaches zero.
|
|
||||||
// Ok when the block closed; otherwise the status that ended the input.
|
|
||||||
ReadStatus skip_block(int depth) {
|
|
||||||
Raw t;
|
|
||||||
while (depth > 0) {
|
|
||||||
ReadStatus rc = read_token(t);
|
|
||||||
if (rc == ReadStatus::NoInput) return rc;
|
|
||||||
if (t.text == "{") ++depth;
|
|
||||||
else if (t.text == "}") --depth;
|
|
||||||
if (rc != ReadStatus::Ok) return depth == 0 ? ReadStatus::Ok : rc;
|
|
||||||
}
|
|
||||||
return ReadStatus::Ok;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
void skip_space() {
|
|
||||||
while (pos_ < s_.size() && is_space(s_[pos_])) {
|
|
||||||
if (s_[pos_] == '\n') ++line_;
|
|
||||||
++pos_;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
void skip_line() {
|
|
||||||
while (pos_ < s_.size() && s_[pos_] != '\r' && s_[pos_] != '\n') ++pos_;
|
|
||||||
while (pos_ < s_.size() && (s_[pos_] == '\r' || s_[pos_] == '\n')) {
|
|
||||||
if (s_[pos_] == '\n') ++line_;
|
|
||||||
++pos_;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
void advance_to(std::size_t p) {
|
|
||||||
for (; pos_ < p; ++pos_)
|
|
||||||
if (s_[pos_] == '\n') ++line_;
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string_view s_;
|
|
||||||
std::size_t pos_ = 0;
|
|
||||||
int line_ = 1;
|
|
||||||
};
|
|
||||||
|
|
||||||
} // namespace mars::parse
|
|
||||||
|
|
@ -1,6 +1,5 @@
|
||||||
#include "mars/text/flat_kv.h"
|
#include "mars/text/flat_kv.h"
|
||||||
|
|
||||||
#include "mars/parse/script.h"
|
|
||||||
#include "mars/text/lines.h"
|
#include "mars/text/lines.h"
|
||||||
|
|
||||||
namespace mars::text {
|
namespace mars::text {
|
||||||
|
|
@ -49,46 +48,64 @@ bool has_unbalanced_quote(std::string_view line) {
|
||||||
return quotes % 2 == 1;
|
return quotes % 2 == 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::string quote(std::string_view s) { return "'" + std::string(s) + "'"; }
|
// Shared line pass: yields (line number, tokens) for every non-blank line.
|
||||||
|
template <class Fn>
|
||||||
|
void tokenised_lines(std::string_view text, std::vector<Problem>& problems, Fn&& fn) {
|
||||||
|
for_each_line(text, [&](int lineno, std::string_view raw) {
|
||||||
|
std::string_view line = strip(strip_comment(raw));
|
||||||
|
if (line.empty()) return;
|
||||||
|
if (has_unbalanced_quote(line))
|
||||||
|
problems.push_back(Problem{Problem::Kind::UnbalancedQuote, lineno, -1,
|
||||||
|
"line " + std::to_string(lineno) + ": unbalanced quote"});
|
||||||
|
fn(lineno, split_tokens(line));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
// ---- FlatKV --------------------------------------------------------------
|
// ---- FlatKV --------------------------------------------------------------
|
||||||
|
|
||||||
void FlatKV::add(KvEntry entry) {
|
void FlatKV::add(KvEntry entry) {
|
||||||
first_.emplace(fold_case(entry.key), entries_.size()); // keeps the first index
|
last_[fold_case(entry.key)] = entries_.size();
|
||||||
entries_.push_back(std::move(entry));
|
entries_.push_back(std::move(entry));
|
||||||
}
|
}
|
||||||
|
|
||||||
const KvEntry* FlatKV::find(std::string_view key) const {
|
const KvEntry* FlatKV::find(std::string_view key) const {
|
||||||
auto it = first_.find(fold_case(key));
|
auto it = last_.find(fold_case(key));
|
||||||
return it == first_.end() ? nullptr : &entries_[it->second];
|
return it == last_.end() ? nullptr : &entries_[it->second];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
namespace {
|
||||||
|
const Token* first_token(const FlatKV& kv, std::string_view key) {
|
||||||
|
const KvEntry* e = kv.find(key);
|
||||||
|
return (e && !e->values.empty()) ? &e->values.front() : nullptr;
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
|
|
||||||
std::optional<std::int64_t> FlatKV::get_int(std::string_view key) const {
|
std::optional<std::int64_t> FlatKV::get_int(std::string_view key) const {
|
||||||
const KvEntry* e = find(key);
|
const Token* t = first_token(*this, key);
|
||||||
return e ? e->value.as_int() : std::nullopt;
|
return t ? t->as_int() : std::nullopt;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<double> FlatKV::get_float(std::string_view key) const {
|
std::optional<double> FlatKV::get_float(std::string_view key) const {
|
||||||
const KvEntry* e = find(key);
|
const Token* t = first_token(*this, key);
|
||||||
return e ? e->value.as_float() : std::nullopt;
|
return t ? t->as_float() : std::nullopt;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<bool> FlatKV::get_bool(std::string_view key) const {
|
std::optional<bool> FlatKV::get_bool(std::string_view key) const {
|
||||||
const KvEntry* e = find(key);
|
const Token* t = first_token(*this, key);
|
||||||
return e ? e->value.as_bool() : std::nullopt;
|
return t ? t->as_bool() : std::nullopt;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<std::string_view> FlatKV::get_string(std::string_view key) const {
|
std::optional<std::string_view> FlatKV::get_string(std::string_view key) const {
|
||||||
const KvEntry* e = find(key);
|
const Token* t = first_token(*this, key);
|
||||||
if (!e) return std::nullopt;
|
if (!t) return std::nullopt;
|
||||||
return std::string_view(e->value.text);
|
return std::string_view(t->text);
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<Color> FlatKV::get_color(std::string_view key) const {
|
std::optional<Color> FlatKV::get_color(std::string_view key) const {
|
||||||
const KvEntry* e = find(key);
|
const Token* t = first_token(*this, key);
|
||||||
return e ? parse_color(e->value.text) : std::nullopt;
|
return t ? parse_color(t->text) : std::nullopt;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<std::pair<std::string, std::vector<int>>> FlatKV::duplicates() const {
|
std::vector<std::pair<std::string, std::vector<int>>> FlatKV::duplicates() const {
|
||||||
|
|
@ -106,55 +123,16 @@ std::vector<std::pair<std::string, std::vector<int>>> FlatKV::duplicates() const
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
||||||
// The GlobalConsts loader loop: step with Script::next(); a pair is consumed
|
|
||||||
// on first sight, a block is skipped, a stray `}` is ignored, and the loop
|
|
||||||
// ends on the first step that does not complete (which drops that step).
|
|
||||||
Result<FlatKV> parse_flat_kv(std::string_view text) {
|
Result<FlatKV> parse_flat_kv(std::string_view text) {
|
||||||
using mars::parse::ReadStatus;
|
|
||||||
using mars::parse::ScriptToken;
|
|
||||||
|
|
||||||
Result<FlatKV> r;
|
Result<FlatKV> r;
|
||||||
mars::parse::Script script(text);
|
tokenised_lines(text, r.problems, [&](int lineno, std::vector<Token> toks) {
|
||||||
for (;;) {
|
|
||||||
ScriptToken t;
|
|
||||||
const ReadStatus rc = script.next(t);
|
|
||||||
if (t.key_unterminated)
|
|
||||||
r.problems.push_back(Problem{Problem::Kind::UnterminatedQuote, t.key_line, -1,
|
|
||||||
"line " + std::to_string(t.key_line) + ": unterminated quote"});
|
|
||||||
if (t.value_unterminated)
|
|
||||||
r.problems.push_back(Problem{Problem::Kind::UnterminatedQuote, t.value_line, -1,
|
|
||||||
"line " + std::to_string(t.value_line) + ": unterminated quote"});
|
|
||||||
if (rc != ReadStatus::Ok) {
|
|
||||||
const bool key_read = t.key_status != ReadStatus::NoInput;
|
|
||||||
if (key_read && t.key != "}") {
|
|
||||||
std::string what = t.value_status == ReadStatus::NoInput || t.key_status == ReadStatus::AtEnd
|
|
||||||
? "trailing key " + quote(t.key) + " without a value: dropped"
|
|
||||||
: "final pair " + quote(t.key) + " " + quote(t.value) +
|
|
||||||
" touches end of input (no trailing newline): dropped";
|
|
||||||
r.problems.push_back(Problem{Problem::Kind::DroppedTrailingPair, t.key_line, -1,
|
|
||||||
"line " + std::to_string(t.key_line) + ": " + what});
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (t.type == ScriptToken::Type::Pair) {
|
|
||||||
if (r.value.has(t.key)) {
|
|
||||||
r.problems.push_back(Problem{Problem::Kind::DuplicateKey, t.key_line, -1,
|
|
||||||
"line " + std::to_string(t.key_line) + ": " + quote(t.key) +
|
|
||||||
" multiply defined (first occurrence kept)"});
|
|
||||||
}
|
|
||||||
KvEntry e;
|
KvEntry e;
|
||||||
e.line = t.key_line;
|
e.line = lineno;
|
||||||
e.key = std::string(t.key);
|
e.key = std::move(toks.front().text);
|
||||||
e.value = Token{std::string(t.value), t.value_quoted};
|
toks.erase(toks.begin());
|
||||||
|
e.values = std::move(toks);
|
||||||
r.value.add(std::move(e));
|
r.value.add(std::move(e));
|
||||||
} else if (t.type == ScriptToken::Type::Open) {
|
});
|
||||||
r.problems.push_back(Problem{Problem::Kind::SkippedBlock, t.key_line, -1,
|
|
||||||
"line " + std::to_string(t.key_line) + ": block " + quote(t.key) +
|
|
||||||
" skipped"});
|
|
||||||
if (script.skip_block(1) != ReadStatus::Ok) break;
|
|
||||||
}
|
|
||||||
// Close at top level: ignored
|
|
||||||
}
|
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -162,13 +140,8 @@ Result<FlatKV> parse_flat_kv(std::string_view text) {
|
||||||
|
|
||||||
Result<Rows> parse_rows(std::string_view text) {
|
Result<Rows> parse_rows(std::string_view text) {
|
||||||
Result<Rows> r;
|
Result<Rows> r;
|
||||||
for_each_line(text, [&](int lineno, std::string_view raw) {
|
tokenised_lines(text, r.problems, [&](int lineno, std::vector<Token> toks) {
|
||||||
std::string_view line = strip(strip_comment(raw));
|
r.value.push_back(Row{lineno, std::move(toks)});
|
||||||
if (line.empty()) return;
|
|
||||||
if (has_unbalanced_quote(line))
|
|
||||||
r.problems.push_back(Problem{Problem::Kind::UnbalancedQuote, lineno, -1,
|
|
||||||
"line " + std::to_string(lineno) + ": unbalanced quote"});
|
|
||||||
r.value.push_back(Row{lineno, split_tokens(line)});
|
|
||||||
});
|
});
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -1,26 +1,14 @@
|
||||||
// mars::text -- the flat "KEY value" tuning tables (Data/**/*.txt, read by the
|
// mars::text -- the flat "KEY value" tuning tables and whitespace-positional
|
||||||
// engine's GlobalConsts loader) and the whitespace-positional row tables
|
// row tables (Data/**/*.txt, Weapons/_turrets.txt, Badges/BadgeTable.txt ...).
|
||||||
// (Weapons/_turrets.txt, Badges/BadgeTable.txt ...).
|
|
||||||
//
|
//
|
||||||
// KEY value tables (parse_flat_kv) follow the engine's loader exactly. The
|
// Line grammar (both shapes):
|
||||||
// file is not read by lines: the same pull tokenizer as the brace-block
|
|
||||||
// catalogs (mars/parse/script.h) steps through it and every step is
|
|
||||||
// KEY value one token each -- so `COLOR "48 29 2"` must be quoted, and
|
|
||||||
// `LIST 1 2 3` would be the pairs LIST=1 and 2=3
|
|
||||||
// NAME { a block; skipped up to its matching `}`
|
|
||||||
// } ignored
|
|
||||||
// Keys are matched case-insensitively and the FIRST occurrence wins: the
|
|
||||||
// loader erases a key once it has consumed it, so a later duplicate is
|
|
||||||
// "multiply defined" and ignored (Problem::Kind::DuplicateKey). A final pair
|
|
||||||
// whose value touches the end of the file (no trailing newline) is dropped
|
|
||||||
// (DroppedTrailingPair) -- two shipped files lose their last key this way.
|
|
||||||
// `//` comments, `"`/`'`/backtick quotes without escapes, 1023-byte tokens:
|
|
||||||
// see script.h.
|
|
||||||
//
|
|
||||||
// Row tables (parse_rows) are line-based:
|
|
||||||
// line := token* ('//' comment)? -- '//' inside "..." is not a comment
|
// line := token* ('//' comment)? -- '//' inside "..." is not a comment
|
||||||
// token := '"' [^"]* '"' | non-whitespace+ -- a quoted token may be empty ("")
|
// token := '"' [^"]* '"' | non-whitespace+ -- a quoted token may be empty ("")
|
||||||
// -- an unpaired '"' is just a bareword char
|
// -- an unpaired '"' is just a bareword char
|
||||||
|
// KEY value tables take the first token of a line as the key and the rest as
|
||||||
|
// its value(s); a key alone is allowed (empty value). Keys are looked up
|
||||||
|
// case-insensitively; when a key repeats, the last line wins (earlier entries
|
||||||
|
// stay visible through entries() / duplicates()).
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include <optional>
|
#include <optional>
|
||||||
|
|
@ -37,22 +25,21 @@ namespace mars::text {
|
||||||
|
|
||||||
struct KvEntry {
|
struct KvEntry {
|
||||||
int line = 0;
|
int line = 0;
|
||||||
std::string key; // original spelling (quotes stripped if it was quoted)
|
std::string key; // original spelling
|
||||||
Token value; // exactly one token
|
std::vector<Token> values; // 0 = key only, 1 = scalar, n = list
|
||||||
};
|
};
|
||||||
|
|
||||||
class FlatKV {
|
class FlatKV {
|
||||||
public:
|
public:
|
||||||
// Every pair in file order, later duplicates included.
|
|
||||||
const std::vector<KvEntry>& entries() const { return entries_; }
|
const std::vector<KvEntry>& entries() const { return entries_; }
|
||||||
std::size_t size() const { return entries_.size(); }
|
std::size_t size() const { return entries_.size(); }
|
||||||
|
|
||||||
// Case-insensitive lookup; the FIRST entry with that key (engine rule).
|
// Case-insensitive lookup; the last entry with that key.
|
||||||
const KvEntry* find(std::string_view key) const;
|
const KvEntry* find(std::string_view key) const;
|
||||||
bool has(std::string_view key) const { return find(key) != nullptr; }
|
bool has(std::string_view key) const { return find(key) != nullptr; }
|
||||||
|
|
||||||
// Convenience accessors on the entry's value token.
|
// Convenience accessors on the first value token of the entry.
|
||||||
// nullopt when the key is missing or the token has the wrong shape.
|
// nullopt when the key is missing, has no value, or the token has the wrong shape.
|
||||||
std::optional<std::int64_t> get_int(std::string_view key) const;
|
std::optional<std::int64_t> get_int(std::string_view key) const;
|
||||||
std::optional<double> get_float(std::string_view key) const;
|
std::optional<double> get_float(std::string_view key) const;
|
||||||
std::optional<bool> get_bool(std::string_view key) const;
|
std::optional<bool> get_bool(std::string_view key) const;
|
||||||
|
|
@ -66,7 +53,7 @@ public:
|
||||||
|
|
||||||
private:
|
private:
|
||||||
std::vector<KvEntry> entries_;
|
std::vector<KvEntry> entries_;
|
||||||
std::unordered_map<std::string, std::size_t> first_; // folded key -> index of first entry
|
std::unordered_map<std::string, std::size_t> last_; // folded key -> index of last entry
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Row {
|
struct Row {
|
||||||
|
|
@ -78,8 +65,8 @@ using Rows = std::vector<Row>;
|
||||||
Result<FlatKV> parse_flat_kv(std::string_view text);
|
Result<FlatKV> parse_flat_kv(std::string_view text);
|
||||||
Result<Rows> parse_rows(std::string_view text);
|
Result<Rows> parse_rows(std::string_view text);
|
||||||
|
|
||||||
// Line-based building blocks used by parse_rows, exposed for one-off readers.
|
// Building blocks, exposed for tests and for one-off line readers.
|
||||||
std::string_view strip_comment(std::string_view line); // drop a trailing // comment (quote-aware)
|
std::string_view strip_comment(std::string_view line); // drop a trailing // comment (quote-aware)
|
||||||
std::vector<Token> split_tokens(std::string_view line); // tokenise per the row grammar above
|
std::vector<Token> split_tokens(std::string_view line); // tokenise per the grammar above
|
||||||
|
|
||||||
} // namespace mars::text
|
} // namespace mars::text
|
||||||
|
|
|
||||||
|
|
@ -15,12 +15,8 @@ struct Problem {
|
||||||
Unrecognised, // manifest: line is neither entry, comment nor tombstone
|
Unrecognised, // manifest: line is neither entry, comment nor tombstone
|
||||||
DuplicateId, // manifest: id assigned twice (id = the repeated id)
|
DuplicateId, // manifest: id assigned twice (id = the repeated id)
|
||||||
DeletedAndAssigned, // manifest: id is both a tombstone and an entry (line = 0)
|
DeletedAndAssigned, // manifest: id is both a tombstone and an entry (line = 0)
|
||||||
UnbalancedQuote, // rows: odd number of '"' on a line
|
UnbalancedQuote, // flat kv / rows: odd number of '"' on a line
|
||||||
UnterminatedQuotedCell, // csv: end of input inside a quoted cell
|
UnterminatedQuotedCell // csv: end of input inside a quoted cell
|
||||||
DuplicateKey, // flat kv: key already defined (this later line is ignored)
|
|
||||||
DroppedTrailingPair, // flat kv: final pair touches end of input (no newline): dropped
|
|
||||||
UnterminatedQuote, // flat kv: a quote that runs to end of input
|
|
||||||
SkippedBlock // flat kv: a `name {` block was skipped
|
|
||||||
};
|
};
|
||||||
Kind kind;
|
Kind kind;
|
||||||
int line = 0; // 1-based; 0 when the problem is not tied to a line
|
int line = 0; // 1-based; 0 when the problem is not tied to a line
|
||||||
|
|
|
||||||
|
|
@ -155,13 +155,9 @@ TEST(weapon_no_block_is_fatal) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(weapon_syntax_error_is_fatal_and_recovery_is_reported) {
|
TEST(weapon_syntax_error_is_fatal_and_recovery_is_reported) {
|
||||||
// Engine parity: an unterminated quote is NOT fatal -- the string runs to EOF (swallowing the
|
|
||||||
// '}') and EOF closes the block; the recovery is reported as a Syntax-kind problem.
|
|
||||||
auto bad = parse_weapon("weapon { name \"unterminated }");
|
auto bad = parse_weapon("weapon { name \"unterminated }");
|
||||||
CHECK(bad.ok());
|
CHECK(!bad.ok());
|
||||||
bool reported = false;
|
CHECK(bad.problems[0].kind == Problem::Kind::Syntax);
|
||||||
for (const auto& p : bad.problems) if (p.kind == Problem::Kind::Syntax) reported = true;
|
|
||||||
CHECK(reported);
|
|
||||||
auto rec = parse_weapon("weapon { name @W weaponclass bullet turretsize small turretclass standard\n"); // EOF closes
|
auto rec = parse_weapon("weapon { name @W weaponclass bullet turretsize small turretclass standard\n"); // EOF closes
|
||||||
CHECK(rec.ok());
|
CHECK(rec.ok());
|
||||||
CHECK_EQ(rec.problems.size(), std::size_t{1});
|
CHECK_EQ(rec.problems.size(), std::size_t{1});
|
||||||
|
|
|
||||||
|
|
@ -108,6 +108,9 @@ Json node_to_json(const Node& n) {
|
||||||
Json obj = Json::object();
|
Json obj = Json::object();
|
||||||
for (const Entry& e : n.entries) {
|
for (const Entry& e : n.entries) {
|
||||||
switch (e.kind) {
|
switch (e.kind) {
|
||||||
|
case Entry::Kind::Item:
|
||||||
|
obj.add("_items", Json::string(e.value));
|
||||||
|
break;
|
||||||
case Entry::Kind::Pair:
|
case Entry::Kind::Pair:
|
||||||
obj.add(lower(e.key), e.quoted ? Json::string(e.value) : typed(e.value));
|
obj.add(lower(e.key), e.quoted ? Json::string(e.value) : typed(e.value));
|
||||||
break;
|
break;
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,6 @@
|
||||||
|
|
||||||
#include "canon.h"
|
#include "canon.h"
|
||||||
#include "mars/parse/blocks.h"
|
#include "mars/parse/blocks.h"
|
||||||
#include "mars/parse/script.h"
|
|
||||||
#include "test_main.h"
|
#include "test_main.h"
|
||||||
|
|
||||||
using namespace mars::parse;
|
using namespace mars::parse;
|
||||||
|
|
@ -24,74 +23,8 @@ std::string canon_of(std::string_view text, Options o = {}) {
|
||||||
return canon::blocks_json(must_parse(text, o).root);
|
return canon::blocks_json(must_parse(text, o).root);
|
||||||
}
|
}
|
||||||
|
|
||||||
Options strict_opts() {
|
|
||||||
Options o;
|
|
||||||
o.strict = true;
|
|
||||||
return o;
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
// ---- Script: the engine's pull tokenizer -------------------------------------
|
|
||||||
|
|
||||||
TEST(script_read_token_statuses) {
|
|
||||||
Script s("a b\n");
|
|
||||||
Script::Raw t;
|
|
||||||
CHECK(s.read_token(t) == ReadStatus::Ok);
|
|
||||||
CHECK_EQ(std::string(t.text), "a");
|
|
||||||
CHECK(s.read_token(t) == ReadStatus::Ok); // "b" is followed by the newline
|
|
||||||
CHECK_EQ(std::string(t.text), "b");
|
|
||||||
CHECK(s.read_token(t) == ReadStatus::NoInput);
|
|
||||||
CHECK(t.text.empty());
|
|
||||||
|
|
||||||
Script e("a b");
|
|
||||||
CHECK(e.read_token(t) == ReadStatus::Ok);
|
|
||||||
CHECK(e.read_token(t) == ReadStatus::AtEnd); // "b" touches the end of input
|
|
||||||
CHECK_EQ(std::string(t.text), "b");
|
|
||||||
CHECK(e.read_token(t) == ReadStatus::NoInput);
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(script_next_returns_the_value_read_status) {
|
|
||||||
Script s("k v\nname {\n}\nlast 1");
|
|
||||||
ScriptToken t;
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok);
|
|
||||||
CHECK(t.type == ScriptToken::Type::Pair);
|
|
||||||
CHECK_EQ(std::string(t.key), "k");
|
|
||||||
CHECK_EQ(std::string(t.value), "v");
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok);
|
|
||||||
CHECK(t.type == ScriptToken::Type::Open);
|
|
||||||
CHECK_EQ(std::string(t.key), "name");
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok);
|
|
||||||
CHECK(t.type == ScriptToken::Type::Close);
|
|
||||||
CHECK(s.next(t) == ReadStatus::AtEnd); // the step was read but is incomplete
|
|
||||||
CHECK_EQ(std::string(t.key), "last"); // ... its text is still available
|
|
||||||
CHECK_EQ(std::string(t.value), "1");
|
|
||||||
CHECK(s.next(t) == ReadStatus::NoInput);
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(script_skip_block_counts_whole_brace_tokens_only) {
|
|
||||||
Script s("x 1 a{ } b } after 2\n");
|
|
||||||
ScriptToken t;
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok); // x 1
|
|
||||||
CHECK(s.skip_block(1) == ReadStatus::Ok); // a{ is a word; first whole } closes
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok);
|
|
||||||
CHECK_EQ(std::string(t.key), "b"); // b } -> pair b="}"
|
|
||||||
CHECK_EQ(std::string(t.value), "}");
|
|
||||||
Script u("a { b {");
|
|
||||||
CHECK(u.skip_block(1) != ReadStatus::Ok); // input ends inside the block
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(script_token_text_is_capped_at_1023_bytes) {
|
|
||||||
const std::string text = std::string(1100, 'x') + " 1\n";
|
|
||||||
Script s(text);
|
|
||||||
ScriptToken t;
|
|
||||||
CHECK(s.next(t) == ReadStatus::Ok);
|
|
||||||
CHECK_EQ(t.key.size(), Script::kMaxTokenLength);
|
|
||||||
CHECK_EQ(std::string(t.value), "1"); // the whole word was consumed
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- tree ---------------------------------------------------------------------
|
|
||||||
|
|
||||||
TEST(basic_block_pairs_and_nesting) {
|
TEST(basic_block_pairs_and_nesting) {
|
||||||
Document d = must_parse(
|
Document d = must_parse(
|
||||||
"weapon\n{\n\tname @WEAPON_X\n\tcost 50\n\tbank { mount { node N1 } }\n}\n");
|
"weapon\n{\n\tname @WEAPON_X\n\tcost 50\n\tbank { mount { node N1 } }\n}\n");
|
||||||
|
|
@ -121,33 +54,17 @@ TEST(quoted_values_keep_spaces_backslashes_and_slashes) {
|
||||||
CHECK_EQ(*d.root.first_value("d"), "");
|
CHECK_EQ(*d.root.first_value("d"), "");
|
||||||
CHECK(d.root.first("a")->quoted);
|
CHECK(d.root.first("a")->quoted);
|
||||||
CHECK(d.root.first("d")->quoted);
|
CHECK(d.root.first("d")->quoted);
|
||||||
CHECK(!d.root.first("a")->key_quoted);
|
|
||||||
CHECK_EQ(d.root.entries.size(), std::size_t{4});
|
CHECK_EQ(d.root.entries.size(), std::size_t{4});
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(single_quote_and_backtick_also_quote) {
|
|
||||||
Document d = must_parse("a 'x y'\nb `p q`\nc \"it's\"\nd 'say \"hi\"'\n");
|
|
||||||
CHECK_EQ(*d.root.first_value("a"), "x y");
|
|
||||||
CHECK(d.root.first("a")->quoted);
|
|
||||||
CHECK_EQ(*d.root.first_value("b"), "p q");
|
|
||||||
CHECK_EQ(*d.root.first_value("c"), "it's"); // only the opening character closes
|
|
||||||
CHECK_EQ(*d.root.first_value("d"), "say \"hi\"");
|
|
||||||
CHECK(d.warnings.empty());
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(comments_are_stripped) {
|
TEST(comments_are_stripped) {
|
||||||
std::string s = canon_of("// leading\nx 1 // trailing\n//y 2\nz 3\n");
|
std::string s = canon_of("// leading\nx 1 // trailing\n//y 2\nz 3\n");
|
||||||
CHECK_EQ(s, R"({"x":1,"z":3})");
|
CHECK_EQ(s, R"({"x":1,"z":3})");
|
||||||
// The comment test is made on the extracted token: glued to a word it is
|
// '//' only starts a comment at a token boundary: glued to a bareword it is
|
||||||
// part of the word ...
|
// part of the word (the reference reader behaves this way; the shipped data
|
||||||
|
// never relies on either reading).
|
||||||
CHECK_EQ(canon_of("z 3//glued\n"), R"({"z":"3//glued"})");
|
CHECK_EQ(canon_of("z 3//glued\n"), R"({"z":"3//glued"})");
|
||||||
CHECK_EQ(canon_of("z 3 //not glued\n"), R"({"z":3})");
|
CHECK_EQ(canon_of("z 3 //not glued\n"), R"({"z":3})");
|
||||||
// ... and a quoted token whose content starts with // is a comment too.
|
|
||||||
CHECK_EQ(canon_of("\"// whole line\" junk\na 1\n"), R"({"a":1})");
|
|
||||||
// a closing quote glued to // : the quoted token ends, then the comment starts
|
|
||||||
CHECK_EQ(canon_of("c \"0 0 0\"// \" 92 76 20\"\n"), R"({"c":"0 0 0"})");
|
|
||||||
// comment on the last line without a newline: the pair before it survives
|
|
||||||
CHECK_EQ(canon_of("a 1 // end"), R"({"a":1})");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(repeated_keys_stay_in_order) {
|
TEST(repeated_keys_stay_in_order) {
|
||||||
|
|
@ -171,35 +88,22 @@ TEST(keys_are_case_insensitive_but_case_is_kept) {
|
||||||
CHECK_EQ(canon_of("tech { Requires X requires Y }"), R"({"tech":{"requires":["X","Y"]}})");
|
CHECK_EQ(canon_of("tech { Requires X requires Y }"), R"({"tech":{"requires":["X","Y"]}})");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(quoted_strings_in_key_position_are_keys) {
|
TEST(bare_quoted_items_and_trailing_bareword) {
|
||||||
// systemnames.txt lists: the engine pairs them up like any KEY value step
|
Document d = must_parse("names { \"Sol\" \"Alpha Centauri\" }\nflags { on off }");
|
||||||
Document d = must_parse("names { \"Sol\" \"Alpha Centauri\" \"Vega\" \"Deneb\" }\n");
|
auto items = d.root.first_block("names")->items();
|
||||||
const Node* n = d.root.first_block("names");
|
CHECK_EQ(items.size(), std::size_t{2});
|
||||||
CHECK_EQ(n->entries.size(), std::size_t{2});
|
CHECK_EQ(std::string(items[1]), "Alpha Centauri");
|
||||||
CHECK_EQ(n->entries[0].key, "Sol");
|
// "on off": 'on' takes 'off' as its value; nothing is left over.
|
||||||
CHECK(n->entries[0].key_quoted);
|
const Node* f = d.root.first_block("flags");
|
||||||
CHECK_EQ(n->entries[0].value, "Alpha Centauri");
|
CHECK_EQ(*f->first_value("on"), "off");
|
||||||
CHECK(n->entries[0].quoted);
|
// A lone word right before '}' is an item, not a key.
|
||||||
CHECK_EQ(*n->first_value("vega"), "Deneb");
|
Document e = must_parse("flags { a 1 solo }");
|
||||||
CHECK_EQ(canon_of("n { \"x\" \"y\" }\n"), R"({"n":{"x":"y"}})");
|
CHECK_EQ(e.root.first_block("flags")->items().size(), std::size_t{1});
|
||||||
}
|
CHECK_EQ(std::string(e.root.first_block("flags")->items()[0]), "solo");
|
||||||
|
CHECK_EQ(canon_of("flags { a 1 solo }"), R"({"flags":{"_items":"solo","a":1}})");
|
||||||
TEST(odd_item_count_pairs_the_last_item_with_the_brace) {
|
CHECK_EQ(canon_of("n { \"x\" \"y\" }"), R"({"n":{"_items":["x","y"]}})");
|
||||||
// The `}` becomes a value, the block stays open and swallows what follows
|
// lone bareword at EOF, top level
|
||||||
// (hiver/liir lists in the shipped systemnames.txt).
|
CHECK_EQ(canon_of("a 1 trailing"), R"({"_items":"trailing","a":1})");
|
||||||
auto r = parse_blocks("n { \"A\" \"B\" \"C\" }\nm { \"D\" \"E\" }\n");
|
|
||||||
CHECK(r.ok());
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"n":{"a":"B","c":"}","m":{"d":"E"}}})");
|
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1}); // n never closes
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(lone_word_before_a_brace_takes_the_brace_as_its_value) {
|
|
||||||
auto r = parse_blocks("flags { a 1 solo }\n");
|
|
||||||
CHECK(r.ok());
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"flags":{"a":1,"solo":"}"}})");
|
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1}); // flags is left open
|
|
||||||
// "on off": 'on' takes 'off' as its value
|
|
||||||
CHECK_EQ(canon_of("flags { on off }\n"), R"({"flags":{"on":"off"}})");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(option_scalar_and_block_share_a_key) {
|
TEST(option_scalar_and_block_share_a_key) {
|
||||||
|
|
@ -219,45 +123,22 @@ TEST(option_scalar_and_block_share_a_key) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(tokenizer_is_not_line_based) {
|
TEST(tokenizer_is_not_line_based) {
|
||||||
// block opens on the same line as a preceding pair; several entries per
|
// block opens on the same line as a preceding pair; name glued to brace;
|
||||||
// line; entries split across lines
|
// several entries per line; entries split across lines
|
||||||
std::string s = canon_of(
|
std::string s = canon_of(
|
||||||
"bank { turretsize small mount { node N }\n}\n"
|
"bank { turretsize small mount { node N }\n}\n"
|
||||||
"weapon { name\n\nX }\n"
|
"weapon{ name\n\nX }\n"
|
||||||
"a\n{\nb\n1\n}\n");
|
"a\n{\nb\n1\n}");
|
||||||
CHECK_EQ(s, R"({"a":{"b":1},"bank":{"mount":{"node":"N"},"turretsize":"small"},"weapon":{"name":"X"}})");
|
CHECK_EQ(s, R"({"a":{"b":1},"bank":{"mount":{"node":"N"},"turretsize":"small"},"weapon":{"name":"X"}})");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(braces_are_not_delimiters) {
|
TEST(braces_split_barewords) {
|
||||||
// Only a whole-token { or } is a brace: glued to a word it is part of the word.
|
CHECK_EQ(canon_of("a{b 1}c 2"), R"({"a":{"b":1},"c":2})");
|
||||||
// a{b -> key, 1}c -> value, then `2` is a trailing key without a value (dropped)
|
|
||||||
auto r = parse_blocks("a{b 1}c 2\n");
|
|
||||||
CHECK(r.ok());
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"a{b":"1}c"})");
|
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1});
|
|
||||||
auto w = parse_blocks("weapon{ name X }\n");
|
|
||||||
CHECK_EQ(canon::blocks_json(w->root), R"({"weapon{":"name","x":"}"})");
|
|
||||||
CHECK(w->root.first_block("weapon") == nullptr);
|
|
||||||
CHECK_EQ(canon_of("weapon { name X }\n"), R"({"weapon":{"name":"X"}})");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(quotes_do_not_split_barewords) {
|
TEST(quotes_split_barewords) {
|
||||||
// abc"def" is one word; a quote only opens a string at the start of a token
|
// abc"def" -> bareword abc, then quoted def (a pair abc="def")
|
||||||
CHECK_EQ(canon_of("abc\"def\" 1\n"), R"({"abc\"def\"":1})");
|
CHECK_EQ(canon_of("abc\"def\""), R"({"abc":"def"})");
|
||||||
// a closing quote followed directly by text: the next token starts there
|
|
||||||
CHECK_EQ(canon_of("k \"v\"x 1\n"), R"({"k":"v","x":1})");
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(quoted_braces_count_as_braces) {
|
|
||||||
// Next() compares the stripped text, so "{" and "}" open and close blocks.
|
|
||||||
CHECK_EQ(canon_of("a \"{\" x 1 \"}\"\n"), R"({"a":{"x":1}})");
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(whitespace_is_space_tab_cr_lf_only) {
|
|
||||||
Document d = must_parse("a\v1 2\nb\f3 4\n");
|
|
||||||
CHECK(d.root.first("a\v1") != nullptr);
|
|
||||||
CHECK_EQ(*d.root.first_value("a\v1"), "2");
|
|
||||||
CHECK_EQ(*d.root.first_value("b\f3"), "4");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(eof_closes_open_blocks_lenient) {
|
TEST(eof_closes_open_blocks_lenient) {
|
||||||
|
|
@ -272,26 +153,13 @@ TEST(eof_closes_open_blocks_lenient) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(eof_inside_block_is_an_error_when_strict) {
|
TEST(eof_inside_block_is_an_error_when_strict) {
|
||||||
auto r = parse_blocks("shipsection {\n model M\n", strict_opts());
|
Options strict;
|
||||||
|
strict.strict = true;
|
||||||
|
auto r = parse_blocks("shipsection {\n model M\n", strict);
|
||||||
CHECK(!r.ok());
|
CHECK(!r.ok());
|
||||||
CHECK_EQ(r.error().line, 3);
|
CHECK_EQ(r.error().line, 3);
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(closing_brace_touching_eof_is_a_plain_close) {
|
|
||||||
auto r = parse_blocks("a { x 1 }");
|
|
||||||
CHECK(r.ok());
|
|
||||||
CHECK(r->warnings.empty());
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"a":{"x":1}})");
|
|
||||||
auto two = parse_blocks("a { b { x 1 } }");
|
|
||||||
CHECK(two->warnings.empty());
|
|
||||||
auto one_short = parse_blocks("a { b { x 1 }");
|
|
||||||
CHECK_EQ(one_short->warnings.size(), std::size_t{1}); // a is left open
|
|
||||||
CHECK(one_short->root.first_block("a")->first_block("b") != nullptr);
|
|
||||||
auto stray = parse_blocks("a { x 1 }\n}");
|
|
||||||
CHECK_EQ(stray->warnings.size(), std::size_t{1}); // stray at top level
|
|
||||||
CHECK(parse_blocks("a { x 1 }", strict_opts()).ok());
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(stray_top_level_close_is_ignored_lenient) {
|
TEST(stray_top_level_close_is_ignored_lenient) {
|
||||||
auto r = parse_blocks("a { x 1 }\n}\nb { y 2 }\n");
|
auto r = parse_blocks("a { x 1 }\n}\nb { y 2 }\n");
|
||||||
CHECK(r.ok());
|
CHECK(r.ok());
|
||||||
|
|
@ -303,63 +171,28 @@ TEST(stray_top_level_close_is_ignored_lenient) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(stray_top_level_close_is_an_error_when_strict) {
|
TEST(stray_top_level_close_is_an_error_when_strict) {
|
||||||
auto r = parse_blocks("a { x 1 }\n}\n", strict_opts());
|
Options strict;
|
||||||
|
strict.strict = true;
|
||||||
|
auto r = parse_blocks("a { x 1 }\n}\n", strict);
|
||||||
CHECK(!r.ok());
|
CHECK(!r.ok());
|
||||||
CHECK_EQ(r.error().line, 2);
|
CHECK_EQ(r.error().line, 2);
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(final_pair_without_trailing_newline_is_dropped) {
|
TEST(unterminated_string_is_always_an_error) {
|
||||||
auto r = parse_blocks("a 1\nb 2");
|
|
||||||
CHECK(r.ok());
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"a":1})");
|
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1});
|
|
||||||
CHECK_EQ(r->warnings[0].line, 2);
|
|
||||||
CHECK(!parse_blocks("a 1\nb 2", strict_opts()).ok());
|
|
||||||
// any trailing whitespace saves it
|
|
||||||
CHECK_EQ(canon_of("a 1\nb 2 "), R"({"a":1,"b":2})");
|
|
||||||
CHECK_EQ(canon_of("a 1\nb 2\r\n"), R"({"a":1,"b":2})");
|
|
||||||
// a quoted value whose closing quote is the last byte is dropped too
|
|
||||||
CHECK_EQ(canon_of("a 1\nb \"2\""), R"({"a":1})");
|
|
||||||
// inside a block the pair is dropped and the block is left open
|
|
||||||
auto in_block = parse_blocks("s {\n a 1\n b 2");
|
|
||||||
CHECK_EQ(canon::blocks_json(in_block->root), R"({"s":{"a":1}})");
|
|
||||||
CHECK_EQ(in_block->warnings.size(), std::size_t{2});
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(trailing_key_without_value_is_dropped) {
|
|
||||||
auto r = parse_blocks("a 1\nb\n");
|
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"a":1})");
|
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1});
|
|
||||||
auto t = parse_blocks("a 1\nb"); // key itself touches the end
|
|
||||||
CHECK_EQ(canon::blocks_json(t->root), R"({"a":1})");
|
|
||||||
CHECK_EQ(t->warnings.size(), std::size_t{1});
|
|
||||||
auto h = parse_blocks("a 1\nb {"); // block header touching the end
|
|
||||||
CHECK_EQ(canon::blocks_json(h->root), R"({"a":1})");
|
|
||||||
CHECK_EQ(h->warnings.size(), std::size_t{1});
|
|
||||||
CHECK_EQ(canon_of("a 1 trailing\n"), R"({"a":1})");
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST(unterminated_quote_runs_to_end_of_input) {
|
|
||||||
// The value swallows the rest of the file and then touches EOF -> dropped.
|
|
||||||
auto r = parse_blocks("a \"never closed\n b 1\n");
|
auto r = parse_blocks("a \"never closed\n b 1\n");
|
||||||
CHECK(r.ok());
|
CHECK(!r.ok());
|
||||||
CHECK_EQ(canon::blocks_json(r->root), "{}");
|
CHECK_EQ(r.error().line, 1);
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{2}); // unterminated + dropped
|
Options strict;
|
||||||
CHECK(!parse_blocks("a \"never closed", strict_opts()).ok());
|
strict.strict = true;
|
||||||
// the text is still visible through the tokenizer
|
CHECK(!parse_blocks("a \"never closed", strict).ok());
|
||||||
Script s("a \"x\ny");
|
|
||||||
ScriptToken t;
|
|
||||||
CHECK(s.next(t) == ReadStatus::AtEnd);
|
|
||||||
CHECK(t.value_unterminated);
|
|
||||||
CHECK_EQ(std::string(t.value), "x\ny");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(open_brace_in_key_position_is_a_key) {
|
TEST(open_brace_without_name_is_always_an_error) {
|
||||||
auto r = parse_blocks("a 1\n{ b 2 }\n");
|
auto r = parse_blocks("a 1\n{ b 2 }");
|
||||||
CHECK(r.ok());
|
CHECK(!r.ok());
|
||||||
CHECK_EQ(canon::blocks_json(r->root), R"({"2":"}","a":1,"{":"b"})");
|
CHECK_EQ(r.error().line, 2);
|
||||||
CHECK_EQ(r->warnings.size(), std::size_t{1});
|
// ... also when it follows a complete pair on the same line
|
||||||
CHECK(!parse_blocks("a 1\n{ b 2 }\n", strict_opts()).ok());
|
CHECK(!parse_blocks("a 1 { b 2 }").ok());
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(crlf_and_line_numbers) {
|
TEST(crlf_and_line_numbers) {
|
||||||
|
|
@ -384,7 +217,6 @@ TEST(empty_and_comment_only_input) {
|
||||||
Document e = must_parse(" // nothing here\n\n");
|
Document e = must_parse(" // nothing here\n\n");
|
||||||
CHECK(e.root.entries.empty());
|
CHECK(e.root.entries.empty());
|
||||||
CHECK_EQ(canon_of(""), "{}");
|
CHECK_EQ(canon_of(""), "{}");
|
||||||
CHECK_EQ(canon_of("// no newline at all"), "{}");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(top_level_pairs_and_blocks_mix) {
|
TEST(top_level_pairs_and_blocks_mix) {
|
||||||
|
|
@ -395,7 +227,7 @@ TEST(top_level_pairs_and_blocks_mix) {
|
||||||
|
|
||||||
TEST(numbers_are_typed_only_in_canonical_form_and_only_when_bare) {
|
TEST(numbers_are_typed_only_in_canonical_form_and_only_when_bare) {
|
||||||
// The Node keeps raw text; the canonical form types barewords like the oracle
|
// The Node keeps raw text; the canonical form types barewords like the oracle
|
||||||
std::string s = canon_of("i 7 f .5 e 7e+8 t TRUE q \"8\" n -.8 d 5.\n");
|
std::string s = canon_of("i 7 f .5 e 7e+8 t TRUE q \"8\" n -.8 d 5.");
|
||||||
CHECK_EQ(s, "{\"d\":\"\\u00015\",\"e\":\"\\u0001700000000\",\"f\":\"\\u00010.5\","
|
CHECK_EQ(s, "{\"d\":\"\\u00015\",\"e\":\"\\u0001700000000\",\"f\":\"\\u00010.5\","
|
||||||
"\"i\":7,\"n\":\"\\u0001-0.80000000000000004\",\"q\":\"8\",\"t\":true}");
|
"\"i\":7,\"n\":\"\\u0001-0.80000000000000004\",\"q\":\"8\",\"t\":true}");
|
||||||
}
|
}
|
||||||
|
|
@ -409,7 +241,7 @@ TEST(high_bytes_pass_through) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(deep_nesting_returns_to_the_right_parent) {
|
TEST(deep_nesting_returns_to_the_right_parent) {
|
||||||
Document d = must_parse("a { b { c { d 1 } e 2 } f 3 } g 4\n");
|
Document d = must_parse("a { b { c { d 1 } e 2 } f 3 } g 4");
|
||||||
const Node* a = d.root.first_block("a");
|
const Node* a = d.root.first_block("a");
|
||||||
CHECK_EQ(*a->first_block("b")->first_block("c")->first_value("d"), "1");
|
CHECK_EQ(*a->first_block("b")->first_block("c")->first_value("d"), "1");
|
||||||
CHECK_EQ(*a->first_block("b")->first_value("e"), "2");
|
CHECK_EQ(*a->first_block("b")->first_value("e"), "2");
|
||||||
|
|
@ -419,7 +251,7 @@ TEST(deep_nesting_returns_to_the_right_parent) {
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(lookup_misses_return_null_or_empty) {
|
TEST(lookup_misses_return_null_or_empty) {
|
||||||
Document d = must_parse("a 1\n");
|
Document d = must_parse("a 1");
|
||||||
CHECK(d.root.first("zzz") == nullptr);
|
CHECK(d.root.first("zzz") == nullptr);
|
||||||
CHECK(d.root.first_value("zzz") == nullptr);
|
CHECK(d.root.first_value("zzz") == nullptr);
|
||||||
CHECK(d.root.first_block("a") == nullptr); // a is a pair, not a block
|
CHECK(d.root.first_block("a") == nullptr); // a is a pair, not a block
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,6 @@
|
||||||
#include <iterator>
|
#include <iterator>
|
||||||
#include <map>
|
#include <map>
|
||||||
#include <string>
|
#include <string>
|
||||||
#include <vector>
|
|
||||||
|
|
||||||
#include "mars/text/csv.h"
|
#include "mars/text/csv.h"
|
||||||
#include "mars/text/flat_kv.h"
|
#include "mars/text/flat_kv.h"
|
||||||
|
|
@ -65,21 +64,24 @@ void json_tokens(const std::vector<Token>& toks, std::string& out) {
|
||||||
out.push_back(']');
|
out.push_back(']');
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// kv value: null / scalar / list, matching the reference reader's shape
|
||||||
|
void json_kv_value(const std::vector<Token>& toks, std::string& out) {
|
||||||
|
if (toks.empty()) out += "null";
|
||||||
|
else if (toks.size() == 1) json_token(toks[0], out);
|
||||||
|
else json_tokens(toks, out);
|
||||||
|
}
|
||||||
|
|
||||||
std::string dump_kv(const FlatKV& kv) {
|
std::string dump_kv(const FlatKV& kv) {
|
||||||
// engine rule: keys compared case-insensitively, first occurrence wins,
|
std::map<std::string, const KvEntry*> last; // exact key, last wins
|
||||||
// reported under its first spelling (the reference reader's dict shape)
|
for (const KvEntry& e : kv.entries()) last[e.key] = &e;
|
||||||
std::map<std::string, const KvEntry*> first_seen;
|
|
||||||
std::vector<const KvEntry*> order;
|
|
||||||
for (const KvEntry& e : kv.entries())
|
|
||||||
if (first_seen.emplace(fold_case(e.key), &e).second) order.push_back(&e);
|
|
||||||
std::string out = "{\"values\":{";
|
std::string out = "{\"values\":{";
|
||||||
bool first = true;
|
bool first = true;
|
||||||
for (const KvEntry* e : order) {
|
for (const auto& [key, e] : last) {
|
||||||
if (!first) out.push_back(',');
|
if (!first) out.push_back(',');
|
||||||
first = false;
|
first = false;
|
||||||
json_string(e->key, out);
|
json_string(key, out);
|
||||||
out.push_back(':');
|
out.push_back(':');
|
||||||
json_token(e->value, out);
|
json_kv_value(e->values, out);
|
||||||
}
|
}
|
||||||
out += "},\"duplicates\":{";
|
out += "},\"duplicates\":{";
|
||||||
first = true;
|
first = true;
|
||||||
|
|
@ -112,10 +114,6 @@ const char* problem_kind(Problem::Kind k) {
|
||||||
case Problem::Kind::DeletedAndAssigned: return "deleted_and_assigned";
|
case Problem::Kind::DeletedAndAssigned: return "deleted_and_assigned";
|
||||||
case Problem::Kind::UnbalancedQuote: return "unbalanced_quote";
|
case Problem::Kind::UnbalancedQuote: return "unbalanced_quote";
|
||||||
case Problem::Kind::UnterminatedQuotedCell: return "unterminated_quoted_cell";
|
case Problem::Kind::UnterminatedQuotedCell: return "unterminated_quoted_cell";
|
||||||
case Problem::Kind::DuplicateKey: return "duplicate_key";
|
|
||||||
case Problem::Kind::DroppedTrailingPair: return "dropped_trailing_pair";
|
|
||||||
case Problem::Kind::UnterminatedQuote: return "unterminated_quote";
|
|
||||||
case Problem::Kind::SkippedBlock: return "skipped_block";
|
|
||||||
}
|
}
|
||||||
return "?";
|
return "?";
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -56,21 +56,12 @@ int main() {
|
||||||
auto endgame = r.value.get_color("ENDGAME_FILL_COLOR");
|
auto endgame = r.value.get_color("ENDGAME_FILL_COLOR");
|
||||||
CHECK(endgame && endgame->components == 3);
|
CHECK(endgame && endgame->components == 3);
|
||||||
}
|
}
|
||||||
// StrategyVars.txt: typed access; the file has no trailing newline, so the
|
// StrategyVars.txt: typed access
|
||||||
// engine drops its last pair (CIVILIAN_BURDEN_RATIO keeps the image default)
|
|
||||||
{
|
{
|
||||||
auto r = parse_flat_kv(read_file(base + "Data/Strategy/StrategyVars.txt"));
|
auto r = parse_flat_kv(read_file(base + "Data/Strategy/StrategyVars.txt"));
|
||||||
CHECK(r.value.size() == 96);
|
CHECK(r.ok());
|
||||||
|
CHECK(r.value.size() == 97);
|
||||||
CHECK(r.value.get_float("SLAVES_DEATH_RATE").has_value());
|
CHECK(r.value.get_float("SLAVES_DEATH_RATE").has_value());
|
||||||
CHECK(!r.value.has("CIVILIAN_BURDEN_RATIO"));
|
|
||||||
CHECK(r.problems.size() == 1 && r.problems[0].kind == Problem::Kind::DroppedTrailingPair);
|
|
||||||
}
|
|
||||||
// encounters.txt: same, HERALD_SPEECH_MAX_INVERVAL is never read
|
|
||||||
{
|
|
||||||
auto r = parse_flat_kv(read_file(base + "Data/encounters.txt"));
|
|
||||||
CHECK(!r.value.has("HERALD_SPEECH_MAX_INVERVAL"));
|
|
||||||
CHECK(r.problems.size() == 1 && r.problems[0].kind == Problem::Kind::DroppedTrailingPair);
|
|
||||||
CHECK(r.value.duplicates().empty());
|
|
||||||
}
|
}
|
||||||
// _turrets.txt: 42 rows, 8 columns, quoted model in the last column
|
// _turrets.txt: 42 rows, 8 columns, quoted model in the last column
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -121,13 +121,14 @@ static void test_flat_kv() {
|
||||||
"EMPTY \"\"\n"
|
"EMPTY \"\"\n"
|
||||||
"COLOR \"255\t177\t\t39\"\n"
|
"COLOR \"255\t177\t\t39\"\n"
|
||||||
"FLAG true\n"
|
"FLAG true\n"
|
||||||
|
"NAME_ONLY\n"
|
||||||
|
"LIST 1 2 three\n"
|
||||||
"path \"Data\\\\x.txt\"\n"
|
"path \"Data\\\\x.txt\"\n"
|
||||||
"slaves_death_rate 0.10\n"
|
"SLAVES_DEATH_RATE 0.10\n"
|
||||||
"ENDGAME \"0 0 0\"// \" 92 76 20\"\n"
|
"BADQ \"oops\n";
|
||||||
"LAST 7\n";
|
|
||||||
auto r = parse_flat_kv(text);
|
auto r = parse_flat_kv(text);
|
||||||
const FlatKV& kv = r.value;
|
const FlatKV& kv = r.value;
|
||||||
CHECK_EQ(kv.size(), 10u); // every pair in file order, the duplicate included
|
CHECK_EQ(kv.size(), 11u);
|
||||||
|
|
||||||
// typed access, case-insensitive lookup, original spelling preserved
|
// typed access, case-insensitive lookup, original spelling preserved
|
||||||
CHECK_EQ(kv.get_int("max_ships").value(), 12);
|
CHECK_EQ(kv.get_int("max_ships").value(), 12);
|
||||||
|
|
@ -135,118 +136,49 @@ static void test_flat_kv() {
|
||||||
CHECK_EQ(kv.find("MAX_SHIPS")->line, 4);
|
CHECK_EQ(kv.find("MAX_SHIPS")->line, 4);
|
||||||
CHECK(!kv.get_int("TITLE"));
|
CHECK(!kv.get_int("TITLE"));
|
||||||
CHECK_EQ(kv.get_string("TITLE").value(), std::string_view("Sword // of the Stars"));
|
CHECK_EQ(kv.get_string("TITLE").value(), std::string_view("Sword // of the Stars"));
|
||||||
CHECK(kv.find("TITLE")->value.quoted);
|
CHECK(kv.find("TITLE")->values[0].quoted);
|
||||||
CHECK_EQ(kv.get_string("EMPTY").value(), std::string_view(""));
|
CHECK_EQ(kv.get_string("EMPTY").value(), std::string_view(""));
|
||||||
auto c = kv.get_color("COLOR");
|
auto c = kv.get_color("COLOR");
|
||||||
CHECK(c && c->r == 255 && c->g == 177 && c->b == 39);
|
CHECK(c && c->r == 255 && c->g == 177 && c->b == 39);
|
||||||
CHECK_EQ(kv.get_bool("flag").value(), true);
|
CHECK_EQ(kv.get_bool("flag").value(), true);
|
||||||
|
CHECK(kv.has("NAME_ONLY"));
|
||||||
|
CHECK(kv.find("NAME_ONLY")->values.empty());
|
||||||
|
CHECK(!kv.get_string("NAME_ONLY"));
|
||||||
|
CHECK_EQ(kv.find("LIST")->values.size(), 3u);
|
||||||
|
CHECK(kv.find("LIST")->values[2].kind() == ValueKind::Bareword);
|
||||||
|
CHECK_EQ(kv.get_int("LIST").value(), 1); // first token
|
||||||
CHECK_EQ(kv.get_string("path").value(), std::string_view("Data\\\\x.txt")); // no escape processing
|
CHECK_EQ(kv.get_string("path").value(), std::string_view("Data\\\\x.txt")); // no escape processing
|
||||||
CHECK_EQ(kv.get_string("ENDGAME").value(), std::string_view("0 0 0")); // closing quote glued to //
|
|
||||||
CHECK_EQ(kv.get_int("LAST").value(), 7);
|
|
||||||
CHECK(!kv.has("missing"));
|
CHECK(!kv.has("missing"));
|
||||||
CHECK(!kv.get_int("missing"));
|
CHECK(!kv.get_int("missing"));
|
||||||
|
|
||||||
// duplicates: FIRST occurrence wins (engine erases a key once consumed);
|
// duplicates: last wins for lookup, both entries kept
|
||||||
// the later line is reported and stays visible in entries()/duplicates()
|
CHECK_EQ(kv.get_float("slaves_death_rate").value(), 0.10);
|
||||||
CHECK_EQ(kv.get_float("slaves_death_rate").value(), 0.05);
|
|
||||||
CHECK_EQ(kv.find("SLAVES_DEATH_RATE")->line, 3);
|
|
||||||
auto dups = kv.duplicates();
|
auto dups = kv.duplicates();
|
||||||
CHECK_EQ(dups.size(), 1u);
|
CHECK_EQ(dups.size(), 1u);
|
||||||
CHECK(dups[0].first == "slaves_death_rate");
|
CHECK(dups[0].first == "slaves_death_rate");
|
||||||
CHECK((dups[0].second == std::vector<int>{3, 10}));
|
CHECK((dups[0].second == std::vector<int>{3, 12}));
|
||||||
CHECK_EQ(r.problems.size(), 1u);
|
|
||||||
CHECK(r.problems[0].kind == Problem::Kind::DuplicateKey && r.problems[0].line == 10);
|
|
||||||
|
|
||||||
// key case only differs: the same key to the engine (_stricmp)
|
// unbalanced quote: line still parsed, warning reported
|
||||||
|
CHECK_EQ(r.problems.size(), 1u);
|
||||||
|
CHECK(r.problems[0].kind == Problem::Kind::UnbalancedQuote && r.problems[0].line == 13);
|
||||||
|
CHECK_EQ(kv.get_string("BADQ").value(), std::string_view("\"oops"));
|
||||||
|
|
||||||
|
// key case only differs: a duplicate for us
|
||||||
auto r2 = parse_flat_kv("Foo 1\nFOO 2\n");
|
auto r2 = parse_flat_kv("Foo 1\nFOO 2\n");
|
||||||
CHECK_EQ(r2.value.get_int("foo").value(), 1);
|
CHECK_EQ(r2.value.get_int("foo").value(), 2);
|
||||||
CHECK_EQ(r2.value.duplicates().size(), 1u);
|
CHECK_EQ(r2.value.duplicates().size(), 1u);
|
||||||
CHECK(r2.problems.size() == 1 && r2.problems[0].kind == Problem::Kind::DuplicateKey);
|
|
||||||
|
|
||||||
// empty / comment-only input
|
// empty / comment-only input
|
||||||
CHECK_EQ(parse_flat_kv("").value.size(), 0u);
|
CHECK_EQ(parse_flat_kv("").value.size(), 0u);
|
||||||
CHECK_EQ(parse_flat_kv("// nothing\n\n \n").value.size(), 0u);
|
CHECK_EQ(parse_flat_kv("// nothing\n\n \n").value.size(), 0u);
|
||||||
CHECK(parse_flat_kv("// no newline").ok());
|
// line numbers with mixed terminators
|
||||||
// line numbers count '\n' only (mixed terminators)
|
auto r3 = parse_flat_kv("A 1\rB 2\r\nC 3\nD 4");
|
||||||
auto r3 = parse_flat_kv("A 1\rB 2\r\nC 3\nD 4\n");
|
CHECK_EQ(r3.value.find("D")->line, 4);
|
||||||
CHECK_EQ(r3.value.find("B")->line, 1);
|
|
||||||
CHECK_EQ(r3.value.find("D")->line, 3);
|
|
||||||
// high bytes pass through untouched
|
// high bytes pass through untouched
|
||||||
auto r4 = parse_flat_kv("K \"caf\xe9 \x93q\x94\"\n");
|
auto r4 = parse_flat_kv("K \"caf\xe9 \x93q\x94\"");
|
||||||
CHECK_EQ(r4.value.get_string("K").value(), std::string_view("caf\xe9 \x93q\x94"));
|
CHECK_EQ(r4.value.get_string("K").value(), std::string_view("caf\xe9 \x93q\x94"));
|
||||||
}
|
}
|
||||||
|
|
||||||
// The loader is the engine's token stepper, not a line reader.
|
|
||||||
static void test_flat_kv_engine_steps() {
|
|
||||||
// one value token per key: extra tokens start the next pair
|
|
||||||
auto r = parse_flat_kv("LIST 1 2 three\n");
|
|
||||||
CHECK_EQ(r.value.size(), 2u);
|
|
||||||
CHECK_EQ(r.value.get_int("LIST").value(), 1);
|
|
||||||
CHECK_EQ(r.value.get_string("2").value(), std::string_view("three"));
|
|
||||||
// a key alone on a line takes the next line's key as its value
|
|
||||||
auto k = parse_flat_kv("A\nB 2\nC 3\n");
|
|
||||||
CHECK_EQ(k.value.get_string("A").value(), std::string_view("B"));
|
|
||||||
CHECK_EQ(k.value.get_string("2").value(), std::string_view("C"));
|
|
||||||
CHECK(!k.value.has("B"));
|
|
||||||
CHECK(k.problems.size() == 1 && k.problems[0].kind == Problem::Kind::DroppedTrailingPair); // "3"
|
|
||||||
// braces are not delimiters
|
|
||||||
auto b = parse_flat_kv("A{ 1\n");
|
|
||||||
CHECK(b.value.has("A{") && !b.value.has("A"));
|
|
||||||
|
|
||||||
// the final pair is dropped when its value touches the end of input
|
|
||||||
auto d = parse_flat_kv("A 1\nB 2");
|
|
||||||
CHECK(d.value.has("A") && !d.value.has("B"));
|
|
||||||
CHECK_EQ(d.value.size(), 1u);
|
|
||||||
CHECK(d.problems.size() == 1 && d.problems[0].kind == Problem::Kind::DroppedTrailingPair &&
|
|
||||||
d.problems[0].line == 2);
|
|
||||||
CHECK(parse_flat_kv("A 1\nB \"2\"").problems.size() == 1); // closing quote as last byte: same
|
|
||||||
CHECK(parse_flat_kv("A 1\nB 2 ").value.has("B")); // any trailing whitespace saves it
|
|
||||||
CHECK(parse_flat_kv("A 1\nB 2 // c").value.has("B")); // ... or a comment
|
|
||||||
auto lone = parse_flat_kv("A 1\nB\n");
|
|
||||||
CHECK(!lone.value.has("B") && lone.problems.size() == 1);
|
|
||||||
auto lone2 = parse_flat_kv("A 1\nB");
|
|
||||||
CHECK(!lone2.value.has("B") && lone2.problems.size() == 1);
|
|
||||||
|
|
||||||
// quotes: ", ' and backtick, no escapes; a quote inside a word is literal
|
|
||||||
auto q = parse_flat_kv("A 'x y'\nB `p q`\nC it's\nD \"say \\\"hi\" 5\n");
|
|
||||||
CHECK_EQ(q.value.get_string("A").value(), std::string_view("x y"));
|
|
||||||
CHECK(q.value.find("A")->value.quoted);
|
|
||||||
CHECK_EQ(q.value.get_string("B").value(), std::string_view("p q"));
|
|
||||||
CHECK_EQ(q.value.get_string("C").value(), std::string_view("it's"));
|
|
||||||
CHECK_EQ(q.value.get_string("D").value(), std::string_view("say \\")); // backslash does not escape
|
|
||||||
CHECK_EQ(q.value.get_int("hi\"").value(), 5); // the rest is a new key
|
|
||||||
CHECK(q.ok());
|
|
||||||
// an unterminated quote runs to the end of input (and so the pair is dropped)
|
|
||||||
auto u = parse_flat_kv("A \"never\nB 2\n");
|
|
||||||
CHECK_EQ(u.value.size(), 0u);
|
|
||||||
CHECK_EQ(u.problems.size(), 2u);
|
|
||||||
CHECK(u.problems[0].kind == Problem::Kind::UnterminatedQuote);
|
|
||||||
CHECK(u.problems[1].kind == Problem::Kind::DroppedTrailingPair);
|
|
||||||
|
|
||||||
// comments: the test is on the extracted token
|
|
||||||
auto c = parse_flat_kv("A 3//x\n\"// whole line\" junk\nB 2\n");
|
|
||||||
CHECK_EQ(c.value.get_string("A").value(), std::string_view("3//x"));
|
|
||||||
CHECK(c.value.find("A")->value.kind() == ValueKind::Bareword);
|
|
||||||
CHECK_EQ(c.value.get_int("B").value(), 2);
|
|
||||||
CHECK_EQ(c.value.size(), 2u);
|
|
||||||
|
|
||||||
// a block is skipped; a stray } is ignored
|
|
||||||
auto blk = parse_flat_kv("A 1\nblk {\n X 2\n sub { Y 3 }\n}\n}\nB 4\n");
|
|
||||||
CHECK_EQ(blk.value.size(), 2u);
|
|
||||||
CHECK(blk.value.has("A") && blk.value.has("B") && !blk.value.has("X"));
|
|
||||||
CHECK(blk.problems.size() == 1 && blk.problems[0].kind == Problem::Kind::SkippedBlock);
|
|
||||||
|
|
||||||
// tokens are capped at 1023 bytes
|
|
||||||
std::string longkey(1100, 'k');
|
|
||||||
auto cap = parse_flat_kv(longkey + " 1\n");
|
|
||||||
CHECK_EQ(cap.value.entries()[0].key.size(), 1023u);
|
|
||||||
CHECK(cap.value.has(longkey.substr(0, 1023)));
|
|
||||||
|
|
||||||
// whitespace is space, tab, CR, LF only
|
|
||||||
auto ws = parse_flat_kv("A\v1 2\n");
|
|
||||||
CHECK(ws.value.has("A\v1"));
|
|
||||||
}
|
|
||||||
|
|
||||||
static void test_rows() {
|
static void test_rows() {
|
||||||
const char* text =
|
const char* text =
|
||||||
"// size weapon-size class health \"model\"\r\n"
|
"// size weapon-size class health \"model\"\r\n"
|
||||||
|
|
@ -397,7 +329,6 @@ int main() {
|
||||||
test_color();
|
test_color();
|
||||||
test_strip_comment_and_tokens();
|
test_strip_comment_and_tokens();
|
||||||
test_flat_kv();
|
test_flat_kv();
|
||||||
test_flat_kv_engine_steps();
|
|
||||||
test_rows();
|
test_rows();
|
||||||
test_manifest();
|
test_manifest();
|
||||||
test_csv_records();
|
test_csv_records();
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue