Parser rewrite
This commit is contained in:
@@ -20,13 +20,10 @@ public class ExtractionBenchmarks
|
|||||||
private byte[] _json = null!;
|
private byte[] _json = null!;
|
||||||
private ISearchExpression _query = null!;
|
private ISearchExpression _query = null!;
|
||||||
|
|
||||||
private Regex _flatInterpreted = null!;
|
private Regex _technologyBlock = null!;
|
||||||
private Regex _flatCompiled = null!;
|
|
||||||
private Regex _flatSourceGen = null!;
|
|
||||||
private Regex _pathAwareCompiled = null!;
|
private Regex _pathAwareCompiled = null!;
|
||||||
private Regex _pathAwareNonBacktracking = null!;
|
private Regex _pathAwareNonBacktracking = null!;
|
||||||
private Regex _countriesBlock = null!;
|
private Regex _countriesBlock = null!;
|
||||||
private PcreRegex _pcreFlat = null!;
|
|
||||||
private PcreRegex _pcrePathAware = null!;
|
private PcreRegex _pcrePathAware = null!;
|
||||||
|
|
||||||
[GlobalSetup]
|
[GlobalSetup]
|
||||||
@@ -37,13 +34,10 @@ public class ExtractionBenchmarks
|
|||||||
_json = File.ReadAllBytes(BenchData.JsonPath);
|
_json = File.ReadAllBytes(BenchData.JsonPath);
|
||||||
_query = SearchExpressionCompiler.Compile(BenchData.PdxQuery);
|
_query = SearchExpressionCompiler.Compile(BenchData.PdxQuery);
|
||||||
|
|
||||||
_flatInterpreted = new Regex(Extractors.FlatPattern, RegexOptions.None);
|
_technologyBlock = new Regex(Extractors.TechnologyBlockPattern, RegexOptions.Compiled);
|
||||||
_flatCompiled = new Regex(Extractors.FlatPattern, RegexOptions.Compiled);
|
|
||||||
_flatSourceGen = Extractors.FlatSourceGen();
|
|
||||||
_pathAwareCompiled = new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled);
|
_pathAwareCompiled = new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled);
|
||||||
_pathAwareNonBacktracking = new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking);
|
_pathAwareNonBacktracking = new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking);
|
||||||
_countriesBlock = new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled);
|
_countriesBlock = new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled);
|
||||||
_pcreFlat = new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled);
|
|
||||||
_pcrePathAware = new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled);
|
_pcrePathAware = new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -53,26 +47,14 @@ public class ExtractionBenchmarks
|
|||||||
[Benchmark(Description = "Full parse, then select")]
|
[Benchmark(Description = "Full parse, then select")]
|
||||||
public long FullParse() => Extractors.FullParseThenSelect(_pdx);
|
public long FullParse() => Extractors.FullParseThenSelect(_pdx);
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex flat, interpreted")]
|
|
||||||
public long RegexFlatInterpreted() => Extractors.RegexScan(_flatInterpreted, _pdxText);
|
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex flat, compiled")]
|
|
||||||
public long RegexFlatCompiled() => Extractors.RegexScan(_flatCompiled, _pdxText);
|
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex flat, source generated")]
|
|
||||||
public long RegexFlatSourceGen() => Extractors.RegexScan(_flatSourceGen, _pdxText);
|
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex path aware, compiled")]
|
[Benchmark(Description = ".NET Regex path aware, compiled")]
|
||||||
public long RegexPathAwareCompiled() => Extractors.RegexScan(_pathAwareCompiled, _pdxText);
|
public long RegexPathAwareCompiled() => Extractors.RegexScan(_pathAwareCompiled, _pdxText);
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex path aware, NonBacktracking")]
|
[Benchmark(Description = ".NET Regex path aware, NonBacktracking")]
|
||||||
public long RegexPathAwareNonBacktracking() => Extractors.RegexScan(_pathAwareNonBacktracking, _pdxText);
|
public long RegexPathAwareNonBacktracking() => Extractors.RegexScan(_pathAwareNonBacktracking, _pdxText);
|
||||||
|
|
||||||
[Benchmark(Description = ".NET Regex balanced block + flat")]
|
[Benchmark(Description = ".NET Regex balanced block + inner scan")]
|
||||||
public long RegexBalancedTwoStage() => Extractors.RegexTwoStage(_countriesBlock, _flatCompiled, _pdxText);
|
public long RegexBalancedTwoStage() => Extractors.RegexTwoStage(_countriesBlock, _technologyBlock, _pdxText);
|
||||||
|
|
||||||
[Benchmark(Description = "PCRE.NET flat, JIT compiled")]
|
|
||||||
public long PcreFlat() => Extractors.PcreScan(_pcreFlat, _pdxText);
|
|
||||||
|
|
||||||
[Benchmark(Description = "PCRE.NET path aware, JIT compiled")]
|
[Benchmark(Description = "PCRE.NET path aware, JIT compiled")]
|
||||||
public long PcrePathAware() => Extractors.PcreScan(_pcrePathAware, _pdxText);
|
public long PcrePathAware() => Extractors.PcreScan(_pcrePathAware, _pdxText);
|
||||||
|
|||||||
@@ -65,10 +65,11 @@ public static partial class Extractors
|
|||||||
// ---------------------------------------------------------------- .NET Regex
|
// ---------------------------------------------------------------- .NET Regex
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Flat scan: matches technology blocks anywhere in the file.
|
/// Matches a single technology block. On its own it would match anywhere in the file,
|
||||||
/// Fastest regex option, but it does not verify that the match really sits under <c>countries</c>.
|
/// so it is only used as the second stage of <see cref="RegexTwoStage" />, inside the
|
||||||
|
/// already extracted <c>countries</c> block.
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public const string FlatPattern =
|
public const string TechnologyBlockPattern =
|
||||||
@"\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}";
|
@"\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}";
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
@@ -109,9 +110,6 @@ public static partial class Extractors
|
|||||||
return RegexScan(innerRegex, block.Value);
|
return RegexScan(innerRegex, block.Value);
|
||||||
}
|
}
|
||||||
|
|
||||||
[GeneratedRegex(FlatPattern)]
|
|
||||||
public static partial Regex FlatSourceGen();
|
|
||||||
|
|
||||||
[GeneratedRegex(PathAwarePattern)]
|
[GeneratedRegex(PathAwarePattern)]
|
||||||
public static partial Regex PathAwareSourceGen();
|
public static partial Regex PathAwareSourceGen();
|
||||||
|
|
||||||
|
|||||||
@@ -10,7 +10,7 @@
|
|||||||
</PropertyGroup>
|
</PropertyGroup>
|
||||||
|
|
||||||
<ItemGroup>
|
<ItemGroup>
|
||||||
<PackageReference Include="BenchmarkDotNet" Version="0.15.4" />
|
<PackageReference Include="BenchmarkDotNet" Version="0.15.8" />
|
||||||
<PackageReference Include="PCRE.NET" Version="1.6.0" />
|
<PackageReference Include="PCRE.NET" Version="1.6.0" />
|
||||||
</ItemGroup>
|
</ItemGroup>
|
||||||
|
|
||||||
|
|||||||
@@ -43,16 +43,13 @@ public static class Program
|
|||||||
|
|
||||||
Time("SearchExpression", () => Extractors.SearchExpression(pdx, query));
|
Time("SearchExpression", () => Extractors.SearchExpression(pdx, query));
|
||||||
Time("FullParse", () => Extractors.FullParseThenSelect(pdx));
|
Time("FullParse", () => Extractors.FullParseThenSelect(pdx));
|
||||||
Time("Regex flat compiled",
|
|
||||||
() => Extractors.RegexScan(new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text));
|
|
||||||
Time("Regex path aware compiled",
|
Time("Regex path aware compiled",
|
||||||
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled), text));
|
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled), text));
|
||||||
Time("Regex path aware nonbacktracking",
|
Time("Regex path aware nonbacktracking",
|
||||||
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking), text));
|
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking), text));
|
||||||
Time("Regex balanced two stage",
|
Time("Regex balanced two stage",
|
||||||
() => Extractors.RegexTwoStage(new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled),
|
() => Extractors.RegexTwoStage(new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled),
|
||||||
new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text));
|
new Regex(Extractors.TechnologyBlockPattern, RegexOptions.Compiled), text));
|
||||||
Time("PCRE flat", () => Extractors.PcreScan(new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled), text));
|
|
||||||
Time("PCRE path aware",
|
Time("PCRE path aware",
|
||||||
() => Extractors.PcreScan(new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled), text));
|
() => Extractors.PcreScan(new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled), text));
|
||||||
Time("Utf8JsonReader", () => Extractors.Utf8JsonReaderScan(json));
|
Time("Utf8JsonReader", () => Extractors.Utf8JsonReaderScan(json));
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ regex engines and `jq` on one realistic task:
|
|||||||
|---|---|
|
|---|---|
|
||||||
| this repo | `countries.*.technology` |
|
| this repo | `countries.*.technology` |
|
||||||
| jq | `.countries \| map_values(.technology)` |
|
| jq | `.countries \| map_values(.technology)` |
|
||||||
| .NET `Regex` / PCRE.NET | see `Extractors.FlatPattern` / `Extractors.PathAwarePattern` |
|
| .NET `Regex` / PCRE.NET | see `Extractors.PathAwarePattern` / `Extractors.CountriesBlockPattern` |
|
||||||
| `Utf8JsonReader` / `JsonDocument` | hand written navigation |
|
| `Utf8JsonReader` / `JsonDocument` | hand written navigation |
|
||||||
|
|
||||||
## Corpus
|
## Corpus
|
||||||
@@ -48,64 +48,70 @@ shows immediately when an engine finds something different from the others.
|
|||||||
|
|
||||||
## Results
|
## Results
|
||||||
|
|
||||||
AMD Ryzen 7 5700X, 16 logical cores, Windows 10 21H2, .NET 10.0.12, jq 1.8.2,
|
AMD Ryzen 7 5700X, 8 physical cores (16 logical), Windows 10 21H2, .NET 10.0.12, jq 1.8.2,
|
||||||
109 MB save / 92 MB JSON twin, BenchmarkDotNet `RunStrategy.Monitoring`, 5 iterations.
|
109 MB save / 92 MB JSON twin, BenchmarkDotNet `RunStrategy.Monitoring`,
|
||||||
|
3 warmups and 10 iterations for the in-process engines, 5 for jq.
|
||||||
|
|
||||||
| engine | mean | vs this project | allocated | correct? |
|
| engine | mean | vs this project | allocated | correct? |
|
||||||
|---|---:|---:|---:|---|
|
|---|---:|---:|---:|---|
|
||||||
| .NET Regex flat, source generated | 14.8 ms | 0.02x | 1.2 MB | path-blind |
|
| **SearchExpression (this project)** | **49.6 ms** | **1.00x** | **1.46 MB** | yes |
|
||||||
| .NET Regex flat, compiled | 17.0 ms | 0.03x | 1.2 MB | path-blind |
|
| .NET Regex balanced block + inner scan | 114.0 ms | 2.30x | 213 MB | yes |
|
||||||
| .NET Regex flat, interpreted | 19.3 ms | 0.03x | 1.2 MB | path-blind |
|
| PCRE.NET path aware, JIT compiled | 123.4 ms | 2.49x | 0.90 MB | 1869/1870 |
|
||||||
| PCRE.NET flat, JIT compiled | 97.4 ms | 0.15x | 0.9 MB | path-blind |
|
| .NET Regex path aware, compiled | 135.1 ms | 2.72x | 1.34 MB | 1869/1870 |
|
||||||
| .NET Regex balanced block + flat | 112.3 ms | 0.17x | 224 MB | yes |
|
| Utf8JsonReader over JSON twin | 155.5 ms | 3.13x | 16 KB | yes |
|
||||||
| .NET Regex path aware, compiled | 116.2 ms | 0.18x | 1.4 MB | 1869/1870 |
|
| JsonDocument over JSON twin | 323.2 ms | 6.51x | 72 B (native) | yes |
|
||||||
| PCRE.NET path aware, JIT compiled | 122.9 ms | 0.19x | 0.9 MB | 1869/1870 |
|
| .NET Regex path aware, NonBacktracking | 440.5 ms | 8.87x | 30.6 MB | 1869/1870 |
|
||||||
| Utf8JsonReader over JSON twin | 155.7 ms | 0.24x | 0 B | yes |
|
| Full parse, then select | 1204 ms | 24.3x | 962 MB | yes |
|
||||||
| JsonDocument over JSON twin | 321.4 ms | 0.50x | 0 B (native) | yes |
|
| jq `empty` (parse the file, emit nothing) | 4188 ms | 84.4x | n/a | n/a |
|
||||||
| .NET Regex path aware, NonBacktracking | 444.5 ms | 0.69x | 32 MB | 1869/1870 |
|
| jq `.countries \| map_values(.technology)` | 4339 ms | 87.4x | n/a | yes |
|
||||||
| **SearchExpression (this project)** | **645.1 ms** | **1.00x** | **1.6 MB** | yes |
|
| jq `--stream` | 17497 ms | 352x | n/a | n/a |
|
||||||
| Full parse, then select | 1831 ms | 2.84x | 1153 MB | yes |
|
| jq process startup only | 4.1 ms | 0.08x | n/a | n/a |
|
||||||
| jq `.countries \| map_values(.technology)` | 4340 ms | 6.73x | n/a | yes |
|
|
||||||
| jq `empty` (parse the file, emit nothing) | 4227 ms | 6.55x | n/a | n/a |
|
Reading the save from disk instead of a preloaded `byte[]` costs the parser ~9 ms more:
|
||||||
| jq `--stream` | 17490 ms | 27.1x | n/a | n/a |
|
58.4 ms via `FileStream` (`ParserInputBenchmarks`).
|
||||||
| jq process startup only | 4.2 ms | 0.01x | n/a | n/a |
|
|
||||||
|
|
||||||
### Reading the table
|
### Reading the table
|
||||||
|
|
||||||
* **jq is ~6.7x slower than this parser** and 97% of that time is JSON parsing, not the
|
* **This parser is the fastest engine measured here.** It is 2.3x faster than the only
|
||||||
query: `jq empty` on the same file costs 4227 ms of the 4340 ms. Process startup is
|
regex approach that enforces the path, 3.1x faster than a hand written `Utf8JsonReader`
|
||||||
negligible (4 ms). jq's `--stream` mode, often recommended for large inputs, is 4x
|
over the equivalent JSON, 6.5x faster than `JsonDocument`, and 87x faster than jq — while
|
||||||
*slower* still. jq also needs the data converted to JSON first, which this parser has to
|
allocating 1.46 MB for a 109 MB input.
|
||||||
do anyway — so end to end jq is strictly more expensive here.
|
* **jq is ~87x slower and 97% of that is JSON parsing, not the query**: `jq empty` on the
|
||||||
* **The regexes are 5-40x faster, but they are not doing the same job.** A regex never
|
same file costs 4188 ms of the 4339 ms. Process startup is negligible (4 ms). jq's
|
||||||
parses the structure; it scans bytes for a literal and validates a short window around it.
|
`--stream` mode, often recommended for large inputs, is 4x *slower* still. jq also cannot
|
||||||
`Regex` with a literal prefix (`technology={`) is vectorized, so 109 MB is scanned at
|
read the Paradox format, so it first needs the save converted to JSON — a conversion this
|
||||||
several GB/s. That speed is real and the cost is real too: nothing verifies that the hit
|
parser has to perform anyway.
|
||||||
is under `countries`, at the right depth, or belongs to the tag matched before it.
|
* **Only one regex approach here is actually correct**, and it is the expensive one: cut the
|
||||||
* **Path-aware regexes lose both the speed and the correctness.** Forcing the tag into the
|
`countries={...}` block out with a balancing-group pattern (a .NET-only feature; PCRE would
|
||||||
pattern costs 7x (17 ms -> 116 ms) and still returns 1869 of 1870 countries: the tag class
|
need recursion), then scan inside it. It allocates 213 MB, because stage one materialises
|
||||||
`[A-Z0-9]{3}` silently drops EU4's `---` pseudo-country. Pairing survives here only by
|
the whole block as a string.
|
||||||
luck of the file layout — all 933 tech-less country blocks happen to be grouped at the end
|
* **The cheaper path-aware pattern pairs a country tag with the next technology block over a
|
||||||
of the save, so the lazy gap never runs across one. A save that interleaves them would
|
lazy gap, and silently gets it wrong**: 1869 of 1870 countries, because the tag class
|
||||||
make the regex report a technology block under the wrong country tag with no error.
|
`[A-Z0-9]{3}` drops EU4's `---` pseudo-country. Even that much only holds by luck of the
|
||||||
* **`RegexOptions.NonBacktracking`** guarantees linear time but is 26x slower than the
|
file layout — all 933 tech-less country blocks happen to be grouped at the end of the save,
|
||||||
compiled backtracking engine on this pattern and allocates 32 MB.
|
so the lazy gap never runs across one. A save that interleaved them would report a
|
||||||
* **PCRE.NET** (the most used non-BCL regex engine in .NET, ~460k downloads) is 6x slower
|
technology block under the wrong country tag, with no error.
|
||||||
than `System.Text.RegularExpressions` on the flat pattern, because .NET's vectorized
|
* **`RegexOptions.NonBacktracking`** guarantees linear time but is 3.3x slower than the
|
||||||
literal prefix search beats PCRE2's JIT here. On the path-aware pattern the two are equal.
|
compiled backtracking engine on this pattern and allocates 31 MB.
|
||||||
There is no reason to leave the BCL engine for this workload.
|
* **PCRE.NET** (the most used non-BCL regex engine in .NET, ~460k downloads) is within ~10%
|
||||||
* **The query is what makes this parser fast, not the parsing.** Same parser, same file:
|
of `System.Text.RegularExpressions` on the path-aware pattern (123 ms vs 135 ms). Nothing
|
||||||
645 ms with `countries.*.technology`, 1831 ms and 1.1 GB allocated without a query. The
|
here justifies leaving the BCL engine.
|
||||||
search expression prunes ~65% of the work and 99.9% of the allocations.
|
* **The query is what makes this parser fast.** Same parser, same file: 50 ms with
|
||||||
* **Against a JSON reader on equivalent data**, the parser is 4x slower than `Utf8JsonReader`
|
`countries.*.technology`, 1204 ms and 962 MB allocated without a query — 24x and 660x.
|
||||||
and 2x slower than `JsonDocument`. Those numbers exclude the pdx -> JSON conversion
|
|
||||||
(~4.7 s), so they are a ceiling for the format, not a usable alternative.
|
|
||||||
|
|
||||||
### Where this parser's time goes
|
### How the parser got here
|
||||||
|
|
||||||
645 ms for 109 MB is ~170 MB/s, or ~13 cycles per input byte. The lexer reads the save one
|
Three measurements on the same benchmark (`ParserInputBenchmarks`, reading a `FileStream`):
|
||||||
byte at a time through `Stream.ReadByte()` (`SaveParserEU4.LexTextSave`), which is a virtual
|
|
||||||
call plus bounds check per byte, and appends char by char into a `StringBuilder`. Reading
|
| lexer | mean | allocated |
|
||||||
into a `byte[]` buffer and scanning it with `ReadOnlySpan<byte>.IndexOfAny` would be the
|
|---|---:|---:|
|
||||||
first thing to try — the regex numbers above show what the same hardware does when it scans
|
| `Stream.ReadByte()` per byte | 834 ms | 1.55 MB |
|
||||||
a span instead of a stream.
|
| copy the file into a `MemoryStream` first, still `ReadByte()` | 738 ms | 257 MB |
|
||||||
|
| 64 KB buffer + `Tokenizer` with byte-level block skipping | 58 ms | 1.46 MB |
|
||||||
|
|
||||||
|
The first jump came from removing a virtual call per byte. The large one came from making
|
||||||
|
the query prune *work* rather than just data: `Tokenizer.SkipBlock` throws away a rejected
|
||||||
|
`{...}` block by scanning raw bytes for brace depth with `SearchValues<byte>.IndexOfAny`,
|
||||||
|
so blocks the search expression rejects are never turned into tokens or strings at all.
|
||||||
|
Retained tokens are spans into the read buffer, decoded only when their value is kept, and
|
||||||
|
numbers are parsed straight from UTF8.
|
||||||
|
|||||||
@@ -1,4 +1,6 @@
|
|||||||
|
using System.Collections.Generic;
|
||||||
using System.IO;
|
using System.IO;
|
||||||
|
using System.Text;
|
||||||
using System.Text.Encodings.Web;
|
using System.Text.Encodings.Web;
|
||||||
using System.Text.Json;
|
using System.Text.Json;
|
||||||
using System.Text.Json.Serialization;
|
using System.Text.Json.Serialization;
|
||||||
@@ -41,12 +43,64 @@ public class SearchExpressionTests
|
|||||||
[TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")]
|
[TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")]
|
||||||
public void TestSearchOnSmallData(string input, string expectedOutput)
|
public void TestSearchOnSmallData(string input, string expectedOutput)
|
||||||
{
|
{
|
||||||
using var saveStream = new MemoryStream(_smallSaveData, false);
|
Assert.That(Search(_smallSaveData, input), Is.EqualTo(expectedOutput));
|
||||||
var se = SearchExpressionCompiler.Compile(input);
|
}
|
||||||
var parser = new SaveParserEU4(saveStream, se);
|
|
||||||
|
/// <summary>
|
||||||
|
/// The same queries, but with a read buffer so small that tokens, quoted strings and
|
||||||
|
/// skipped blocks are split across refills.
|
||||||
|
/// </summary>
|
||||||
|
[TestCase("a", "a={ b={ c=0 d=1 e=2 } f=3 }")]
|
||||||
|
[TestCase("a.b", "a={ b={ c=0 d=1 e=2 } }")]
|
||||||
|
[TestCase("a.[1]", "a={ f=3 }")]
|
||||||
|
[TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")]
|
||||||
|
public void TestSearchAcrossBufferRefills(string input, string expectedOutput)
|
||||||
|
{
|
||||||
|
foreach (int bufferSize in new[] { 16, 17, 23, 64 })
|
||||||
|
Assert.That(Search(_smallSaveData, input, bufferSize), Is.EqualTo(expectedOutput),
|
||||||
|
$"buffer size {bufferSize}");
|
||||||
|
}
|
||||||
|
|
||||||
|
[Test]
|
||||||
|
public void BracesInsideQuotedStringAreText()
|
||||||
|
{
|
||||||
|
byte[] data = "EU4txt a={ name=\"x{y}z\" b=1 }".ToBytes();
|
||||||
|
using var saveStream = new MemoryStream(data, false);
|
||||||
|
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("a"));
|
||||||
|
var a = (Dictionary<string, object>)parser.Parse()["a"];
|
||||||
|
Assert.Multiple(() =>
|
||||||
|
{
|
||||||
|
Assert.That(a["name"], Is.EqualTo("x{y}z"));
|
||||||
|
Assert.That(a["b"], Is.EqualTo(1L));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
[Test]
|
||||||
|
public void SkippedBlockIgnoresBracesInsideQuotedStrings()
|
||||||
|
{
|
||||||
|
byte[] data = "EU4txt a={ s=\"{{{\" } b=2".ToBytes();
|
||||||
|
using var saveStream = new MemoryStream(data, false);
|
||||||
|
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("b"));
|
||||||
|
Assert.That(parser.Parse()["b"], Is.EqualTo(2L));
|
||||||
|
}
|
||||||
|
|
||||||
|
[Test]
|
||||||
|
public void EncodingIsConfigurable()
|
||||||
|
{
|
||||||
|
byte[] data = Encoding.UTF8.GetBytes("EU4txt a={ name=\"Ä\" }");
|
||||||
|
using var saveStream = new MemoryStream(data, false);
|
||||||
|
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("a"), Encoding.UTF8);
|
||||||
|
var a = (Dictionary<string, object>)parser.Parse()["a"];
|
||||||
|
Assert.That(a["name"], Is.EqualTo("Ä"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static string Search(byte[] saveData, string query, int bufferSize = 64 * 1024)
|
||||||
|
{
|
||||||
|
using var saveStream = new MemoryStream(saveData, false);
|
||||||
|
var se = SearchExpressionCompiler.Compile(query);
|
||||||
|
var parser = new SaveParserEU4(saveStream, se, bufferSize: bufferSize);
|
||||||
var rootNode = parser.Parse();
|
var rootNode = parser.Parse();
|
||||||
string json = JsonSerializer.Serialize(rootNode, _smallSaveSerializerOptions);
|
string json = JsonSerializer.Serialize(rootNode, _smallSaveSerializerOptions);
|
||||||
string pdx = JsonToPdx(json);
|
return JsonToPdx(json);
|
||||||
Assert.That(pdx, Is.EqualTo(expectedOutput));
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1,121 +0,0 @@
|
|||||||
using System.Collections;
|
|
||||||
|
|
||||||
namespace ParadoxSaveParser.Lib;
|
|
||||||
|
|
||||||
/// <summary>
|
|
||||||
/// Enumerator wrapper that stores <c>N/2</c> items before and <c>N/2-1</c> after <c>Current</c> item.
|
|
||||||
/// </summary>
|
|
||||||
/// <code language="cs">
|
|
||||||
/// IEnumerator<int> Enumerator()
|
|
||||||
/// {
|
|
||||||
/// for(int i = 0; i < 6; i++)
|
|
||||||
/// yield return i;
|
|
||||||
/// }
|
|
||||||
///
|
|
||||||
/// var en = Enumerator();
|
|
||||||
/// var bufen = new BufferedEnumerator<int>(en, 5);
|
|
||||||
///
|
|
||||||
/// while(bufen.MoveNext())
|
|
||||||
/// {
|
|
||||||
/// var cur = bufen.Current;
|
|
||||||
/// for (var prev = cur.List?.First; prev != cur; prev = prev?.Next)
|
|
||||||
/// Console.Write($"{prev?.Value} ");
|
|
||||||
///
|
|
||||||
/// Console.Write($"| {cur.Value} |");
|
|
||||||
///
|
|
||||||
/// for (var next = cur.Next; next is not null; next = next.Next)
|
|
||||||
/// Console.Write($" {next.Value}");
|
|
||||||
/// Console.WriteLine();
|
|
||||||
/// }
|
|
||||||
/// </code>
|
|
||||||
/// Output:
|
|
||||||
/// <code>
|
|
||||||
/// | 0 | 1 2 3 4
|
|
||||||
/// 0 | 1 | 2 3 4
|
|
||||||
/// 0 1 | 2 | 3 4
|
|
||||||
/// 1 2 | 3 | 4 5
|
|
||||||
/// 2 3 | 4 | 5
|
|
||||||
/// 3 4 | 5 |
|
|
||||||
/// </code>
|
|
||||||
public class BufferedEnumerator<T> : IEnumerator<BufferedEnumerator<T>.Node>
|
|
||||||
{
|
|
||||||
public class Node
|
|
||||||
{
|
|
||||||
#nullable disable
|
|
||||||
public Node Previous;
|
|
||||||
public Node Next;
|
|
||||||
public T Value;
|
|
||||||
#nullable enable
|
|
||||||
}
|
|
||||||
|
|
||||||
private readonly IEnumerator<T> _enumerator;
|
|
||||||
private readonly Node[] _ringBuffer;
|
|
||||||
private Node? _currentNode;
|
|
||||||
private int _currentBufferIndex = -1;
|
|
||||||
private int _lastValueIndex = -1;
|
|
||||||
|
|
||||||
public BufferedEnumerator(IEnumerator<T> enumerator, int bufferSize)
|
|
||||||
{
|
|
||||||
_enumerator = enumerator;
|
|
||||||
_ringBuffer = new Node[bufferSize];
|
|
||||||
}
|
|
||||||
|
|
||||||
private void InitBuffer()
|
|
||||||
{
|
|
||||||
_ringBuffer[0] = new Node
|
|
||||||
{
|
|
||||||
Value = default!
|
|
||||||
};
|
|
||||||
for (int i = 1; i < _ringBuffer.Length; i++)
|
|
||||||
{
|
|
||||||
_ringBuffer[i] = new Node
|
|
||||||
{
|
|
||||||
Previous = _ringBuffer[i - 1],
|
|
||||||
Value = default!,
|
|
||||||
};
|
|
||||||
_ringBuffer[i - 1].Next = _ringBuffer[i];
|
|
||||||
}
|
|
||||||
_ringBuffer[^1].Next = _ringBuffer[0];
|
|
||||||
_ringBuffer[0].Previous = _ringBuffer[^1];
|
|
||||||
}
|
|
||||||
|
|
||||||
public bool MoveNext()
|
|
||||||
{
|
|
||||||
if (_currentBufferIndex == -1)
|
|
||||||
{
|
|
||||||
InitBuffer();
|
|
||||||
|
|
||||||
int beforeMidpoint = _ringBuffer.Length / 2 - 1;
|
|
||||||
for (int i = 0; i <= beforeMidpoint && _enumerator.MoveNext(); i++)
|
|
||||||
{
|
|
||||||
_ringBuffer[i].Value = _enumerator.Current;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_currentBufferIndex = (_currentBufferIndex + 1) % _ringBuffer.Length;
|
|
||||||
if (_enumerator.MoveNext())
|
|
||||||
{
|
|
||||||
int midpoint = (_currentBufferIndex + _ringBuffer.Length / 2) % _ringBuffer.Length;
|
|
||||||
_ringBuffer[midpoint].Value = _enumerator.Current;
|
|
||||||
_lastValueIndex = midpoint;
|
|
||||||
}
|
|
||||||
if(_currentBufferIndex == (_lastValueIndex + 1) % _ringBuffer.Length)
|
|
||||||
return false;
|
|
||||||
|
|
||||||
_currentNode = _ringBuffer[_currentBufferIndex];
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
public void Reset()
|
|
||||||
{
|
|
||||||
throw new NotImplementedException();
|
|
||||||
}
|
|
||||||
|
|
||||||
public Node Current => _currentNode!;
|
|
||||||
|
|
||||||
object IEnumerator.Current => Current;
|
|
||||||
|
|
||||||
public void Dispose()
|
|
||||||
{
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -8,6 +8,6 @@
|
|||||||
</PropertyGroup>
|
</PropertyGroup>
|
||||||
|
|
||||||
<ItemGroup>
|
<ItemGroup>
|
||||||
<PackageReference Include="Microsoft.Extensions.ObjectPool" Version="10.0.12" />
|
<InternalsVisibleTo Include="ParadoxSaveParser.Lib.Tests" />
|
||||||
</ItemGroup>
|
</ItemGroup>
|
||||||
</Project>
|
</Project>
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ global using System;
|
|||||||
global using System.Collections.Generic;
|
global using System.Collections.Generic;
|
||||||
global using System.IO;
|
global using System.IO;
|
||||||
global using System.Text;
|
global using System.Text;
|
||||||
using Microsoft.Extensions.ObjectPool;
|
using System.Globalization;
|
||||||
|
|
||||||
namespace ParadoxSaveParser.Lib;
|
namespace ParadoxSaveParser.Lib;
|
||||||
|
|
||||||
@@ -11,11 +11,19 @@ namespace ParadoxSaveParser.Lib;
|
|||||||
/// </summary>
|
/// </summary>
|
||||||
public class SaveParserEU4
|
public class SaveParserEU4
|
||||||
{
|
{
|
||||||
protected readonly Stream _saveFile;
|
private const int DefaultBufferSize = 64 * 1024;
|
||||||
private readonly BufferedEnumerator<Token> _tokens;
|
private static ReadOnlySpan<byte> Header => "EU4txt"u8;
|
||||||
private readonly ObjectPool<StringBuilder> _stringBuilderPool;
|
|
||||||
|
private readonly Tokenizer _tokens;
|
||||||
private ISearchExpression? _searchExprCurrent;
|
private ISearchExpression? _searchExprCurrent;
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Encoding of the strings inside the save. Saves of different localizations use
|
||||||
|
/// different ones, so it can be changed; <see cref="System.Text.Encoding.Latin1" /> maps
|
||||||
|
/// every byte to one character and never fails, which makes it a safe default.
|
||||||
|
/// </summary>
|
||||||
|
public Encoding Encoding { get; }
|
||||||
|
|
||||||
/// <param name="savefile">
|
/// <param name="savefile">
|
||||||
/// Uncompressed stream of <c>gamestate</c> file which can be extracted from save archive
|
/// Uncompressed stream of <c>gamestate</c> file which can be extracted from save archive
|
||||||
/// </param>
|
/// </param>
|
||||||
@@ -23,211 +31,89 @@ public class SaveParserEU4
|
|||||||
/// Parsing whole save takes 10 seconds on mid pc and takes 1GB of RAM,
|
/// Parsing whole save takes 10 seconds on mid pc and takes 1GB of RAM,
|
||||||
/// so you should specify what exactly you want to get from save file
|
/// so you should specify what exactly you want to get from save file
|
||||||
/// </param>
|
/// </param>
|
||||||
public SaveParserEU4(Stream savefile, ISearchExpression? query)
|
/// <param name="encoding">Encoding of the strings inside the save. Latin1 by default.</param>
|
||||||
|
/// <param name="bufferSize">Size of the read buffer. Mostly useful for tests.</param>
|
||||||
|
public SaveParserEU4(Stream savefile, ISearchExpression? query,
|
||||||
|
Encoding? encoding = null, int bufferSize = DefaultBufferSize)
|
||||||
{
|
{
|
||||||
_saveFile = savefile;
|
Encoding = encoding ?? Encoding.Latin1;
|
||||||
_searchExprCurrent = query;
|
_searchExprCurrent = query;
|
||||||
const int tokenBufSize = 5;
|
_tokens = new Tokenizer(savefile, bufferSize);
|
||||||
_tokens = new BufferedEnumerator<Token>(LexTextSave(), tokenBufSize);
|
|
||||||
_stringBuilderPool = new DefaultObjectPool<StringBuilder>(
|
|
||||||
new StringBuilderPooledObjectPolicy
|
|
||||||
{
|
|
||||||
InitialCapacity = tokenBufSize * 13,
|
|
||||||
MaximumRetainedCapacity = tokenBufSize * 13,
|
|
||||||
});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
protected IEnumerator<Token> LexTextSave()
|
public Dictionary<string, object> Parse()
|
||||||
{
|
{
|
||||||
string expectedHeader = "EU4txt";
|
_tokens.ReadHeader(Header, Encoding);
|
||||||
byte[] headBytes = new byte[expectedHeader.Length];
|
return ParseDict();
|
||||||
_saveFile.ReadExactly(headBytes);
|
|
||||||
string headStr = Encoding.UTF8.GetString(headBytes);
|
|
||||||
if (headStr != expectedHeader)
|
|
||||||
throw new Exception($"Invalid gamestate header. Expected '{expectedHeader}', got '{headStr}'.");
|
|
||||||
|
|
||||||
StringBuilder strb = _stringBuilderPool.Get();
|
|
||||||
int line = 2;
|
|
||||||
int column = 0;
|
|
||||||
bool isQuoteOpen = false;
|
|
||||||
bool isStrInQuotes = false;
|
|
||||||
Token strToken = new()
|
|
||||||
{
|
|
||||||
type = TokenType.Invalid,
|
|
||||||
column = -1,
|
|
||||||
line = -1,
|
|
||||||
value = null,
|
|
||||||
};
|
|
||||||
|
|
||||||
bool TryCompleteStringToken()
|
|
||||||
{
|
|
||||||
if (isQuoteOpen)
|
|
||||||
return false;
|
|
||||||
|
|
||||||
// strings in quotes may be empty
|
|
||||||
if (!isStrInQuotes && (strb.Length <= 0 || strb[0] == '#'))
|
|
||||||
return false;
|
|
||||||
|
|
||||||
strToken = new Token
|
|
||||||
{
|
|
||||||
type = TokenType.StringOrNumber,
|
|
||||||
column = (short)(column - strb.Length),
|
|
||||||
line = line,
|
|
||||||
value = strb,
|
|
||||||
};
|
|
||||||
strb = _stringBuilderPool.Get();
|
|
||||||
isStrInQuotes = false;
|
|
||||||
return true;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Reading the save one byte at a time through Stream.ReadByte() costs a virtual call
|
|
||||||
// per byte, which dominated the parsing time. Bytes are pulled into this buffer
|
|
||||||
// instead, so the stream is touched once per 64 KB and the inner loop reads an array.
|
|
||||||
byte[] buffer = new byte[64 * 1024];
|
|
||||||
int bufferLength;
|
|
||||||
while ((bufferLength = _saveFile.Read(buffer, 0, buffer.Length)) > 0)
|
|
||||||
{
|
|
||||||
for (int i = 0; i < bufferLength; i++)
|
|
||||||
{
|
|
||||||
int c = buffer[i];
|
|
||||||
column++;
|
|
||||||
switch (c)
|
|
||||||
{
|
|
||||||
case '\"':
|
|
||||||
isQuoteOpen = !isQuoteOpen;
|
|
||||||
isStrInQuotes = true;
|
|
||||||
break;
|
|
||||||
case ' ':
|
|
||||||
case '\t':
|
|
||||||
case '\r':
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
break;
|
|
||||||
case '\n':
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
line++;
|
|
||||||
column = 0;
|
|
||||||
break;
|
|
||||||
case '=':
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
yield return new Token
|
|
||||||
{
|
|
||||||
type = TokenType.Equals,
|
|
||||||
line = line, column = (short)column
|
|
||||||
};
|
|
||||||
break;
|
|
||||||
case '{':
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
yield return new Token
|
|
||||||
{
|
|
||||||
type = TokenType.BracketOpen,
|
|
||||||
line = line, column = (short)column
|
|
||||||
};
|
|
||||||
break;
|
|
||||||
case '}':
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
yield return new Token
|
|
||||||
{
|
|
||||||
type = TokenType.BracketClose,
|
|
||||||
line = line, column = (short)column
|
|
||||||
};
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
// Skip control characters, which are invisible and causing frontend bugs.
|
|
||||||
// I dont know why there are so many of them in strings.
|
|
||||||
if (c >= 0x20)
|
|
||||||
strb.Append((char)c);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// end of file: the last token may still be unterminated
|
|
||||||
if (TryCompleteStringToken())
|
|
||||||
yield return strToken;
|
|
||||||
_stringBuilderPool.Return(strb);
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
// doesn't move next
|
// doesn't move next
|
||||||
private object? ParseValue()
|
private object? ParseValue()
|
||||||
{
|
{
|
||||||
var tok = _tokens.Current.Value;
|
switch (_tokens.Type)
|
||||||
switch (tok.type)
|
|
||||||
{
|
{
|
||||||
case TokenType.StringOrNumber:
|
case TokenType.StringOrNumber:
|
||||||
try
|
return ParseScalar(_tokens.Text);
|
||||||
{
|
|
||||||
// string values can be empty
|
|
||||||
if (tok.value!.Length == 0)
|
|
||||||
return string.Empty;
|
|
||||||
if (tok.value.Equals("yes"))
|
|
||||||
return true;
|
|
||||||
if (tok.value.Equals("no"))
|
|
||||||
return false;
|
|
||||||
|
|
||||||
string tokStr = tok.value.ToString();
|
|
||||||
if (tokStr[0] != '-' && !char.IsDigit(tokStr[0]))
|
|
||||||
return tokStr;
|
|
||||||
if (tokStr.Contains('.') && double.TryParse(tokStr, out double d))
|
|
||||||
return d;
|
|
||||||
if (long.TryParse(tokStr, out long l))
|
|
||||||
return l;
|
|
||||||
return tokStr;
|
|
||||||
}
|
|
||||||
finally
|
|
||||||
{
|
|
||||||
_stringBuilderPool.Return(tok.value!);
|
|
||||||
}
|
|
||||||
case TokenType.BracketOpen:
|
case TokenType.BracketOpen:
|
||||||
object obj = ParseListOrDict();
|
return ParseListOrDict();
|
||||||
return obj;
|
|
||||||
case TokenType.BracketClose:
|
case TokenType.BracketClose:
|
||||||
return null;
|
return null;
|
||||||
default:
|
default:
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private object ParseScalar(ReadOnlySpan<byte> text)
|
||||||
|
{
|
||||||
|
// string values can be empty
|
||||||
|
if (text.Length == 0)
|
||||||
|
return string.Empty;
|
||||||
|
if (text.SequenceEqual("yes"u8))
|
||||||
|
return true;
|
||||||
|
if (text.SequenceEqual("no"u8))
|
||||||
|
return false;
|
||||||
|
|
||||||
|
byte first = text[0];
|
||||||
|
if (first != (byte)'-' && !char.IsAsciiDigit((char)first))
|
||||||
|
return DecodeString(text);
|
||||||
|
if (text.Contains((byte)'.')
|
||||||
|
&& double.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out double d))
|
||||||
|
return d;
|
||||||
|
if (long.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out long l))
|
||||||
|
return l;
|
||||||
|
return DecodeString(text);
|
||||||
|
}
|
||||||
|
|
||||||
|
private string DecodeString(ReadOnlySpan<byte> text)
|
||||||
|
{
|
||||||
|
// Skip control characters, which are invisible and causing frontend bugs.
|
||||||
|
// I dont know why there are so many of them in strings.
|
||||||
|
if (!text.ContainsAnyInRange((byte)0, (byte)0x1F))
|
||||||
|
return Encoding.GetString(text);
|
||||||
|
|
||||||
|
Span<byte> cleaned = text.Length <= 256 ? stackalloc byte[text.Length] : new byte[text.Length];
|
||||||
|
int length = 0;
|
||||||
|
foreach (byte b in text)
|
||||||
|
if (b >= 0x20)
|
||||||
|
cleaned[length++] = b;
|
||||||
|
return Encoding.GetString(cleaned[..length]);
|
||||||
|
}
|
||||||
|
|
||||||
// skips next value
|
// skips next value
|
||||||
/// <returns>true if skipped value, false if current token is closing bracket</returns>
|
/// <returns>true if skipped value, false if current token is closing bracket</returns>
|
||||||
private bool SkipValue()
|
private bool SkipValue()
|
||||||
{
|
{
|
||||||
var tok = _tokens.Current.Value;
|
switch (_tokens.Type)
|
||||||
switch (tok.type)
|
|
||||||
{
|
{
|
||||||
case TokenType.BracketOpen:
|
case TokenType.BracketOpen:
|
||||||
SkipObject();
|
_tokens.SkipBlock();
|
||||||
return true;
|
return true;
|
||||||
case TokenType.StringOrNumber:
|
case TokenType.StringOrNumber:
|
||||||
_stringBuilderPool.Return(tok.value!);
|
|
||||||
return true;
|
return true;
|
||||||
case TokenType.BracketClose:
|
case TokenType.BracketClose:
|
||||||
return false;
|
return false;
|
||||||
default:
|
default:
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// skips all tokens inside curly braces block
|
|
||||||
private void SkipObject(int bracketBalance = 1)
|
|
||||||
{
|
|
||||||
while (bracketBalance != 0 && _tokens.MoveNext())
|
|
||||||
{
|
|
||||||
var tok = _tokens.Current.Value;
|
|
||||||
if (tok.type == TokenType.BracketOpen)
|
|
||||||
bracketBalance++;
|
|
||||||
else if (tok.type == TokenType.BracketClose)
|
|
||||||
bracketBalance--;
|
|
||||||
else if (tok.type == TokenType.StringOrNumber)
|
|
||||||
{
|
|
||||||
_stringBuilderPool.Return(tok.value!);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -237,9 +123,7 @@ public class SaveParserEU4
|
|||||||
// doesn't move next
|
// doesn't move next
|
||||||
private object ParseListOrDict()
|
private object ParseListOrDict()
|
||||||
{
|
{
|
||||||
var first = _tokens.Current.Next;
|
if (_tokens.PeekType(1) == TokenType.StringOrNumber && _tokens.PeekType(2) == TokenType.Equals)
|
||||||
var second = _tokens.Current.Next?.Next;
|
|
||||||
if (first?.Value.type == TokenType.StringOrNumber && second?.Value.type == TokenType.Equals)
|
|
||||||
return ParseDict();
|
return ParseDict();
|
||||||
|
|
||||||
return ParseList();
|
return ParseList();
|
||||||
@@ -251,17 +135,18 @@ public class SaveParserEU4
|
|||||||
List<object> list = new();
|
List<object> list = new();
|
||||||
for (int i = 0; ; i++)
|
for (int i = 0; ; i++)
|
||||||
{
|
{
|
||||||
if (!_tokens.MoveNext())
|
if (!_tokens.Read())
|
||||||
throw new Exception("Unexpected end of file");
|
throw new Exception("Unexpected end of file");
|
||||||
|
|
||||||
ISearchExpression? searchExprNext = null;
|
ISearchExpression? searchExprNext = null;
|
||||||
if (_searchExprCurrent != null
|
if (_searchExprCurrent != null
|
||||||
&& !_searchExprCurrent.DoesMatch(new SearchArgs(i, string.Empty), out searchExprNext))
|
&& !_searchExprCurrent.DoesMatch(new MatchCandidate(i), out searchExprNext))
|
||||||
{
|
{
|
||||||
if (!SkipValue())
|
if (!SkipValue())
|
||||||
break;
|
break;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
var searchExprPrev = _searchExprCurrent;
|
var searchExprPrev = _searchExprCurrent;
|
||||||
_searchExprCurrent = searchExprNext;
|
_searchExprCurrent = searchExprNext;
|
||||||
object? value = ParseValue();
|
object? value = ParseValue();
|
||||||
@@ -284,53 +169,54 @@ public class SaveParserEU4
|
|||||||
{
|
{
|
||||||
Dictionary<string, object> dict = new();
|
Dictionary<string, object> dict = new();
|
||||||
|
|
||||||
// root is a dict without closing bracket, so this method must check _tokenIndex < _tokens.Count
|
// root is a dict without closing bracket, so this method must check for end of file
|
||||||
for (int localIndex = 0; _tokens.MoveNext(); localIndex++)
|
for (int localIndex = 0; _tokens.Read(); localIndex++)
|
||||||
{
|
{
|
||||||
var tok = _tokens.Current.Value;
|
|
||||||
// end of dictionary
|
// end of dictionary
|
||||||
if (tok.type == TokenType.BracketClose)
|
if (_tokens.Type == TokenType.BracketClose)
|
||||||
break;
|
break;
|
||||||
|
|
||||||
// Saves may contain some blocks without key.
|
// Saves may contain some blocks without key.
|
||||||
// Such blocks are skipped because idk where to put them.
|
// Such blocks are skipped because idk where to put them.
|
||||||
// Example: `technology_group=tech_cannorian{ }
|
// Example: `technology_group=tech_cannorian{ }
|
||||||
// { } { } { }`
|
// { } { } { }`
|
||||||
if (tok.type == TokenType.BracketOpen)
|
if (_tokens.Type == TokenType.BracketOpen)
|
||||||
{
|
{
|
||||||
SkipObject();
|
_tokens.SkipBlock();
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (tok.type != TokenType.StringOrNumber)
|
if (_tokens.Type != TokenType.StringOrNumber)
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
|
|
||||||
var keySB = tok.value!;
|
// The key is matched before the value is read, so that a key rejected by the query
|
||||||
|
// never has to become a string.
|
||||||
|
ISearchExpression? searchExprNext = null;
|
||||||
|
bool matches = _searchExprCurrent == null
|
||||||
|
|| _searchExprCurrent.DoesMatch(
|
||||||
|
new MatchCandidate(localIndex, _tokens.Text, Encoding), out searchExprNext);
|
||||||
|
string? keyStr = matches ? DecodeString(_tokens.Text) : null;
|
||||||
|
|
||||||
// next token should be `=` or `{`
|
// next token should be `=` or `{`
|
||||||
if (!_tokens.MoveNext())
|
if (!_tokens.Read())
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
tok = _tokens.Current.Value;
|
if (_tokens.Type == TokenType.Equals)
|
||||||
if (tok.type == TokenType.Equals)
|
|
||||||
{
|
{
|
||||||
// skip `=`
|
// skip `=`
|
||||||
if (!_tokens.MoveNext())
|
if (!_tokens.Read())
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
}
|
}
|
||||||
// Saves may contain object definition without `=`.
|
// Saves may contain object definition without `=`.
|
||||||
// Example: `map_area_data {` instead of `map_area_data = {`
|
// Example: `map_area_data {` instead of `map_area_data = {`
|
||||||
else if (tok.type != TokenType.BracketOpen)
|
else if (_tokens.Type != TokenType.BracketOpen)
|
||||||
{
|
{
|
||||||
throw new UnexpectedTokenException(tok);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
}
|
}
|
||||||
|
|
||||||
ISearchExpression? searchExprNext = null;
|
if (!matches)
|
||||||
if (_searchExprCurrent != null
|
|
||||||
&& !_searchExprCurrent.DoesMatch(new SearchArgs(localIndex, keySB), out searchExprNext))
|
|
||||||
{
|
{
|
||||||
if (!SkipValue())
|
if (!SkipValue())
|
||||||
throw new UnexpectedTokenException(_tokens.Current.Value);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
_stringBuilderPool.Return(keySB);
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -338,17 +224,14 @@ public class SaveParserEU4
|
|||||||
_searchExprCurrent = searchExprNext;
|
_searchExprCurrent = searchExprNext;
|
||||||
object? value = ParseValue();
|
object? value = ParseValue();
|
||||||
if (value is null)
|
if (value is null)
|
||||||
throw new UnexpectedTokenException(_tokens.Current.Value);
|
throw new UnexpectedTokenException(_tokens, Encoding);
|
||||||
_searchExprCurrent = searExpressionPrevious;
|
_searchExprCurrent = searExpressionPrevious;
|
||||||
|
|
||||||
string keyStr = keySB.ToString();
|
|
||||||
_stringBuilderPool.Return(keySB);
|
|
||||||
|
|
||||||
// Paradox save format has another way of defining list:
|
// Paradox save format has another way of defining list:
|
||||||
// a = 1
|
// a = 1
|
||||||
// a = 2
|
// a = 2
|
||||||
// It means `a = { 1 2 }`
|
// It means `a = { 1 2 }`
|
||||||
if (dict.TryGetValue(keyStr, out var firstValue))
|
if (dict.TryGetValue(keyStr!, out var firstValue))
|
||||||
{
|
{
|
||||||
// Do dot add empty collections into list.
|
// Do dot add empty collections into list.
|
||||||
// `key:{}` is okay, but i don't want to see `key:[{},{},{},{},{},{}]`
|
// `key:{}` is okay, but i don't want to see `key:[{},{},{},{},{},{}]`
|
||||||
@@ -357,73 +240,21 @@ public class SaveParserEU4
|
|||||||
|
|
||||||
if (firstValue is List<object> existingList)
|
if (firstValue is List<object> existingList)
|
||||||
existingList.Add(value);
|
existingList.Add(value);
|
||||||
else dict[keyStr] = new List<object> { firstValue, value };
|
else dict[keyStr!] = new List<object> { firstValue, value };
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
dict.Add(keyStr, value);
|
dict.Add(keyStr!, value);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return dict;
|
return dict;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Dictionary<string, object> Parse()
|
internal class UnexpectedTokenException : Exception
|
||||||
{
|
{
|
||||||
var root = ParseDict();
|
public UnexpectedTokenException(Tokenizer tokens, Encoding encoding) :
|
||||||
return root;
|
base($"Unexpected token: {tokens.Line}:{tokens.Column} '{tokens.Describe(encoding)}'")
|
||||||
}
|
|
||||||
|
|
||||||
protected enum TokenType : byte
|
|
||||||
{
|
|
||||||
Invalid,
|
|
||||||
StringOrNumber,
|
|
||||||
Equals,
|
|
||||||
BracketOpen,
|
|
||||||
BracketClose
|
|
||||||
}
|
|
||||||
|
|
||||||
protected struct Token
|
|
||||||
{
|
|
||||||
public required TokenType type;
|
|
||||||
public required short column;
|
|
||||||
public required int line;
|
|
||||||
public StringBuilder? value;
|
|
||||||
|
|
||||||
public override string ToString()
|
|
||||||
{
|
|
||||||
string s;
|
|
||||||
switch (type)
|
|
||||||
{
|
|
||||||
case TokenType.Invalid:
|
|
||||||
s = "INVALID_TOKEN";
|
|
||||||
break;
|
|
||||||
case TokenType.StringOrNumber:
|
|
||||||
if (value == null || value.Length == 0)
|
|
||||||
s = "NULL";
|
|
||||||
else s = value.ToString();
|
|
||||||
break;
|
|
||||||
case TokenType.Equals:
|
|
||||||
s = "=";
|
|
||||||
break;
|
|
||||||
case TokenType.BracketOpen:
|
|
||||||
s = "{";
|
|
||||||
break;
|
|
||||||
case TokenType.BracketClose:
|
|
||||||
s = "}";
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
throw new ArgumentOutOfRangeException(type.ToString());
|
|
||||||
}
|
|
||||||
|
|
||||||
return $"{line}:{column} '{s}'";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
protected class UnexpectedTokenException : Exception
|
|
||||||
{
|
|
||||||
public UnexpectedTokenException(Token token) :
|
|
||||||
base($"Unexpected token: {token}")
|
|
||||||
{
|
{
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,29 +1,37 @@
|
|||||||
namespace ParadoxSaveParser.Lib;
|
namespace ParadoxSaveParser.Lib;
|
||||||
|
|
||||||
public readonly record struct SearchArgs
|
/// <summary>
|
||||||
|
/// The node a search expression is tested against: its key inside the parent dictionary,
|
||||||
|
/// still in the raw bytes of the save, and its position inside the parent list.
|
||||||
|
/// Keys stay undecoded so that a key rejected by the query never becomes a string.
|
||||||
|
/// </summary>
|
||||||
|
public readonly ref struct MatchCandidate
|
||||||
{
|
{
|
||||||
public readonly string KeyStr;
|
public readonly ReadOnlySpan<byte> Key;
|
||||||
public readonly StringBuilder? KeySB;
|
public readonly int Index;
|
||||||
public readonly int LocalIndex;
|
|
||||||
|
|
||||||
public SearchArgs(int localIndex, string keyStr)
|
/// <summary>Encoding <see cref="Key" /> is written in.</summary>
|
||||||
|
public readonly Encoding Encoding;
|
||||||
|
|
||||||
|
/// <summary>A list item, which has a position but no key.</summary>
|
||||||
|
public MatchCandidate(int index)
|
||||||
{
|
{
|
||||||
KeyStr = keyStr;
|
Key = default;
|
||||||
KeySB = null;
|
Index = index;
|
||||||
LocalIndex = localIndex;
|
Encoding = Encoding.Latin1;
|
||||||
}
|
}
|
||||||
|
|
||||||
public SearchArgs(int localIndex, StringBuilder keySb)
|
public MatchCandidate(int index, ReadOnlySpan<byte> key, Encoding encoding)
|
||||||
{
|
{
|
||||||
KeyStr = string.Empty;
|
Key = key;
|
||||||
KeySB = keySb;
|
Index = index;
|
||||||
LocalIndex = localIndex;
|
Encoding = encoding;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
public interface ISearchExpression
|
public interface ISearchExpression
|
||||||
{
|
{
|
||||||
bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression);
|
bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression);
|
||||||
}
|
}
|
||||||
|
|
||||||
public static class SearchExpressionCompiler
|
public static class SearchExpressionCompiler
|
||||||
@@ -108,7 +116,7 @@ public static class SearchExpressionCompiler
|
|||||||
|
|
||||||
private record AnyMatchExpression(ISearchExpression? next) : ISearchExpression
|
private record AnyMatchExpression(ISearchExpression? next) : ISearchExpression
|
||||||
{
|
{
|
||||||
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression)
|
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
|
||||||
{
|
{
|
||||||
nextSearchExpression = next;
|
nextSearchExpression = next;
|
||||||
return true;
|
return true;
|
||||||
@@ -117,7 +125,7 @@ public static class SearchExpressionCompiler
|
|||||||
|
|
||||||
private record NoMatchExpression : ISearchExpression
|
private record NoMatchExpression : ISearchExpression
|
||||||
{
|
{
|
||||||
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression)
|
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
|
||||||
{
|
{
|
||||||
nextSearchExpression = null;
|
nextSearchExpression = null;
|
||||||
return false;
|
return false;
|
||||||
@@ -126,10 +134,10 @@ public static class SearchExpressionCompiler
|
|||||||
|
|
||||||
private record MultipleMatchExpression(List<ISearchExpression> subExprs) : ISearchExpression
|
private record MultipleMatchExpression(List<ISearchExpression> subExprs) : ISearchExpression
|
||||||
{
|
{
|
||||||
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression)
|
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
|
||||||
{
|
{
|
||||||
foreach (var e in subExprs)
|
foreach (var e in subExprs)
|
||||||
if (e.DoesMatch(args, out nextSearchExpression))
|
if (e.DoesMatch(candidate, out nextSearchExpression))
|
||||||
return true;
|
return true;
|
||||||
|
|
||||||
nextSearchExpression = null;
|
nextSearchExpression = null;
|
||||||
@@ -139,9 +147,9 @@ public static class SearchExpressionCompiler
|
|||||||
|
|
||||||
private record IndexMatchExpression(int index, ISearchExpression? next) : ISearchExpression
|
private record IndexMatchExpression(int index, ISearchExpression? next) : ISearchExpression
|
||||||
{
|
{
|
||||||
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression)
|
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
|
||||||
{
|
{
|
||||||
if (args.LocalIndex == index)
|
if (candidate.Index == index)
|
||||||
{
|
{
|
||||||
nextSearchExpression = next;
|
nextSearchExpression = next;
|
||||||
return true;
|
return true;
|
||||||
@@ -154,9 +162,20 @@ public static class SearchExpressionCompiler
|
|||||||
|
|
||||||
private record ExactMatchExpression(string key, ISearchExpression? next) : ISearchExpression
|
private record ExactMatchExpression(string key, ISearchExpression? next) : ISearchExpression
|
||||||
{
|
{
|
||||||
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression)
|
// the key is compared as bytes, so it is encoded once for whatever encoding the
|
||||||
|
// parser reads the save in
|
||||||
|
private byte[]? _keyBytes;
|
||||||
|
private Encoding? _keyEncoding;
|
||||||
|
|
||||||
|
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
|
||||||
{
|
{
|
||||||
if ((args.KeySB != null && args.KeySB.Equals(key)) || args.KeyStr == key)
|
if (!ReferenceEquals(_keyEncoding, candidate.Encoding))
|
||||||
|
{
|
||||||
|
_keyEncoding = candidate.Encoding;
|
||||||
|
_keyBytes = candidate.Encoding.GetBytes(key);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (candidate.Key.SequenceEqual(_keyBytes))
|
||||||
{
|
{
|
||||||
nextSearchExpression = next;
|
nextSearchExpression = next;
|
||||||
return true;
|
return true;
|
||||||
|
|||||||
@@ -0,0 +1,456 @@
|
|||||||
|
using System.Buffers;
|
||||||
|
|
||||||
|
namespace ParadoxSaveParser.Lib;
|
||||||
|
|
||||||
|
internal enum TokenType : byte
|
||||||
|
{
|
||||||
|
// default value, so a slot that was never scanned is recognizably empty
|
||||||
|
Invalid,
|
||||||
|
// any bare or quoted value: the tokenizer does not tell numbers from strings
|
||||||
|
StringOrNumber,
|
||||||
|
Equals,
|
||||||
|
BracketOpen,
|
||||||
|
BracketClose,
|
||||||
|
EndOfFile
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Splits a Paradox text save into tokens, reading the stream through a reusable buffer.
|
||||||
|
/// Token text is exposed as a span into that buffer, so a token costs no allocation at all
|
||||||
|
/// unless it happens to straddle a buffer refill.
|
||||||
|
/// <para>
|
||||||
|
/// The parser can tell the tokenizer to throw away a whole <c>{...}</c> block with
|
||||||
|
/// <see cref="SkipBlock" />. Blocks rejected by the search expression are then never
|
||||||
|
/// turned into tokens or strings, which is what makes a query cheaper than a full parse.
|
||||||
|
/// </para>
|
||||||
|
/// </summary>
|
||||||
|
internal sealed class Tokenizer
|
||||||
|
{
|
||||||
|
/// <summary>Bytes that end an unquoted token.</summary>
|
||||||
|
private static readonly SearchValues<byte> TokenEnd = SearchValues.Create(" \t\r\n={}\""u8);
|
||||||
|
|
||||||
|
private static readonly SearchValues<byte> Whitespace = SearchValues.Create(" \t\r\n"u8);
|
||||||
|
|
||||||
|
/// <summary>Everything <see cref="SkipBlock" /> has to look at while counting depth.</summary>
|
||||||
|
private static readonly SearchValues<byte> BlockChars = SearchValues.Create("{}\""u8);
|
||||||
|
|
||||||
|
/// <summary>One scanned token. Reused forever, so scanning a token allocates nothing.</summary>
|
||||||
|
private sealed class Slot
|
||||||
|
{
|
||||||
|
public TokenType Type;
|
||||||
|
// position of the token's first byte in the file, for error messages
|
||||||
|
public int Line;
|
||||||
|
public short Column;
|
||||||
|
|
||||||
|
// text is either a range of the shared buffer, or, once a refill would overwrite it,
|
||||||
|
// a copy in Scratch
|
||||||
|
public int Start;
|
||||||
|
public int Length;
|
||||||
|
// grown on demand and kept between tokens, so long tokens stop reallocating after a while
|
||||||
|
public byte[] Scratch = [];
|
||||||
|
public bool InScratch;
|
||||||
|
}
|
||||||
|
|
||||||
|
private readonly Stream _stream;
|
||||||
|
private readonly byte[] _buffer;
|
||||||
|
|
||||||
|
// current token plus up to two lookahead tokens
|
||||||
|
private readonly Slot[] _slots = [new Slot(), new Slot(), new Slot()];
|
||||||
|
// index of the current token; the slots are used as a ring, so the array is never shifted
|
||||||
|
private int _current;
|
||||||
|
// how many slots after _current already hold a scanned, not yet consumed token
|
||||||
|
private int _lookahead;
|
||||||
|
|
||||||
|
// read position and amount of valid data in _buffer
|
||||||
|
private int _pos;
|
||||||
|
private int _length;
|
||||||
|
// set once the stream has no more bytes; stops Fill() from calling Read() again
|
||||||
|
private bool _eof;
|
||||||
|
// position of _pos in the file; _column is 0-based here and reported 1-based
|
||||||
|
private int _line = 1;
|
||||||
|
private int _column;
|
||||||
|
|
||||||
|
public Tokenizer(Stream stream, int bufferSize)
|
||||||
|
{
|
||||||
|
if (bufferSize < 16)
|
||||||
|
throw new ArgumentOutOfRangeException(nameof(bufferSize), bufferSize, "buffer is too small");
|
||||||
|
_stream = stream;
|
||||||
|
_buffer = new byte[bufferSize];
|
||||||
|
}
|
||||||
|
|
||||||
|
public TokenType Type => _slots[_current].Type;
|
||||||
|
public int Line => _slots[_current].Line;
|
||||||
|
public short Column => _slots[_current].Column;
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Bytes of the current token, without quotes. Valid only until the next
|
||||||
|
/// <see cref="Read" />, because the buffer underneath it gets reused.
|
||||||
|
/// </summary>
|
||||||
|
public ReadOnlySpan<byte> Text
|
||||||
|
{
|
||||||
|
get
|
||||||
|
{
|
||||||
|
var slot = _slots[_current];
|
||||||
|
// a token that survived a refill was copied out; everything else still points into the buffer
|
||||||
|
return slot.InScratch
|
||||||
|
? slot.Scratch.AsSpan(0, slot.Length)
|
||||||
|
: _buffer.AsSpan(slot.Start, slot.Length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Consumes the file's magic header and checks it, before any token is scanned.</summary>
|
||||||
|
public void ReadHeader(ReadOnlySpan<byte> expected, Encoding encoding)
|
||||||
|
{
|
||||||
|
// read straight from the stream: this runs before the buffer holds anything
|
||||||
|
Span<byte> head = stackalloc byte[expected.Length];
|
||||||
|
_stream.ReadExactly(head);
|
||||||
|
if (!head.SequenceEqual(expected))
|
||||||
|
throw new Exception($"Invalid gamestate header. " +
|
||||||
|
$"Expected '{encoding.GetString(expected)}', got '{encoding.GetString(head)}'.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <returns>false at end of file</returns>
|
||||||
|
public bool Read()
|
||||||
|
{
|
||||||
|
_current = NextSlot(_current);
|
||||||
|
// the next slot may already be filled by an earlier PeekType, then there is nothing to scan
|
||||||
|
if (_lookahead > 0)
|
||||||
|
_lookahead--;
|
||||||
|
else Scan(_slots[_current]);
|
||||||
|
return _slots[_current].Type != TokenType.EndOfFile;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Type of a token that has not been consumed yet, 1 or 2 tokens ahead of the current one.
|
||||||
|
/// Only types are available: the parser never needs the text of a token it has not reached.
|
||||||
|
/// </summary>
|
||||||
|
public TokenType PeekType(int offset)
|
||||||
|
{
|
||||||
|
// scan only as far as asked, so lookahead never runs into a block SkipBlock is about to drop
|
||||||
|
while (_lookahead < offset)
|
||||||
|
{
|
||||||
|
// first slot after the ones that are already filled
|
||||||
|
int slot = _current;
|
||||||
|
for (int i = 0; i <= _lookahead; i++)
|
||||||
|
slot = NextSlot(slot);
|
||||||
|
Scan(_slots[slot]);
|
||||||
|
_lookahead++;
|
||||||
|
}
|
||||||
|
|
||||||
|
int index = _current;
|
||||||
|
for (int i = 0; i < offset; i++)
|
||||||
|
index = NextSlot(index);
|
||||||
|
return _slots[index].Type;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Throws away the block opened by the current <c>{</c> token without tokenizing it:
|
||||||
|
/// raw bytes are scanned for braces until the depth returns to zero.
|
||||||
|
/// Braces inside quoted strings are text and do not change the depth.
|
||||||
|
/// Leaves the closing <c>}</c> as the current token.
|
||||||
|
/// </summary>
|
||||||
|
public void SkipBlock()
|
||||||
|
{
|
||||||
|
int depth = 1;
|
||||||
|
|
||||||
|
// tokens that lookahead already pulled out of the buffer still count towards the depth
|
||||||
|
while (depth > 0 && _lookahead > 0)
|
||||||
|
{
|
||||||
|
_current = NextSlot(_current);
|
||||||
|
_lookahead--;
|
||||||
|
switch (_slots[_current].Type)
|
||||||
|
{
|
||||||
|
case TokenType.BracketOpen:
|
||||||
|
depth++;
|
||||||
|
break;
|
||||||
|
case TokenType.BracketClose:
|
||||||
|
depth--;
|
||||||
|
break;
|
||||||
|
case TokenType.EndOfFile:
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bool inQuotes = false;
|
||||||
|
while (depth > 0)
|
||||||
|
{
|
||||||
|
if (_pos >= _length && !Fill())
|
||||||
|
break; // unbalanced braces: the file ended inside the block
|
||||||
|
|
||||||
|
var span = _buffer.AsSpan(_pos, _length - _pos);
|
||||||
|
// inside a string only the closing quote matters, braces there are ordinary characters
|
||||||
|
int i = inQuotes ? span.IndexOf((byte)'\"') : span.IndexOfAny(BlockChars);
|
||||||
|
if (i < 0)
|
||||||
|
{
|
||||||
|
// nothing interesting in this bufferful, drop all of it and refill
|
||||||
|
Consume(span.Length);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
byte b = span[i];
|
||||||
|
Consume(i + 1);
|
||||||
|
if (b == (byte)'\"')
|
||||||
|
inQuotes = !inQuotes;
|
||||||
|
else if (b == (byte)'{')
|
||||||
|
depth++;
|
||||||
|
else depth--;
|
||||||
|
}
|
||||||
|
|
||||||
|
// hand the parser the closing brace it expects, without having tokenized anything inside
|
||||||
|
var current = _slots[_current];
|
||||||
|
current.Type = depth == 0 ? TokenType.BracketClose : TokenType.EndOfFile;
|
||||||
|
current.Length = 0;
|
||||||
|
current.InScratch = false;
|
||||||
|
current.Line = _line;
|
||||||
|
current.Column = (short)_column;
|
||||||
|
}
|
||||||
|
|
||||||
|
public string Describe(Encoding encoding) => Type switch
|
||||||
|
{
|
||||||
|
TokenType.StringOrNumber => Text.Length == 0 ? "NULL" : encoding.GetString(Text),
|
||||||
|
TokenType.Equals => "=",
|
||||||
|
TokenType.BracketOpen => "{",
|
||||||
|
TokenType.BracketClose => "}",
|
||||||
|
TokenType.EndOfFile => "END_OF_FILE",
|
||||||
|
_ => "INVALID_TOKEN",
|
||||||
|
};
|
||||||
|
|
||||||
|
/// <summary>Slot indices wrap around: the three slots form a ring buffer.</summary>
|
||||||
|
private int NextSlot(int i) => i + 1 == _slots.Length ? 0 : i + 1;
|
||||||
|
|
||||||
|
/// <summary>Reads the next token from the stream into <paramref name="slot" />.</summary>
|
||||||
|
private void Scan(Slot slot)
|
||||||
|
{
|
||||||
|
// loops only to skip comments, which produce no token
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (!SkipWhitespace())
|
||||||
|
{
|
||||||
|
slot.Type = TokenType.EndOfFile;
|
||||||
|
slot.Length = 0;
|
||||||
|
slot.InScratch = false;
|
||||||
|
slot.Line = _line;
|
||||||
|
slot.Column = (short)_column;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
slot.Line = _line;
|
||||||
|
slot.Column = (short)(_column + 1); // columns are reported 1-based
|
||||||
|
// SkipWhitespace guarantees at least one buffered byte here
|
||||||
|
byte b = _buffer[_pos];
|
||||||
|
switch (b)
|
||||||
|
{
|
||||||
|
case (byte)'=':
|
||||||
|
ConsumeFlat(1);
|
||||||
|
SetDelimiter(slot, TokenType.Equals);
|
||||||
|
return;
|
||||||
|
case (byte)'{':
|
||||||
|
ConsumeFlat(1);
|
||||||
|
SetDelimiter(slot, TokenType.BracketOpen);
|
||||||
|
return;
|
||||||
|
case (byte)'}':
|
||||||
|
ConsumeFlat(1);
|
||||||
|
SetDelimiter(slot, TokenType.BracketClose);
|
||||||
|
return;
|
||||||
|
case (byte)'\"':
|
||||||
|
ReadQuoted(slot);
|
||||||
|
return;
|
||||||
|
default:
|
||||||
|
ReadBare(slot);
|
||||||
|
// comments are dropped, same as before: a token starting with '#' is not emitted
|
||||||
|
if (slot.Length > 0 && FirstByte(slot) == (byte)'#')
|
||||||
|
continue;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Single-character tokens carry no text of their own.</summary>
|
||||||
|
private static void SetDelimiter(Slot slot, TokenType type)
|
||||||
|
{
|
||||||
|
slot.Type = type;
|
||||||
|
slot.Length = 0;
|
||||||
|
slot.InScratch = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Reads an unquoted token, which ends at the first delimiter byte.</summary>
|
||||||
|
private void ReadBare(Slot slot)
|
||||||
|
{
|
||||||
|
StartText(slot);
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
var span = _buffer.AsSpan(_pos, _length - _pos);
|
||||||
|
// one vectorized search replaces a loop over the token's bytes
|
||||||
|
int end = span.IndexOfAny(TokenEnd);
|
||||||
|
if (end >= 0)
|
||||||
|
{
|
||||||
|
// the delimiter itself is left for the next Scan to classify
|
||||||
|
AppendText(slot, span[..end]);
|
||||||
|
ConsumeFlat(end);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// token runs to the end of the buffer and continues in the next one
|
||||||
|
AppendText(slot, span);
|
||||||
|
ConsumeFlat(span.Length);
|
||||||
|
if (!Fill())
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Reads a quoted token. Everything up to the next <c>"</c> is text, braces included;
|
||||||
|
/// the format has no escape sequences, so a quote always closes the string.
|
||||||
|
/// </summary>
|
||||||
|
private void ReadQuoted(Slot slot)
|
||||||
|
{
|
||||||
|
ConsumeFlat(1); // opening quote
|
||||||
|
StartText(slot);
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (_pos >= _length && !Fill())
|
||||||
|
return; // unterminated string at end of file
|
||||||
|
|
||||||
|
var span = _buffer.AsSpan(_pos, _length - _pos);
|
||||||
|
int end = span.IndexOf((byte)'\"');
|
||||||
|
if (end >= 0)
|
||||||
|
{
|
||||||
|
AppendText(slot, span[..end]);
|
||||||
|
Consume(end + 1); // content plus the closing quote
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Consume, not ConsumeFlat: a quoted value is allowed to span several lines
|
||||||
|
AppendText(slot, span);
|
||||||
|
Consume(span.Length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Begins a text token at the current read position.</summary>
|
||||||
|
private void StartText(Slot slot)
|
||||||
|
{
|
||||||
|
slot.Type = TokenType.StringOrNumber;
|
||||||
|
slot.Start = _pos;
|
||||||
|
slot.Length = 0;
|
||||||
|
slot.InScratch = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Extends a text token by one chunk of bytes taken from the buffer.</summary>
|
||||||
|
private void AppendText(Slot slot, ReadOnlySpan<byte> chunk)
|
||||||
|
{
|
||||||
|
if (!slot.InScratch)
|
||||||
|
{
|
||||||
|
// chunks always continue where the previous one ended, so the range just grows
|
||||||
|
// still contiguous in the buffer, nothing to copy
|
||||||
|
if (slot.Length == 0)
|
||||||
|
slot.Start = _pos;
|
||||||
|
slot.Length += chunk.Length;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// the start of this token is already out of the buffer, so the rest has to follow it
|
||||||
|
EnsureScratch(slot, slot.Length + chunk.Length);
|
||||||
|
chunk.CopyTo(slot.Scratch.AsSpan(slot.Length));
|
||||||
|
slot.Length += chunk.Length;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>First byte of a token, wherever its text currently lives. Used to spot comments.</summary>
|
||||||
|
private byte FirstByte(Slot slot) => slot.InScratch ? slot.Scratch[0] : _buffer[slot.Start];
|
||||||
|
|
||||||
|
private static void EnsureScratch(Slot slot, int size)
|
||||||
|
{
|
||||||
|
if (slot.Scratch.Length >= size)
|
||||||
|
return;
|
||||||
|
// doubling keeps a token that is appended chunk by chunk from resizing on every chunk
|
||||||
|
int capacity = Math.Max(size, slot.Scratch.Length * 2);
|
||||||
|
Array.Resize(ref slot.Scratch, capacity);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Moves the read position onto the next non-whitespace byte.</summary>
|
||||||
|
/// <returns>false at end of file</returns>
|
||||||
|
private bool SkipWhitespace()
|
||||||
|
{
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (_pos >= _length && !Fill())
|
||||||
|
return false;
|
||||||
|
|
||||||
|
var span = _buffer.AsSpan(_pos, _length - _pos);
|
||||||
|
int i = span.IndexOfAnyExcept(Whitespace);
|
||||||
|
if (i < 0)
|
||||||
|
{
|
||||||
|
// whitespace to the end of the buffer, keep going in the next one
|
||||||
|
Consume(span.Length);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (i > 0)
|
||||||
|
Consume(i);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Refills the buffer from the stream.</summary>
|
||||||
|
/// <returns>false if the stream is exhausted</returns>
|
||||||
|
private bool Fill()
|
||||||
|
{
|
||||||
|
if (_eof)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// text of live tokens lives in the buffer, so it has to be copied out before overwriting
|
||||||
|
MaterializeSlots();
|
||||||
|
// a short read is fine, the next Fill picks up the rest
|
||||||
|
_length = _stream.Read(_buffer, 0, _buffer.Length);
|
||||||
|
_pos = 0;
|
||||||
|
if (_length > 0)
|
||||||
|
return true;
|
||||||
|
|
||||||
|
_eof = true;
|
||||||
|
_length = 0;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Copies every token that still points into the buffer out to its own scratch array.
|
||||||
|
/// Called just before a refill, which is the only moment that text can be lost.
|
||||||
|
/// </summary>
|
||||||
|
private void MaterializeSlots()
|
||||||
|
{
|
||||||
|
foreach (var slot in _slots)
|
||||||
|
{
|
||||||
|
// delimiters and empty tokens own no text, and a copied one needs no second copy
|
||||||
|
if (slot.InScratch || slot.Length == 0 || slot.Type != TokenType.StringOrNumber)
|
||||||
|
continue;
|
||||||
|
EnsureScratch(slot, slot.Length);
|
||||||
|
_buffer.AsSpan(slot.Start, slot.Length).CopyTo(slot.Scratch);
|
||||||
|
slot.InScratch = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Consumes bytes that cannot contain a line break.</summary>
|
||||||
|
private void ConsumeFlat(int count)
|
||||||
|
{
|
||||||
|
_pos += count;
|
||||||
|
_column += count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>Consumes bytes that may contain line breaks, keeping line and column exact.</summary>
|
||||||
|
private void Consume(int count)
|
||||||
|
{
|
||||||
|
var slice = _buffer.AsSpan(_pos, count);
|
||||||
|
// the last break decides the column, and only then is counting all of them worth it
|
||||||
|
int lastBreak = slice.LastIndexOf((byte)'\n');
|
||||||
|
if (lastBreak < 0)
|
||||||
|
{
|
||||||
|
_column += count;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
_line += slice.Count((byte)'\n');
|
||||||
|
// bytes left after the final break
|
||||||
|
_column = count - lastBreak - 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
_pos += count;
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user