Parser rewrite

This commit is contained in:
2026-09-15 00:14:26 +02:00
parent 3dde5e5acf
commit 1d53f6930a
11 changed files with 889 additions and 667 deletions
@@ -20,13 +20,10 @@ public class ExtractionBenchmarks
private byte[] _json = null!; private byte[] _json = null!;
private ISearchExpression _query = null!; private ISearchExpression _query = null!;
private Regex _flatInterpreted = null!; private Regex _technologyBlock = null!;
private Regex _flatCompiled = null!;
private Regex _flatSourceGen = null!;
private Regex _pathAwareCompiled = null!; private Regex _pathAwareCompiled = null!;
private Regex _pathAwareNonBacktracking = null!; private Regex _pathAwareNonBacktracking = null!;
private Regex _countriesBlock = null!; private Regex _countriesBlock = null!;
private PcreRegex _pcreFlat = null!;
private PcreRegex _pcrePathAware = null!; private PcreRegex _pcrePathAware = null!;
[GlobalSetup] [GlobalSetup]
@@ -37,13 +34,10 @@ public class ExtractionBenchmarks
_json = File.ReadAllBytes(BenchData.JsonPath); _json = File.ReadAllBytes(BenchData.JsonPath);
_query = SearchExpressionCompiler.Compile(BenchData.PdxQuery); _query = SearchExpressionCompiler.Compile(BenchData.PdxQuery);
_flatInterpreted = new Regex(Extractors.FlatPattern, RegexOptions.None); _technologyBlock = new Regex(Extractors.TechnologyBlockPattern, RegexOptions.Compiled);
_flatCompiled = new Regex(Extractors.FlatPattern, RegexOptions.Compiled);
_flatSourceGen = Extractors.FlatSourceGen();
_pathAwareCompiled = new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled); _pathAwareCompiled = new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled);
_pathAwareNonBacktracking = new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking); _pathAwareNonBacktracking = new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking);
_countriesBlock = new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled); _countriesBlock = new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled);
_pcreFlat = new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled);
_pcrePathAware = new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled); _pcrePathAware = new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled);
} }
@@ -53,26 +47,14 @@ public class ExtractionBenchmarks
[Benchmark(Description = "Full parse, then select")] [Benchmark(Description = "Full parse, then select")]
public long FullParse() => Extractors.FullParseThenSelect(_pdx); public long FullParse() => Extractors.FullParseThenSelect(_pdx);
[Benchmark(Description = ".NET Regex flat, interpreted")]
public long RegexFlatInterpreted() => Extractors.RegexScan(_flatInterpreted, _pdxText);
[Benchmark(Description = ".NET Regex flat, compiled")]
public long RegexFlatCompiled() => Extractors.RegexScan(_flatCompiled, _pdxText);
[Benchmark(Description = ".NET Regex flat, source generated")]
public long RegexFlatSourceGen() => Extractors.RegexScan(_flatSourceGen, _pdxText);
[Benchmark(Description = ".NET Regex path aware, compiled")] [Benchmark(Description = ".NET Regex path aware, compiled")]
public long RegexPathAwareCompiled() => Extractors.RegexScan(_pathAwareCompiled, _pdxText); public long RegexPathAwareCompiled() => Extractors.RegexScan(_pathAwareCompiled, _pdxText);
[Benchmark(Description = ".NET Regex path aware, NonBacktracking")] [Benchmark(Description = ".NET Regex path aware, NonBacktracking")]
public long RegexPathAwareNonBacktracking() => Extractors.RegexScan(_pathAwareNonBacktracking, _pdxText); public long RegexPathAwareNonBacktracking() => Extractors.RegexScan(_pathAwareNonBacktracking, _pdxText);
[Benchmark(Description = ".NET Regex balanced block + flat")] [Benchmark(Description = ".NET Regex balanced block + inner scan")]
public long RegexBalancedTwoStage() => Extractors.RegexTwoStage(_countriesBlock, _flatCompiled, _pdxText); public long RegexBalancedTwoStage() => Extractors.RegexTwoStage(_countriesBlock, _technologyBlock, _pdxText);
[Benchmark(Description = "PCRE.NET flat, JIT compiled")]
public long PcreFlat() => Extractors.PcreScan(_pcreFlat, _pdxText);
[Benchmark(Description = "PCRE.NET path aware, JIT compiled")] [Benchmark(Description = "PCRE.NET path aware, JIT compiled")]
public long PcrePathAware() => Extractors.PcreScan(_pcrePathAware, _pdxText); public long PcrePathAware() => Extractors.PcreScan(_pcrePathAware, _pdxText);
+4 -6
View File
@@ -65,10 +65,11 @@ public static partial class Extractors
// ---------------------------------------------------------------- .NET Regex // ---------------------------------------------------------------- .NET Regex
/// <summary> /// <summary>
/// Flat scan: matches technology blocks anywhere in the file. /// Matches a single technology block. On its own it would match anywhere in the file,
/// Fastest regex option, but it does not verify that the match really sits under <c>countries</c>. /// so it is only used as the second stage of <see cref="RegexTwoStage" />, inside the
/// already extracted <c>countries</c> block.
/// </summary> /// </summary>
public const string FlatPattern = public const string TechnologyBlockPattern =
@"\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}"; @"\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}";
/// <summary> /// <summary>
@@ -109,9 +110,6 @@ public static partial class Extractors
return RegexScan(innerRegex, block.Value); return RegexScan(innerRegex, block.Value);
} }
[GeneratedRegex(FlatPattern)]
public static partial Regex FlatSourceGen();
[GeneratedRegex(PathAwarePattern)] [GeneratedRegex(PathAwarePattern)]
public static partial Regex PathAwareSourceGen(); public static partial Regex PathAwareSourceGen();
@@ -10,7 +10,7 @@
</PropertyGroup> </PropertyGroup>
<ItemGroup> <ItemGroup>
<PackageReference Include="BenchmarkDotNet" Version="0.15.4" /> <PackageReference Include="BenchmarkDotNet" Version="0.15.8" />
<PackageReference Include="PCRE.NET" Version="1.6.0" /> <PackageReference Include="PCRE.NET" Version="1.6.0" />
</ItemGroup> </ItemGroup>
+1 -4
View File
@@ -43,16 +43,13 @@ public static class Program
Time("SearchExpression", () => Extractors.SearchExpression(pdx, query)); Time("SearchExpression", () => Extractors.SearchExpression(pdx, query));
Time("FullParse", () => Extractors.FullParseThenSelect(pdx)); Time("FullParse", () => Extractors.FullParseThenSelect(pdx));
Time("Regex flat compiled",
() => Extractors.RegexScan(new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text));
Time("Regex path aware compiled", Time("Regex path aware compiled",
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled), text)); () => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled), text));
Time("Regex path aware nonbacktracking", Time("Regex path aware nonbacktracking",
() => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking), text)); () => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking), text));
Time("Regex balanced two stage", Time("Regex balanced two stage",
() => Extractors.RegexTwoStage(new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled), () => Extractors.RegexTwoStage(new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled),
new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text)); new Regex(Extractors.TechnologyBlockPattern, RegexOptions.Compiled), text));
Time("PCRE flat", () => Extractors.PcreScan(new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled), text));
Time("PCRE path aware", Time("PCRE path aware",
() => Extractors.PcreScan(new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled), text)); () => Extractors.PcreScan(new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled), text));
Time("Utf8JsonReader", () => Extractors.Utf8JsonReaderScan(json)); Time("Utf8JsonReader", () => Extractors.Utf8JsonReaderScan(json));
+60 -54
View File
@@ -9,7 +9,7 @@ regex engines and `jq` on one realistic task:
|---|---| |---|---|
| this repo | `countries.*.technology` | | this repo | `countries.*.technology` |
| jq | `.countries \| map_values(.technology)` | | jq | `.countries \| map_values(.technology)` |
| .NET `Regex` / PCRE.NET | see `Extractors.FlatPattern` / `Extractors.PathAwarePattern` | | .NET `Regex` / PCRE.NET | see `Extractors.PathAwarePattern` / `Extractors.CountriesBlockPattern` |
| `Utf8JsonReader` / `JsonDocument` | hand written navigation | | `Utf8JsonReader` / `JsonDocument` | hand written navigation |
## Corpus ## Corpus
@@ -48,64 +48,70 @@ shows immediately when an engine finds something different from the others.
## Results ## Results
AMD Ryzen 7 5700X, 16 logical cores, Windows 10 21H2, .NET 10.0.12, jq 1.8.2, AMD Ryzen 7 5700X, 8 physical cores (16 logical), Windows 10 21H2, .NET 10.0.12, jq 1.8.2,
109 MB save / 92 MB JSON twin, BenchmarkDotNet `RunStrategy.Monitoring`, 5 iterations. 109 MB save / 92 MB JSON twin, BenchmarkDotNet `RunStrategy.Monitoring`,
3 warmups and 10 iterations for the in-process engines, 5 for jq.
| engine | mean | vs this project | allocated | correct? | | engine | mean | vs this project | allocated | correct? |
|---|---:|---:|---:|---| |---|---:|---:|---:|---|
| .NET Regex flat, source generated | 14.8 ms | 0.02x | 1.2 MB | path-blind | | **SearchExpression (this project)** | **49.6 ms** | **1.00x** | **1.46 MB** | yes |
| .NET Regex flat, compiled | 17.0 ms | 0.03x | 1.2 MB | path-blind | | .NET Regex balanced block + inner scan | 114.0 ms | 2.30x | 213 MB | yes |
| .NET Regex flat, interpreted | 19.3 ms | 0.03x | 1.2 MB | path-blind | | PCRE.NET path aware, JIT compiled | 123.4 ms | 2.49x | 0.90 MB | 1869/1870 |
| PCRE.NET flat, JIT compiled | 97.4 ms | 0.15x | 0.9 MB | path-blind | | .NET Regex path aware, compiled | 135.1 ms | 2.72x | 1.34 MB | 1869/1870 |
| .NET Regex balanced block + flat | 112.3 ms | 0.17x | 224 MB | yes | | Utf8JsonReader over JSON twin | 155.5 ms | 3.13x | 16 KB | yes |
| .NET Regex path aware, compiled | 116.2 ms | 0.18x | 1.4 MB | 1869/1870 | | JsonDocument over JSON twin | 323.2 ms | 6.51x | 72 B (native) | yes |
| PCRE.NET path aware, JIT compiled | 122.9 ms | 0.19x | 0.9 MB | 1869/1870 | | .NET Regex path aware, NonBacktracking | 440.5 ms | 8.87x | 30.6 MB | 1869/1870 |
| Utf8JsonReader over JSON twin | 155.7 ms | 0.24x | 0 B | yes | | Full parse, then select | 1204 ms | 24.3x | 962 MB | yes |
| JsonDocument over JSON twin | 321.4 ms | 0.50x | 0 B (native) | yes | | jq `empty` (parse the file, emit nothing) | 4188 ms | 84.4x | n/a | n/a |
| .NET Regex path aware, NonBacktracking | 444.5 ms | 0.69x | 32 MB | 1869/1870 | | jq `.countries \| map_values(.technology)` | 4339 ms | 87.4x | n/a | yes |
| **SearchExpression (this project)** | **645.1 ms** | **1.00x** | **1.6 MB** | yes | | jq `--stream` | 17497 ms | 352x | n/a | n/a |
| Full parse, then select | 1831 ms | 2.84x | 1153 MB | yes | | jq process startup only | 4.1 ms | 0.08x | n/a | n/a |
| jq `.countries \| map_values(.technology)` | 4340 ms | 6.73x | n/a | yes |
| jq `empty` (parse the file, emit nothing) | 4227 ms | 6.55x | n/a | n/a | Reading the save from disk instead of a preloaded `byte[]` costs the parser ~9 ms more:
| jq `--stream` | 17490 ms | 27.1x | n/a | n/a | 58.4 ms via `FileStream` (`ParserInputBenchmarks`).
| jq process startup only | 4.2 ms | 0.01x | n/a | n/a |
### Reading the table ### Reading the table
* **jq is ~6.7x slower than this parser** and 97% of that time is JSON parsing, not the * **This parser is the fastest engine measured here.** It is 2.3x faster than the only
query: `jq empty` on the same file costs 4227 ms of the 4340 ms. Process startup is regex approach that enforces the path, 3.1x faster than a hand written `Utf8JsonReader`
negligible (4 ms). jq's `--stream` mode, often recommended for large inputs, is 4x over the equivalent JSON, 6.5x faster than `JsonDocument`, and 87x faster than jq — while
*slower* still. jq also needs the data converted to JSON first, which this parser has to allocating 1.46 MB for a 109 MB input.
do anyway — so end to end jq is strictly more expensive here. * **jq is ~87x slower and 97% of that is JSON parsing, not the query**: `jq empty` on the
* **The regexes are 5-40x faster, but they are not doing the same job.** A regex never same file costs 4188 ms of the 4339 ms. Process startup is negligible (4 ms). jq's
parses the structure; it scans bytes for a literal and validates a short window around it. `--stream` mode, often recommended for large inputs, is 4x *slower* still. jq also cannot
`Regex` with a literal prefix (`technology={`) is vectorized, so 109 MB is scanned at read the Paradox format, so it first needs the save converted to JSON — a conversion this
several GB/s. That speed is real and the cost is real too: nothing verifies that the hit parser has to perform anyway.
is under `countries`, at the right depth, or belongs to the tag matched before it. * **Only one regex approach here is actually correct**, and it is the expensive one: cut the
* **Path-aware regexes lose both the speed and the correctness.** Forcing the tag into the `countries={...}` block out with a balancing-group pattern (a .NET-only feature; PCRE would
pattern costs 7x (17 ms -> 116 ms) and still returns 1869 of 1870 countries: the tag class need recursion), then scan inside it. It allocates 213 MB, because stage one materialises
`[A-Z0-9]{3}` silently drops EU4's `---` pseudo-country. Pairing survives here only by the whole block as a string.
luck of the file layout — all 933 tech-less country blocks happen to be grouped at the end * **The cheaper path-aware pattern pairs a country tag with the next technology block over a
of the save, so the lazy gap never runs across one. A save that interleaves them would lazy gap, and silently gets it wrong**: 1869 of 1870 countries, because the tag class
make the regex report a technology block under the wrong country tag with no error. `[A-Z0-9]{3}` drops EU4's `---` pseudo-country. Even that much only holds by luck of the
* **`RegexOptions.NonBacktracking`** guarantees linear time but is 26x slower than the file layout — all 933 tech-less country blocks happen to be grouped at the end of the save,
compiled backtracking engine on this pattern and allocates 32 MB. so the lazy gap never runs across one. A save that interleaved them would report a
* **PCRE.NET** (the most used non-BCL regex engine in .NET, ~460k downloads) is 6x slower technology block under the wrong country tag, with no error.
than `System.Text.RegularExpressions` on the flat pattern, because .NET's vectorized * **`RegexOptions.NonBacktracking`** guarantees linear time but is 3.3x slower than the
literal prefix search beats PCRE2's JIT here. On the path-aware pattern the two are equal. compiled backtracking engine on this pattern and allocates 31 MB.
There is no reason to leave the BCL engine for this workload. * **PCRE.NET** (the most used non-BCL regex engine in .NET, ~460k downloads) is within ~10%
* **The query is what makes this parser fast, not the parsing.** Same parser, same file: of `System.Text.RegularExpressions` on the path-aware pattern (123 ms vs 135 ms). Nothing
645 ms with `countries.*.technology`, 1831 ms and 1.1 GB allocated without a query. The here justifies leaving the BCL engine.
search expression prunes ~65% of the work and 99.9% of the allocations. * **The query is what makes this parser fast.** Same parser, same file: 50 ms with
* **Against a JSON reader on equivalent data**, the parser is 4x slower than `Utf8JsonReader` `countries.*.technology`, 1204 ms and 962 MB allocated without a query — 24x and 660x.
and 2x slower than `JsonDocument`. Those numbers exclude the pdx -> JSON conversion
(~4.7 s), so they are a ceiling for the format, not a usable alternative.
### Where this parser's time goes ### How the parser got here
645 ms for 109 MB is ~170 MB/s, or ~13 cycles per input byte. The lexer reads the save one Three measurements on the same benchmark (`ParserInputBenchmarks`, reading a `FileStream`):
byte at a time through `Stream.ReadByte()` (`SaveParserEU4.LexTextSave`), which is a virtual
call plus bounds check per byte, and appends char by char into a `StringBuilder`. Reading | lexer | mean | allocated |
into a `byte[]` buffer and scanning it with `ReadOnlySpan<byte>.IndexOfAny` would be the |---|---:|---:|
first thing to try — the regex numbers above show what the same hardware does when it scans | `Stream.ReadByte()` per byte | 834 ms | 1.55 MB |
a span instead of a stream. | copy the file into a `MemoryStream` first, still `ReadByte()` | 738 ms | 257 MB |
| 64 KB buffer + `Tokenizer` with byte-level block skipping | 58 ms | 1.46 MB |
The first jump came from removing a virtual call per byte. The large one came from making
the query prune *work* rather than just data: `Tokenizer.SkipBlock` throws away a rejected
`{...}` block by scanning raw bytes for brace depth with `SearchValues<byte>.IndexOfAny`,
so blocks the search expression rejects are never turned into tokens or strings at all.
Retained tokens are spans into the read buffer, decoded only when their value is kept, and
numbers are parsed straight from UTF8.
@@ -1,4 +1,6 @@
using System.Collections.Generic;
using System.IO; using System.IO;
using System.Text;
using System.Text.Encodings.Web; using System.Text.Encodings.Web;
using System.Text.Json; using System.Text.Json;
using System.Text.Json.Serialization; using System.Text.Json.Serialization;
@@ -41,12 +43,64 @@ public class SearchExpressionTests
[TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")] [TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")]
public void TestSearchOnSmallData(string input, string expectedOutput) public void TestSearchOnSmallData(string input, string expectedOutput)
{ {
using var saveStream = new MemoryStream(_smallSaveData, false); Assert.That(Search(_smallSaveData, input), Is.EqualTo(expectedOutput));
var se = SearchExpressionCompiler.Compile(input); }
var parser = new SaveParserEU4(saveStream, se);
/// <summary>
/// The same queries, but with a read buffer so small that tokens, quoted strings and
/// skipped blocks are split across refills.
/// </summary>
[TestCase("a", "a={ b={ c=0 d=1 e=2 } f=3 }")]
[TestCase("a.b", "a={ b={ c=0 d=1 e=2 } }")]
[TestCase("a.[1]", "a={ f=3 }")]
[TestCase("a.(b.e|f)", "a={ b={ e=2 } f=3 }")]
public void TestSearchAcrossBufferRefills(string input, string expectedOutput)
{
foreach (int bufferSize in new[] { 16, 17, 23, 64 })
Assert.That(Search(_smallSaveData, input, bufferSize), Is.EqualTo(expectedOutput),
$"buffer size {bufferSize}");
}
[Test]
public void BracesInsideQuotedStringAreText()
{
byte[] data = "EU4txt a={ name=\"x{y}z\" b=1 }".ToBytes();
using var saveStream = new MemoryStream(data, false);
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("a"));
var a = (Dictionary<string, object>)parser.Parse()["a"];
Assert.Multiple(() =>
{
Assert.That(a["name"], Is.EqualTo("x{y}z"));
Assert.That(a["b"], Is.EqualTo(1L));
});
}
[Test]
public void SkippedBlockIgnoresBracesInsideQuotedStrings()
{
byte[] data = "EU4txt a={ s=\"{{{\" } b=2".ToBytes();
using var saveStream = new MemoryStream(data, false);
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("b"));
Assert.That(parser.Parse()["b"], Is.EqualTo(2L));
}
[Test]
public void EncodingIsConfigurable()
{
byte[] data = Encoding.UTF8.GetBytes("EU4txt a={ name=\"Ä\" }");
using var saveStream = new MemoryStream(data, false);
var parser = new SaveParserEU4(saveStream, SearchExpressionCompiler.Compile("a"), Encoding.UTF8);
var a = (Dictionary<string, object>)parser.Parse()["a"];
Assert.That(a["name"], Is.EqualTo("Ä"));
}
private static string Search(byte[] saveData, string query, int bufferSize = 64 * 1024)
{
using var saveStream = new MemoryStream(saveData, false);
var se = SearchExpressionCompiler.Compile(query);
var parser = new SaveParserEU4(saveStream, se, bufferSize: bufferSize);
var rootNode = parser.Parse(); var rootNode = parser.Parse();
string json = JsonSerializer.Serialize(rootNode, _smallSaveSerializerOptions); string json = JsonSerializer.Serialize(rootNode, _smallSaveSerializerOptions);
string pdx = JsonToPdx(json); return JsonToPdx(json);
Assert.That(pdx, Is.EqualTo(expectedOutput));
} }
} }
-121
View File
@@ -1,121 +0,0 @@
using System.Collections;
namespace ParadoxSaveParser.Lib;
/// <summary>
/// Enumerator wrapper that stores <c>N/2</c> items before and <c>N/2-1</c> after <c>Current</c> item.
/// </summary>
/// <code language="cs">
/// IEnumerator&lt;int&gt; Enumerator()
/// {
/// for(int i = 0; i &lt; 6; i++)
/// yield return i;
/// }
///
/// var en = Enumerator();
/// var bufen = new BufferedEnumerator&lt;int&gt;(en, 5);
///
/// while(bufen.MoveNext())
/// {
/// var cur = bufen.Current;
/// for (var prev = cur.List?.First; prev != cur; prev = prev?.Next)
/// Console.Write($"{prev?.Value} ");
///
/// Console.Write($"| {cur.Value} |");
///
/// for (var next = cur.Next; next is not null; next = next.Next)
/// Console.Write($" {next.Value}");
/// Console.WriteLine();
/// }
/// </code>
/// Output:
/// <code>
/// | 0 | 1 2 3 4
/// 0 | 1 | 2 3 4
/// 0 1 | 2 | 3 4
/// 1 2 | 3 | 4 5
/// 2 3 | 4 | 5
/// 3 4 | 5 |
/// </code>
public class BufferedEnumerator<T> : IEnumerator<BufferedEnumerator<T>.Node>
{
public class Node
{
#nullable disable
public Node Previous;
public Node Next;
public T Value;
#nullable enable
}
private readonly IEnumerator<T> _enumerator;
private readonly Node[] _ringBuffer;
private Node? _currentNode;
private int _currentBufferIndex = -1;
private int _lastValueIndex = -1;
public BufferedEnumerator(IEnumerator<T> enumerator, int bufferSize)
{
_enumerator = enumerator;
_ringBuffer = new Node[bufferSize];
}
private void InitBuffer()
{
_ringBuffer[0] = new Node
{
Value = default!
};
for (int i = 1; i < _ringBuffer.Length; i++)
{
_ringBuffer[i] = new Node
{
Previous = _ringBuffer[i - 1],
Value = default!,
};
_ringBuffer[i - 1].Next = _ringBuffer[i];
}
_ringBuffer[^1].Next = _ringBuffer[0];
_ringBuffer[0].Previous = _ringBuffer[^1];
}
public bool MoveNext()
{
if (_currentBufferIndex == -1)
{
InitBuffer();
int beforeMidpoint = _ringBuffer.Length / 2 - 1;
for (int i = 0; i <= beforeMidpoint && _enumerator.MoveNext(); i++)
{
_ringBuffer[i].Value = _enumerator.Current;
}
}
_currentBufferIndex = (_currentBufferIndex + 1) % _ringBuffer.Length;
if (_enumerator.MoveNext())
{
int midpoint = (_currentBufferIndex + _ringBuffer.Length / 2) % _ringBuffer.Length;
_ringBuffer[midpoint].Value = _enumerator.Current;
_lastValueIndex = midpoint;
}
if(_currentBufferIndex == (_lastValueIndex + 1) % _ringBuffer.Length)
return false;
_currentNode = _ringBuffer[_currentBufferIndex];
return true;
}
public void Reset()
{
throw new NotImplementedException();
}
public Node Current => _currentNode!;
object IEnumerator.Current => Current;
public void Dispose()
{
}
}
@@ -8,6 +8,6 @@
</PropertyGroup> </PropertyGroup>
<ItemGroup> <ItemGroup>
<PackageReference Include="Microsoft.Extensions.ObjectPool" Version="10.0.12" /> <InternalsVisibleTo Include="ParadoxSaveParser.Lib.Tests" />
</ItemGroup> </ItemGroup>
</Project> </Project>
+99 -268
View File
@@ -2,7 +2,7 @@ global using System;
global using System.Collections.Generic; global using System.Collections.Generic;
global using System.IO; global using System.IO;
global using System.Text; global using System.Text;
using Microsoft.Extensions.ObjectPool; using System.Globalization;
namespace ParadoxSaveParser.Lib; namespace ParadoxSaveParser.Lib;
@@ -11,11 +11,19 @@ namespace ParadoxSaveParser.Lib;
/// </summary> /// </summary>
public class SaveParserEU4 public class SaveParserEU4
{ {
protected readonly Stream _saveFile; private const int DefaultBufferSize = 64 * 1024;
private readonly BufferedEnumerator<Token> _tokens; private static ReadOnlySpan<byte> Header => "EU4txt"u8;
private readonly ObjectPool<StringBuilder> _stringBuilderPool;
private readonly Tokenizer _tokens;
private ISearchExpression? _searchExprCurrent; private ISearchExpression? _searchExprCurrent;
/// <summary>
/// Encoding of the strings inside the save. Saves of different localizations use
/// different ones, so it can be changed; <see cref="System.Text.Encoding.Latin1" /> maps
/// every byte to one character and never fails, which makes it a safe default.
/// </summary>
public Encoding Encoding { get; }
/// <param name="savefile"> /// <param name="savefile">
/// Uncompressed stream of <c>gamestate</c> file which can be extracted from save archive /// Uncompressed stream of <c>gamestate</c> file which can be extracted from save archive
/// </param> /// </param>
@@ -23,211 +31,89 @@ public class SaveParserEU4
/// Parsing whole save takes 10 seconds on mid pc and takes 1GB of RAM, /// Parsing whole save takes 10 seconds on mid pc and takes 1GB of RAM,
/// so you should specify what exactly you want to get from save file /// so you should specify what exactly you want to get from save file
/// </param> /// </param>
public SaveParserEU4(Stream savefile, ISearchExpression? query) /// <param name="encoding">Encoding of the strings inside the save. Latin1 by default.</param>
/// <param name="bufferSize">Size of the read buffer. Mostly useful for tests.</param>
public SaveParserEU4(Stream savefile, ISearchExpression? query,
Encoding? encoding = null, int bufferSize = DefaultBufferSize)
{ {
_saveFile = savefile; Encoding = encoding ?? Encoding.Latin1;
_searchExprCurrent = query; _searchExprCurrent = query;
const int tokenBufSize = 5; _tokens = new Tokenizer(savefile, bufferSize);
_tokens = new BufferedEnumerator<Token>(LexTextSave(), tokenBufSize);
_stringBuilderPool = new DefaultObjectPool<StringBuilder>(
new StringBuilderPooledObjectPolicy
{
InitialCapacity = tokenBufSize * 13,
MaximumRetainedCapacity = tokenBufSize * 13,
});
} }
protected IEnumerator<Token> LexTextSave() public Dictionary<string, object> Parse()
{ {
string expectedHeader = "EU4txt"; _tokens.ReadHeader(Header, Encoding);
byte[] headBytes = new byte[expectedHeader.Length]; return ParseDict();
_saveFile.ReadExactly(headBytes);
string headStr = Encoding.UTF8.GetString(headBytes);
if (headStr != expectedHeader)
throw new Exception($"Invalid gamestate header. Expected '{expectedHeader}', got '{headStr}'.");
StringBuilder strb = _stringBuilderPool.Get();
int line = 2;
int column = 0;
bool isQuoteOpen = false;
bool isStrInQuotes = false;
Token strToken = new()
{
type = TokenType.Invalid,
column = -1,
line = -1,
value = null,
};
bool TryCompleteStringToken()
{
if (isQuoteOpen)
return false;
// strings in quotes may be empty
if (!isStrInQuotes && (strb.Length <= 0 || strb[0] == '#'))
return false;
strToken = new Token
{
type = TokenType.StringOrNumber,
column = (short)(column - strb.Length),
line = line,
value = strb,
};
strb = _stringBuilderPool.Get();
isStrInQuotes = false;
return true;
} }
// Reading the save one byte at a time through Stream.ReadByte() costs a virtual call
// per byte, which dominated the parsing time. Bytes are pulled into this buffer
// instead, so the stream is touched once per 64 KB and the inner loop reads an array.
byte[] buffer = new byte[64 * 1024];
int bufferLength;
while ((bufferLength = _saveFile.Read(buffer, 0, buffer.Length)) > 0)
{
for (int i = 0; i < bufferLength; i++)
{
int c = buffer[i];
column++;
switch (c)
{
case '\"':
isQuoteOpen = !isQuoteOpen;
isStrInQuotes = true;
break;
case ' ':
case '\t':
case '\r':
if (TryCompleteStringToken())
yield return strToken;
break;
case '\n':
if (TryCompleteStringToken())
yield return strToken;
line++;
column = 0;
break;
case '=':
if (TryCompleteStringToken())
yield return strToken;
yield return new Token
{
type = TokenType.Equals,
line = line, column = (short)column
};
break;
case '{':
if (TryCompleteStringToken())
yield return strToken;
yield return new Token
{
type = TokenType.BracketOpen,
line = line, column = (short)column
};
break;
case '}':
if (TryCompleteStringToken())
yield return strToken;
yield return new Token
{
type = TokenType.BracketClose,
line = line, column = (short)column
};
break;
default:
// Skip control characters, which are invisible and causing frontend bugs.
// I dont know why there are so many of them in strings.
if (c >= 0x20)
strb.Append((char)c);
break;
}
}
}
// end of file: the last token may still be unterminated
if (TryCompleteStringToken())
yield return strToken;
_stringBuilderPool.Return(strb);
}
// doesn't move next // doesn't move next
private object? ParseValue() private object? ParseValue()
{ {
var tok = _tokens.Current.Value; switch (_tokens.Type)
switch (tok.type)
{ {
case TokenType.StringOrNumber: case TokenType.StringOrNumber:
try return ParseScalar(_tokens.Text);
{
// string values can be empty
if (tok.value!.Length == 0)
return string.Empty;
if (tok.value.Equals("yes"))
return true;
if (tok.value.Equals("no"))
return false;
string tokStr = tok.value.ToString();
if (tokStr[0] != '-' && !char.IsDigit(tokStr[0]))
return tokStr;
if (tokStr.Contains('.') && double.TryParse(tokStr, out double d))
return d;
if (long.TryParse(tokStr, out long l))
return l;
return tokStr;
}
finally
{
_stringBuilderPool.Return(tok.value!);
}
case TokenType.BracketOpen: case TokenType.BracketOpen:
object obj = ParseListOrDict(); return ParseListOrDict();
return obj;
case TokenType.BracketClose: case TokenType.BracketClose:
return null; return null;
default: default:
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
} }
} }
private object ParseScalar(ReadOnlySpan<byte> text)
{
// string values can be empty
if (text.Length == 0)
return string.Empty;
if (text.SequenceEqual("yes"u8))
return true;
if (text.SequenceEqual("no"u8))
return false;
byte first = text[0];
if (first != (byte)'-' && !char.IsAsciiDigit((char)first))
return DecodeString(text);
if (text.Contains((byte)'.')
&& double.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out double d))
return d;
if (long.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out long l))
return l;
return DecodeString(text);
}
private string DecodeString(ReadOnlySpan<byte> text)
{
// Skip control characters, which are invisible and causing frontend bugs.
// I dont know why there are so many of them in strings.
if (!text.ContainsAnyInRange((byte)0, (byte)0x1F))
return Encoding.GetString(text);
Span<byte> cleaned = text.Length <= 256 ? stackalloc byte[text.Length] : new byte[text.Length];
int length = 0;
foreach (byte b in text)
if (b >= 0x20)
cleaned[length++] = b;
return Encoding.GetString(cleaned[..length]);
}
// skips next value // skips next value
/// <returns>true if skipped value, false if current token is closing bracket</returns> /// <returns>true if skipped value, false if current token is closing bracket</returns>
private bool SkipValue() private bool SkipValue()
{ {
var tok = _tokens.Current.Value; switch (_tokens.Type)
switch (tok.type)
{ {
case TokenType.BracketOpen: case TokenType.BracketOpen:
SkipObject(); _tokens.SkipBlock();
return true; return true;
case TokenType.StringOrNumber: case TokenType.StringOrNumber:
_stringBuilderPool.Return(tok.value!);
return true; return true;
case TokenType.BracketClose: case TokenType.BracketClose:
return false; return false;
default: default:
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
}
}
// skips all tokens inside curly braces block
private void SkipObject(int bracketBalance = 1)
{
while (bracketBalance != 0 && _tokens.MoveNext())
{
var tok = _tokens.Current.Value;
if (tok.type == TokenType.BracketOpen)
bracketBalance++;
else if (tok.type == TokenType.BracketClose)
bracketBalance--;
else if (tok.type == TokenType.StringOrNumber)
{
_stringBuilderPool.Return(tok.value!);
}
} }
} }
@@ -237,9 +123,7 @@ public class SaveParserEU4
// doesn't move next // doesn't move next
private object ParseListOrDict() private object ParseListOrDict()
{ {
var first = _tokens.Current.Next; if (_tokens.PeekType(1) == TokenType.StringOrNumber && _tokens.PeekType(2) == TokenType.Equals)
var second = _tokens.Current.Next?.Next;
if (first?.Value.type == TokenType.StringOrNumber && second?.Value.type == TokenType.Equals)
return ParseDict(); return ParseDict();
return ParseList(); return ParseList();
@@ -251,17 +135,18 @@ public class SaveParserEU4
List<object> list = new(); List<object> list = new();
for (int i = 0; ; i++) for (int i = 0; ; i++)
{ {
if (!_tokens.MoveNext()) if (!_tokens.Read())
throw new Exception("Unexpected end of file"); throw new Exception("Unexpected end of file");
ISearchExpression? searchExprNext = null; ISearchExpression? searchExprNext = null;
if (_searchExprCurrent != null if (_searchExprCurrent != null
&& !_searchExprCurrent.DoesMatch(new SearchArgs(i, string.Empty), out searchExprNext)) && !_searchExprCurrent.DoesMatch(new MatchCandidate(i), out searchExprNext))
{ {
if(!SkipValue()) if (!SkipValue())
break; break;
continue; continue;
} }
var searchExprPrev = _searchExprCurrent; var searchExprPrev = _searchExprCurrent;
_searchExprCurrent = searchExprNext; _searchExprCurrent = searchExprNext;
object? value = ParseValue(); object? value = ParseValue();
@@ -284,53 +169,54 @@ public class SaveParserEU4
{ {
Dictionary<string, object> dict = new(); Dictionary<string, object> dict = new();
// root is a dict without closing bracket, so this method must check _tokenIndex < _tokens.Count // root is a dict without closing bracket, so this method must check for end of file
for (int localIndex = 0; _tokens.MoveNext(); localIndex++) for (int localIndex = 0; _tokens.Read(); localIndex++)
{ {
var tok = _tokens.Current.Value;
// end of dictionary // end of dictionary
if (tok.type == TokenType.BracketClose) if (_tokens.Type == TokenType.BracketClose)
break; break;
// Saves may contain some blocks without key. // Saves may contain some blocks without key.
// Such blocks are skipped because idk where to put them. // Such blocks are skipped because idk where to put them.
// Example: `technology_group=tech_cannorian{ } // Example: `technology_group=tech_cannorian{ }
// { } { } { }` // { } { } { }`
if (tok.type == TokenType.BracketOpen) if (_tokens.Type == TokenType.BracketOpen)
{ {
SkipObject(); _tokens.SkipBlock();
continue; continue;
} }
if (tok.type != TokenType.StringOrNumber) if (_tokens.Type != TokenType.StringOrNumber)
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
var keySB = tok.value!; // The key is matched before the value is read, so that a key rejected by the query
// never has to become a string.
ISearchExpression? searchExprNext = null;
bool matches = _searchExprCurrent == null
|| _searchExprCurrent.DoesMatch(
new MatchCandidate(localIndex, _tokens.Text, Encoding), out searchExprNext);
string? keyStr = matches ? DecodeString(_tokens.Text) : null;
// next token should be `=` or `{` // next token should be `=` or `{`
if (!_tokens.MoveNext()) if (!_tokens.Read())
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
tok = _tokens.Current.Value; if (_tokens.Type == TokenType.Equals)
if (tok.type == TokenType.Equals)
{ {
// skip `=` // skip `=`
if (!_tokens.MoveNext()) if (!_tokens.Read())
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
} }
// Saves may contain object definition without `=`. // Saves may contain object definition without `=`.
// Example: `map_area_data {` instead of `map_area_data = {` // Example: `map_area_data {` instead of `map_area_data = {`
else if (tok.type != TokenType.BracketOpen) else if (_tokens.Type != TokenType.BracketOpen)
{ {
throw new UnexpectedTokenException(tok); throw new UnexpectedTokenException(_tokens, Encoding);
} }
ISearchExpression? searchExprNext = null; if (!matches)
if (_searchExprCurrent != null
&& !_searchExprCurrent.DoesMatch(new SearchArgs(localIndex, keySB), out searchExprNext))
{ {
if(!SkipValue()) if (!SkipValue())
throw new UnexpectedTokenException(_tokens.Current.Value); throw new UnexpectedTokenException(_tokens, Encoding);
_stringBuilderPool.Return(keySB);
continue; continue;
} }
@@ -338,17 +224,14 @@ public class SaveParserEU4
_searchExprCurrent = searchExprNext; _searchExprCurrent = searchExprNext;
object? value = ParseValue(); object? value = ParseValue();
if (value is null) if (value is null)
throw new UnexpectedTokenException(_tokens.Current.Value); throw new UnexpectedTokenException(_tokens, Encoding);
_searchExprCurrent = searExpressionPrevious; _searchExprCurrent = searExpressionPrevious;
string keyStr = keySB.ToString();
_stringBuilderPool.Return(keySB);
// Paradox save format has another way of defining list: // Paradox save format has another way of defining list:
// a = 1 // a = 1
// a = 2 // a = 2
// It means `a = { 1 2 }` // It means `a = { 1 2 }`
if (dict.TryGetValue(keyStr, out var firstValue)) if (dict.TryGetValue(keyStr!, out var firstValue))
{ {
// Do dot add empty collections into list. // Do dot add empty collections into list.
// `key:{}` is okay, but i don't want to see `key:[{},{},{},{},{},{}]` // `key:{}` is okay, but i don't want to see `key:[{},{},{},{},{},{}]`
@@ -357,73 +240,21 @@ public class SaveParserEU4
if (firstValue is List<object> existingList) if (firstValue is List<object> existingList)
existingList.Add(value); existingList.Add(value);
else dict[keyStr] = new List<object> { firstValue, value }; else dict[keyStr!] = new List<object> { firstValue, value };
} }
else else
{ {
dict.Add(keyStr, value); dict.Add(keyStr!, value);
} }
} }
return dict; return dict;
} }
public Dictionary<string, object> Parse() internal class UnexpectedTokenException : Exception
{ {
var root = ParseDict(); public UnexpectedTokenException(Tokenizer tokens, Encoding encoding) :
return root; base($"Unexpected token: {tokens.Line}:{tokens.Column} '{tokens.Describe(encoding)}'")
}
protected enum TokenType : byte
{
Invalid,
StringOrNumber,
Equals,
BracketOpen,
BracketClose
}
protected struct Token
{
public required TokenType type;
public required short column;
public required int line;
public StringBuilder? value;
public override string ToString()
{
string s;
switch (type)
{
case TokenType.Invalid:
s = "INVALID_TOKEN";
break;
case TokenType.StringOrNumber:
if (value == null || value.Length == 0)
s = "NULL";
else s = value.ToString();
break;
case TokenType.Equals:
s = "=";
break;
case TokenType.BracketOpen:
s = "{";
break;
case TokenType.BracketClose:
s = "}";
break;
default:
throw new ArgumentOutOfRangeException(type.ToString());
}
return $"{line}:{column} '{s}'";
}
}
protected class UnexpectedTokenException : Exception
{
public UnexpectedTokenException(Token token) :
base($"Unexpected token: {token}")
{ {
} }
} }
+40 -21
View File
@@ -1,29 +1,37 @@
namespace ParadoxSaveParser.Lib; namespace ParadoxSaveParser.Lib;
public readonly record struct SearchArgs /// <summary>
/// The node a search expression is tested against: its key inside the parent dictionary,
/// still in the raw bytes of the save, and its position inside the parent list.
/// Keys stay undecoded so that a key rejected by the query never becomes a string.
/// </summary>
public readonly ref struct MatchCandidate
{ {
public readonly string KeyStr; public readonly ReadOnlySpan<byte> Key;
public readonly StringBuilder? KeySB; public readonly int Index;
public readonly int LocalIndex;
public SearchArgs(int localIndex, string keyStr) /// <summary>Encoding <see cref="Key" /> is written in.</summary>
public readonly Encoding Encoding;
/// <summary>A list item, which has a position but no key.</summary>
public MatchCandidate(int index)
{ {
KeyStr = keyStr; Key = default;
KeySB = null; Index = index;
LocalIndex = localIndex; Encoding = Encoding.Latin1;
} }
public SearchArgs(int localIndex, StringBuilder keySb) public MatchCandidate(int index, ReadOnlySpan<byte> key, Encoding encoding)
{ {
KeyStr = string.Empty; Key = key;
KeySB = keySb; Index = index;
LocalIndex = localIndex; Encoding = encoding;
} }
} }
public interface ISearchExpression public interface ISearchExpression
{ {
bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression); bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression);
} }
public static class SearchExpressionCompiler public static class SearchExpressionCompiler
@@ -108,7 +116,7 @@ public static class SearchExpressionCompiler
private record AnyMatchExpression(ISearchExpression? next) : ISearchExpression private record AnyMatchExpression(ISearchExpression? next) : ISearchExpression
{ {
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression) public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
{ {
nextSearchExpression = next; nextSearchExpression = next;
return true; return true;
@@ -117,7 +125,7 @@ public static class SearchExpressionCompiler
private record NoMatchExpression : ISearchExpression private record NoMatchExpression : ISearchExpression
{ {
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression) public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
{ {
nextSearchExpression = null; nextSearchExpression = null;
return false; return false;
@@ -126,10 +134,10 @@ public static class SearchExpressionCompiler
private record MultipleMatchExpression(List<ISearchExpression> subExprs) : ISearchExpression private record MultipleMatchExpression(List<ISearchExpression> subExprs) : ISearchExpression
{ {
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression) public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
{ {
foreach (var e in subExprs) foreach (var e in subExprs)
if (e.DoesMatch(args, out nextSearchExpression)) if (e.DoesMatch(candidate, out nextSearchExpression))
return true; return true;
nextSearchExpression = null; nextSearchExpression = null;
@@ -139,9 +147,9 @@ public static class SearchExpressionCompiler
private record IndexMatchExpression(int index, ISearchExpression? next) : ISearchExpression private record IndexMatchExpression(int index, ISearchExpression? next) : ISearchExpression
{ {
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression) public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
{ {
if (args.LocalIndex == index) if (candidate.Index == index)
{ {
nextSearchExpression = next; nextSearchExpression = next;
return true; return true;
@@ -154,9 +162,20 @@ public static class SearchExpressionCompiler
private record ExactMatchExpression(string key, ISearchExpression? next) : ISearchExpression private record ExactMatchExpression(string key, ISearchExpression? next) : ISearchExpression
{ {
public bool DoesMatch(SearchArgs args, out ISearchExpression? nextSearchExpression) // the key is compared as bytes, so it is encoded once for whatever encoding the
// parser reads the save in
private byte[]? _keyBytes;
private Encoding? _keyEncoding;
public bool DoesMatch(MatchCandidate candidate, out ISearchExpression? nextSearchExpression)
{ {
if ((args.KeySB != null && args.KeySB.Equals(key)) || args.KeyStr == key) if (!ReferenceEquals(_keyEncoding, candidate.Encoding))
{
_keyEncoding = candidate.Encoding;
_keyBytes = candidate.Encoding.GetBytes(key);
}
if (candidate.Key.SequenceEqual(_keyBytes))
{ {
nextSearchExpression = next; nextSearchExpression = next;
return true; return true;
+456
View File
@@ -0,0 +1,456 @@
using System.Buffers;
namespace ParadoxSaveParser.Lib;
internal enum TokenType : byte
{
// default value, so a slot that was never scanned is recognizably empty
Invalid,
// any bare or quoted value: the tokenizer does not tell numbers from strings
StringOrNumber,
Equals,
BracketOpen,
BracketClose,
EndOfFile
}
/// <summary>
/// Splits a Paradox text save into tokens, reading the stream through a reusable buffer.
/// Token text is exposed as a span into that buffer, so a token costs no allocation at all
/// unless it happens to straddle a buffer refill.
/// <para>
/// The parser can tell the tokenizer to throw away a whole <c>{...}</c> block with
/// <see cref="SkipBlock" />. Blocks rejected by the search expression are then never
/// turned into tokens or strings, which is what makes a query cheaper than a full parse.
/// </para>
/// </summary>
internal sealed class Tokenizer
{
/// <summary>Bytes that end an unquoted token.</summary>
private static readonly SearchValues<byte> TokenEnd = SearchValues.Create(" \t\r\n={}\""u8);
private static readonly SearchValues<byte> Whitespace = SearchValues.Create(" \t\r\n"u8);
/// <summary>Everything <see cref="SkipBlock" /> has to look at while counting depth.</summary>
private static readonly SearchValues<byte> BlockChars = SearchValues.Create("{}\""u8);
/// <summary>One scanned token. Reused forever, so scanning a token allocates nothing.</summary>
private sealed class Slot
{
public TokenType Type;
// position of the token's first byte in the file, for error messages
public int Line;
public short Column;
// text is either a range of the shared buffer, or, once a refill would overwrite it,
// a copy in Scratch
public int Start;
public int Length;
// grown on demand and kept between tokens, so long tokens stop reallocating after a while
public byte[] Scratch = [];
public bool InScratch;
}
private readonly Stream _stream;
private readonly byte[] _buffer;
// current token plus up to two lookahead tokens
private readonly Slot[] _slots = [new Slot(), new Slot(), new Slot()];
// index of the current token; the slots are used as a ring, so the array is never shifted
private int _current;
// how many slots after _current already hold a scanned, not yet consumed token
private int _lookahead;
// read position and amount of valid data in _buffer
private int _pos;
private int _length;
// set once the stream has no more bytes; stops Fill() from calling Read() again
private bool _eof;
// position of _pos in the file; _column is 0-based here and reported 1-based
private int _line = 1;
private int _column;
public Tokenizer(Stream stream, int bufferSize)
{
if (bufferSize < 16)
throw new ArgumentOutOfRangeException(nameof(bufferSize), bufferSize, "buffer is too small");
_stream = stream;
_buffer = new byte[bufferSize];
}
public TokenType Type => _slots[_current].Type;
public int Line => _slots[_current].Line;
public short Column => _slots[_current].Column;
/// <summary>
/// Bytes of the current token, without quotes. Valid only until the next
/// <see cref="Read" />, because the buffer underneath it gets reused.
/// </summary>
public ReadOnlySpan<byte> Text
{
get
{
var slot = _slots[_current];
// a token that survived a refill was copied out; everything else still points into the buffer
return slot.InScratch
? slot.Scratch.AsSpan(0, slot.Length)
: _buffer.AsSpan(slot.Start, slot.Length);
}
}
/// <summary>Consumes the file's magic header and checks it, before any token is scanned.</summary>
public void ReadHeader(ReadOnlySpan<byte> expected, Encoding encoding)
{
// read straight from the stream: this runs before the buffer holds anything
Span<byte> head = stackalloc byte[expected.Length];
_stream.ReadExactly(head);
if (!head.SequenceEqual(expected))
throw new Exception($"Invalid gamestate header. " +
$"Expected '{encoding.GetString(expected)}', got '{encoding.GetString(head)}'.");
}
/// <returns>false at end of file</returns>
public bool Read()
{
_current = NextSlot(_current);
// the next slot may already be filled by an earlier PeekType, then there is nothing to scan
if (_lookahead > 0)
_lookahead--;
else Scan(_slots[_current]);
return _slots[_current].Type != TokenType.EndOfFile;
}
/// <summary>
/// Type of a token that has not been consumed yet, 1 or 2 tokens ahead of the current one.
/// Only types are available: the parser never needs the text of a token it has not reached.
/// </summary>
public TokenType PeekType(int offset)
{
// scan only as far as asked, so lookahead never runs into a block SkipBlock is about to drop
while (_lookahead < offset)
{
// first slot after the ones that are already filled
int slot = _current;
for (int i = 0; i <= _lookahead; i++)
slot = NextSlot(slot);
Scan(_slots[slot]);
_lookahead++;
}
int index = _current;
for (int i = 0; i < offset; i++)
index = NextSlot(index);
return _slots[index].Type;
}
/// <summary>
/// Throws away the block opened by the current <c>{</c> token without tokenizing it:
/// raw bytes are scanned for braces until the depth returns to zero.
/// Braces inside quoted strings are text and do not change the depth.
/// Leaves the closing <c>}</c> as the current token.
/// </summary>
public void SkipBlock()
{
int depth = 1;
// tokens that lookahead already pulled out of the buffer still count towards the depth
while (depth > 0 && _lookahead > 0)
{
_current = NextSlot(_current);
_lookahead--;
switch (_slots[_current].Type)
{
case TokenType.BracketOpen:
depth++;
break;
case TokenType.BracketClose:
depth--;
break;
case TokenType.EndOfFile:
return;
}
}
bool inQuotes = false;
while (depth > 0)
{
if (_pos >= _length && !Fill())
break; // unbalanced braces: the file ended inside the block
var span = _buffer.AsSpan(_pos, _length - _pos);
// inside a string only the closing quote matters, braces there are ordinary characters
int i = inQuotes ? span.IndexOf((byte)'\"') : span.IndexOfAny(BlockChars);
if (i < 0)
{
// nothing interesting in this bufferful, drop all of it and refill
Consume(span.Length);
continue;
}
byte b = span[i];
Consume(i + 1);
if (b == (byte)'\"')
inQuotes = !inQuotes;
else if (b == (byte)'{')
depth++;
else depth--;
}
// hand the parser the closing brace it expects, without having tokenized anything inside
var current = _slots[_current];
current.Type = depth == 0 ? TokenType.BracketClose : TokenType.EndOfFile;
current.Length = 0;
current.InScratch = false;
current.Line = _line;
current.Column = (short)_column;
}
public string Describe(Encoding encoding) => Type switch
{
TokenType.StringOrNumber => Text.Length == 0 ? "NULL" : encoding.GetString(Text),
TokenType.Equals => "=",
TokenType.BracketOpen => "{",
TokenType.BracketClose => "}",
TokenType.EndOfFile => "END_OF_FILE",
_ => "INVALID_TOKEN",
};
/// <summary>Slot indices wrap around: the three slots form a ring buffer.</summary>
private int NextSlot(int i) => i + 1 == _slots.Length ? 0 : i + 1;
/// <summary>Reads the next token from the stream into <paramref name="slot" />.</summary>
private void Scan(Slot slot)
{
// loops only to skip comments, which produce no token
while (true)
{
if (!SkipWhitespace())
{
slot.Type = TokenType.EndOfFile;
slot.Length = 0;
slot.InScratch = false;
slot.Line = _line;
slot.Column = (short)_column;
return;
}
slot.Line = _line;
slot.Column = (short)(_column + 1); // columns are reported 1-based
// SkipWhitespace guarantees at least one buffered byte here
byte b = _buffer[_pos];
switch (b)
{
case (byte)'=':
ConsumeFlat(1);
SetDelimiter(slot, TokenType.Equals);
return;
case (byte)'{':
ConsumeFlat(1);
SetDelimiter(slot, TokenType.BracketOpen);
return;
case (byte)'}':
ConsumeFlat(1);
SetDelimiter(slot, TokenType.BracketClose);
return;
case (byte)'\"':
ReadQuoted(slot);
return;
default:
ReadBare(slot);
// comments are dropped, same as before: a token starting with '#' is not emitted
if (slot.Length > 0 && FirstByte(slot) == (byte)'#')
continue;
return;
}
}
}
/// <summary>Single-character tokens carry no text of their own.</summary>
private static void SetDelimiter(Slot slot, TokenType type)
{
slot.Type = type;
slot.Length = 0;
slot.InScratch = false;
}
/// <summary>Reads an unquoted token, which ends at the first delimiter byte.</summary>
private void ReadBare(Slot slot)
{
StartText(slot);
while (true)
{
var span = _buffer.AsSpan(_pos, _length - _pos);
// one vectorized search replaces a loop over the token's bytes
int end = span.IndexOfAny(TokenEnd);
if (end >= 0)
{
// the delimiter itself is left for the next Scan to classify
AppendText(slot, span[..end]);
ConsumeFlat(end);
return;
}
// token runs to the end of the buffer and continues in the next one
AppendText(slot, span);
ConsumeFlat(span.Length);
if (!Fill())
return;
}
}
/// <summary>
/// Reads a quoted token. Everything up to the next <c>"</c> is text, braces included;
/// the format has no escape sequences, so a quote always closes the string.
/// </summary>
private void ReadQuoted(Slot slot)
{
ConsumeFlat(1); // opening quote
StartText(slot);
while (true)
{
if (_pos >= _length && !Fill())
return; // unterminated string at end of file
var span = _buffer.AsSpan(_pos, _length - _pos);
int end = span.IndexOf((byte)'\"');
if (end >= 0)
{
AppendText(slot, span[..end]);
Consume(end + 1); // content plus the closing quote
return;
}
// Consume, not ConsumeFlat: a quoted value is allowed to span several lines
AppendText(slot, span);
Consume(span.Length);
}
}
/// <summary>Begins a text token at the current read position.</summary>
private void StartText(Slot slot)
{
slot.Type = TokenType.StringOrNumber;
slot.Start = _pos;
slot.Length = 0;
slot.InScratch = false;
}
/// <summary>Extends a text token by one chunk of bytes taken from the buffer.</summary>
private void AppendText(Slot slot, ReadOnlySpan<byte> chunk)
{
if (!slot.InScratch)
{
// chunks always continue where the previous one ended, so the range just grows
// still contiguous in the buffer, nothing to copy
if (slot.Length == 0)
slot.Start = _pos;
slot.Length += chunk.Length;
return;
}
// the start of this token is already out of the buffer, so the rest has to follow it
EnsureScratch(slot, slot.Length + chunk.Length);
chunk.CopyTo(slot.Scratch.AsSpan(slot.Length));
slot.Length += chunk.Length;
}
/// <summary>First byte of a token, wherever its text currently lives. Used to spot comments.</summary>
private byte FirstByte(Slot slot) => slot.InScratch ? slot.Scratch[0] : _buffer[slot.Start];
private static void EnsureScratch(Slot slot, int size)
{
if (slot.Scratch.Length >= size)
return;
// doubling keeps a token that is appended chunk by chunk from resizing on every chunk
int capacity = Math.Max(size, slot.Scratch.Length * 2);
Array.Resize(ref slot.Scratch, capacity);
}
/// <summary>Moves the read position onto the next non-whitespace byte.</summary>
/// <returns>false at end of file</returns>
private bool SkipWhitespace()
{
while (true)
{
if (_pos >= _length && !Fill())
return false;
var span = _buffer.AsSpan(_pos, _length - _pos);
int i = span.IndexOfAnyExcept(Whitespace);
if (i < 0)
{
// whitespace to the end of the buffer, keep going in the next one
Consume(span.Length);
continue;
}
if (i > 0)
Consume(i);
return true;
}
}
/// <summary>Refills the buffer from the stream.</summary>
/// <returns>false if the stream is exhausted</returns>
private bool Fill()
{
if (_eof)
return false;
// text of live tokens lives in the buffer, so it has to be copied out before overwriting
MaterializeSlots();
// a short read is fine, the next Fill picks up the rest
_length = _stream.Read(_buffer, 0, _buffer.Length);
_pos = 0;
if (_length > 0)
return true;
_eof = true;
_length = 0;
return false;
}
/// <summary>
/// Copies every token that still points into the buffer out to its own scratch array.
/// Called just before a refill, which is the only moment that text can be lost.
/// </summary>
private void MaterializeSlots()
{
foreach (var slot in _slots)
{
// delimiters and empty tokens own no text, and a copied one needs no second copy
if (slot.InScratch || slot.Length == 0 || slot.Type != TokenType.StringOrNumber)
continue;
EnsureScratch(slot, slot.Length);
_buffer.AsSpan(slot.Start, slot.Length).CopyTo(slot.Scratch);
slot.InScratch = true;
}
}
/// <summary>Consumes bytes that cannot contain a line break.</summary>
private void ConsumeFlat(int count)
{
_pos += count;
_column += count;
}
/// <summary>Consumes bytes that may contain line breaks, keeping line and column exact.</summary>
private void Consume(int count)
{
var slice = _buffer.AsSpan(_pos, count);
// the last break decides the column, and only then is counting all of them worth it
int lastBreak = slice.LastIndexOf((byte)'\n');
if (lastBreak < 0)
{
_column += count;
}
else
{
_line += slice.Count((byte)'\n');
// bytes left after the final break
_column = count - lastBreak - 1;
}
_pos += count;
}
}