From 3dde5e5acfaca1d59fb0fbc2b50623978b8e47be Mon Sep 17 00:00:00 2001 From: Timerix Date: Mon, 14 Sep 2026 23:09:25 +0200 Subject: [PATCH] added benchmarks --- ParadoxSaveParser.Benchmarks/BenchData.cs | 77 +++++++ .../ExtractionBenchmarks.cs | 85 +++++++ ParadoxSaveParser.Benchmarks/Extractors.cs | 209 ++++++++++++++++++ ParadoxSaveParser.Benchmarks/JqBenchmarks.cs | 79 +++++++ ParadoxSaveParser.Benchmarks/Mispairing.cs | 60 +++++ .../ParadoxSaveParser.Benchmarks.csproj | 20 ++ .../ParserInputBenchmarks.cs | 29 +++ ParadoxSaveParser.Benchmarks/Program.cs | 80 +++++++ ParadoxSaveParser.Benchmarks/README.md | 111 ++++++++++ ParadoxSaveParser.sln | 53 +++++ README.md | 38 ++-- 11 files changed, 823 insertions(+), 18 deletions(-) create mode 100644 ParadoxSaveParser.Benchmarks/BenchData.cs create mode 100644 ParadoxSaveParser.Benchmarks/ExtractionBenchmarks.cs create mode 100644 ParadoxSaveParser.Benchmarks/Extractors.cs create mode 100644 ParadoxSaveParser.Benchmarks/JqBenchmarks.cs create mode 100644 ParadoxSaveParser.Benchmarks/Mispairing.cs create mode 100644 ParadoxSaveParser.Benchmarks/ParadoxSaveParser.Benchmarks.csproj create mode 100644 ParadoxSaveParser.Benchmarks/ParserInputBenchmarks.cs create mode 100644 ParadoxSaveParser.Benchmarks/Program.cs create mode 100644 ParadoxSaveParser.Benchmarks/README.md diff --git a/ParadoxSaveParser.Benchmarks/BenchData.cs b/ParadoxSaveParser.Benchmarks/BenchData.cs new file mode 100644 index 0000000..7eac775 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/BenchData.cs @@ -0,0 +1,77 @@ +using System; +using System.IO; +using System.Text; +using System.Text.Encodings.Web; +using System.Text.Json; +using System.Text.Json.Serialization; +using ParadoxSaveParser.Lib; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// Locates and lazily loads the benchmark corpus: +/// a real EU4 gamestate file and its compact JSON twin (for jq). +/// Override the directory with the PSP_BENCH_DATA environment variable. +/// +public static class BenchData +{ + public const string PdxQuery = "countries.*.technology"; + + /// jq program extracting the same data as . + public const string JqQuery = ".countries | map_values(.technology)"; + + public static string Dir { get; } = + Environment.GetEnvironmentVariable("PSP_BENCH_DATA") + ?? Path.Combine(FindRepoRoot(), "ParadoxSaveParser.CLI", "bin", "Debug", "net10.0"); + + public static string PdxPath => Path.Combine(Dir, "gamestate"); + public static string JsonPath => Path.Combine(Dir, "gamestate.min.json"); + + private static byte[]? _pdxBytes; + private static string? _pdxText; + + /// Raw bytes of the save, kept in memory so benchmarks don't measure disk I/O. + public static byte[] PdxBytes => _pdxBytes ??= File.ReadAllBytes(PdxPath); + + /// + /// The save decoded as text for the regex engines. + /// Latin1 is used because EU4 saves contain non-UTF8 bytes inside dynasty names, + /// and it maps every byte to exactly one char. + /// + public static string PdxText => _pdxText ??= Encoding.Latin1.GetString(PdxBytes); + + private static string FindRepoRoot() + { + var dir = new DirectoryInfo(AppContext.BaseDirectory); + while (dir != null && !File.Exists(Path.Combine(dir.FullName, "ParadoxSaveParser.sln"))) + dir = dir.Parent; + return dir?.FullName ?? Directory.GetCurrentDirectory(); + } + + /// + /// Produces the compact JSON twin of the save file, which is what jq is given. + /// Compact (not indented) so jq isn't penalized by parsing whitespace. + /// + public static void PrepareJson() + { + if (File.Exists(JsonPath)) + { + Console.WriteLine($"{JsonPath} already exists, skipping"); + return; + } + + Console.WriteLine($"parsing {PdxPath} ..."); + using var input = File.OpenRead(PdxPath); + var root = new SaveParserEU4(input, null).Parse(); + Console.WriteLine($"serializing to {JsonPath} ..."); + using var output = File.Create(JsonPath); + JsonSerializer.Serialize(output, root, new JsonSerializerOptions + { + WriteIndented = false, + Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping, + MaxDepth = 1024, + DefaultIgnoreCondition = JsonIgnoreCondition.WhenWritingNull, + }); + Console.WriteLine($"done: {new FileInfo(JsonPath).Length / 1024 / 1024} MB"); + } +} diff --git a/ParadoxSaveParser.Benchmarks/ExtractionBenchmarks.cs b/ParadoxSaveParser.Benchmarks/ExtractionBenchmarks.cs new file mode 100644 index 0000000..ef7c2a5 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/ExtractionBenchmarks.cs @@ -0,0 +1,85 @@ +using System.IO; +using System.Text.RegularExpressions; +using BenchmarkDotNet.Attributes; +using BenchmarkDotNet.Engines; +using ParadoxSaveParser.Lib; +using PCRE; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// In-process engines, all extracting countries.*.technology from the same 110 MB save. +/// Input is preloaded into memory, so disk I/O is not part of any measurement. +/// +[MemoryDiagnoser] +[SimpleJob(RunStrategy.Monitoring, launchCount: 1, warmupCount: 1, iterationCount: 5)] +public class ExtractionBenchmarks +{ + private byte[] _pdx = null!; + private string _pdxText = null!; + private byte[] _json = null!; + private ISearchExpression _query = null!; + + private Regex _flatInterpreted = null!; + private Regex _flatCompiled = null!; + private Regex _flatSourceGen = null!; + private Regex _pathAwareCompiled = null!; + private Regex _pathAwareNonBacktracking = null!; + private Regex _countriesBlock = null!; + private PcreRegex _pcreFlat = null!; + private PcreRegex _pcrePathAware = null!; + + [GlobalSetup] + public void Setup() + { + _pdx = BenchData.PdxBytes; + _pdxText = BenchData.PdxText; + _json = File.ReadAllBytes(BenchData.JsonPath); + _query = SearchExpressionCompiler.Compile(BenchData.PdxQuery); + + _flatInterpreted = new Regex(Extractors.FlatPattern, RegexOptions.None); + _flatCompiled = new Regex(Extractors.FlatPattern, RegexOptions.Compiled); + _flatSourceGen = Extractors.FlatSourceGen(); + _pathAwareCompiled = new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled); + _pathAwareNonBacktracking = new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking); + _countriesBlock = new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled); + _pcreFlat = new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled); + _pcrePathAware = new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled); + } + + [Benchmark(Baseline = true, Description = "SearchExpression (this project)")] + public long SearchExpression() => Extractors.SearchExpression(_pdx, _query); + + [Benchmark(Description = "Full parse, then select")] + public long FullParse() => Extractors.FullParseThenSelect(_pdx); + + [Benchmark(Description = ".NET Regex flat, interpreted")] + public long RegexFlatInterpreted() => Extractors.RegexScan(_flatInterpreted, _pdxText); + + [Benchmark(Description = ".NET Regex flat, compiled")] + public long RegexFlatCompiled() => Extractors.RegexScan(_flatCompiled, _pdxText); + + [Benchmark(Description = ".NET Regex flat, source generated")] + public long RegexFlatSourceGen() => Extractors.RegexScan(_flatSourceGen, _pdxText); + + [Benchmark(Description = ".NET Regex path aware, compiled")] + public long RegexPathAwareCompiled() => Extractors.RegexScan(_pathAwareCompiled, _pdxText); + + [Benchmark(Description = ".NET Regex path aware, NonBacktracking")] + public long RegexPathAwareNonBacktracking() => Extractors.RegexScan(_pathAwareNonBacktracking, _pdxText); + + [Benchmark(Description = ".NET Regex balanced block + flat")] + public long RegexBalancedTwoStage() => Extractors.RegexTwoStage(_countriesBlock, _flatCompiled, _pdxText); + + [Benchmark(Description = "PCRE.NET flat, JIT compiled")] + public long PcreFlat() => Extractors.PcreScan(_pcreFlat, _pdxText); + + [Benchmark(Description = "PCRE.NET path aware, JIT compiled")] + public long PcrePathAware() => Extractors.PcreScan(_pcrePathAware, _pdxText); + + [Benchmark(Description = "Utf8JsonReader over JSON twin")] + public long Utf8JsonReader() => Extractors.Utf8JsonReaderScan(_json); + + [Benchmark(Description = "JsonDocument over JSON twin")] + public long JsonDocument() => Extractors.JsonDocumentScan(_json); +} diff --git a/ParadoxSaveParser.Benchmarks/Extractors.cs b/ParadoxSaveParser.Benchmarks/Extractors.cs new file mode 100644 index 0000000..e180733 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/Extractors.cs @@ -0,0 +1,209 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Text.Json; +using System.Text.RegularExpressions; +using ParadoxSaveParser.Lib; +using PCRE; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// Every extractor solves the same task: pull the technology block +/// (adm/dip/mil levels) of every country out of an EU4 save, i.e. the path +/// countries.*.technology. +/// They all return the same checksum so the benchmarks can be verified against each other: +/// number_of_countries * 1_000_000 + sum_of_all_tech_levels. +/// +public static partial class Extractors +{ + public static long Checksum(long count, long techSum) => count * 1_000_000 + techSum; + + // ---------------------------------------------------------------- this project + + /// The parser of this repo: single streaming pass, guided by a compiled search expression. + public static long SearchExpression(byte[] pdx, ISearchExpression query) + { + using var stream = new MemoryStream(pdx, false); + var root = new SaveParserEU4(stream, query).Parse(); + return ChecksumOfParsedTree(root); + } + + /// Same query, but the parser reads from the given stream instead of a preloaded byte[]. + public static long SearchExpressionFromStream(Stream stream, ISearchExpression query) + { + var root = new SaveParserEU4(stream, query).Parse(); + return ChecksumOfParsedTree(root); + } + + /// Same parser without a query: parses the whole save into memory (worst case reference point). + public static long FullParseThenSelect(byte[] pdx) + { + using var stream = new MemoryStream(pdx, false); + var root = new SaveParserEU4(stream, null).Parse(); + return ChecksumOfParsedTree(root); + } + + private static long ChecksumOfParsedTree(Dictionary root) + { + long count = 0, sum = 0; + var countries = (Dictionary)root["countries"]; + foreach (var country in countries.Values) + { + if (country is not Dictionary c + || !c.TryGetValue("technology", out var t) + || t is not Dictionary tech) + continue; + count++; + foreach (var level in tech.Values) + sum += (long)level; + } + + return Checksum(count, sum); + } + + // ---------------------------------------------------------------- .NET Regex + + /// + /// Flat scan: matches technology blocks anywhere in the file. + /// Fastest regex option, but it does not verify that the match really sits under countries. + /// + public const string FlatPattern = + @"\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}"; + + /// + /// Path-aware: pairs a depth-1 country tag with the next technology block via a lazy gap. + /// This is what a regex user writes when the key must be associated with the value. + /// + public const string PathAwarePattern = + @"\n\t([A-Z0-9]{3})=\{[\s\S]*?\n\t\ttechnology=\{\n\t\t\tadm_tech=(\d+)\n\t\t\tdip_tech=(\d+)\n\t\t\tmil_tech=(\d+)\n\t\t\}"; + + /// + /// Balancing-group pattern isolating the whole countries={...} block. + /// .NET-only feature; PCRE would need recursion instead. + /// + public const string CountriesBlockPattern = + @"\ncountries=\{(?>[^{}]+|\{(?)|\}(?<-d>))*(?(d)(?!))\n\}"; + + public static long RegexScan(Regex regex, string text) + { + long count = 0, sum = 0; + foreach (Match m in regex.Matches(text)) + { + count++; + int g = m.Groups.Count - 3; + sum += int.Parse(m.Groups[g].ValueSpan) + + int.Parse(m.Groups[g + 1].ValueSpan) + + int.Parse(m.Groups[g + 2].ValueSpan); + } + + return Checksum(count, sum); + } + + /// Two stages: cut out the countries block with balanced braces, then scan inside it. + public static long RegexTwoStage(Regex blockRegex, Regex innerRegex, string text) + { + var block = blockRegex.Match(text); + if (!block.Success) + throw new Exception("countries block not found"); + return RegexScan(innerRegex, block.Value); + } + + [GeneratedRegex(FlatPattern)] + public static partial Regex FlatSourceGen(); + + [GeneratedRegex(PathAwarePattern)] + public static partial Regex PathAwareSourceGen(); + + // ---------------------------------------------------------------- PCRE.NET + + public static long PcreScan(PcreRegex regex, string text) + { + long count = 0, sum = 0; + foreach (var m in regex.Matches(text)) + { + count++; + int g = m.Groups.Count - 3; + sum += int.Parse(m[g].Value) + + int.Parse(m[g + 1].Value) + + int.Parse(m[g + 2].Value); + } + + return Checksum(count, sum); + } + + // ---------------------------------------------------------------- System.Text.Json + + private static readonly byte[] CountriesUtf8 = "countries"u8.ToArray(); + private static readonly byte[] TechnologyUtf8 = "technology"u8.ToArray(); + + /// Hand written streaming reader over the JSON twin: the closest JSON analogue of the search expression. + public static long Utf8JsonReaderScan(byte[] json) + { + long count = 0, sum = 0; + var reader = new Utf8JsonReader(json, isFinalBlock: true, state: default); + reader.Read(); // StartObject (root) + + while (reader.Read() && reader.TokenType == JsonTokenType.PropertyName) + { + bool isCountries = reader.ValueTextEquals(CountriesUtf8); + reader.Read(); + if (!isCountries) + { + reader.Skip(); + continue; + } + + // inside countries: property name = country tag + while (reader.Read() && reader.TokenType == JsonTokenType.PropertyName) + { + reader.Read(); + if (reader.TokenType != JsonTokenType.StartObject) + { + reader.Skip(); + continue; + } + + while (reader.Read() && reader.TokenType == JsonTokenType.PropertyName) + { + bool isTech = reader.ValueTextEquals(TechnologyUtf8); + reader.Read(); + if (!isTech) + { + reader.Skip(); + continue; + } + + count++; + while (reader.Read() && reader.TokenType == JsonTokenType.PropertyName) + { + reader.Read(); + sum += reader.GetInt64(); + } + } + } + + break; // countries handled, nothing else is needed + } + + return Checksum(count, sum); + } + + /// DOM approach: parse the whole JSON document, then navigate. The .NET equivalent of what jq does. + public static long JsonDocumentScan(byte[] json) + { + long count = 0, sum = 0; + using var doc = JsonDocument.Parse(json, new JsonDocumentOptions { MaxDepth = 1024 }); + foreach (var country in doc.RootElement.GetProperty("countries").EnumerateObject()) + { + if (country.Value.ValueKind != JsonValueKind.Object + || !country.Value.TryGetProperty("technology", out var tech)) + continue; + count++; + foreach (var level in tech.EnumerateObject()) + sum += level.Value.GetInt64(); + } + + return Checksum(count, sum); + } +} diff --git a/ParadoxSaveParser.Benchmarks/JqBenchmarks.cs b/ParadoxSaveParser.Benchmarks/JqBenchmarks.cs new file mode 100644 index 0000000..7ae7f22 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/JqBenchmarks.cs @@ -0,0 +1,79 @@ +using System; +using System.Diagnostics; +using BenchmarkDotNet.Attributes; +using BenchmarkDotNet.Engines; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// jq runs as an external process on the JSON twin of the save. +/// Numbers therefore include process start and reading the file from the OS page cache, +/// which is exactly what you pay when you shell out to jq. +/// measures that fixed overhead so it can be subtracted. +/// +[SimpleJob(RunStrategy.Monitoring, launchCount: 1, warmupCount: 1, iterationCount: 5)] +public class JqBenchmarks +{ + public static string JqPath { get; set; } = "jq"; + + [GlobalSetup] + public void Setup() => Run("--version", null); + + [Benchmark(Description = "jq .countries | map_values(.technology)")] + public int Jq() => Run("-c", BenchData.JqQuery); + + [Benchmark(Description = "jq empty (parse only)")] + public int JqParseOnly() => Run("-c", "empty"); + + [Benchmark(Description = "jq --stream (no DOM)")] + public int JqStream() => Run("-c --stream", "empty"); + + [Benchmark(Description = "jq process startup only (tiny input)")] + public int JqNoop() + { + var psi = new ProcessStartInfo(JqPath) + { + RedirectStandardOutput = true, + RedirectStandardInput = true, + UseShellExecute = false, + }; + psi.ArgumentList.Add("-c"); + psi.ArgumentList.Add("empty"); + using var p = Process.Start(psi)!; + p.StandardInput.Write("{}"); + p.StandardInput.Close(); + p.StandardOutput.ReadToEnd(); + p.WaitForExit(); + return p.ExitCode; + } + + private static int Run(string flags, string? program) + { + var psi = new ProcessStartInfo(JqPath) + { + RedirectStandardOutput = true, + RedirectStandardError = true, + UseShellExecute = false, + }; + foreach (var f in flags.Split(' ', StringSplitOptions.RemoveEmptyEntries)) + psi.ArgumentList.Add(f); + if (program != null) + { + psi.ArgumentList.Add(program); + psi.ArgumentList.Add(BenchData.JsonPath); + } + + using var p = Process.Start(psi)!; + // output must be drained, otherwise jq blocks on a full pipe + long bytes = 0; + var buf = new char[1 << 16]; + int read; + while ((read = p.StandardOutput.Read(buf, 0, buf.Length)) > 0) + bytes += read; + p.StandardError.ReadToEnd(); + p.WaitForExit(); + if (p.ExitCode != 0) + throw new Exception($"jq exited with {p.ExitCode}"); + return (int)(bytes & int.MaxValue); + } +} diff --git a/ParadoxSaveParser.Benchmarks/Mispairing.cs b/ParadoxSaveParser.Benchmarks/Mispairing.cs new file mode 100644 index 0000000..475ee47 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/Mispairing.cs @@ -0,0 +1,60 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Text.RegularExpressions; +using ParadoxSaveParser.Lib; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// Checks whether the regex approach associates the right country tag with the right +/// technology block. It cannot: 933 of the 2803 country blocks in the test save have no +/// technology block at all, so a lazy "tag ... technology" pattern silently pairs a tag +/// with a technology block belonging to a later country. +/// +public static class Mispairing +{ + public static Dictionary FromParser(byte[] pdx) + { + var result = new Dictionary(); + using var stream = new MemoryStream(pdx, false); + var root = new SaveParserEU4(stream, SearchExpressionCompiler.Compile(BenchData.PdxQuery)).Parse(); + foreach (var (tag, value) in (Dictionary)root["countries"]) + { + if (value is Dictionary c + && c.TryGetValue("technology", out var t) + && t is Dictionary tech + && tech.TryGetValue("adm_tech", out var adm)) + result[tag] = (long)adm; + } + + return result; + } + + public static Dictionary FromRegex(string text, Regex pathAware) + { + var result = new Dictionary(); + foreach (Match m in pathAware.Matches(text)) + result[m.Groups[1].Value] = long.Parse(m.Groups[2].ValueSpan); + return result; + } + + public static void Report(byte[] pdx, string text) + { + var expected = FromParser(pdx); + var actual = FromRegex(text, new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled)); + int missing = 0, wrong = 0, extra = 0; + foreach (var (tag, adm) in expected) + { + if (!actual.TryGetValue(tag, out long got)) { missing++; Console.WriteLine($" missing tag: {tag}"); } + else if (got != adm) wrong++; + } + + foreach (var tag in actual.Keys) + if (!expected.ContainsKey(tag)) + extra++; + + Console.WriteLine($"path aware regex pairing: {expected.Count} countries have technology, " + + $"regex reported {actual.Count}; missing={missing} wrong_value={wrong} invented={extra}"); + } +} diff --git a/ParadoxSaveParser.Benchmarks/ParadoxSaveParser.Benchmarks.csproj b/ParadoxSaveParser.Benchmarks/ParadoxSaveParser.Benchmarks.csproj new file mode 100644 index 0000000..3edacc6 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/ParadoxSaveParser.Benchmarks.csproj @@ -0,0 +1,20 @@ + + + Exe + net10.0 + latest + enable + disable + true + true + + + + + + + + + + + diff --git a/ParadoxSaveParser.Benchmarks/ParserInputBenchmarks.cs b/ParadoxSaveParser.Benchmarks/ParserInputBenchmarks.cs new file mode 100644 index 0000000..642a164 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/ParserInputBenchmarks.cs @@ -0,0 +1,29 @@ +using System.IO; +using BenchmarkDotNet.Attributes; +using BenchmarkDotNet.Engines; +using ParadoxSaveParser.Lib; + +namespace ParadoxSaveParser.Benchmarks; + +/// +/// SearchExpression only, parsing straight from the save file on disk: +/// the file is opened inside the benchmark and is never preloaded into a byte[]. +/// Run it once with and once without the internal MemoryStream copy in +/// to see what that copy buys. +/// +[MemoryDiagnoser] +[SimpleJob(RunStrategy.Monitoring, launchCount: 1, warmupCount: 1, iterationCount: 5)] +public class ParserInputBenchmarks +{ + private ISearchExpression _query = null!; + + [GlobalSetup] + public void Setup() => _query = SearchExpressionCompiler.Compile(BenchData.PdxQuery); + + [Benchmark(Description = "SearchExpression over FileStream")] + public long FromFile() + { + using var file = File.OpenRead(BenchData.PdxPath); + return Extractors.SearchExpressionFromStream(file, _query); + } +} diff --git a/ParadoxSaveParser.Benchmarks/Program.cs b/ParadoxSaveParser.Benchmarks/Program.cs new file mode 100644 index 0000000..f152946 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/Program.cs @@ -0,0 +1,80 @@ +using System; +using System.IO; +using System.Linq; +using System.Text.RegularExpressions; +using BenchmarkDotNet.Running; +using PCRE; +using ParadoxSaveParser.Lib; + +namespace ParadoxSaveParser.Benchmarks; + +public static class Program +{ + public static void Main(string[] args) + { + switch (args.FirstOrDefault()) + { + case "prepare": + BenchData.PrepareJson(); + return; + case "verify": + Verify(); + return; + default: + BenchmarkSwitcher.FromTypes([typeof(ExtractionBenchmarks), typeof(JqBenchmarks), typeof(ParserInputBenchmarks)]) + .Run(args); + return; + } + } + + /// + /// Runs every extractor once and prints its checksum and wall time, + /// so that differences in what each engine actually finds are visible before benchmarking. + /// + private static void Verify() + { + Console.WriteLine($"save: {BenchData.PdxPath} {new FileInfo(BenchData.PdxPath).Length / 1024 / 1024} MB"); + Console.WriteLine($"json: {BenchData.JsonPath} {new FileInfo(BenchData.JsonPath).Length / 1024 / 1024} MB"); + + var pdx = BenchData.PdxBytes; + var text = BenchData.PdxText; + var json = File.ReadAllBytes(BenchData.JsonPath); + var query = SearchExpressionCompiler.Compile(BenchData.PdxQuery); + + Time("SearchExpression", () => Extractors.SearchExpression(pdx, query)); + Time("FullParse", () => Extractors.FullParseThenSelect(pdx)); + Time("Regex flat compiled", + () => Extractors.RegexScan(new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text)); + Time("Regex path aware compiled", + () => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.Compiled), text)); + Time("Regex path aware nonbacktracking", + () => Extractors.RegexScan(new Regex(Extractors.PathAwarePattern, RegexOptions.NonBacktracking), text)); + Time("Regex balanced two stage", + () => Extractors.RegexTwoStage(new Regex(Extractors.CountriesBlockPattern, RegexOptions.Compiled), + new Regex(Extractors.FlatPattern, RegexOptions.Compiled), text)); + Time("PCRE flat", () => Extractors.PcreScan(new PcreRegex(Extractors.FlatPattern, PcreOptions.Compiled), text)); + Time("PCRE path aware", + () => Extractors.PcreScan(new PcreRegex(Extractors.PathAwarePattern, PcreOptions.Compiled), text)); + Time("Utf8JsonReader", () => Extractors.Utf8JsonReaderScan(json)); + Mispairing.Report(pdx, text); + Time("JsonDocument", () => Extractors.JsonDocumentScan(json)); + } + + private static void Time(string name, Func f) + { + var sw = System.Diagnostics.Stopwatch.StartNew(); + long checksum; + try + { + checksum = f(); + } + catch (Exception ex) + { + Console.WriteLine($"{name,-34} FAILED: {ex.Message}"); + return; + } + + sw.Stop(); + Console.WriteLine($"{name,-34} countries={checksum / 1_000_000,5} techSum={checksum % 1_000_000,7} {sw.ElapsedMilliseconds,7} ms"); + } +} diff --git a/ParadoxSaveParser.Benchmarks/README.md b/ParadoxSaveParser.Benchmarks/README.md new file mode 100644 index 0000000..d110d74 --- /dev/null +++ b/ParadoxSaveParser.Benchmarks/README.md @@ -0,0 +1,111 @@ +# ParadoxSaveParser.Benchmarks + +Compares the query engine of this repo (`SearchExpression` + `SaveParserEU4`) against +regex engines and `jq` on one realistic task: + +> extract the `technology` block (adm/dip/mil level) of every country from an EU4 save + +| engine | query | +|---|---| +| this repo | `countries.*.technology` | +| jq | `.countries \| map_values(.technology)` | +| .NET `Regex` / PCRE.NET | see `Extractors.FlatPattern` / `Extractors.PathAwarePattern` | +| `Utf8JsonReader` / `JsonDocument` | hand written navigation | + +## Corpus + +* `gamestate` — a real 109 MB EU4 1.37 save (text format, 6.0M lines, 2803 country blocks, + 1870 of which have a `technology` block). +* `gamestate.min.json` — 92 MB compact JSON twin, produced from the save by this project's + own parser. It exists because jq and the `System.Text.Json` engines cannot read the + Paradox format at all. + +The save is **not** in the repository. Point `PSP_BENCH_DATA` at the directory holding it, +or drop it in `ParadoxSaveParser.CLI/bin/Debug/net10.0/` (the default location). + +## Usage + +```sh +dotnet build ParadoxSaveParser.Benchmarks -c Release +cd ParadoxSaveParser.Benchmarks/bin/Release/net10.0 + +./ParadoxSaveParser.Benchmarks.exe prepare # build the JSON twin (once) +./ParadoxSaveParser.Benchmarks.exe verify # one run of every engine + correctness check +./ParadoxSaveParser.Benchmarks.exe --filter '*' # full BenchmarkDotNet run +``` + +Every engine returns the checksum `countries * 1_000_000 + sum_of_tech_levels`, so `verify` +shows immediately when an engine finds something different from the others. + +## Fairness notes + +* All in-process engines read from memory; file I/O is excluded. jq is an external process, + so its numbers include process start (measured separately by the `jq startup only` + benchmark) and reading the JSON from the OS page cache. +* The regex engines work on the save text, which is 109 MB; jq works on 92 MB of JSON. +* jq is the only engine here that is a general query language: it can do arithmetic, + filtering and reshaping, which neither the regexes nor `SearchExpression` can. + +## Results + +AMD Ryzen 7 5700X, 16 logical cores, Windows 10 21H2, .NET 10.0.12, jq 1.8.2, +109 MB save / 92 MB JSON twin, BenchmarkDotNet `RunStrategy.Monitoring`, 5 iterations. + +| engine | mean | vs this project | allocated | correct? | +|---|---:|---:|---:|---| +| .NET Regex flat, source generated | 14.8 ms | 0.02x | 1.2 MB | path-blind | +| .NET Regex flat, compiled | 17.0 ms | 0.03x | 1.2 MB | path-blind | +| .NET Regex flat, interpreted | 19.3 ms | 0.03x | 1.2 MB | path-blind | +| PCRE.NET flat, JIT compiled | 97.4 ms | 0.15x | 0.9 MB | path-blind | +| .NET Regex balanced block + flat | 112.3 ms | 0.17x | 224 MB | yes | +| .NET Regex path aware, compiled | 116.2 ms | 0.18x | 1.4 MB | 1869/1870 | +| PCRE.NET path aware, JIT compiled | 122.9 ms | 0.19x | 0.9 MB | 1869/1870 | +| Utf8JsonReader over JSON twin | 155.7 ms | 0.24x | 0 B | yes | +| JsonDocument over JSON twin | 321.4 ms | 0.50x | 0 B (native) | yes | +| .NET Regex path aware, NonBacktracking | 444.5 ms | 0.69x | 32 MB | 1869/1870 | +| **SearchExpression (this project)** | **645.1 ms** | **1.00x** | **1.6 MB** | yes | +| Full parse, then select | 1831 ms | 2.84x | 1153 MB | yes | +| jq `.countries \| map_values(.technology)` | 4340 ms | 6.73x | n/a | yes | +| jq `empty` (parse the file, emit nothing) | 4227 ms | 6.55x | n/a | n/a | +| jq `--stream` | 17490 ms | 27.1x | n/a | n/a | +| jq process startup only | 4.2 ms | 0.01x | n/a | n/a | + +### Reading the table + +* **jq is ~6.7x slower than this parser** and 97% of that time is JSON parsing, not the + query: `jq empty` on the same file costs 4227 ms of the 4340 ms. Process startup is + negligible (4 ms). jq's `--stream` mode, often recommended for large inputs, is 4x + *slower* still. jq also needs the data converted to JSON first, which this parser has to + do anyway — so end to end jq is strictly more expensive here. +* **The regexes are 5-40x faster, but they are not doing the same job.** A regex never + parses the structure; it scans bytes for a literal and validates a short window around it. + `Regex` with a literal prefix (`technology={`) is vectorized, so 109 MB is scanned at + several GB/s. That speed is real and the cost is real too: nothing verifies that the hit + is under `countries`, at the right depth, or belongs to the tag matched before it. +* **Path-aware regexes lose both the speed and the correctness.** Forcing the tag into the + pattern costs 7x (17 ms -> 116 ms) and still returns 1869 of 1870 countries: the tag class + `[A-Z0-9]{3}` silently drops EU4's `---` pseudo-country. Pairing survives here only by + luck of the file layout — all 933 tech-less country blocks happen to be grouped at the end + of the save, so the lazy gap never runs across one. A save that interleaves them would + make the regex report a technology block under the wrong country tag with no error. +* **`RegexOptions.NonBacktracking`** guarantees linear time but is 26x slower than the + compiled backtracking engine on this pattern and allocates 32 MB. +* **PCRE.NET** (the most used non-BCL regex engine in .NET, ~460k downloads) is 6x slower + than `System.Text.RegularExpressions` on the flat pattern, because .NET's vectorized + literal prefix search beats PCRE2's JIT here. On the path-aware pattern the two are equal. + There is no reason to leave the BCL engine for this workload. +* **The query is what makes this parser fast, not the parsing.** Same parser, same file: + 645 ms with `countries.*.technology`, 1831 ms and 1.1 GB allocated without a query. The + search expression prunes ~65% of the work and 99.9% of the allocations. +* **Against a JSON reader on equivalent data**, the parser is 4x slower than `Utf8JsonReader` + and 2x slower than `JsonDocument`. Those numbers exclude the pdx -> JSON conversion + (~4.7 s), so they are a ceiling for the format, not a usable alternative. + +### Where this parser's time goes + +645 ms for 109 MB is ~170 MB/s, or ~13 cycles per input byte. The lexer reads the save one +byte at a time through `Stream.ReadByte()` (`SaveParserEU4.LexTextSave`), which is a virtual +call plus bounds check per byte, and appends char by char into a `StringBuilder`. Reading +into a `byte[]` buffer and scanning it with `ReadOnlySpan.IndexOfAny` would be the +first thing to try — the regex numbers above show what the same hardware does when it scans +a span instead of a stream. diff --git a/ParadoxSaveParser.sln b/ParadoxSaveParser.sln index 6ce891b..f11acf5 100644 --- a/ParadoxSaveParser.sln +++ b/ParadoxSaveParser.sln @@ -15,27 +15,80 @@ Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "ParadoxSaveParser.Lib.Tests EndProject Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "ParadoxSaveParser.CLI", "ParadoxSaveParser.CLI\ParadoxSaveParser.CLI.csproj", "{2D4448A6-390D-47F3-9BB7-6266669719DE}" EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "ParadoxSaveParser.Benchmarks", "ParadoxSaveParser.Benchmarks\ParadoxSaveParser.Benchmarks.csproj", "{0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}" +EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution Debug|Any CPU = Debug|Any CPU + Debug|x64 = Debug|x64 + Debug|x86 = Debug|x86 Release|Any CPU = Release|Any CPU + Release|x64 = Release|x64 + Release|x86 = Release|x86 EndGlobalSection GlobalSection(ProjectConfigurationPlatforms) = postSolution {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|Any CPU.ActiveCfg = Debug|Any CPU {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|Any CPU.Build.0 = Debug|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|x64.ActiveCfg = Debug|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|x64.Build.0 = Debug|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|x86.ActiveCfg = Debug|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Debug|x86.Build.0 = Debug|Any CPU {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|Any CPU.ActiveCfg = Release|Any CPU {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|Any CPU.Build.0 = Release|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|x64.ActiveCfg = Release|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|x64.Build.0 = Release|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|x86.ActiveCfg = Release|Any CPU + {7D377558-CE40-4D7E-B0B1-FB33C475AE6A}.Release|x86.Build.0 = Release|Any CPU {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|Any CPU.ActiveCfg = Debug|Any CPU {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|Any CPU.Build.0 = Debug|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|x64.ActiveCfg = Debug|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|x64.Build.0 = Debug|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|x86.ActiveCfg = Debug|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Debug|x86.Build.0 = Debug|Any CPU {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|Any CPU.ActiveCfg = Release|Any CPU {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|Any CPU.Build.0 = Release|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|x64.ActiveCfg = Release|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|x64.Build.0 = Release|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|x86.ActiveCfg = Release|Any CPU + {53ED0135-9513-4DE2-9187-CF2899F179B3}.Release|x86.Build.0 = Release|Any CPU {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|Any CPU.ActiveCfg = Debug|Any CPU {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|Any CPU.Build.0 = Debug|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|x64.ActiveCfg = Debug|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|x64.Build.0 = Debug|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|x86.ActiveCfg = Debug|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Debug|x86.Build.0 = Debug|Any CPU {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|Any CPU.ActiveCfg = Release|Any CPU {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|Any CPU.Build.0 = Release|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|x64.ActiveCfg = Release|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|x64.Build.0 = Release|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|x86.ActiveCfg = Release|Any CPU + {23F4BE1B-3043-4821-9F65-74FF5F57FA59}.Release|x86.Build.0 = Release|Any CPU {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|Any CPU.ActiveCfg = Debug|Any CPU {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|Any CPU.Build.0 = Debug|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|x64.ActiveCfg = Debug|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|x64.Build.0 = Debug|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|x86.ActiveCfg = Debug|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Debug|x86.Build.0 = Debug|Any CPU {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|Any CPU.ActiveCfg = Release|Any CPU {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|Any CPU.Build.0 = Release|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|x64.ActiveCfg = Release|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|x64.Build.0 = Release|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|x86.ActiveCfg = Release|Any CPU + {2D4448A6-390D-47F3-9BB7-6266669719DE}.Release|x86.Build.0 = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|Any CPU.Build.0 = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|x64.ActiveCfg = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|x64.Build.0 = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|x86.ActiveCfg = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Debug|x86.Build.0 = Debug|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|Any CPU.ActiveCfg = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|Any CPU.Build.0 = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|x64.ActiveCfg = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|x64.Build.0 = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|x86.ActiveCfg = Release|Any CPU + {0E35F6F0-3D74-4C18-BFC8-A1C45A04A410}.Release|x86.Build.0 = Release|Any CPU + EndGlobalSection + GlobalSection(SolutionProperties) = preSolution + HideSolutionNode = FALSE EndGlobalSection EndGlobal diff --git a/README.md b/README.md index e28356a..eba3d4e 100644 --- a/README.md +++ b/README.md @@ -1,18 +1,20 @@ -# Paradox Save Parser -Yet another save files parser. - -## Supported games: -- EU4 - -## Project structure -- **[ParadoxSaveParser.Lib](./ParadoxSaveParser.Lib)** - - Parser itself -- **[ParadoxSaveParser.Lib.Tests](./ParadoxSaveParser.Lib.Tests)** - - Tests for parser -- **[ParadoxSaveParser.CLI](./ParadoxSaveParser.CLI)** - - Command line tool to parse save files. Can be used in interactive mode. -- **[ParadoxSaveParser.WebAPI](./ParadoxSaveParser.WebAPI)** - - Backend for my save file analytics website (TODO: add repo link). - -## Building -1. Install protobuf compiler https://protobuf.dev/installation/ +# Paradox Save Parser +Yet another save files parser. + +## Supported games: +- EU4 + +## Project structure +- **[ParadoxSaveParser.WebAPI](./ParadoxSaveParser.WebAPI)** - + Backend for my save file analytics website (TODO: add repo link). +- **[ParadoxSaveParser.CLI](./ParadoxSaveParser.CLI)** - + Command line tool to parse save files. Can be used in interactive mode. +- **[ParadoxSaveParser.Lib](./ParadoxSaveParser.Lib)** - + Parser itself +- **[ParadoxSaveParser.Lib.Tests](./ParadoxSaveParser.Lib.Tests)** - + Tests for parser +- **[ParadoxSaveParser.Benchmarks](./ParadoxSaveParser.Benchmarks)** - + Performance comparison of the search expression engine against regex engines and jq + +## Building +1. Install protobuf compiler https://protobuf.dev/installation/