Files
ParadoxSaveParser/ParadoxSaveParser.Lib/SaveParserEU4.cs
T
2026-09-15 13:18:09 +02:00

271 lines
9.3 KiB
C#

global using System;
global using System.Collections.Generic;
global using System.IO;
global using System.Text;
using System.Globalization;
namespace ParadoxSaveParser.Lib;
/// <summary>
/// Sequential parser that doesn't cache anything.
/// </summary>
public class SaveParserEU4
{
private const int DefaultBufferSize = 64 * 1024;
private static ReadOnlySpan<byte> Header => "EU4txt"u8;
/// <summary>
/// Windows-1252 is the default encoding of the save files
/// </summary>
public static Encoding DefaultEncoding => ParadoxEncodings.Windows1252;
private readonly Tokenizer _tokens;
private readonly SearchExpressionCompilation? _query;
// step of the query the parser is currently at, compiled by Parse()
private ISearchExpression? _searchExprCurrent;
/// <summary>
/// Encoding of the strings inside the save. Saves of different localizations use
/// different ones, so it can be changed; <see cref="DefaultEncoding" /> is what EU4 writes.
/// </summary>
public Encoding Encoding { get; }
/// <param name="savefile">
/// Uncompressed stream of <c>gamestate</c> file which can be extracted from save archive
/// </param>
/// <param name="query">
/// Parsing whole save may take a few seconds on mid pc and takes 1GB of RAM,
/// so you should specify what exactly you want to get from save file
/// </param>
/// <param name="encoding">Encoding of the strings inside the save. <see cref="DefaultEncoding"/></param>
/// <param name="bufferSize">Size of the read buffer. Mostly useful for tests.</param>
public SaveParserEU4(Stream savefile, SearchExpressionCompilation? query,
Encoding? encoding = null, int bufferSize = DefaultBufferSize)
{
Encoding = encoding ?? DefaultEncoding;
_query = query;
_tokens = new Tokenizer(savefile, bufferSize);
}
public Dictionary<string, object> Parse()
{
// compiled here, so the keys of the query are always encoded like the save
_searchExprCurrent = _query?.Compile(Encoding);
_tokens.ReadHeader(Header, Encoding);
return ParseDict();
}
// doesn't move next
private object? ParseValue()
{
switch (_tokens.Type)
{
case TokenType.StringOrNumber:
return ParseScalar(_tokens.Text);
case TokenType.BracketOpen:
return ParseListOrDict();
case TokenType.BracketClose:
return null;
default:
throw new UnexpectedTokenException(_tokens, Encoding);
}
}
private object ParseScalar(ReadOnlySpan<byte> text)
{
// string values can be empty
if (text.Length == 0)
return string.Empty;
if (text.SequenceEqual("yes"u8))
return true;
if (text.SequenceEqual("no"u8))
return false;
byte first = text[0];
if (first != (byte)'-' && !char.IsAsciiDigit((char)first))
return DecodeString(text);
if (text.Contains((byte)'.')
&& double.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out double d))
return d;
if (long.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out long l))
return l;
return DecodeString(text);
}
private string DecodeString(ReadOnlySpan<byte> text)
{
// Skip control characters, which are invisible and causing frontend bugs.
// I dont know why there are so many of them in strings.
if (!text.ContainsAnyInRange((byte)0, (byte)0x1F))
return Encoding.GetString(text);
Span<byte> cleaned = text.Length <= 256 ? stackalloc byte[text.Length] : new byte[text.Length];
int length = 0;
foreach (byte b in text)
if (b >= 0x20)
cleaned[length++] = b;
return Encoding.GetString(cleaned[..length]);
}
// skips next value
/// <returns>true if skipped value, false if current token is closing bracket</returns>
private bool SkipValue()
{
switch (_tokens.Type)
{
case TokenType.BracketOpen:
_tokens.SkipBlock();
return true;
case TokenType.StringOrNumber:
return true;
case TokenType.BracketClose:
return false;
default:
throw new UnexpectedTokenException(_tokens, Encoding);
}
}
private static bool IsEmptyCollection(object value)
=> value is Dictionary<string, object> { Count: 0 } or List<object> { Count: 0 };
// doesn't move next
private object ParseListOrDict()
{
if (_tokens.PeekType(1) == TokenType.StringOrNumber && _tokens.PeekType(2) == TokenType.Equals)
return ParseDict();
return ParseList();
}
// moves next
private List<object> ParseList()
{
List<object> list = new();
for (int i = 0; ; i++)
{
if (!_tokens.Read())
throw new Exception("Unexpected end of file");
ISearchExpression? searchExprNext = null;
if (_searchExprCurrent != null
&& !_searchExprCurrent.DoesMatch(new MatchCandidate(i), out searchExprNext))
{
if (!SkipValue())
break;
continue;
}
var searchExprPrev = _searchExprCurrent;
_searchExprCurrent = searchExprNext;
object? value = ParseValue();
_searchExprCurrent = searchExprPrev;
if (value is null)
break;
// do dot add empty collections into list
if (IsEmptyCollection(value))
continue;
list.Add(value);
}
return list;
}
// moves next
private Dictionary<string, object> ParseDict()
{
Dictionary<string, object> dict = new();
// root is a dict without closing bracket, so this method must check for end of file
for (int localIndex = 0; _tokens.Read(); localIndex++)
{
// end of dictionary
if (_tokens.Type == TokenType.BracketClose)
break;
// Saves may contain some blocks without key.
// Such blocks are skipped because idk where to put them.
// Example: `technology_group=tech_cannorian{ }
// { } { } { }`
if (_tokens.Type == TokenType.BracketOpen)
{
_tokens.SkipBlock();
continue;
}
if (_tokens.Type != TokenType.StringOrNumber)
throw new UnexpectedTokenException(_tokens, Encoding);
// The key is matched before the value is read, so that a key rejected by the query
// never has to become a string.
ISearchExpression? searchExprNext = null;
bool matches = _searchExprCurrent == null
|| _searchExprCurrent.DoesMatch(
new MatchCandidate(localIndex, _tokens.Text), out searchExprNext);
string? keyStr = matches ? DecodeString(_tokens.Text) : null;
// next token should be `=` or `{`
if (!_tokens.Read())
throw new UnexpectedTokenException(_tokens, Encoding);
if (_tokens.Type == TokenType.Equals)
{
// skip `=`
if (!_tokens.Read())
throw new UnexpectedTokenException(_tokens, Encoding);
}
// Saves may contain object definition without `=`.
// Example: `map_area_data {` instead of `map_area_data = {`
else if (_tokens.Type != TokenType.BracketOpen)
{
throw new UnexpectedTokenException(_tokens, Encoding);
}
if (!matches)
{
if (!SkipValue())
throw new UnexpectedTokenException(_tokens, Encoding);
continue;
}
var searExpressionPrevious = _searchExprCurrent;
_searchExprCurrent = searchExprNext;
object? value = ParseValue();
if (value is null)
throw new UnexpectedTokenException(_tokens, Encoding);
_searchExprCurrent = searExpressionPrevious;
// Paradox save format has another way of defining list:
// a = 1
// a = 2
// It means `a = { 1 2 }`
if (dict.TryGetValue(keyStr!, out var firstValue))
{
// Do dot add empty collections into list.
// `key:{}` is okay, but i don't want to see `key:[{},{},{},{},{},{}]`
if (IsEmptyCollection(value))
continue;
if (firstValue is List<object> existingList)
existingList.Add(value);
else dict[keyStr!] = new List<object> { firstValue, value };
}
else
{
dict.Add(keyStr!, value);
}
}
return dict;
}
internal class UnexpectedTokenException : Exception
{
public UnexpectedTokenException(Tokenizer tokens, Encoding encoding) :
base($"Unexpected token: {tokens.Line}:{tokens.Column} '{tokens.Describe(encoding)}'")
{
}
}
}