feat: Adds Latin1 text support (#2317)
## Summary - Adds proper Latin1 encoding support, replacing CP1252 usage throughout the codebase - Adds specialized, optimized string decoding methods with safe string filtering for each encoding type - Filters invalid Unicode characters (C0/C1 control codes, non-characters) by removal rather than replacement since the UO client renders nothing for these characters - Fixes UTF-16 null terminator position handling to correctly advance by 2 bytes ## Changes TextEncoding.cs - Added SearchValues-based invalid byte/char detection for efficient filtering - Added encoding-specific GetString methods: GetStringAscii, GetStringLatin1, GetStringUtf8, GetStringBigUni, GetStringLittleUni - Each method supports a safeString parameter for filtering invalid characters - Little-endian UTF-16 uses direct memory cast for zero-copy decoding on LE systems - Invalid characters are removed (not replaced with U+FFFD) since the client renders nothing for them SpanReader.cs - Added ReadLatin1() and ReadLatin1Safe() methods - Rewrote encoding-specific read methods to use optimized TextEncoding.GetString* methods - Fixed UTF-16 null terminator handling: position now correctly advances by byteLength (2) instead of 1 SpanWriter.cs - Added WriteLatin1 and WriteLatin1Null methods ## Packet Updates - Updated all packet code to use Latin1 encoding instead of CP1252 - Affected: account packets, equipment packets, menu packets, message packets, mobile packets, player packets, secure trade packets, vendor packets, gump packets, book packets, mahjong packets ## Filtering Behavior Invalid characters filtered in safe mode: ``` ┌───────────────┬────────────────────────┐ │ Range │ Description │ ├───────────────┼────────────────────────┤ │ 0x00-0x1F │ C0 control codes │ ├───────────────┼────────────────────────┤ │ 0x7F │ DEL │ ├───────────────┼────────────────────────┤ │ 0x80-0x9F │ C1 control codes │ ├───────────────┼────────────────────────┤ │ 0xFFFE-0xFFFF │ Unicode non-characters │ └───────────────┴────────────────────────┘ ``` Note: Surrogate pairs (0xD800-0xDFFF) are not filtered because proper validation requires context checking for paired vs unpaired surrogates. The UO client renders nothing for these anyway. ## Test Plan - All 631 Server.Tests pass - Verified client rendering behavior using TestUnicodeGump command (pages 1-5) - Confirmed U+FFFD, unpaired surrogates, and non-characters all render as blank in client - Verified Latin1 characters (0xA0-0xFF) display correctly - Verified C1 control codes (0x80-0x9F) are filtered and don't display
This commit is contained in:
parent
1e4c32c809
commit
0404251638
27 changed files with 735 additions and 96 deletions
|
|
@ -291,7 +291,8 @@ public class SpanReaderTests
|
|||
var result = reader.ReadLittleUni();
|
||||
|
||||
Assert.Equal("Hi", result);
|
||||
Assert.Equal(5, reader.Position);
|
||||
// Position 6: 4 bytes for "Hi" + 2 bytes for UTF-16 null terminator
|
||||
Assert.Equal(6, reader.Position);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
|
|
@ -309,12 +310,14 @@ public class SpanReaderTests
|
|||
[Fact]
|
||||
public void TestReadLittleUniSafe()
|
||||
{
|
||||
ReadOnlySpan<byte> buffer = [(byte)'H', 0, 0xFF, 0xD8, (byte)'i', 0];
|
||||
// Test with C1 control code (0x85 = NEL) which should be filtered
|
||||
ReadOnlySpan<byte> buffer = [(byte)'H', 0, 0x85, 0x00, (byte)'i', 0];
|
||||
var reader = new SpanReader(buffer);
|
||||
|
||||
var result = reader.ReadLittleUniSafe();
|
||||
|
||||
Assert.Equal("H\uFFFDi", result);
|
||||
// C1 control (0x0085) removed - client renders nothing for invalid chars
|
||||
Assert.Equal("Hi", result);
|
||||
Assert.Equal(6, reader.Position);
|
||||
}
|
||||
|
||||
|
|
@ -339,7 +342,8 @@ public class SpanReaderTests
|
|||
var result = reader.ReadBigUni();
|
||||
|
||||
Assert.Equal("Hi", result);
|
||||
Assert.Equal(5, reader.Position);
|
||||
// Position 6: 4 bytes for "Hi" + 2 bytes for UTF-16 null terminator
|
||||
Assert.Equal(6, reader.Position);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
|
|
@ -357,12 +361,14 @@ public class SpanReaderTests
|
|||
[Fact]
|
||||
public void TestReadBigUniSafe()
|
||||
{
|
||||
ReadOnlySpan<byte> buffer = [0, (byte)'H', 0xD8, 0xFF, 0, (byte)'i'];
|
||||
// Test with C1 control code (0x0085 = NEL) which should be filtered
|
||||
ReadOnlySpan<byte> buffer = [0, (byte)'H', 0x00, 0x85, 0, (byte)'i'];
|
||||
var reader = new SpanReader(buffer);
|
||||
|
||||
var result = reader.ReadBigUniSafe();
|
||||
|
||||
Assert.Equal("H\uFFFDi", result);
|
||||
// C1 control (0x0085) removed - client renders nothing for invalid chars
|
||||
Assert.Equal("Hi", result);
|
||||
Assert.Equal(6, reader.Position);
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue