Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,8 @@ float confidence = resultDetected.Confidence;
IList<DetectionDetail> allDetails = result.Details;
```

Byte-array and span detection use the same 1024-byte batches as stream detection. Detection can stop once a confident match is found, without inspecting the remaining input. This is an encoding heuristic, not validation that every byte belongs to the detected encoding.

### Asynchronous Methods

```c#
Expand Down Expand Up @@ -142,3 +144,7 @@ For some encodings no alias is available: `cp949`, `iso-2022-cn`, `euc-tw`, `iso
The library is subject to the Mozilla Public License Version 1.1 (the "License"). Alternatively, it may be used under the terms of either the GNU General Public License Version 2 or later (the "GPL"), or the GNU Lesser General Public License Version 2.1 or later (the "LGPL").

Test data has been extracted from [Wikipedia](https://wikipedia.org) and [The Project Gutenberg](https://www.gutenberg.org/) books and is subject to their licenses.

### Modifications

2026-10-09: Byte-array and span detection now process input in stream-sized batches and can stop at a confident match; input-method consistency tests and the sampling documentation were added. These modifications are derived from the Mozilla Universal charset detector code originally provided by Netscape Communications Corporation.
8 changes: 7 additions & 1 deletion src/CharsetDetector.cs
Original file line number Diff line number Diff line change
Expand Up @@ -127,7 +127,13 @@ protected CharsetDetector()
public static DetectionResult DetectFromBytes(ReadOnlySpan<byte> bytes)
{
var detector = new CharsetDetector();
detector.Feed(bytes);
while (!bytes.IsEmpty && !detector._done)
{
var toRead = Math.Min(bytes.Length, BufferSize);
detector.Feed(bytes.Slice(0, toRead));
bytes = bytes.Slice(toRead);
}

return detector.DataEnd();
}

Expand Down
103 changes: 103 additions & 0 deletions tests/InputMethodConsistencyTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
using System;
using System.IO;
using System.Text;
using System.Threading.Tasks;
using NUnit.Framework;
using UtfUnknown.Core;

namespace UtfUnknown.Tests;

public class InputMethodConsistencyTests
{
[TestCase("array")]
[TestCase("span")]
[TestCase("slice")]
public void LateInvalidByteMatchesStreamEarlyDetection(string inputMethod)
{
// Streams already stop probing after a confident prefix. A whole-buffer
// feed sees the late invalid byte before reaching that decision.
var prefix = Encoding.UTF8.GetBytes(new string('é', 2048));
var bytes = new byte[prefix.Length + 1];
prefix.CopyTo(bytes, 0);
bytes[bytes.Length - 1] = 0xff;

using var stream = new MemoryStream(bytes);
var expected = CharsetDetector.DetectFromStream(stream);
Assert.That(expected.Detected.EncodingName, Is.EqualTo(CodepageName.UTF8));
Assert.That(stream.Position, Is.LessThan(bytes.Length));
AssertSameResult(Detect(bytes, inputMethod), expected);
}

[Test]
public void Utf8WithLateEmojiMatchesStream()
{
var bytes = Encoding.UTF8.GetBytes(new string('é', 2048) + "😀");
using var stream = new MemoryStream(bytes);
AssertSameResult(CharsetDetector.DetectFromBytes(bytes), CharsetDetector.DetectFromStream(stream));
}

[TestCase(1023)]
[TestCase(1024)]
[TestCase(1025)]
public void MultibyteCharacterCrossingChunkBoundaryMatchesStream(int prefixLength)
{
var bytes = Encoding.UTF8.GetBytes(new string('a', prefixLength) + "€é日本語");
using var stream = new MemoryStream(bytes);
var expected = CharsetDetector.DetectFromStream(stream);
Assert.That(expected.Detected.EncodingName, Is.EqualTo(CodepageName.UTF8));
AssertSameResult(CharsetDetector.DetectFromBytes(bytes.AsSpan()), expected);
}

[Test]
public async Task ByteDetectionAlsoMatchesAsyncStream()
{
var bytes = Encoding.UTF8.GetBytes(new string('é', 2048) + "😀");
using var stream = new MemoryStream(bytes);
AssertSameResult(CharsetDetector.DetectFromBytes(bytes), await CharsetDetector.DetectFromStreamAsync(stream));
}

[Test]
public void SelectedSliceStillDetectsBomAndExcludesSurroundingBytes()
{
var bytes = new byte[] { 0xff, 0xef, 0xbb, 0xbf, (byte)'a', 0xff };
var result = CharsetDetector.DetectFromBytes(bytes, 1, 4);
Assert.That(result.Detected.EncodingName, Is.EqualTo(CodepageName.UTF8));
Assert.That(result.Detected.Confidence, Is.EqualTo(1.0f));
Assert.That(result.Detected.HasBOM, Is.True);
}

[Test]
public void EmptyInputStillHasNoDetectedEncoding()
{
Assert.That(CharsetDetector.DetectFromBytes(Array.Empty<byte>()).Detected, Is.Null);
Assert.That(CharsetDetector.DetectFromBytes(ReadOnlySpan<byte>.Empty).Detected, Is.Null);
}

[Test]
public void NullArrayStillThrows()
{
Assert.Throws<ArgumentNullException>(() => CharsetDetector.DetectFromBytes((byte[])null));
}

private static DetectionResult Detect(byte[] bytes, string inputMethod)
{
if (inputMethod == "span")
return CharsetDetector.DetectFromBytes(bytes.AsSpan());
if (inputMethod == "slice")
{
var surrounded = new byte[bytes.Length + 2];
surrounded[0] = surrounded[surrounded.Length - 1] = 0x00;
bytes.CopyTo(surrounded, 1);
return CharsetDetector.DetectFromBytes(surrounded, 1, bytes.Length);
}
return CharsetDetector.DetectFromBytes(bytes);
}

private static void AssertSameResult(DetectionResult actual, DetectionResult expected)
{
Assert.That(actual.Detected, Is.Not.Null);
Assert.That(actual.Detected.EncodingName, Is.EqualTo(expected.Detected.EncodingName));
Assert.That(actual.Detected.Confidence, Is.EqualTo(expected.Detected.Confidence));
Assert.That(actual.Detected.HasBOM, Is.EqualTo(expected.Detected.HasBOM));
}
}
Loading