diff --git a/sources/EncodingChecker.Tests/EncodingAmbiguityTests.cs b/sources/EncodingChecker.Tests/EncodingAmbiguityTests.cs new file mode 100644 index 0000000..2ffc329 --- /dev/null +++ b/sources/EncodingChecker.Tests/EncodingAmbiguityTests.cs @@ -0,0 +1,120 @@ +using System.Text; + +namespace EncodingChecker.Tests; + +/// +/// A corpus audit of 5,078 files found 262 where the bytes do not identify the encoding +/// that wrote them: single-byte code pages map 256 values independently, so a file valid +/// in windows-1252 is equally valid in iso-8859-1 and no inspection decides between them. +/// Detection still answers, and converting on that answer rewrites the file into one of +/// several possible readings without saying so. +/// +/// These pin the distinction the refusal rests on. Two failure directions matter equally: +/// converting a genuinely ambiguous file, and refusing one whose encoding the bytes do +/// determine. A gate that refuses everything is not safe, it is broken. +/// +public sealed class EncodingAmbiguityTests +{ + private static AmbiguityAnalysis Analyze(string text, string charset) => + EncodingAmbiguity.Analyze( + Encoding.GetEncoding(charset).GetBytes(text), + Encoding.GetEncoding(charset)); + + [Theory] + [InlineData("utf-8", "Hello 世界 café")] + [InlineData("shift_jis", "こんにちは世界。日本語のテキスト")] + [InlineData("euc-jp", "こんにちは世界")] + [InlineData("big5", "你好世界。這是繁體中文")] + [InlineData("gb18030", "这是简体中文文本")] + [InlineData("euc-kr", "안녕하세요 세계")] + [InlineData("utf-16", "Hello 世界 café")] + public void StructuredEncodingsAreNotTreatedAsAmbiguous(string charset, string text) + { + // These constrain their byte sequences, so a file valid under one was not valid + // by accident. Other codecs "reading" it differently are codecs that cannot + // refuse anything, which is not a competing claim. + AmbiguityAnalysis analysis = Analyze(text, charset); + + Assert.NotEqual(AmbiguityClass.TextChanging, analysis.Class); + Assert.True(analysis.IsSafeToConvertAutomatically); + } + + [Theory] + [InlineData("windows-1252", "Le café était déjà prêt")] + [InlineData("koi8-r", "Привет мир")] + [InlineData("iso-8859-7", "Γειά σου κόσμε")] + public void SingleByteTextWithNoDistinguishingStructureIsAmbiguous( + string charset, string text) + { + AmbiguityAnalysis analysis = Analyze(text, charset); + + Assert.Equal(AmbiguityClass.TextChanging, analysis.Class); + Assert.False(analysis.IsSafeToConvertAutomatically); + Assert.NotEmpty(analysis.CompetingCandidates); + } + + [Fact] + public void PureAsciiIsAmbiguousInLabelButNotInText() + { + // Every candidate agrees on what this file says, so the label is undetermined and + // the content is not. Refusing here would protect nothing. + AmbiguityAnalysis analysis = EncodingAmbiguity.Analyze( + "plain ascii, no high bytes at all"u8, Encoding.ASCII); + + Assert.NotEqual(AmbiguityClass.TextChanging, analysis.Class); + Assert.True(analysis.IsSafeToConvertAutomatically); + Assert.Empty(analysis.CompetingCandidates); + } + + [Fact] + public void TheRefusalNamesTheEncodingsActuallyInConflict() + { + // "Low confidence" gives a user nothing to act on. The competing encodings and + // the next step do. + AmbiguityAnalysis analysis = Analyze("Le café était déjà prêt", "windows-1252"); + + string message = analysis.Describe("windows-1252"); + + Assert.Contains("could not be determined uniquely", message); + Assert.Contains("windows-1252", message); + Assert.Contains("would produce different text", message); + Assert.Contains("specify the source encoding explicitly", message); + } + + [Fact] + public void TheSameBytesAlwaysGetTheSameAnswer() + { + // An earlier version sampled probe positions randomly and moved with the seed on + // short files. Whether a conversion is refused must not depend on a random draw. + byte[] bytes = Encoding.GetEncoding("windows-1252").GetBytes("Le café était déjà prêt"); + Encoding detected = Encoding.GetEncoding("windows-1252"); + + AmbiguityClass first = EncodingAmbiguity.Analyze(bytes, detected).Class; + + for (int i = 0; i < 8; i++) + Assert.Equal(first, EncodingAmbiguity.Analyze(bytes, detected).Class); + } + + [Fact] + public void AliasesOfOneEncodingAreNotCountedAsRivalReadings() + { + // The candidate set is deduplicated by code page. cp949 and ks_c_5601-1987 name + // one encoding, and listing both would manufacture a disagreement out of a + // spelling difference - a mistake this project has made before, in the audit that + // compared detector output by label. + AmbiguityAnalysis analysis = Analyze("안녕하세요 세계", "euc-kr"); + + Assert.Equal( + analysis.CompetingCandidates.Count, + analysis.CompetingCandidates.Distinct().Count()); + } + + [Fact] + public void AnEmptySampleIsNotClaimedToBeAmbiguous() + { + AmbiguityAnalysis analysis = EncodingAmbiguity.Analyze( + ReadOnlySpan.Empty, Encoding.UTF8); + + Assert.True(analysis.IsSafeToConvertAutomatically); + } +} diff --git a/sources/EncodingChecker.Tests/ExplicitSourceEncodingTests.cs b/sources/EncodingChecker.Tests/ExplicitSourceEncodingTests.cs new file mode 100644 index 0000000..c71a356 --- /dev/null +++ b/sources/EncodingChecker.Tests/ExplicitSourceEncodingTests.cs @@ -0,0 +1,204 @@ +using System.Text; + +namespace EncodingChecker.Tests; + +/// +/// The contract for -From, and the escape from an ambiguity refusal. +/// +/// The refusal message tells a user to specify the source encoding, so specifying it has +/// to work — otherwise the safety feature issues advice its own interface cannot take. +/// +/// But it replaces detection and nothing else. Every guarantee from the conversion engine +/// still holds: the bytes must strictly decode as the named encoding, the output must +/// re-decode to exactly the same text, and a failed backup still aborts. -From +/// answers "which encoding is this?", not "convert it regardless". +/// +public sealed class ExplicitSourceEncodingTests : IDisposable +{ + private readonly string _root = + Directory.CreateTempSubdirectory("ec_from_").FullName; + + public void Dispose() + { + try + { + Directory.Delete(_root, recursive: true); + } + catch (IOException) + { + // Best-effort cleanup. + } + } + + private string Write(string name, byte[] content) + { + string path = Path.Combine(_root, name); + File.WriteAllBytes(path, content); + return path; + } + + private List Scan(string? from, bool backup = false) + { + var results = new List(); + + ScanEngine.ScanDirectory( + new ScanDirectoryOptions + { + BaseDirectory = _root, + IncludeSubdirectories = true, + IncludePatterns = ["*"], + Action = ScanAction.Convert, + TargetCharset = "utf-8", + TargetWriteBom = false, + SourceCharset = from, + Backup = backup, + }, + results.Add, + CancellationToken.None); + + return results; + } + + [Fact] + public void AmbiguousAutoDetection_IsRefused() + { + Write("ambiguous.txt", + Encoding.GetEncoding("windows-1252").GetBytes("Le café était déjà prêt")); + + ConversionReportEntry entry = Assert.Single(Scan(from: null)); + + Assert.Equal(ConversionRowResult.Error, entry.Result); + Assert.Contains("could not be determined uniquely", entry.Diagnostic); + } + + [Fact] + public void SameTextAlternativeCodecs_AreAllowed() + { + // Every candidate agrees on the content, so the label is undetermined and the + // text is not. There is nothing to protect the user from. + Write("ascii.txt", "plain ascii, no high bytes"u8.ToArray()); + + ConversionReportEntry entry = Assert.Single(Scan(from: null)); + + Assert.NotEqual(ConversionRowResult.Error, entry.Result); + } + + [Fact] + public void ExplicitSource_LetsTheConversionProceed() + { + // The same file the detector refuses. Somebody has now said which encoding it is. + const string text = "Le café était déjà prêt"; + string path = Write("resolved.txt", + Encoding.GetEncoding("windows-1252").GetBytes(text)); + + ConversionReportEntry entry = Assert.Single(Scan(from: "windows-1252")); + + Assert.Equal(ConversionRowResult.Converted, entry.Result); + Assert.Equal(text, Encoding.UTF8.GetString(File.ReadAllBytes(path))); + Assert.True(entry.SourceEncodingWasSpecified); + } + + [Fact] + public void ExplicitSource_ChangesTheAnswerNotJustThePermission() + { + // Naming a different encoding for the same bytes must produce different text. + // If it did not, -From would be decoration rather than a decision. + byte[] bytes = Encoding.GetEncoding("windows-1252").GetBytes("café"); + string path = Write("interpretation.txt", bytes); + + Assert.Equal(ConversionRowResult.Converted, + Assert.Single(Scan(from: "koi8-r")).Result); + + string asKoi8 = Encoding.UTF8.GetString(File.ReadAllBytes(path)); + + Assert.NotEqual("café", asKoi8); + Assert.Equal(Encoding.GetEncoding("koi8-r").GetString(bytes), asKoi8); + } + + [Fact] + public void ExplicitSource_WithBytesItCannotDecode_IsStillRefused() + { + // EUC-JP bytes carrying a JIS X 0212 sequence code page 51932 cannot map. + // Naming the encoding does not make the bytes representable. + byte[] unrepresentable = + [0x8F, 0xB0, 0xDF, 0xB9, 0xA5, 0xA1, 0xA4, 0xC0, 0xA4, 0xB3]; + string path = Write("undecodable.txt", unrepresentable); + + ConversionReportEntry entry = Assert.Single(Scan(from: "euc-jp")); + + Assert.Equal(ConversionRowResult.Error, entry.Result); + Assert.Equal(unrepresentable, File.ReadAllBytes(path)); + } + + [Fact] + public void ExplicitSource_WithContentTheTargetCannotHold_IsStillRefused() + { + // Converting to a target that cannot represent the text must fail whether the + // source encoding was detected or chosen. + string path = Path.Combine(_root, "unencodable.txt"); + byte[] original = Encoding.UTF8.GetBytes("世界 مرحبا"); + File.WriteAllBytes(path, original); + + var results = new List(); + + ScanEngine.ScanDirectory( + new ScanDirectoryOptions + { + BaseDirectory = _root, + IncludeSubdirectories = true, + IncludePatterns = ["*"], + Action = ScanAction.Convert, + TargetCharset = "windows-1252", + TargetWriteBom = false, + SourceCharset = "utf-8", + }, + results.Add, + CancellationToken.None); + + ConversionReportEntry entry = Assert.Single(results); + + Assert.Equal(ConversionRowResult.Error, entry.Result); + Assert.Equal(original, File.ReadAllBytes(path)); + } + + [Fact] + public void ExplicitSource_WithAFailingBackup_IsStillRefused() + { + byte[] original = Encoding.GetEncoding("windows-1252").GetBytes("café"); + string path = Write("backupfail.txt", original); + Directory.CreateDirectory(path + ".bak"); + + ConversionReportEntry entry = + Assert.Single(Scan(from: "windows-1252", backup: true)); + + Assert.Equal(ConversionRowResult.Error, entry.Result); + Assert.Equal(original, File.ReadAllBytes(path)); + } + + [Fact] + public void ExplicitSource_StillWritesTheRecoveryRecord() + { + // Choosing the encoding does not opt out of being able to undo the result. + string path = Write("recorded.txt", + Encoding.GetEncoding("windows-1252").GetBytes("café")); + + Assert.Equal(ConversionRowResult.Converted, + Assert.Single(Scan(from: "windows-1252", backup: true)).Result); + + RestoreStatus status = ConversionMetadataStore.Inspect(path); + + Assert.Equal(RestoreAvailability.Available, status.Availability); + Assert.Equal(1252, status.Metadata!.DetectedCodePage); + } + + [Fact] + public void DetectionModeIsRecordedSoTheTwoClaimsStayDistinct() + { + // Detection can be wrong in ways an explicit choice cannot. A later journal + // needs to know which one produced a conversion. + Write("ascii.txt", "plain ascii"u8.ToArray()); + + Assert.False(Assert.Single(Scan(from: null)).SourceEncodingWasSpecified); + Assert.True(Assert.Single(Scan(from: "windows-1252")).SourceEncodingWasSpecified); + } +} diff --git a/sources/EncodingChecker/ConversionReport.cs b/sources/EncodingChecker/ConversionReport.cs index ef6a853..b35d3cb 100644 --- a/sources/EncodingChecker/ConversionReport.cs +++ b/sources/EncodingChecker/ConversionReport.cs @@ -1,4 +1,4 @@ -using System; +using System; using System.Collections.Generic; using System.IO; using System.Text; @@ -48,6 +48,37 @@ internal sealed class ConversionReportEntry /// internal string? CurrentCharsetLabel { get; set; } + /// + /// How far this file's bytes identify the encoding that wrote them, decided during + /// detection. Defaults to , which is correct + /// for a caller that supplies the source encoding rather than having it detected: + /// there is nothing ambiguous about an answer somebody gave. + /// Internal state; not included in CSV output. + /// + internal AmbiguityClass Ambiguity { get; set; } = AmbiguityClass.Unambiguous; + + /// + /// Encodings that read this file differently from the one detected. Empty unless + /// is . + /// Internal state; not included in CSV output. + /// + internal IReadOnlyList CompetingEncodings { get; set; } = []; + + /// + /// Why this file received its classification. + /// Internal state; not included in CSV output. + /// + internal AmbiguityReason AmbiguityReason { get; set; } = + AmbiguityReason.ExplicitlySpecified; + + /// + /// Whether the source encoding was chosen by the caller rather than detected. + /// Recorded because the two are different claims: detection can be wrong in ways an + /// explicit choice cannot, and a later journal should be able to say which was used. + /// Internal state; not included in CSV output. + /// + internal bool SourceEncodingWasSpecified { get; set; } + /// Additional error detail; not included in CSV output. internal string? Diagnostic { get; set; } } diff --git a/sources/EncodingChecker/EncodingAmbiguity.cs b/sources/EncodingChecker/EncodingAmbiguity.cs new file mode 100644 index 0000000..35f9062 --- /dev/null +++ b/sources/EncodingChecker/EncodingAmbiguity.cs @@ -0,0 +1,363 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Security.Cryptography; +using System.Text; + +namespace EncodingChecker; + +/// How far the file's bytes determine which encoding wrote them. +internal enum AmbiguityClass +{ + /// Only one supported codec reads these bytes at all. + Unambiguous, + + /// + /// Several codecs read them, and every one yields the same text. The label is + /// undetermined; the content is not, so a conversion cannot lose anything. + /// + TextEquivalent, + + /// + /// Several codecs read them and disagree about what they say. The bytes do not + /// determine the answer, and choosing wrongly changes the user's text. + /// + TextChanging, +} + +/// Why a file was placed in its . +internal enum AmbiguityReason +{ + /// Only one supported codec reads these bytes. + SingleCandidate, + + /// The detected codec constrains these bytes; rivals do not compete. + StructurallyDetermined, + + /// Several codecs read them and produce the same text. + MultipleCodecsSameText, + + /// Several codecs read them and produce different text. + MultipleCodecsDifferentText, + + /// The source encoding was chosen rather than detected. + ExplicitlySpecified, +} + +internal sealed record AmbiguityAnalysis +{ + internal required AmbiguityClass Class { get; init; } + + internal required AmbiguityReason Reason { get; init; } + + /// Supported codecs that strictly decode the sample, by name. + internal required IReadOnlyList Candidates { get; init; } + + /// Distinct readings those candidates produce. + internal required int DistinctReadings { get; init; } + + /// + /// Codecs whose reading differs from the detected one. These are the competing + /// interpretations worth naming to a user, rather than the full candidate list. + /// + internal required IReadOnlyList CompetingCandidates { get; init; } + + internal bool IsSafeToConvertAutomatically => Class != AmbiguityClass.TextChanging; + + /// A reason a person can act on, naming the codecs actually in conflict. + internal string Describe(string detectedName) => + Class == AmbiguityClass.TextChanging + ? DescribeRefusal(detectedName, CompetingCandidates) + : string.Empty; + + /// + /// The refusal message, phrased so the next step is obvious. "Low confidence" tells a + /// user nothing they can act on; naming the encodings actually in conflict does. + /// + internal static string DescribeRefusal( + string detectedName, IReadOnlyList competingCandidates) + { + string competing = string.Join(", ", competingCandidates.Take(4)); + + if (competingCandidates.Count > 4) + competing += $", and {competingCandidates.Count - 4} more"; + + return "The encoding could not be determined uniquely from the file's contents. " + + $"{detectedName} and {competing} all match this file and would produce " + + "different text. No conversion was performed; specify the source " + + "encoding explicitly to convert it."; + } +} + +/// +/// Decides whether a file's bytes actually identify the encoding that wrote them. +/// +/// +/// A corpus audit of 5,078 files found 262 where they do not: single-byte code pages map +/// 256 values independently, so a file valid in windows-1252 is equally valid in +/// iso-8859-1, and no inspection of the bytes decides between them. Detection heuristics +/// still answer, and on short or ASCII-heavy input that answer is close to a guess. +/// +/// The distinction that matters is not how many codecs *can* read the bytes but whether +/// they *disagree* about the result. A pure-ASCII file read as us-ascii or as UTF-8 is +/// ambiguous in label and identical in text; the same file read as iso-8859-1 or +/// windows-1252 with bytes in 0x80-0x9F is not. +/// +/// +internal static class EncodingAmbiguity +{ + /// + /// Bytes examined. Matches the detector's own sample: asking a different question of + /// a different span could conclude "unambiguous" about a region the detector never saw. + /// + internal const int SampleBytes = 64 * 1024; + + /// + /// The codecs EC can name, deduplicated by code page so aliases do not appear as + /// rival interpretations of the same bytes. Derived from what EC supports rather + /// than from any corpus - a candidate set drawn from test data makes the answer + /// depend on what the tests happened to contain. + /// + private static readonly Lazy Universe = new(() => + { + var seen = new HashSet(); + var result = new List(); + + foreach (string name in TextEncoding.SupportedCharsets) + { + Encoding encoding; + + try + { + encoding = Encoding.GetEncoding( + name, EncoderFallback.ExceptionFallback, DecoderFallback.ExceptionFallback); + } + catch (Exception ex) when (ex is ArgumentException or NotSupportedException) + { + continue; + } + + if (seen.Add(encoding.CodePage)) + result.Add(encoding); + } + + return [.. result]; + }); + + /// + /// Above this, the codec has some hold on the input. + /// + /// + /// Set just above zero because the observed separation is not a matter of degree: a + /// single-byte code page with no undefined positions rejects *nothing*, measuring + /// exactly 0.000, while every multi-byte and Unicode encoding measured on real text + /// lands at 0.111 or above. A larger margin was tried first and put Shift_JIS at + /// 0.146 - above the line, but close enough that a different probe order moved it + /// across. The gap is zero versus non-zero; the threshold should say so rather than + /// invent a cutoff in the middle of empty space. + /// + private const double ConstraintFloor = 0.02; + + /// Probe positions; enough to characterise the input without scanning it all. + private const int ProbePositions = 256; + + /// + /// How tightly a codec constrains these particular bytes: the fraction of small + /// mutations it rejects. + /// + /// + /// This is what separates a real rival reading from a meaningless one. Every byte + /// sequence is "valid" iso-8859-1, so iso-8859-1 offering a different reading of a + /// UTF-8 file is not a competing claim - it is a codec that cannot refuse anything. + /// Valid UTF-8 is improbable by accident, so a file that survives mutation under it + /// was not valid by chance. + /// + /// Asked of the bytes rather than of the codec, because single-byte does not imply + /// unconstrained: windows-1252 leaves 0x81, 0x8D, 0x8F, 0x90 and 0x9D undefined, so a + /// file containing one of them is constrained under it while another file is not. + /// + /// + /// Probes evenly-spaced positions rather than random ones. An earlier version sampled + /// randomly and gave answers that moved with the seed on short files - a conversion + /// refusing or proceeding should not depend on a random draw. + /// + /// + private static double ConstraintOn(ReadOnlySpan sample, Encoding encoding) + { + if (sample.Length < 2) + return 0.0; + + int step = Math.Max(1, sample.Length / ProbePositions); + int rejected = 0; + int probes = 0; + + byte[] flipped = sample.ToArray(); + byte[] shortened = new byte[sample.Length - 1]; + + for (int i = 0; i < sample.Length; i += step) + { + // Deleting a byte tests alignment, which is the only structure a fixed-width + // encoding has: UTF-16 survives nearly any bit flip, because most flips just + // produce a different valid character, but losing one byte shifts every unit + // after it. + sample[..i].CopyTo(shortened); + sample[(i + 1)..].CopyTo(shortened.AsSpan(i)); + + probes++; + if (TryHash(shortened, encoding, flush: true) is null) + rejected++; + + // Flipping the high bit tests the value, which is what the multi-byte pages + // constrain. + byte original = flipped[i]; + flipped[i] ^= 0x80; + + probes++; + if (TryHash(flipped, encoding, flush: true) is null) + rejected++; + + flipped[i] = original; + } + + return probes == 0 ? 0.0 : (double)rejected / probes; + } + + internal static AmbiguityAnalysis Analyze(ReadOnlySpan sample, Encoding detected) + { + ArgumentNullException.ThrowIfNull(detected); + + if (sample.Length > SampleBytes) + sample = sample[..SampleBytes]; + + string? detectedHash = null; + var byHash = new Dictionary>(StringComparer.Ordinal); + + foreach (Encoding candidate in Universe.Value) + { + string? hash = TryHash(sample, candidate); + + if (hash is null) + continue; + + if (!byHash.TryGetValue(hash, out List? names)) + byHash[hash] = names = []; + + names.Add(candidate.WebName); + + if (candidate.CodePage == detected.CodePage) + detectedHash = hash; + } + + // The detected codec failing to decode its own sample is a detection problem, + // not an ambiguity one; leave it to strict decoding to reject. + if (detectedHash is null) + { + return new AmbiguityAnalysis + { + Class = AmbiguityClass.Unambiguous, + Reason = AmbiguityReason.SingleCandidate, + Candidates = [], + DistinctReadings = byHash.Count, + CompetingCandidates = [], + }; + } + + List candidates = [.. byHash.Values.SelectMany(v => v)]; + + List competing = + [ + .. byHash + .Where(pair => !string.Equals(pair.Key, detectedHash, StringComparison.Ordinal)) + .SelectMany(pair => pair.Value) + .Order(StringComparer.Ordinal) + ]; + + // Rival readings only compete when the detected codec has no hold on these + // bytes. If it does - valid UTF-8, valid Shift_JIS - the file's structure + // already picked it out, and codecs that accept every byte sequence are not + // offering an alternative so much as failing to object. + bool detectedIsDetermined = + competing.Count > 0 && ConstraintOn(sample, detected) >= ConstraintFloor; + + if (detectedIsDetermined) + competing = []; + + AmbiguityClass classification = + competing.Count > 0 ? AmbiguityClass.TextChanging + : candidates.Count > 1 ? AmbiguityClass.TextEquivalent + : AmbiguityClass.Unambiguous; + + AmbiguityReason reason = + competing.Count > 0 ? AmbiguityReason.MultipleCodecsDifferentText + : detectedIsDetermined ? AmbiguityReason.StructurallyDetermined + : candidates.Count > 1 ? AmbiguityReason.MultipleCodecsSameText + : AmbiguityReason.SingleCandidate; + + return new AmbiguityAnalysis + { + Class = classification, + Reason = reason, + Candidates = candidates, + DistinctReadings = byHash.Count, + CompetingCandidates = competing, + }; + } + + /// + /// SHA-256 over the decoded text, or if this codec cannot + /// strictly decode the sample. + /// + /// + /// Whether an incomplete trailing sequence counts as invalid. + /// + /// False when asking which codecs read the real sample: it may genuinely cut a + /// character in half at 64 KiB, and a boundary artifact is not evidence. + /// + /// + /// True when measuring constraint against mutated copies, where an incomplete tail + /// is the whole point. Deleting a byte from a UTF-16 file leaves an odd length, and + /// tolerating that made fixed-width encodings look unconstrained - they survive + /// almost any bit flip, so alignment is the only structure they have to test. + /// + /// + private static string? TryHash( + ReadOnlySpan sample, Encoding encoding, bool flush = false) + { + char[] buffer; + + try + { + buffer = new char[encoding.GetMaxCharCount(sample.Length)]; + } + catch (ArgumentOutOfRangeException) + { + return null; + } + + int written; + + try + { + Decoder decoder = TextEncoding.Strict(encoding).GetDecoder(); + written = decoder.GetChars(sample, buffer, flush); + } + catch (Exception ex) when (ex is DecoderFallbackException or ArgumentException) + { + return null; + } + + if (written == 0) + return null; + + Span bytes = stackalloc byte[sizeof(char)]; + using var sha = IncrementalHash.CreateHash(HashAlgorithmName.SHA256); + + for (int i = 0; i < written; i++) + { + BitConverter.TryWriteBytes(bytes, buffer[i]); + sha.AppendData(bytes); + } + + return Convert.ToHexStringLower(sha.GetHashAndReset()); + } +} diff --git a/sources/EncodingChecker/MainForm.cs b/sources/EncodingChecker/MainForm.cs index 3d866ed..80abdad 100644 --- a/sources/EncodingChecker/MainForm.cs +++ b/sources/EncodingChecker/MainForm.cs @@ -1132,32 +1132,9 @@ private void UpdateControlsOnActionDone(string statusMessage) // Matches encodings reported by UtfUnknown.Core.CodepageName. // UTF-7 is intentionally excluded because .NET disables it by default (SYSLIB0001) // and Encoding.GetEncoding throws NotSupportedException. - private static readonly string[] SupportedCharsets = - [ - "ascii", "utf-8", "utf-16le", "utf-16be", - "utf-32le", "utf-32be", - "euc-jp", "euc-kr", "euc-tw", - "iso-2022-cn", "iso-2022-kr", "iso-2022-jp", - "x-cp50227", - "big5", "gb18030", "hz-gb-2312", "shift-jis", - "ks_c_5601-1987", "cp949", - "ibm852", "ibm855", "ibm866", - "iso-8859-1", "iso-8859-2", "iso-8859-3", - "iso-8859-4", "iso-8859-5", "iso-8859-6", - "iso-8859-7", "iso-8859-8", "iso-8859-9", - "iso-8859-10", "iso-8859-11", "iso-8859-13", - "iso-8859-15", "iso-8859-16", - "windows-1250", "windows-1251", "windows-1252", - "windows-1253", "windows-1255", "windows-1256", - "windows-1257", "windows-1258", - "x-mac-ce", "x-mac-cyrillic", - "koi8-r", "tis-620", "viscii", - "X-ISO-10646-UCS-4-3412", - "X-ISO-10646-UCS-4-2143" - ]; private static string[] GetSupportedCharsets() => - SupportedCharsets; + TextEncoding.SupportedCharsets; private void ShowWarning( string message, diff --git a/sources/EncodingChecker/Program.cs b/sources/EncodingChecker/Program.cs index 16cd75c..f3cddba 100644 --- a/sources/EncodingChecker/Program.cs +++ b/sources/EncodingChecker/Program.cs @@ -99,6 +99,7 @@ internal sealed class CliOptions internal List Include = []; internal List Exclude = []; internal string? Target; + internal string? From; internal string? ValidateCharsets; internal bool DetectOnly; internal string? ReportPath; @@ -134,6 +135,18 @@ containing a separator matches the path relative "utf-8" or "utf-8-bom". Required unless -Validate or -DetectOnly is given. + [-From ""] + Treat every file as this encoding instead of + detecting it. Use when detection reports that a + file's encoding cannot be determined from its + contents, or when you already know it. + + This replaces detection and nothing else: the bytes + must still decode strictly as this encoding, the + output is still verified to hold exactly the same + text, and a failed backup still aborts. Convert mode + only. + Modes: Conversion is the default mode. [-Validate ""] @@ -203,6 +216,8 @@ final summary line. -Report is unaffected. EncodingChecker.exe -BasePath . -Include "*" -Validate "utf-8,utf-8-bom" -Report report.csv + EncodingChecker.exe -BasePath . -Include "*.txt" -From "windows-1252" -Target "utf-8" + EncodingChecker.exe -BasePath D:\NetworkShare -Include "*.txt" -Target "utf-8" -MaxParallelism 2 -FailOnChanges """; @@ -263,6 +278,7 @@ .. options.ValidateCharsets! var scanOptions = new ScanDirectoryOptions { + SourceCharset = options.From, BaseDirectory = options.BasePath!, IncludeSubdirectories = true, IncludePatterns = options.Include, @@ -469,6 +485,17 @@ internal static bool TryParseArguments( } break; + case "from": + if (!TryTakeValue( + args, + ref i, + out options.From)) + { + error = "-From requires a value."; + return false; + } + break; + case "validate": if (!TryTakeValue( args, @@ -546,7 +573,7 @@ internal static bool TryParseArguments( private static readonly HashSet KnownFlagNames = new(StringComparer.OrdinalIgnoreCase) { - "basepath", "include", "exclude", "target", "validate", + "basepath", "include", "exclude", "target", "from", "validate", "detectonly", "report", "maxparallelism", "failonchanges", "whatif", "backup", "quiet", "verbose", }; @@ -589,6 +616,26 @@ internal static bool TryValidateOptions( CliOptions options, [NotNullWhen(false)] out string? error) { + if (!string.IsNullOrWhiteSpace(options.From)) + { + if (options.DetectOnly || !string.IsNullOrWhiteSpace(options.ValidateCharsets)) + { + error = "-From applies to conversion only; it cannot be combined with " + + "-DetectOnly or -Validate, which report what the detector finds."; + return false; + } + + try + { + Encoding.GetEncoding(options.From!); + } + catch (ArgumentException) + { + error = $"'{options.From}' is not a recognized encoding."; + return false; + } + } + if (string.IsNullOrWhiteSpace(options.BasePath)) { error = "-BasePath is required."; diff --git a/sources/EncodingChecker/ScanEngine.cs b/sources/EncodingChecker/ScanEngine.cs index effe811..8f1468b 100644 --- a/sources/EncodingChecker/ScanEngine.cs +++ b/sources/EncodingChecker/ScanEngine.cs @@ -60,6 +60,23 @@ internal sealed class ScanDirectoryOptions /// Target charset for conversion, without "-bom". internal string? TargetCharset { get; init; } + /// + /// Source charset chosen by the caller, used instead of detection. + /// + /// + /// This replaces detection, and nothing else. Every verification still applies: the + /// bytes must strictly decode as this encoding, the output must re-decode to exactly + /// the same text, the backup must verify, and the record must be written before + /// anything is installed. It answers "which encoding is this?", not "convert it + /// regardless". + /// + /// It is also the escape from an ambiguity refusal. Where the bytes cannot identify + /// the encoding, someone who knows has to say - and saying so must be possible, or + /// the refusal is advice the user cannot take. + /// + /// + internal string? SourceCharset { get; init; } + internal bool TargetWriteBom { get; init; } /// Simulate conversion without writing. @@ -296,7 +313,25 @@ internal static string FormatCharsetLabel( Encoding? targetEncoding, CancellationToken cancellationToken) { - Encoding? detected = TextEncoding.DetectFromFile(path); + bool sourceWasSpecified = !string.IsNullOrWhiteSpace(options.SourceCharset); + + Encoding? detected; + + if (sourceWasSpecified) + { + try + { + detected = Encoding.GetEncoding(options.SourceCharset!); + } + catch (ArgumentException) + { + detected = null; + } + } + else + { + detected = TextEncoding.DetectFromFile(path); + } bool hasBom = detected != null && @@ -315,6 +350,26 @@ internal static string FormatCharsetLabel( Result = ConversionRowResult.Unchanged, }; + // Classified here, beside the detection it qualifies, and carried on the entry. + // The conversion path then reads a decision rather than re-deriving one, and a + // caller who supplies the source encoding instead of detecting it gets the + // default - correctly, since there is nothing ambiguous about an answer somebody + // gave. + entry.SourceEncodingWasSpecified = sourceWasSpecified; + + if (detected is not null && options.Action == ScanAction.Convert + && !sourceWasSpecified) + { + AmbiguityAnalysis? ambiguity = AnalyzeAmbiguity(path, detected); + + if (ambiguity is not null) + { + entry.Ambiguity = ambiguity.Class; + entry.AmbiguityReason = ambiguity.Reason; + entry.CompetingEncodings = ambiguity.CompetingCandidates; + } + } + switch (options.Action) { case ScanAction.Detect: @@ -403,6 +458,21 @@ private static void ApplyConversion( return; } + // Refuse before touching anything when the file's bytes do not identify the + // encoding that wrote them and the rival readings disagree about the text. + // + // Detection still produces an answer for these; on short or ASCII-heavy input + // that answer is close to a guess, and acting on it rewrites the user's file + // into one of several possible readings without saying so. Skipped when the user + // named the source encoding, since then it is their answer, not a guess. + if (entry.Ambiguity == AmbiguityClass.TextChanging) + { + entry.Result = ConversionRowResult.Error; + entry.Diagnostic = AmbiguityAnalysis.DescribeRefusal( + sourceCharset, entry.CompetingEncodings); + return; + } + if (whatIf) { entry.Result = ConversionRowResult.Converted; // "would be converted" @@ -546,6 +616,38 @@ private static void ApplyConversion( }); } + /// + /// Reads the detector's sample again and asks whether it identifies one encoding. + /// Returns when the file cannot be read, leaving the decision + /// to the conversion itself rather than refusing on an I/O error. + /// + private static AmbiguityAnalysis? AnalyzeAmbiguity(string path, Encoding sourceEncoding) + { + try + { + using FileStream stream = new( + path, FileMode.Open, FileAccess.Read, + FileShare.ReadWrite | FileShare.Delete, + 4096, FileOptions.SequentialScan); + + int length = (int)Math.Min(stream.Length, EncodingAmbiguity.SampleBytes); + + if (length == 0) + return null; + + byte[] buffer = new byte[length]; + int read = stream.ReadAtLeast(buffer, length, throwOnEndOfStream: false); + + return read == 0 + ? null + : EncodingAmbiguity.Analyze(buffer.AsSpan(0, read), sourceEncoding); + } + catch (Exception ex) when (ex is IOException or UnauthorizedAccessException) + { + return null; + } + } + private static void CreateBackup(string path) { string? directory = Path.GetDirectoryName(path); diff --git a/sources/EncodingChecker/TextEncoding.cs b/sources/EncodingChecker/TextEncoding.cs index ca1b388..76797b3 100644 --- a/sources/EncodingChecker/TextEncoding.cs +++ b/sources/EncodingChecker/TextEncoding.cs @@ -226,6 +226,41 @@ internal static class TextEncoding #region Helpers + /// + /// Every charset EC can name or convert to. + /// + /// + /// Shared with , which needs the same set to decide + /// whether a file's bytes identify one encoding or several. That question must be + /// answered against what EC actually supports; deriving the candidates from any + /// other source would make the answer depend on something the user cannot see. + /// + internal static readonly string[] SupportedCharsets = + [ + "ascii", "utf-8", "utf-16le", "utf-16be", + "utf-32le", "utf-32be", + "euc-jp", "euc-kr", "euc-tw", + "iso-2022-cn", "iso-2022-kr", "iso-2022-jp", + "x-cp50227", + "big5", "gb18030", "hz-gb-2312", "shift-jis", + "ks_c_5601-1987", "cp949", + "ibm852", "ibm855", "ibm866", + "iso-8859-1", "iso-8859-2", "iso-8859-3", + "iso-8859-4", "iso-8859-5", "iso-8859-6", + "iso-8859-7", "iso-8859-8", "iso-8859-9", + "iso-8859-10", "iso-8859-11", "iso-8859-13", + "iso-8859-15", "iso-8859-16", + "windows-1250", "windows-1251", "windows-1252", + "windows-1253", "windows-1255", "windows-1256", + "windows-1257", "windows-1258", + "x-mac-ce", "x-mac-cyrillic", + "koi8-r", "tis-620", "viscii", + "X-ISO-10646-UCS-4-3412", + "X-ISO-10646-UCS-4-2143" + ]; + + + /// /// Returns an encoding whose decoder and encoder actually enforce strict fallback. ///