diff --git a/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs b/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs
index 91cc09d374..e64903f076 100644
--- a/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs
+++ b/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs
@@ -25,6 +25,8 @@ namespace SIL.FieldWorks.Common.RootSites.RenderBenchmark
public abstract class RenderBenchmarkTestsBase : RealDataTestsBase
{
protected const string DeterministicRenderFontFamily = "Segoe UI";
+ // Second Latin font for scenarios whose runs alternate font family.
+ protected const string SecondaryRenderFontFamily = "Times New Roman";
// Pinned Arabic font (loaded privately by RenderTestAssemblySetup). Used for Arabic runs so
// they don't depend on the host's Segoe UI Arabic version / font fallback.
protected const string ArabicRenderFontFamily = "Scheherazade New";
@@ -320,6 +322,15 @@ protected void SetupScenarioData(string scenarioId)
case "multi-ws":
CreateMultiWsScenario();
break;
+ case "single-para-mixed-ws":
+ CreateSingleParaMixedWsScenario();
+ break;
+ case "nfc-composable-diacritics":
+ CreateNfcComposableDiacriticsScenario();
+ break;
+ case "multi-line-wrap-single-ws":
+ CreateMultiLineWrapSingleWsScenario();
+ break;
case "lex-shallow":
CreateLexEntryScenario(depth: 2, breadth: 3);
break;
@@ -539,6 +550,27 @@ private void CreateMultiWsScenario()
AddMultiWsSections(book, 5, versesPerSection: 8, chapterStart: 1);
}
+ private void CreateSingleParaMixedWsScenario()
+ {
+ var book = CreateBook(19); // PSA
+ m_hvoRoot = book.Hvo;
+ AddSingleMixedWsParagraph(book, sentenceCount: 236);
+ }
+
+ private void CreateNfcComposableDiacriticsScenario()
+ {
+ var book = CreateBook(15); // EZR
+ m_hvoRoot = book.Hvo;
+ AddNfcComposableDiacriticsParagraph(book, wordCount: 80);
+ }
+
+ private void CreateMultiLineWrapSingleWsScenario()
+ {
+ var book = CreateBook(17); // EST
+ m_hvoRoot = book.Hvo;
+ AddSingleWsProseParagraph(book, sentenceCount: 200);
+ }
+
#region Rich Data Factories
protected IScrBook CreateBook(int bookNum)
@@ -928,6 +960,147 @@ protected void AddMultiWsSections(IScrBook book, int sectionCount,
}
}
+ ///
+ /// Returns the canonical decomposition (NFD) of , and fails the
+ /// fixture unless the result contains combining marks that NFC would compose. A scenario
+ /// built from it therefore always exercises text that normalization rewrites, whatever
+ /// normalization form the source file itself is stored in.
+ ///
+ protected static string Decomposed(string text)
+ {
+ string nfd = CustomIcu.GetIcuNormalizer(FwNormalizationMode.knmNFD).Normalize(text);
+ string nfc = CustomIcu.GetIcuNormalizer(FwNormalizationMode.knmNFC).Normalize(text);
+ Assert.That(nfd, Is.Not.EqualTo(nfc),
+ $"Scenario text '{text}' must contain characters that decompose.");
+ return nfd;
+ }
+
+ ///
+ /// Adds a section whose content is one paragraph of
+ /// sentences. Each sentence's text and run properties come from the two callbacks, which
+ /// receive the sentence index. With the paragraph
+ /// opens with a chapter-number run.
+ ///
+ private void AddSingleParagraphSection(IScrBook book, string heading, int sentenceCount,
+ Func sentence, Func runProps, bool withChapterNumber = false)
+ {
+ var section = Cache.ServiceLocator.GetInstance().Create();
+ book.SectionsOS.Add(section);
+
+ var stTextFactory = Cache.ServiceLocator.GetInstance();
+
+ section.HeadingOA = stTextFactory.Create();
+ var headingBldr = new StTxtParaBldr(Cache) { ParaStyleName = ScrStyleNames.SectionHead };
+ headingBldr.AppendRun(heading, StyleUtils.CharStyleTextProps(null, m_wsEng));
+ headingBldr.CreateParagraph(section.HeadingOA);
+
+ section.ContentOA = stTextFactory.Create();
+ var paraBldr = new StTxtParaBldr(Cache) { ParaStyleName = ScrStyleNames.NormalParagraph };
+ if (withChapterNumber)
+ paraBldr.AppendRun("1", StyleUtils.CharStyleTextProps(ScrStyleNames.ChapterNumber, m_wsEng));
+
+ for (int i = 0; i < sentenceCount; i++)
+ paraBldr.AppendRun(sentence(i), runProps(i));
+
+ paraBldr.CreateParagraph(section.ContentOA);
+ }
+
+ ///
+ /// Adds one paragraph of short decomposed sentences whose runs alternate between two
+ /// writing systems and two font families.
+ ///
+ protected void AddSingleMixedWsParagraph(IScrBook book, int sentenceCount)
+ {
+ string[] subjects =
+ {
+ "the \u00E9lder",
+ "the h\u00E8rder",
+ "the s\u00EFnger",
+ "the tea\u00E7her",
+ "the trav\u00EAler",
+ "the wom\u00E3n"
+ };
+ string[] predicates =
+ {
+ "spoke of the l\u00F3ng rains",
+ "walked to the f\u00E0r well",
+ "named the sev\u00EBn hills",
+ "counted the cattl\u00E9 at dusk",
+ "kept the \u00F4ld story",
+ "asked for a bless\u0129ng"
+ };
+
+ AddSingleParagraphSection(book, "Single Paragraph, Mixed Writing Systems", sentenceCount,
+ i => Decomposed($"{subjects[i % subjects.Length]} {predicates[i % predicates.Length]} {i + 1}. "),
+ i => AlternatingFontRunProps(i % 2 == 0),
+ withChapterNumber: true);
+ }
+
+ private ITsTextProps AlternatingFontRunProps(bool first)
+ {
+ var bldr = TsStringUtils.MakePropsBldr();
+ bldr.SetIntPropValues((int)FwTextPropType.ktptWs, (int)FwTextPropVar.ktpvDefault,
+ first ? m_wsEng : m_wsFr);
+ bldr.SetStrPropValue((int)FwTextPropType.ktptFontFamily,
+ first ? DeterministicRenderFontFamily : SecondaryRenderFontFamily);
+ return bldr.GetTextProps();
+ }
+
+ ///
+ /// Adds one wrapped paragraph of Latin sentences, each naming a word spelled with
+ /// decomposed diacritics.
+ ///
+ protected void AddNfcComposableDiacriticsParagraph(IScrBook book, int wordCount)
+ {
+ string[] words =
+ {
+ "caf\u00E9",
+ "d\u00E9j\u00E0",
+ "no\u00EBl",
+ "fran\u00E7ais",
+ "gar\u00E7on",
+ "h\u00F4tel",
+ "a\u00F1o",
+ "cr\u00E9\u00E9e",
+ "\u00E9l\u00E9gant",
+ "fa\u00E7ade"
+ };
+
+ AddSingleParagraphSection(book, "Decomposed Diacritics Microbenchmark", wordCount,
+ i => Decomposed($"The {words[i % words.Length]} recorded here is entry {i + 1}. "),
+ i => StyleUtils.CharStyleTextProps(null, m_wsEng));
+ }
+
+ ///
+ /// Adds one long wrapped paragraph of decomposed sentences in a single writing system
+ /// and font.
+ ///
+ protected void AddSingleWsProseParagraph(IScrBook book, int sentenceCount)
+ {
+ string[] subjects =
+ {
+ "the mer\u00E7hant",
+ "the mas\u00F3n",
+ "the scrib\u00EB",
+ "the sheph\u00E8rd",
+ "the weav\u00EAr",
+ "the pott\u00E9r"
+ };
+ string[] predicates =
+ {
+ "measured the grain by the riv\u00E9r",
+ "repaired the east\u00E8rn wall before dusk",
+ "copied the ledger onto fresh parchm\u00EBnt",
+ "counted the flock past the old gat\u00EA",
+ "dyed the cloth a deep saffr\u00F5n",
+ "shaped the jar on the slow whe\u00EBl"
+ };
+
+ AddSingleParagraphSection(book, "Single Writing-System Line-Wrap Microbenchmark", sentenceCount,
+ i => Decomposed($"{subjects[i % subjects.Length]} {predicates[i % predicates.Length]} on day {i + 1}. "),
+ i => StyleUtils.CharStyleTextProps(null, m_wsEng));
+ }
+
#endregion
#region Lex Entry Scenario Data
diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png
new file mode 100644
index 0000000000..9d19d891b4
Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png differ
diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png
new file mode 100644
index 0000000000..1925ebb15f
Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png differ
diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png
new file mode 100644
index 0000000000..38c21c2c42
Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png differ
diff --git a/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json b/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json
index 782b92e682..305b23157f 100644
--- a/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json
+++ b/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json
@@ -89,6 +89,21 @@
"description": "Lex entry with senses nested 6 levels deep, 2-wide (depth 6, breadth 2 = 126 senses)",
"tags": ["lex-entry", "nested-senses", "exponential-cost", "stress"],
"viewType": "LexEntry"
+ },
+ {
+ "id": "single-para-mixed-ws",
+ "description": "One paragraph of 236 short decomposed (NFD) sentences whose runs alternate two writing systems and two font families",
+ "tags": ["stress", "layout-stress", "multi-ws", "single-paragraph", "nfd"]
+ },
+ {
+ "id": "nfc-composable-diacritics",
+ "description": "One wrapped paragraph of Latin prose spelled with decomposed (NFD) diacritics that NFC composes",
+ "tags": ["stress", "layout-stress", "nfd", "diacritics", "line-breaking"]
+ },
+ {
+ "id": "multi-line-wrap-single-ws",
+ "description": "One long paragraph of 200 decomposed (NFD) sentences in one writing system and font, wrapping over many lines",
+ "tags": ["stress", "layout-stress", "single-paragraph", "line-breaking", "nfd"]
}
]
}
diff --git a/Src/views/Test/TestUniscribeEngine.h b/Src/views/Test/TestUniscribeEngine.h
index 4780767501..d627bb9a37 100644
--- a/Src/views/Test/TestUniscribeEngine.h
+++ b/Src/views/Test/TestUniscribeEngine.h
@@ -16,6 +16,7 @@ Last reviewed:
#include "testViews.h"
#include "RenderEngineTestBase.h"
+#include "LayoutCache.h"
namespace TestViews
{
@@ -396,6 +397,94 @@ namespace TestViews
#endif
}
+ // Installs a fresh layout-pass cache for the lifetime of the object, restoring the
+ // previous one even when a test assertion throws.
+ class ScopedLayoutPassCache
+ {
+ public:
+ ScopedLayoutPassCache() : m_pPrev(SetCurrentLayoutPassCache(&m_cache)) {}
+ ~ScopedLayoutPassCache() { SetCurrentLayoutPassCache(m_pPrev); }
+ LayoutPassCache & Cache() { return m_cache; }
+
+ private:
+ LayoutPassCache m_cache;
+ LayoutPassCache * m_pPrev;
+ };
+
+ // Breaks one text source twice in one layout pass and reports cache hits and misses.
+ // The engine is built here, not via COM, so it sees the cache this module installs.
+ void BreakTwiceInOneLayoutPass(const wchar_t * pszText, int * pcHit, int * pcMiss)
+ {
+ *pcHit = *pcMiss = 0;
+#if defined(WIN32) || defined(_M_X64)
+ HWND hwndDesktop = ::GetDesktopWindow();
+ HDC hdcDesktop = ::GetDC(hwndDesktop);
+ HDC hdc = ::CreateCompatibleDC(hdcDesktop);
+ HBITMAP hbm = ::CreateCompatibleBitmap(hdcDesktop, 100, 100);
+ HGDIOBJ hbmOld = ::SelectObject(hdc, hbm);
+
+ IVwGraphicsWin32Ptr qvg;
+ qvg.CreateInstance(CLSID_VwGraphicsWin32);
+ qvg->Initialize(hdc);
+
+ ILgWritingSystemFactoryPtr qwsf;
+ m_qre->get_WritingSystemFactory(&qwsf);
+ IRenderEnginePtr qre;
+ UniscribeEngine::CreateCom(NULL, IID_IRenderEngine, (void **)&qre);
+ CheckHr(qre->putref_WritingSystemFactory(qwsf));
+ CheckHr(qre->putref_RenderEngineFactory(m_qref));
+ TxtSrc ts(pszText, qwsf);
+ IVwTextSourcePtr qts;
+ ts.QueryInterface(IID_IVwTextSource, (void **)&qts);
+ int cch;
+ CheckHr(qts->get_Length(&cch));
+
+ {
+ ScopedLayoutPassCache scope;
+ for (int ibreak = 0; ibreak < 2; ++ibreak)
+ {
+ ILgSegmentPtr qseg;
+ int dichLimSeg;
+ int dxWidth;
+ LgEndSegmentType est;
+ CheckHr(qre->FindBreakPoint(qvg, qts, NULL, 0, cch, cch, TRUE, TRUE, 4000,
+ klbWordBreak, klbLetterBreak, ktwshAll, FALSE, &qseg, &dichLimSeg,
+ &dxWidth, &est, NULL));
+ }
+ *pcHit = scope.Cache().AnalysisCache().HitCount();
+ *pcMiss = scope.Cache().AnalysisCache().MissCount();
+ }
+
+ qts.Clear();
+ qre.Clear();
+ qvg.Clear();
+ ::SelectObject(hdc, hbmOld);
+ ::DeleteObject(hbm);
+ ::DeleteDC(hdc);
+ ::ReleaseDC(hwndDesktop, hdcDesktop);
+#endif
+ }
+
+ void testNfcTextAnalysisIsReusedWithinLayoutPass()
+ {
+#if defined(WIN32) || defined(_M_X64)
+ int cHit, cMiss;
+ BreakTwiceInOneLayoutPass(L"cafe deja vu", &cHit, &cMiss);
+ unitpp::assert_true("text NFC leaves unchanged should be reused on the second break",
+ cHit > 0);
+#endif
+ }
+
+ void testDecomposedTextAnalysisIsReusedWithinLayoutPass()
+ {
+#if defined(WIN32) || defined(_M_X64)
+ int cHit, cMiss;
+ // Decomposed e-acute and a-grave, which NFC composes.
+ BreakTwiceInOneLayoutPass(L"cafe\u0301 de\u0301ja\u0300 vu", &cHit, &cMiss);
+ unitpp::assert_true("text NFC rewrites should be reused on the second break", cHit > 0);
+#endif
+ }
+
virtual void Setup()
{
RenderEngineTestBase::Setup();
diff --git a/Src/views/Test/TestViewCaches.h b/Src/views/Test/TestViewCaches.h
index 2b108c1d8d..a3311e3de5 100644
--- a/Src/views/Test/TestViewCaches.h
+++ b/Src/views/Test/TestViewCaches.h
@@ -9,9 +9,11 @@ This software is licensed under the LGPL, version 2.1 or later
#pragma once
#include "testViews.h"
+#include "RenderEngineTestBase.h"
#include "ColorStateCache.h"
#include "FontHandleCache.h"
#include "LayoutCache.h"
+#include "NfcOffsetMap.h"
namespace TestViews
{
@@ -218,6 +220,195 @@ namespace TestViews
public:
TestShapeRunCache();
};
+
+ class TestTextAnalysisCache : public unitpp::suite
+ {
+ static IVwTextSource * FakeSource(int n)
+ {
+ return reinterpret_cast(static_cast(0x1000 + n));
+ }
+
+ void StoreTenChars(TextAnalysisCache & cache, IVwTextSource * pts, int ichMin, int ws)
+ {
+ SCRIPT_ITEM rgscri[2];
+ ZeroMemory(rgscri, sizeof(rgscri));
+ rgscri[1].iCharPos = 10;
+ cache.Store(pts, ichMin, 10, ws, false, L"abcdefghij", 10, true, rgscri, 1);
+ }
+
+ void testStoredRangeIsFound()
+ {
+ TextAnalysisCache cache;
+ StoreTenChars(cache, FakeSource(1), 0, 1);
+ TextAnalysisEntry * pentry = cache.Find(FakeSource(1), 0, 10, 1, false);
+ unitpp::assert_true("an identical request should hit", pentry != NULL);
+ unitpp::assert_eq("the hit should carry the stored text length", 10,
+ pentry->m_vchText.Size());
+ unitpp::assert_true("the hit should carry the stored text",
+ ::memcmp(pentry->m_vchText.Begin(), L"abcdefghij", 10 * isizeof(OLECHAR)) == 0);
+ unitpp::assert_eq("the hit should carry the stored item count", 1, pentry->m_citem);
+ }
+
+ void testShorterRequestFromSameStartIsCovered()
+ {
+ TextAnalysisCache cache;
+ StoreTenChars(cache, FakeSource(1), 0, 1);
+ unitpp::assert_true("a shorter request from the same start should hit",
+ cache.Find(FakeSource(1), 0, 4, 1, false) != NULL);
+ unitpp::assert_true("a longer request from the same start should miss",
+ cache.Find(FakeSource(1), 0, 11, 1, false) == NULL);
+ }
+
+ void testKeyMismatchMisses()
+ {
+ TextAnalysisCache cache;
+ StoreTenChars(cache, FakeSource(1), 0, 1);
+ unitpp::assert_true("a different start offset should miss",
+ cache.Find(FakeSource(1), 1, 4, 1, false) == NULL);
+ unitpp::assert_true("a different text source should miss",
+ cache.Find(FakeSource(2), 0, 4, 1, false) == NULL);
+ unitpp::assert_true("a different writing system should miss",
+ cache.Find(FakeSource(1), 0, 4, 2, false) == NULL);
+ unitpp::assert_true("a different direction should miss",
+ cache.Find(FakeSource(1), 0, 4, 1, true) == NULL);
+ }
+
+ void testEvictionIsBounded()
+ {
+ TextAnalysisCache cache(2);
+ StoreTenChars(cache, FakeSource(1), 0, 1);
+ StoreTenChars(cache, FakeSource(2), 0, 1);
+ StoreTenChars(cache, FakeSource(3), 0, 1);
+ unitpp::assert_eq("storing past capacity should evict", 1, cache.EvictionCount());
+ unitpp::assert_true("the oldest entry should be the one evicted",
+ cache.Find(FakeSource(1), 0, 10, 1, false) == NULL);
+ unitpp::assert_true("the newest entry should survive",
+ cache.Find(FakeSource(3), 0, 10, 1, false) != NULL);
+ }
+
+ public:
+ TestTextAnalysisCache();
+ };
+
+ class TestNfcOffsetMap : public unitpp::suite
+ {
+ // NFC length of the text in [ichMin, ichLim), normalized from scratch: the definition
+ // the map must reproduce.
+ static int OracleNfcLength(const StrUni & stu, int ichMin, int ichLim)
+ {
+ StrUni stuRange(stu.Chars() + ichMin, ichLim - ichMin);
+ StrUtil::NormalizeStrUni(stuRange, UNORM_NFC);
+ return stuRange.Length();
+ }
+
+ void VerifyAgainstOracle(const wchar_t * pszText, const char * pszLabel)
+ {
+ StrUni stu(pszText);
+ int cch = stu.Length();
+ TxtSrc ts(pszText, g_qwsf);
+ IVwTextSourcePtr qts;
+ ts.QueryInterface(IID_IVwTextSource, (void **)&qts);
+ NfcOffsetMap map;
+ map.Reset(qts);
+
+ StrAnsi staMsg;
+ int cBoundaries = 0;
+ for (int ichBase = 0; ichBase <= cch; ++ichBase)
+ {
+ int ichNfc;
+ if (!map.IsBoundary(ichBase))
+ {
+ staMsg.Format("%s: base %d is not a boundary so the map must decline",
+ pszLabel, ichBase);
+ unitpp::assert_true(staMsg.Chars(), !map.TryOffsetInNfc(cch, ichBase, &ichNfc));
+ continue;
+ }
+ ++cBoundaries;
+ for (int ich = ichBase; ich <= cch; ++ich)
+ {
+ int cchExpected = OracleNfcLength(stu, ichBase, ich);
+ staMsg.Format("%s: OffsetInNfc(%d, %d) must answer", pszLabel, ich, ichBase);
+ unitpp::assert_true(staMsg.Chars(), map.TryOffsetInNfc(ich, ichBase, &ichNfc));
+ staMsg.Format("%s: OffsetInNfc(%d, %d)", pszLabel, ich, ichBase);
+ unitpp::assert_eq(staMsg.Chars(), cchExpected, ichNfc);
+ }
+ int cchNfcAll = OracleNfcLength(stu, ichBase, cch);
+ for (int ichNfcReq = 0; ichNfcReq <= cchNfcAll + 2; ++ichNfcReq)
+ {
+ int ichExpected = ichBase;
+ for (int ich = ichBase; ich <= cch; ++ich)
+ {
+ if (OracleNfcLength(stu, ichBase, ich) <= ichNfcReq)
+ ichExpected = ich;
+ }
+ int ichOrig;
+ staMsg.Format("%s: OffsetToOrig(%d, %d) must answer", pszLabel, ichNfcReq, ichBase);
+ unitpp::assert_true(staMsg.Chars(),
+ map.TryOffsetToOrig(ichNfcReq, ichBase, &ichOrig));
+ staMsg.Format("%s: OffsetToOrig(%d, %d)", pszLabel, ichNfcReq, ichBase);
+ unitpp::assert_eq(staMsg.Chars(), ichExpected, ichOrig);
+ }
+ }
+ staMsg.Format("%s: text start and end are always boundaries", pszLabel);
+ unitpp::assert_true(staMsg.Chars(), cBoundaries >= (cch == 0 ? 1 : 2));
+ }
+
+ void testMatchesFromScratchNormalization()
+ {
+ VerifyAgainstOracle(L"plain ascii text, no marks", "ascii only");
+ VerifyAgainstOracle(L"cafe\u0301 de\u0301ja\u0300 vu no\u0308el", "decomposed latin");
+ VerifyAgainstOracle(L"caf\u00E9 d\u00E9j\u00E0 vu", "precomposed latin");
+ VerifyAgainstOracle(L"a\u0301\u0327b a\u0327\u0301c", "non-canonical mark order");
+ VerifyAgainstOracle(L"\u1e38 L\u0323\u0304 L\u0304\u0323\u0323 x L\u0304\u0323", "marks that reorder and shorten the form");
+ VerifyAgainstOracle(L"\u0301\u0300abc", "marks with no base");
+ VerifyAgainstOracle(L"\u0628\u064E\u0651 \u0644\u0651\u064E", "arabic harakat");
+ VerifyAgainstOracle(L"\u1112\u1161\u11AB \u1100\u1161", "hangul jamo");
+ VerifyAgainstOracle(L"x\U0001D400\u0301y\U0001D401", "surrogate pairs");
+ VerifyAgainstOracle(L"a\u0344b\u0344", "lengthening mark");
+ VerifyAgainstOracle(L"\u212Bngstro\u0308m \u2126", "singleton decomposition");
+ VerifyAgainstOracle(L"The \u0628\u064E caf\u00E9 e\u0301\u1112\u1161\u11AB\U0001D400\u0344!", "mixed everything");
+ }
+
+ void testCombiningMarkAndLowSurrogateAreNotBoundaries()
+ {
+ TxtSrc ts(L"e\u0301\U0001D400", g_qwsf);
+ IVwTextSourcePtr qts;
+ ts.QueryInterface(IID_IVwTextSource, (void **)&qts);
+ NfcOffsetMap map;
+ map.Reset(qts);
+ unitpp::assert_true("start is a boundary", map.IsBoundary(0));
+ unitpp::assert_true("before a combining mark is not a boundary", !map.IsBoundary(1));
+ unitpp::assert_true("before a high surrogate is a boundary", map.IsBoundary(2));
+ unitpp::assert_true("between surrogates is not a boundary", !map.IsBoundary(3));
+ unitpp::assert_true("end is a boundary", map.IsBoundary(4));
+ }
+
+ void testResetRebindsToAnotherSource()
+ {
+ TxtSrc ts1(L"ab", g_qwsf);
+ TxtSrc ts2(L"e\u0301e\u0301", g_qwsf);
+ IVwTextSourcePtr qts1;
+ IVwTextSourcePtr qts2;
+ ts1.QueryInterface(IID_IVwTextSource, (void **)&qts1);
+ ts2.QueryInterface(IID_IVwTextSource, (void **)&qts2);
+ NfcOffsetMap map;
+ map.Reset(qts1);
+ unitpp::assert_eq("two ascii characters", 2, map.NfcLengthOfPrefix(2));
+ map.Reset(qts2);
+ unitpp::assert_eq("rebinding discards the old table", 2, map.NfcLengthOfPrefix(4));
+ }
+
+ public:
+ TestNfcOffsetMap();
+ virtual void Setup()
+ {
+ CreateTestWritingSystemFactory();
+ }
+ virtual void Teardown()
+ {
+ CloseTestWritingSystemFactory();
+ }
+ };
}
#endif // TESTVIEWCACHES_H_INCLUDED
diff --git a/Src/views/Test/testViews.mak b/Src/views/Test/testViews.mak
index 1b2c2adf54..4835c5ea4b 100644
--- a/Src/views/Test/testViews.mak
+++ b/Src/views/Test/testViews.mak
@@ -154,6 +154,7 @@ $(VIEWSTEST_SRC)\Collection.cpp: $(VIEWSTEST_SRC)\DummyBaseVc.h $(VIEWSTEST_SRC)
$(VIEWSTEST_SRC)\TestTsStrBldr.h\
$(VIEWSTEST_SRC)\TestTsString.h\
$(VIEWSTEST_SRC)\TestTsPropsBldr.h\
- $(VIEWSTEST_SRC)\TestTsTextProps.h
+ $(VIEWSTEST_SRC)\TestTsTextProps.h\
+ $(VIEWSTEST_SRC)\TestViewCaches.h
$(DISPLAY) Collecting tests for $(BUILD_PRODUCT).$(BUILD_EXTENSION)
$(COLLECT) $** $(VIEWSTEST_SRC)\Collection.cpp
diff --git a/Src/views/lib/LayoutCache.h b/Src/views/lib/LayoutCache.h
index c1bc0b6f44..112311140d 100644
--- a/Src/views/lib/LayoutCache.h
+++ b/Src/views/lib/LayoutCache.h
@@ -2,6 +2,10 @@
#ifndef LAYOUTCACHE_INCLUDED
#define LAYOUTCACHE_INCLUDED
+#include "NfcOffsetMap.h"
+
+// Itemization of a text range and its NFC form. With m_fTextIsNfc set, m_vchText equals the
+// source text; otherwise offsets into it translate through the pass's NfcOffsetMap.
class TextAnalysisEntry
{
public:
@@ -12,7 +16,6 @@ class TextAnalysisEntry
m_ws(0),
m_fWsRtl(false),
m_fTextIsNfc(true),
- m_cchNfc(0),
m_citem(0)
{
}
@@ -22,47 +25,6 @@ class TextAnalysisEntry
return m_pts == pts && m_ichMin == ichMin && m_cch >= cch && m_ws == ws && m_fWsRtl == fWsRtl;
}
- int RequestedNfcLength(int cchRequested) const
- {
- if (cchRequested <= 0)
- return 0;
- if (m_fTextIsNfc)
- return cchRequested;
- if (m_vichOrigToNfc.Size() == 0)
- return cchRequested;
- if (cchRequested >= m_vichOrigToNfc.Size())
- return m_vichOrigToNfc[m_vichOrigToNfc.Size() - 1];
- return m_vichOrigToNfc[cchRequested];
- }
-
- int OffsetInNfc(int ich, int ichBase) const
- {
- Assert(ich >= ichBase);
- if (m_fTextIsNfc)
- return ich - ichBase;
- int ichRelative = ich - ichBase;
- if (ichRelative <= 0)
- return 0;
- if (m_vichOrigToNfc.Size() == 0)
- return ichRelative;
- if (ichRelative >= m_vichOrigToNfc.Size())
- return m_vichOrigToNfc[m_vichOrigToNfc.Size() - 1];
- return m_vichOrigToNfc[ichRelative];
- }
-
- int OffsetToOrig(int ich, int ichBase) const
- {
- if (m_fTextIsNfc)
- return ich + ichBase;
- if (ich <= 0)
- return ichBase;
- if (m_vichNfcToOrig.Size() == 0)
- return ich + ichBase;
- if (ich >= m_vichNfcToOrig.Size())
- return m_cch + ichBase;
- return m_vichNfcToOrig[ich] + ichBase;
- }
-
void CopyScriptItemsTo(Vector & vscri, int & citem) const
{
citem = m_citem;
@@ -80,13 +42,13 @@ class TextAnalysisEntry
int m_cch;
int m_ws;
bool m_fWsRtl;
+ // True when NFC normalization leaves the source text unchanged, so m_vchText equals it and
+ // offsets into m_vchText are source offsets.
bool m_fTextIsNfc;
- int m_cchNfc;
int m_citem;
- Vector m_vchNfc;
+ // The NFC form of the range.
+ Vector m_vchText;
Vector m_vscri;
- Vector m_vichOrigToNfc;
- Vector m_vichNfcToOrig;
};
class ShapeRunEntry
@@ -195,7 +157,7 @@ class TextAnalysisCache
TextAnalysisEntry * Store(IVwTextSource * pts, int ichMin, int cch, int ws, bool fWsRtl,
const OLECHAR * prgchNfc, int cchNfc, bool fTextIsNfc, const SCRIPT_ITEM * prgscri,
- int citem, const Vector * pvichOrigToNfc, const Vector * pvichNfcToOrig)
+ int citem)
{
TextAnalysisEntry * pentry = NULL;
for (int ientry = 0; ientry < m_ventry.Size(); ++ientry)
@@ -231,12 +193,11 @@ class TextAnalysisCache
pentry->m_ws = ws;
pentry->m_fWsRtl = fWsRtl;
pentry->m_fTextIsNfc = fTextIsNfc;
- pentry->m_cchNfc = cchNfc;
pentry->m_citem = citem;
- pentry->m_vchNfc.Resize(cchNfc);
+ pentry->m_vchText.Resize(cchNfc);
if (cchNfc > 0)
- ::memcpy(pentry->m_vchNfc.Begin(), prgchNfc, cchNfc * isizeof(OLECHAR));
+ ::memcpy(pentry->m_vchText.Begin(), prgchNfc, cchNfc * isizeof(OLECHAR));
int cscri = citem + 1;
if (cscri < 2)
@@ -245,24 +206,6 @@ class TextAnalysisCache
if (cscri > 0)
::memcpy(pentry->m_vscri.Begin(), prgscri, cscri * isizeof(SCRIPT_ITEM));
- if (pvichOrigToNfc)
- {
- pentry->m_vichOrigToNfc.Resize(pvichOrigToNfc->Size());
- for (int i = 0; i < pvichOrigToNfc->Size(); ++i)
- pentry->m_vichOrigToNfc[i] = (*pvichOrigToNfc)[i];
- }
- else
- pentry->m_vichOrigToNfc.Delete(0, pentry->m_vichOrigToNfc.Size());
-
- if (pvichNfcToOrig)
- {
- pentry->m_vichNfcToOrig.Resize(pvichNfcToOrig->Size());
- for (int i = 0; i < pvichNfcToOrig->Size(); ++i)
- pentry->m_vichNfcToOrig[i] = (*pvichNfcToOrig)[i];
- }
- else
- pentry->m_vichNfcToOrig.Delete(0, pentry->m_vichNfcToOrig.Size());
-
return pentry;
}
@@ -420,6 +363,7 @@ class LayoutPassCache
{
m_analysisCache.Reset();
m_shapeRunCache.Reset();
+ m_nfcOffsets.Clear();
}
TextAnalysisCache & AnalysisCache()
@@ -427,6 +371,14 @@ class LayoutPassCache
return m_analysisCache;
}
+ // The offset map for pts. The pass keeps one map, so asking for a different text source
+ // than the last call discards the table and the next lookup rebuilds it.
+ NfcOffsetMap & NfcOffsetsFor(IVwTextSource * pts)
+ {
+ m_nfcOffsets.Reset(pts);
+ return m_nfcOffsets;
+ }
+
ShapeRunCache & ShapeCache()
{
return m_shapeRunCache;
@@ -435,6 +387,7 @@ class LayoutPassCache
private:
TextAnalysisCache m_analysisCache;
ShapeRunCache m_shapeRunCache;
+ NfcOffsetMap m_nfcOffsets;
};
extern __declspec(thread) LayoutPassCache * g_pCurrentLayoutPassCache;
@@ -457,6 +410,7 @@ inline bool IsPath1ShapeCacheEnabled()
return s_nEnabled == 1;
}
+// Also governs the NfcOffsetMap, which exists to serve the analysis cache's decomposed entries.
inline bool IsPath2AnalysisCacheEnabled()
{
static int s_nEnabled = -1;
diff --git a/Src/views/lib/NfcOffsetMap.h b/Src/views/lib/NfcOffsetMap.h
new file mode 100644
index 0000000000..32eeab7330
--- /dev/null
+++ b/Src/views/lib/NfcOffsetMap.h
@@ -0,0 +1,208 @@
+/*--------------------------------------------------------------------*//*:Ignore this sentence.
+Copyright (c) 2026 SIL International
+This software is licensed under the LGPL, version 2.1 or later
+(http://www.gnu.org/licenses/lgpl-2.1.html)
+
+File: NfcOffsetMap.h
+Responsibility:
+Last reviewed: Not yet.
+
+Description:
+ Translates offsets between a text source and its NFC-normalized form without
+ renormalizing a prefix of the text on every request (LT-22674).
+-------------------------------------------------------------------------------*//*:End Ignore*/
+#pragma once
+#ifndef NFCOFFSETMAP_INCLUDED
+#define NFCOFFSETMAP_INCLUDED
+
+/*----------------------------------------------------------------------------------------------
+ Offset translation for one text source between its own (typically NFD) offsets and offsets
+ into its NFC-normalized form.
+
+ An offset into the NFC form of a range is defined as the NFC length of that range, exactly
+ as normalizing the range from scratch would give. The map reproduces that definition while
+ normalizing each character at most once: it splits the text where ICU says normalization
+ cannot cross a boundary, records the NFC length of the text before each boundary, and
+ normalizes only the tail from the nearest boundary when a request lands inside a chunk.
+ Requests whose base offset is not a boundary cannot be answered from the table and report
+ failure, so the caller can fall back to normalizing the range directly.
+
+ The map is bound to one text source whose content does not change while the map is in
+ use; Reset rebinds it.
+----------------------------------------------------------------------------------------------*/
+class NfcOffsetMap
+{
+public:
+ NfcOffsetMap() : m_pts(NULL), m_cchText(0), m_fBuilt(false)
+ {
+ }
+
+ // Binds the map to a text source, discarding any table built for a different one.
+ void Reset(IVwTextSource * pts)
+ {
+ if (pts == m_pts && m_fBuilt)
+ return;
+ Clear();
+ m_pts = pts;
+ }
+
+ // Unbinds the map and discards its table.
+ void Clear()
+ {
+ m_pts = NULL;
+ m_cchText = 0;
+ m_fBuilt = false;
+ m_vchText.Clear();
+ m_vichBoundary.Clear();
+ m_vcchNfcBefore.Clear();
+ }
+
+ // True when normalization of the text before ich is independent of the text from ich on.
+ // The start and end of the text are boundaries; the middle of a surrogate pair is not.
+ bool IsBoundary(int ich)
+ {
+ if (!Build() || ich < 0 || ich > m_cchText)
+ return false;
+ if (ich == 0 || ich == m_cchText)
+ return true;
+ return BoundaryIndexAt(ich) >= 0;
+ }
+
+ // NFC length of the text from ichBase to ich. Fails unless ichBase is a boundary.
+ bool TryOffsetInNfc(int ich, int ichBase, int * pichNfc)
+ {
+ *pichNfc = 0;
+ if (!Build() || ichBase < 0 || ich < ichBase || ich > m_cchText || !IsBoundary(ichBase))
+ return false;
+ *pichNfc = NfcLengthOfPrefix(ich) - NfcLengthOfPrefix(ichBase);
+ return true;
+ }
+
+ // The longest range starting at ichBase whose NFC form is at most ichNfc characters long,
+ // reported as the offset of its end. Fails unless ichBase is a boundary.
+ bool TryOffsetToOrig(int ichNfc, int ichBase, int * pichOrig)
+ {
+ *pichOrig = ichBase;
+ if (!Build() || ichBase < 0 || ichBase > m_cchText || ichNfc < 0 || !IsBoundary(ichBase))
+ return false;
+ int cchNfcBase = NfcLengthOfPrefix(ichBase);
+ // Find the last boundary at or after ichBase whose prefix length still fits, then
+ // walk forward one character at a time inside that chunk.
+ int iLow = BoundaryIndexAtOrBefore(ichBase);
+ int iHigh = m_vichBoundary.Size() - 1;
+ while (iLow < iHigh)
+ {
+ int iMid = iLow + (iHigh - iLow + 1) / 2;
+ if (m_vcchNfcBefore[iMid] - cchNfcBase <= ichNfc)
+ iLow = iMid;
+ else
+ iHigh = iMid - 1;
+ }
+ // Prefix lengths need not grow inside a chunk: a mark that reorders ahead of another can
+ // compose with the base and shorten the form, so the walk keeps the last offset that
+ // fits.
+ int ichFit = max(m_vichBoundary[iLow], ichBase);
+ int ichChunkLim = iLow + 1 < m_vichBoundary.Size() ? m_vichBoundary[iLow + 1] : m_cchText;
+ for (int ich = ichFit; ich < ichChunkLim; ++ich)
+ {
+ if (NfcLengthOfPrefix(ich + 1) - cchNfcBase <= ichNfc)
+ ichFit = ich + 1;
+ }
+ *pichOrig = ichFit;
+ return true;
+ }
+
+ // NFC length of the text before ich, computed from the nearest boundary at or before it.
+ int NfcLengthOfPrefix(int ich)
+ {
+ if (!Build() || ich <= 0)
+ return 0;
+ Assert(ich <= m_cchText);
+ int iBoundary = BoundaryIndexAtOrBefore(ich);
+ int ichBoundary = m_vichBoundary[iBoundary];
+ int cchNfc = m_vcchNfcBefore[iBoundary];
+ if (ich > ichBoundary)
+ cchNfc += NfcLength(ichBoundary, ich);
+ return cchNfc;
+ }
+
+private:
+ // Fetches the whole text and records every normalization boundary with the NFC length
+ // of the text before it. Chunks between boundaries are normalized independently.
+ bool Build()
+ {
+ if (m_fBuilt)
+ return true;
+ if (!m_pts)
+ return false;
+ CheckHr(m_pts->get_Length(&m_cchText));
+ m_vchText.Resize(m_cchText);
+ if (m_cchText > 0)
+ CheckHr(m_pts->Fetch(0, m_cchText, m_vchText.Begin()));
+
+ const icu::Normalizer2 * pnorm = SilUtil::GetIcuNormalizer(UNORM_NFC);
+ m_vichBoundary.Push(0);
+ m_vcchNfcBefore.Push(0);
+ int ichChunk = 0;
+ int cchNfc = 0;
+ int ich = 0;
+ while (ich < m_cchText)
+ {
+ int ichNext = ich;
+ UChar32 ch;
+ U16_NEXT(m_vchText.Begin(), ichNext, m_cchText, ch);
+ if (ich > 0 && pnorm->hasBoundaryBefore(ch))
+ {
+ cchNfc += NfcLength(ichChunk, ich);
+ m_vichBoundary.Push(ich);
+ m_vcchNfcBefore.Push(cchNfc);
+ ichChunk = ich;
+ }
+ ich = ichNext;
+ }
+ m_fBuilt = true;
+ return true;
+ }
+
+ // NFC length of the text in [ichMin, ichLim).
+ int NfcLength(int ichMin, int ichLim)
+ {
+ if (ichLim <= ichMin)
+ return 0;
+ StrUni stu(m_vchText.Begin() + ichMin, ichLim - ichMin);
+ StrUtil::NormalizeStrUni(stu, UNORM_NFC);
+ return stu.Length();
+ }
+
+ // Index of the last recorded boundary at or before ich; the table always holds 0.
+ int BoundaryIndexAtOrBefore(int ich)
+ {
+ int iLow = 0;
+ int iHigh = m_vichBoundary.Size() - 1;
+ while (iLow < iHigh)
+ {
+ int iMid = iLow + (iHigh - iLow + 1) / 2;
+ if (m_vichBoundary[iMid] <= ich)
+ iLow = iMid;
+ else
+ iHigh = iMid - 1;
+ }
+ return iLow;
+ }
+
+ // Index of the boundary recorded exactly at ich, or -1.
+ int BoundaryIndexAt(int ich)
+ {
+ int i = BoundaryIndexAtOrBefore(ich);
+ return m_vichBoundary[i] == ich ? i : -1;
+ }
+
+ IVwTextSource * m_pts;
+ int m_cchText;
+ bool m_fBuilt;
+ Vector m_vchText;
+ Vector m_vichBoundary;
+ Vector m_vcchNfcBefore;
+};
+
+#endif // NFCOFFSETMAP_INCLUDED
diff --git a/Src/views/lib/UniscribeEngine.cpp b/Src/views/lib/UniscribeEngine.cpp
index 570217645b..03b6077374 100644
--- a/Src/views/lib/UniscribeEngine.cpp
+++ b/Src/views/lib/UniscribeEngine.cpp
@@ -414,9 +414,8 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
// clear.
// PATH-N1: Get NFC flag to avoid redundant OffsetInNfc/OffsetToOrig normalization below.
bool fTextIsNfc = false;
- const TextAnalysisEntry * pAnalysis = NULL;
int cchNfc = UniscribeSegment::CallScriptItemize(rgchBuf, INIT_BUF_SIZE, vch, pts, ichMinSeg,
- ichLimText - ichMinSeg, &prgchBuf, citem, (bool)fParaRtoL, &fTextIsNfc, &pAnalysis);
+ ichLimText - ichMinSeg, &prgchBuf, citem, (bool)fParaRtoL, &fTextIsNfc);
Vector vichBreak;
ILgLineBreakerPtr qlb;
@@ -514,7 +513,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
}
ichLim = min(ichLimNext, ichLimBT2);
// Optimize JohnT: if ichLim==ichBase+m_dichLim, can use cchNfc.
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
if (ichLimNfc == ichMinNfc)
{
// This can happen if later characters in a composite have different properties than the first.
@@ -538,7 +537,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
{
// Script item is smaller than run; shorten the amount we treat as a 'run'.
ichLimNfc = (pscri + 1)->iCharPos;
- ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc);
}
// Set up the characters of the run, if any.
@@ -765,7 +764,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
vdxRun.Pop();
ichLimBT2 = ichMin;
ichLim = *(vichRun.Top());
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
vichRun.Pop();
cglyph = *(viglyphRun.Top());
viglyphRun.Pop();
@@ -870,7 +869,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
vdxRun.Pop();
ichLimBT2 = ichMin;
ichLim = *(vichRun.Top());
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
vichRun.Pop();
cglyph = *(viglyphRun.Top());
viglyphRun.Pop();
@@ -984,11 +983,11 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
}
}
}
- ichLim = UniscribeSegment::OffsetToOrig(ichMinUri + ichRun, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLim = UniscribeSegment::OffsetToOrig(ichMinUri + ichRun, ichMinSeg, pts, fTextIsNfc);
break;
}
- int ichLineBreak = UniscribeSegment::OffsetToOrig(ichLineBreakNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ int ichLineBreak = UniscribeSegment::OffsetToOrig(ichLineBreakNfc, ichMinSeg, pts, fTextIsNfc);
if (ichLineBreak <= ichMin)
{
@@ -1005,7 +1004,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
viglyphRun.Pop();
}
ichLim = ichMin; // Required to get correct values at start of loop.
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
ichLimBT2 = ichLineBreak;
fRemovedWs = false;
fBacktracking = true;
@@ -1013,12 +1012,12 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
}
ichLim = ichLineBreak; // We limit the segment to not exceed the latest line break point.
Assert(ichLim <= ichLimBacktrack);
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
fOkBreak = true; // Means we have a good line break.
// Store the glyph-specific information: stretch values.
int cchRunTotalTmp = uri.cch;
- uri.cch = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis) - ichMinNfc;
+ uri.cch = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc) - ichMinNfc;
UniscribeSegment::ShapePlaceRun(uri, true);
viglyphRun.Push(cglyph);
cglyph += uri.cglyph;
@@ -1044,7 +1043,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
if (twsh == ktwshNoWs)
{
fRemovedWs = RemoveTrailingWhiteSpace(ichMinUri, &ichLimNfc, uri);
- ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc);
// Usually the worst case is that ichLimNfc == ichMinUri, indicating that the whole run is
// white space. However, in at least one pathological case, we have observed uniscribe
// strip of more than one run of white space. Hence the <=.
@@ -1068,7 +1067,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
dxSegWidth = *(vdxRun.Top());
vdxRun.Pop();
ichLim = *(vichRun.Top());
- ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc);
vichRun.Pop();
cglyph = *(viglyphRun.Top());
viglyphRun.Pop();
@@ -1097,7 +1096,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint(
Assert(irun == 1);
Assert(ichMinUri == 0);
RemoveNonWhiteSpace(ichMinUri, &ichLimNfc, uri);
- ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis);
+ ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc);
if (ichLim == ichMinSeg)
return S_OK; // failure to create a valid segment
fOkBreak = true;
diff --git a/Src/views/lib/UniscribeSegment.cpp b/Src/views/lib/UniscribeSegment.cpp
index f40bd24bfc..17687b0625 100644
--- a/Src/views/lib/UniscribeSegment.cpp
+++ b/Src/views/lib/UniscribeSegment.cpp
@@ -29,8 +29,6 @@ DEFINE_THIS_FILE
//:>********************************************************************************************
//:> Forward declarations
//:>********************************************************************************************
-static void BuildNfcOffsetMaps(const StrUni & stuOrig, Vector & vichOrigToNfc,
- Vector & vichNfcToOrig);
static void ApplyShapeRunCacheEntry(ShapeRunEntry & entry, UniscribeRunInfo & uri);
//:>********************************************************************************************
@@ -1631,18 +1629,21 @@ int UniscribeSegment::OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, boo
Assert(ich >= ichBase);
if (fTextIsNfc)
return ich - ichBase;
+ int ichNfc;
+ NfcOffsetMap * pmap = CurrentNfcOffsetMap(pts);
+ if (pmap && pmap->TryOffsetInNfc(ich, ichBase, &ichNfc))
+ return ichNfc;
return OffsetInNfc(ich, ichBase, pts);
}
-int UniscribeSegment::OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc,
- const TextAnalysisEntry * pAnalysis)
+// The layout pass's offset map for pts, or NULL outside a layout pass. FW_PERF_P125_PATH2=0
+// turns the map off together with the analysis cache it serves.
+NfcOffsetMap * UniscribeSegment::CurrentNfcOffsetMap(IVwTextSource * pts)
{
- Assert(ich >= ichBase);
- if (fTextIsNfc)
- return ich - ichBase;
- if (pAnalysis)
- return pAnalysis->OffsetInNfc(ich, ichBase);
- return OffsetInNfc(ich, ichBase, pts, fTextIsNfc);
+ if (!IsPath2AnalysisCacheEnabled())
+ return NULL;
+ LayoutPassCache * pLayoutPassCache = GetCurrentLayoutPassCache();
+ return pLayoutPassCache ? &pLayoutPassCache->NfcOffsetsFor(pts) : NULL;
}
// ich is an offset into the (NFC normalized) characters of this segment.
@@ -1689,43 +1690,13 @@ int UniscribeSegment::OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bo
{
if (fTextIsNfc)
return ich + ichBase;
+ int ichOrig;
+ NfcOffsetMap * pmap = CurrentNfcOffsetMap(pts);
+ if (pmap && pmap->TryOffsetToOrig(ich, ichBase, &ichOrig))
+ return ichOrig;
return OffsetToOrig(ich, ichBase, pts);
}
-int UniscribeSegment::OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc,
- const TextAnalysisEntry * pAnalysis)
-{
- if (fTextIsNfc)
- return ich + ichBase;
- if (pAnalysis)
- return pAnalysis->OffsetToOrig(ich, ichBase);
- return OffsetToOrig(ich, ichBase, pts, fTextIsNfc);
-}
-
-static void BuildNfcOffsetMaps(const StrUni & stuOrig, Vector & vichOrigToNfc,
- Vector & vichNfcToOrig)
-{
- int cchOrig = stuOrig.Length();
- vichOrigToNfc.Resize(cchOrig + 1);
- vichOrigToNfc[0] = 0;
- for (int ich = 1; ich <= cchOrig; ++ich)
- {
- StrUni stuPrefix(stuOrig.Chars(), ich);
- StrUtil::NormalizeStrUni(stuPrefix, UNORM_NFC);
- vichOrigToNfc[ich] = stuPrefix.Length();
- }
-
- int cchNfc = vichOrigToNfc[cchOrig];
- vichNfcToOrig.Resize(cchNfc + 1);
- int ichOrig = 0;
- for (int ichNfc = 0; ichNfc <= cchNfc; ++ichNfc)
- {
- while (ichOrig + 1 <= cchOrig && vichOrigToNfc[ichOrig + 1] <= ichNfc)
- ++ichOrig;
- vichNfcToOrig[ichNfc] = ichOrig;
- }
-}
-
static void ApplyShapeRunCacheEntry(ShapeRunEntry & entry, UniscribeRunInfo & uri)
{
if (uri.CGlyphMax() < entry.m_cglyph)
@@ -3092,11 +3063,9 @@ void UniscribeSegment::AdjustEndForWidth(int ichBase, IVwGraphics * pvg)
----------------------------------------------------------------------------------------------*/
int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf,
Vector & vch, IVwTextSource * pts, int ichMin, int cch, OLECHAR ** pprgchBuf,
- int & citem, bool fParaRTL, bool * pfTextIsNfc, const TextAnalysisEntry ** ppAnalysis)
+ int & citem, bool fParaRTL, bool * pfTextIsNfc)
{
* pprgchBuf = prgchDefBuf; // Use on-stack variable if big enough
- if (ppAnalysis)
- *ppAnalysis = NULL;
bool fNeedOpenTypeScriptTags = cch > 0 && TextRangeHasOpenTypeFeatures(pts, ichMin, cch);
LayoutPassCache * pLayoutPassCache = (!fNeedOpenTypeScriptTags && IsPath2AnalysisCacheEnabled()) ?
@@ -3114,27 +3083,41 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf,
ws = chrp.ws;
fWsRtl = chrp.fWsRtl;
pCachedAnalysis = pLayoutPassCache->AnalysisCache().Find(pts, ichMin, cchOrig, ws, fWsRtl);
+ int cchNfcHit = cchOrig;
+ if (pCachedAnalysis && !pCachedAnalysis->m_fTextIsNfc)
+ {
+ // A shorter request is served from a longer entry only when its end is a
+ // normalization boundary; otherwise the entry's NFC form is not a prefix of the
+ // request's own NFC form.
+ NfcOffsetMap & map = pLayoutPassCache->NfcOffsetsFor(pts);
+ if (cchOrig == pCachedAnalysis->m_cch)
+ cchNfcHit = pCachedAnalysis->m_vchText.Size();
+ else if (!map.IsBoundary(ichMin + cchOrig) ||
+ !map.TryOffsetInNfc(ichMin + cchOrig, ichMin, &cchNfcHit))
+ {
+ pCachedAnalysis = NULL;
+ }
+ }
if (pCachedAnalysis)
{
if (pfTextIsNfc)
*pfTextIsNfc = pCachedAnalysis->m_fTextIsNfc;
- *pprgchBuf = pCachedAnalysis->m_vchNfc.Size() > 0 ? pCachedAnalysis->m_vchNfc.Begin() : prgchDefBuf;
+ *pprgchBuf = pCachedAnalysis->m_vchText.Size() > 0 ? pCachedAnalysis->m_vchText.Begin() : prgchDefBuf;
pCachedAnalysis->CopyScriptItemsTo(g_vscri, citem);
if (g_votScriptTags.Size() < citem)
g_votScriptTags.Resize(citem);
for (int itag = 0; itag < citem; ++itag)
g_votScriptTags[itag] = 0;
g_cscri = citem;
- if (ppAnalysis)
- *ppAnalysis = pCachedAnalysis;
- return pCachedAnalysis->RequestedNfcLength(cchOrig);
+ return cchNfcHit;
}
}
if (pLayoutPassCache && !pCachedAnalysis)
dwStartMs = ::GetTickCount();
- Vector vichOrigToNfc;
- Vector vichNfcToOrig;
+ // True when NFC normalization leaves the fetched text unchanged, so offsets into the NFC
+ // buffer are source offsets.
+ bool fTextIsNfc = true;
#ifdef UNISCRIBE_NFC
if (cch)
@@ -3154,8 +3137,7 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf,
bool fComputedTextIsNfc = (stu == stuOrig);
if (pfTextIsNfc)
*pfTextIsNfc = fComputedTextIsNfc;
- if (!fComputedTextIsNfc && pLayoutPassCache)
- BuildNfcOffsetMaps(stuOrig, vichOrigToNfc, vichNfcToOrig);
+ fTextIsNfc = fComputedTextIsNfc;
if (cch > cchBuf)
{
cchBuf = cch;
@@ -3168,13 +3150,6 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf,
{
if (pfTextIsNfc)
*pfTextIsNfc = true; // Empty text is trivially NFC
- if (pLayoutPassCache)
- {
- vichOrigToNfc.Resize(1);
- vichOrigToNfc[0] = 0;
- vichNfcToOrig.Resize(1);
- vichNfcToOrig[0] = 0;
- }
}
#else
if (pfTextIsNfc)
@@ -3317,12 +3292,8 @@ typedef struct tag_SCRIPT_STATE {
if (pLayoutPassCache)
{
pLayoutPassCache->AnalysisCache().AddComputeMs(::GetTickCount() - dwStartMs);
- TextAnalysisEntry * pStoredAnalysis = pLayoutPassCache->AnalysisCache().Store(pts, ichMin,
- cchOrig, ws, fWsRtl, *pprgchBuf, cch, pfTextIsNfc ? *pfTextIsNfc : true,
- g_vscri.Begin(), citem, vichOrigToNfc.Size() ? &vichOrigToNfc : NULL,
- vichNfcToOrig.Size() ? &vichNfcToOrig : NULL);
- if (ppAnalysis)
- *ppAnalysis = pStoredAnalysis;
+ pLayoutPassCache->AnalysisCache().Store(pts, ichMin, cchOrig, ws, fWsRtl, *pprgchBuf, cch,
+ fTextIsNfc, g_vscri.Begin(), citem);
}
return cch;
}
@@ -3391,9 +3362,8 @@ template int UniscribeSegment::DoAllRuns(int ichBase, IVwGraphics * pv
OLECHAR * prgchBuf; // Where text actually goes.
// PATH-N1: Get NFC flag from CallScriptItemize to skip redundant OffsetInNfc calls below.
bool fTextIsNfc = false;
- const TextAnalysisEntry * pAnalysis = NULL;
int cchNfc = CallScriptItemize(rgchBuf, INIT_BUF_SIZE, vch, m_qts, ichBase, m_dichLim, &prgchBuf,
- citem, m_fParaRTL, &fTextIsNfc, &pAnalysis);
+ citem, m_fParaRTL, &fTextIsNfc);
// If dxdExpectedWidth is not 0, then the segment will try its best to stretch to the
// specified size.
@@ -3453,7 +3423,7 @@ template int UniscribeSegment::DoAllRuns(int ichBase, IVwGraphics * pv
if (ichLim - ichBase > m_dichLim)
ichLim = ichBase + m_dichLim;
- ichLimNfc = OffsetInNfc(ichLim, ichBase, m_qts, fTextIsNfc, pAnalysis);
+ ichLimNfc = OffsetInNfc(ichLim, ichBase, m_qts, fTextIsNfc);
if (ichLimNfc == ichMinNfc && m_dichLim > 0)
{
// This can happen pathologically where later characters in a composition have different
diff --git a/Src/views/lib/UniscribeSegment.h b/Src/views/lib/UniscribeSegment.h
index a62b44c3f1..c9f4b72fdf 100644
--- a/Src/views/lib/UniscribeSegment.h
+++ b/Src/views/lib/UniscribeSegment.h
@@ -30,7 +30,7 @@ typedef Vector ScrItemVec; // Hungarian vscri;
typedef Vector ScrLogAttrVec; // Hungarian vsla.
typedef Vector OpenTypeTagVec; // Hungarian vot.
-class TextAnalysisEntry;
+class NfcOffsetMap;
/*----------------------------------------------------------------------------------------------
Class: UniscribeRunInfo
@@ -237,12 +237,9 @@ class UniscribeSegment : public ILgSegment
static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts);
static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc);
- static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc,
- const TextAnalysisEntry * pAnalysis);
+ static NfcOffsetMap * CurrentNfcOffsetMap(IVwTextSource * pts);
static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts);
static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc);
- static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc,
- const TextAnalysisEntry * pAnalysis);
protected:
// Static variables
@@ -310,7 +307,7 @@ class UniscribeSegment : public ILgSegment
static void ShapePlaceRun(UniscribeRunInfo& uri, bool fCreatingSeg = false);
static int CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, Vector & vch,
IVwTextSource * pts, int ichMin, int cch, OLECHAR ** pprgchBuf, int & citem,
- bool fParaRTL, bool * pfTextIsNfc = NULL, const TextAnalysisEntry ** ppAnalysis = NULL);
+ bool fParaRTL, bool * pfTextIsNfc = NULL);
int NumStretchableGlyphs();
int StretchGlyphs(UniscribeRunInfo & uri,