diff --git a/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs b/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs index 91cc09d374..e64903f076 100644 --- a/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs +++ b/Src/Common/RootSite/RootSiteTests/RenderBenchmarkTestsBase.cs @@ -25,6 +25,8 @@ namespace SIL.FieldWorks.Common.RootSites.RenderBenchmark public abstract class RenderBenchmarkTestsBase : RealDataTestsBase { protected const string DeterministicRenderFontFamily = "Segoe UI"; + // Second Latin font for scenarios whose runs alternate font family. + protected const string SecondaryRenderFontFamily = "Times New Roman"; // Pinned Arabic font (loaded privately by RenderTestAssemblySetup). Used for Arabic runs so // they don't depend on the host's Segoe UI Arabic version / font fallback. protected const string ArabicRenderFontFamily = "Scheherazade New"; @@ -320,6 +322,15 @@ protected void SetupScenarioData(string scenarioId) case "multi-ws": CreateMultiWsScenario(); break; + case "single-para-mixed-ws": + CreateSingleParaMixedWsScenario(); + break; + case "nfc-composable-diacritics": + CreateNfcComposableDiacriticsScenario(); + break; + case "multi-line-wrap-single-ws": + CreateMultiLineWrapSingleWsScenario(); + break; case "lex-shallow": CreateLexEntryScenario(depth: 2, breadth: 3); break; @@ -539,6 +550,27 @@ private void CreateMultiWsScenario() AddMultiWsSections(book, 5, versesPerSection: 8, chapterStart: 1); } + private void CreateSingleParaMixedWsScenario() + { + var book = CreateBook(19); // PSA + m_hvoRoot = book.Hvo; + AddSingleMixedWsParagraph(book, sentenceCount: 236); + } + + private void CreateNfcComposableDiacriticsScenario() + { + var book = CreateBook(15); // EZR + m_hvoRoot = book.Hvo; + AddNfcComposableDiacriticsParagraph(book, wordCount: 80); + } + + private void CreateMultiLineWrapSingleWsScenario() + { + var book = CreateBook(17); // EST + m_hvoRoot = book.Hvo; + AddSingleWsProseParagraph(book, sentenceCount: 200); + } + #region Rich Data Factories protected IScrBook CreateBook(int bookNum) @@ -928,6 +960,147 @@ protected void AddMultiWsSections(IScrBook book, int sectionCount, } } + /// + /// Returns the canonical decomposition (NFD) of , and fails the + /// fixture unless the result contains combining marks that NFC would compose. A scenario + /// built from it therefore always exercises text that normalization rewrites, whatever + /// normalization form the source file itself is stored in. + /// + protected static string Decomposed(string text) + { + string nfd = CustomIcu.GetIcuNormalizer(FwNormalizationMode.knmNFD).Normalize(text); + string nfc = CustomIcu.GetIcuNormalizer(FwNormalizationMode.knmNFC).Normalize(text); + Assert.That(nfd, Is.Not.EqualTo(nfc), + $"Scenario text '{text}' must contain characters that decompose."); + return nfd; + } + + /// + /// Adds a section whose content is one paragraph of + /// sentences. Each sentence's text and run properties come from the two callbacks, which + /// receive the sentence index. With the paragraph + /// opens with a chapter-number run. + /// + private void AddSingleParagraphSection(IScrBook book, string heading, int sentenceCount, + Func sentence, Func runProps, bool withChapterNumber = false) + { + var section = Cache.ServiceLocator.GetInstance().Create(); + book.SectionsOS.Add(section); + + var stTextFactory = Cache.ServiceLocator.GetInstance(); + + section.HeadingOA = stTextFactory.Create(); + var headingBldr = new StTxtParaBldr(Cache) { ParaStyleName = ScrStyleNames.SectionHead }; + headingBldr.AppendRun(heading, StyleUtils.CharStyleTextProps(null, m_wsEng)); + headingBldr.CreateParagraph(section.HeadingOA); + + section.ContentOA = stTextFactory.Create(); + var paraBldr = new StTxtParaBldr(Cache) { ParaStyleName = ScrStyleNames.NormalParagraph }; + if (withChapterNumber) + paraBldr.AppendRun("1", StyleUtils.CharStyleTextProps(ScrStyleNames.ChapterNumber, m_wsEng)); + + for (int i = 0; i < sentenceCount; i++) + paraBldr.AppendRun(sentence(i), runProps(i)); + + paraBldr.CreateParagraph(section.ContentOA); + } + + /// + /// Adds one paragraph of short decomposed sentences whose runs alternate between two + /// writing systems and two font families. + /// + protected void AddSingleMixedWsParagraph(IScrBook book, int sentenceCount) + { + string[] subjects = + { + "the \u00E9lder", + "the h\u00E8rder", + "the s\u00EFnger", + "the tea\u00E7her", + "the trav\u00EAler", + "the wom\u00E3n" + }; + string[] predicates = + { + "spoke of the l\u00F3ng rains", + "walked to the f\u00E0r well", + "named the sev\u00EBn hills", + "counted the cattl\u00E9 at dusk", + "kept the \u00F4ld story", + "asked for a bless\u0129ng" + }; + + AddSingleParagraphSection(book, "Single Paragraph, Mixed Writing Systems", sentenceCount, + i => Decomposed($"{subjects[i % subjects.Length]} {predicates[i % predicates.Length]} {i + 1}. "), + i => AlternatingFontRunProps(i % 2 == 0), + withChapterNumber: true); + } + + private ITsTextProps AlternatingFontRunProps(bool first) + { + var bldr = TsStringUtils.MakePropsBldr(); + bldr.SetIntPropValues((int)FwTextPropType.ktptWs, (int)FwTextPropVar.ktpvDefault, + first ? m_wsEng : m_wsFr); + bldr.SetStrPropValue((int)FwTextPropType.ktptFontFamily, + first ? DeterministicRenderFontFamily : SecondaryRenderFontFamily); + return bldr.GetTextProps(); + } + + /// + /// Adds one wrapped paragraph of Latin sentences, each naming a word spelled with + /// decomposed diacritics. + /// + protected void AddNfcComposableDiacriticsParagraph(IScrBook book, int wordCount) + { + string[] words = + { + "caf\u00E9", + "d\u00E9j\u00E0", + "no\u00EBl", + "fran\u00E7ais", + "gar\u00E7on", + "h\u00F4tel", + "a\u00F1o", + "cr\u00E9\u00E9e", + "\u00E9l\u00E9gant", + "fa\u00E7ade" + }; + + AddSingleParagraphSection(book, "Decomposed Diacritics Microbenchmark", wordCount, + i => Decomposed($"The {words[i % words.Length]} recorded here is entry {i + 1}. "), + i => StyleUtils.CharStyleTextProps(null, m_wsEng)); + } + + /// + /// Adds one long wrapped paragraph of decomposed sentences in a single writing system + /// and font. + /// + protected void AddSingleWsProseParagraph(IScrBook book, int sentenceCount) + { + string[] subjects = + { + "the mer\u00E7hant", + "the mas\u00F3n", + "the scrib\u00EB", + "the sheph\u00E8rd", + "the weav\u00EAr", + "the pott\u00E9r" + }; + string[] predicates = + { + "measured the grain by the riv\u00E9r", + "repaired the east\u00E8rn wall before dusk", + "copied the ledger onto fresh parchm\u00EBnt", + "counted the flock past the old gat\u00EA", + "dyed the cloth a deep saffr\u00F5n", + "shaped the jar on the slow whe\u00EBl" + }; + + AddSingleParagraphSection(book, "Single Writing-System Line-Wrap Microbenchmark", sentenceCount, + i => Decomposed($"{subjects[i % subjects.Length]} {predicates[i % predicates.Length]} on day {i + 1}. "), + i => StyleUtils.CharStyleTextProps(null, m_wsEng)); + } + #endregion #region Lex Entry Scenario Data diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png new file mode 100644 index 0000000000..9d19d891b4 Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_multi-line-wrap-single-ws.verified.png differ diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png new file mode 100644 index 0000000000..1925ebb15f Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_nfc-composable-diacritics.verified.png differ diff --git a/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png new file mode 100644 index 0000000000..38c21c2c42 Binary files /dev/null and b/Src/Common/RootSite/RootSiteTests/RenderVerifyTests.VerifyScenario_single-para-mixed-ws.verified.png differ diff --git a/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json b/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json index 782b92e682..305b23157f 100644 --- a/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json +++ b/Src/Common/RootSite/RootSiteTests/TestData/RenderBenchmarkScenarios.json @@ -89,6 +89,21 @@ "description": "Lex entry with senses nested 6 levels deep, 2-wide (depth 6, breadth 2 = 126 senses)", "tags": ["lex-entry", "nested-senses", "exponential-cost", "stress"], "viewType": "LexEntry" + }, + { + "id": "single-para-mixed-ws", + "description": "One paragraph of 236 short decomposed (NFD) sentences whose runs alternate two writing systems and two font families", + "tags": ["stress", "layout-stress", "multi-ws", "single-paragraph", "nfd"] + }, + { + "id": "nfc-composable-diacritics", + "description": "One wrapped paragraph of Latin prose spelled with decomposed (NFD) diacritics that NFC composes", + "tags": ["stress", "layout-stress", "nfd", "diacritics", "line-breaking"] + }, + { + "id": "multi-line-wrap-single-ws", + "description": "One long paragraph of 200 decomposed (NFD) sentences in one writing system and font, wrapping over many lines", + "tags": ["stress", "layout-stress", "single-paragraph", "line-breaking", "nfd"] } ] } diff --git a/Src/views/Test/TestUniscribeEngine.h b/Src/views/Test/TestUniscribeEngine.h index 4780767501..d627bb9a37 100644 --- a/Src/views/Test/TestUniscribeEngine.h +++ b/Src/views/Test/TestUniscribeEngine.h @@ -16,6 +16,7 @@ Last reviewed: #include "testViews.h" #include "RenderEngineTestBase.h" +#include "LayoutCache.h" namespace TestViews { @@ -396,6 +397,94 @@ namespace TestViews #endif } + // Installs a fresh layout-pass cache for the lifetime of the object, restoring the + // previous one even when a test assertion throws. + class ScopedLayoutPassCache + { + public: + ScopedLayoutPassCache() : m_pPrev(SetCurrentLayoutPassCache(&m_cache)) {} + ~ScopedLayoutPassCache() { SetCurrentLayoutPassCache(m_pPrev); } + LayoutPassCache & Cache() { return m_cache; } + + private: + LayoutPassCache m_cache; + LayoutPassCache * m_pPrev; + }; + + // Breaks one text source twice in one layout pass and reports cache hits and misses. + // The engine is built here, not via COM, so it sees the cache this module installs. + void BreakTwiceInOneLayoutPass(const wchar_t * pszText, int * pcHit, int * pcMiss) + { + *pcHit = *pcMiss = 0; +#if defined(WIN32) || defined(_M_X64) + HWND hwndDesktop = ::GetDesktopWindow(); + HDC hdcDesktop = ::GetDC(hwndDesktop); + HDC hdc = ::CreateCompatibleDC(hdcDesktop); + HBITMAP hbm = ::CreateCompatibleBitmap(hdcDesktop, 100, 100); + HGDIOBJ hbmOld = ::SelectObject(hdc, hbm); + + IVwGraphicsWin32Ptr qvg; + qvg.CreateInstance(CLSID_VwGraphicsWin32); + qvg->Initialize(hdc); + + ILgWritingSystemFactoryPtr qwsf; + m_qre->get_WritingSystemFactory(&qwsf); + IRenderEnginePtr qre; + UniscribeEngine::CreateCom(NULL, IID_IRenderEngine, (void **)&qre); + CheckHr(qre->putref_WritingSystemFactory(qwsf)); + CheckHr(qre->putref_RenderEngineFactory(m_qref)); + TxtSrc ts(pszText, qwsf); + IVwTextSourcePtr qts; + ts.QueryInterface(IID_IVwTextSource, (void **)&qts); + int cch; + CheckHr(qts->get_Length(&cch)); + + { + ScopedLayoutPassCache scope; + for (int ibreak = 0; ibreak < 2; ++ibreak) + { + ILgSegmentPtr qseg; + int dichLimSeg; + int dxWidth; + LgEndSegmentType est; + CheckHr(qre->FindBreakPoint(qvg, qts, NULL, 0, cch, cch, TRUE, TRUE, 4000, + klbWordBreak, klbLetterBreak, ktwshAll, FALSE, &qseg, &dichLimSeg, + &dxWidth, &est, NULL)); + } + *pcHit = scope.Cache().AnalysisCache().HitCount(); + *pcMiss = scope.Cache().AnalysisCache().MissCount(); + } + + qts.Clear(); + qre.Clear(); + qvg.Clear(); + ::SelectObject(hdc, hbmOld); + ::DeleteObject(hbm); + ::DeleteDC(hdc); + ::ReleaseDC(hwndDesktop, hdcDesktop); +#endif + } + + void testNfcTextAnalysisIsReusedWithinLayoutPass() + { +#if defined(WIN32) || defined(_M_X64) + int cHit, cMiss; + BreakTwiceInOneLayoutPass(L"cafe deja vu", &cHit, &cMiss); + unitpp::assert_true("text NFC leaves unchanged should be reused on the second break", + cHit > 0); +#endif + } + + void testDecomposedTextAnalysisIsReusedWithinLayoutPass() + { +#if defined(WIN32) || defined(_M_X64) + int cHit, cMiss; + // Decomposed e-acute and a-grave, which NFC composes. + BreakTwiceInOneLayoutPass(L"cafe\u0301 de\u0301ja\u0300 vu", &cHit, &cMiss); + unitpp::assert_true("text NFC rewrites should be reused on the second break", cHit > 0); +#endif + } + virtual void Setup() { RenderEngineTestBase::Setup(); diff --git a/Src/views/Test/TestViewCaches.h b/Src/views/Test/TestViewCaches.h index 2b108c1d8d..a3311e3de5 100644 --- a/Src/views/Test/TestViewCaches.h +++ b/Src/views/Test/TestViewCaches.h @@ -9,9 +9,11 @@ This software is licensed under the LGPL, version 2.1 or later #pragma once #include "testViews.h" +#include "RenderEngineTestBase.h" #include "ColorStateCache.h" #include "FontHandleCache.h" #include "LayoutCache.h" +#include "NfcOffsetMap.h" namespace TestViews { @@ -218,6 +220,195 @@ namespace TestViews public: TestShapeRunCache(); }; + + class TestTextAnalysisCache : public unitpp::suite + { + static IVwTextSource * FakeSource(int n) + { + return reinterpret_cast(static_cast(0x1000 + n)); + } + + void StoreTenChars(TextAnalysisCache & cache, IVwTextSource * pts, int ichMin, int ws) + { + SCRIPT_ITEM rgscri[2]; + ZeroMemory(rgscri, sizeof(rgscri)); + rgscri[1].iCharPos = 10; + cache.Store(pts, ichMin, 10, ws, false, L"abcdefghij", 10, true, rgscri, 1); + } + + void testStoredRangeIsFound() + { + TextAnalysisCache cache; + StoreTenChars(cache, FakeSource(1), 0, 1); + TextAnalysisEntry * pentry = cache.Find(FakeSource(1), 0, 10, 1, false); + unitpp::assert_true("an identical request should hit", pentry != NULL); + unitpp::assert_eq("the hit should carry the stored text length", 10, + pentry->m_vchText.Size()); + unitpp::assert_true("the hit should carry the stored text", + ::memcmp(pentry->m_vchText.Begin(), L"abcdefghij", 10 * isizeof(OLECHAR)) == 0); + unitpp::assert_eq("the hit should carry the stored item count", 1, pentry->m_citem); + } + + void testShorterRequestFromSameStartIsCovered() + { + TextAnalysisCache cache; + StoreTenChars(cache, FakeSource(1), 0, 1); + unitpp::assert_true("a shorter request from the same start should hit", + cache.Find(FakeSource(1), 0, 4, 1, false) != NULL); + unitpp::assert_true("a longer request from the same start should miss", + cache.Find(FakeSource(1), 0, 11, 1, false) == NULL); + } + + void testKeyMismatchMisses() + { + TextAnalysisCache cache; + StoreTenChars(cache, FakeSource(1), 0, 1); + unitpp::assert_true("a different start offset should miss", + cache.Find(FakeSource(1), 1, 4, 1, false) == NULL); + unitpp::assert_true("a different text source should miss", + cache.Find(FakeSource(2), 0, 4, 1, false) == NULL); + unitpp::assert_true("a different writing system should miss", + cache.Find(FakeSource(1), 0, 4, 2, false) == NULL); + unitpp::assert_true("a different direction should miss", + cache.Find(FakeSource(1), 0, 4, 1, true) == NULL); + } + + void testEvictionIsBounded() + { + TextAnalysisCache cache(2); + StoreTenChars(cache, FakeSource(1), 0, 1); + StoreTenChars(cache, FakeSource(2), 0, 1); + StoreTenChars(cache, FakeSource(3), 0, 1); + unitpp::assert_eq("storing past capacity should evict", 1, cache.EvictionCount()); + unitpp::assert_true("the oldest entry should be the one evicted", + cache.Find(FakeSource(1), 0, 10, 1, false) == NULL); + unitpp::assert_true("the newest entry should survive", + cache.Find(FakeSource(3), 0, 10, 1, false) != NULL); + } + + public: + TestTextAnalysisCache(); + }; + + class TestNfcOffsetMap : public unitpp::suite + { + // NFC length of the text in [ichMin, ichLim), normalized from scratch: the definition + // the map must reproduce. + static int OracleNfcLength(const StrUni & stu, int ichMin, int ichLim) + { + StrUni stuRange(stu.Chars() + ichMin, ichLim - ichMin); + StrUtil::NormalizeStrUni(stuRange, UNORM_NFC); + return stuRange.Length(); + } + + void VerifyAgainstOracle(const wchar_t * pszText, const char * pszLabel) + { + StrUni stu(pszText); + int cch = stu.Length(); + TxtSrc ts(pszText, g_qwsf); + IVwTextSourcePtr qts; + ts.QueryInterface(IID_IVwTextSource, (void **)&qts); + NfcOffsetMap map; + map.Reset(qts); + + StrAnsi staMsg; + int cBoundaries = 0; + for (int ichBase = 0; ichBase <= cch; ++ichBase) + { + int ichNfc; + if (!map.IsBoundary(ichBase)) + { + staMsg.Format("%s: base %d is not a boundary so the map must decline", + pszLabel, ichBase); + unitpp::assert_true(staMsg.Chars(), !map.TryOffsetInNfc(cch, ichBase, &ichNfc)); + continue; + } + ++cBoundaries; + for (int ich = ichBase; ich <= cch; ++ich) + { + int cchExpected = OracleNfcLength(stu, ichBase, ich); + staMsg.Format("%s: OffsetInNfc(%d, %d) must answer", pszLabel, ich, ichBase); + unitpp::assert_true(staMsg.Chars(), map.TryOffsetInNfc(ich, ichBase, &ichNfc)); + staMsg.Format("%s: OffsetInNfc(%d, %d)", pszLabel, ich, ichBase); + unitpp::assert_eq(staMsg.Chars(), cchExpected, ichNfc); + } + int cchNfcAll = OracleNfcLength(stu, ichBase, cch); + for (int ichNfcReq = 0; ichNfcReq <= cchNfcAll + 2; ++ichNfcReq) + { + int ichExpected = ichBase; + for (int ich = ichBase; ich <= cch; ++ich) + { + if (OracleNfcLength(stu, ichBase, ich) <= ichNfcReq) + ichExpected = ich; + } + int ichOrig; + staMsg.Format("%s: OffsetToOrig(%d, %d) must answer", pszLabel, ichNfcReq, ichBase); + unitpp::assert_true(staMsg.Chars(), + map.TryOffsetToOrig(ichNfcReq, ichBase, &ichOrig)); + staMsg.Format("%s: OffsetToOrig(%d, %d)", pszLabel, ichNfcReq, ichBase); + unitpp::assert_eq(staMsg.Chars(), ichExpected, ichOrig); + } + } + staMsg.Format("%s: text start and end are always boundaries", pszLabel); + unitpp::assert_true(staMsg.Chars(), cBoundaries >= (cch == 0 ? 1 : 2)); + } + + void testMatchesFromScratchNormalization() + { + VerifyAgainstOracle(L"plain ascii text, no marks", "ascii only"); + VerifyAgainstOracle(L"cafe\u0301 de\u0301ja\u0300 vu no\u0308el", "decomposed latin"); + VerifyAgainstOracle(L"caf\u00E9 d\u00E9j\u00E0 vu", "precomposed latin"); + VerifyAgainstOracle(L"a\u0301\u0327b a\u0327\u0301c", "non-canonical mark order"); + VerifyAgainstOracle(L"\u1e38 L\u0323\u0304 L\u0304\u0323\u0323 x L\u0304\u0323", "marks that reorder and shorten the form"); + VerifyAgainstOracle(L"\u0301\u0300abc", "marks with no base"); + VerifyAgainstOracle(L"\u0628\u064E\u0651 \u0644\u0651\u064E", "arabic harakat"); + VerifyAgainstOracle(L"\u1112\u1161\u11AB \u1100\u1161", "hangul jamo"); + VerifyAgainstOracle(L"x\U0001D400\u0301y\U0001D401", "surrogate pairs"); + VerifyAgainstOracle(L"a\u0344b\u0344", "lengthening mark"); + VerifyAgainstOracle(L"\u212Bngstro\u0308m \u2126", "singleton decomposition"); + VerifyAgainstOracle(L"The \u0628\u064E caf\u00E9 e\u0301\u1112\u1161\u11AB\U0001D400\u0344!", "mixed everything"); + } + + void testCombiningMarkAndLowSurrogateAreNotBoundaries() + { + TxtSrc ts(L"e\u0301\U0001D400", g_qwsf); + IVwTextSourcePtr qts; + ts.QueryInterface(IID_IVwTextSource, (void **)&qts); + NfcOffsetMap map; + map.Reset(qts); + unitpp::assert_true("start is a boundary", map.IsBoundary(0)); + unitpp::assert_true("before a combining mark is not a boundary", !map.IsBoundary(1)); + unitpp::assert_true("before a high surrogate is a boundary", map.IsBoundary(2)); + unitpp::assert_true("between surrogates is not a boundary", !map.IsBoundary(3)); + unitpp::assert_true("end is a boundary", map.IsBoundary(4)); + } + + void testResetRebindsToAnotherSource() + { + TxtSrc ts1(L"ab", g_qwsf); + TxtSrc ts2(L"e\u0301e\u0301", g_qwsf); + IVwTextSourcePtr qts1; + IVwTextSourcePtr qts2; + ts1.QueryInterface(IID_IVwTextSource, (void **)&qts1); + ts2.QueryInterface(IID_IVwTextSource, (void **)&qts2); + NfcOffsetMap map; + map.Reset(qts1); + unitpp::assert_eq("two ascii characters", 2, map.NfcLengthOfPrefix(2)); + map.Reset(qts2); + unitpp::assert_eq("rebinding discards the old table", 2, map.NfcLengthOfPrefix(4)); + } + + public: + TestNfcOffsetMap(); + virtual void Setup() + { + CreateTestWritingSystemFactory(); + } + virtual void Teardown() + { + CloseTestWritingSystemFactory(); + } + }; } #endif // TESTVIEWCACHES_H_INCLUDED diff --git a/Src/views/Test/testViews.mak b/Src/views/Test/testViews.mak index 1b2c2adf54..4835c5ea4b 100644 --- a/Src/views/Test/testViews.mak +++ b/Src/views/Test/testViews.mak @@ -154,6 +154,7 @@ $(VIEWSTEST_SRC)\Collection.cpp: $(VIEWSTEST_SRC)\DummyBaseVc.h $(VIEWSTEST_SRC) $(VIEWSTEST_SRC)\TestTsStrBldr.h\ $(VIEWSTEST_SRC)\TestTsString.h\ $(VIEWSTEST_SRC)\TestTsPropsBldr.h\ - $(VIEWSTEST_SRC)\TestTsTextProps.h + $(VIEWSTEST_SRC)\TestTsTextProps.h\ + $(VIEWSTEST_SRC)\TestViewCaches.h $(DISPLAY) Collecting tests for $(BUILD_PRODUCT).$(BUILD_EXTENSION) $(COLLECT) $** $(VIEWSTEST_SRC)\Collection.cpp diff --git a/Src/views/lib/LayoutCache.h b/Src/views/lib/LayoutCache.h index c1bc0b6f44..112311140d 100644 --- a/Src/views/lib/LayoutCache.h +++ b/Src/views/lib/LayoutCache.h @@ -2,6 +2,10 @@ #ifndef LAYOUTCACHE_INCLUDED #define LAYOUTCACHE_INCLUDED +#include "NfcOffsetMap.h" + +// Itemization of a text range and its NFC form. With m_fTextIsNfc set, m_vchText equals the +// source text; otherwise offsets into it translate through the pass's NfcOffsetMap. class TextAnalysisEntry { public: @@ -12,7 +16,6 @@ class TextAnalysisEntry m_ws(0), m_fWsRtl(false), m_fTextIsNfc(true), - m_cchNfc(0), m_citem(0) { } @@ -22,47 +25,6 @@ class TextAnalysisEntry return m_pts == pts && m_ichMin == ichMin && m_cch >= cch && m_ws == ws && m_fWsRtl == fWsRtl; } - int RequestedNfcLength(int cchRequested) const - { - if (cchRequested <= 0) - return 0; - if (m_fTextIsNfc) - return cchRequested; - if (m_vichOrigToNfc.Size() == 0) - return cchRequested; - if (cchRequested >= m_vichOrigToNfc.Size()) - return m_vichOrigToNfc[m_vichOrigToNfc.Size() - 1]; - return m_vichOrigToNfc[cchRequested]; - } - - int OffsetInNfc(int ich, int ichBase) const - { - Assert(ich >= ichBase); - if (m_fTextIsNfc) - return ich - ichBase; - int ichRelative = ich - ichBase; - if (ichRelative <= 0) - return 0; - if (m_vichOrigToNfc.Size() == 0) - return ichRelative; - if (ichRelative >= m_vichOrigToNfc.Size()) - return m_vichOrigToNfc[m_vichOrigToNfc.Size() - 1]; - return m_vichOrigToNfc[ichRelative]; - } - - int OffsetToOrig(int ich, int ichBase) const - { - if (m_fTextIsNfc) - return ich + ichBase; - if (ich <= 0) - return ichBase; - if (m_vichNfcToOrig.Size() == 0) - return ich + ichBase; - if (ich >= m_vichNfcToOrig.Size()) - return m_cch + ichBase; - return m_vichNfcToOrig[ich] + ichBase; - } - void CopyScriptItemsTo(Vector & vscri, int & citem) const { citem = m_citem; @@ -80,13 +42,13 @@ class TextAnalysisEntry int m_cch; int m_ws; bool m_fWsRtl; + // True when NFC normalization leaves the source text unchanged, so m_vchText equals it and + // offsets into m_vchText are source offsets. bool m_fTextIsNfc; - int m_cchNfc; int m_citem; - Vector m_vchNfc; + // The NFC form of the range. + Vector m_vchText; Vector m_vscri; - Vector m_vichOrigToNfc; - Vector m_vichNfcToOrig; }; class ShapeRunEntry @@ -195,7 +157,7 @@ class TextAnalysisCache TextAnalysisEntry * Store(IVwTextSource * pts, int ichMin, int cch, int ws, bool fWsRtl, const OLECHAR * prgchNfc, int cchNfc, bool fTextIsNfc, const SCRIPT_ITEM * prgscri, - int citem, const Vector * pvichOrigToNfc, const Vector * pvichNfcToOrig) + int citem) { TextAnalysisEntry * pentry = NULL; for (int ientry = 0; ientry < m_ventry.Size(); ++ientry) @@ -231,12 +193,11 @@ class TextAnalysisCache pentry->m_ws = ws; pentry->m_fWsRtl = fWsRtl; pentry->m_fTextIsNfc = fTextIsNfc; - pentry->m_cchNfc = cchNfc; pentry->m_citem = citem; - pentry->m_vchNfc.Resize(cchNfc); + pentry->m_vchText.Resize(cchNfc); if (cchNfc > 0) - ::memcpy(pentry->m_vchNfc.Begin(), prgchNfc, cchNfc * isizeof(OLECHAR)); + ::memcpy(pentry->m_vchText.Begin(), prgchNfc, cchNfc * isizeof(OLECHAR)); int cscri = citem + 1; if (cscri < 2) @@ -245,24 +206,6 @@ class TextAnalysisCache if (cscri > 0) ::memcpy(pentry->m_vscri.Begin(), prgscri, cscri * isizeof(SCRIPT_ITEM)); - if (pvichOrigToNfc) - { - pentry->m_vichOrigToNfc.Resize(pvichOrigToNfc->Size()); - for (int i = 0; i < pvichOrigToNfc->Size(); ++i) - pentry->m_vichOrigToNfc[i] = (*pvichOrigToNfc)[i]; - } - else - pentry->m_vichOrigToNfc.Delete(0, pentry->m_vichOrigToNfc.Size()); - - if (pvichNfcToOrig) - { - pentry->m_vichNfcToOrig.Resize(pvichNfcToOrig->Size()); - for (int i = 0; i < pvichNfcToOrig->Size(); ++i) - pentry->m_vichNfcToOrig[i] = (*pvichNfcToOrig)[i]; - } - else - pentry->m_vichNfcToOrig.Delete(0, pentry->m_vichNfcToOrig.Size()); - return pentry; } @@ -420,6 +363,7 @@ class LayoutPassCache { m_analysisCache.Reset(); m_shapeRunCache.Reset(); + m_nfcOffsets.Clear(); } TextAnalysisCache & AnalysisCache() @@ -427,6 +371,14 @@ class LayoutPassCache return m_analysisCache; } + // The offset map for pts. The pass keeps one map, so asking for a different text source + // than the last call discards the table and the next lookup rebuilds it. + NfcOffsetMap & NfcOffsetsFor(IVwTextSource * pts) + { + m_nfcOffsets.Reset(pts); + return m_nfcOffsets; + } + ShapeRunCache & ShapeCache() { return m_shapeRunCache; @@ -435,6 +387,7 @@ class LayoutPassCache private: TextAnalysisCache m_analysisCache; ShapeRunCache m_shapeRunCache; + NfcOffsetMap m_nfcOffsets; }; extern __declspec(thread) LayoutPassCache * g_pCurrentLayoutPassCache; @@ -457,6 +410,7 @@ inline bool IsPath1ShapeCacheEnabled() return s_nEnabled == 1; } +// Also governs the NfcOffsetMap, which exists to serve the analysis cache's decomposed entries. inline bool IsPath2AnalysisCacheEnabled() { static int s_nEnabled = -1; diff --git a/Src/views/lib/NfcOffsetMap.h b/Src/views/lib/NfcOffsetMap.h new file mode 100644 index 0000000000..32eeab7330 --- /dev/null +++ b/Src/views/lib/NfcOffsetMap.h @@ -0,0 +1,208 @@ +/*--------------------------------------------------------------------*//*:Ignore this sentence. +Copyright (c) 2026 SIL International +This software is licensed under the LGPL, version 2.1 or later +(http://www.gnu.org/licenses/lgpl-2.1.html) + +File: NfcOffsetMap.h +Responsibility: +Last reviewed: Not yet. + +Description: + Translates offsets between a text source and its NFC-normalized form without + renormalizing a prefix of the text on every request (LT-22674). +-------------------------------------------------------------------------------*//*:End Ignore*/ +#pragma once +#ifndef NFCOFFSETMAP_INCLUDED +#define NFCOFFSETMAP_INCLUDED + +/*---------------------------------------------------------------------------------------------- + Offset translation for one text source between its own (typically NFD) offsets and offsets + into its NFC-normalized form. + + An offset into the NFC form of a range is defined as the NFC length of that range, exactly + as normalizing the range from scratch would give. The map reproduces that definition while + normalizing each character at most once: it splits the text where ICU says normalization + cannot cross a boundary, records the NFC length of the text before each boundary, and + normalizes only the tail from the nearest boundary when a request lands inside a chunk. + Requests whose base offset is not a boundary cannot be answered from the table and report + failure, so the caller can fall back to normalizing the range directly. + + The map is bound to one text source whose content does not change while the map is in + use; Reset rebinds it. +----------------------------------------------------------------------------------------------*/ +class NfcOffsetMap +{ +public: + NfcOffsetMap() : m_pts(NULL), m_cchText(0), m_fBuilt(false) + { + } + + // Binds the map to a text source, discarding any table built for a different one. + void Reset(IVwTextSource * pts) + { + if (pts == m_pts && m_fBuilt) + return; + Clear(); + m_pts = pts; + } + + // Unbinds the map and discards its table. + void Clear() + { + m_pts = NULL; + m_cchText = 0; + m_fBuilt = false; + m_vchText.Clear(); + m_vichBoundary.Clear(); + m_vcchNfcBefore.Clear(); + } + + // True when normalization of the text before ich is independent of the text from ich on. + // The start and end of the text are boundaries; the middle of a surrogate pair is not. + bool IsBoundary(int ich) + { + if (!Build() || ich < 0 || ich > m_cchText) + return false; + if (ich == 0 || ich == m_cchText) + return true; + return BoundaryIndexAt(ich) >= 0; + } + + // NFC length of the text from ichBase to ich. Fails unless ichBase is a boundary. + bool TryOffsetInNfc(int ich, int ichBase, int * pichNfc) + { + *pichNfc = 0; + if (!Build() || ichBase < 0 || ich < ichBase || ich > m_cchText || !IsBoundary(ichBase)) + return false; + *pichNfc = NfcLengthOfPrefix(ich) - NfcLengthOfPrefix(ichBase); + return true; + } + + // The longest range starting at ichBase whose NFC form is at most ichNfc characters long, + // reported as the offset of its end. Fails unless ichBase is a boundary. + bool TryOffsetToOrig(int ichNfc, int ichBase, int * pichOrig) + { + *pichOrig = ichBase; + if (!Build() || ichBase < 0 || ichBase > m_cchText || ichNfc < 0 || !IsBoundary(ichBase)) + return false; + int cchNfcBase = NfcLengthOfPrefix(ichBase); + // Find the last boundary at or after ichBase whose prefix length still fits, then + // walk forward one character at a time inside that chunk. + int iLow = BoundaryIndexAtOrBefore(ichBase); + int iHigh = m_vichBoundary.Size() - 1; + while (iLow < iHigh) + { + int iMid = iLow + (iHigh - iLow + 1) / 2; + if (m_vcchNfcBefore[iMid] - cchNfcBase <= ichNfc) + iLow = iMid; + else + iHigh = iMid - 1; + } + // Prefix lengths need not grow inside a chunk: a mark that reorders ahead of another can + // compose with the base and shorten the form, so the walk keeps the last offset that + // fits. + int ichFit = max(m_vichBoundary[iLow], ichBase); + int ichChunkLim = iLow + 1 < m_vichBoundary.Size() ? m_vichBoundary[iLow + 1] : m_cchText; + for (int ich = ichFit; ich < ichChunkLim; ++ich) + { + if (NfcLengthOfPrefix(ich + 1) - cchNfcBase <= ichNfc) + ichFit = ich + 1; + } + *pichOrig = ichFit; + return true; + } + + // NFC length of the text before ich, computed from the nearest boundary at or before it. + int NfcLengthOfPrefix(int ich) + { + if (!Build() || ich <= 0) + return 0; + Assert(ich <= m_cchText); + int iBoundary = BoundaryIndexAtOrBefore(ich); + int ichBoundary = m_vichBoundary[iBoundary]; + int cchNfc = m_vcchNfcBefore[iBoundary]; + if (ich > ichBoundary) + cchNfc += NfcLength(ichBoundary, ich); + return cchNfc; + } + +private: + // Fetches the whole text and records every normalization boundary with the NFC length + // of the text before it. Chunks between boundaries are normalized independently. + bool Build() + { + if (m_fBuilt) + return true; + if (!m_pts) + return false; + CheckHr(m_pts->get_Length(&m_cchText)); + m_vchText.Resize(m_cchText); + if (m_cchText > 0) + CheckHr(m_pts->Fetch(0, m_cchText, m_vchText.Begin())); + + const icu::Normalizer2 * pnorm = SilUtil::GetIcuNormalizer(UNORM_NFC); + m_vichBoundary.Push(0); + m_vcchNfcBefore.Push(0); + int ichChunk = 0; + int cchNfc = 0; + int ich = 0; + while (ich < m_cchText) + { + int ichNext = ich; + UChar32 ch; + U16_NEXT(m_vchText.Begin(), ichNext, m_cchText, ch); + if (ich > 0 && pnorm->hasBoundaryBefore(ch)) + { + cchNfc += NfcLength(ichChunk, ich); + m_vichBoundary.Push(ich); + m_vcchNfcBefore.Push(cchNfc); + ichChunk = ich; + } + ich = ichNext; + } + m_fBuilt = true; + return true; + } + + // NFC length of the text in [ichMin, ichLim). + int NfcLength(int ichMin, int ichLim) + { + if (ichLim <= ichMin) + return 0; + StrUni stu(m_vchText.Begin() + ichMin, ichLim - ichMin); + StrUtil::NormalizeStrUni(stu, UNORM_NFC); + return stu.Length(); + } + + // Index of the last recorded boundary at or before ich; the table always holds 0. + int BoundaryIndexAtOrBefore(int ich) + { + int iLow = 0; + int iHigh = m_vichBoundary.Size() - 1; + while (iLow < iHigh) + { + int iMid = iLow + (iHigh - iLow + 1) / 2; + if (m_vichBoundary[iMid] <= ich) + iLow = iMid; + else + iHigh = iMid - 1; + } + return iLow; + } + + // Index of the boundary recorded exactly at ich, or -1. + int BoundaryIndexAt(int ich) + { + int i = BoundaryIndexAtOrBefore(ich); + return m_vichBoundary[i] == ich ? i : -1; + } + + IVwTextSource * m_pts; + int m_cchText; + bool m_fBuilt; + Vector m_vchText; + Vector m_vichBoundary; + Vector m_vcchNfcBefore; +}; + +#endif // NFCOFFSETMAP_INCLUDED diff --git a/Src/views/lib/UniscribeEngine.cpp b/Src/views/lib/UniscribeEngine.cpp index 570217645b..03b6077374 100644 --- a/Src/views/lib/UniscribeEngine.cpp +++ b/Src/views/lib/UniscribeEngine.cpp @@ -414,9 +414,8 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( // clear. // PATH-N1: Get NFC flag to avoid redundant OffsetInNfc/OffsetToOrig normalization below. bool fTextIsNfc = false; - const TextAnalysisEntry * pAnalysis = NULL; int cchNfc = UniscribeSegment::CallScriptItemize(rgchBuf, INIT_BUF_SIZE, vch, pts, ichMinSeg, - ichLimText - ichMinSeg, &prgchBuf, citem, (bool)fParaRtoL, &fTextIsNfc, &pAnalysis); + ichLimText - ichMinSeg, &prgchBuf, citem, (bool)fParaRtoL, &fTextIsNfc); Vector vichBreak; ILgLineBreakerPtr qlb; @@ -514,7 +513,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( } ichLim = min(ichLimNext, ichLimBT2); // Optimize JohnT: if ichLim==ichBase+m_dichLim, can use cchNfc. - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); if (ichLimNfc == ichMinNfc) { // This can happen if later characters in a composite have different properties than the first. @@ -538,7 +537,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( { // Script item is smaller than run; shorten the amount we treat as a 'run'. ichLimNfc = (pscri + 1)->iCharPos; - ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc); } // Set up the characters of the run, if any. @@ -765,7 +764,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( vdxRun.Pop(); ichLimBT2 = ichMin; ichLim = *(vichRun.Top()); - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); vichRun.Pop(); cglyph = *(viglyphRun.Top()); viglyphRun.Pop(); @@ -870,7 +869,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( vdxRun.Pop(); ichLimBT2 = ichMin; ichLim = *(vichRun.Top()); - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); vichRun.Pop(); cglyph = *(viglyphRun.Top()); viglyphRun.Pop(); @@ -984,11 +983,11 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( } } } - ichLim = UniscribeSegment::OffsetToOrig(ichMinUri + ichRun, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLim = UniscribeSegment::OffsetToOrig(ichMinUri + ichRun, ichMinSeg, pts, fTextIsNfc); break; } - int ichLineBreak = UniscribeSegment::OffsetToOrig(ichLineBreakNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis); + int ichLineBreak = UniscribeSegment::OffsetToOrig(ichLineBreakNfc, ichMinSeg, pts, fTextIsNfc); if (ichLineBreak <= ichMin) { @@ -1005,7 +1004,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( viglyphRun.Pop(); } ichLim = ichMin; // Required to get correct values at start of loop. - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); ichLimBT2 = ichLineBreak; fRemovedWs = false; fBacktracking = true; @@ -1013,12 +1012,12 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( } ichLim = ichLineBreak; // We limit the segment to not exceed the latest line break point. Assert(ichLim <= ichLimBacktrack); - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); fOkBreak = true; // Means we have a good line break. // Store the glyph-specific information: stretch values. int cchRunTotalTmp = uri.cch; - uri.cch = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis) - ichMinNfc; + uri.cch = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc) - ichMinNfc; UniscribeSegment::ShapePlaceRun(uri, true); viglyphRun.Push(cglyph); cglyph += uri.cglyph; @@ -1044,7 +1043,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( if (twsh == ktwshNoWs) { fRemovedWs = RemoveTrailingWhiteSpace(ichMinUri, &ichLimNfc, uri); - ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc); // Usually the worst case is that ichLimNfc == ichMinUri, indicating that the whole run is // white space. However, in at least one pathological case, we have observed uniscribe // strip of more than one run of white space. Hence the <=. @@ -1068,7 +1067,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( dxSegWidth = *(vdxRun.Top()); vdxRun.Pop(); ichLim = *(vichRun.Top()); - ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLimNfc = UniscribeSegment::OffsetInNfc(ichLim, ichMinSeg, pts, fTextIsNfc); vichRun.Pop(); cglyph = *(viglyphRun.Top()); viglyphRun.Pop(); @@ -1097,7 +1096,7 @@ STDMETHODIMP UniscribeEngine::FindBreakPoint( Assert(irun == 1); Assert(ichMinUri == 0); RemoveNonWhiteSpace(ichMinUri, &ichLimNfc, uri); - ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc, pAnalysis); + ichLim = UniscribeSegment::OffsetToOrig(ichLimNfc, ichMinSeg, pts, fTextIsNfc); if (ichLim == ichMinSeg) return S_OK; // failure to create a valid segment fOkBreak = true; diff --git a/Src/views/lib/UniscribeSegment.cpp b/Src/views/lib/UniscribeSegment.cpp index f40bd24bfc..17687b0625 100644 --- a/Src/views/lib/UniscribeSegment.cpp +++ b/Src/views/lib/UniscribeSegment.cpp @@ -29,8 +29,6 @@ DEFINE_THIS_FILE //:>******************************************************************************************** //:> Forward declarations //:>******************************************************************************************** -static void BuildNfcOffsetMaps(const StrUni & stuOrig, Vector & vichOrigToNfc, - Vector & vichNfcToOrig); static void ApplyShapeRunCacheEntry(ShapeRunEntry & entry, UniscribeRunInfo & uri); //:>******************************************************************************************** @@ -1631,18 +1629,21 @@ int UniscribeSegment::OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, boo Assert(ich >= ichBase); if (fTextIsNfc) return ich - ichBase; + int ichNfc; + NfcOffsetMap * pmap = CurrentNfcOffsetMap(pts); + if (pmap && pmap->TryOffsetInNfc(ich, ichBase, &ichNfc)) + return ichNfc; return OffsetInNfc(ich, ichBase, pts); } -int UniscribeSegment::OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc, - const TextAnalysisEntry * pAnalysis) +// The layout pass's offset map for pts, or NULL outside a layout pass. FW_PERF_P125_PATH2=0 +// turns the map off together with the analysis cache it serves. +NfcOffsetMap * UniscribeSegment::CurrentNfcOffsetMap(IVwTextSource * pts) { - Assert(ich >= ichBase); - if (fTextIsNfc) - return ich - ichBase; - if (pAnalysis) - return pAnalysis->OffsetInNfc(ich, ichBase); - return OffsetInNfc(ich, ichBase, pts, fTextIsNfc); + if (!IsPath2AnalysisCacheEnabled()) + return NULL; + LayoutPassCache * pLayoutPassCache = GetCurrentLayoutPassCache(); + return pLayoutPassCache ? &pLayoutPassCache->NfcOffsetsFor(pts) : NULL; } // ich is an offset into the (NFC normalized) characters of this segment. @@ -1689,43 +1690,13 @@ int UniscribeSegment::OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bo { if (fTextIsNfc) return ich + ichBase; + int ichOrig; + NfcOffsetMap * pmap = CurrentNfcOffsetMap(pts); + if (pmap && pmap->TryOffsetToOrig(ich, ichBase, &ichOrig)) + return ichOrig; return OffsetToOrig(ich, ichBase, pts); } -int UniscribeSegment::OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc, - const TextAnalysisEntry * pAnalysis) -{ - if (fTextIsNfc) - return ich + ichBase; - if (pAnalysis) - return pAnalysis->OffsetToOrig(ich, ichBase); - return OffsetToOrig(ich, ichBase, pts, fTextIsNfc); -} - -static void BuildNfcOffsetMaps(const StrUni & stuOrig, Vector & vichOrigToNfc, - Vector & vichNfcToOrig) -{ - int cchOrig = stuOrig.Length(); - vichOrigToNfc.Resize(cchOrig + 1); - vichOrigToNfc[0] = 0; - for (int ich = 1; ich <= cchOrig; ++ich) - { - StrUni stuPrefix(stuOrig.Chars(), ich); - StrUtil::NormalizeStrUni(stuPrefix, UNORM_NFC); - vichOrigToNfc[ich] = stuPrefix.Length(); - } - - int cchNfc = vichOrigToNfc[cchOrig]; - vichNfcToOrig.Resize(cchNfc + 1); - int ichOrig = 0; - for (int ichNfc = 0; ichNfc <= cchNfc; ++ichNfc) - { - while (ichOrig + 1 <= cchOrig && vichOrigToNfc[ichOrig + 1] <= ichNfc) - ++ichOrig; - vichNfcToOrig[ichNfc] = ichOrig; - } -} - static void ApplyShapeRunCacheEntry(ShapeRunEntry & entry, UniscribeRunInfo & uri) { if (uri.CGlyphMax() < entry.m_cglyph) @@ -3092,11 +3063,9 @@ void UniscribeSegment::AdjustEndForWidth(int ichBase, IVwGraphics * pvg) ----------------------------------------------------------------------------------------------*/ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, Vector & vch, IVwTextSource * pts, int ichMin, int cch, OLECHAR ** pprgchBuf, - int & citem, bool fParaRTL, bool * pfTextIsNfc, const TextAnalysisEntry ** ppAnalysis) + int & citem, bool fParaRTL, bool * pfTextIsNfc) { * pprgchBuf = prgchDefBuf; // Use on-stack variable if big enough - if (ppAnalysis) - *ppAnalysis = NULL; bool fNeedOpenTypeScriptTags = cch > 0 && TextRangeHasOpenTypeFeatures(pts, ichMin, cch); LayoutPassCache * pLayoutPassCache = (!fNeedOpenTypeScriptTags && IsPath2AnalysisCacheEnabled()) ? @@ -3114,27 +3083,41 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, ws = chrp.ws; fWsRtl = chrp.fWsRtl; pCachedAnalysis = pLayoutPassCache->AnalysisCache().Find(pts, ichMin, cchOrig, ws, fWsRtl); + int cchNfcHit = cchOrig; + if (pCachedAnalysis && !pCachedAnalysis->m_fTextIsNfc) + { + // A shorter request is served from a longer entry only when its end is a + // normalization boundary; otherwise the entry's NFC form is not a prefix of the + // request's own NFC form. + NfcOffsetMap & map = pLayoutPassCache->NfcOffsetsFor(pts); + if (cchOrig == pCachedAnalysis->m_cch) + cchNfcHit = pCachedAnalysis->m_vchText.Size(); + else if (!map.IsBoundary(ichMin + cchOrig) || + !map.TryOffsetInNfc(ichMin + cchOrig, ichMin, &cchNfcHit)) + { + pCachedAnalysis = NULL; + } + } if (pCachedAnalysis) { if (pfTextIsNfc) *pfTextIsNfc = pCachedAnalysis->m_fTextIsNfc; - *pprgchBuf = pCachedAnalysis->m_vchNfc.Size() > 0 ? pCachedAnalysis->m_vchNfc.Begin() : prgchDefBuf; + *pprgchBuf = pCachedAnalysis->m_vchText.Size() > 0 ? pCachedAnalysis->m_vchText.Begin() : prgchDefBuf; pCachedAnalysis->CopyScriptItemsTo(g_vscri, citem); if (g_votScriptTags.Size() < citem) g_votScriptTags.Resize(citem); for (int itag = 0; itag < citem; ++itag) g_votScriptTags[itag] = 0; g_cscri = citem; - if (ppAnalysis) - *ppAnalysis = pCachedAnalysis; - return pCachedAnalysis->RequestedNfcLength(cchOrig); + return cchNfcHit; } } if (pLayoutPassCache && !pCachedAnalysis) dwStartMs = ::GetTickCount(); - Vector vichOrigToNfc; - Vector vichNfcToOrig; + // True when NFC normalization leaves the fetched text unchanged, so offsets into the NFC + // buffer are source offsets. + bool fTextIsNfc = true; #ifdef UNISCRIBE_NFC if (cch) @@ -3154,8 +3137,7 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, bool fComputedTextIsNfc = (stu == stuOrig); if (pfTextIsNfc) *pfTextIsNfc = fComputedTextIsNfc; - if (!fComputedTextIsNfc && pLayoutPassCache) - BuildNfcOffsetMaps(stuOrig, vichOrigToNfc, vichNfcToOrig); + fTextIsNfc = fComputedTextIsNfc; if (cch > cchBuf) { cchBuf = cch; @@ -3168,13 +3150,6 @@ int UniscribeSegment::CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, { if (pfTextIsNfc) *pfTextIsNfc = true; // Empty text is trivially NFC - if (pLayoutPassCache) - { - vichOrigToNfc.Resize(1); - vichOrigToNfc[0] = 0; - vichNfcToOrig.Resize(1); - vichNfcToOrig[0] = 0; - } } #else if (pfTextIsNfc) @@ -3317,12 +3292,8 @@ typedef struct tag_SCRIPT_STATE { if (pLayoutPassCache) { pLayoutPassCache->AnalysisCache().AddComputeMs(::GetTickCount() - dwStartMs); - TextAnalysisEntry * pStoredAnalysis = pLayoutPassCache->AnalysisCache().Store(pts, ichMin, - cchOrig, ws, fWsRtl, *pprgchBuf, cch, pfTextIsNfc ? *pfTextIsNfc : true, - g_vscri.Begin(), citem, vichOrigToNfc.Size() ? &vichOrigToNfc : NULL, - vichNfcToOrig.Size() ? &vichNfcToOrig : NULL); - if (ppAnalysis) - *ppAnalysis = pStoredAnalysis; + pLayoutPassCache->AnalysisCache().Store(pts, ichMin, cchOrig, ws, fWsRtl, *pprgchBuf, cch, + fTextIsNfc, g_vscri.Begin(), citem); } return cch; } @@ -3391,9 +3362,8 @@ template int UniscribeSegment::DoAllRuns(int ichBase, IVwGraphics * pv OLECHAR * prgchBuf; // Where text actually goes. // PATH-N1: Get NFC flag from CallScriptItemize to skip redundant OffsetInNfc calls below. bool fTextIsNfc = false; - const TextAnalysisEntry * pAnalysis = NULL; int cchNfc = CallScriptItemize(rgchBuf, INIT_BUF_SIZE, vch, m_qts, ichBase, m_dichLim, &prgchBuf, - citem, m_fParaRTL, &fTextIsNfc, &pAnalysis); + citem, m_fParaRTL, &fTextIsNfc); // If dxdExpectedWidth is not 0, then the segment will try its best to stretch to the // specified size. @@ -3453,7 +3423,7 @@ template int UniscribeSegment::DoAllRuns(int ichBase, IVwGraphics * pv if (ichLim - ichBase > m_dichLim) ichLim = ichBase + m_dichLim; - ichLimNfc = OffsetInNfc(ichLim, ichBase, m_qts, fTextIsNfc, pAnalysis); + ichLimNfc = OffsetInNfc(ichLim, ichBase, m_qts, fTextIsNfc); if (ichLimNfc == ichMinNfc && m_dichLim > 0) { // This can happen pathologically where later characters in a composition have different diff --git a/Src/views/lib/UniscribeSegment.h b/Src/views/lib/UniscribeSegment.h index a62b44c3f1..c9f4b72fdf 100644 --- a/Src/views/lib/UniscribeSegment.h +++ b/Src/views/lib/UniscribeSegment.h @@ -30,7 +30,7 @@ typedef Vector ScrItemVec; // Hungarian vscri; typedef Vector ScrLogAttrVec; // Hungarian vsla. typedef Vector OpenTypeTagVec; // Hungarian vot. -class TextAnalysisEntry; +class NfcOffsetMap; /*---------------------------------------------------------------------------------------------- Class: UniscribeRunInfo @@ -237,12 +237,9 @@ class UniscribeSegment : public ILgSegment static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts); static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc); - static int OffsetInNfc(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc, - const TextAnalysisEntry * pAnalysis); + static NfcOffsetMap * CurrentNfcOffsetMap(IVwTextSource * pts); static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts); static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc); - static int OffsetToOrig(int ich, int ichBase, IVwTextSource * pts, bool fTextIsNfc, - const TextAnalysisEntry * pAnalysis); protected: // Static variables @@ -310,7 +307,7 @@ class UniscribeSegment : public ILgSegment static void ShapePlaceRun(UniscribeRunInfo& uri, bool fCreatingSeg = false); static int CallScriptItemize(OLECHAR * prgchDefBuf, int cchBuf, Vector & vch, IVwTextSource * pts, int ichMin, int cch, OLECHAR ** pprgchBuf, int & citem, - bool fParaRTL, bool * pfTextIsNfc = NULL, const TextAnalysisEntry ** ppAnalysis = NULL); + bool fParaRTL, bool * pfTextIsNfc = NULL); int NumStretchableGlyphs(); int StretchGlyphs(UniscribeRunInfo & uri,