From 685507922b0b75da5935076395a5b1ec1ef58356 Mon Sep 17 00:00:00 2001 From: Robin Haberkorn Date: Sat, 25 Jul 2026 01:55:28 +0200 Subject: revised and improved the Unicode glyph-to-byte conversion heuristics Previously almost all glyph-to-byte offset conversions consulted Scintilla's line index and counted characters on the resulting line. For instance a simple expression like `.+1J` would scan the same line twice completely, which would be very slow on pathologically long lines. Even insertions did that due to having to update the ^Y ranges. If you repeat such an operation over all characters as in `<.+1:J;>` you would have complexity O(n^2) for n = line length. Only commands with an explicit relative nature like `C` and `A` would use teco_view_glyph2bytes_relative() which scans beginning at dot as long as the relative movement is less than 1024 glyphs. Wit the new heuristics almost all glyph-to-byte and byte-to-glyph conversions can make use of that optimization. This requires that dot must at all times be known in glyphs as well - the byte position is managed by Scintilla (SCI_GETCURRENTPOS). We therefore introduced teco_current_doc_set_dot() and teco_current_doc_get_dot() to update dot in the current buffer or Q-Register -- it cannot be stored along with the view since Q-Registers share a single view. A number of auxiliary functions have been introduced for converting relative to a known (glyphs,bytes) offset pair and for converting absolute and relative positions with regard to the current doc and SCI_GETCURRENTPOS position. Of course this is error-prone since the glyph and dot positions are interdependant - they must always be kept in sync. With these new optimizations even pathologically long lines can (usually) be managed even in UTF-8 documents. It does not address slow-downs in Scintilla's line layout, yet. grosciteco.tes for instance runs twice as fast now. --- tests/testsuite.at | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) (limited to 'tests') diff --git a/tests/testsuite.at b/tests/testsuite.at index a1a563d..cbb12d4 100644 --- a/tests/testsuite.at +++ b/tests/testsuite.at @@ -394,6 +394,11 @@ TE_CHECK([[0EE 255@I/A/J 65001EE 0A-(-2)"N(0/0)' 1A-^^A"N(0/0)' 2A-(-1)"N(0/0)'] TE_CHECK([[@EQa// 0EE 128@I/A/J 65001EE 0Qa-(-2)"N(0/0)' 1Qa-^^A"N(0/0)' 2Qa-(-1)"N(0/0)']], 0, ignore, ignore) AT_CLEANUP +AT_SETUP([Glyphs to byte heuristics]) +# Should be fast. +TE_CHECK([[100000<@I"Ж"> J<:C;> J<.-1:J;> .*2-^E"N(0/0)']], 0, ignore, ignore) +AT_CLEANUP + AT_SETUP([Automatic EOL normalization]) TE_CHECK([[@EB'^EQ[$srcdir]/autoeol-input.txt' EL-2"N(0/0)' 2LR 13@I'' 0EL @EW'autoeol-sciteco.txt']], 0, ignore, ignore) @@ -542,9 +547,7 @@ AT_CLEANUP # It could fail because the memory limit is exceeed, # but not in this case since the match string isn't too large. AT_SETUP([Pattern matching overflow]) -# NOTE: Creating very long lines would currently be ineffective -# at least in UTF-8 mode. -TE_CHECK([[100000<@I"^J">J @S"^EM^X"]], 0, ignore, ignore) +TE_CHECK([[100000<@I"X">J @S"^EM^X"]], 0, ignore, ignore) AT_CLEANUP AT_SETUP([Block-wise backwards search]) -- cgit v1.2.3