From d19361ac855df3414fbf57fc0c4eda0afefa1526 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?J=C3=B6rg=20Breitbart?= Date: Sun, 4 Dec 2022 19:53:36 +0100 Subject: [PATCH] initial rewrite of addon mechs --- .../src/WebLinkProvider.ts | 129 ++++++++++++------ .../src/WebLinksAddon.ts | 23 +--- bin/test_weblinks.sh | 22 +++ 3 files changed, 111 insertions(+), 63 deletions(-) create mode 100755 bin/test_weblinks.sh diff --git a/addons/xterm-addon-web-links/src/WebLinkProvider.ts b/addons/xterm-addon-web-links/src/WebLinkProvider.ts index fafbb614..501eba99 100644 --- a/addons/xterm-addon-web-links/src/WebLinkProvider.ts +++ b/addons/xterm-addon-web-links/src/WebLinkProvider.ts @@ -45,7 +45,10 @@ export class LinkComputer { public static computeLink(y: number, regex: RegExp, terminal: Terminal, activate: (event: MouseEvent, uri: string) => void): ILink[] { const rex = new RegExp(regex.source, (regex.flags || '') + 'g'); - const [line, startLineIndex] = LinkComputer._translateBufferLineToStringWithWrap(y - 1, false, terminal); + const [lines, startLineIndex] = LinkComputer._getFullLineString(y - 1, terminal); + + // TODO: do locally limited search if string is too long + const line = lines.join(''); // Don't try if the wrapped line if excessively large as the regex matching will block the main // thread. @@ -58,7 +61,7 @@ export class LinkComputer { const result: ILink[] = []; while ((match = rex.exec(line)) !== null) { - const text = match[1]; + const text = match[0]; if (!text) { // something matched but does not comply with the given matchIndex // since this is most likely a bug the regex itself we simply do nothing here @@ -77,28 +80,39 @@ export class LinkComputer { break; } - let endX = stringIndex + text.length; - let endY = startLineIndex + 1; - - while (endX > terminal.cols) { - endX -= terminal.cols; - endY++; + // check via URL if the matched text would form a proper url + // NOTE: This outsources the ugly url parsing to the browser. + // To avoid surprising auto expansion from URL we additionally + // check afterwards if the provided string resembles the parsed + // one close enough: + // - decodeURI decode path segement back to byte repr + // to detect unicode auto conversion correctly + // - append / also match domain urls w'o any path notion + try { + const url = new URL(text); + const urlText = decodeURI(url.toString()); + if (text !== urlText && text + '/' !== urlText) { + continue; + } + } catch (e) { + continue; } - let startX = stringIndex + 1; - let startY = startLineIndex + 1; - while (startX > terminal.cols) { - startX -= terminal.cols; - startY++; + + const [startY, startX] = LinkComputer._mapStringIndexToBuffer(startLineIndex, stringIndex, terminal); + const [endY, endX] = LinkComputer._mapStringIndexToBuffer(startLineIndex, stringIndex + text.length, terminal); + + if (startY === -1 || startX === -1 || endY === -1 || endX === -1) { + continue; } const range = { start: { - x: startX, - y: startY + x: startX + 1, + y: startY + 1 }, end: { - x: endX, + x: endX + 1, y: endY } }; @@ -112,39 +126,66 @@ export class LinkComputer { /** * Gets the entire line for the buffer line * @param lineIndex The index of the line being translated. - * @param trimRight Whether to trim whitespace to the right. */ - private static _translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean, terminal: Terminal): [string, number] { - let lineString = ''; - let lineWrapsToNext: boolean; - let prevLinesToWrap: boolean; + private static _getFullLineString(lineIndex: number, terminal: Terminal): [string[], number] { + let line: any; - do { - const line = terminal.buffer.active.getLine(lineIndex); + // expand top + let topIdx = lineIndex; + while ((line = terminal.buffer.active.getLine(topIdx)) && line.isWrapped) { + topIdx--; + } + + // expand bottom + let bottomIdx = lineIndex + 1; + while ((line = terminal.buffer.active.getLine(bottomIdx)) && line.isWrapped) { + bottomIdx++; + } + + const lines: string[] = []; + for (let idx = topIdx; idx < bottomIdx; ++idx) { + lines.push(terminal.buffer.active.getLine(idx)?.translateToString(true)!); + } + + return [lines, topIdx]; + } + + /** + * Map a string index back to buffer positions. + * Returns buffer position as [lineIndex, columnIndex] 0-based, + * or [-1, -1] in case the lookup ran into a non-existing line. + */ + private static _mapStringIndexToBuffer(lineIndex: number, stringIndex: number, terminal: Terminal): [number, number] { + const buf = terminal.buffer.active; + const cell = buf.getNullCell(); + while (stringIndex) { + const line = buf.getLine(lineIndex); if (!line) { - break; + return [-1, -1]; } - - if (line.isWrapped) { - lineIndex--; + for (let i = 0; i < line.length; ++i) { + line.getCell(i, cell); + const chars = cell.getChars(); + const width = cell.getWidth(); + if (width) { + stringIndex -= chars.length || 1; + } + // look ahead for early wrap around of wide chars + if (i === line.length - 1 && chars === '' && width) { + const line = buf.getLine(lineIndex + 1); + if (line && line.isWrapped) { + line.getCell(0, cell); + if (cell.getWidth() === 2) { + stringIndex += 1; + } + } + } + if (stringIndex < 0) { + return [lineIndex, i]; + } } - - prevLinesToWrap = line.isWrapped; - } while (prevLinesToWrap); - - const startLineIndex = lineIndex; - - do { - const nextLine = terminal.buffer.active.getLine(lineIndex + 1); - lineWrapsToNext = nextLine ? nextLine.isWrapped : false; - const line = terminal.buffer.active.getLine(lineIndex); - if (!line) { - break; - } - lineString += line.translateToString(!lineWrapsToNext && trimRight).substring(0, terminal.cols); lineIndex++; - } while (lineWrapsToNext); - - return [lineString, startLineIndex]; + } + return [lineIndex, 0]; } } diff --git a/addons/xterm-addon-web-links/src/WebLinksAddon.ts b/addons/xterm-addon-web-links/src/WebLinksAddon.ts index 1e3c877d..c71753ff 100644 --- a/addons/xterm-addon-web-links/src/WebLinksAddon.ts +++ b/addons/xterm-addon-web-links/src/WebLinksAddon.ts @@ -6,25 +6,10 @@ import { Terminal, ITerminalAddon, IDisposable } from 'xterm'; import { ILinkProviderOptions, WebLinkProvider } from './WebLinkProvider'; -const protocolClause = '(https?:\\/\\/)'; -const domainCharacterSet = '[\\da-z\\.-]+'; -const negatedDomainCharacterSet = '[^\\da-z\\.-]+'; -const domainBodyClause = '(' + domainCharacterSet + ')'; -const tldClause = '([a-z\\.]{2,18})'; -const ipClause = '((\\d{1,3}\\.){3}\\d{1,3})'; -const localHostClause = '(localhost)'; -const portClause = '(:\\d{1,5})'; -const hostClause = '((' + domainBodyClause + '\\.' + tldClause + ')|' + ipClause + '|' + localHostClause + ')' + portClause + '?'; -const pathCharacterSet = '(\\/[\\/\\w\\.\\-%~:+@]*)*([^:"\'\\s])'; -const pathClause = '(' + pathCharacterSet + ')?'; -const queryStringHashFragmentCharacterSet = '[0-9\\w\\[\\]\\(\\)\\/\\?\\!#@$%&\'*+,:;~\\=\\.\\-]*'; -const queryStringClause = '(\\?' + queryStringHashFragmentCharacterSet + ')?'; -const hashFragmentClause = '(#' + queryStringHashFragmentCharacterSet + ')?'; -const negatedPathCharacterSet = '[^\\/\\w\\.\\-%]+'; -const bodyClause = hostClause + pathClause + queryStringClause + hashFragmentClause; -const start = '(?:^|' + negatedDomainCharacterSet + ')('; -const end = ')($|' + negatedPathCharacterSet + ')'; -const strictUrlRegex = new RegExp(start + protocolClause + bodyClause + end); +// consider everthing starting with http:// or https:// +// up to first whitespace as url +// gets further narrowed down with URL later on +const strictUrlRegex = /https?:[/]{2}\S*/; function handleLink(event: MouseEvent, uri: string): void { const newWindow = window.open(); diff --git a/bin/test_weblinks.sh b/bin/test_weblinks.sh new file mode 100755 index 00000000..a50db788 --- /dev/null +++ b/bin/test_weblinks.sh @@ -0,0 +1,22 @@ +#!/bin/bash + +# all half width - only good case +echo "aaa http://example.com aaa http://example.com aaa" + +# full width before - wrong offset +echo "¥¥¥ http://example.com aaa http://example.com aaa" + +# full width between - wrong offset +echo "aaa http://example.com ¥¥¥ http://example.com aaa" + +# full width before and between - error in offsets adding up +echo "¥¥¥ http://example.com ¥¥¥ http://example.com aaa" + +# full width within url - partial wrong match +echo "aaa https://ko.wikipedia.org/wiki/위키백과:대문 aaa https://ko.wikipedia.org/wiki/위키백과:대문" + +# full width within and before - partial wrong match + wrong offsets +echo "¥¥¥ https://ko.wikipedia.org/wiki/위키백과:대문 aaa https://ko.wikipedia.org/wiki/위키백과:대문" + +# not matching at all +echo "http://test:password@example.com/some_path"