feat(search): tighten the grammar and use Unicode case folding
This commit is contained in:
1 parent
97f79db3f1
commit
78e5faa3c4
7 files changed
+305
-105
No files matched your search
@@ -130,8 +130,9 @@ public class SearchQueryParserTests {
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void QualifierValueMayContainColons() {
|
||||
AssertParsesTo(new SearchNode.TagName("a:b"), "tag:a:b");
|
||||
public void QuotedQualifierValueMayContainColonsAndOperators() {
|
||||
AssertParsesTo(new SearchNode.TagName("a:b"), "tag:\"a:b\"");
|
||||
AssertParsesTo(new SearchNode.TagName("AND"), "tag:\"AND\"");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
@@ -148,12 +149,18 @@ public class SearchQueryParserTests {
|
||||
[TestMethod]
|
||||
public void IdQualifierParsesNumber() {
|
||||
AssertParsesTo(new SearchNode.CardId(42), "id:42");
|
||||
AssertParsesTo(new SearchNode.CardId(12), "id:\"12\"");
|
||||
AssertParsesTo(new SearchNode.CardId(61), "id:#61");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void ColonWithoutKeyPrefixIsFreeText() {
|
||||
AssertParsesTo(new SearchNode.FreeText(":foo"), ":foo");
|
||||
public void QuotedTextWithColonIsFreeText() {
|
||||
AssertParsesTo(new SearchNode.FreeText("http://x"), "\"http://x\"");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void WordMayContainHashAndPunctuation() {
|
||||
AssertParsesTo(new SearchNode.FreeText("#61"), "#61");
|
||||
AssertParsesTo(new SearchNode.FreeText("c#/.net"), "c#/.net");
|
||||
}
|
||||
|
||||
#endregion
|
||||
@@ -227,8 +234,84 @@ public class SearchQueryParserTests {
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void NonNumericIdIsAnError() {
|
||||
AssertSyntaxError("id:12x");
|
||||
[DataRow("tag: \"x\"")]
|
||||
[DataRow("title: \"a b\"")]
|
||||
public void SpacedQualifierPhraseValueIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("a\"b\"")]
|
||||
[DataRow("\"a\"b")]
|
||||
[DataRow("\"a\"\"b\"")]
|
||||
[DataRow("tag:\"x\"y")]
|
||||
[DataRow("tag:\"x\"\"y\"")]
|
||||
public void PhraseGluedToOtherTextIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("\"\"")]
|
||||
[DataRow("a \"\"")]
|
||||
[DataRow("tag:\"\"")]
|
||||
public void EmptyPhraseIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void PhraseMayTouchParentheses() {
|
||||
AssertParsesTo(
|
||||
new SearchNode.And([new SearchNode.FreeText("a b"), new SearchNode.FreeText("c")]),
|
||||
"(\"a b\")(c)");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("id:12x")]
|
||||
[DataRow("id:abc")]
|
||||
[DataRow("id:0")]
|
||||
[DataRow("id:-1")]
|
||||
[DataRow("id:+1")]
|
||||
[DataRow("id:#")]
|
||||
[DataRow("id:##1")]
|
||||
[DataRow("id:1.5")]
|
||||
[DataRow("id:٣")]
|
||||
[DataRow("id:99999999999999999999")]
|
||||
public void InvalidIdValueIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void QuotedIdValueIsAnError() {
|
||||
AssertSyntaxError("id:\"61\"");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("tag:AND")]
|
||||
[DataRow("tag:OR")]
|
||||
[DataRow("title:AND")]
|
||||
public void BareOperatorAsQualifierValueIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void UnquotedQualifierValueWithColonIsAnError() {
|
||||
AssertSyntaxError("tag:a:b");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow(":foo")]
|
||||
[DataRow("http://x")]
|
||||
[DataRow("a:")]
|
||||
public void ColonAfterNonKeyIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("a\\b")]
|
||||
[DataRow("\\")]
|
||||
[DataRow("tag:a\\b")]
|
||||
public void BackslashInBareWordIsAnError(string text) {
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
|
||||
@@ -26,12 +26,14 @@ public class SearchSqlCompilerTests {
|
||||
// Columns: 1 "To Do", 2 "In Progress".
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO columns (title, description, created_at, updated_at) VALUES ('To Do', '', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO columns (title, description, created_at, updated_at) VALUES ('In Progress', '', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO columns (title, description, created_at, updated_at) VALUES ('Archive', '', 1, 1);");
|
||||
|
||||
// Cards: 1 titled, 2 title-or-content matches, 3 untitled, 4 wildcard chars.
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO cards (column_id, title, content, created_at, updated_at) VALUES (1, 'Fix login', 'Cannot sign in', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO cards (column_id, title, content, created_at, updated_at) VALUES (2, 'Green apple', 'the apple pie', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO cards (column_id, title, content, created_at, updated_at) VALUES (2, '', 'banana split', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO cards (column_id, title, content, created_at, updated_at) VALUES (1, '100% done', 'a_b test', 1, 1);");
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO cards (column_id, title, content, created_at, updated_at) VALUES (3, 'ÄÖÜ Straße', '', 1, 1);");
|
||||
|
||||
// Tags: mixed case names plus one containing spaces.
|
||||
SqliteTestHelper.Exec(setup, "INSERT INTO tags (name, color, description) VALUES ('Bug', '#ff0000', '');");
|
||||
@@ -143,20 +145,27 @@ public class SearchSqlCompilerTests {
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void LikeWildcardsInUserTextMatchLiterally() {
|
||||
public void PercentAndUnderscoreInUserTextMatchLiterally() {
|
||||
using var fixture = new SearchFixture();
|
||||
CollectionAssert.AreEqual(new[] { 4L }, fixture.Search("100%"));
|
||||
CollectionAssert.AreEqual(new[] { 4L }, fixture.Search("a_b"));
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void UnicodeCaseFoldingMatchesNonAscii() {
|
||||
using var fixture = new SearchFixture();
|
||||
|
||||
// SQLite's LIKE would only fold ASCII; the ykb_fold function folds these too.
|
||||
CollectionAssert.AreEqual(new[] { 5L }, fixture.Search("äöü"));
|
||||
CollectionAssert.AreEqual(new[] { 5L }, fixture.Search("straße"));
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void CompiledPredicateUsesNamedParameters() {
|
||||
CompiledSearch compiled = SearchSqlCompiler.Compile(new SearchNode.FreeText("x"));
|
||||
|
||||
StringAssert.Contains(compiled.Predicate, "$p0");
|
||||
StringAssert.Contains(compiled.Predicate, "$p1");
|
||||
Assert.AreEqual(2, compiled.Parameters.Count);
|
||||
Assert.AreEqual("%x%", compiled.Parameters[0].Value);
|
||||
Assert.AreEqual("%x%", compiled.Parameters[1].Value);
|
||||
Assert.AreEqual(1, compiled.Parameters.Count);
|
||||
Assert.AreEqual("x", compiled.Parameters[0].Value);
|
||||
}
|
||||
}
|
||||
@@ -34,21 +34,20 @@ public abstract record SearchToken {
|
||||
/// quoted phrases and parentheses; whitespace only separates tokens.
|
||||
///
|
||||
/// The lexer is strict: inside a phrase only <c>\"</c> and <c>\\</c> are valid
|
||||
/// escapes, a bare backslash or an unknown escape is rejected, and an
|
||||
/// unterminated phrase is rejected. Any input not covered by the token rules
|
||||
/// is rejected rather than silently ignored.
|
||||
/// escapes, a bare backslash or an unknown escape is rejected, a backslash in a
|
||||
/// bare word is rejected, and an unterminated phrase is rejected. A phrase
|
||||
/// must not be empty and must be separated from neighbouring words and
|
||||
/// phrases by whitespace or a parenthesis, except as a qualifier value glued to
|
||||
/// its key (<c>tag:"a b"</c>). Any input not covered by the token rules is
|
||||
/// rejected rather than silently ignored.
|
||||
/// </summary>
|
||||
public static class SearchLexer {
|
||||
/// <summary>
|
||||
/// A valid phrase escape sequence: <c>\"</c> or <c>\\</c>.
|
||||
/// </summary>
|
||||
/// <summary>A valid phrase escape sequence: <c>\"</c> or <c>\\</c>.</summary>
|
||||
private static readonly Parser<char, char> PhraseEscape =
|
||||
Parser.Try(Parser.String("\\\"").Select(_ => '"'))
|
||||
.Or(Parser.String("\\\\").Select(_ => '\\'));
|
||||
|
||||
/// <summary>
|
||||
/// One phrase character: an escape or any character except a quote or backslash.
|
||||
/// </summary>
|
||||
/// <summary>One phrase character: an escape or any character except a quote or backslash.</summary>
|
||||
private static readonly Parser<char, char> PhraseCharacter =
|
||||
PhraseEscape.Or(Parser<char>.Token(character => character != '"' && character != '\\'));
|
||||
|
||||
@@ -64,39 +63,99 @@ public static class SearchLexer {
|
||||
private static readonly Parser<char, SearchToken> CloseParenParser =
|
||||
Parser.Char(')').Select(_ => (SearchToken)new SearchToken.CloseParen());
|
||||
|
||||
// ':' stays inside the word token (the parser splits qualifiers); a backslash
|
||||
// is not a word character, so a bare one is left over and rejected by End.
|
||||
private static readonly Parser<char, SearchToken> WordParser =
|
||||
Parser<char>.Token(character =>
|
||||
!char.IsWhiteSpace(character) && character != '(' && character != ')' && character != '"')
|
||||
!char.IsWhiteSpace(character) && character != '(' && character != ')' && character != '"' && character != '\\')
|
||||
.AtLeastOnce()
|
||||
.Select(chars => (SearchToken)new SearchToken.Word(new string(chars.ToArray())));
|
||||
|
||||
private static readonly Parser<char, SearchToken> TokenParser =
|
||||
// Each token with its source span, so adjacency can be checked afterwards.
|
||||
private static readonly Parser<char, PositionedToken> TokenParser =
|
||||
Parser.SkipWhitespaces.Then(
|
||||
PhraseParser
|
||||
from start in Parser<char>.CurrentOffset
|
||||
from token in PhraseParser
|
||||
.Or(OpenParenParser)
|
||||
.Or(CloseParenParser)
|
||||
.Or(WordParser));
|
||||
.Or(WordParser)
|
||||
from end in Parser<char>.CurrentOffset
|
||||
select new PositionedToken(token, start, end));
|
||||
|
||||
// Try(...) keeps a failed token non-consuming so Many stops gracefully; End
|
||||
// then rejects any leftover input, which turns dangling quotes or invalid
|
||||
// escapes into hard failures instead of silently dropped text.
|
||||
private static readonly Parser<char, IReadOnlyList<SearchToken>> Lexer =
|
||||
private static readonly Parser<char, IReadOnlyList<PositionedToken>> Lexer =
|
||||
Parser.Try(TokenParser).Many()
|
||||
.Before(Parser.SkipWhitespaces)
|
||||
.Before(Parser<char>.End)
|
||||
.Select(tokens => (IReadOnlyList<SearchToken>)tokens.ToList());
|
||||
.Select(tokens => (IReadOnlyList<PositionedToken>)tokens.ToList());
|
||||
|
||||
/// <summary>
|
||||
/// Tokenizes the search text.
|
||||
/// </summary>
|
||||
/// <param name="input">The raw search text.</param>
|
||||
/// <returns>The token stream.</returns>
|
||||
/// <exception cref="SearchSyntaxException">The text contains an invalid escape or an unterminated phrase.</exception>
|
||||
/// <exception cref="SearchSyntaxException">
|
||||
/// The text contains an invalid escape, an unterminated or empty phrase, a
|
||||
/// phrase glued to other text, or whitespace between a qualifier and its phrase value.
|
||||
/// </exception>
|
||||
public static IReadOnlyList<SearchToken> Tokenize(string input) {
|
||||
IReadOnlyList<PositionedToken> tokens;
|
||||
try {
|
||||
return Lexer.ParseOrThrow(input);
|
||||
tokens = Lexer.ParseOrThrow(input);
|
||||
} catch (ParseException exception) {
|
||||
throw new SearchSyntaxException(exception.Message, exception);
|
||||
}
|
||||
|
||||
EnsurePhrasesAreDelimited(tokens);
|
||||
return tokens.Select(positioned => positioned.Token).ToList();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Rejects the phrase placements the grammar does not allow but the token
|
||||
/// rules alone would accept: an empty phrase, a phrase touching a word or
|
||||
/// another phrase (<c>a"b"</c>, <c>"a"b</c>, <c>"a""b"</c>), and whitespace
|
||||
/// between <c>key:</c> and its phrase value (<c>tag: "x"</c>). The only
|
||||
/// legal contact is <c>key:"value"</c>.
|
||||
/// </summary>
|
||||
/// <param name="tokens">The positioned token stream.</param>
|
||||
/// <exception cref="SearchSyntaxException">A phrase is misplaced.</exception>
|
||||
private static void EnsurePhrasesAreDelimited(IReadOnlyList<PositionedToken> tokens) {
|
||||
for (int index = 0; index < tokens.Count; index++) {
|
||||
if (tokens[index].Token is SearchToken.Phrase { Text.Length: 0 }) {
|
||||
throw new SearchSyntaxException("Empty phrase \"\" is not allowed.");
|
||||
}
|
||||
|
||||
if (index == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
PositionedToken previous = tokens[index - 1];
|
||||
PositionedToken current = tokens[index];
|
||||
bool qualifierValue = previous.Token is SearchToken.Word word && word.Text.EndsWith(':')
|
||||
&& current.Token is SearchToken.Phrase;
|
||||
bool touching = previous.End == current.Start;
|
||||
|
||||
if (qualifierValue && !touching) {
|
||||
throw new SearchSyntaxException(
|
||||
$"Qualifier '{((SearchToken.Word)previous.Token).Text}' must be followed directly by its value, without whitespace.");
|
||||
}
|
||||
|
||||
bool phraseInvolved = previous.Token is SearchToken.Phrase || current.Token is SearchToken.Phrase;
|
||||
bool parenInvolved = previous.Token is SearchToken.OpenParen or SearchToken.CloseParen
|
||||
|| current.Token is SearchToken.OpenParen or SearchToken.CloseParen;
|
||||
if (touching && phraseInvolved && !parenInvolved && !qualifierValue) {
|
||||
throw new SearchSyntaxException("A phrase must be separated from adjacent text by whitespace.");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A token with its source span.
|
||||
/// </summary>
|
||||
/// <param name="Token">The token.</param>
|
||||
/// <param name="Start">Offset of the first character.</param>
|
||||
/// <param name="End">Offset just past the last character.</param>
|
||||
private sealed record PositionedToken(SearchToken Token, int Start, int End);
|
||||
}
|
||||
@@ -7,8 +7,10 @@ namespace YKanBan.Search;
|
||||
/// <summary>
|
||||
/// Parses the search grammar with Pidgin, strictly: any malformed input
|
||||
/// (unbalanced parentheses, dangling or consecutive operators, unterminated
|
||||
/// phrases, invalid escapes, qualifier keys without a value or with an
|
||||
/// unknown/non-numeric id value) raises <see cref="SearchSyntaxException"/>.
|
||||
/// phrases, invalid escapes, a backslash in a bare word, unknown qualifier
|
||||
/// keys, qualifiers without a value, unquoted values containing ':' or equal
|
||||
/// to AND/OR, and id values that are not an unquoted positive integer with an
|
||||
/// optional '#') raises <see cref="SearchSyntaxException"/>.
|
||||
///
|
||||
/// Only uppercase AND/OR are operators (lowercase ones are plain words); AND —
|
||||
/// explicit or by adjacency — binds tighter than OR; parentheses group. Blank
|
||||
@@ -16,9 +18,7 @@ namespace YKanBan.Search;
|
||||
/// qualifier keys are fixed English grammar and are never localized.
|
||||
/// </summary>
|
||||
public static class SearchQueryParser {
|
||||
/// <summary>
|
||||
/// The recognized qualifier keys; add new entries here to extend the grammar.
|
||||
/// </summary>
|
||||
/// <summary>The recognized qualifier keys; add new entries here to extend the grammar.</summary>
|
||||
private static readonly (string Key, QualifierKind Kind)[] QualifierTable =
|
||||
[
|
||||
("tag", QualifierKind.Tag),
|
||||
@@ -42,16 +42,12 @@ public static class SearchQueryParser {
|
||||
private static readonly Parser<SearchToken, SearchToken> CloseParen =
|
||||
Parser<SearchToken>.Token(token => token is SearchToken.CloseParen);
|
||||
|
||||
/// <summary>
|
||||
/// The text of a phrase token.
|
||||
/// </summary>
|
||||
/// <summary>The text of a phrase token.</summary>
|
||||
private static readonly Parser<SearchToken, string> PhraseText =
|
||||
Parser<SearchToken>.Token(token => token is SearchToken.Phrase)
|
||||
.Select(token => ((SearchToken.Phrase)token).Text);
|
||||
|
||||
/// <summary>
|
||||
/// The text of a word token, operator keywords excluded (they never start an operand).
|
||||
/// </summary>
|
||||
/// <summary>The text of a word token, operator keywords excluded (they never start an operand).</summary>
|
||||
private static readonly Parser<SearchToken, string> NonKeywordWord =
|
||||
Parser<SearchToken>.Token(token => token is SearchToken.Word word && word.Text != "AND" && word.Text != "OR")
|
||||
.Select(token => ((SearchToken.Word)token).Text);
|
||||
@@ -96,9 +92,7 @@ public static class SearchQueryParser {
|
||||
private static readonly Parser<SearchToken, SearchNode> Query =
|
||||
OrExpr.Before(Parser<SearchToken>.End);
|
||||
|
||||
/// <summary>
|
||||
/// Defers the OrExpr field read to parse time (recursion cycle breaker).
|
||||
/// </summary>
|
||||
/// <summary>Defers the OrExpr field read to parse time (recursion cycle breaker).</summary>
|
||||
/// <returns>The OR-expression parser.</returns>
|
||||
private static Parser<SearchToken, SearchNode> OrExprDeferred() => OrExpr;
|
||||
|
||||
@@ -115,6 +109,11 @@ public static class SearchQueryParser {
|
||||
return null;
|
||||
}
|
||||
|
||||
// YYC MARK: the expression is deliberately not capped by parenthesis
|
||||
// depth or atom count. The app is local-only with no untrusted search
|
||||
// input, so there is no attack surface; a pathologically deep or long
|
||||
// expression can still overflow the parser stack or exceed SQLite's
|
||||
// expression nesting limit, which is accepted here.
|
||||
try {
|
||||
List<SearchToken> tokens = SearchLexer.Tokenize(text).ToList();
|
||||
return Query.ParseOrThrow(tokens);
|
||||
@@ -133,52 +132,77 @@ public static class SearchQueryParser {
|
||||
/// <returns>The parser producing the operand node.</returns>
|
||||
private static Parser<SearchToken, SearchNode> QualifierFrom(string word) {
|
||||
int colonIndex = word.IndexOf(':');
|
||||
|
||||
// Only a leading run of letters/digits/underscores before ':' can be a qualifier key.
|
||||
if (colonIndex <= 0 || !IsQualifierKeyCandidate(word.AsSpan(0, colonIndex))) {
|
||||
if (colonIndex < 0) {
|
||||
return Parser<SearchToken>.Return<SearchNode>(new SearchNode.FreeText(word));
|
||||
}
|
||||
|
||||
// Any ':' in a bare word makes it a qualifier, so the text before it must be a known key.
|
||||
string key = word[..colonIndex];
|
||||
string value = word[(colonIndex + 1)..];
|
||||
|
||||
if (!TryLookupQualifier(key, out QualifierKind kind)) {
|
||||
return Parser<SearchToken>.Fail<SearchNode>($"Unknown qualifier key '{key}'.");
|
||||
return Parser<SearchToken>.Fail<SearchNode>($"Unknown qualifier key '{key}'; quote the text to search for it literally.");
|
||||
}
|
||||
|
||||
// "key:" without an inline value takes the value from the following phrase, e.g. tag:"a b".
|
||||
return value.Length == 0
|
||||
? PhraseText.Bind(text => AtomParser(kind, text))
|
||||
: AtomParser(kind, value);
|
||||
if (value.Length == 0) {
|
||||
return kind == QualifierKind.Id
|
||||
? Parser<SearchToken>.Fail<SearchNode>("The id qualifier expects an unquoted positive integer.")
|
||||
: PhraseText.Select(phrase => TextAtom(kind, phrase));
|
||||
}
|
||||
|
||||
if (value.Contains(':')) {
|
||||
return Parser<SearchToken>.Fail<SearchNode>($"Unquoted value '{value}' of qualifier '{key}' contains ':'; quote the value.");
|
||||
}
|
||||
|
||||
if (value is "AND" or "OR") {
|
||||
return Parser<SearchToken>.Fail<SearchNode>($"Operator {value} cannot be a qualifier value; quote it to search for it literally.");
|
||||
}
|
||||
|
||||
if (kind == QualifierKind.Id) {
|
||||
return TryParseCardId(value, out long id)
|
||||
? Parser<SearchToken>.Return<SearchNode>(new SearchNode.CardId(id))
|
||||
: Parser<SearchToken>.Fail<SearchNode>($"The id qualifier expects a positive integer, got '{value}'.");
|
||||
}
|
||||
|
||||
return Parser<SearchToken>.Return(TextAtom(kind, value));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Builds the atom parser for a qualifier value, failing strictly on a non-numeric id.
|
||||
/// Builds the atom for a textual qualifier value.
|
||||
/// </summary>
|
||||
/// <param name="kind">The qualifier kind.</param>
|
||||
/// <param name="kind">The qualifier kind; never <see cref="QualifierKind.Id"/>.</param>
|
||||
/// <param name="value">The qualifier value.</param>
|
||||
/// <returns>The parser producing the atom node.</returns>
|
||||
private static Parser<SearchToken, SearchNode> AtomParser(QualifierKind kind, string value) =>
|
||||
AtomFrom(kind, value) is { } atom
|
||||
? Parser<SearchToken>.Return(atom)
|
||||
: Parser<SearchToken>.Fail<SearchNode>($"The id qualifier expects an integer, got '{value}'.");
|
||||
|
||||
/// <summary>
|
||||
/// Builds the atom for a qualifier value; null when an id value is not an integer.
|
||||
/// </summary>
|
||||
/// <param name="kind">The qualifier kind.</param>
|
||||
/// <param name="value">The qualifier value.</param>
|
||||
/// <returns>The atom node, or <see langword="null"/> for an invalid id value.</returns>
|
||||
private static SearchNode? AtomFrom(QualifierKind kind, string value) => kind switch {
|
||||
/// <returns>The atom node.</returns>
|
||||
private static SearchNode TextAtom(QualifierKind kind, string value) => kind switch {
|
||||
QualifierKind.Tag => new SearchNode.TagName(value),
|
||||
QualifierKind.Title => new SearchNode.FieldText(SearchField.Title, value),
|
||||
QualifierKind.Content => new SearchNode.FieldText(SearchField.Content, value),
|
||||
QualifierKind.Column => new SearchNode.ColumnTitle(value),
|
||||
QualifierKind.Id when long.TryParse(value, NumberStyles.Integer, CultureInfo.InvariantCulture, out long id) =>
|
||||
new SearchNode.CardId(id),
|
||||
_ => null,
|
||||
_ => throw new InvalidOperationException($"Qualifier kind {kind} has no textual value."),
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Parses an id qualifier value: decimal digits with an optional leading
|
||||
/// <c>#</c>, denoting a positive integer.
|
||||
/// </summary>
|
||||
/// <param name="value">The inline qualifier value.</param>
|
||||
/// <param name="id">The parsed card id.</param>
|
||||
/// <returns><see langword="true"/> when the value is a valid positive card id.</returns>
|
||||
private static bool TryParseCardId(string value, out long id) {
|
||||
ReadOnlySpan<char> digits = value.AsSpan();
|
||||
if (digits.StartsWith("#")) {
|
||||
digits = digits[1..];
|
||||
}
|
||||
|
||||
// Digits only: the span check rules out signs, whitespace and other NumberStyles leniency.
|
||||
id = 0;
|
||||
return digits.Length > 0
|
||||
&& !digits.ContainsAnyExceptInRange('0', '9')
|
||||
&& long.TryParse(digits, NumberStyles.None, CultureInfo.InvariantCulture, out id)
|
||||
&& id > 0;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Looks up a qualifier key in the extension table with ordinal (case-sensitive) matching.
|
||||
/// </summary>
|
||||
@@ -197,22 +221,6 @@ public static class SearchQueryParser {
|
||||
return false;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Returns whether the text looks like a qualifier key: a non-empty run of
|
||||
/// letters, digits or underscores.
|
||||
/// </summary>
|
||||
/// <param name="text">The candidate key text.</param>
|
||||
/// <returns><see langword="true"/> when every character is a word character.</returns>
|
||||
private static bool IsQualifierKeyCandidate(ReadOnlySpan<char> text) {
|
||||
foreach (char character in text) {
|
||||
if (!(char.IsLetterOrDigit(character) || character == '_')) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Collapses an AND chain into a single node.
|
||||
/// </summary>
|
||||
@@ -236,9 +244,7 @@ public static class SearchQueryParser {
|
||||
|
||||
#endregion
|
||||
|
||||
/// <summary>
|
||||
/// The recognized qualifier kinds.
|
||||
/// </summary>
|
||||
/// <summary>The recognized qualifier kinds.</summary>
|
||||
private enum QualifierKind {
|
||||
Tag,
|
||||
Title,
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
using YKanBan.Storage;
|
||||
|
||||
namespace YKanBan.Search;
|
||||
|
||||
/// <summary>
|
||||
@@ -10,10 +12,11 @@ public sealed record CompiledSearch(string Predicate, IReadOnlyList<(string Name
|
||||
|
||||
/// <summary>
|
||||
/// Compiles a parsed search AST into a parameterized SQL predicate. Matching is
|
||||
/// case-insensitive with SQLite's built-in semantics: LIKE's default case
|
||||
/// folding for substring matches plus COLLATE NOCASE for exact matches. LIKE
|
||||
/// wildcards and the escape character in user text are escaped so they match
|
||||
/// literally.
|
||||
/// Unicode case-insensitive: both sides are folded with
|
||||
/// <see cref="SqliteDatabase.Fold"/> (the column side through the per-connection
|
||||
/// <see cref="SqliteDatabase.FoldFunctionName"/> function) and then compared
|
||||
/// ordinally — <c>instr</c> for substrings, <c>=</c> for exact matches. LIKE is
|
||||
/// deliberately not used, so <c>%</c> and <c>_</c> carry no wildcard meaning.
|
||||
/// </summary>
|
||||
public static class SearchSqlCompiler {
|
||||
/// <summary>
|
||||
@@ -43,26 +46,25 @@ public static class SearchSqlCompiler {
|
||||
|
||||
case SearchNode.FreeText freeText: {
|
||||
// A bare word matches a substring of the title OR the content.
|
||||
string titleParam = AddParameter(parameters, "%" + EscapeLike(freeText.Text) + "%");
|
||||
string contentParam = AddParameter(parameters, "%" + EscapeLike(freeText.Text) + "%");
|
||||
return $"(c.title LIKE {titleParam} ESCAPE '\\' OR c.content LIKE {contentParam} ESCAPE '\\')";
|
||||
string param = AddParameter(parameters, SqliteDatabase.Fold(freeText.Text));
|
||||
return $"({Contains("c.title", param)} OR {Contains("c.content", param)})";
|
||||
}
|
||||
|
||||
case SearchNode.FieldText field: {
|
||||
string column = field.Field == SearchField.Title ? "c.title" : "c.content";
|
||||
string param = AddParameter(parameters, "%" + EscapeLike(field.Text) + "%");
|
||||
return $"{column} LIKE {param} ESCAPE '\\'";
|
||||
string param = AddParameter(parameters, SqliteDatabase.Fold(field.Text));
|
||||
return Contains(column, param);
|
||||
}
|
||||
|
||||
case SearchNode.TagName tag: {
|
||||
string param = AddParameter(parameters, tag.Name);
|
||||
string param = AddParameter(parameters, SqliteDatabase.Fold(tag.Name));
|
||||
return "EXISTS (SELECT 1 FROM card_tags ct JOIN tags t ON t.id = ct.tag_id " +
|
||||
$"WHERE ct.card_id = c.id AND t.name COLLATE NOCASE = {param})";
|
||||
$"WHERE ct.card_id = c.id AND {Folded("t.name")} = {param})";
|
||||
}
|
||||
|
||||
case SearchNode.ColumnTitle column: {
|
||||
string param = AddParameter(parameters, column.Title);
|
||||
return $"c.column_id IN (SELECT id FROM columns WHERE title COLLATE NOCASE = {param})";
|
||||
string param = AddParameter(parameters, SqliteDatabase.Fold(column.Title));
|
||||
return $"c.column_id IN (SELECT id FROM columns WHERE {Folded("title")} = {param})";
|
||||
}
|
||||
|
||||
case SearchNode.CardId id: {
|
||||
@@ -88,10 +90,17 @@ public static class SearchSqlCompiler {
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Escapes LIKE wildcards and the escape character inside user text.
|
||||
/// Wraps a column expression in the case-folding SQL function.
|
||||
/// </summary>
|
||||
/// <param name="text">The raw user text.</param>
|
||||
/// <returns>The text with <c>\</c>, <c>%</c> and <c>_</c> escaped.</returns>
|
||||
private static string EscapeLike(string text) =>
|
||||
text.Replace("\\", "\\\\").Replace("%", "\\%").Replace("_", "\\_");
|
||||
/// <param name="column">The column expression.</param>
|
||||
/// <returns>The folded column expression.</returns>
|
||||
private static string Folded(string column) => $"{SqliteDatabase.FoldFunctionName}({column})";
|
||||
|
||||
/// <summary>
|
||||
/// Builds an ordinal substring test of a folded column against an already folded parameter.
|
||||
/// </summary>
|
||||
/// <param name="column">The column expression.</param>
|
||||
/// <param name="param">The placeholder of the folded search text.</param>
|
||||
/// <returns>The SQL boolean expression.</returns>
|
||||
private static string Contains(string column, string param) => $"instr({Folded(column)}, {param}) > 0";
|
||||
}
|
||||
@@ -7,6 +7,14 @@ namespace YKanBan.Search;
|
||||
/// localized.
|
||||
/// </summary>
|
||||
public sealed class SearchSyntaxException : Exception {
|
||||
/// <summary>
|
||||
/// Initializes the exception with a descriptive English message.
|
||||
/// </summary>
|
||||
/// <param name="message">The English diagnostic message.</param>
|
||||
public SearchSyntaxException(string message)
|
||||
: base(message) {
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Initializes the exception with a descriptive English message.
|
||||
/// </summary>
|
||||
|
||||
@@ -8,6 +8,20 @@ namespace YKanBan.Storage;
|
||||
/// foreign keys on) and applies incremental <c>user_version</c> migrations.
|
||||
/// </summary>
|
||||
public static class SqliteDatabase {
|
||||
/// <summary>
|
||||
/// Name of the per-connection SQL function that applies <see cref="Fold"/>;
|
||||
/// search uses it for Unicode case-insensitive matching.
|
||||
/// </summary>
|
||||
public const string FoldFunctionName = "ykb_fold";
|
||||
|
||||
/// <summary>
|
||||
/// Unicode case folding shared by the SQL function and the search
|
||||
/// parameters it is compared against; comparisons after folding are ordinal.
|
||||
/// </summary>
|
||||
/// <param name="text">The text to fold.</param>
|
||||
/// <returns>The invariant lowercase form.</returns>
|
||||
public static string Fold(string text) => text.ToLowerInvariant();
|
||||
|
||||
/// <summary>
|
||||
/// Opens (creating if needed) a database file, applies the mandatory
|
||||
/// PRAGMAs and runs any pending migrations.
|
||||
@@ -27,6 +41,7 @@ public static class SqliteDatabase {
|
||||
connection.Open();
|
||||
try {
|
||||
ApplyPragmas(connection);
|
||||
RegisterFunctions(connection);
|
||||
ApplyMigrations(connection, migrations);
|
||||
return connection;
|
||||
} catch {
|
||||
@@ -59,6 +74,17 @@ public static class SqliteDatabase {
|
||||
Execute(connection, "PRAGMA foreign_keys=ON;");
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Registers the custom SQL functions; like the PRAGMAs they live on the
|
||||
/// connection and must be registered on every one.
|
||||
/// </summary>
|
||||
/// <param name="connection">An open database connection.</param>
|
||||
internal static void RegisterFunctions(SqliteConnection connection) {
|
||||
// NULL never reaches it from the NOT NULL text columns, but stays NULL-safe regardless.
|
||||
connection.CreateFunction<string?, string?>(
|
||||
FoldFunctionName, text => text is null ? null : Fold(text), isDeterministic: true);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs every migration newer than the stored <c>user_version</c>, each in
|
||||
/// its own transaction together with its version bump.
|
||||
|
||||
Reference in new issue
Block a user