fix(search): reject empty phrases and phrases glued to other text
The lexer now keeps token offsets and rejects what the grammar never allowed but the token rules accepted: - an empty phrase: "" (matched every card), tag:"" - a phrase touching a word or another phrase: a"b", "a"b, "a""b" - whitespace between a qualifier and its phrase value: tag: "x" key:"value" stays the only legal contact; parentheses still delimit phrases. Listed in PLAN 6.1/6.2 and USAGE.
This commit is contained in:
1 parent
9277e2a331
commit
1d744ba504
4 files changed
+124
-12
No files matched your search
@@ -159,7 +159,7 @@ value := word | phrase -- tag / title / content / column
|
||||
| '#'? digit+ -- id: a positive decimal integer
|
||||
word := one or more characters, except whitespace and ( ) " : \ ;
|
||||
not equal to AND or OR
|
||||
phrase := '"' any text '"' -- escapes: \" and \\
|
||||
phrase := '"' one or more characters '"' -- escapes: \" and \\
|
||||
```
|
||||
|
||||
### Operators
|
||||
@@ -198,6 +198,10 @@ phrase := '"' any text '"' -- escapes: \" and \\
|
||||
`\\` for a backslash; any other backslash is an error.
|
||||
- Outside quotes, `\` is not allowed at all, and `:` always starts a
|
||||
qualifier. Quote any text containing them: `"C:\\temp"`, `"a:b"`.
|
||||
- A phrase cannot be empty, and must be separated from neighbouring words
|
||||
and phrases by a space or a parenthesis: write `a "b"`, not `a"b"`.
|
||||
- A qualifier's phrase value follows the `:` directly: `tag:"needs review"`,
|
||||
not `tag: "needs review"`.
|
||||
|
||||
### Case
|
||||
|
||||
@@ -227,6 +231,9 @@ These are invalid:
|
||||
| A qualifier without a value | `tag:` |
|
||||
| `AND` / `OR` as an unquoted qualifier value | `tag:AND` (write `tag:"AND"`) |
|
||||
| An unterminated phrase | `"dark mode` |
|
||||
| An empty phrase | `""`, `tag:""` |
|
||||
| A phrase touching other text | `a"b"`, `"a"b`, `"a""b"` |
|
||||
| A space between a qualifier and its phrase value | `tag: "needs review"` |
|
||||
| An invalid escape in a phrase | `tag:"a\x"` |
|
||||
| A backslash outside a phrase | `C:\temp` |
|
||||
| An unknown qualifier key | `label:bug`, `Tag:bug`, `http://x` |
|
||||
|
||||
@@ -296,6 +296,42 @@ public class SearchQueryParserTests
|
||||
AssertSyntaxError("tag: bug");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("tag: \"x\"")]
|
||||
[DataRow("title: \"a b\"")]
|
||||
public void SpacedQualifierPhraseValueIsAnError(string text)
|
||||
{
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("a\"b\"")]
|
||||
[DataRow("\"a\"b")]
|
||||
[DataRow("\"a\"\"b\"")]
|
||||
[DataRow("tag:\"x\"y")]
|
||||
[DataRow("tag:\"x\"\"y\"")]
|
||||
public void PhraseGluedToOtherTextIsAnError(string text)
|
||||
{
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("\"\"")]
|
||||
[DataRow("a \"\"")]
|
||||
[DataRow("tag:\"\"")]
|
||||
public void EmptyPhraseIsAnError(string text)
|
||||
{
|
||||
AssertSyntaxError(text);
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
public void PhraseMayTouchParentheses()
|
||||
{
|
||||
AssertParsesTo(
|
||||
new SearchNode.And([new SearchNode.FreeText("a b"), new SearchNode.FreeText("c")]),
|
||||
"(\"a b\")(c)");
|
||||
}
|
||||
|
||||
[TestMethod]
|
||||
[DataRow("id:12x")]
|
||||
[DataRow("id:abc")]
|
||||
|
||||
@@ -36,8 +36,11 @@ public abstract record SearchToken
|
||||
///
|
||||
/// The lexer is strict: inside a phrase only <c>\"</c> and <c>\\</c> are valid
|
||||
/// escapes, a bare backslash or an unknown escape is rejected, a backslash in a
|
||||
/// bare word is rejected, and an unterminated phrase is rejected. Any input not covered by the token rules
|
||||
/// is rejected rather than silently ignored.
|
||||
/// bare word is rejected, and an unterminated phrase is rejected. A phrase
|
||||
/// must not be empty and must be separated from neighbouring words and
|
||||
/// phrases by whitespace or a parenthesis, except as a qualifier value glued to
|
||||
/// its key (<c>tag:"a b"</c>). Any input not covered by the token rules is
|
||||
/// rejected rather than silently ignored.
|
||||
/// </summary>
|
||||
public static class SearchLexer
|
||||
{
|
||||
@@ -70,37 +73,101 @@ public static class SearchLexer
|
||||
.AtLeastOnce()
|
||||
.Select(chars => (SearchToken)new SearchToken.Word(new string(chars.ToArray())));
|
||||
|
||||
private static readonly Parser<char, SearchToken> TokenParser =
|
||||
// Each token with its source span, so adjacency can be checked afterwards.
|
||||
private static readonly Parser<char, PositionedToken> TokenParser =
|
||||
Parser.SkipWhitespaces.Then(
|
||||
PhraseParser
|
||||
from start in Parser<char>.CurrentOffset
|
||||
from token in PhraseParser
|
||||
.Or(OpenParenParser)
|
||||
.Or(CloseParenParser)
|
||||
.Or(WordParser));
|
||||
.Or(WordParser)
|
||||
from end in Parser<char>.CurrentOffset
|
||||
select new PositionedToken(token, start, end));
|
||||
|
||||
// Try(...) keeps a failed token non-consuming so Many stops gracefully; End
|
||||
// then rejects any leftover input, which turns dangling quotes or invalid
|
||||
// escapes into hard failures instead of silently dropped text.
|
||||
private static readonly Parser<char, IReadOnlyList<SearchToken>> Lexer =
|
||||
private static readonly Parser<char, IReadOnlyList<PositionedToken>> Lexer =
|
||||
Parser.Try(TokenParser).Many()
|
||||
.Before(Parser.SkipWhitespaces)
|
||||
.Before(Parser<char>.End)
|
||||
.Select(tokens => (IReadOnlyList<SearchToken>)tokens.ToList());
|
||||
.Select(tokens => (IReadOnlyList<PositionedToken>)tokens.ToList());
|
||||
|
||||
/// <summary>
|
||||
/// Tokenizes the search text.
|
||||
/// </summary>
|
||||
/// <param name="input">The raw search text.</param>
|
||||
/// <returns>The token stream.</returns>
|
||||
/// <exception cref="SearchSyntaxException">The text contains an invalid escape or an unterminated phrase.</exception>
|
||||
/// <exception cref="SearchSyntaxException">
|
||||
/// The text contains an invalid escape, an unterminated or empty phrase, a
|
||||
/// phrase glued to other text, or whitespace between a qualifier and its phrase value.
|
||||
/// </exception>
|
||||
public static IReadOnlyList<SearchToken> Tokenize(string input)
|
||||
{
|
||||
IReadOnlyList<PositionedToken> tokens;
|
||||
try
|
||||
{
|
||||
return Lexer.ParseOrThrow(input);
|
||||
tokens = Lexer.ParseOrThrow(input);
|
||||
}
|
||||
catch (ParseException exception)
|
||||
{
|
||||
throw new SearchSyntaxException(exception.Message, exception);
|
||||
}
|
||||
|
||||
EnsurePhrasesAreDelimited(tokens);
|
||||
return tokens.Select(positioned => positioned.Token).ToList();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Rejects the phrase placements the grammar does not allow but the token
|
||||
/// rules alone would accept: an empty phrase, a phrase touching a word or
|
||||
/// another phrase (<c>a"b"</c>, <c>"a"b</c>, <c>"a""b"</c>), and whitespace
|
||||
/// between <c>key:</c> and its phrase value (<c>tag: "x"</c>). The only
|
||||
/// legal contact is <c>key:"value"</c>.
|
||||
/// </summary>
|
||||
/// <param name="tokens">The positioned token stream.</param>
|
||||
/// <exception cref="SearchSyntaxException">A phrase is misplaced.</exception>
|
||||
private static void EnsurePhrasesAreDelimited(IReadOnlyList<PositionedToken> tokens)
|
||||
{
|
||||
for (int index = 0; index < tokens.Count; index++)
|
||||
{
|
||||
if (tokens[index].Token is SearchToken.Phrase { Text.Length: 0 })
|
||||
{
|
||||
throw new SearchSyntaxException("Empty phrase \"\" is not allowed.");
|
||||
}
|
||||
|
||||
if (index == 0)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
PositionedToken previous = tokens[index - 1];
|
||||
PositionedToken current = tokens[index];
|
||||
bool qualifierValue = previous.Token is SearchToken.Word word && word.Text.EndsWith(':')
|
||||
&& current.Token is SearchToken.Phrase;
|
||||
bool touching = previous.End == current.Start;
|
||||
|
||||
if (qualifierValue && !touching)
|
||||
{
|
||||
throw new SearchSyntaxException(
|
||||
$"Qualifier '{((SearchToken.Word)previous.Token).Text}' must be followed directly by its value, without whitespace.");
|
||||
}
|
||||
|
||||
bool phraseInvolved = previous.Token is SearchToken.Phrase || current.Token is SearchToken.Phrase;
|
||||
bool parenInvolved = previous.Token is SearchToken.OpenParen or SearchToken.CloseParen
|
||||
|| current.Token is SearchToken.OpenParen or SearchToken.CloseParen;
|
||||
if (touching && phraseInvolved && !parenInvolved && !qualifierValue)
|
||||
{
|
||||
throw new SearchSyntaxException("A phrase must be separated from adjacent text by whitespace.");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A token with its source span.
|
||||
/// </summary>
|
||||
/// <param name="Token">The token.</param>
|
||||
/// <param name="Start">Offset of the first character.</param>
|
||||
/// <param name="End">Offset just past the last character.</param>
|
||||
private sealed record PositionedToken(SearchToken Token, int Start, int End);
|
||||
}
|
||||
+4
-2
@@ -16,7 +16,7 @@
|
||||
- 网络共享明确不支持;命令行参数边界;macOS 无参启动行为写入文档(§2.3、§4.1、附录)
|
||||
- 补充:导出转义规则、卡片列表虚拟化、模态窗口 FlowDirection、`Resources` 类名使用规则、CI osx-x64 运行器(§5、§7、§8)
|
||||
- 依赖:Avalonia 11.3.6 → 11.3.22,Semi.Avalonia 11.2.1.10 → 11.3.22(§7)
|
||||
- 实现回补:搜索括号嵌套上限 64 层(§6.2);初始化中途锁冲突进锁冲突页、数据库文件缺失不静默新建(§4.4)
|
||||
- 实现回补:搜索括号嵌套上限 64 层,空短语、短语紧贴文本、`key:` 后空白均为非法(§6.1、§6.2);导出表格单元格转义 `\`(§5.8);初始化中途锁冲突进锁冲突页、数据库文件缺失不静默新建(§4.4)
|
||||
- 里程碑:标注 M0–M4 已完成,新增 **M4R(v1.9 回补)**(§9)
|
||||
|
||||
## 1. 项目概述
|
||||
@@ -279,10 +279,11 @@ key := tag | title | content | id | column
|
||||
value := word | phrase (tag/title/content/column)
|
||||
| '#'? digit+ (id:十进制正整数,可带前缀 #)
|
||||
word := 一个或多个字符,不含空白及 ( ) " : \ ;且不等于 AND / OR
|
||||
phrase := '"' 任意文本 '"' (内部支持转义:\" → 字面引号,\\ → 字面反斜杠)
|
||||
phrase := '"' 一个或多个字符 '"' (内部支持转义:\" → 字面引号,\\ → 字面反斜杠)
|
||||
```
|
||||
|
||||
- 裸词中出现 `:` 时其左侧必须是合法 key,否则为"未知限定符"错误;含 `:` `(` `)` `\` 等字符的搜索值须用引号(如 `"http://x"`)
|
||||
- 短语不得为空(`""`);短语与相邻裸词/短语之间须有空白或括号(`a"b"`、`"a"b`、`"a""b"` 非法),唯一例外是紧贴 key 的限定符值 `key:"…"`;`key:` 与短语值之间不得有空白(`tag: "x"` 非法)
|
||||
|
||||
### 6.2 语义
|
||||
|
||||
@@ -297,6 +298,7 @@ phrase := '"' 任意文本 '"' (内部支持转义:\" → 字面引号,
|
||||
- 运算符悬空(`bug AND`)、连续运算符(`a AND OR b`)
|
||||
- 限定符无值(`tag:`)、限定符值为裸 `AND`/`OR`(`tag:AND`,须写 `tag:"AND"`)
|
||||
- 短语引号未闭合、无效转义(`tag:"a\x"`)、裸词中出现 `\`
|
||||
- 空短语(`""`、`tag:""`);短语紧贴其他文本(`a"b"`、`"a"b`);`key:` 与短语值之间有空白(`tag: "x"`)
|
||||
- 未知限定符 key(合法 key 仅 tag / title / content / id / column)
|
||||
- `id:` 值不是十进制正整数(`id:abc`、`id:0`、`id:"61"`)
|
||||
- 错误对话框:模态、父窗口居中;文案 = 资源模板"查询表达式有误,详情:{0}"(i18n),{0} 处拼接解析器返回的**非 i18n 英文错误描述串**(对位置信息无要求)
|
||||
|
||||
Reference in new issue
Block a user