From aae80d330e58012f3088c3cd4e58ea1c8aa4a161 Mon Sep 17 00:00:00 2001 From: A Tobey Date: Tue, 8 Sep 2026 10:05:57 -0400 Subject: [PATCH] fix: classify complete literal words consistently The lexer classified prefixes of ordinary literal words as numbers, keywords, or flags, then rejected the pieces as adjacent arguments. Recognize numeric filenames and version strings as whole words, accept embedded plus signs in word classes, and let numeric and keyword prefixes participate in existing colon and glob fusion. These words now retain their complete spelling: ```sh echo 123.txt 1.2.3 .123.txt true:foo do:foo a+b 123+b 123@host echo ~/a:b ls 1.0* true* ``` A numeric filename can exceed the integer range because it is text. Complete scalar numbers still follow the existing numeric rules: incomplete floats and integer overflow remain errors, leading zeros remain text, and ordinary integers and floats keep their types. Glob fusion slices the original source, so numeric formatting cannot change which filenames match. Standalone keywords, booleans, plus-prefixed flags, collection delimiters, and quote-to-join checks retain their roles. Validation: numeric and keyword regression suites failed before the fix; the final tests check exact token spans, argument values, assignment and redirect values, actual glob matches with misleading numeric-prefix controls, tilde expansion, and control-flow/record behavior. Workspace tests, clippy across all targets with warnings denied, and rustdoc with warnings denied passed. Help fragments and generated syntax are synchronized. Reviewed through kaibo's default cast, GLM-5.2 with DeepSeek V4 Flash investigation. Review exposed the numeric prefix followed directly by punctuation; it is fixed with failing-then-passing regression tests rather than preserved as another word-character exception. This layer builds on #438 in the native stack rooted at #437. Tilde assignment and its regex-operator collision are a separate layer. --- CHANGELOG.md | 7 ++ crates/kaish-help/content/en/syntax.md | 4 +- crates/kaish-help/src/fragments.rs | 4 +- crates/kaish-kernel/src/lexer.rs | 21 ++-- .../tests/keyword_literal_words_tests.rs | 54 ++++++++++ .../tests/numeric_word_boundaries_tests.rs | 100 ++++++++++++++++++ docs/LANGUAGE.md | 2 + 7 files changed, 181 insertions(+), 11 deletions(-) create mode 100644 crates/kaish-kernel/tests/keyword_literal_words_tests.rs create mode 100644 crates/kaish-kernel/tests/numeric_word_boundaries_tests.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index a71cdbe2..5cd8bcda 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,13 @@ breaking entries are marked **BREAKING**. ### Fixed +- Numeric filenames and versions such as `123.txt` and `1.2.3` now stay + literal words. Float-prefixed globs such as `1.0*` preserve their exact + spelling and match filenames without changing scalar number rules. +- Keywords inside colon words and globs (`true:foo`, `do*`) now remain + literal text. Embedded `+` and tilde paths containing `:` are accepted + consistently; standalone keywords and plus-prefixed flags keep their meaning. + - For-loop items now require whitespace between them. Quote a whole word to join text with interpolation; adjacent fragments are refused before the substitution or loop body runs, instead of becoming separate iterations. diff --git a/crates/kaish-help/content/en/syntax.md b/crates/kaish-help/content/en/syntax.md index 92cf3af3..1431b8e1 100644 --- a/crates/kaish-help/content/en/syntax.md +++ b/crates/kaish-help/content/en/syntax.md @@ -172,7 +172,9 @@ echo "/tmp/$(id -u).sock" # one argument echo $dir/file.txt # error — quote "$dir/file.txt" echo /tmp/$(id -u).sock # error — quote "/tmp/$(id -u).sock" cmd > $dir/out.txt # error — quote "$dir/out.txt" -# Literal words like file.txt, v1.2.3, and .git/HEAD need no quotes. +# Literal words keep their complete spelling, without quotes: +echo 123.txt 1.2.3 true:foo a+b .git/HEAD +ls 1.0* # glob keeps the written numeric prefix ``` ## Comments diff --git a/crates/kaish-help/src/fragments.rs b/crates/kaish-help/src/fragments.rs index 1df2c145..d4370adc 100644 --- a/crates/kaish-help/src/fragments.rs +++ b/crates/kaish-help/src/fragments.rs @@ -471,7 +471,9 @@ echo "/tmp/$(id -u).sock" # one argument echo $dir/file.txt # error — quote "$dir/file.txt" echo /tmp/$(id -u).sock # error — quote "/tmp/$(id -u).sock" cmd > $dir/out.txt # error — quote "$dir/out.txt" -# Literal words like file.txt, v1.2.3, and .git/HEAD need no quotes. +# Literal words keep their complete spelling, without quotes: +echo 123.txt 1.2.3 true:foo a+b .git/HEAD +ls 1.0* # glob keeps the written numeric prefix ```"#, ), syntax_section( diff --git a/crates/kaish-kernel/src/lexer.rs b/crates/kaish-kernel/src/lexer.rs index b3a0200a..148acc32 100644 --- a/crates/kaish-kernel/src/lexer.rs +++ b/crates/kaish-kernel/src/lexer.rs @@ -458,7 +458,8 @@ pub enum Token { /// (the POSIX `.` source alias) which only matches a bare `.` — the source /// alias requires whitespace before its file argument (`. script`), so /// `.parent` (no space) is unambiguously a single bareword. - #[regex(r"\.[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.#\-\u{80}-\u{10FFFF}]*", lex_dotted_ident, priority = 3)] + #[regex(r"\.[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.@+#\-\u{80}-\u{10FFFF}]*", lex_dotted_ident, priority = 3)] + #[regex(r"\.[0-9]+\.[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.@+#\-\u{80}-\u{10FFFF}]*", lex_dotted_ident, priority = 3)] DottedIdent(String), #[token("{")] @@ -684,10 +685,10 @@ pub enum Token { // ═══════════════════════════════════════════════════════════════════ /// Digit-leading bareword: `019dda1c` (SHA prefix), UUIDs, version-ish - /// strings. Distinguished from `Int` because at least one alpha character - /// follows the leading digits — the lexer commits to "this is a string, - /// not a number." Treated as a bareword string in expression position. - #[regex(r"[0-9]+[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.#\-\u{80}-\u{10FFFF}]*", lex_number_ident, priority = 3)] + /// strings and numeric filenames (`123.txt`, `1.2.3`). A nonnumeric + /// suffix or multiple dot-separated numeric components makes the whole + /// word text. Complete scalar numerals retain their numeric rules. + #[regex(r"[0-9]+(\.[0-9]+)*\.?[a-zA-Z_+@\u{80}-\u{10FFFF}][a-zA-Z0-9_.@+#\-\u{80}-\u{10FFFF}]*|[0-9]+(\.[0-9]+){2,}[a-zA-Z0-9_.@+#\-\u{80}-\u{10FFFF}]*", lex_number_ident, priority = 3)] NumberIdent(String), /// Numeric word containing an embedded hyphen run, or a minus-led numeric @@ -738,7 +739,7 @@ pub enum Token { /// `a@b.com` (bare `@` is an ordinary word character, as in bash). The /// leading class excludes digits — `NumberIdent`/`Int` own digit-leading /// words — and the ASCII operator/whitespace set. - #[regex(r"[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.@#\-\u{80}-\u{10FFFF}]*", lex_ident)] + #[regex(r"[a-zA-Z_\u{80}-\u{10FFFF}][a-zA-Z0-9_.@+#\-\u{80}-\u{10FFFF}]*", lex_ident)] Ident(String), // ═══════════════════════════════════════════════════════════════════ @@ -3171,7 +3172,7 @@ fn compute_value_context(tokens: &[Spanned]) -> Vec { /// True for token types that can participate in colon-adjacent merging. fn is_colon_mergeable(token: &Token) -> bool { - matches!( + token.is_keyword() || token.is_type() || matches!( token, Token::Ident(_) | Token::NumberIdent(_) @@ -3179,6 +3180,7 @@ fn is_colon_mergeable(token: &Token) -> bool { | Token::AtWord(_) | Token::DottedIdent(_) | Token::Colon + | Token::TildePath(_) | Token::Int(_) | Token::RelativePath(_) | Token::DotSlashPath(_) @@ -3272,7 +3274,7 @@ fn flush_colon_run( /// True for token types that can participate in a glob word. fn is_glob_mergeable(token: &Token) -> bool { - matches!( + token.is_keyword() || token.is_type() || matches!( token, Token::Star | Token::Question @@ -3285,6 +3287,7 @@ fn is_glob_mergeable(token: &Token) -> bool { | Token::DottedIdent(_) | Token::Path(_) | Token::Int(_) + | Token::Float(_) | Token::LBracket | Token::RBracket | Token::Bang @@ -3739,7 +3742,7 @@ pub(crate) fn is_leading_zero_numeral(word: &str) -> bool { /// /// Runs as the LAST step of `tokenize_impl`, after every fusion pass: /// `is_colon_mergeable` matches `Int` and `Float` directly and -/// `is_glob_mergeable` matches `Int`, so a numeral must still present its +/// `is_glob_mergeable` matches both too, so a numeral must still present its /// ordinary shape while fusion decides. /// Spans are original-source coordinates by now, so `source[span]` is the /// exact word the author typed. diff --git a/crates/kaish-kernel/tests/keyword_literal_words_tests.rs b/crates/kaish-kernel/tests/keyword_literal_words_tests.rs new file mode 100644 index 00000000..65f49065 --- /dev/null +++ b/crates/kaish-kernel/tests/keyword_literal_words_tests.rs @@ -0,0 +1,54 @@ +//! Reserved words remain reserved only when they occupy the whole word. +#![allow(clippy::unwrap_used, clippy::expect_used)] +use kaish_kernel::ast::{Arg, Expr, Stmt, Value}; +use kaish_kernel::lexer::{tokenize, Token}; +use kaish_kernel::parser::parse; +use kaish_kernel::{Kernel, KernelConfig}; +use rstest::rstest; + +#[rstest] +#[case("true:foo")] +#[case("false:foo")] +#[case("do:foo")] +#[case("for:foo")] +#[case("string:foo")] +#[case("~/a:b")] +#[case("a+b")] +#[case("a+b+c")] +#[case(".a+b")] +#[case("123a+b")] +fn literal_word_keeps_one_argument(#[case] word: &str) { + let program = parse(&format!("echo {word}")).unwrap(); + let Stmt::Command(command) = &program.statements[0] else { panic!("{program:?}") }; + assert_eq!(command.args, vec![Arg::Positional(Expr::Literal(Value::String(word.into())))]); +} + +#[rstest] +#[case("true*")] +#[case("do?")] +#[case("false[ab]")] +fn keyword_prefixed_pattern_is_a_glob(#[case] word: &str) { + let tokens = tokenize(word).unwrap(); + assert_eq!(tokens.len(), 1, "{tokens:?}"); + assert_eq!(tokens[0].token, Token::GlobWord(word.into())); + parse(&format!("echo {word}")).unwrap(); +} + +#[tokio::test] +async fn whole_keywords_records_and_plus_flags_keep_their_meaning() { + let kernel = Kernel::new(KernelConfig::isolated()).unwrap(); + let result = kernel.execute("set +e; x={enabled:true,disabled:false}; if true; then for item in do:foo true:foo; do echo $item; done; fi; echo ${x[enabled]} ${x[disabled]}").await.unwrap(); + assert!(result.ok(), "{result:?}"); + assert_eq!(result.text_out(), "do:foo\ntrue:foo\ntrue false\n"); + assert_eq!(tokenize("+e").unwrap()[0].token, Token::PlusFlag("e".into())); + assert_eq!(tokenize("true").unwrap()[0].token, Token::True); +} + +#[tokio::test] +async fn colon_tilde_path_expands_home_and_keyword_glob_matches_files() { + let kernel = Kernel::new(KernelConfig::isolated().with_initial_vars( + [("HOME".into(), Value::String("/home/fixture".into()))].into())).unwrap(); + let result = kernel.execute("echo ~/a:b; touch /true-one /true-two /other; cd /; echo true*").await.unwrap(); + assert!(result.ok(), "{result:?}"); + assert_eq!(result.text_out(), "/home/fixture/a:b\ntrue-one true-two\n"); +} diff --git a/crates/kaish-kernel/tests/numeric_word_boundaries_tests.rs b/crates/kaish-kernel/tests/numeric_word_boundaries_tests.rs new file mode 100644 index 00000000..e694ca9a --- /dev/null +++ b/crates/kaish-kernel/tests/numeric_word_boundaries_tests.rs @@ -0,0 +1,100 @@ +//! Numeric prefixes do not turn a literal filename into a malformed number. +#![allow(clippy::unwrap_used, clippy::expect_used)] +use kaish_kernel::ast::{Arg, Expr, PipelineStage, Stmt, Value}; +use kaish_kernel::lexer::{tokenize, NumericLiteralData, Token}; +use kaish_kernel::parser::parse; +use kaish_kernel::{Kernel, KernelConfig}; +use rstest::rstest; + +#[rstest] +#[case("123.txt")] +#[case("123+b")] +#[case("123@host")] +#[case("1.0+2")] +#[case("1.2.3")] +#[case("1.0.txt")] +#[case("007.0644.txt")] +#[case("1.2rc1")] +#[case("1.2.3-rc1")] +#[case("123.日本語")] +#[case("9223372036854775808.txt")] +#[case(".123.txt")] +fn numeric_prefixed_word_keeps_its_exact_text(#[case] word: &str) { + let tokens = tokenize(word).unwrap(); + assert_eq!(tokens.len(), 1, "{word}: {tokens:?}"); + assert_eq!(tokens[0].span, 0..word.len()); + let program = parse(&format!("echo {word}")).unwrap(); + let Stmt::Command(command) = &program.statements[0] else { panic!("{program:?}") }; + assert_eq!(command.args, vec![Arg::Positional(Expr::Literal(Value::String(word.into())))]); + let program = parse(&format!("p={word}")).unwrap(); + let Stmt::Assignment(assignment) = &program.statements[0] else { panic!("{program:?}") }; + assert_eq!(assignment.value, Expr::Literal(Value::String(word.into()))); + let program = parse(&format!("cat <{word}")).unwrap(); + let Stmt::Pipeline(pipeline) = &program.statements[0] else { panic!("{program:?}") }; + let PipelineStage::Command(command) = &pipeline.stages[0] else { panic!("{pipeline:?}") }; + assert_eq!(command.redirects[0].target, Expr::Literal(Value::String(word.into()))); +} + +#[rstest] +#[case("1.0*")] +#[case("007.00*")] +#[case("1.0?")] +#[case("1.0[ab]")] +fn float_prefixed_glob_keeps_its_source(#[case] word: &str) { + let tokens = tokenize(word).unwrap(); + assert_eq!(tokens.len(), 1, "{tokens:?}"); + assert_eq!(tokens[0].token, Token::GlobWord(word.into())); + assert_eq!(tokens[0].span, 0..word.len()); + parse(&format!("echo {word}")).unwrap(); +} + +#[tokio::test] +async fn glob_matches_the_written_prefix_and_quotes_keep_it_literal() { + let kernel = Kernel::new(KernelConfig::isolated()).unwrap(); + let result = kernel.execute("touch /1.0a /1.0b /1a /007.00a /7a; cd /").await.unwrap(); + assert!(result.ok(), "{result:?}"); + for (source, expected) in [("echo 1.0*", "1.0a 1.0b\n"), ("echo 007.00*", "007.00a\n"), ("echo \"1.0*\"", "1.0*\n")] { + let result = kernel.execute(source).await.unwrap(); + assert!(result.ok(), "{result:?}"); + assert_eq!(result.text_out(), expected); + } +} + +#[test] +fn scalar_number_contract_is_unchanged() { + for word in [".5", "5.", "9223372036854775808"] { + assert!(tokenize(word).is_err(), "{word} must remain invalid"); + } + assert_eq!(tokenize("1.0").unwrap()[0].token, Token::NumericLiteral(NumericLiteralData { + value: Value::Float(1.0), raw: "1.0".into(), + })); + assert_eq!(tokenize("123").unwrap()[0].token, Token::Int(123)); + assert_eq!(tokenize("1.25").unwrap()[0].token, Token::Float(1.25)); + assert_eq!(tokenize("007").unwrap()[0].token, Token::NumberIdent("007".into())); + parse("x=[1.0,2.0]; y={version:1.2.3}").unwrap(); + assert!(parse("echo 1.2.3$tag").is_err()); +} + +#[rstest] +#[case("echo 123.txt$x")] +#[case("echo true:foo$x")] +#[case("echo a+b$x")] +#[case("echo 1.0*$x")] +#[case("for x in 123.txt$tag; do echo $x; done")] +#[case("cat <123.txt$x")] +fn literal_word_still_cannot_join_an_expansion(#[case] source: &str) { + assert!(parse(source).is_err(), "{source}"); +} + +#[rstest] +#[case("123.txt", "123.txt")] +#[case("true:foo", "true:foo")] +#[case("+e", "+e")] +#[case("1.0a", "1.0*")] +#[tokio::test] +async fn literal_words_and_globs_work_as_case_patterns(#[case] word: &str, #[case] pattern: &str) { + let kernel = Kernel::new(KernelConfig::isolated()).unwrap(); + let result = kernel.execute(&format!("case '{word}' in {pattern}) echo match ;; *) echo miss ;; esac")).await.unwrap(); + assert!(result.ok(), "{result:?}"); + assert_eq!(result.text_out(), "match\n"); +} diff --git a/docs/LANGUAGE.md b/docs/LANGUAGE.md index c0c1026c..737e55a9 100644 --- a/docs/LANGUAGE.md +++ b/docs/LANGUAGE.md @@ -408,6 +408,8 @@ punctuation outside that set is still not a word character: `echo 100%` is an error, as it was before.) ```sh +echo 123.txt 1.2.3 true:foo a+b # complete literal words +ls 1.0* # glob keeps the written numeric prefix echo café cd ~/文書 ls /tmp/日本語