Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 1 addition & 31 deletions crates/tsv_debug/src/cli/commands/gap_audit_known.txt
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@
# the gate rather than being pinned.
#
# Format: KIND<TAB>SHAPE<TAB>PAYLOADS
# shapes: 570
# shapes: 540
DROPPED !)⟨⟩!. annotation,block,jsdoc_cast
DROPPED #⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
DROPPED #⟨⟩\ annotation,block,jsdoc_cast,line,multiline
Expand Down Expand Up @@ -61,8 +61,6 @@ DROPPED ()⟨⟩) annotation,block,jsdoc_cast,line,multiline
DROPPED ()⟨⟩)); annotation,block,jsdoc_cast,line,multiline
DROPPED ()⟨⟩); annotation,block,jsdoc_cast,line,multiline
DROPPED ()⟨⟩␣ annotation,block,jsdoc_cast,line,multiline
DROPPED (++⟨⟩IDENT line
DROPPED (--⟨⟩IDENT line
DROPPED (/)⟨⟩␣ annotation,block,jsdoc_cast
DROPPED (<⟨⟩IDENT annotation,block,jsdoc_cast,multiline
DROPPED ([(⟨⟩IDENT line
Expand Down Expand Up @@ -110,7 +108,6 @@ DROPPED )⟨⟩) annotation,block,jsdoc_cast,line,multiline
DROPPED )⟨⟩)); annotation,block,jsdoc_cast,line,multiline
DROPPED )⟨⟩); annotation,block,jsdoc_cast,line,multiline
DROPPED )⟨⟩)?] annotation,block,jsdoc_cast,line,multiline
DROPPED )⟨⟩++; annotation,block,jsdoc_cast
DROPPED )⟨⟩, annotation,block,jsdoc_cast,line,multiline
DROPPED )⟨⟩=> annotation,block,jsdoc_cast
DROPPED )⟨⟩=>( annotation,block,jsdoc_cast
Expand All @@ -131,12 +128,7 @@ DROPPED */,⟨⟩␣ annotation,block,jsdoc_cast
DROPPED */⟨⟩, annotation,block,jsdoc_cast,line,multiline
DROPPED */⟨⟩,/* line
DROPPED */⟨⟩/* line,multiline
DROPPED ++(⟨⟩IDENT line
DROPPED ++(⟨⟩␣ line
DROPPED ++⟨⟩( annotation,block,line,multiline
DROPPED ++⟨⟩); annotation,block,jsdoc_cast,multiline
DROPPED ++⟨⟩IDENT line
DROPPED ++⟨⟩␣ line
DROPPED +⟨⟩IDENT line
DROPPED +⟨⟩␣ annotation,block,jsdoc_cast,line,multiline
DROPPED ,(⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
Expand All @@ -157,9 +149,6 @@ DROPPED ,⟨⟩] annotation,block,jsdoc_cast,multiline
DROPPED ,⟨⟩]); multiline
DROPPED ,⟨⟩]; annotation,block,jsdoc_cast,multiline
DROPPED ,⟨⟩␣ annotation,block,jsdoc_cast,line,multiline
DROPPED --⟨⟩IDENT line
DROPPED --⟨⟩␣ line
DROPPED -⟨⟩NUM annotation,block,jsdoc_cast,line,multiline
DROPPED .#⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
DROPPED .()⟨⟩)!( annotation,block,jsdoc_cast,line,multiline
DROPPED .()⟨⟩)!` annotation,block,jsdoc_cast,line,multiline
Expand All @@ -171,16 +160,12 @@ DROPPED :⟨⟩{ annotation,block,jsdoc_cast,multiline
DROPPED :⟨⟩␣ annotation,block,jsdoc_cast,multiline
DROPPED ;#⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
DROPPED ;}⟨⟩} annotation,block,jsdoc_cast,line,multiline
DROPPED <-⟨⟩NUM annotation,block,jsdoc_cast,line,multiline
DROPPED <⟨⟩IDENT annotation,block,jsdoc_cast,multiline
DROPPED <⟨⟩const annotation,block,jsdoc_cast,multiline
DROPPED <⟨⟩␣ annotation,block,jsdoc_cast,multiline
DROPPED ="{⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
DROPPED =(<⟨⟩IDENT annotation,block,jsdoc_cast,multiline
DROPPED =(⟨⟩< line
DROPPED =++⟨⟩IDENT line
DROPPED =--⟨⟩IDENT line
DROPPED =-⟨⟩NUM annotation,block,jsdoc_cast,line,multiline
DROPPED =<⟨⟩IDENT annotation,block,jsdoc_cast,multiline
DROPPED =>(⟨⟩{ line
DROPPED =>(⟨⟩{}) line
Expand Down Expand Up @@ -236,25 +221,16 @@ DROPPED IDENT⟨⟩)!` annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)) annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)). annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)); annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)++ annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩): annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩); annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)=> annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩)} annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩+ annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩++ annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩++) annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩++, annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩++; annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩++} annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩, annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩,/* line,multiline
DROPPED IDENT⟨⟩,// line
DROPPED IDENT⟨⟩,>( annotation,block,jsdoc_cast,multiline
DROPPED IDENT⟨⟩,>/ annotation,block,jsdoc_cast,multiline
DROPPED IDENT⟨⟩-- annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩--) annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩--; annotation,block,jsdoc_cast
DROPPED IDENT⟨⟩/ annotation,block,jsdoc_cast,line,multiline
DROPPED IDENT⟨⟩/* line,multiline
DROPPED IDENT⟨⟩: annotation,block,jsdoc_cast,line,multiline
Expand Down Expand Up @@ -347,7 +323,6 @@ DROPPED ][⟨⟩]); annotation,block,jsdoc_cast,line,multiline
DROPPED ][⟨⟩]; annotation,block,jsdoc_cast,line,multiline
DROPPED ]⟨⟩) annotation,block,jsdoc_cast,line,multiline
DROPPED ]⟨⟩); annotation,block,jsdoc_cast,multiline
DROPPED ]⟨⟩++; annotation,block,jsdoc_cast
DROPPED ]⟨⟩/* line
DROPPED ]⟨⟩: annotation,block,jsdoc_cast,line,multiline
DROPPED ]⟨⟩[]) annotation,block,jsdoc_cast
Expand Down Expand Up @@ -388,7 +363,6 @@ DROPPED void⟨⟩␣ annotation,block,jsdoc_cast
DROPPED yield⟨⟩* annotation,block,jsdoc_cast,line,multiline
DROPPED {#⟨⟩IDENT annotation,block,jsdoc_cast,line,multiline
DROPPED {(<⟨⟩IDENT annotation,block,jsdoc_cast,multiline
DROPPED {++⟨⟩IDENT line
DROPPED {})⟨⟩); annotation,block,jsdoc_cast,multiline
DROPPED {},⟨⟩␣ annotation,block,jsdoc_cast,multiline
DROPPED {}⟨⟩); annotation,block,jsdoc_cast,line,multiline
Expand Down Expand Up @@ -442,7 +416,6 @@ DROPPED ␣⟨⟩) annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩)!` annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩)!} annotation,block,jsdoc_cast,multiline
DROPPED ␣⟨⟩)( annotation,block,jsdoc_cast,multiline
DROPPED ␣⟨⟩)++ annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩), annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩). annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩)// annotation,block,jsdoc_cast,line,multiline
Expand All @@ -452,13 +425,10 @@ DROPPED ␣⟨⟩)} annotation,block,jsdoc_cast,multiline
DROPPED ␣⟨⟩)}' annotation,block,jsdoc_cast,multiline
DROPPED ␣⟨⟩)}` annotation,block,jsdoc_cast,multiline
DROPPED ␣⟨⟩+ annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩++ annotation,block,jsdoc_cast
DROPPED ␣⟨⟩++; annotation,block,jsdoc_cast
DROPPED ␣⟨⟩, annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩,' annotation,block,jsdoc_cast
DROPPED ␣⟨⟩,/* annotation,block,jsdoc_cast
DROPPED ␣⟨⟩,]; annotation,block,jsdoc_cast
DROPPED ␣⟨⟩--; annotation,block,jsdoc_cast
DROPPED ␣⟨⟩... multiline
DROPPED ␣⟨⟩/ annotation,block,jsdoc_cast,line,multiline
DROPPED ␣⟨⟩/* annotation,block,jsdoc_cast,line,multiline
Expand Down
13 changes: 3 additions & 10 deletions crates/tsv_ts/src/parser/expression_lookahead.rs
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
//
// All functions operate on byte slices for performance (no tokenization needed).

use super::scan::{is_identifier_start, skip_identifier, skip_whitespace_and_comments};
use super::scan::{is_identifier_start, is_word_at, skip_identifier, skip_whitespace_and_comments};
use tsv_lang::source_scan::{TriviaProfile, is_regex_start, skip_regex_literal, skip_trivia};

/// `<` at `pos` is `<=` comparison operator, not an angle bracket open
Expand Down Expand Up @@ -430,17 +430,10 @@ fn paren_list_then_arrow(bytes: &[u8], paren: usize) -> bool {
/// past the `abstract` keyword.
pub(super) fn is_construct_type_start(bytes: &[u8], pos: usize) -> bool {
// Whole-word `new` (not an identifier like `newType`).
if !bytes[pos..].starts_with(b"new") {
if !is_word_at(bytes, pos, b"new") {
return false;
}
let after_new = pos + b"new".len();
if bytes
.get(after_new)
.is_some_and(|&b| b.is_ascii_alphanumeric() || b == b'_' || b == b'$')
{
return false;
}
let paren = skip_whitespace_and_comments(bytes, after_new);
let paren = skip_whitespace_and_comments(bytes, pos + b"new".len());
if bytes.get(paren) != Some(&b'(') {
return false;
}
Expand Down
137 changes: 92 additions & 45 deletions crates/tsv_ts/src/parser/expression_type_args.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,10 @@ use super::expression_lookahead::{
is_construct_type_start, is_function_type_start, is_generic_function_type_start,
scan_for_closing_angle_bracket,
};
use super::scan::{is_identifier_start, skip_identifier, skip_whitespace_and_comments};
use super::scan::{
is_identifier_start, is_word_at, skip_identifier, skip_numeric_literal,
skip_whitespace_and_comments,
};

impl<'a, 'arena> Parser<'a, 'arena> {
/// Check if current position starts type arguments: `<Type, ...>`
Expand Down Expand Up @@ -170,70 +173,82 @@ impl<'a, 'arena> Parser<'a, 'arena> {
// starting at `pos` is equivalent to starting past the separator.)
b'>' | b'<' | b',' | b'|' | b'&' => scan_for_closing_angle_bracket(bytes, pos),

// Indexed type vs array access: `T[K]` vs `arr[0]`
b'[' => self.check_indexed_type_pattern(bytes, pos),
// Indexed type vs array access: `T[K]` vs `arr[0]`. Confirmed by the same
// closing-`>` scan as the arms above — `T[K]` shaped bytes are equally a
// member access on a comparison's right operand, so only the matching `>`
// (and its follow token) tells them apart: `f(a < B[c], d)` and
// `a < B[c] > d` stay comparisons, `f<A[B], C>(x)` is an instantiation.
b'[' => {
self.check_indexed_type_pattern(bytes, pos)
&& scan_for_closing_angle_bracket(bytes, pos)
}

// Type constraint: `T extends U`
b'e' if bytes[pos..].starts_with(b"extends") => true,
// Type constraint: `T extends U`. Whole-word — an identifier that merely
// starts with `extends` is an ordinary operand (`a < b` ⏎ `extendsFoo()`,
// where ASI ends the statement) — and confirmed by the closing-`>` scan
// like every sibling arm.
b'e' if is_word_at(bytes, pos, b"extends") => {
scan_for_closing_angle_bracket(bytes, pos)
}

_ => false,
}
}

/// Check if `[` at `pos` starts an indexed type (not array access).
/// Whether the `[` at `pos` can open an indexed-access type rather than an array
/// index. A pre-filter only: every shape that stays grammatical both ways is handed
/// to the caller's closing-`>` scan, which arbitrates.
///
/// - `arr[0]`: numeric index → array access
/// - `arr[i]` followed by `<` or `;`: array access
/// - `T[K]` followed by `>` or `,`: indexed type
/// - `T["key"]`, `T[keyof U]`, `T[typeof x]`: indexed type
/// - `a[b - 1]`: complex expression → array access (default)
/// - `T[]`, `T["key"]`, `T[keyof U]`, `T[typeof x]`: indexed type
/// - `T[K]`, `T[0]`, `T[-1]` followed by `>`, `,`, or another `[`: indexed type
/// - `T[A | B]`, `T[0 | 1]`, `T[A[B]]`, `T[A.B]`, `T[A<B>]`,
/// `T[A extends B ? C : D]`: the index is itself a type, so the scan decides
/// - `a[b - 1]`, `a[0 + 1]`, `a[c || d]`, `a[c <= d]`: arithmetic, or a
/// logical/shift/relational operator — an expression, never a type → array access
/// - `a[-b]`: a unary negation, not a negative literal → array access
fn check_indexed_type_pattern(&self, bytes: &[u8], pos: usize) -> bool {
let inside = skip_whitespace_and_comments(bytes, pos + 1);
if inside >= bytes.len() {
let Some(&first) = bytes.get(inside) else {
return false;
}
};

// Empty brackets `T[]` — array type
if bytes[inside] == b']' {
return true;
}
match first {
// Empty brackets `T[]` — array type
b']' => true,

// Numeric index is definitely array access
if bytes[inside].is_ascii_digit() {
return false;
}
// Numeric literal index: `T[0]`, `T[-1]`, `T[0 | 1]`. A numeric literal is as
// valid a type as it is an array index, so the literal alone decides nothing —
// what FOLLOWS it does, under the same rule a reference index answers to.
b'0'..=b'9' | b'-' => {
let after_number = skip_numeric_literal(bytes, inside);
// No literal starts here at all (`-b`) — a unary negation, so the index is
// an expression. Guarding on this is what stops the `-` from swallowing an
// identifier and landing on the same `]` a real literal ends at.
if after_number == inside {
return false;
}
continues_as_type(bytes, skip_whitespace_and_comments(bytes, after_number))
}

// Identifier index: check for type keywords then what follows `]`
if is_identifier_start(bytes[inside]) {
let after_id = skip_identifier(bytes, inside);
// String literal key: `T["key"]`, `T['key']` — indexed access type
b'\'' | b'"' | b'`' => true,

// Type operator keywords: `T[keyof U]`, `T[typeof x]`
let kw = &bytes[inside..after_id];
if kw == b"keyof" || kw == b"typeof" {
return true;
}
// Identifier index: check for type keywords then what follows the identifier
_ if is_identifier_start(first) => {
let after_id = skip_identifier(bytes, inside);

let after_bracket = skip_whitespace_and_comments(bytes, after_id);
if after_bracket < bytes.len() && bytes[after_bracket] == b']' {
let after_close = skip_whitespace_and_comments(bytes, after_bracket + 1);
// Type args end with `>` or continue with `,`
if after_close < bytes.len() && matches!(bytes[after_close], b'>' | b',') {
// Type operator keywords: `T[keyof U]`, `T[typeof x]`
let kw = &bytes[inside..after_id];
if kw == b"keyof" || kw == b"typeof" {
return true;
}
return false;

continues_as_type(bytes, skip_whitespace_and_comments(bytes, after_id))
}
// Identifier followed by something other than `]` (e.g., `b - 1]`)
// is a complex expression — array access, not indexed type
return false;
}

// String literal key: `T["key"]`, `T['key']` — indexed access type
if matches!(bytes[inside], b'\'' | b'"' | b'`') {
return true;
// Unknown pattern — default to NOT type args (safer for JS expressions)
_ => false,
}

// Unknown pattern — default to NOT type args (safer for JS expressions)
false
}

/// Check if position points to a TypeScript type keyword.
Expand Down Expand Up @@ -273,3 +288,35 @@ impl<'a, 'arena> Parser<'a, 'arena> {
}
}
}

/// Whether the byte at `after_operand` — the first non-trivia byte past an index operand —
/// continues a TYPE rather than an expression.
///
/// Shared by both operand kinds, and it must stay shared: `T[K | J]` and `T[0 | 1]` are
/// the same question, and answering it in one place is what keeps a numeric index from
/// being read more narrowly than a reference one.
fn continues_as_type(bytes: &[u8], after_operand: usize) -> bool {
match bytes.get(after_operand) {
Some(b']') => {
let after_close = skip_whitespace_and_comments(bytes, after_operand + 1);
// Type args end with `>`, continue with `,`, or chain another index group
// (`T[K][J]`) — the caller's closing-`>` scan arbitrates all three
matches!(bytes.get(after_close), Some(b'>' | b',' | b'['))
}
// `||` and `&&` are logical operators, so the index is an expression — only the
// single `|`/`&` are type operators (as in the caller's own arm). Likewise `<<` is
// a shift and `<=` a comparison, neither a type's `<`.
Some(b'|' | b'&') if bytes.get(after_operand + 1) == bytes.get(after_operand) => false,
Some(b'<') if matches!(bytes.get(after_operand + 1), Some(b'<' | b'=')) => false,
// Type-continuation tokens: the index is a union or intersection (`T[A | B]`), a
// nested index (`T[A[B]]`), a qualified name (`T[A.B]`), a generic reference
// (`T[A<B>]`), or a conditional (`T[A extends B ? C : D]`). None of these can be
// arithmetic, so hand the decision to the caller's closing-`>` scan.
Some(b'|' | b'&' | b'[' | b'.' | b'<') => true,
Some(b'e') if is_word_at(bytes, after_operand, b"extends") => true,
// Anything else after the operand (e.g. `b - 1]`) is arithmetic — an expression,
// never a type, and the one case the closing-`>` scan cannot arbitrate
// (`a < arr[b - 1] > (c)` is grammatical both ways).
_ => false,
}
}
51 changes: 51 additions & 0 deletions crates/tsv_ts/src/parser/scan.rs
Original file line number Diff line number Diff line change
Expand Up @@ -96,6 +96,57 @@ pub(super) fn is_identifier_continue(b: u8) -> bool {
b.is_ascii_alphanumeric() || b == b'_' || b == b'$' || b > 127
}

/// Check if `word` sits at `pos` as a whole identifier, not as the prefix of a longer
/// one — `is_word_at(b"extendsFoo", 0, b"extends")` is false.
///
/// A bare `starts_with` is the trap this exists to close: the byte after the word decides
/// whether a lookahead is looking at a keyword or at an ordinary identifier that happens
/// to share its opening bytes.
#[inline]
pub(super) fn is_word_at(bytes: &[u8], pos: usize, word: &[u8]) -> bool {
bytes[pos..].starts_with(word)
&& bytes
.get(pos + word.len())
.is_none_or(|&b| !is_identifier_continue(b))
}

/// Skip a numeric literal, returning the position after it — or `pos` unchanged when no
/// literal starts there. Handles a leading `-` (the only sign a literal *type* may carry —
/// `A[+1]` is not one), radix prefixes (`0x`), separators (`1_000`), a BigInt `n`, and an
/// exponent whose own sign (`1e-3`) would otherwise end the scan.
///
/// Deliberately loose about a literal's INTERIOR — it accepts more than the grammar does.
/// Callers use it to find where a literal ENDS, then check what follows, so over-consuming
/// a malformed literal only makes that follow-check fail. The FIRST character is the one
/// place it is strict, and must stay so: `-b` would otherwise scan as a literal ending at
/// the very `]` a real one ends at, leaving the follow-check no way to tell a negated
/// identifier from a negative number.
#[inline]
pub(super) fn skip_numeric_literal(bytes: &[u8], pos: usize) -> usize {
let mut cursor = pos;
if bytes.get(cursor) == Some(&b'-') {
cursor += 1;
}
// A literal starts with a digit, or a bare `.` for `-.5`. Anything else (`-b`, `-$b`)
// is a unary negation; reporting "nothing skipped" is how the caller learns that.
if !matches!(bytes.get(cursor), Some(b'0'..=b'9' | b'.')) {
return pos;
}
while cursor < bytes.len() {
let b = bytes[cursor];
if b.is_ascii_alphanumeric() || b == b'.' || b == b'_' {
// An exponent's sign belongs to the literal, not to a following operator.
if matches!(b, b'e' | b'E') && matches!(bytes.get(cursor + 1), Some(b'-' | b'+')) {
cursor += 1;
}
cursor += 1;
} else {
break;
}
}
cursor
}

/// Skip an identifier, returning position after the identifier
/// Assumes `pos` is at the start of an identifier
#[inline]
Expand Down
Loading