All files / utils ParserUtils.ts

97.87% Statements 46/47
97.95% Branches 48/49
100% Functions 8/8
100% Lines 46/46

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274                                                            21250x                                                                               24725x 24725x   24725x 3x     24722x 24725x 24725x                                                                               2543x                                         24722x 24716x   6x                             16x 16x 4x     12x 12x 4x     8x 8x 8x 1x     7x 7x 2x     5x                                       1182x 1176x     6x 6x 1x     5x 5x 5x     5x 1x       4x 4x                               1982x 1982x 1982x                                                 322x 322x 980x 329x     322x          
/**
 * Utility class for extracting information from parser contexts.
 *
 * Centralizes common patterns for getting source positions from ANTLR
 * parser contexts, providing consistent null handling across the codebase.
 */
 
import { ParserRuleContext, TerminalNode } from "antlr4ng";
 
import * as Parser from "../PARSE/2-Parse/grammar/CNextParser";
import ISourcePosition from "./types/ISourcePosition";
import type ISourceSpan from "../types/ISourceSpan";
 
/**
 * Static utility methods for parser context operations
 */
class ParserUtils {
  /**
   * Extract source position from a parser context.
   *
   * Handles null/undefined start tokens gracefully, returning 0 for
   * missing values. This is the standard pattern used throughout C-Next
   * for error reporting.
   *
   * @param ctx - Any parser context with a start token
   * @returns Position with line and column (defaults to 0 if unavailable)
   */
  static getPosition(ctx: {
    start?: { line?: number; column?: number } | null;
  }): ISourcePosition {
    return {
      line: ctx.start?.line ?? 0,
      column: ctx.start?.column ?? 0,
    };
  }
 
  /**
   * Extract a full source span from a parser context.
   *
   * Takes the context STRUCTURALLY rather than as an ANTLR type, the same as
   * `getPosition` above. It was not a style choice: `transpiler/data/` could
   * import `utils/`, and `data-cannot-import-logic` was `reachable: true`, so an
   * ANTLR import here would have let `data/ -> utils/ -> logic/parser/` fail the
   * layer gate from a module that never mentions the parser (#1297). #1444
   * moved both layers into 1.1 Discover and retired the rule. The structural
   * type stays because it costs nothing and needs no grammar import.
   *
   * ANTLR's `stop` token is the LAST token of the rule, and its `column` is
   * where that token BEGINS. The exclusive end is therefore its column plus its
   * own width -- taking `stop.column` directly would underline every declaration
   * one token short, which reads as correct on a single-character final token
   * and is wrong everywhere else.
   *
   * A context with no `stop` (an error node, mid-recovery) yields a zero-width
   * span at `start`, so a caller always gets a well-ordered span and never has
   * to test for half-populated positions.
   *
   * @param ctx - Any parser context with start and stop tokens
   * @returns The span, defaulting to 0 for anything unavailable
   */
  static getSpan(ctx: {
    start?: { line?: number; column?: number } | null;
    stop?: {
      line?: number;
      column?: number;
      text?: string | null;
      start?: number;
      stop?: number;
    } | null;
  }): ISourceSpan {
    const line = ctx.start?.line ?? 0;
    const column = ctx.start?.column ?? 0;
 
    if (!ctx.stop) {
      return { line, column, endLine: line, endColumn: column };
    }
 
    const endLine = ctx.stop.line ?? line;
    const stopColumn = ctx.stop.column ?? column;
    return {
      line,
      column,
      endLine,
      endColumn: stopColumn + ParserUtils.tokenWidth(ctx.stop),
    };
  }
 
  /**
   * A context's span, or `fallback` when the context has no position at all.
   *
   * The single owner of one question: *what span does a member with no start
   * token get?* #1318 answered it twice and differently -- the C enum collector
   * inherited the enclosing enum's span, `MemberSymbolBase` called `getSpan`
   * unconditionally and produced `0:0` -- in the same change that existed to
   * stop members carrying four different construction paths.
   *
   * Inheriting is the right answer, and the C collector's reason is the one
   * that generalizes: a diagnostic aimed at the top of the file is worse than
   * one aimed at the enclosing declaration. `0:0` is also indistinguishable
   * from `UNSET_SOURCE_SPAN`, so it would read as "position not yet collected"
   * to anything checking that sentinel.
   *
   * Reachable only through error recovery, where a context can hold an error
   * node with no start token -- which is exactly when a diagnostic is being
   * produced and its position matters most.
   */
  static getSpanOr(
    ctx: {
      start?: { line?: number; column?: number } | null;
      stop?: {
        line?: number;
        column?: number;
        text?: string | null;
        start?: number;
        stop?: number;
      } | null;
    },
    fallback: ISourceSpan,
  ): ISourceSpan {
    return ctx.start ? ParserUtils.getSpan(ctx) : fallback;
  }
 
  /**
   * How many characters the stop token actually occupies.
   *
   * Prefers the token's character offsets over `text.length`, because the two
   * disagree on the one token every file ends with: ANTLR's EOF reports
   * `text === "<EOF>"` -- five characters -- while occupying zero, with
   * `stop === start - 1`. Measured on `enum EColor { RED }`: `text.length` gives
   * 5 and `stop - start + 1` gives 0, so a context ending at EOF was reporting an
   * `endColumn` five past the end of the file. Reachable under error recovery.
   *
   * Falls back to `text.length` when offsets are absent, which is the case for
   * the structural literals unit tests construct.
   */
  private static tokenWidth(stop: {
    text?: string | null;
    start?: number;
    stop?: number;
  }): number {
    if (typeof stop.start === "number" && typeof stop.stop === "number") {
      return Math.max(0, stop.stop - stop.start + 1);
    }
    return stop.text?.length ?? 0;
  }
 
  /**
   * Parse a "line:column message" prefix from an error message.
   *
   * CodeGenerator validation errors embed location as "line:col message".
   * This extracts the location and returns the clean message, or defaults
   * to line 1, column 0 if no prefix is found.
   */
  static parseErrorLocation(message: string): {
    line: number;
    column: number;
    message: string;
  } {
    const colonIdx = message.indexOf(":");
    if (colonIdx < 1) {
      return { line: 1, column: 0, message };
    }
 
    const lineStr = message.substring(0, colonIdx);
    if (!/^\d+$/.test(lineStr)) {
      return { line: 1, column: 0, message };
    }
 
    const afterColon = message.substring(colonIdx + 1);
    const spaceIdx = afterColon.indexOf(" ");
    if (spaceIdx < 1) {
      return { line: 1, column: 0, message };
    }
 
    const colStr = afterColon.substring(0, spaceIdx);
    if (!/^\d+$/.test(colStr)) {
      return { line: 1, column: 0, message };
    }
 
    return {
      line: Number.parseInt(lineStr, 10),
      column: Number.parseInt(colStr, 10),
      message: afterColon.substring(spaceIdx + 1),
    };
  }
  /**
   * Whether this is the main function with its command-line args parameter
   * -- the one place a trailing `[]` on a parameter is the language's own
   * form (ADR-030), lowered to `int main(int argc, char *argv[])`.
   * Supports: u8 args[][] (legacy) or string args[] (preferred)
   *
   * @param name - Function name
   * @param paramList - Parameter list context
   * @returns true if this is main with args parameter
   */
  static isMainFunctionWithArgs(
    name: string,
    paramList: Parser.ParameterListContext | null,
  ): boolean {
    if (name !== "main" || !paramList) {
      return false;
    }
 
    const params = paramList.parameter();
    if (params.length !== 1) {
      return false;
    }
 
    const param = params[0];
    const typeCtx = param.type();
    const dims = param.arrayDimension();
 
    // Check for string args[] (preferred - array of strings)
    if (typeCtx.stringType() && dims.length === 1) {
      return true;
    }
 
    // Check for u8 args[][] (legacy - 2D array of bytes)
    const type = typeCtx.getText();
    return (type === "u8" || type === "i8") && dims.length === 2;
  }
 
  /**
   * The two VALUE arms of a real ternary, or null for anything else.
   *
   * A ternary's value is one of its arms; its condition is a separate
   * expression that is never a value operand of the enclosing operator.
   * Addressed through `orExpression()`, never child indices: the condition is
   * parenthesized, so child 0 is `(` (CLAUDE.md). #1668: 2.1 and 2.2 each
   * picked the arms by hand, and E0810 did not skip the condition at all, so
   * `((i > 0) ? k : k) * i` read as internally mixed and went unchecked.
   */
  static ternaryValueArms(
    node: ParserRuleContext,
  ): [Parser.OrExpressionContext, Parser.OrExpressionContext] | null {
    Iif (!(node instanceof Parser.TernaryExpressionContext)) return null;
    const branches = node.orExpression();
    return branches.length === 3 ? [branches[1], branches[2]] : null;
  }
 
  /**
   * Extract operators from parse tree children in order.
   *
   * When parsing expressions like "a + b - c", ANTLR creates children
   * with operands interleaved: [a, +, b, -, c]. This method extracts
   * just the operators as terminal nodes.
   *
   * Note: Using children.filter() loses operator ordering when operators
   * are detected using text.includes(), so we iterate explicitly.
   *
   * #1445: moved here from `CodegenParserUtils`, which held this one method
   * and existed because it was "separated from src/utils/ParserUtils.ts to
   * avoid circular dependencies". No cycle is possible: it imported `antlr4ng`
   * and nothing else. So the module was a module for a reason that had stopped
   * being true, and merging it is a deletion rather than a relocation -- the
   * render layer still walks the tree here, through its orchestrator, exactly
   * as before.
   *
   * @param ctx - The parser rule context containing operands and operators
   * @returns Array of operator strings in the order they appear
   */
  static getOperatorsFromChildren(ctx: ParserRuleContext): string[] {
    const operators: string[] = [];
    for (const child of ctx.children) {
      if (child instanceof TerminalNode) {
        operators.push(child.getText());
      }
    }
    return operators;
  }
}
 
export default ParserUtils;