|
|
@@ -0,0 +1,146 @@
|
|
|
+/** Best-effort semantic checks over parsed Modula-2 units.
|
|
|
+ *
|
|
|
+ * Works on the lenient declaration parser: statement and expression bodies
|
|
|
+ * are not parsed, so checks are limited to declaration structure and
|
|
|
+ * identifier usage. The guiding rule is zero false positives — uncertain
|
|
|
+ * cases are skipped, never flagged:
|
|
|
+ * - qualified names (`M.X`, `r.field`) are never flagged;
|
|
|
+ * - identifiers inside `WITH` regions are skipped (unqualified field access);
|
|
|
+ * - FROM-imported names are skipped when the defining `.def` is unavailable;
|
|
|
+ * - predefined identifiers (`INTEGER`, `INC`, `NIL`, ...) are always accepted.
|
|
|
+ *
|
|
|
+ * Full type checking (assignment compatibility, call arity) needs statement
|
|
|
+ * and expression parsing and is deliberately out of scope here.
|
|
|
+ */
|
|
|
+
|
|
|
+import { lex, Token } from './lexer';
|
|
|
+import { M2Symbol, M2Unit, parseUnitText, rangeContains } from './parser';
|
|
|
+import { flattenUnit, resolveName } from './resolve';
|
|
|
+
|
|
|
+export interface M2Issue {
|
|
|
+ line: number; ch: number; endLine: number; endCh: number;
|
|
|
+ message: string;
|
|
|
+ severity: 'error' | 'warning';
|
|
|
+}
|
|
|
+
|
|
|
+/** Predefined identifiers available without declaration (GNU/PIM/ISO common core). */
|
|
|
+const BUILTINS = new Set([
|
|
|
+ // Types
|
|
|
+ 'INTEGER', 'CARDINAL', 'LONGINT', 'SHORTINT', 'LONGCARD', 'SHORTCARD',
|
|
|
+ 'REAL', 'LONGREAL', 'CHAR', 'BOOLEAN', 'BITSET', 'ADDRESS', 'WORD', 'BYTE',
|
|
|
+ 'OCTET',
|
|
|
+ // Constants
|
|
|
+ 'NIL', 'TRUE', 'FALSE',
|
|
|
+ // Procedures and functions
|
|
|
+ 'ABS', 'ADR', 'CAP', 'CHR', 'DEC', 'DISPOSE', 'EXCL', 'FLOAT', 'HALT',
|
|
|
+ 'HIGH', 'INC', 'INCL', 'LFLOAT', 'MAX', 'MIN', 'NEW', 'ODD', 'ORD',
|
|
|
+ 'SIZE', 'TRUNC', 'TSIZE', 'VAL', 'CODE',
|
|
|
+]);
|
|
|
+
|
|
|
+const OPENERS = new Set(['BEGIN', 'IF', 'CASE', 'LOOP', 'WHILE', 'FOR', 'WITH', 'RECORD', 'MODULE']);
|
|
|
+
|
|
|
+export function analyseUnit(text: string, filePath: string, docDir: string): M2Issue[] {
|
|
|
+ const unit = parseUnitText(text, filePath);
|
|
|
+ return [
|
|
|
+ ...duplicateDeclarations(unit),
|
|
|
+ ...unknownIdentifiers(text, unit, docDir),
|
|
|
+ ];
|
|
|
+}
|
|
|
+
|
|
|
+/** Same name declared twice in one scope (Modula-2 has no overloading). */
|
|
|
+function duplicateDeclarations(unit: M2Unit): M2Issue[] {
|
|
|
+ const issues: M2Issue[] = [];
|
|
|
+ const scopes = new Map<M2Symbol | null, Map<string, M2Symbol>>();
|
|
|
+ for (const { sym, chain } of flattenUnit(unit)) {
|
|
|
+ if (!sym.name) continue;
|
|
|
+ // The file's own module symbol is the root scope, not a declaration.
|
|
|
+ if (sym.kind === 'module' && chain.length === 0) continue;
|
|
|
+ const parent = chain[chain.length - 1] ?? null;
|
|
|
+ // Enumeration literals live in the scope enclosing their type.
|
|
|
+ const scope = sym.kind === 'constant' && parent?.kind === 'type'
|
|
|
+ ? chain[chain.length - 2] ?? null
|
|
|
+ : parent;
|
|
|
+ let seen = scopes.get(scope);
|
|
|
+ if (!seen) { seen = new Map(); scopes.set(scope, seen); }
|
|
|
+ const first = seen.get(sym.name);
|
|
|
+ if (first) {
|
|
|
+ issues.push({
|
|
|
+ line: sym.nameRange.startLine, ch: sym.nameRange.startCh,
|
|
|
+ endLine: sym.nameRange.endLine, endCh: sym.nameRange.endCh,
|
|
|
+ message: `Duplicate declaration "${sym.name}" (first declared at line ${first.nameRange.startLine + 1})`,
|
|
|
+ severity: 'error',
|
|
|
+ });
|
|
|
+ } else {
|
|
|
+ seen.set(sym.name, sym);
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return issues;
|
|
|
+}
|
|
|
+
|
|
|
+/** Identifier uses that resolve to no visible declaration. */
|
|
|
+function unknownIdentifiers(text: string, unit: M2Unit, docDir: string): M2Issue[] {
|
|
|
+ const issues: M2Issue[] = [];
|
|
|
+ const tokens = lex(text);
|
|
|
+ const declared = new Set<string>();
|
|
|
+ for (const { sym } of flattenUnit(unit)) {
|
|
|
+ declared.add(`${sym.nameRange.startLine}:${sym.nameRange.startCh}`);
|
|
|
+ }
|
|
|
+ const fromNames = new Set<string>();
|
|
|
+ for (const imp of unit.imports) for (const n of imp.names) fromNames.add(n);
|
|
|
+ const withRegions = withRegionsOf(tokens);
|
|
|
+ for (let k = 0; k < tokens.length; k++) {
|
|
|
+ const t = tokens[k];
|
|
|
+ if (t.kind !== 'ident') continue;
|
|
|
+ const prev = tokens[k - 1];
|
|
|
+ const next = tokens[k + 1];
|
|
|
+ // Qualified access (`M.X`, `r.field`): unverifiable without deeper analysis.
|
|
|
+ if (prev?.text === '.' || next?.text === '.') continue;
|
|
|
+ // Declaration sites.
|
|
|
+ if (declared.has(`${t.line}:${t.ch}`)) continue;
|
|
|
+ // Import clauses.
|
|
|
+ if (unit.imports.some(imp => rangeContains(imp.range, t.line, t.ch))) continue;
|
|
|
+ // WITH regions (unqualified record field access).
|
|
|
+ if (withRegions.some(([a, b]) => a <= t.offset && t.endOffset <= b)) continue;
|
|
|
+ // FROM-imported: unverifiable when the defining `.def` is unavailable.
|
|
|
+ if (fromNames.has(t.text)) continue;
|
|
|
+ if (BUILTINS.has(t.text)) continue;
|
|
|
+ // Closing `END Name`.
|
|
|
+ if (prev?.kind === 'keyword' && prev.text === 'END') continue;
|
|
|
+ const resolved = resolveName(unit, docDir, t.line, t.ch, null, t.text);
|
|
|
+ if (!resolved) {
|
|
|
+ issues.push({
|
|
|
+ line: t.line, ch: t.ch, endLine: t.endLine, endCh: t.endCh,
|
|
|
+ message: `Unknown identifier "${t.text}"`,
|
|
|
+ severity: 'error',
|
|
|
+ });
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return issues;
|
|
|
+}
|
|
|
+
|
|
|
+/** Offset spans of `WITH ... DO ... END` blocks (conservative bracket matching). */
|
|
|
+function withRegionsOf(tokens: Token[]): Array<[number, number]> {
|
|
|
+ const regions: Array<[number, number]> = [];
|
|
|
+ for (let k = 0; k < tokens.length; k++) {
|
|
|
+ const t = tokens[k];
|
|
|
+ if (t.kind !== 'keyword' || t.text !== 'WITH') continue;
|
|
|
+ let doIdx = -1;
|
|
|
+ for (let j = k + 1; j < Math.min(k + 13, tokens.length); j++) {
|
|
|
+ const u = tokens[j];
|
|
|
+ if (u.kind === 'keyword' && u.text === 'DO') { doIdx = j; break; }
|
|
|
+ if ((u.kind === 'symbol' && u.text === ';') ||
|
|
|
+ (u.kind === 'keyword' && (u.text === 'BEGIN' || u.text === 'END'))) break;
|
|
|
+ }
|
|
|
+ if (doIdx < 0) continue;
|
|
|
+ let depth = 1;
|
|
|
+ for (let m = doIdx + 1; m < tokens.length; m++) {
|
|
|
+ const u = tokens[m];
|
|
|
+ if (u.kind === 'keyword' && OPENERS.has(u.text)) depth++;
|
|
|
+ else if (u.kind === 'keyword' && u.text === 'END') {
|
|
|
+ depth--;
|
|
|
+ if (depth === 0) { regions.push([t.offset, u.endOffset]); break; }
|
|
|
+ }
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return regions;
|
|
|
+}
|