|
|
@@ -0,0 +1,386 @@
|
|
|
+/** Token-based Modula-2 document formatter with user-settable rules.
|
|
|
+ *
|
|
|
+ * Design: line structure is preserved (statements are never joined or
|
|
|
+ * split), while indentation, keyword case, intra-line spacing, blank
|
|
|
+ * lines and trailing whitespace are normalised per `FormatOptions`.
|
|
|
+ * Comments are preserved verbatim (only repositioned); continuation
|
|
|
+ * lines of multi-line comments are emitted as-is. The formatter is
|
|
|
+ * best-effort on broken code: it always terminates and never drops
|
|
|
+ * tokens (verified by the preservation tests).
|
|
|
+ */
|
|
|
+
|
|
|
+import { Token, lex, M2_KEYWORDS } from './lexer';
|
|
|
+import { M2Unit, parseUnitText } from './parser';
|
|
|
+import { flattenUnit, resolveName } from './resolve';
|
|
|
+import { procTypeFormalPositions } from './analyse';
|
|
|
+import * as path from 'path';
|
|
|
+
|
|
|
+export type KeywordCase = 'upper' | 'preserve';
|
|
|
+
|
|
|
+export interface FormatOptions {
|
|
|
+ indentSize: number;
|
|
|
+ useTabs: boolean;
|
|
|
+ keywordCase: KeywordCase;
|
|
|
+ spaceAroundOperators: boolean;
|
|
|
+ spaceAfterComma: boolean;
|
|
|
+ spaceBeforeColon: boolean;
|
|
|
+ spaceAroundRange: boolean;
|
|
|
+ emptyLineLimit: number;
|
|
|
+ trimTrailingWhitespace: boolean;
|
|
|
+ insertFinalNewline: boolean;
|
|
|
+}
|
|
|
+
|
|
|
+export const DEFAULT_FORMAT_OPTIONS: FormatOptions = {
|
|
|
+ indentSize: 2,
|
|
|
+ useTabs: false,
|
|
|
+ keywordCase: 'upper',
|
|
|
+ spaceAroundOperators: true,
|
|
|
+ spaceAfterComma: true,
|
|
|
+ spaceBeforeColon: false,
|
|
|
+ spaceAroundRange: true,
|
|
|
+ emptyLineLimit: 1,
|
|
|
+ trimTrailingWhitespace: true,
|
|
|
+ insertFinalNewline: true,
|
|
|
+};
|
|
|
+
|
|
|
+/** Sanitise arbitrary settings JSON into valid options (never throws). */
|
|
|
+export function normalizeFormatOptions(raw: unknown): FormatOptions {
|
|
|
+ const r = (raw ?? {}) as Partial<FormatOptions>;
|
|
|
+ const num = (v: unknown, lo: number, hi: number, dflt: number): number => {
|
|
|
+ const n = typeof v === 'number' && Number.isFinite(v) ? Math.round(v) : dflt;
|
|
|
+ return Math.min(hi, Math.max(lo, n));
|
|
|
+ };
|
|
|
+ const bool = (v: unknown, dflt: boolean): boolean =>
|
|
|
+ typeof v === 'boolean' ? v : dflt;
|
|
|
+ return {
|
|
|
+ indentSize: num(r.indentSize, 0, 8, DEFAULT_FORMAT_OPTIONS.indentSize),
|
|
|
+ useTabs: bool(r.useTabs, false),
|
|
|
+ keywordCase: r.keywordCase === 'preserve' || r.keywordCase === 'upper'
|
|
|
+ ? r.keywordCase : 'upper',
|
|
|
+ spaceAroundOperators: bool(r.spaceAroundOperators, true),
|
|
|
+ spaceAfterComma: bool(r.spaceAfterComma, true),
|
|
|
+ spaceBeforeColon: bool(r.spaceBeforeColon, false),
|
|
|
+ spaceAroundRange: bool(r.spaceAroundRange, true),
|
|
|
+ emptyLineLimit: num(r.emptyLineLimit, 0, 10, DEFAULT_FORMAT_OPTIONS.emptyLineLimit),
|
|
|
+ trimTrailingWhitespace: bool(r.trimTrailingWhitespace, true),
|
|
|
+ insertFinalNewline: bool(r.insertFinalNewline, true),
|
|
|
+ };
|
|
|
+}
|
|
|
+
|
|
|
+interface RichToken extends Token {
|
|
|
+ comment?: boolean;
|
|
|
+}
|
|
|
+
|
|
|
+/** Structural words for indentation (matched case-insensitively). */
|
|
|
+const STRUCT = new Set([
|
|
|
+ 'END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY',
|
|
|
+ 'THEN', 'LOOP', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'CASE', 'OF',
|
|
|
+]);
|
|
|
+const DISPLAY_DEDENT = new Set(['END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY']);
|
|
|
+const OPENERS = new Set(['THEN', 'LOOP', 'BEGIN', 'RECORD', 'REPEAT', 'DO']);
|
|
|
+const CLOSERS = new Set(['END', 'UNTIL']);
|
|
|
+const UNARY_AFTER = new Set([
|
|
|
+ '(', '[', ',', ';', ':=', '=', '..', 'THEN', 'DO', 'OF', 'RETURN',
|
|
|
+ 'NOT', 'AND', 'OR', 'DIV', 'MOD', 'IN', '+', '-', '*', '/', '<', '<=',
|
|
|
+ '>', '>=', '<>', '#', 'BY', 'TO',
|
|
|
+]);
|
|
|
+
|
|
|
+function wordOf(t: RichToken): string {
|
|
|
+ return t.kind === 'keyword' && STRUCT.has(t.text.toUpperCase()) ? t.text.toUpperCase() : '';
|
|
|
+}
|
|
|
+
|
|
|
+function applyCase(text: string, mode: KeywordCase): string {
|
|
|
+ return mode === 'upper' ? text.toUpperCase() : text;
|
|
|
+}
|
|
|
+
|
|
|
+/** Promote keyword-shaped identifiers to keywords unless they resolve.
|
|
|
+ *
|
|
|
+ * Lowercase keywords lex as identifiers, but real code also uses words
|
|
|
+ * like `mod`, `In` or `Type` as identifiers (accepted by gm2). An
|
|
|
+ * identifier that is declared here, is a type-expression formal, or
|
|
|
+ * resolves (qualifier-aware) stays an identifier; anything else shaped
|
|
|
+ * like a reserved word is a keyword. This keeps formatting token-exact.
|
|
|
+ */
|
|
|
+function classifyKeywords(tokens: RichToken[], text: string, filePath: string, docDir: string): void {
|
|
|
+ const unit = parseUnitText(text, filePath);
|
|
|
+ const declared = new Set<string>();
|
|
|
+ for (const { sym } of flattenUnit(unit)) {
|
|
|
+ declared.add(`${sym.nameRange.startLine}:${sym.nameRange.startCh}`);
|
|
|
+ }
|
|
|
+ for (const pos of procTypeFormalPositions(tokens)) declared.add(pos);
|
|
|
+ const prevCode = (i: number): { tok: RichToken; idx: number } | null => {
|
|
|
+ for (let j = i - 1; j >= 0; j--) {
|
|
|
+ const t = tokens[j];
|
|
|
+ if (!t.comment) return { tok: t, idx: j };
|
|
|
+ }
|
|
|
+ return null;
|
|
|
+ };
|
|
|
+ tokens.forEach((t, i) => {
|
|
|
+ if (t.comment || t.kind !== 'ident' || !M2_KEYWORDS.has(t.text.toUpperCase())) return;
|
|
|
+ if (declared.has(`${t.line}:${t.ch}`)) return;
|
|
|
+ let qualifier: string | null = null;
|
|
|
+ const name = t.text;
|
|
|
+ const dot = prevCode(i);
|
|
|
+ if (dot && dot.tok.text === '.') {
|
|
|
+ const head = prevCode(dot.idx);
|
|
|
+ if (!head || head.tok.kind !== 'ident') return;
|
|
|
+ qualifier = head.tok.text;
|
|
|
+ }
|
|
|
+ const r = resolveName(unit, docDir, t.line, t.ch, qualifier, name);
|
|
|
+ if (!r) t.kind = 'keyword';
|
|
|
+ });
|
|
|
+}
|
|
|
+/** Comment ranges (nesting-aware; Modula-2 has no string escapes). */
|
|
|
+function scanComments(text: string): RichToken[] {
|
|
|
+ const out: RichToken[] = [];
|
|
|
+ let i = 0;
|
|
|
+ let line = 0;
|
|
|
+ let col = 0;
|
|
|
+ const skipString = (quote: string): void => {
|
|
|
+ i++;
|
|
|
+ col++;
|
|
|
+ while (i < text.length && text[i] !== '\n' && text[i] !== '\r') {
|
|
|
+ if (text[i] === quote) {
|
|
|
+ i++;
|
|
|
+ col++;
|
|
|
+ break;
|
|
|
+ }
|
|
|
+ i++;
|
|
|
+ col++;
|
|
|
+ }
|
|
|
+ };
|
|
|
+ while (i < text.length) {
|
|
|
+ const c = text[i];
|
|
|
+ if (c === "'" || c === '"') {
|
|
|
+ skipString(c);
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (c === '(' && text[i + 1] === '*') {
|
|
|
+ const s = i;
|
|
|
+ const sl = line;
|
|
|
+ const sc = col;
|
|
|
+ let level = 1;
|
|
|
+ i += 2;
|
|
|
+ col += 2;
|
|
|
+ while (i < text.length && level > 0) {
|
|
|
+ if (text[i] === '(' && text[i + 1] === '*') {
|
|
|
+ level++;
|
|
|
+ i += 2;
|
|
|
+ col += 2;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (text[i] === '*' && text[i + 1] === ')') {
|
|
|
+ level--;
|
|
|
+ i += 2;
|
|
|
+ col += 2;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (text[i] === '\n') {
|
|
|
+ line++;
|
|
|
+ col = 0;
|
|
|
+ } else if (text[i] !== '\r') {
|
|
|
+ col++;
|
|
|
+ } else {
|
|
|
+ col = 0;
|
|
|
+ }
|
|
|
+ i++;
|
|
|
+ }
|
|
|
+ out.push({
|
|
|
+ text: text.slice(s, i), kind: 'ident', comment: true,
|
|
|
+ line: sl, ch: sc, endLine: line, endCh: col,
|
|
|
+ offset: s, endOffset: i,
|
|
|
+ });
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (c === '\n') {
|
|
|
+ line++;
|
|
|
+ col = 0;
|
|
|
+ } else if (c !== '\r') {
|
|
|
+ col++;
|
|
|
+ } else {
|
|
|
+ col = 0;
|
|
|
+ }
|
|
|
+ i++;
|
|
|
+ }
|
|
|
+ return out;
|
|
|
+}
|
|
|
+
|
|
|
+export function formatDocument(
|
|
|
+ text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
|
|
|
+): string {
|
|
|
+ const opts = normalizeFormatOptions(rawOptions ?? {});
|
|
|
+ if (text.length === 0) return '';
|
|
|
+ const eol = text.includes('\r\n') ? '\r\n' : '\n';
|
|
|
+ const dir = docDir || (filePath ? path.dirname(filePath) : '');
|
|
|
+ const tokens: RichToken[] = [
|
|
|
+ ...lex(text).filter(t => t.kind !== 'eof'),
|
|
|
+ ...scanComments(text),
|
|
|
+ ].sort((a, b) => a.offset - b.offset);
|
|
|
+ classifyKeywords(tokens, text, filePath, dir);
|
|
|
+
|
|
|
+ const byLine = new Map<number, RichToken[]>();
|
|
|
+ let maxLine = 0;
|
|
|
+ for (const t of tokens) {
|
|
|
+ maxLine = Math.max(maxLine, t.endLine);
|
|
|
+ const list = byLine.get(t.line) ?? [];
|
|
|
+ list.push(t);
|
|
|
+ byLine.set(t.line, list);
|
|
|
+ }
|
|
|
+ const srcLines = text.split(/\r?\n/);
|
|
|
+ const lastLine = Math.max(maxLine, srcLines.length - 1);
|
|
|
+
|
|
|
+ // Continuation lines of multi-line comments: verbatim, but counted.
|
|
|
+ const verbatim = new Set<number>();
|
|
|
+ for (const t of tokens) {
|
|
|
+ if (t.comment && t.endLine > t.line) {
|
|
|
+ for (let l = t.line + 1; l <= t.endLine; l++) verbatim.add(l);
|
|
|
+ }
|
|
|
+ }
|
|
|
+
|
|
|
+ const indentOf = (level: number): string =>
|
|
|
+ opts.useTabs ? '\t'.repeat(level) : ' '.repeat(opts.indentSize * level);
|
|
|
+
|
|
|
+ const out: string[] = [];
|
|
|
+ let level = 0;
|
|
|
+ let casePending = false;
|
|
|
+ let blanks = 0;
|
|
|
+
|
|
|
+ for (let i = 0; i <= lastLine; i++) {
|
|
|
+ const lineToks = (byLine.get(i) ?? []).filter(t => !t.comment || t.line === i);
|
|
|
+ const code = lineToks.filter(t => !t.comment);
|
|
|
+ const startsHere = lineToks.filter(t => t.comment && t.line === i);
|
|
|
+
|
|
|
+ if (verbatim.has(i)) {
|
|
|
+ out.push(opts.trimTrailingWhitespace ? srcLines[i]?.replace(/[ \t]+$/, '') ?? '' : srcLines[i] ?? '');
|
|
|
+ level = applyDelta(code, level, { casePending }).level;
|
|
|
+ casePending = applyDelta(code, level, { casePending }).casePending;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+
|
|
|
+ if (code.length === 0 && startsHere.length === 0) {
|
|
|
+ blanks++;
|
|
|
+ if (blanks <= opts.emptyLineLimit) out.push('');
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ blanks = 0;
|
|
|
+
|
|
|
+ if (code.length === 0) {
|
|
|
+ // Comment-only line: indent, no level change.
|
|
|
+ const parts = startsHere.map(c => firstFragment(c));
|
|
|
+ out.push(indentOf(level) + parts.join(' '));
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+
|
|
|
+ const first = wordOf(code[0]);
|
|
|
+ const emitLevel = Math.max(0, level - (DISPLAY_DEDENT.has(first) ? 1 : 0));
|
|
|
+ let body = emitTokens(code, opts);
|
|
|
+ const trailing = startsHere.filter(c => code[0].offset < c.offset);
|
|
|
+ for (const c of trailing) body += ' ' + firstFragment(c);
|
|
|
+ let line = indentOf(emitLevel) + body;
|
|
|
+ if (opts.trimTrailingWhitespace) line = line.replace(/[ \t]+$/, '');
|
|
|
+ out.push(line);
|
|
|
+
|
|
|
+ const applied = applyDelta(code, level, { casePending });
|
|
|
+ level = applied.level;
|
|
|
+ casePending = applied.casePending;
|
|
|
+ }
|
|
|
+
|
|
|
+ if (!opts.insertFinalNewline) return out.join('\n').split('\n').join(eol);
|
|
|
+ const stripped = out.join('\n').replace(/[ \t\n]+$/, '');
|
|
|
+ return (stripped.length > 0 ? stripped + '\n' : '').split('\n').join(eol);
|
|
|
+}
|
|
|
+
|
|
|
+function firstFragment(c: RichToken): string {
|
|
|
+ const idx = c.text.search(/\r?\n/);
|
|
|
+ return idx < 0 ? c.text : c.text.slice(0, idx);
|
|
|
+}
|
|
|
+
|
|
|
+function applyDelta(
|
|
|
+ code: RichToken[], level: number, state: { casePending: boolean },
|
|
|
+): { level: number; casePending: boolean } {
|
|
|
+ let opens = 0;
|
|
|
+ let closes = 0;
|
|
|
+ let casePending = state.casePending;
|
|
|
+ for (const t of code) {
|
|
|
+ const w = wordOf(t);
|
|
|
+ if (!w) continue;
|
|
|
+ if (w === 'CASE') {
|
|
|
+ casePending = true;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (w === 'OF') {
|
|
|
+ if (casePending) {
|
|
|
+ opens++;
|
|
|
+ casePending = false;
|
|
|
+ }
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (OPENERS.has(w)) {
|
|
|
+ opens++;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ if (CLOSERS.has(w)) {
|
|
|
+ closes++;
|
|
|
+ continue;
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return { level: Math.max(0, level + opens - closes), casePending };
|
|
|
+}
|
|
|
+
|
|
|
+function isWordish(t: RichToken): boolean {
|
|
|
+ return t.kind === 'ident' || t.kind === 'keyword' || t.kind === 'number' || t.kind === 'string';
|
|
|
+}
|
|
|
+
|
|
|
+function emitTokens(code: RichToken[], opts: FormatOptions): string {
|
|
|
+ let out = '';
|
|
|
+ const gap = (prev: RichToken | null, cur: RichToken): string => {
|
|
|
+ if (!prev) return '';
|
|
|
+ const pt = prev.text;
|
|
|
+ const ct = cur.text;
|
|
|
+ // Opening brackets: attached, except after a keyword (`ARRAY [`).
|
|
|
+ if (ct === '(') return prev.kind === 'keyword' ? ' ' : '';
|
|
|
+ if (ct === '[') return prev.kind === 'keyword' ? ' ' : '';
|
|
|
+ if (ct === '{') return '';
|
|
|
+ if (ct === '<*') return ' ';
|
|
|
+ // After openers: attached.
|
|
|
+ if (pt === '(' || pt === '[' || pt === '{' || pt === '<*') return '';
|
|
|
+ if (pt === '.' || pt === '^') return '';
|
|
|
+ // Closing brackets and separators: attached before...
|
|
|
+ if (ct === ')' || ct === ']' || ct === '}' || ct === '*>') return '';
|
|
|
+ if (ct === '.' || ct === '^') return '';
|
|
|
+ if (ct === ',') return '';
|
|
|
+ if (ct === ';') return '';
|
|
|
+ // ...with a space after closers and `;` (except before tighter tokens).
|
|
|
+ if (pt === ')' || pt === ']' || pt === '}' || pt === '*>' || pt === ';') {
|
|
|
+ return isWordish(cur) ? ' ' : '';
|
|
|
+ }
|
|
|
+ if (ct === ':=') return opts.spaceAroundOperators ? ' ' : '';
|
|
|
+ if (pt === ':=') return opts.spaceAroundOperators ? ' ' : '';
|
|
|
+ if (pt === ',') return opts.spaceAfterComma ? ' ' : '';
|
|
|
+ if (ct === ':') return opts.spaceBeforeColon ? ' ' : '';
|
|
|
+ if (pt === ':') return ' ';
|
|
|
+ if (ct === '..' || pt === '..') return opts.spaceAroundRange ? ' ' : '';
|
|
|
+ if (ct === '+' || ct === '-') return isUnary(prev) ? '' : (opts.spaceAroundOperators ? ' ' : '');
|
|
|
+ if (ct === '^') return '';
|
|
|
+ // Symbol operators: spaced iff the setting is on.
|
|
|
+ if (cur.kind === 'symbol' || prev.kind === 'symbol') {
|
|
|
+ return opts.spaceAroundOperators ? ' ' : '';
|
|
|
+ }
|
|
|
+ // Word pairs (ident/keyword/number/string): single space.
|
|
|
+ return ' ';
|
|
|
+ };
|
|
|
+ let prev: RichToken | null = null;
|
|
|
+ for (const t of code) {
|
|
|
+ const text = t.kind === 'keyword' ? applyCase(t.text, opts.keywordCase) : t.text;
|
|
|
+ out += gap(prev, t) + text;
|
|
|
+ prev = t;
|
|
|
+ }
|
|
|
+ return out;
|
|
|
+}
|
|
|
+
|
|
|
+function isUnary(prev: RichToken | null): boolean {
|
|
|
+ if (!prev) return true;
|
|
|
+ if (prev.kind === 'keyword') return UNARY_AFTER.has(prev.text.toUpperCase());
|
|
|
+ if (prev.kind !== 'symbol') return false;
|
|
|
+ return UNARY_AFTER.has(prev.text);
|
|
|
+}
|