| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528 |
- /** Token-based Oberon document formatter with user-settable rules.
- *
- * Twin of the Pascal formatter: line structure is preserved
- * (statements are never joined or split), while indentation, keyword
- * case, intra-line spacing, blank lines and trailing whitespace are
- * normalised per `FormatOptions`. Comments are preserved verbatim
- * (only repositioned); continuation lines of multi-line comments are
- * emitted as-is. Best-effort on broken code: always terminates and
- * never drops tokens. `*`/`-` export marks stay attached to their
- * declaration; the same glyphs as operators keep their spaces.
- */
- import * as path from 'path';
- import { parseImports, parseOberon, ObSymbol } from './oberonSymbols';
- import { resolveAt } from './oberonResolve';
- export type KeywordCase = 'upper' | 'preserve';
- export interface FormatOptions {
- indentSize: number;
- useTabs: boolean;
- keywordCase: KeywordCase;
- spaceAroundOperators: boolean;
- spaceAfterComma: boolean;
- spaceBeforeColon: boolean;
- spaceAroundRange: boolean;
- emptyLineLimit: number;
- trimTrailingWhitespace: boolean;
- insertFinalNewline: boolean;
- }
- export const DEFAULT_FORMAT_OPTIONS: FormatOptions = {
- indentSize: 2,
- useTabs: false,
- keywordCase: 'upper',
- spaceAroundOperators: true,
- spaceAfterComma: true,
- spaceBeforeColon: false,
- spaceAroundRange: true,
- emptyLineLimit: 1,
- trimTrailingWhitespace: true,
- insertFinalNewline: true,
- };
- /** Sanitise arbitrary settings JSON into valid options (never throws). */
- export function normalizeFormatOptions(raw: unknown): FormatOptions {
- const r = (raw ?? {}) as Partial<FormatOptions>;
- const num = (v: unknown, lo: number, hi: number, dflt: number): number => {
- const n = typeof v === 'number' && Number.isFinite(v) ? Math.round(v) : dflt;
- return Math.min(hi, Math.max(lo, n));
- };
- const bool = (v: unknown, dflt: boolean): boolean =>
- typeof v === 'boolean' ? v : dflt;
- return {
- indentSize: num(r.indentSize, 0, 8, DEFAULT_FORMAT_OPTIONS.indentSize),
- useTabs: bool(r.useTabs, false),
- keywordCase: r.keywordCase === 'preserve' || r.keywordCase === 'upper'
- ? r.keywordCase : 'upper',
- spaceAroundOperators: bool(r.spaceAroundOperators, true),
- spaceAfterComma: bool(r.spaceAfterComma, true),
- spaceBeforeColon: bool(r.spaceBeforeColon, false),
- spaceAroundRange: bool(r.spaceAroundRange, true),
- emptyLineLimit: num(r.emptyLineLimit, 0, 10, DEFAULT_FORMAT_OPTIONS.emptyLineLimit),
- trimTrailingWhitespace: bool(r.trimTrailingWhitespace, true),
- insertFinalNewline: bool(r.insertFinalNewline, true),
- };
- }
- type TokenKind = 'keyword' | 'ident' | 'number' | 'string' | 'symbol';
- interface RichToken {
- text: string;
- kind: TokenKind;
- comment?: boolean;
- line: number; ch: number; endLine: number; endCh: number;
- offset: number; endOffset: number;
- }
- /** Reserved words matched case-insensitively by the formatter lexer. */
- const OBERON_KEYWORDS = new Set([
- 'ARRAY', 'BEGIN', 'BY', 'CASE', 'CONST', 'DEFINITION', 'DIV', 'DO',
- 'ELSE', 'ELSIF', 'END', 'EXIT', 'FOR', 'IF', 'IMPORT', 'IN', 'IS',
- 'LOOP', 'MOD', 'MODULE', 'NIL', 'OF', 'OR', 'POINTER', 'PROCEDURE',
- 'RECORD', 'REPEAT', 'RETURN', 'THEN', 'TO', 'TYPE', 'UNTIL', 'VAR',
- 'WHILE', 'WITH',
- ]);
- const isWordStart = (c: string) => /[A-Za-z_]/.test(c);
- const isWordChar = (c: string) => /[A-Za-z0-9_]/.test(c);
- const isDigit = (c: string) => /[0-9]/.test(c);
- const isHex = (c: string) => /[0-9A-Fa-f]/.test(c);
- /** Lex code tokens (comments land in `comments`, strings are skipped). */
- function lexCode(text: string): { tokens: RichToken[]; comments: RichToken[] } {
- const tokens: RichToken[] = [];
- const comments: RichToken[] = [];
- let i = 0;
- let line = 0;
- let col = 0;
- const push = (t: string, kind: TokenKind, sl: number, sc: number, so: number): void => {
- tokens.push({ text: t, kind, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
- };
- const pushComment = (t: string, sl: number, sc: number, so: number): void => {
- comments.push({ text: t, kind: 'ident', comment: true, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
- };
- const step = (n: number): void => {
- for (let k = 0; k < n; k++) {
- if (text[i] === '\n') { line++; col = 0; }
- else if (text[i] !== '\r') { col++; }
- else { col = 0; }
- i++;
- }
- };
- while (i < text.length) {
- const c = text[i];
- const nx = i + 1 < text.length ? text[i + 1] : '';
- if (c === '\n' || c === '\r' || c === ' ' || c === '\t') { step(1); continue; }
- // Nested (* *) comment (the only Oberon comment form).
- if (c === '(' && nx === '*') {
- const sl = line; const sc = col; const so = i;
- let level = 1;
- step(2);
- while (i < text.length && level > 0) {
- if (text[i] === '(' && text[i + 1] === '*') { level++; step(2); continue; }
- if (text[i] === '*' && text[i + 1] === ')') { level--; step(2); continue; }
- step(1);
- }
- pushComment(text.slice(so, i), sl, sc, so);
- continue;
- }
- // "..." / '...' strings (no escapes; unterminated runs to end of line).
- if (c === '"' || c === "'") {
- const sl = line; const sc = col; const so = i;
- step(1);
- while (i < text.length && text[i] !== c && text[i] !== '\n' && text[i] !== '\r') step(1);
- if (i < text.length && text[i] === c) step(1);
- push(text.slice(so, i), 'string', sl, sc, so);
- continue;
- }
- if (isWordStart(c)) {
- const sl = line; const sc = col; const so = i;
- while (i < text.length && isWordChar(text[i])) step(1);
- const word = text.slice(so, i);
- push(word, OBERON_KEYWORDS.has(word.toUpperCase()) ? 'keyword' : 'ident', sl, sc, so);
- continue;
- }
- // Numbers: decimal, .fraction, E exponents, H/X suffixes (0FFH, 41X).
- if (isDigit(c)) {
- const sl = line; const sc = col; const so = i;
- while (i < text.length && isHex(text[i])) step(1);
- if ((text[i] === 'H' || text[i] === 'X') && !isWordChar(text[i + 1] ?? '')) {
- step(1);
- } else {
- // Rewind hex letters: plain decimal (optionally real).
- while (i > so && /[A-Fa-f]/.test(text[i - 1] ?? '')) {
- i--; col--;
- }
- if (text[i] === '.' && text[i + 1] !== '.' && isDigit(text[i + 1] ?? '')) {
- step(1);
- while (i < text.length && isDigit(text[i])) step(1);
- }
- if (text[i] === 'E' || text[i] === 'e') {
- let j = i + 1;
- if (text[j] === '+' || text[j] === '-') j++;
- if (isDigit(text[j] ?? '')) {
- while (j < text.length && isDigit(text[j])) j++;
- step(j - i);
- }
- }
- }
- push(text.slice(so, i), 'number', sl, sc, so);
- continue;
- }
- // Multi-char symbols.
- const sl = line; const sc = col; const so = i;
- const two = text.slice(i, i + 2);
- if (two === ':=' || two === '<=' || two === '>=') {
- step(2);
- push(two, 'symbol', sl, sc, so);
- continue;
- }
- if (c === '.' && nx === '.') {
- step(2);
- push('..', 'symbol', sl, sc, so);
- continue;
- }
- step(1);
- push(c, 'symbol', sl, sc, so);
- }
- return { tokens, comments };
- }
- /** Structural words for indentation (matched case-insensitively). */
- const STRUCT = new Set([
- 'END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY',
- 'THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'CASE', 'OF', 'LOOP', 'TRY',
- ]);
- const DISPLAY_DEDENT = new Set(['END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY']);
- const OPENERS = new Set(['THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'LOOP', 'TRY']);
- const CLOSERS = new Set(['END', 'UNTIL']);
- const UNARY_AFTER = new Set([
- '(', '[', ',', ';', ':=', '=', '..', 'THEN', 'DO', 'OF',
- 'NOT', 'AND', 'OR', 'DIV', 'MOD', 'IN', '+', '-', '*', '/', '<', '<=',
- '>', '>=', '#', '&', '~', 'BY', 'TO',
- ]);
- function wordOf(t: RichToken): string {
- return t.kind === 'keyword' && STRUCT.has(t.text.toUpperCase()) ? t.text.toUpperCase() : '';
- }
- function applyCase(text: string, mode: KeywordCase): string {
- return mode === 'upper' ? text.toUpperCase() : text;
- }
- /** Promote keyword-shaped identifiers to keywords unless they resolve.
- *
- * Lowercase keywords lex as identifiers, but code may use words like
- * `to` as identifiers. An identifier that is declared here or resolves
- * (qualifier-aware) stays an identifier; anything else shaped like a
- * reserved word is a keyword. This keeps formatting token-exact.
- */
- function classifyKeywords(tokens: RichToken[], text: string, filePath: string, docDir: string, marks: Set<string>): void {
- let symbols: ObSymbol[];
- try {
- symbols = parseOberon(text, marks);
- } catch {
- return;
- }
- const imports = parseImports(text);
- const declared = new Set<string>();
- const flatten = (ss: ObSymbol[]): void => {
- for (const s of ss) {
- if (s.name) declared.add(`${s.line}:${s.ch}`);
- flatten(s.children);
- }
- };
- flatten(symbols);
- const prevCode = (i: number): { tok: RichToken; idx: number } | null => {
- for (let j = i - 1; j >= 0; j--) {
- const t = tokens[j];
- if (!t.comment) return { tok: t, idx: j };
- }
- return null;
- };
- tokens.forEach((t, i) => {
- if (t.comment || t.kind !== 'ident' || !OBERON_KEYWORDS.has(t.text.toUpperCase())) return;
- if (declared.has(`${t.line}:${t.ch}`)) return;
- let qualifier: string | null = null;
- const dot = prevCode(i);
- if (dot && dot.tok.text === '.') {
- const head = prevCode(dot.idx);
- if (!head || head.tok.kind !== 'ident') return;
- qualifier = head.tok.text;
- }
- const r = resolveAt(symbols, imports, filePath, docDir, t.line, t.ch, qualifier, t.text);
- if (!r) t.kind = 'keyword';
- });
- }
- export function formatDocument(
- text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
- ): string {
- const opts = normalizeFormatOptions(rawOptions ?? {});
- const { entries, eol } = formatLineEntries(text, opts, filePath, docDir);
- const joined = entries.map(l => l.text).join('\n');
- if (!opts.insertFinalNewline) return joined.split('\n').join(eol);
- return (joined.length > 0 ? joined + '\n' : '').split('\n').join(eol);
- }
- /** One formatted output line and the 0-based input line it came from. */
- export interface FormattedLine {
- src: number;
- text: string;
- }
- /** One edit for `textDocument/rangeFormatting`, directly mappable to LSP. */
- export interface RangeEdit {
- startLine: number;
- startCh: number;
- endLine: number;
- endCh: number;
- newText: string;
- }
- /** Format a line range with full-document context. Never throws. */
- export function formatRangeEdits(
- text: string, startLine: number, endLine: number,
- rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
- ): RangeEdit[] {
- if (text.length === 0) return [];
- const opts = normalizeFormatOptions(rawOptions ?? {});
- const srcLines = text.split('\n');
- const srcCount = srcLines.length;
- const from = Math.min(Math.max(0, Math.min(startLine, endLine)), srcCount - 1);
- const to = Math.min(Math.max(Math.max(startLine, endLine), from), srcCount - 1);
- const { entries } = formatLineEntries(text, rawOptions, filePath, docDir);
- const bySrc = new Map<number, string>();
- for (const e of entries) {
- if (e.src >= from && e.src <= to && !bySrc.has(e.src)) bySrc.set(e.src, e.text);
- }
- const edits: RangeEdit[] = [];
- let runFrom = -1;
- const parts: string[] = [];
- let tailNewlined = false;
- const flush = (endExclusive: number): void => {
- if (runFrom < 0) return;
- const atEof = endExclusive >= srcCount;
- if (atEof && parts.length === 0 && runFrom > 0) {
- const prevLen = srcLines[runFrom - 1].length;
- edits.push({
- startLine: runFrom - 1, startCh: prevLen,
- endLine: srcCount - 1, endCh: srcLines[srcCount - 1].length,
- newText: opts.insertFinalNewline && entries.length > 0 ? '\n' : '',
- });
- runFrom = -1;
- return;
- }
- let newText = parts.join('\n');
- if (parts.length > 0 && (atEof ? opts.insertFinalNewline : true)) newText += '\n';
- if (atEof && newText.endsWith('\n')) tailNewlined = true;
- const end = atEof
- ? { line: srcCount - 1, ch: srcLines[srcCount - 1].length }
- : { line: endExclusive, ch: 0 };
- edits.push({
- startLine: runFrom, startCh: 0,
- endLine: end.line, endCh: end.ch, newText,
- });
- runFrom = -1;
- parts.length = 0;
- };
- for (let s = from; s <= to; s++) {
- const want = bySrc.get(s);
- if (want !== undefined && want === srcLines[s]) {
- flush(s);
- continue;
- }
- if (runFrom < 0) runFrom = s;
- if (want !== undefined) parts.push(want);
- }
- flush(to + 1);
- if (to === srcCount - 1 && opts.insertFinalNewline && !text.endsWith('\n') &&
- entries.length > 0 && !tailNewlined) {
- const last = srcLines[srcCount - 1].length;
- edits.push({
- startLine: srcCount - 1, startCh: last,
- endLine: srcCount - 1, endCh: last, newText: '\n',
- });
- }
- return edits;
- }
- function formatLineEntries(
- text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
- ): { entries: FormattedLine[]; eol: string } {
- const opts = normalizeFormatOptions(rawOptions ?? {});
- if (text.length === 0) return { entries: [], eol: '\n' };
- const eol = text.includes('\r\n') ? '\r\n' : '\n';
- const dir = docDir || (filePath ? path.dirname(filePath) : '');
- const { tokens, comments } = lexCode(text);
- const all: RichToken[] = [...tokens, ...comments].sort((a, b) => a.offset - b.offset);
- const marks = new Set<string>();
- classifyKeywords(all, text, filePath, dir, marks);
- const byLine = new Map<number, RichToken[]>();
- let maxLine = 0;
- for (const t of all) {
- maxLine = Math.max(maxLine, t.endLine);
- const list = byLine.get(t.line) ?? [];
- list.push(t);
- byLine.set(t.line, list);
- }
- const srcLines = text.split(/\r?\n/);
- const lastLine = Math.max(maxLine, srcLines.length - 1);
- const verbatim = new Set<number>();
- for (const t of all) {
- if (t.comment && t.endLine > t.line) {
- for (let l = t.line + 1; l <= t.endLine; l++) verbatim.add(l);
- }
- }
- const indentOf = (level: number): string =>
- opts.useTabs ? '\t'.repeat(level) : ' '.repeat(opts.indentSize * level);
- const entries: FormattedLine[] = [];
- let level = 0;
- let casePending = false;
- let blanks = 0;
- for (let i = 0; i <= lastLine; i++) {
- const lineToks = (byLine.get(i) ?? []).filter(t => !t.comment || t.line === i);
- const code = lineToks.filter(t => !t.comment);
- const startsHere = lineToks.filter(t => t.comment && t.line === i);
- if (verbatim.has(i)) {
- entries.push({ src: i, text: opts.trimTrailingWhitespace ? srcLines[i]?.replace(/[ \t]+$/, '') ?? '' : srcLines[i] ?? '' });
- const applied = applyDelta(code, level, { casePending });
- level = applied.level;
- casePending = applied.casePending;
- continue;
- }
- if (code.length === 0 && startsHere.length === 0) {
- blanks++;
- if (blanks <= opts.emptyLineLimit) entries.push({ src: i, text: '' });
- continue;
- }
- blanks = 0;
- if (code.length === 0) {
- const parts = startsHere.map(c => firstFragment(c));
- entries.push({ src: i, text: indentOf(level) + parts.join(' ') });
- continue;
- }
- const first = wordOf(code[0]);
- const emitLevel = Math.max(0, level - (DISPLAY_DEDENT.has(first) ? 1 : 0));
- let body = emitTokens(code, opts, marks);
- const trailing = startsHere.filter(c => code[0].offset < c.offset);
- for (const c of trailing) body += ' ' + firstFragment(c);
- let lineText = indentOf(emitLevel) + body;
- if (opts.trimTrailingWhitespace) lineText = lineText.replace(/[ \t]+$/, '');
- entries.push({ src: i, text: lineText });
- const applied = applyDelta(code, level, { casePending });
- level = applied.level;
- casePending = applied.casePending;
- }
- if (!opts.insertFinalNewline) return { entries, eol };
- while (entries.length > 0 && /^[ \t]*$/.test(entries[entries.length - 1].text)) {
- entries.pop();
- }
- if (entries.length > 0) {
- entries[entries.length - 1].text = entries[entries.length - 1].text.replace(/[ \t]+$/, '');
- }
- return { entries, eol };
- }
- function firstFragment(c: RichToken): string {
- const idx = c.text.search(/\r?\n/);
- return idx < 0 ? c.text : c.text.slice(0, idx);
- }
- function applyDelta(
- code: RichToken[], level: number, state: { casePending: boolean },
- ): { level: number; casePending: boolean } {
- let opens = 0;
- let closes = 0;
- let casePending = state.casePending;
- for (const t of code) {
- const w = wordOf(t);
- if (!w) continue;
- if (w === 'CASE') {
- casePending = true;
- continue;
- }
- if (w === 'OF') {
- if (casePending) {
- opens++;
- casePending = false;
- }
- continue;
- }
- if (OPENERS.has(w)) {
- opens++;
- continue;
- }
- if (CLOSERS.has(w)) {
- closes++;
- continue;
- }
- }
- return { level: Math.max(0, level + opens - closes), casePending };
- }
- function isWordish(t: RichToken): boolean {
- return t.kind === 'ident' || t.kind === 'keyword' || t.kind === 'number' || t.kind === 'string';
- }
- function emitTokens(code: RichToken[], opts: FormatOptions, marks: Set<string>): string {
- let out = '';
- const gap = (prev: RichToken | null, cur: RichToken): string => {
- if (!prev) return '';
- const pt = prev.text;
- const ct = cur.text;
- if (ct === '(') return prev.kind === 'keyword' ? ' ' : '';
- if (ct === '[') return prev.kind === 'keyword' ? ' ' : '';
- if (pt === '(' || pt === '[') return '';
- if (pt === '.' || pt === '^' || pt === '@') return '';
- if (ct === ')' || ct === ']') return '';
- if (ct === '.' || ct === '^') return '';
- if (ct === ',') return '';
- if (ct === ';') return '';
- if (pt === ')' || pt === ']' || pt === ';') {
- return isWordish(cur) ? ' ' : '';
- }
- // `*`/`-` export marks attach to the declared name (positions
- // recorded by the parser); the same glyphs as operators keep spaces.
- if ((ct === '*' || ct === '-') && prev.kind === 'ident') {
- return marks.has(`${cur.line}:${cur.ch}`) ? '' : (opts.spaceAroundOperators ? ' ' : '');
- }
- if (ct === ':=') return opts.spaceAroundOperators ? ' ' : '';
- if (pt === ':=') return opts.spaceAroundOperators ? ' ' : '';
- if (pt === ',') return opts.spaceAfterComma ? ' ' : '';
- if (ct === ':') return opts.spaceBeforeColon ? ' ' : '';
- if (pt === ':') return ' ';
- if (ct === '..' || pt === '..') return opts.spaceAroundRange ? ' ' : '';
- if (ct === '+' || ct === '-') return isUnary(prev) ? '' : (opts.spaceAroundOperators ? ' ' : '');
- if (cur.kind === 'symbol' || prev.kind === 'symbol') {
- return opts.spaceAroundOperators ? ' ' : '';
- }
- return ' ';
- };
- for (let k = 0; k < code.length; k++) {
- const t = code[k];
- const text = t.kind === 'keyword' ? applyCase(t.text, opts.keywordCase) : t.text;
- out += gap(k > 0 ? code[k - 1] : null, t) + text;
- }
- return out;
- }
- function isUnary(prev: RichToken | null): boolean {
- if (!prev) return true;
- if (prev.kind === 'keyword') return UNARY_AFTER.has(prev.text.toUpperCase());
- if (prev.kind !== 'symbol') return false;
- return UNARY_AFTER.has(prev.text);
- }
|