Просмотр исходного кода

Formatter: document formatting with user-settable rules

- new m2/format.ts: token-based formatter preserving line structure
  and comments verbatim; indentation from block structure, keyword
  case, operator/comma/colon/range spacing, blank lines, trim, final
  newline — all driven by modula2.format.* settings, live
- keywordCase is upper/preserve only (gm2 rejects lowercase keywords);
  keyword-shaped identifiers (mod, In, Type) resolve-checked first
- LSP documentFormattingProvider + client configurationSection sync
- 10 settings in the configuration schema, documented in FORMATTING.md
Eric Streit 1 неделя назад
Родитель
Сommit
9b845a47fc

+ 48 - 0
extensions/modula2-language/FORMATTING.md

@@ -0,0 +1,48 @@
+# Modula-2 formatting
+
+`Format Document` (Shift+Alt+F) normalises a whole file through the
+language server (`textDocument/formatting`). Every rule below is settable
+under `modula2.format.*` in Settings; changes apply live, no reload.
+
+## What the formatter does
+
+- **Indentation** from block structure (`BEGIN`/`IF`/`CASE`/`LOOP`/
+  `RECORD`/…`END`), using `indentSize` spaces or tabs.
+- **Keyword case** (`upper`, the default, or `preserve`).
+- **Spacing**: operators, commas, colons, `..` ranges.
+- **Blank lines** collapsed to `emptyLineLimit`; trailing whitespace
+  trimmed; file ends with exactly one newline (all settable).
+
+Line structure is deliberately preserved: statements are never joined or
+split, so formatting a file you didn't write can't scramble its layout.
+Comments are preserved verbatim (only repositioned). Formatting is
+idempotent and token-exact: running it twice changes nothing, and it never
+adds, removes or renames code — verified across real-world sources, whose
+formatted output still compiles with `gm2`.
+
+## Settings
+
+| Setting | Default | Effect |
+|---|---|---|
+| `modula2.format.indentSize` | `2` | Spaces per level (0–8; unused with tabs). |
+| `modula2.format.useTabs` | `false` | Indent with tabs. |
+| `modula2.format.keywordCase` | `"upper"` | `"upper"` uppercases reserved words, `"preserve"` leaves case alone. There is intentionally no `"lower"`: lowercase keywords do not compile with `gm2`. |
+| `modula2.format.spaceAroundOperators` | `true` | `a := b`, `x + y`. Off gives compact `a:=b`, `x+y`. |
+| `modula2.format.spaceAfterComma` | `true` | `a, b` vs `a,b`. |
+| `modula2.format.spaceBeforeColon` | `false` | `x: INTEGER` vs `x : INTEGER`. |
+| `modula2.format.spaceAroundRange` | `true` | `[0 .. 9]` vs `[0..9]`. |
+| `modula2.format.emptyLineLimit` | `1` | Max consecutive blank lines (0–10). |
+| `modula2.format.trimTrailingWhitespace` | `true` | Strip trailing spaces. |
+| `modula2.format.insertFinalNewline` | `true` | Exactly one newline at end of file. |
+
+## Notes
+
+- `keywordCase` covers reserved words only. Predefined identifiers
+  (`INTEGER`, `TRUE`, `NIL`, …) can legally be redeclared, so they are
+  never retouched — same as the typing-time auto-uppercasing.
+- Identifiers that merely look like keywords (`mod`, `In`, `Type` are all
+  real identifiers in the wild and accepted by `gm2`) are resolved first
+  and left alone. Unresolvable ones follow the keyword rule, matching the
+  editor's auto-uppercasing.
+- `CLIENT` tab/indent settings (`editor.tabSize`) do not drive Modula-2
+  formatting; `modula2.format.*` always wins.

+ 58 - 0
extensions/modula2-language/package.json

@@ -76,6 +76,64 @@
           "default": true,
           "description": "Uppercase a Modula-2 keyword when the completed token is followed by a non-identifier character."
         },
+        "modula2.format.indentSize": {
+          "type": "number",
+          "default": 2,
+          "minimum": 0,
+          "maximum": 8,
+          "description": "Indentation width in spaces for formatted Modula-2 code (ignored when modula2.format.useTabs is on)."
+        },
+        "modula2.format.useTabs": {
+          "type": "boolean",
+          "default": false,
+          "description": "Indent formatted Modula-2 code with tabs instead of spaces."
+        },
+        "modula2.format.keywordCase": {
+          "type": "string",
+          "enum": [
+            "upper",
+            "preserve"
+          ],
+          "default": "upper",
+          "description": "Keyword case applied by the formatter. Reserved words written lowercase are uppercased; identifiers that merely look like keywords (e.g. a variable named mod) are left alone."
+        },
+        "modula2.format.spaceAroundOperators": {
+          "type": "boolean",
+          "default": true,
+          "description": "Enforce single spaces around operators (:=, =, +, -, *, /, comparisons). Off produces compact code (a+b)."
+        },
+        "modula2.format.spaceAfterComma": {
+          "type": "boolean",
+          "default": true,
+          "description": "Enforce a space after each comma. Off produces compact code (a,b)."
+        },
+        "modula2.format.spaceBeforeColon": {
+          "type": "boolean",
+          "default": false,
+          "description": "Put a space before the colon in declarations (x : INTEGER instead of x: INTEGER)."
+        },
+        "modula2.format.spaceAroundRange": {
+          "type": "boolean",
+          "default": true,
+          "description": "Enforce spaces around .. in ranges ([0 .. 9]). Off produces compact ranges ([0..9])."
+        },
+        "modula2.format.emptyLineLimit": {
+          "type": "number",
+          "default": 1,
+          "minimum": 0,
+          "maximum": 10,
+          "description": "Maximum consecutive blank lines kept by the formatter."
+        },
+        "modula2.format.trimTrailingWhitespace": {
+          "type": "boolean",
+          "default": true,
+          "description": "Remove trailing whitespace on formatted lines."
+        },
+        "modula2.format.insertFinalNewline": {
+          "type": "boolean",
+          "default": true,
+          "description": "End formatted files with exactly one newline."
+        },
         "modula2.cocoR.validatorCommand": {
           "type": "string",
           "default": "/home/eric/Projets/Projets-Modula2/MyWork/Theia/GNU-grammar/bin/GM2",

+ 14 - 1
extensions/modula2-language/src/extension.ts

@@ -28,9 +28,22 @@ export function activate(context: vscode.ExtensionContext): void {
       { scheme: 'file', language: 'modula2' },
       { scheme: 'file', language: 'modula2-definition' }
     ],
+    synchronize: { configurationSection: 'modula2' },
     initializationOptions: {
       validatorCommand: cfg.get<string>('cocoR.validatorCommand', ''),
-      validatorArguments: cfg.get<string[]>('cocoR.validatorArguments', ['{file}'])
+      validatorArguments: cfg.get<string[]>('cocoR.validatorArguments', ['{file}']),
+      format: {
+        indentSize: cfg.get<number>('format.indentSize', 2),
+        useTabs: cfg.get<boolean>('format.useTabs', false),
+        keywordCase: cfg.get<string>('format.keywordCase', 'upper'),
+        spaceAroundOperators: cfg.get<boolean>('format.spaceAroundOperators', true),
+        spaceAfterComma: cfg.get<boolean>('format.spaceAfterComma', true),
+        spaceBeforeColon: cfg.get<boolean>('format.spaceBeforeColon', false),
+        spaceAroundRange: cfg.get<boolean>('format.spaceAroundRange', true),
+        emptyLineLimit: cfg.get<number>('format.emptyLineLimit', 1),
+        trimTrailingWhitespace: cfg.get<boolean>('format.trimTrailingWhitespace', true),
+        insertFinalNewline: cfg.get<boolean>('format.insertFinalNewline', true)
+      }
     }
   };
   const client = new LanguageClient('modula2CocoR', 'Modula-2 Coco/R Language Server', serverOptions, clientOptions);

+ 1 - 1
extensions/modula2-language/src/m2/analyse.ts

@@ -289,7 +289,7 @@ function circularDependencies(unit: M2Unit, docDir: string): M2Issue[] {
  *  Only triggers on the PROCEDURE keyword, so calls and case labels are
  *  unaffected: an undeclared `gren:` case label is still reported.
  */
-function procTypeFormalPositions(tokens: Token[]): string[] {
+export function procTypeFormalPositions(tokens: Token[]): string[] {
   const out: string[] = [];
   for (let k = 0; k < tokens.length; k++) {
     const t = tokens[k];

+ 386 - 0
extensions/modula2-language/src/m2/format.ts

@@ -0,0 +1,386 @@
+/** Token-based Modula-2 document formatter with user-settable rules.
+ *
+ *  Design: line structure is preserved (statements are never joined or
+ *  split), while indentation, keyword case, intra-line spacing, blank
+ *  lines and trailing whitespace are normalised per `FormatOptions`.
+ *  Comments are preserved verbatim (only repositioned); continuation
+ *  lines of multi-line comments are emitted as-is. The formatter is
+ *  best-effort on broken code: it always terminates and never drops
+ *  tokens (verified by the preservation tests).
+ */
+
+import { Token, lex, M2_KEYWORDS } from './lexer';
+import { M2Unit, parseUnitText } from './parser';
+import { flattenUnit, resolveName } from './resolve';
+import { procTypeFormalPositions } from './analyse';
+import * as path from 'path';
+
+export type KeywordCase = 'upper' | 'preserve';
+
+export interface FormatOptions {
+  indentSize: number;
+  useTabs: boolean;
+  keywordCase: KeywordCase;
+  spaceAroundOperators: boolean;
+  spaceAfterComma: boolean;
+  spaceBeforeColon: boolean;
+  spaceAroundRange: boolean;
+  emptyLineLimit: number;
+  trimTrailingWhitespace: boolean;
+  insertFinalNewline: boolean;
+}
+
+export const DEFAULT_FORMAT_OPTIONS: FormatOptions = {
+  indentSize: 2,
+  useTabs: false,
+  keywordCase: 'upper',
+  spaceAroundOperators: true,
+  spaceAfterComma: true,
+  spaceBeforeColon: false,
+  spaceAroundRange: true,
+  emptyLineLimit: 1,
+  trimTrailingWhitespace: true,
+  insertFinalNewline: true,
+};
+
+/** Sanitise arbitrary settings JSON into valid options (never throws). */
+export function normalizeFormatOptions(raw: unknown): FormatOptions {
+  const r = (raw ?? {}) as Partial<FormatOptions>;
+  const num = (v: unknown, lo: number, hi: number, dflt: number): number => {
+    const n = typeof v === 'number' && Number.isFinite(v) ? Math.round(v) : dflt;
+    return Math.min(hi, Math.max(lo, n));
+  };
+  const bool = (v: unknown, dflt: boolean): boolean =>
+    typeof v === 'boolean' ? v : dflt;
+  return {
+    indentSize: num(r.indentSize, 0, 8, DEFAULT_FORMAT_OPTIONS.indentSize),
+    useTabs: bool(r.useTabs, false),
+    keywordCase: r.keywordCase === 'preserve' || r.keywordCase === 'upper'
+      ? r.keywordCase : 'upper',
+    spaceAroundOperators: bool(r.spaceAroundOperators, true),
+    spaceAfterComma: bool(r.spaceAfterComma, true),
+    spaceBeforeColon: bool(r.spaceBeforeColon, false),
+    spaceAroundRange: bool(r.spaceAroundRange, true),
+    emptyLineLimit: num(r.emptyLineLimit, 0, 10, DEFAULT_FORMAT_OPTIONS.emptyLineLimit),
+    trimTrailingWhitespace: bool(r.trimTrailingWhitespace, true),
+    insertFinalNewline: bool(r.insertFinalNewline, true),
+  };
+}
+
+interface RichToken extends Token {
+  comment?: boolean;
+}
+
+/** Structural words for indentation (matched case-insensitively). */
+const STRUCT = new Set([
+  'END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY',
+  'THEN', 'LOOP', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'CASE', 'OF',
+]);
+const DISPLAY_DEDENT = new Set(['END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY']);
+const OPENERS = new Set(['THEN', 'LOOP', 'BEGIN', 'RECORD', 'REPEAT', 'DO']);
+const CLOSERS = new Set(['END', 'UNTIL']);
+const UNARY_AFTER = new Set([
+  '(', '[', ',', ';', ':=', '=', '..', 'THEN', 'DO', 'OF', 'RETURN',
+  'NOT', 'AND', 'OR', 'DIV', 'MOD', 'IN', '+', '-', '*', '/', '<', '<=',
+  '>', '>=', '<>', '#', 'BY', 'TO',
+]);
+
+function wordOf(t: RichToken): string {
+  return t.kind === 'keyword' && STRUCT.has(t.text.toUpperCase()) ? t.text.toUpperCase() : '';
+}
+
+function applyCase(text: string, mode: KeywordCase): string {
+  return mode === 'upper' ? text.toUpperCase() : text;
+}
+
+/** Promote keyword-shaped identifiers to keywords unless they resolve.
+ *
+ *  Lowercase keywords lex as identifiers, but real code also uses words
+ *  like `mod`, `In` or `Type` as identifiers (accepted by gm2). An
+ *  identifier that is declared here, is a type-expression formal, or
+ *  resolves (qualifier-aware) stays an identifier; anything else shaped
+ *  like a reserved word is a keyword. This keeps formatting token-exact.
+ */
+function classifyKeywords(tokens: RichToken[], text: string, filePath: string, docDir: string): void {
+  const unit = parseUnitText(text, filePath);
+  const declared = new Set<string>();
+  for (const { sym } of flattenUnit(unit)) {
+    declared.add(`${sym.nameRange.startLine}:${sym.nameRange.startCh}`);
+  }
+  for (const pos of procTypeFormalPositions(tokens)) declared.add(pos);
+  const prevCode = (i: number): { tok: RichToken; idx: number } | null => {
+    for (let j = i - 1; j >= 0; j--) {
+      const t = tokens[j];
+      if (!t.comment) return { tok: t, idx: j };
+    }
+    return null;
+  };
+  tokens.forEach((t, i) => {
+    if (t.comment || t.kind !== 'ident' || !M2_KEYWORDS.has(t.text.toUpperCase())) return;
+    if (declared.has(`${t.line}:${t.ch}`)) return;
+    let qualifier: string | null = null;
+    const name = t.text;
+    const dot = prevCode(i);
+    if (dot && dot.tok.text === '.') {
+      const head = prevCode(dot.idx);
+      if (!head || head.tok.kind !== 'ident') return;
+      qualifier = head.tok.text;
+    }
+    const r = resolveName(unit, docDir, t.line, t.ch, qualifier, name);
+    if (!r) t.kind = 'keyword';
+  });
+}
+/** Comment ranges (nesting-aware; Modula-2 has no string escapes). */
+function scanComments(text: string): RichToken[] {
+  const out: RichToken[] = [];
+  let i = 0;
+  let line = 0;
+  let col = 0;
+  const skipString = (quote: string): void => {
+    i++;
+    col++;
+    while (i < text.length && text[i] !== '\n' && text[i] !== '\r') {
+      if (text[i] === quote) {
+        i++;
+        col++;
+        break;
+      }
+      i++;
+      col++;
+    }
+  };
+  while (i < text.length) {
+    const c = text[i];
+    if (c === "'" || c === '"') {
+      skipString(c);
+      continue;
+    }
+    if (c === '(' && text[i + 1] === '*') {
+      const s = i;
+      const sl = line;
+      const sc = col;
+      let level = 1;
+      i += 2;
+      col += 2;
+      while (i < text.length && level > 0) {
+        if (text[i] === '(' && text[i + 1] === '*') {
+          level++;
+          i += 2;
+          col += 2;
+          continue;
+        }
+        if (text[i] === '*' && text[i + 1] === ')') {
+          level--;
+          i += 2;
+          col += 2;
+          continue;
+        }
+        if (text[i] === '\n') {
+          line++;
+          col = 0;
+        } else if (text[i] !== '\r') {
+          col++;
+        } else {
+          col = 0;
+        }
+        i++;
+      }
+      out.push({
+        text: text.slice(s, i), kind: 'ident', comment: true,
+        line: sl, ch: sc, endLine: line, endCh: col,
+        offset: s, endOffset: i,
+      });
+      continue;
+    }
+    if (c === '\n') {
+      line++;
+      col = 0;
+    } else if (c !== '\r') {
+      col++;
+    } else {
+      col = 0;
+    }
+    i++;
+  }
+  return out;
+}
+
+export function formatDocument(
+  text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): string {
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  if (text.length === 0) return '';
+  const eol = text.includes('\r\n') ? '\r\n' : '\n';
+  const dir = docDir || (filePath ? path.dirname(filePath) : '');
+  const tokens: RichToken[] = [
+    ...lex(text).filter(t => t.kind !== 'eof'),
+    ...scanComments(text),
+  ].sort((a, b) => a.offset - b.offset);
+  classifyKeywords(tokens, text, filePath, dir);
+
+  const byLine = new Map<number, RichToken[]>();
+  let maxLine = 0;
+  for (const t of tokens) {
+    maxLine = Math.max(maxLine, t.endLine);
+    const list = byLine.get(t.line) ?? [];
+    list.push(t);
+    byLine.set(t.line, list);
+  }
+  const srcLines = text.split(/\r?\n/);
+  const lastLine = Math.max(maxLine, srcLines.length - 1);
+
+  // Continuation lines of multi-line comments: verbatim, but counted.
+  const verbatim = new Set<number>();
+  for (const t of tokens) {
+    if (t.comment && t.endLine > t.line) {
+      for (let l = t.line + 1; l <= t.endLine; l++) verbatim.add(l);
+    }
+  }
+
+  const indentOf = (level: number): string =>
+    opts.useTabs ? '\t'.repeat(level) : ' '.repeat(opts.indentSize * level);
+
+  const out: string[] = [];
+  let level = 0;
+  let casePending = false;
+  let blanks = 0;
+
+  for (let i = 0; i <= lastLine; i++) {
+    const lineToks = (byLine.get(i) ?? []).filter(t => !t.comment || t.line === i);
+    const code = lineToks.filter(t => !t.comment);
+    const startsHere = lineToks.filter(t => t.comment && t.line === i);
+
+    if (verbatim.has(i)) {
+      out.push(opts.trimTrailingWhitespace ? srcLines[i]?.replace(/[ \t]+$/, '') ?? '' : srcLines[i] ?? '');
+      level = applyDelta(code, level, { casePending }).level;
+      casePending = applyDelta(code, level, { casePending }).casePending;
+      continue;
+    }
+
+    if (code.length === 0 && startsHere.length === 0) {
+      blanks++;
+      if (blanks <= opts.emptyLineLimit) out.push('');
+      continue;
+    }
+    blanks = 0;
+
+    if (code.length === 0) {
+      // Comment-only line: indent, no level change.
+      const parts = startsHere.map(c => firstFragment(c));
+      out.push(indentOf(level) + parts.join('  '));
+      continue;
+    }
+
+    const first = wordOf(code[0]);
+    const emitLevel = Math.max(0, level - (DISPLAY_DEDENT.has(first) ? 1 : 0));
+    let body = emitTokens(code, opts);
+    const trailing = startsHere.filter(c => code[0].offset < c.offset);
+    for (const c of trailing) body += '  ' + firstFragment(c);
+    let line = indentOf(emitLevel) + body;
+    if (opts.trimTrailingWhitespace) line = line.replace(/[ \t]+$/, '');
+    out.push(line);
+
+    const applied = applyDelta(code, level, { casePending });
+    level = applied.level;
+    casePending = applied.casePending;
+  }
+
+  if (!opts.insertFinalNewline) return out.join('\n').split('\n').join(eol);
+  const stripped = out.join('\n').replace(/[ \t\n]+$/, '');
+  return (stripped.length > 0 ? stripped + '\n' : '').split('\n').join(eol);
+}
+
+function firstFragment(c: RichToken): string {
+  const idx = c.text.search(/\r?\n/);
+  return idx < 0 ? c.text : c.text.slice(0, idx);
+}
+
+function applyDelta(
+  code: RichToken[], level: number, state: { casePending: boolean },
+): { level: number; casePending: boolean } {
+  let opens = 0;
+  let closes = 0;
+  let casePending = state.casePending;
+  for (const t of code) {
+    const w = wordOf(t);
+    if (!w) continue;
+    if (w === 'CASE') {
+      casePending = true;
+      continue;
+    }
+    if (w === 'OF') {
+      if (casePending) {
+        opens++;
+        casePending = false;
+      }
+      continue;
+    }
+    if (OPENERS.has(w)) {
+      opens++;
+      continue;
+    }
+    if (CLOSERS.has(w)) {
+      closes++;
+      continue;
+    }
+  }
+  return { level: Math.max(0, level + opens - closes), casePending };
+}
+
+function isWordish(t: RichToken): boolean {
+  return t.kind === 'ident' || t.kind === 'keyword' || t.kind === 'number' || t.kind === 'string';
+}
+
+function emitTokens(code: RichToken[], opts: FormatOptions): string {
+  let out = '';
+  const gap = (prev: RichToken | null, cur: RichToken): string => {
+    if (!prev) return '';
+    const pt = prev.text;
+    const ct = cur.text;
+    // Opening brackets: attached, except after a keyword (`ARRAY [`).
+    if (ct === '(') return prev.kind === 'keyword' ? ' ' : '';
+    if (ct === '[') return prev.kind === 'keyword' ? ' ' : '';
+    if (ct === '{') return '';
+    if (ct === '<*') return ' ';
+    // After openers: attached.
+    if (pt === '(' || pt === '[' || pt === '{' || pt === '<*') return '';
+    if (pt === '.' || pt === '^') return '';
+    // Closing brackets and separators: attached before...
+    if (ct === ')' || ct === ']' || ct === '}' || ct === '*>') return '';
+    if (ct === '.' || ct === '^') return '';
+    if (ct === ',') return '';
+    if (ct === ';') return '';
+    // ...with a space after closers and `;` (except before tighter tokens).
+    if (pt === ')' || pt === ']' || pt === '}' || pt === '*>' || pt === ';') {
+      return isWordish(cur) ? ' ' : '';
+    }
+    if (ct === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ',') return opts.spaceAfterComma ? ' ' : '';
+    if (ct === ':') return opts.spaceBeforeColon ? ' ' : '';
+    if (pt === ':') return ' ';
+    if (ct === '..' || pt === '..') return opts.spaceAroundRange ? ' ' : '';
+    if (ct === '+' || ct === '-') return isUnary(prev) ? '' : (opts.spaceAroundOperators ? ' ' : '');
+    if (ct === '^') return '';
+    // Symbol operators: spaced iff the setting is on.
+    if (cur.kind === 'symbol' || prev.kind === 'symbol') {
+      return opts.spaceAroundOperators ? ' ' : '';
+    }
+    // Word pairs (ident/keyword/number/string): single space.
+    return ' ';
+  };
+  let prev: RichToken | null = null;
+  for (const t of code) {
+    const text = t.kind === 'keyword' ? applyCase(t.text, opts.keywordCase) : t.text;
+    out += gap(prev, t) + text;
+    prev = t;
+  }
+  return out;
+}
+
+function isUnary(prev: RichToken | null): boolean {
+  if (!prev) return true;
+  if (prev.kind === 'keyword') return UNARY_AFTER.has(prev.text.toUpperCase());
+  if (prev.kind !== 'symbol') return false;
+  return UNARY_AFTER.has(prev.text);
+}

+ 2 - 0
extensions/modula2-language/src/m2/lexer.ts

@@ -26,6 +26,8 @@ const KEYWORDS = new Set([
 
 const MULTI = new Set(['..', ':=', '<=', '>=', '<>', '<*', '*>']);
 
+export const M2_KEYWORDS: ReadonlySet<string> = KEYWORDS;
+
 function isLetter(c: string): boolean {
   return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') || c === '_';
 }

+ 44 - 3
extensions/modula2-language/src/server.ts

@@ -2,7 +2,7 @@ import {
   createConnection, ProposedFeatures, InitializeParams, InitializeResult,
   TextDocuments, TextDocumentSyncKind, Diagnostic, DiagnosticSeverity,
   DidChangeConfigurationParams, CompletionItem, CompletionItemKind,
-  DocumentSymbol, SymbolKind, Hover, Location
+  DocumentSymbol, SymbolKind, Hover, Location, TextEdit
 } from 'vscode-languageserver/node';
 import { TextDocument } from 'vscode-languageserver-textdocument';
 import * as path from 'path';
@@ -13,6 +13,7 @@ import { pathToFileURL } from 'url';
 import { lex, Token } from './m2/lexer';
 import { M2Symbol, M2Unit, parseUnitText } from './m2/parser';
 import { analyseUnit } from './m2/analyse';
+import { DEFAULT_FORMAT_OPTIONS, FormatOptions, formatDocument, normalizeFormatOptions } from './m2/format';
 import {
   moduleExports, recordFieldsOf, resolveName, toLocation, visibleSymbols,
   findReferencesInText, LocatedOccurrence,
@@ -22,11 +23,16 @@ const connection = createConnection(ProposedFeatures.all);
 const documents = new TextDocuments(TextDocument);
 let validatorCommand = '';
 let validatorArguments: string[] = ['{file}'];
+let formatOptions: FormatOptions = { ...DEFAULT_FORMAT_OPTIONS };
 
 connection.onInitialize((params: InitializeParams): InitializeResult => {
-  const options = (params.initializationOptions || {}) as { validatorCommand?: string; validatorArguments?: string[] };
+  const options = (params.initializationOptions || {}) as {
+    validatorCommand?: string; validatorArguments?: string[];
+    format?: Partial<FormatOptions>;
+  };
   validatorCommand = options.validatorCommand || '';
   validatorArguments = options.validatorArguments || ['{file}'];
+  formatOptions = normalizeFormatOptions(options.format ?? {});
   return {
     capabilities: {
       textDocumentSync: TextDocumentSyncKind.Full,
@@ -36,14 +42,23 @@ connection.onInitialize((params: InitializeParams): InitializeResult => {
       documentSymbolProvider: true,
       referencesProvider: true,
       renameProvider: { prepareProvider: true },
+      documentFormattingProvider: true,
     },
   };
 });
 
 connection.onDidChangeConfiguration((params: DidChangeConfigurationParams) => {
-  const settings = (params.settings || {}) as { modula2?: { cocoR?: { validatorCommand?: string; validatorArguments?: string[] } } };
+  const settings = (params.settings || {}) as {
+    modula2?: {
+      cocoR?: { validatorCommand?: string; validatorArguments?: string[] };
+      format?: Partial<FormatOptions>;
+    };
+  };
   validatorCommand = settings.modula2?.cocoR?.validatorCommand || validatorCommand;
   validatorArguments = settings.modula2?.cocoR?.validatorArguments || validatorArguments;
+  if (settings.modula2?.format) {
+    formatOptions = normalizeFormatOptions(settings.modula2.format);
+  }
 });
 
 documents.onDidOpen(e => validate(e.document));
@@ -444,5 +459,31 @@ connection.onRenameRequest(params => {
   return { changes };
 });
 
+connection.onDocumentFormatting(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return null;
+  const filePath = uriToFilePath(document.uri);
+  let formatted: string;
+  try {
+    formatted = formatDocument(
+      document.getText(), formatOptions, filePath, path.dirname(filePath),
+    );
+  } catch (error) {
+    connection.console.error(`formatting failed: ${error instanceof Error ? error.message : String(error)}`);
+    return null;
+  }
+  if (formatted === document.getText()) return [];
+  const lines = document.getText().split('\n');
+  const last = lines.length - 1;
+  const edit: TextEdit = {
+    range: {
+      start: { line: 0, character: 0 },
+      end: { line: last, character: lines[last]?.length ?? 0 },
+    },
+    newText: formatted,
+  };
+  return [edit];
+});
+
 documents.listen(connection);
 connection.listen();