Переглянути джерело

Pascal LSP: formatter, semantic tokens, diagnostics, settings, snippets (P2+P3)

Eric Streit 2 днів тому
батько
коміт
bb943fd910

+ 79 - 1
extensions/pascal-language/package.json

@@ -33,6 +33,65 @@
     "configuration": {
       "title": "Pascal",
       "properties": {
+        "modula2.pascal.uppercaseKeywords": {
+          "type": "boolean",
+          "default": true,
+          "description": "Uppercase a Pascal keyword when the completed token is followed by a non-identifier character."
+        },
+        "modula2.pascal.format.indentSize": {
+          "type": "number",
+          "default": 2,
+          "description": "Indentation width applied by the Pascal formatter."
+        },
+        "modula2.pascal.format.useTabs": {
+          "type": "boolean",
+          "default": false,
+          "description": "Indent with tabs instead of spaces."
+        },
+        "modula2.pascal.format.keywordCase": {
+          "type": "string",
+          "enum": [
+            "upper",
+            "preserve"
+          ],
+          "default": "upper",
+          "description": "Keyword case applied by the formatter. Reserved words written lowercase are uppercased; identifiers that merely look like keywords are left alone."
+        },
+        "modula2.pascal.format.spaceAroundOperators": {
+          "type": "boolean",
+          "default": true,
+          "description": "Spaces around operators (:=, +, -, ...)."
+        },
+        "modula2.pascal.format.spaceAfterComma": {
+          "type": "boolean",
+          "default": true,
+          "description": "Space after commas."
+        },
+        "modula2.pascal.format.spaceBeforeColon": {
+          "type": "boolean",
+          "default": false,
+          "description": "Space before colons in declarations."
+        },
+        "modula2.pascal.format.spaceAroundRange": {
+          "type": "boolean",
+          "default": true,
+          "description": "Spaces around the .. range operator."
+        },
+        "modula2.pascal.format.emptyLineLimit": {
+          "type": "number",
+          "default": 1,
+          "description": "Maximum consecutive blank lines kept."
+        },
+        "modula2.pascal.format.trimTrailingWhitespace": {
+          "type": "boolean",
+          "default": true,
+          "description": "Remove trailing whitespace."
+        },
+        "modula2.pascal.format.insertFinalNewline": {
+          "type": "boolean",
+          "default": true,
+          "description": "End the file with a newline."
+        },
         "modula2.pascal.dialect": {
           "type": "string",
           "enum": [
@@ -54,7 +113,26 @@
           "description": "Validator command per Pascal dialect, e.g. {\"freepascal\": \"/path/to/FPC\"}. Each command receives the temporary source file as its single argument. Never point these at pcom/psc oracle compilers (they can hang)."
         }
       }
-    }
+    },
+    "grammars": [
+      {
+        "language": "pascal",
+        "scopeName": "source.pascal",
+        "path": "./syntaxes/pascal.tmLanguage.json"
+      }
+    ],
+    "snippets": [
+      {
+        "language": "pascal",
+        "path": "./snippets/pascal.json"
+      }
+    ],
+    "jsonValidation": [
+      {
+        "fileMatch": "pascal.json",
+        "url": "./pascal-schema.json"
+      }
+    ]
   },
   "scripts": {
     "clean": "rimraf out",

+ 55 - 0
extensions/pascal-language/pascal-schema.json

@@ -0,0 +1,55 @@
+{
+  "$schema": "http://json-schema.org/draft-07/schema#",
+  "title": "Pascal project",
+  "description": "Project description consumed by the Pascal IDE (build, run, tasks, wizards).",
+  "type": "object",
+  "required": ["name", "main"],
+  "properties": {
+    "name": {
+      "type": "string",
+      "minLength": 1,
+      "description": "Project name. Defaults the built executable to bin/<name>."
+    },
+    "main": {
+      "type": "string",
+      "minLength": 1,
+      "description": "Main file relative to the project root, e.g. \"src/Main.pas\"."
+    },
+    "output": {
+      "type": "string",
+      "description": "Built executable relative to the root. Defaults to bin/<name>."
+    },
+    "compiler": {
+      "type": "object",
+      "description": "Compiler selection: Free Pascal (fpc) or Blaise.",
+      "properties": {
+        "type": {
+          "type": "string",
+          "enum": ["fpc", "blaise"],
+          "description": "Compiler family."
+        },
+        "path": {
+          "type": "string",
+          "description": "Compiler executable. Defaults to \"fpc\" (or \"blaise\" for that family, or the pascal.compiler.path preference)."
+        },
+        "options": {
+          "type": "array",
+          "items": { "type": "string" },
+          "description": "Extra compiler options appended to the build command."
+        }
+      },
+      "additionalProperties": false
+    },
+    "sourceDirectories": {
+      "type": "array",
+      "items": { "type": "string" },
+      "description": "Source folders searched by the New Unit wizard. Defaults to [\"src\"]."
+    },
+    "libraryDirectories": {
+      "type": "array",
+      "items": { "type": "string" },
+      "description": "Third-party library folders searched by language services after the source folders."
+    }
+  },
+  "additionalProperties": false
+}

+ 42 - 0
extensions/pascal-language/snippets/pascal.json

@@ -0,0 +1,42 @@
+{
+  "program": {
+    "prefix": "program",
+    "body": ["PROGRAM ${1:ProgramName};", "", "BEGIN", "    ${2}", "END."],
+    "description": "Create a Pascal program"
+  },
+  "unit": {
+    "prefix": "unit",
+    "body": ["UNIT ${1:UnitName};", "", "INTERFACE", "", "${2}", "", "IMPLEMENTATION", "", "${3}", "", "BEGIN", "END."],
+    "description": "Create a Pascal unit"
+  },
+  "procedure": {
+    "prefix": "proc",
+    "body": ["PROCEDURE ${1:ProcedureName}(${2});", "BEGIN", "    ${3}", "END;"],
+    "description": "Create a Pascal procedure"
+  },
+  "function": {
+    "prefix": "func",
+    "body": ["FUNCTION ${1:FunctionName}(${2}): ${3:Integer};", "BEGIN", "    ${4}", "    ${1:FunctionName} := ${5:0};", "END;"],
+    "description": "Create a Pascal function"
+  },
+  "if": {
+    "prefix": "if",
+    "body": ["IF ${1:condition} THEN", "BEGIN", "    ${2}", "END;"],
+    "description": "Create an IF statement"
+  },
+  "for": {
+    "prefix": "for",
+    "body": ["FOR ${1:i} := ${2:start} TO ${3:end} DO", "BEGIN", "    ${4}", "END;"],
+    "description": "Create a FOR loop"
+  },
+  "while": {
+    "prefix": "while",
+    "body": ["WHILE ${1:condition} DO", "BEGIN", "    ${2}", "END;"],
+    "description": "Create a WHILE loop"
+  },
+  "record": {
+    "prefix": "record",
+    "body": ["${1:T} = RECORD", "    ${2:field}: ${3:Integer};", "END;"],
+    "description": "Create a Pascal record type"
+  }
+}

+ 161 - 0
extensions/pascal-language/src/diagnostics.ts

@@ -0,0 +1,161 @@
+import * as vscode from 'vscode';
+
+const PROGRAM_HEADER = /\b(PROGRAM|UNIT|LIBRARY)\s+([A-Za-z_][A-Za-z0-9_]*)/i;
+const END_NAME = /\bEND\s+([A-Za-z_][A-Za-z0-9_]*)\s*(\.)?/gi;
+
+interface ScanResult {
+  diagnostics: vscode.Diagnostic[];
+  code: string;
+}
+
+function positionAt(document: vscode.TextDocument, offset: number): vscode.Position {
+  return document.positionAt(Math.max(0, Math.min(offset, document.getText().length)));
+}
+
+function rangeAt(document: vscode.TextDocument, start: number, end: number): vscode.Range {
+  return new vscode.Range(positionAt(document, start), positionAt(document, end));
+}
+
+function diagnostic(
+  document: vscode.TextDocument,
+  start: number,
+  end: number,
+  message: string,
+  severity = vscode.DiagnosticSeverity.Error
+): vscode.Diagnostic {
+  const item = new vscode.Diagnostic(rangeAt(document, start, end), message, severity);
+  item.source = 'Pascal';
+  return item;
+}
+
+/**
+ * Lightweight diagnostics used alongside the Coco/R dialect validators.
+ *
+ * Only constructs recognisable without pretending to be a complete Pascal
+ * parser: unterminated comments/strings, the program/unit header, and a
+ * trailing `END name.` mismatch. Real syntax diagnostics come from the
+ * configured dialect validator.
+ */
+function scan(document: vscode.TextDocument): ScanResult {
+  const text = document.getText();
+  const diagnostics: vscode.Diagnostic[] = [];
+  let code = '';
+  let i = 0;
+
+  while (i < text.length) {
+    const ch = text[i];
+    const next = text[i + 1];
+
+    if (ch === '(' && next === '*') {
+      const start = i;
+      i += 2;
+      let depth = 1;
+      while (i < text.length && depth > 0) {
+        if (text[i] === '(' && text[i + 1] === '*') { depth++; i += 2; continue; }
+        if (text[i] === '*' && text[i + 1] === ')') { depth--; i += 2; continue; }
+        i++;
+      }
+      if (depth > 0) {
+        diagnostics.push(diagnostic(document, start, Math.min(start + 2, text.length), 'Unterminated Pascal comment.'));
+      }
+      code += '  ';
+      continue;
+    }
+
+    if (ch === '{') {
+      const start = i;
+      i++;
+      let closed = false;
+      while (i < text.length) {
+        if (text[i] === '}') { i++; closed = true; break; }
+        i++;
+      }
+      if (!closed) {
+        diagnostics.push(diagnostic(document, start, Math.min(start + 1, text.length), 'Unterminated Pascal comment.'));
+      }
+      code += '  ';
+      continue;
+    }
+
+    if (ch === "'") {
+      const start = i;
+      i++;
+      let closed = false;
+      while (i < text.length && text[i] !== '\n') {
+        if (text[i] === "'") {
+          if (text[i + 1] === "'") { i += 2; continue; }
+          i++;
+          closed = true;
+          break;
+        }
+        i++;
+      }
+      if (!closed) {
+        diagnostics.push(diagnostic(document, start, Math.min(start + 1, text.length), 'Unterminated string literal.'));
+      }
+      code += '  ';
+      continue;
+    }
+
+    code += ch;
+    i++;
+  }
+
+  const header = PROGRAM_HEADER.exec(code);
+  if (!header) {
+    diagnostics.push(diagnostic(document, 0, Math.min(code.length, 1), 'Expected a PROGRAM, UNIT or LIBRARY header.'));
+    return { diagnostics, code };
+  }
+
+  const unitName = header[2];
+  const headerEnd = header.index + header[0].length;
+  const semicolon = code.indexOf(';', headerEnd);
+  if (semicolon < 0) {
+    diagnostics.push(diagnostic(document, headerEnd, Math.min(headerEnd + unitName.length, code.length), 'Expected ";" after the program header.'));
+  }
+
+  END_NAME.lastIndex = 0;
+  let match: RegExpExecArray | null = null;
+  let lastEnd: RegExpExecArray | null = null;
+  while ((match = END_NAME.exec(code)) !== null) {
+    lastEnd = match;
+  }
+
+  // Only units/library headers name their END; plain PROGRAM ... BEGIN/END. has none.
+  const kind = header[1].toUpperCase();
+  if (lastEnd && (kind === 'UNIT' || kind === 'LIBRARY')) {
+    const endName = lastEnd[1];
+    const nameStart = lastEnd.index + lastEnd[0].indexOf(endName);
+    if (endName.toUpperCase() !== unitName.toUpperCase()) {
+      diagnostics.push(diagnostic(document, nameStart, nameStart + endName.length, `Unit name "${unitName}" expected after END, found "${endName}".`));
+    }
+    if (!lastEnd[2]) {
+      const dot = lastEnd.index + lastEnd[0].length;
+      diagnostics.push(diagnostic(document, dot, Math.min(dot + 1, code.length), `Expected "." after END ${endName}.`));
+    }
+  }
+
+  return { diagnostics, code };
+}
+
+export function createDiagnostics(context: vscode.ExtensionContext): void {
+  const collection = vscode.languages.createDiagnosticCollection('pascal');
+  context.subscriptions.push(collection);
+
+  const update = (document: vscode.TextDocument): void => {
+    if (document.languageId !== 'pascal') {
+      return;
+    }
+    collection.set(document.uri, scan(document).diagnostics);
+  };
+
+  for (const document of vscode.workspace.textDocuments) {
+    update(document);
+  }
+
+  context.subscriptions.push(
+    vscode.workspace.onDidOpenTextDocument(update),
+    vscode.workspace.onDidChangeTextDocument(event => update(event.document)),
+    vscode.workspace.onDidCloseTextDocument(document => collection.delete(document.uri))
+  );
+}

+ 80 - 1
extensions/pascal-language/src/extension.ts

@@ -1,7 +1,23 @@
 import * as vscode from 'vscode';
+import { createDiagnostics } from './diagnostics';
 import { LanguageClient, LanguageClientOptions, ServerOptions, TransportKind } from 'vscode-languageclient/node';
 
+const KEYWORDS = new Set([
+  'AND', 'ARRAY', 'BEGIN', 'CASE', 'CONST', 'DIV', 'DO', 'DOWNTO',
+  'ELSE', 'END', 'FILE', 'FOR', 'FUNCTION', 'GOTO', 'IF', 'IN',
+  'LABEL', 'MOD', 'NIL', 'NOT', 'OF', 'OR', 'PACKED', 'PROCEDURE',
+  'PROGRAM', 'RECORD', 'REPEAT', 'SET', 'THEN', 'TO', 'TYPE', 'UNIT',
+  'UNTIL', 'USES', 'VAR', 'WHILE', 'WITH', 'FORWARD'
+]);
+
+let applyingEdit = false;
+
+function isIdentifierCharacter(ch: string | undefined): boolean {
+  return !!ch && /[A-Za-z0-9_]/.test(ch);
+}
+
 export function activate(context: vscode.ExtensionContext): void {
+  createDiagnostics(context);
   const serverModule = context.asAbsolutePath('out/server.js');
   const serverOptions: ServerOptions = {
     run: { module: serverModule, transport: TransportKind.ipc },
@@ -15,12 +31,75 @@ export function activate(context: vscode.ExtensionContext): void {
     synchronize: { configurationSection: 'modula2' },
     initializationOptions: {
       pascalDialect: cfg.get<string>('pascal.dialect', 'freepascal'),
-      pascalValidators: cfg.get<object>('pascal.validators', {})
+      pascalValidators: cfg.get<object>('pascal.validators', {}),
+      format: {
+        indentSize: cfg.get<number>('pascal.format.indentSize', 2),
+        useTabs: cfg.get<boolean>('pascal.format.useTabs', false),
+        keywordCase: cfg.get<string>('pascal.format.keywordCase', 'upper'),
+        spaceAroundOperators: cfg.get<boolean>('pascal.format.spaceAroundOperators', true),
+        spaceAfterComma: cfg.get<boolean>('pascal.format.spaceAfterComma', true),
+        spaceBeforeColon: cfg.get<boolean>('pascal.format.spaceBeforeColon', false),
+        spaceAroundRange: cfg.get<boolean>('pascal.format.spaceAroundRange', true),
+        emptyLineLimit: cfg.get<number>('pascal.format.emptyLineLimit', 1),
+        trimTrailingWhitespace: cfg.get<boolean>('pascal.format.trimTrailingWhitespace', true),
+        insertFinalNewline: cfg.get<boolean>('pascal.format.insertFinalNewline', true)
+      }
     }
   };
   const client = new LanguageClient('pascalCocoR', 'Pascal Coco/R Language Server', serverOptions, clientOptions);
   context.subscriptions.push(client);
   void client.start();
+  context.subscriptions.push(vscode.workspace.onDidChangeTextDocument(async event => {
+    if (applyingEdit || event.document.languageId !== 'pascal' ||
+        !vscode.workspace.getConfiguration('modula2').get<boolean>('pascal.uppercaseKeywords', true)) {
+      return;
+    }
+
+    const editor = vscode.window.activeTextEditor;
+    if (!editor || editor.document.uri.toString() !== event.document.uri.toString()) {
+      return;
+    }
+
+    for (const change of event.contentChanges) {
+      const position = change.range.start.translate(0, change.text.length);
+      const line = event.document.lineAt(position.line).text;
+      let start = position.character;
+
+      while (start > 0 && isIdentifierCharacter(line[start - 1])) {
+        start--;
+      }
+
+      let end = position.character;
+      while (end < line.length && isIdentifierCharacter(line[end])) {
+        end++;
+      }
+
+      // Do not modify a token while it is still being typed.
+      // This is what prevents "beginning" from becoming "BEGINning".
+      if (end >= line.length || isIdentifierCharacter(line[end])) {
+        continue;
+      }
+
+      const token = line.slice(start, end);
+      const upper = token.toUpperCase();
+
+      if (!KEYWORDS.has(upper) || token === upper) {
+        continue;
+      }
+
+      applyingEdit = true;
+      try {
+        await editor.edit(builder => {
+          builder.replace(
+            new vscode.Range(position.line, start, position.line, end),
+            upper
+          );
+        });
+      } finally {
+        applyingEdit = false;
+      }
+    }
+  }));
 }
 
 export function deactivate(): void {}

+ 539 - 0
extensions/pascal-language/src/pascalFormat.ts

@@ -0,0 +1,539 @@
+/** Token-based Pascal document formatter with user-settable rules.
+ *
+ *  Twin of the Modula-2 formatter: line structure is preserved
+ *  (statements are never joined or split), while indentation, keyword
+ *  case, intra-line spacing, blank lines and trailing whitespace are
+ *  normalised per `FormatOptions`. Comments are preserved verbatim
+ *  (only repositioned); continuation lines of multi-line comments are
+ *  emitted as-is. Best-effort on broken code: always terminates and
+ *  never drops tokens.
+ */
+
+import * as path from 'path';
+import { parsePascal, parseUses, PasSymbol } from './pascalSymbols';
+import { resolveAt } from './pascalResolve';
+
+export type KeywordCase = 'upper' | 'preserve';
+
+export interface FormatOptions {
+  indentSize: number;
+  useTabs: boolean;
+  keywordCase: KeywordCase;
+  spaceAroundOperators: boolean;
+  spaceAfterComma: boolean;
+  spaceBeforeColon: boolean;
+  spaceAroundRange: boolean;
+  emptyLineLimit: number;
+  trimTrailingWhitespace: boolean;
+  insertFinalNewline: boolean;
+}
+
+export const DEFAULT_FORMAT_OPTIONS: FormatOptions = {
+  indentSize: 2,
+  useTabs: false,
+  keywordCase: 'upper',
+  spaceAroundOperators: true,
+  spaceAfterComma: true,
+  spaceBeforeColon: false,
+  spaceAroundRange: true,
+  emptyLineLimit: 1,
+  trimTrailingWhitespace: true,
+  insertFinalNewline: true,
+};
+
+/** Sanitise arbitrary settings JSON into valid options (never throws). */
+export function normalizeFormatOptions(raw: unknown): FormatOptions {
+  const r = (raw ?? {}) as Partial<FormatOptions>;
+  const num = (v: unknown, lo: number, hi: number, dflt: number): number => {
+    const n = typeof v === 'number' && Number.isFinite(v) ? Math.round(v) : dflt;
+    return Math.min(hi, Math.max(lo, n));
+  };
+  const bool = (v: unknown, dflt: boolean): boolean =>
+    typeof v === 'boolean' ? v : dflt;
+  return {
+    indentSize: num(r.indentSize, 0, 8, DEFAULT_FORMAT_OPTIONS.indentSize),
+    useTabs: bool(r.useTabs, false),
+    keywordCase: r.keywordCase === 'preserve' || r.keywordCase === 'upper'
+      ? r.keywordCase : 'upper',
+    spaceAroundOperators: bool(r.spaceAroundOperators, true),
+    spaceAfterComma: bool(r.spaceAfterComma, true),
+    spaceBeforeColon: bool(r.spaceBeforeColon, false),
+    spaceAroundRange: bool(r.spaceAroundRange, true),
+    emptyLineLimit: num(r.emptyLineLimit, 0, 10, DEFAULT_FORMAT_OPTIONS.emptyLineLimit),
+    trimTrailingWhitespace: bool(r.trimTrailingWhitespace, true),
+    insertFinalNewline: bool(r.insertFinalNewline, true),
+  };
+}
+
+type TokenKind = 'keyword' | 'ident' | 'number' | 'string' | 'symbol';
+
+interface RichToken {
+  text: string;
+  kind: TokenKind;
+  comment?: boolean;
+  line: number; ch: number; endLine: number; endCh: number;
+  offset: number; endOffset: number;
+}
+
+/** Reserved words matched case-insensitively by the formatter lexer. */
+const PASCAL_KEYWORDS = new Set([
+  'AND', 'ARRAY', 'BEGIN', 'CASE', 'CONST', 'DIV', 'DO', 'DOWNTO',
+  'ELSE', 'END', 'EXCEPT', 'EXPORTS', 'FILE', 'FINALIZATION', 'FINALLY',
+  'FOR', 'FORWARD', 'FUNCTION', 'GOTO', 'IF', 'IMPLEMENTATION', 'IN',
+  'INHERITED', 'INITIALIZATION', 'INTERFACE', 'LABEL', 'LIBRARY', 'MOD',
+  'NIL', 'NOT', 'OF', 'OR', 'PACKED', 'PROCEDURE', 'PROGRAM', 'RECORD',
+  'REPEAT', 'RESOURCESTRING', 'SET', 'SHL', 'SHR', 'THEN', 'THREADVAR',
+  'TO', 'TRY', 'TYPE', 'UNIT', 'UNTIL', 'USES', 'VAR', 'WHILE', 'WITH', 'XOR',
+]);
+
+const isWordStart = (c: string) => /[A-Za-z_]/.test(c);
+const isWordChar = (c: string) => /[A-Za-z0-9_]/.test(c);
+const isDigit = (c: string) => /[0-9]/.test(c);
+
+/** Lex code tokens (comments land in `comments`, strings are skipped). */
+function lexCode(text: string): { tokens: RichToken[]; comments: RichToken[] } {
+  const tokens: RichToken[] = [];
+  const comments: RichToken[] = [];
+  let i = 0;
+  let line = 0;
+  let col = 0;
+  const push = (t: string, kind: TokenKind, sl: number, sc: number, so: number): void => {
+    tokens.push({ text: t, kind, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
+  };
+  const pushComment = (t: string, sl: number, sc: number, so: number): void => {
+    comments.push({ text: t, kind: 'ident', comment: true, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
+  };
+  const step = (n: number): void => {
+    for (let k = 0; k < n; k++) {
+      if (text[i] === '\n') { line++; col = 0; }
+      else if (text[i] !== '\r') { col++; }
+      else { col = 0; }
+      i++;
+    }
+  };
+  while (i < text.length) {
+    const c = text[i];
+    const nx = i + 1 < text.length ? text[i + 1] : '';
+    if (c === '\n' || c === '\r' || c === ' ' || c === '\t') { step(1); continue; }
+    // Nested (* *) comment.
+    if (c === '(' && nx === '*') {
+      const sl = line; const sc = col; const so = i;
+      let level = 1;
+      step(2);
+      while (i < text.length && level > 0) {
+        if (text[i] === '(' && text[i + 1] === '*') { level++; step(2); continue; }
+        if (text[i] === '*' && text[i + 1] === ')') { level--; step(2); continue; }
+        step(1);
+      }
+      pushComment(text.slice(so, i), sl, sc, so);
+      continue;
+    }
+    // { } comment (possibly multi-line, e.g. {$directives}).
+    if (c === '{') {
+      const sl = line; const sc = col; const so = i;
+      step(1);
+      while (i < text.length && text[i] !== '}') step(1);
+      if (i < text.length) step(1);
+      pushComment(text.slice(so, i), sl, sc, so);
+      continue;
+    }
+    // // comment to end of line.
+    if (c === '/' && nx === '/') {
+      const sl = line; const sc = col; const so = i;
+      while (i < text.length && text[i] !== '\n' && text[i] !== '\r') step(1);
+      pushComment(text.slice(so, i), sl, sc, so);
+      continue;
+    }
+    // '...' string with '' escape (unterminated runs to end of line).
+    if (c === "'") {
+      const sl = line; const sc = col; const so = i;
+      step(1);
+      while (i < text.length && text[i] !== '\n' && text[i] !== '\r') {
+        if (text[i] === "'") {
+          if (text[i + 1] === "'") { step(2); continue; }
+          step(1);
+          break;
+        }
+        step(1);
+      }
+      push(text.slice(so, i), 'string', sl, sc, so);
+      continue;
+    }
+    if (isWordStart(c)) {
+      const sl = line; const sc = col; const so = i;
+      while (i < text.length && isWordChar(text[i])) step(1);
+      const word = text.slice(so, i);
+      push(word, PASCAL_KEYWORDS.has(word.toUpperCase()) ? 'keyword' : 'ident', sl, sc, so);
+      continue;
+    }
+    // Numbers: $hex, &octal, %binary, decimal with optional fraction/exponent.
+    if (isDigit(c) || ((c === '$' || c === '&' || c === '%') && /[0-9A-Fa-f]/.test(nx))) {
+      const sl = line; const sc = col; const so = i;
+      if (c === '$' || c === '&' || c === '%') {
+        step(1);
+        while (i < text.length && /[0-9A-Fa-f]/.test(text[i])) step(1);
+      } else {
+        while (i < text.length && isDigit(text[i])) step(1);
+        if (text[i] === '.' && text[i + 1] !== '.' && isDigit(text[i + 1] ?? '')) {
+          step(1);
+          while (i < text.length && isDigit(text[i])) step(1);
+        }
+        if (text[i] === 'E' || text[i] === 'e') {
+          let j = i + 1;
+          if (text[j] === '+' || text[j] === '-') j++;
+          if (isDigit(text[j] ?? '')) {
+            while (j < text.length && isDigit(text[j])) j++;
+            const span = j - i;
+            step(span);
+          }
+        }
+      }
+      push(text.slice(so, i), 'number', sl, sc, so);
+      continue;
+    }
+    // Multi-char symbols.
+    const sl = line; const sc = col; const so = i;
+    const two = text.slice(i, i + 2);
+    if (two === ':=' || two === '<=' || two === '>=' || two === '<>' || two === '..') {
+      step(2);
+      push(two, 'symbol', sl, sc, so);
+      continue;
+    }
+    step(1);
+    push(c, 'symbol', sl, sc, so);
+  }
+  return { tokens, comments };
+}
+
+/** Structural words for indentation (matched case-insensitively). */
+const STRUCT = new Set([
+  'END', 'ELSE', 'UNTIL', 'EXCEPT', 'FINALLY',
+  'THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'CASE', 'OF', 'TRY',
+]);
+const DISPLAY_DEDENT = new Set(['END', 'ELSE', 'UNTIL', 'EXCEPT', 'FINALLY']);
+const OPENERS = new Set(['THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'TRY']);
+const CLOSERS = new Set(['END', 'UNTIL']);
+const UNARY_AFTER = new Set([
+  '(', '[', ',', ';', ':=', '=', '..', 'THEN', 'DO', 'OF',
+  'NOT', 'AND', 'OR', 'DIV', 'MOD', 'IN', '+', '-', '*', '/', '<', '<=',
+  '>', '>=', '<>', '@', 'BY', 'TO', 'DOWNTO',
+]);
+
+function wordOf(t: RichToken): string {
+  return t.kind === 'keyword' && STRUCT.has(t.text.toUpperCase()) ? t.text.toUpperCase() : '';
+}
+
+function applyCase(text: string, mode: KeywordCase): string {
+  return mode === 'upper' ? text.toUpperCase() : text;
+}
+
+/** Promote keyword-shaped identifiers to keywords unless they resolve.
+ *
+ *  Lowercase keywords lex as identifiers, but real code also uses words
+ *  like `to` or `in` as identifiers. An identifier that is declared here
+ *  or resolves (qualifier-aware) stays an identifier; anything else
+ *  shaped like a reserved word is a keyword. This keeps formatting
+ *  token-exact.
+ */
+function classifyKeywords(tokens: RichToken[], text: string, filePath: string, docDir: string): void {
+  let symbols: PasSymbol[];
+  try {
+    symbols = parsePascal(text);
+  } catch {
+    return;
+  }
+  const uses = parseUses(text);
+  const declared = new Set<string>();
+  const flatten = (ss: PasSymbol[]): void => {
+    for (const s of ss) {
+      if (s.name) declared.add(`${s.line}:${s.ch}`);
+      flatten(s.children);
+    }
+  };
+  flatten(symbols);
+  const prevCode = (i: number): { tok: RichToken; idx: number } | null => {
+    for (let j = i - 1; j >= 0; j--) {
+      const t = tokens[j];
+      if (!t.comment) return { tok: t, idx: j };
+    }
+    return null;
+  };
+  tokens.forEach((t, i) => {
+    if (t.comment || t.kind !== 'ident' || !PASCAL_KEYWORDS.has(t.text.toUpperCase())) return;
+    if (declared.has(`${t.line}:${t.ch}`)) return;
+    let qualifier: string | null = null;
+    const dot = prevCode(i);
+    if (dot && dot.tok.text === '.') {
+      const head = prevCode(dot.idx);
+      if (!head || head.tok.kind !== 'ident') return;
+      qualifier = head.tok.text;
+    }
+    const r = resolveAt(symbols, uses, filePath, docDir, t.line, t.ch, qualifier, t.text);
+    if (!r) t.kind = 'keyword';
+  });
+}
+
+export function formatDocument(
+  text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): string {
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  const { entries, eol } = formatLineEntries(text, opts, filePath, docDir);
+  const joined = entries.map(l => l.text).join('\n');
+  if (!opts.insertFinalNewline) return joined.split('\n').join(eol);
+  return (joined.length > 0 ? joined + '\n' : '').split('\n').join(eol);
+}
+
+/** One formatted output line and the 0-based input line it came from. */
+export interface FormattedLine {
+  src: number;
+  text: string;
+}
+
+/** One edit for `textDocument/rangeFormatting`, directly mappable to LSP. */
+export interface RangeEdit {
+  startLine: number;
+  startCh: number;
+  endLine: number;
+  endCh: number;
+  newText: string;
+}
+
+/** Format a line range with full-document context. Never throws. */
+export function formatRangeEdits(
+  text: string, startLine: number, endLine: number,
+  rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): RangeEdit[] {
+  if (text.length === 0) return [];
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  const srcLines = text.split('\n');
+  const srcCount = srcLines.length;
+  const from = Math.min(Math.max(0, Math.min(startLine, endLine)), srcCount - 1);
+  const to = Math.min(Math.max(Math.max(startLine, endLine), from), srcCount - 1);
+  const { entries } = formatLineEntries(text, rawOptions, filePath, docDir);
+  const bySrc = new Map<number, string>();
+  for (const e of entries) {
+    if (e.src >= from && e.src <= to && !bySrc.has(e.src)) bySrc.set(e.src, e.text);
+  }
+  const edits: RangeEdit[] = [];
+  let runFrom = -1;
+  const parts: string[] = [];
+  let tailNewlined = false;
+  const flush = (endExclusive: number): void => {
+    if (runFrom < 0) return;
+    const atEof = endExclusive >= srcCount;
+    if (atEof && parts.length === 0 && runFrom > 0) {
+      const prevLen = srcLines[runFrom - 1].length;
+      edits.push({
+        startLine: runFrom - 1, startCh: prevLen,
+        endLine: srcCount - 1, endCh: srcLines[srcCount - 1].length,
+        newText: opts.insertFinalNewline && entries.length > 0 ? '\n' : '',
+      });
+      runFrom = -1;
+      return;
+    }
+    let newText = parts.join('\n');
+    if (parts.length > 0 && (atEof ? opts.insertFinalNewline : true)) newText += '\n';
+    if (atEof && newText.endsWith('\n')) tailNewlined = true;
+    const end = atEof
+      ? { line: srcCount - 1, ch: srcLines[srcCount - 1].length }
+      : { line: endExclusive, ch: 0 };
+    edits.push({
+      startLine: runFrom, startCh: 0,
+      endLine: end.line, endCh: end.ch, newText,
+    });
+    runFrom = -1;
+    parts.length = 0;
+  };
+  for (let s = from; s <= to; s++) {
+    const want = bySrc.get(s);
+    if (want !== undefined && want === srcLines[s]) {
+      flush(s);
+      continue;
+    }
+    if (runFrom < 0) runFrom = s;
+    if (want !== undefined) parts.push(want);
+  }
+  flush(to + 1);
+  if (to === srcCount - 1 && opts.insertFinalNewline && !text.endsWith('\n') &&
+      entries.length > 0 && !tailNewlined) {
+    const last = srcLines[srcCount - 1].length;
+    edits.push({
+      startLine: srcCount - 1, startCh: last,
+      endLine: srcCount - 1, endCh: last, newText: '\n',
+    });
+  }
+  return edits;
+}
+
+function formatLineEntries(
+  text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): { entries: FormattedLine[]; eol: string } {
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  if (text.length === 0) return { entries: [], eol: '\n' };
+  const eol = text.includes('\r\n') ? '\r\n' : '\n';
+  const dir = docDir || (filePath ? path.dirname(filePath) : '');
+  const { tokens, comments } = lexCode(text);
+  const all: RichToken[] = [...tokens, ...comments].sort((a, b) => a.offset - b.offset);
+  classifyKeywords(all, text, filePath, dir);
+
+  const byLine = new Map<number, RichToken[]>();
+  let maxLine = 0;
+  for (const t of all) {
+    maxLine = Math.max(maxLine, t.endLine);
+    const list = byLine.get(t.line) ?? [];
+    list.push(t);
+    byLine.set(t.line, list);
+  }
+  const srcLines = text.split(/\r?\n/);
+  const lastLine = Math.max(maxLine, srcLines.length - 1);
+
+  const verbatim = new Set<number>();
+  for (const t of all) {
+    if (t.comment && t.endLine > t.line) {
+      for (let l = t.line + 1; l <= t.endLine; l++) verbatim.add(l);
+    }
+  }
+
+  const indentOf = (level: number): string =>
+    opts.useTabs ? '\t'.repeat(level) : ' '.repeat(opts.indentSize * level);
+
+  const entries: FormattedLine[] = [];
+  let level = 0;
+  let casePending = false;
+  let blanks = 0;
+
+  for (let i = 0; i <= lastLine; i++) {
+    const lineToks = (byLine.get(i) ?? []).filter(t => !t.comment || t.line === i);
+    const code = lineToks.filter(t => !t.comment);
+    const startsHere = lineToks.filter(t => t.comment && t.line === i);
+
+    if (verbatim.has(i)) {
+      entries.push({ src: i, text: opts.trimTrailingWhitespace ? srcLines[i]?.replace(/[ \t]+$/, '') ?? '' : srcLines[i] ?? '' });
+      const applied = applyDelta(code, level, { casePending });
+      level = applied.level;
+      casePending = applied.casePending;
+      continue;
+    }
+
+    if (code.length === 0 && startsHere.length === 0) {
+      blanks++;
+      if (blanks <= opts.emptyLineLimit) entries.push({ src: i, text: '' });
+      continue;
+    }
+    blanks = 0;
+
+    if (code.length === 0) {
+      const parts = startsHere.map(c => firstFragment(c));
+      entries.push({ src: i, text: indentOf(level) + parts.join('  ') });
+      continue;
+    }
+
+    const first = wordOf(code[0]);
+    const emitLevel = Math.max(0, level - (DISPLAY_DEDENT.has(first) ? 1 : 0));
+    let body = emitTokens(code, opts);
+    const trailing = startsHere.filter(c => code[0].offset < c.offset);
+    for (const c of trailing) body += '  ' + firstFragment(c);
+    let lineText = indentOf(emitLevel) + body;
+    if (opts.trimTrailingWhitespace) lineText = lineText.replace(/[ \t]+$/, '');
+    entries.push({ src: i, text: lineText });
+
+    const applied = applyDelta(code, level, { casePending });
+    level = applied.level;
+    casePending = applied.casePending;
+  }
+
+  if (!opts.insertFinalNewline) return { entries, eol };
+  while (entries.length > 0 && /^[ \t]*$/.test(entries[entries.length - 1].text)) {
+    entries.pop();
+  }
+  if (entries.length > 0) {
+    entries[entries.length - 1].text = entries[entries.length - 1].text.replace(/[ \t]+$/, '');
+  }
+  return { entries, eol };
+}
+
+function firstFragment(c: RichToken): string {
+  const idx = c.text.search(/\r?\n/);
+  return idx < 0 ? c.text : c.text.slice(0, idx);
+}
+
+function applyDelta(
+  code: RichToken[], level: number, state: { casePending: boolean },
+): { level: number; casePending: boolean } {
+  let opens = 0;
+  let closes = 0;
+  let casePending = state.casePending;
+  for (const t of code) {
+    const w = wordOf(t);
+    if (!w) continue;
+    if (w === 'CASE') {
+      casePending = true;
+      continue;
+    }
+    if (w === 'OF') {
+      if (casePending) {
+        opens++;
+        casePending = false;
+      }
+      continue;
+    }
+    if (OPENERS.has(w)) {
+      opens++;
+      continue;
+    }
+    if (CLOSERS.has(w)) {
+      closes++;
+      continue;
+    }
+  }
+  return { level: Math.max(0, level + opens - closes), casePending };
+}
+
+function isWordish(t: RichToken): boolean {
+  return t.kind === 'ident' || t.kind === 'keyword' || t.kind === 'number' || t.kind === 'string';
+}
+
+function emitTokens(code: RichToken[], opts: FormatOptions): string {
+  let out = '';
+  const gap = (prev: RichToken | null, cur: RichToken): string => {
+    if (!prev) return '';
+    const pt = prev.text;
+    const ct = cur.text;
+    if (ct === '(') return prev.kind === 'keyword' ? ' ' : '';
+    if (ct === '[') return prev.kind === 'keyword' ? ' ' : '';
+    if (pt === '(' || pt === '[') return '';
+    if (pt === '.' || pt === '^' || pt === '@') return '';
+    if (ct === ')' || ct === ']') return '';
+    if (ct === '.' || ct === '^') return '';
+    if (ct === ',') return '';
+    if (ct === ';') return '';
+    if (pt === ')' || pt === ']' || pt === ';') {
+      return isWordish(cur) ? ' ' : '';
+    }
+    if (ct === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ',') return opts.spaceAfterComma ? ' ' : '';
+    if (ct === ':') return opts.spaceBeforeColon ? ' ' : '';
+    if (pt === ':') return ' ';
+    if (ct === '..' || pt === '..') return opts.spaceAroundRange ? ' ' : '';
+    if (ct === '+' || ct === '-') return isUnary(prev) ? '' : (opts.spaceAroundOperators ? ' ' : '');
+    if (cur.kind === 'symbol' || prev.kind === 'symbol') {
+      return opts.spaceAroundOperators ? ' ' : '';
+    }
+    return ' ';
+  };
+  let prev: RichToken | null = null;
+  for (const t of code) {
+    const text = t.kind === 'keyword' ? applyCase(t.text, opts.keywordCase) : t.text;
+    out += gap(prev, t) + text;
+    prev = t;
+  }
+  return out;
+}
+
+function isUnary(prev: RichToken | null): boolean {
+  if (!prev) return true;
+  if (prev.kind === 'keyword') return UNARY_AFTER.has(prev.text.toUpperCase());
+  if (prev.kind !== 'symbol') return false;
+  return UNARY_AFTER.has(prev.text);
+}

+ 158 - 0
extensions/pascal-language/src/pascalSemantic.ts

@@ -0,0 +1,158 @@
+/** Semantic tokens for Pascal editors (LSP `textDocument/semanticTokens`).
+ *
+ *  Declarations come from the symbol tree; identifier *uses* are resolved
+ *  through the same machinery as hover/definition, so variables,
+ *  parameters, fields, enum members and cross-unit names highlight by
+ *  meaning, not just by spelling. Anything unresolvable is skipped, so
+ *  broken code under editing degrades to TextMate highlighting.
+ */
+
+import { parsePascal, parseUses, PasSymbol } from './pascalSymbols';
+import { identifierOccurrences, resolveAt } from './pascalResolve';
+
+export const SEMANTIC_TOKEN_TYPES = [
+  'namespace', // units
+  'type', // declared types
+  'function', // procedures and functions
+  'variable', // variables and constants
+  'parameter', // parameters
+  'property', // record fields
+  'enumMember', // enumeration literals
+];
+
+export const SEMANTIC_TOKEN_MODIFIERS = ['readonly'];
+
+export function semanticTokensLegend(): { tokenTypes: string[]; tokenModifiers: string[] } {
+  return { tokenTypes: [...SEMANTIC_TOKEN_TYPES], tokenModifiers: [...SEMANTIC_TOKEN_MODIFIERS] };
+}
+
+/** Predeclared type names (resolve to no declaration; still typed). */
+const TYPE_BUILTINS = new Set([
+  'INTEGER', 'SHORTINT', 'SMALLINT', 'LONGINT', 'INT64', 'QWORD',
+  'CARDINAL', 'NATURAL', 'POSITIVE', 'BYTE', 'WORD', 'DWORD',
+  'REAL', 'SINGLE', 'DOUBLE', 'EXTENDED', 'LONGREAL', 'COMP',
+  'CHAR', 'WIDECHAR', 'BOOLEAN', 'BYTEBOOL', 'STRING', 'TEXT',
+]);
+
+interface RawToken {
+  line: number;
+  ch: number;
+  length: number;
+  type: number;
+  modifiers: number;
+}
+
+function flatten(symbols: PasSymbol[], chain: PasSymbol[], out: { sym: PasSymbol; chain: PasSymbol[] }[]): void {
+  for (const s of symbols) {
+    out.push({ sym: s, chain: [...chain] });
+    if (s.children.length > 0) flatten(s.children, [...chain, s], out);
+  }
+}
+
+/** LSP delta-encoded token data for a document. Never throws on broken input. */
+export function computeSemanticTokens(text: string, filePath: string, docDir: string): number[] {
+  const symbols = parsePascal(text);
+  const uses = parseUses(text);
+  const out: RawToken[] = [];
+  const declared = new Set<string>();
+  const flat: { sym: PasSymbol; chain: PasSymbol[] }[] = [];
+  flatten(symbols, [], flat);
+  for (const { sym, chain } of flat) {
+    if (!sym.name) continue;
+    // The file's own program/unit symbol is the root scope, not a declaration.
+    if (sym.kind === 'module' && chain.length === 0) continue;
+    declared.add(`${sym.line}:${sym.ch}`);
+    const mapped = symbolToken(sym, chain);
+    if (!mapped) continue;
+    out.push({
+      line: sym.line,
+      ch: sym.ch,
+      length: Math.max(1, sym.endCh - sym.ch),
+      type: mapped.type,
+      modifiers: mapped.modifiers,
+    });
+  }
+  for (const occ of identifierOccurrences(text)) {
+    if (declared.has(`${occ.line}:${occ.ch}`)) continue;
+    const resolved = resolveAt(symbols, uses, filePath, docDir, occ.line, occ.ch, occ.qualifier, occ.name);
+    if (resolved) {
+      const kind = useTokenKind(resolved.sym);
+      if (kind === null) continue;
+      out.push({
+        line: occ.line,
+        ch: occ.ch,
+        length: Math.max(1, occ.endCh - occ.ch),
+        type: kind,
+        modifiers: 0,
+      });
+      continue;
+    }
+    if (TYPE_BUILTINS.has(occ.name.toUpperCase())) {
+      out.push({
+        line: occ.line,
+        ch: occ.ch,
+        length: Math.max(1, occ.endCh - occ.ch),
+        type: typeIndex('type'),
+        modifiers: 0,
+      });
+    }
+  }
+  out.sort((a, b) => a.line - b.line || a.ch - b.ch);
+  const data: number[] = [];
+  let prevLine = 0;
+  let prevCh = 0;
+  for (const t of out) {
+    data.push(
+      t.line - prevLine,
+      t.line === prevLine ? t.ch - prevCh : t.ch,
+      t.length, t.type, t.modifiers,
+    );
+    prevLine = t.line;
+    prevCh = t.ch;
+  }
+  return data;
+}
+
+function typeIndex(name: string): number {
+  const i = SEMANTIC_TOKEN_TYPES.indexOf(name);
+  return i >= 0 ? i : 0;
+}
+
+function symbolToken(
+  sym: PasSymbol, chain: PasSymbol[],
+): { type: number; modifiers: number } | null {
+  switch (sym.kind) {
+    case 'module': return { type: typeIndex('namespace'), modifiers: 0 };
+    case 'procedure':
+    case 'function': return { type: typeIndex('function'), modifiers: 0 };
+    case 'variable': return { type: typeIndex('variable'), modifiers: 0 };
+    case 'parameter': return { type: typeIndex('parameter'), modifiers: 0 };
+    case 'field': return { type: typeIndex('property'), modifiers: 0 };
+    case 'constant': {
+      // Enumeration literals live under their type; other constants read-only.
+      const underType = chain.length > 0 && chain[chain.length - 1].kind === 'type';
+      return underType
+        ? { type: typeIndex('enumMember'), modifiers: 0 }
+        : { type: typeIndex('variable'), modifiers: 1 };
+    }
+    case 'type': return { type: typeIndex('type'), modifiers: 0 };
+    default: return null;
+  }
+}
+
+/** Token type for a resolved identifier use (null = no token). */
+function useTokenKind(sym: PasSymbol): number | null {
+  switch (sym.kind) {
+    case 'module': return typeIndex('namespace');
+    case 'procedure':
+    case 'function': return typeIndex('function');
+    case 'variable': return typeIndex('variable');
+    case 'parameter': return typeIndex('parameter');
+    case 'field': return typeIndex('property');
+    case 'constant':
+      // Enumeration literals are recorded as `Type.Literal`.
+      return /^\w+\.\w+$/.test(sym.detail) ? typeIndex('enumMember') : typeIndex('variable');
+    case 'type': return typeIndex('type');
+    default: return null;
+  }
+}

+ 79 - 2
extensions/pascal-language/src/server.ts

@@ -15,21 +15,26 @@ import {
   findReferencesInText, moduleExports, nameAtPosition, recordFieldsOf,
   resolveAt, tokenAtPosition, unitAt, PasLocatedOccurrence, PasResolved
 } from './pascalResolve';
+import { DEFAULT_FORMAT_OPTIONS, FormatOptions, formatDocument, formatRangeEdits, normalizeFormatOptions } from './pascalFormat';
+import { computeSemanticTokens, semanticTokensLegend } from './pascalSemantic';
 import {
-  CompletionItem, CompletionItemKind, DocumentSymbol, SymbolKind, Hover, Location
+  CompletionItem, CompletionItemKind, DocumentSymbol, SymbolKind, Hover, Location, TextEdit
 } from 'vscode-languageserver/node';
 
 const connection = createConnection(ProposedFeatures.all);
 const documents = new TextDocuments(TextDocument);
 let pascalDialect = 'freepascal';
 let pascalValidators: Record<string, string> = {};
+let formatOptions: FormatOptions = { ...DEFAULT_FORMAT_OPTIONS };
 
 connection.onInitialize((params: InitializeParams): InitializeResult => {
   const options = (params.initializationOptions || {}) as {
     pascalDialect?: string; pascalValidators?: Record<string, string>;
+    format?: Partial<FormatOptions>;
   };
   pascalDialect = options.pascalDialect || 'freepascal';
   pascalValidators = options.pascalValidators || {};
+  formatOptions = normalizeFormatOptions(options.format ?? {});
   return {
     capabilities: {
       textDocumentSync: TextDocumentSyncKind.Full,
@@ -38,7 +43,13 @@ connection.onInitialize((params: InitializeParams): InitializeResult => {
       definitionProvider: true,
       documentSymbolProvider: true,
       referencesProvider: true,
-      renameProvider: { prepareProvider: true }
+      renameProvider: { prepareProvider: true },
+      documentFormattingProvider: true,
+      documentRangeFormattingProvider: true,
+      semanticTokensProvider: {
+        legend: semanticTokensLegend(),
+        full: true,
+      },
     }
   };
 });
@@ -51,6 +62,10 @@ connection.onDidChangeConfiguration((params: DidChangeConfigurationParams) => {
   };
   pascalDialect = settings.modula2?.pascal?.dialect || pascalDialect;
   pascalValidators = settings.modula2?.pascal?.validators || pascalValidators;
+  const fmt = (settings.modula2?.pascal as { format?: Partial<FormatOptions> } | undefined)?.format;
+  if (fmt) {
+    formatOptions = normalizeFormatOptions(fmt);
+  }
 });
 
 documents.onDidOpen(e => validate(e.document));
@@ -435,5 +450,67 @@ connection.onRenameRequest(params => {
   return { changes };
 });
 
+connection.onDocumentFormatting(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return null;
+  const filePath = uriToFilePath(document.uri);
+  let formatted: string;
+  try {
+    formatted = formatDocument(
+      document.getText(), formatOptions, filePath, path.dirname(filePath),
+    );
+  } catch (error) {
+    connection.console.error(`formatting failed: ${error instanceof Error ? error.message : String(error)}`);
+    return null;
+  }
+  if (formatted === document.getText()) return [];
+  const lines = document.getText().split('\n');
+  const last = lines.length - 1;
+  const edit: TextEdit = {
+    range: {
+      start: { line: 0, character: 0 },
+      end: { line: last, character: lines[last]?.length ?? 0 },
+    },
+    newText: formatted,
+  };
+  return [edit];
+});
+
+connection.onDocumentRangeFormatting(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return null;
+  const filePath = uriToFilePath(document.uri);
+  try {
+    const edits = formatRangeEdits(
+      document.getText(), params.range.start.line, params.range.end.line,
+      formatOptions, filePath, path.dirname(filePath),
+    );
+    return edits.map((e): TextEdit => ({
+      range: {
+        start: { line: e.startLine, character: e.startCh },
+        end: { line: e.endLine, character: e.endCh },
+      },
+      newText: e.newText,
+    }));
+  } catch (error) {
+    connection.console.error(`range formatting failed: ${error instanceof Error ? error.message : String(error)}`);
+    return null;
+  }
+});
+
+connection.languages.semanticTokens.on(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return { data: [] };
+  const filePath = uriToFilePath(document.uri);
+  try {
+    return {
+      data: computeSemanticTokens(document.getText(), filePath, path.dirname(filePath)),
+    };
+  } catch (error) {
+    connection.console.error(`semantic tokens failed: ${error instanceof Error ? error.message : String(error)}`);
+    return { data: [] };
+  }
+});
+
 documents.listen(connection);
 connection.listen();