Преглед изворни кода

Oberon LSP: formatter, semantic tokens (O2); export-mark tracking in parser

Eric Streit пре 3 дана
родитељ
комит
7a436a76c7

+ 528 - 0
extensions/oberon-language/src/oberonFormat.ts

@@ -0,0 +1,528 @@
+/** Token-based Oberon document formatter with user-settable rules.
+ *
+ *  Twin of the Pascal formatter: line structure is preserved
+ *  (statements are never joined or split), while indentation, keyword
+ *  case, intra-line spacing, blank lines and trailing whitespace are
+ *  normalised per `FormatOptions`. Comments are preserved verbatim
+ *  (only repositioned); continuation lines of multi-line comments are
+ *  emitted as-is. Best-effort on broken code: always terminates and
+ *  never drops tokens. `*`/`-` export marks stay attached to their
+ *  declaration; the same glyphs as operators keep their spaces.
+ */
+
+import * as path from 'path';
+import { parseImports, parseOberon, ObSymbol } from './oberonSymbols';
+import { resolveAt } from './oberonResolve';
+
+export type KeywordCase = 'upper' | 'preserve';
+
+export interface FormatOptions {
+  indentSize: number;
+  useTabs: boolean;
+  keywordCase: KeywordCase;
+  spaceAroundOperators: boolean;
+  spaceAfterComma: boolean;
+  spaceBeforeColon: boolean;
+  spaceAroundRange: boolean;
+  emptyLineLimit: number;
+  trimTrailingWhitespace: boolean;
+  insertFinalNewline: boolean;
+}
+
+export const DEFAULT_FORMAT_OPTIONS: FormatOptions = {
+  indentSize: 2,
+  useTabs: false,
+  keywordCase: 'upper',
+  spaceAroundOperators: true,
+  spaceAfterComma: true,
+  spaceBeforeColon: false,
+  spaceAroundRange: true,
+  emptyLineLimit: 1,
+  trimTrailingWhitespace: true,
+  insertFinalNewline: true,
+};
+
+/** Sanitise arbitrary settings JSON into valid options (never throws). */
+export function normalizeFormatOptions(raw: unknown): FormatOptions {
+  const r = (raw ?? {}) as Partial<FormatOptions>;
+  const num = (v: unknown, lo: number, hi: number, dflt: number): number => {
+    const n = typeof v === 'number' && Number.isFinite(v) ? Math.round(v) : dflt;
+    return Math.min(hi, Math.max(lo, n));
+  };
+  const bool = (v: unknown, dflt: boolean): boolean =>
+    typeof v === 'boolean' ? v : dflt;
+  return {
+    indentSize: num(r.indentSize, 0, 8, DEFAULT_FORMAT_OPTIONS.indentSize),
+    useTabs: bool(r.useTabs, false),
+    keywordCase: r.keywordCase === 'preserve' || r.keywordCase === 'upper'
+      ? r.keywordCase : 'upper',
+    spaceAroundOperators: bool(r.spaceAroundOperators, true),
+    spaceAfterComma: bool(r.spaceAfterComma, true),
+    spaceBeforeColon: bool(r.spaceBeforeColon, false),
+    spaceAroundRange: bool(r.spaceAroundRange, true),
+    emptyLineLimit: num(r.emptyLineLimit, 0, 10, DEFAULT_FORMAT_OPTIONS.emptyLineLimit),
+    trimTrailingWhitespace: bool(r.trimTrailingWhitespace, true),
+    insertFinalNewline: bool(r.insertFinalNewline, true),
+  };
+}
+
+type TokenKind = 'keyword' | 'ident' | 'number' | 'string' | 'symbol';
+
+interface RichToken {
+  text: string;
+  kind: TokenKind;
+  comment?: boolean;
+  line: number; ch: number; endLine: number; endCh: number;
+  offset: number; endOffset: number;
+}
+
+/** Reserved words matched case-insensitively by the formatter lexer. */
+const OBERON_KEYWORDS = new Set([
+  'ARRAY', 'BEGIN', 'BY', 'CASE', 'CONST', 'DEFINITION', 'DIV', 'DO',
+  'ELSE', 'ELSIF', 'END', 'EXIT', 'FOR', 'IF', 'IMPORT', 'IN', 'IS',
+  'LOOP', 'MOD', 'MODULE', 'NIL', 'OF', 'OR', 'POINTER', 'PROCEDURE',
+  'RECORD', 'REPEAT', 'RETURN', 'THEN', 'TO', 'TYPE', 'UNTIL', 'VAR',
+  'WHILE', 'WITH',
+]);
+
+const isWordStart = (c: string) => /[A-Za-z_]/.test(c);
+const isWordChar = (c: string) => /[A-Za-z0-9_]/.test(c);
+const isDigit = (c: string) => /[0-9]/.test(c);
+const isHex = (c: string) => /[0-9A-Fa-f]/.test(c);
+
+/** Lex code tokens (comments land in `comments`, strings are skipped). */
+function lexCode(text: string): { tokens: RichToken[]; comments: RichToken[] } {
+  const tokens: RichToken[] = [];
+  const comments: RichToken[] = [];
+  let i = 0;
+  let line = 0;
+  let col = 0;
+  const push = (t: string, kind: TokenKind, sl: number, sc: number, so: number): void => {
+    tokens.push({ text: t, kind, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
+  };
+  const pushComment = (t: string, sl: number, sc: number, so: number): void => {
+    comments.push({ text: t, kind: 'ident', comment: true, line: sl, ch: sc, endLine: line, endCh: col, offset: so, endOffset: i });
+  };
+  const step = (n: number): void => {
+    for (let k = 0; k < n; k++) {
+      if (text[i] === '\n') { line++; col = 0; }
+      else if (text[i] !== '\r') { col++; }
+      else { col = 0; }
+      i++;
+    }
+  };
+  while (i < text.length) {
+    const c = text[i];
+    const nx = i + 1 < text.length ? text[i + 1] : '';
+    if (c === '\n' || c === '\r' || c === ' ' || c === '\t') { step(1); continue; }
+    // Nested (* *) comment (the only Oberon comment form).
+    if (c === '(' && nx === '*') {
+      const sl = line; const sc = col; const so = i;
+      let level = 1;
+      step(2);
+      while (i < text.length && level > 0) {
+        if (text[i] === '(' && text[i + 1] === '*') { level++; step(2); continue; }
+        if (text[i] === '*' && text[i + 1] === ')') { level--; step(2); continue; }
+        step(1);
+      }
+      pushComment(text.slice(so, i), sl, sc, so);
+      continue;
+    }
+    // "..." / '...' strings (no escapes; unterminated runs to end of line).
+    if (c === '"' || c === "'") {
+      const sl = line; const sc = col; const so = i;
+      step(1);
+      while (i < text.length && text[i] !== c && text[i] !== '\n' && text[i] !== '\r') step(1);
+      if (i < text.length && text[i] === c) step(1);
+      push(text.slice(so, i), 'string', sl, sc, so);
+      continue;
+    }
+    if (isWordStart(c)) {
+      const sl = line; const sc = col; const so = i;
+      while (i < text.length && isWordChar(text[i])) step(1);
+      const word = text.slice(so, i);
+      push(word, OBERON_KEYWORDS.has(word.toUpperCase()) ? 'keyword' : 'ident', sl, sc, so);
+      continue;
+    }
+    // Numbers: decimal, .fraction, E exponents, H/X suffixes (0FFH, 41X).
+    if (isDigit(c)) {
+      const sl = line; const sc = col; const so = i;
+      while (i < text.length && isHex(text[i])) step(1);
+      if ((text[i] === 'H' || text[i] === 'X') && !isWordChar(text[i + 1] ?? '')) {
+        step(1);
+      } else {
+        // Rewind hex letters: plain decimal (optionally real).
+        while (i > so && /[A-Fa-f]/.test(text[i - 1] ?? '')) {
+          i--; col--;
+        }
+        if (text[i] === '.' && text[i + 1] !== '.' && isDigit(text[i + 1] ?? '')) {
+          step(1);
+          while (i < text.length && isDigit(text[i])) step(1);
+        }
+        if (text[i] === 'E' || text[i] === 'e') {
+          let j = i + 1;
+          if (text[j] === '+' || text[j] === '-') j++;
+          if (isDigit(text[j] ?? '')) {
+            while (j < text.length && isDigit(text[j])) j++;
+            step(j - i);
+          }
+        }
+      }
+      push(text.slice(so, i), 'number', sl, sc, so);
+      continue;
+    }
+    // Multi-char symbols.
+    const sl = line; const sc = col; const so = i;
+    const two = text.slice(i, i + 2);
+    if (two === ':=' || two === '<=' || two === '>=') {
+      step(2);
+      push(two, 'symbol', sl, sc, so);
+      continue;
+    }
+    if (c === '.' && nx === '.') {
+      step(2);
+      push('..', 'symbol', sl, sc, so);
+      continue;
+    }
+    step(1);
+    push(c, 'symbol', sl, sc, so);
+  }
+  return { tokens, comments };
+}
+
+/** Structural words for indentation (matched case-insensitively). */
+const STRUCT = new Set([
+  'END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY',
+  'THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'CASE', 'OF', 'LOOP', 'TRY',
+]);
+const DISPLAY_DEDENT = new Set(['END', 'ELSE', 'ELSIF', 'UNTIL', 'EXCEPT', 'FINALLY']);
+const OPENERS = new Set(['THEN', 'BEGIN', 'RECORD', 'REPEAT', 'DO', 'LOOP', 'TRY']);
+const CLOSERS = new Set(['END', 'UNTIL']);
+const UNARY_AFTER = new Set([
+  '(', '[', ',', ';', ':=', '=', '..', 'THEN', 'DO', 'OF',
+  'NOT', 'AND', 'OR', 'DIV', 'MOD', 'IN', '+', '-', '*', '/', '<', '<=',
+  '>', '>=', '#', '&', '~', 'BY', 'TO',
+]);
+
+function wordOf(t: RichToken): string {
+  return t.kind === 'keyword' && STRUCT.has(t.text.toUpperCase()) ? t.text.toUpperCase() : '';
+}
+
+function applyCase(text: string, mode: KeywordCase): string {
+  return mode === 'upper' ? text.toUpperCase() : text;
+}
+
+/** Promote keyword-shaped identifiers to keywords unless they resolve.
+ *
+ *  Lowercase keywords lex as identifiers, but code may use words like
+ *  `to` as identifiers. An identifier that is declared here or resolves
+ *  (qualifier-aware) stays an identifier; anything else shaped like a
+ *  reserved word is a keyword. This keeps formatting token-exact.
+ */
+function classifyKeywords(tokens: RichToken[], text: string, filePath: string, docDir: string, marks: Set<string>): void {
+  let symbols: ObSymbol[];
+  try {
+    symbols = parseOberon(text, marks);
+  } catch {
+    return;
+  }
+  const imports = parseImports(text);
+  const declared = new Set<string>();
+  const flatten = (ss: ObSymbol[]): void => {
+    for (const s of ss) {
+      if (s.name) declared.add(`${s.line}:${s.ch}`);
+      flatten(s.children);
+    }
+  };
+  flatten(symbols);
+  const prevCode = (i: number): { tok: RichToken; idx: number } | null => {
+    for (let j = i - 1; j >= 0; j--) {
+      const t = tokens[j];
+      if (!t.comment) return { tok: t, idx: j };
+    }
+    return null;
+  };
+  tokens.forEach((t, i) => {
+    if (t.comment || t.kind !== 'ident' || !OBERON_KEYWORDS.has(t.text.toUpperCase())) return;
+    if (declared.has(`${t.line}:${t.ch}`)) return;
+    let qualifier: string | null = null;
+    const dot = prevCode(i);
+    if (dot && dot.tok.text === '.') {
+      const head = prevCode(dot.idx);
+      if (!head || head.tok.kind !== 'ident') return;
+      qualifier = head.tok.text;
+    }
+    const r = resolveAt(symbols, imports, filePath, docDir, t.line, t.ch, qualifier, t.text);
+    if (!r) t.kind = 'keyword';
+  });
+}
+
+export function formatDocument(
+  text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): string {
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  const { entries, eol } = formatLineEntries(text, opts, filePath, docDir);
+  const joined = entries.map(l => l.text).join('\n');
+  if (!opts.insertFinalNewline) return joined.split('\n').join(eol);
+  return (joined.length > 0 ? joined + '\n' : '').split('\n').join(eol);
+}
+
+/** One formatted output line and the 0-based input line it came from. */
+export interface FormattedLine {
+  src: number;
+  text: string;
+}
+
+/** One edit for `textDocument/rangeFormatting`, directly mappable to LSP. */
+export interface RangeEdit {
+  startLine: number;
+  startCh: number;
+  endLine: number;
+  endCh: number;
+  newText: string;
+}
+
+/** Format a line range with full-document context. Never throws. */
+export function formatRangeEdits(
+  text: string, startLine: number, endLine: number,
+  rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): RangeEdit[] {
+  if (text.length === 0) return [];
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  const srcLines = text.split('\n');
+  const srcCount = srcLines.length;
+  const from = Math.min(Math.max(0, Math.min(startLine, endLine)), srcCount - 1);
+  const to = Math.min(Math.max(Math.max(startLine, endLine), from), srcCount - 1);
+  const { entries } = formatLineEntries(text, rawOptions, filePath, docDir);
+  const bySrc = new Map<number, string>();
+  for (const e of entries) {
+    if (e.src >= from && e.src <= to && !bySrc.has(e.src)) bySrc.set(e.src, e.text);
+  }
+  const edits: RangeEdit[] = [];
+  let runFrom = -1;
+  const parts: string[] = [];
+  let tailNewlined = false;
+  const flush = (endExclusive: number): void => {
+    if (runFrom < 0) return;
+    const atEof = endExclusive >= srcCount;
+    if (atEof && parts.length === 0 && runFrom > 0) {
+      const prevLen = srcLines[runFrom - 1].length;
+      edits.push({
+        startLine: runFrom - 1, startCh: prevLen,
+        endLine: srcCount - 1, endCh: srcLines[srcCount - 1].length,
+        newText: opts.insertFinalNewline && entries.length > 0 ? '\n' : '',
+      });
+      runFrom = -1;
+      return;
+    }
+    let newText = parts.join('\n');
+    if (parts.length > 0 && (atEof ? opts.insertFinalNewline : true)) newText += '\n';
+    if (atEof && newText.endsWith('\n')) tailNewlined = true;
+    const end = atEof
+      ? { line: srcCount - 1, ch: srcLines[srcCount - 1].length }
+      : { line: endExclusive, ch: 0 };
+    edits.push({
+      startLine: runFrom, startCh: 0,
+      endLine: end.line, endCh: end.ch, newText,
+    });
+    runFrom = -1;
+    parts.length = 0;
+  };
+  for (let s = from; s <= to; s++) {
+    const want = bySrc.get(s);
+    if (want !== undefined && want === srcLines[s]) {
+      flush(s);
+      continue;
+    }
+    if (runFrom < 0) runFrom = s;
+    if (want !== undefined) parts.push(want);
+  }
+  flush(to + 1);
+  if (to === srcCount - 1 && opts.insertFinalNewline && !text.endsWith('\n') &&
+      entries.length > 0 && !tailNewlined) {
+    const last = srcLines[srcCount - 1].length;
+    edits.push({
+      startLine: srcCount - 1, startCh: last,
+      endLine: srcCount - 1, endCh: last, newText: '\n',
+    });
+  }
+  return edits;
+}
+
+function formatLineEntries(
+  text: string, rawOptions?: Partial<FormatOptions>, filePath = '', docDir = '',
+): { entries: FormattedLine[]; eol: string } {
+  const opts = normalizeFormatOptions(rawOptions ?? {});
+  if (text.length === 0) return { entries: [], eol: '\n' };
+  const eol = text.includes('\r\n') ? '\r\n' : '\n';
+  const dir = docDir || (filePath ? path.dirname(filePath) : '');
+  const { tokens, comments } = lexCode(text);
+  const all: RichToken[] = [...tokens, ...comments].sort((a, b) => a.offset - b.offset);
+  const marks = new Set<string>();
+  classifyKeywords(all, text, filePath, dir, marks);
+
+  const byLine = new Map<number, RichToken[]>();
+  let maxLine = 0;
+  for (const t of all) {
+    maxLine = Math.max(maxLine, t.endLine);
+    const list = byLine.get(t.line) ?? [];
+    list.push(t);
+    byLine.set(t.line, list);
+  }
+  const srcLines = text.split(/\r?\n/);
+  const lastLine = Math.max(maxLine, srcLines.length - 1);
+
+  const verbatim = new Set<number>();
+  for (const t of all) {
+    if (t.comment && t.endLine > t.line) {
+      for (let l = t.line + 1; l <= t.endLine; l++) verbatim.add(l);
+    }
+  }
+
+  const indentOf = (level: number): string =>
+    opts.useTabs ? '\t'.repeat(level) : ' '.repeat(opts.indentSize * level);
+
+  const entries: FormattedLine[] = [];
+  let level = 0;
+  let casePending = false;
+  let blanks = 0;
+
+  for (let i = 0; i <= lastLine; i++) {
+    const lineToks = (byLine.get(i) ?? []).filter(t => !t.comment || t.line === i);
+    const code = lineToks.filter(t => !t.comment);
+    const startsHere = lineToks.filter(t => t.comment && t.line === i);
+
+    if (verbatim.has(i)) {
+      entries.push({ src: i, text: opts.trimTrailingWhitespace ? srcLines[i]?.replace(/[ \t]+$/, '') ?? '' : srcLines[i] ?? '' });
+      const applied = applyDelta(code, level, { casePending });
+      level = applied.level;
+      casePending = applied.casePending;
+      continue;
+    }
+
+    if (code.length === 0 && startsHere.length === 0) {
+      blanks++;
+      if (blanks <= opts.emptyLineLimit) entries.push({ src: i, text: '' });
+      continue;
+    }
+    blanks = 0;
+
+    if (code.length === 0) {
+      const parts = startsHere.map(c => firstFragment(c));
+      entries.push({ src: i, text: indentOf(level) + parts.join('  ') });
+      continue;
+    }
+
+    const first = wordOf(code[0]);
+    const emitLevel = Math.max(0, level - (DISPLAY_DEDENT.has(first) ? 1 : 0));
+    let body = emitTokens(code, opts, marks);
+    const trailing = startsHere.filter(c => code[0].offset < c.offset);
+    for (const c of trailing) body += '  ' + firstFragment(c);
+    let lineText = indentOf(emitLevel) + body;
+    if (opts.trimTrailingWhitespace) lineText = lineText.replace(/[ \t]+$/, '');
+    entries.push({ src: i, text: lineText });
+
+    const applied = applyDelta(code, level, { casePending });
+    level = applied.level;
+    casePending = applied.casePending;
+  }
+
+  if (!opts.insertFinalNewline) return { entries, eol };
+  while (entries.length > 0 && /^[ \t]*$/.test(entries[entries.length - 1].text)) {
+    entries.pop();
+  }
+  if (entries.length > 0) {
+    entries[entries.length - 1].text = entries[entries.length - 1].text.replace(/[ \t]+$/, '');
+  }
+  return { entries, eol };
+}
+
+function firstFragment(c: RichToken): string {
+  const idx = c.text.search(/\r?\n/);
+  return idx < 0 ? c.text : c.text.slice(0, idx);
+}
+
+function applyDelta(
+  code: RichToken[], level: number, state: { casePending: boolean },
+): { level: number; casePending: boolean } {
+  let opens = 0;
+  let closes = 0;
+  let casePending = state.casePending;
+  for (const t of code) {
+    const w = wordOf(t);
+    if (!w) continue;
+    if (w === 'CASE') {
+      casePending = true;
+      continue;
+    }
+    if (w === 'OF') {
+      if (casePending) {
+        opens++;
+        casePending = false;
+      }
+      continue;
+    }
+    if (OPENERS.has(w)) {
+      opens++;
+      continue;
+    }
+    if (CLOSERS.has(w)) {
+      closes++;
+      continue;
+    }
+  }
+  return { level: Math.max(0, level + opens - closes), casePending };
+}
+
+function isWordish(t: RichToken): boolean {
+  return t.kind === 'ident' || t.kind === 'keyword' || t.kind === 'number' || t.kind === 'string';
+}
+
+function emitTokens(code: RichToken[], opts: FormatOptions, marks: Set<string>): string {
+  let out = '';
+  const gap = (prev: RichToken | null, cur: RichToken): string => {
+    if (!prev) return '';
+    const pt = prev.text;
+    const ct = cur.text;
+    if (ct === '(') return prev.kind === 'keyword' ? ' ' : '';
+    if (ct === '[') return prev.kind === 'keyword' ? ' ' : '';
+    if (pt === '(' || pt === '[') return '';
+    if (pt === '.' || pt === '^' || pt === '@') return '';
+    if (ct === ')' || ct === ']') return '';
+    if (ct === '.' || ct === '^') return '';
+    if (ct === ',') return '';
+    if (ct === ';') return '';
+    if (pt === ')' || pt === ']' || pt === ';') {
+      return isWordish(cur) ? ' ' : '';
+    }
+    // `*`/`-` export marks attach to the declared name (positions
+    // recorded by the parser); the same glyphs as operators keep spaces.
+    if ((ct === '*' || ct === '-') && prev.kind === 'ident') {
+      return marks.has(`${cur.line}:${cur.ch}`) ? '' : (opts.spaceAroundOperators ? ' ' : '');
+    }
+    if (ct === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ':=') return opts.spaceAroundOperators ? ' ' : '';
+    if (pt === ',') return opts.spaceAfterComma ? ' ' : '';
+    if (ct === ':') return opts.spaceBeforeColon ? ' ' : '';
+    if (pt === ':') return ' ';
+    if (ct === '..' || pt === '..') return opts.spaceAroundRange ? ' ' : '';
+    if (ct === '+' || ct === '-') return isUnary(prev) ? '' : (opts.spaceAroundOperators ? ' ' : '');
+    if (cur.kind === 'symbol' || prev.kind === 'symbol') {
+      return opts.spaceAroundOperators ? ' ' : '';
+    }
+    return ' ';
+  };
+  for (let k = 0; k < code.length; k++) {
+    const t = code[k];
+    const text = t.kind === 'keyword' ? applyCase(t.text, opts.keywordCase) : t.text;
+    out += gap(k > 0 ? code[k - 1] : null, t) + text;
+  }
+  return out;
+}
+
+function isUnary(prev: RichToken | null): boolean {
+  if (!prev) return true;
+  if (prev.kind === 'keyword') return UNARY_AFTER.has(prev.text.toUpperCase());
+  if (prev.kind !== 'symbol') return false;
+  return UNARY_AFTER.has(prev.text);
+}

+ 146 - 0
extensions/oberon-language/src/oberonSemantic.ts

@@ -0,0 +1,146 @@
+/** Semantic tokens for Oberon editors (LSP `textDocument/semanticTokens`).
+ *
+ *  Declarations come from the symbol tree; identifier *uses* are resolved
+ *  through the same machinery as hover/definition, so variables,
+ *  parameters, fields and cross-module names highlight by meaning, not
+ *  just by spelling. Anything unresolvable is skipped, so broken code
+ *  under editing degrades to TextMate highlighting.
+ */
+
+import { parseImports, parseOberon, ObSymbol } from './oberonSymbols';
+import { identifierOccurrences, resolveAt } from './oberonResolve';
+
+export const SEMANTIC_TOKEN_TYPES = [
+  'namespace', // modules
+  'type', // declared types
+  'function', // procedures
+  'variable', // variables and constants
+  'parameter', // receiver, parameters
+  'property', // record fields
+  'enumMember', // unused in Oberon (kept for legend parity)
+];
+
+export const SEMANTIC_TOKEN_MODIFIERS = ['readonly'];
+
+export function semanticTokensLegend(): { tokenTypes: string[]; tokenModifiers: string[] } {
+  return { tokenTypes: [...SEMANTIC_TOKEN_TYPES], tokenModifiers: [...SEMANTIC_TOKEN_MODIFIERS] };
+}
+
+/** Predeclared names (resolve to no declaration; still typed). */
+const TYPE_BUILTINS = new Set([
+  'BOOLEAN', 'CHAR', 'INTEGER', 'LONGINT', 'REAL', 'LONGREAL', 'SET',
+  'SHORTINT', 'BYTE',
+]);
+
+interface RawToken {
+  line: number;
+  ch: number;
+  length: number;
+  type: number;
+  modifiers: number;
+}
+
+function flatten(symbols: ObSymbol[], chain: ObSymbol[], out: { sym: ObSymbol; chain: ObSymbol[] }[]): void {
+  for (const s of symbols) {
+    out.push({ sym: s, chain: [...chain] });
+    if (s.children.length > 0) flatten(s.children, [...chain, s], out);
+  }
+}
+
+/** LSP delta-encoded token data for a document. Never throws on broken input. */
+export function computeSemanticTokens(text: string, filePath: string, docDir: string): number[] {
+  const symbols = parseOberon(text);
+  const imports = parseImports(text);
+  const out: RawToken[] = [];
+  const declared = new Set<string>();
+  const flat: { sym: ObSymbol; chain: ObSymbol[] }[] = [];
+  flatten(symbols, [], flat);
+  for (const { sym, chain } of flat) {
+    if (!sym.name) continue;
+    // The file's own module symbol is the root scope, not a declaration.
+    if (sym.kind === 'module' && chain.length === 0) continue;
+    declared.add(`${sym.line}:${sym.ch}`);
+    const mapped = symbolToken(sym);
+    if (!mapped) continue;
+    out.push({
+      line: sym.line,
+      ch: sym.ch,
+      length: Math.max(1, sym.endCh - sym.ch),
+      type: mapped.type,
+      modifiers: mapped.modifiers,
+    });
+  }
+  for (const occ of identifierOccurrences(text)) {
+    if (declared.has(`${occ.line}:${occ.ch}`)) continue;
+    const resolved = resolveAt(symbols, imports, filePath, docDir, occ.line, occ.ch, occ.qualifier, occ.name);
+    if (resolved) {
+      const kind = useTokenKind(resolved.sym);
+      if (kind === null) continue;
+      out.push({
+        line: occ.line,
+        ch: occ.ch,
+        length: Math.max(1, occ.endCh - occ.ch),
+        type: kind,
+        modifiers: 0,
+      });
+      continue;
+    }
+    if (TYPE_BUILTINS.has(occ.name.toUpperCase())) {
+      out.push({
+        line: occ.line,
+        ch: occ.ch,
+        length: Math.max(1, occ.endCh - occ.ch),
+        type: typeIndex('type'),
+        modifiers: 0,
+      });
+    }
+  }
+  out.sort((a, b) => a.line - b.line || a.ch - b.ch);
+  const data: number[] = [];
+  let prevLine = 0;
+  let prevCh = 0;
+  for (const t of out) {
+    data.push(
+      t.line - prevLine,
+      t.line === prevLine ? t.ch - prevCh : t.ch,
+      t.length, t.type, t.modifiers,
+    );
+    prevLine = t.line;
+    prevCh = t.ch;
+  }
+  return data;
+}
+
+function typeIndex(name: string): number {
+  const i = SEMANTIC_TOKEN_TYPES.indexOf(name);
+  return i >= 0 ? i : 0;
+}
+
+function symbolToken(sym: ObSymbol): { type: number; modifiers: number } | null {
+  switch (sym.kind) {
+    case 'module': return { type: typeIndex('namespace'), modifiers: 0 };
+    case 'procedure':
+    case 'function': return { type: typeIndex('function'), modifiers: 0 };
+    case 'variable': return { type: typeIndex('variable'), modifiers: 0 };
+    case 'parameter': return { type: typeIndex('parameter'), modifiers: 0 };
+    case 'field': return { type: typeIndex('property'), modifiers: 0 };
+    case 'constant': return { type: typeIndex('variable'), modifiers: 1 };
+    case 'type': return { type: typeIndex('type'), modifiers: 0 };
+    default: return null;
+  }
+}
+
+/** Token type for a resolved identifier use (null = no token). */
+function useTokenKind(sym: ObSymbol): number | null {
+  switch (sym.kind) {
+    case 'module': return typeIndex('namespace');
+    case 'procedure':
+    case 'function': return typeIndex('function');
+    case 'variable': return typeIndex('variable');
+    case 'parameter': return typeIndex('parameter');
+    case 'field': return typeIndex('property');
+    case 'constant': return typeIndex('variable');
+    case 'type': return typeIndex('type');
+    default: return null;
+  }
+}

+ 18 - 7
extensions/oberon-language/src/oberonSymbols.ts

@@ -88,7 +88,7 @@ export const isIdent = (t: Tok) => /^[A-Za-z_]/.test(t.text);
 
 class Parser {
   pos = 0;
-  constructor(readonly toks: Tok[]) {}
+  constructor(readonly toks: Tok[], readonly marks?: Set<string>) {}
 
   peek(off = 0): Tok | null {
     return this.pos + off < this.toks.length ? this.toks[this.pos + off] : null;
@@ -178,14 +178,19 @@ class Parser {
   }
 
   /** Parse an identifier list with optional `*`/`-` export marks,
-   *  returning the name tokens. */
+   *  returning the name tokens. Mark positions are recorded for the
+   *  formatter (export marks attach; operators keep spaces). */
   nameList(): Tok[] {
     const out: Tok[] = [];
     for (;;) {
       const t = this.peek();
       if (!t || !isIdent(t)) return out;
       out.push(this.next());
-      if (this.peek()?.text === '*' || this.peek()?.text === '-') this.next();
+      const mark = this.peek();
+      if (mark?.text === '*' || mark?.text === '-') {
+        this.next();
+        this.marks?.add(`${mark.line}:${mark.ch}`);
+      }
       if (this.at(',')) { this.next(); continue; }
       return out;
     }
@@ -193,7 +198,11 @@ class Parser {
 
   /** Consume an optional `*`/`-` export mark after a declared name. */
   exportMark(): void {
-    if (this.peek()?.text === '*' || this.peek()?.text === '-') this.next();
+    const mark = this.peek();
+    if (mark?.text === '*' || mark?.text === '-') {
+      this.next();
+      this.marks?.add(`${mark.line}:${mark.ch}`);
+    }
   }
 }
 
@@ -521,10 +530,12 @@ function parseScope(p: Parser, out: ObSymbol[], stopAtBegin: boolean): void {
   }
 }
 
-/** Entry: top-level symbols of an Oberon module. */
-export function parseOberon(text: string): ObSymbol[] {
+/** Entry: top-level symbols of an Oberon module.
+ *  When `marks` is given, export-mark `*`/`-` positions (`line:ch`)
+ *  are collected for the formatter. */
+export function parseOberon(text: string, marks?: Set<string>): ObSymbol[] {
   const toks = lex(mask(text));
-  const p = new Parser(toks);
+  const p = new Parser(toks, marks);
   // skip an optional DEFINITION prefix and the MODULE header to ';'
   if (p.at('DEFINITION')) p.next();
   if (p.at('MODULE')) {

+ 79 - 2
extensions/oberon-language/src/server.ts

@@ -15,21 +15,26 @@ import {
   findReferencesInText, moduleExports, nameAtPosition, recordFieldsOf,
   resolveAt, tokenAtPosition, moduleAt, ObLocatedOccurrence
 } from './oberonResolve';
+import { DEFAULT_FORMAT_OPTIONS, FormatOptions, formatDocument, formatRangeEdits, normalizeFormatOptions } from './oberonFormat';
+import { computeSemanticTokens, semanticTokensLegend } from './oberonSemantic';
 import {
-  CompletionItem, CompletionItemKind, DocumentSymbol, SymbolKind, Hover, Location
+  CompletionItem, CompletionItemKind, DocumentSymbol, SymbolKind, Hover, Location, TextEdit
 } from 'vscode-languageserver/node';
 
 const connection = createConnection(ProposedFeatures.all);
 const documents = new TextDocuments(TextDocument);
 let oberonDialect = 'oberon2';
 let oberonValidators: Record<string, string> = {};
+let formatOptions: FormatOptions = { ...DEFAULT_FORMAT_OPTIONS };
 
 connection.onInitialize((params: InitializeParams): InitializeResult => {
   const options = (params.initializationOptions || {}) as {
     oberonDialect?: string; oberonValidators?: Record<string, string>;
+    format?: Partial<FormatOptions>;
   };
   oberonDialect = options.oberonDialect || 'oberon2';
   oberonValidators = options.oberonValidators || {};
+  formatOptions = normalizeFormatOptions(options.format ?? {});
   return {
     capabilities: {
       textDocumentSync: TextDocumentSyncKind.Full,
@@ -38,7 +43,13 @@ connection.onInitialize((params: InitializeParams): InitializeResult => {
       definitionProvider: true,
       documentSymbolProvider: true,
       referencesProvider: true,
-      renameProvider: { prepareProvider: true }
+      renameProvider: { prepareProvider: true },
+      documentFormattingProvider: true,
+      documentRangeFormattingProvider: true,
+      semanticTokensProvider: {
+        legend: semanticTokensLegend(),
+        full: true,
+      },
     }
   };
 });
@@ -51,6 +62,10 @@ connection.onDidChangeConfiguration((params: DidChangeConfigurationParams) => {
   };
   oberonDialect = settings.modula2?.oberon?.dialect || oberonDialect;
   oberonValidators = settings.modula2?.oberon?.validators || oberonValidators;
+  const fmt = (settings.modula2?.oberon as { format?: Partial<FormatOptions> } | undefined)?.format;
+  if (fmt) {
+    formatOptions = normalizeFormatOptions(fmt);
+  }
 });
 
 documents.onDidOpen(e => validate(e.document));
@@ -438,5 +453,67 @@ connection.onRenameRequest(params => {
   return { changes };
 });
 
+connection.onDocumentFormatting(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return null;
+  const filePath = uriToFilePath(document.uri);
+  let formatted: string;
+  try {
+    formatted = formatDocument(
+      document.getText(), formatOptions, filePath, path.dirname(filePath),
+    );
+  } catch (error) {
+    connection.console.error(`formatting failed: ${error instanceof Error ? error.message : String(error)}`);
+    return null;
+  }
+  if (formatted === document.getText()) return [];
+  const lines = document.getText().split('\n');
+  const last = lines.length - 1;
+  const edit: TextEdit = {
+    range: {
+      start: { line: 0, character: 0 },
+      end: { line: last, character: lines[last]?.length ?? 0 },
+    },
+    newText: formatted,
+  };
+  return [edit];
+});
+
+connection.onDocumentRangeFormatting(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return null;
+  const filePath = uriToFilePath(document.uri);
+  try {
+    const edits = formatRangeEdits(
+      document.getText(), params.range.start.line, params.range.end.line,
+      formatOptions, filePath, path.dirname(filePath),
+    );
+    return edits.map((e): TextEdit => ({
+      range: {
+        start: { line: e.startLine, character: e.startCh },
+        end: { line: e.endLine, character: e.endCh },
+      },
+      newText: e.newText,
+    }));
+  } catch (error) {
+    connection.console.error(`range formatting failed: ${error instanceof Error ? error.message : String(error)}`);
+    return null;
+  }
+});
+
+connection.languages.semanticTokens.on(params => {
+  const document = documents.get(params.textDocument.uri);
+  if (!document) return { data: [] };
+  const filePath = uriToFilePath(document.uri);
+  try {
+    return {
+      data: computeSemanticTokens(document.getText(), filePath, path.dirname(filePath)),
+    };
+  } catch (error) {
+    connection.console.error(`semantic tokens failed: ${error instanceof Error ? error.message : String(error)}`);
+    return { data: [] };
+  }
+});
+
 documents.listen(connection);
 connection.listen();