corpus.sh 1.9 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849
  1. #!/bin/sh
  2. # Runs the TopSpeed sidecar front-end (build/TSM2) over a corpus of
  3. # .DEF/.MOD files and reports parse coverage.
  4. #
  5. # A *syntax* error is a Coco/R error (`'x' expected`, `invalid X`); the
  6. # sidecar's own `not supported yet` (230) markers are the deliberate
  7. # parse-now/lower-later discipline and do NOT count as failures.
  8. #
  9. # Usage: ./corpus.sh [corpus-dir] (default: the TS-V3 tree)
  10. set -e
  11. cd "$(dirname "$0")"
  12. TSM2="$(pwd)/build/TSM2"
  13. CORPUS="${1:-$HOME/bin/Dos/M2/TS-V3}"
  14. WORK="/tmp/opencode/tscorpus"
  15. [ -x "$TSM2" ] || { echo "build/TSM2 missing -- run ./build.sh"; exit 1; }
  16. [ -d "$CORPUS" ] || { echo "corpus not found: $CORPUS"; exit 1; }
  17. rm -rf "$WORK"; mkdir -p "$WORK"
  18. : > "$WORK/results.txt"
  19. n=0
  20. find "$CORPUS" \( -iname '*.DEF' -o -iname '*.MOD' \) 2>/dev/null | sort |
  21. while read -r f; do
  22. n=$((n+1))
  23. w="$WORK/f$n"; mkdir -p "$w"; base=$(basename "$f")
  24. cp "$f" "$w/$base"
  25. ( cd "$w" && "$TSM2" "$base" >/dev/null 2>&1 )
  26. lst="$w/$(echo "$base" | sed 's/\.[^.]*$//').LST"
  27. if [ ! -f "$lst" ]; then echo "NOLST|$f" >> "$WORK/results.txt"; continue; fi
  28. if grep -qE "^\*\*\*\*\*.*(expected|invalid [A-Z])" "$lst"; then
  29. m=$(grep -E "^\*\*\*\*\*.*(expected|invalid [A-Z])" "$lst" | head -1 |
  30. sed -E 's/^\*\*\*\*\* *[^ ]* *//; s/^\*\*\*\*\* *//')
  31. echo "SYN|$m|$f" >> "$WORK/results.txt"
  32. else
  33. echo "OK|0|$f" >> "$WORK/results.txt"
  34. fi
  35. done
  36. total=$(wc -l < "$WORK/results.txt")
  37. ok=$(grep -c '^OK' "$WORK/results.txt" || true)
  38. syn=$(grep -c '^SYN' "$WORK/results.txt" || true)
  39. echo "corpus: $CORPUS"
  40. echo "total : $total"
  41. echo "parse-OK (semantic/230 only): $ok"
  42. echo "syntax-bad : $syn"
  43. echo "=== syntax-error kinds ==="
  44. grep '^SYN' "$WORK/results.txt" | cut -d'|' -f2 |
  45. sed -E 's/[0-9]+/N/g' | sort | uniq -c | sort -rn
  46. echo "=== syntax-bad files ==="
  47. grep '^SYN' "$WORK/results.txt" | cut -d'|' -f3 | sed "s|$CORPUS/||"