#!/bin/sh # nonvacuity.sh -- prove the runtime checks can actually fail. # # A test that has never been seen red is not a test. This script breaks the # runtime on purpose, once per check, and asserts that the check goes red and # says something useful about the breakage. Then it restores the source and # asserts everything is green again. # # Each mutation below is a real bug that was in this file at some point, not an # invented one. That is the point: these are the mistakes we actually make # with 16-bit ModRM, so these are the ones the checks have to catch. # # audit_helpers.py name-versus-decode: catches a wrong ModRM that still # decodes cleanly # audit_helpers.py coverage: catches a helper that has silently # dropped OUT of the audit, which is a green report about # a subject nobody looked at # run_com_tests.sh the .COM layout: catches a header that cannot be # located, a runtime size that disagrees with the image, # and an entry jump that starts in the wrong place # check_runtime.py golden: catches the same thing in the built # image # check_runtime.py decode sweep: catches a wrong instruction LENGTH # check_runtime.py branch targets: catches a wrong fixup # check_runtime.py entry goldens: catches a broken prologue # probe/modrm11.py the mod=11 table: catches the ModRM column itself # going wrong, which no amount of decoding will show # check_framedisp.py the BP disp rule: catches a displacement that reads # a different address than the symbol table named # rt_exec.py the BP contract: catches an entry that borrows BP to # reach its argument and does not hand it back. The bytes # are well formed, the golden is satisfied, the audit says # every helper emits what its name says, and the machine # triple-faults on the second call - so nothing short of # running it, or of stating the register contract # explicitly, can see it. # # The mod=11 cases do not need a rebuild -- they read the probe sources # directly -- so they are cheap, and they are the ones that matter most: the # table they guard is the one thing in this project that was wrong in the # documentation while the code was right, and a table that is wrong in the # code produces bytes that decode perfectly. # # Usage: tests/nonvacuity.sh (from shell/; leaves Runtime.mod restored) set -u cd "$(dirname "$0")/.." || exit 1 GM2=/home/eric/bin/Modula2/Gm2/bin/gm2 SAVED=/tmp/opencode/nonvacuity.Runtime.mod PROBE=/tmp/opencode/nonvacuity.rtprobe DUMP=/tmp/opencode/nonvacuity.dump cp Runtime.mod "$SAVED" || exit 1 trap 'cp "$SAVED" Runtime.mod; "$GM2" -fiso -c Runtime.mod >/dev/null 2>&1' EXIT pass=0 fail=0 # mutate -- apply a deliberate breakage and INSIST it landed. # # Four cases in this file were already dead when first run, all the same way: # the helper they name had been renamed or reformatted since the case was # written, the sed matched nothing, the source was unchanged, and the check # correctly passed - so the harness reported "NOT NON-VACUOUS" and, worse, a # reader skimming the output could take "the check still passed" for a passing # test. A case that cannot fire is worse than no case: it is a claim of # coverage that was never tested. # # So the mutation is verified, not assumed. If the file is byte-identical # afterwards, that is reported as a FAILURE of the harness, naming the sed, and # the case is not run - because running it would only produce a meaningless # green. The message says what to do (fix the sed) rather than what it found. mutate () { mf=$1 msed=$2 cp "$mf" /tmp/opencode/nonvacuity.mut.bak sed -i "$msed" "$mf" if cmp -s "$mf" /tmp/opencode/nonvacuity.mut.bak; then echo " BROKEN CASE: the mutation did not change $mf" echo " sed: $msed" echo " the named code has probably been renamed or reformatted -" echo " fix this case, it is asserting nothing" fail=$((fail + 1)) return 1 fi return 0 } # rebuild