#!/bin/sh # nonvacuity.sh -- prove the runtime checks can actually fail. # # A test that has never been seen red is not a test. This script breaks the # runtime on purpose, once per check, and asserts that the check goes red and # says something useful about the breakage. Then it restores the source and # asserts everything is green again. # # Each mutation below is a real bug that was in this file at some point, not an # invented one. That is the point: these are the mistakes we actually make # with 16-bit ModRM, so these are the ones the checks have to catch. # # audit_helpers.py name-versus-decode: catches a wrong ModRM that still # decodes cleanly, and both halves of a name that admits # an operand at all - `34 02' where the name promises # `34 01', and `35 01' where the row's opcode gate admits # only 34h. Both rows are new, and a row nobody has seen # reject anything accepts whatever it is shown. # audit_helpers.py coverage: catches a helper that has silently # dropped OUT of the audit, which is a green report about # a subject nobody looked at # run_com_tests.sh the .COM layout: catches a header that cannot be # located, a runtime size that disagrees with the image, # and an entry jump that starts in the wrong place # check_runtime.py golden: catches the same thing in the built # image # check_runtime.py decode sweep: catches a wrong instruction LENGTH # check_runtime.py branch targets: catches a wrong fixup # check_runtime.py entry goldens: catches a broken prologue # probe/modrm11.py the mod=11 table: catches the ModRM column itself # going wrong, which no amount of decoding will show # check_framedisp.py the BP disp rule: catches a displacement that reads # a different address than the symbol table named # tests/comimage.py the image layout: catches a sweep region built from a # literal instead of from two independent readings of the # file - the old RT_SZ went on reporting PASS while its # region began inside the runtime - and the layout itself # being written down a second time, now that one file # defines it and all three readers import it # rt_exec.py the BP contract: catches an entry that borrows BP to # reach its argument and does not hand it back. The bytes # are well formed, the golden is satisfied, the audit says # every helper emits what its name says, and the machine # triple-faults on the second call - so nothing short of # running it, or of stating the register contract # explicitly, can see it. # check_8086.py the 8086 opcodes: catches a conditional branch or a # SETcc emitted as `0F 8x'/`0F 9x', which are 386-only. # Nothing else here can: qemu-system-i386's lowest CPU # model is 486, where both are ordinary instructions, and # FCML's -m16 mode is a 386 too. Clause H additionally # catches the branch polarity being inverted, which is # still a legal opcode and therefore invisible to every # shape-based check. # run_com_exec.py behaviour: catches a*b that emits an ADD, `>' # and `>=' swapped, REPEAT..UNTIL that stops after one # pass, two procedures whose parameters collide, a left # operand overwritten by the right one's code (SaveLeft), # and a boolean `not' lowered as the integer one. Every # one of them sat in a fixture that COMPILED and was # never RUN, with every byte-level check green. # run_exec86.py behaviour (Exec86): catches the interpreter's OWN # reading of the machine: JE inverted in Exec86's Cond, # which swaps the two arms of every `=' in every program # while the emitted bytes, the sizes and every # byte-level check stay green. The image is fine; only # the thing reading it is wrong, so nothing that looks at # the image can fail this one. # runtest.py the R key: catches CmdRun poking nothing into # the interpreter at all, which faults on the first step # and prints "interpreter fault" instead of the guest's # answer. It is the only case here that needs a rebuilt # SHELL rather than a rebuilt compiler or runtime. # # The mod=11 cases do not need a rebuild -- they read the probe sources # directly -- so they are cheap, and they are the ones that matter most: the # table they guard is the one thing in this project that was wrong in the # documentation while the code was right, and a table that is wrong in the # code produces bytes that decode perfectly. # # Usage: tests/nonvacuity.sh (from shell/; leaves Runtime.mod restored) set -u cd "$(dirname "$0")/.." || exit 1 GM2=/home/eric/bin/Modula2/Gm2/bin/gm2 SAVED=../tmp/nonvacuity.Runtime.mod PROBE=../tmp/nonvacuity.rtprobe DUMP=../tmp/nonvacuity.dump # Exec86.mod is UNTRACKED, so unlike Runtime.mod git cannot put a botched # mutation back: the copy taken here is the only correct one in existence. # Shell.mod is tracked but is mutated by the R-key case below, and a trap that # restored only the sources would leave tpshell BUILT FROM the mutation - so # the trap rebuilds too. All three copies live in the project's own tmp/ and # are taken here, before the first mutation, rather than beside each case. # # comimage.py and comtest.py are the image layout and one of its three readers; # both are mutated by cases further down, and neither is recoverable from git # once the mutation is staged over - so they join the trap here for the same # reason, and an interrupted run cannot leave the layout half-written. mkdir -p ../tmp SAVED_E=../tmp/nonvacuity.Exec86.mod SAVED_SH=../tmp/nonvacuity.Shell.mod SAVED_CI=../tmp/nonvacuity.comimage.py SAVED_T=../tmp/nonvacuity.comtest.py cp Runtime.mod "$SAVED" || exit 1 cp Exec86.mod "$SAVED_E" || exit 1 cp Shell.mod "$SAVED_SH" || exit 1 cp tests/comimage.py "$SAVED_CI" || exit 1 cp tests/comtest.py "$SAVED_T" || exit 1 restore_all () { cp "$SAVED" Runtime.mod cp "$SAVED_E" Exec86.mod cp "$SAVED_SH" Shell.mod cp "$SAVED_CI" tests/comimage.py cp "$SAVED_T" tests/comtest.py "$GM2" -fiso -c Runtime.mod >/dev/null 2>&1 "$GM2" -fiso -c Exec86.mod >/dev/null 2>&1 # The shell is rebuilt as well: a test that runs against a binary built # from a half-restored tree is reporting on the mutation, not on the code. make >/dev/null 2>&1 } trap restore_all EXIT pass=0 fail=0 # mutate -- apply a deliberate breakage and INSIST it landed. # # Four cases in this file were already dead when first run, all the same way: # the helper they name had been renamed or reformatted since the case was # written, the sed matched nothing, the source was unchanged, and the check # correctly passed - so the harness reported "NOT NON-VACUOUS" and, worse, a # reader skimming the output could take "the check still passed" for a passing # test. A case that cannot fire is worse than no case: it is a claim of # coverage that was never tested. # # So the mutation is verified, not assumed. If the file is byte-identical # afterwards, that is reported as a FAILURE of the harness, naming the sed, and # the case is not run - because running it would only produce a meaningless # green. The message says what to do (fix the sed) rather than what it found. mutate () { mf=$1 msed=$2 cp "$mf" ../tmp/nonvacuity.mut.bak sed -i "$msed" "$mf" if cmp -s "$mf" ../tmp/nonvacuity.mut.bak; then echo " BROKEN CASE: the mutation did not change $mf" echo " sed: $msed" echo " the named code has probably been renamed or reformatted -" echo " fix this case, it is asserting nothing" fail=$((fail + 1)) return 1 fi return 0 } # rebuild