The two new fault-by-design programs came into the corpus with the bytes/bytes-view lane and the survey has been comparing them at its own optimisation level, where the store into .rodata is undefined, LLVM deletes it and exits 0, and this backend -- which has no optimiser to delete anything with -- executes it and takes 139. That reads as a lowering disagreement and is not one: at -O0 the two backends agree exactly, and test_acceptance.ml's dies_segv rows already pin that on both of them. So a second exclusion list beside the one for the programs that never stop, with its own reason written down, rather than building the whole corpus at -O0 and changing the measurement every baseline was taken against. dev-segv would belong on it in any case: it calls agent/start, and under --dev it parks in the break loop rather than dying. MATCH 171, DIFFER 0, REFUSED 0, NOX86 0, SKIP 47
216 lines
9.9 KiB
Bash
Executable File
216 lines
9.9 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Does the hand-written backend agree with LLVM?
|
|
#
|
|
# The only honest test of a hand-encoded backend is what the program prints and
|
|
# what it exits with -- docs/DISCUSS.md item 15 and item 16 both say so, and both
|
|
# say it after a disassembly that read perfectly beside a wrong answer. So this
|
|
# builds every program in test/programs twice, runs both, and diffs stdout,
|
|
# stderr and the exit status. objdump is for after a program already has the
|
|
# wrong answer.
|
|
#
|
|
# stderr is not an afterthought: every message the condition machinery produces
|
|
# goes there -- the bounds and slice errors, the three restart refusals, the
|
|
# transfer failure -- and each carries a location string this backend emits by
|
|
# hand as a .rodata label and a length in a register. An exit status of 134
|
|
# with the wrong text beside it is exactly the failure that looks like a
|
|
# match.
|
|
#
|
|
# Both sides get the same bounds-check setting (the default: on). A sweep that
|
|
# compared a checked build against an unchecked one would say nothing about
|
|
# bounds.flan, which is the one program the two backends disagreed about.
|
|
#
|
|
# Five outcomes, and the third is the progress meter:
|
|
#
|
|
# MATCH built both ways, same stdout, same stderr, same exit status
|
|
# DIFFER built both ways, and disagreed
|
|
# REFUSED X86.Unsupported -- a node this backend does not lower (exit 3)
|
|
# NOX86 failed to build through --x86 for some other reason
|
|
# SKIP no main, does not compile at all, or does not terminate
|
|
#
|
|
# Over test/programs, over spike/x86's own probes, which are here for the paths
|
|
# the corpus does not walk, and over spike/js's, which are here for another
|
|
# backend's decisions but are plain flan programs and so are evidence about
|
|
# this one too: p1-int-semantics.flan had been catching a wrong shift for as
|
|
# long as this sweep had been not reading it.
|
|
#
|
|
# Usage: spike/x86/survey.sh [name-substring ...]
|
|
set -u
|
|
orig=$(pwd)
|
|
here=$(cd "$(dirname "$0")" && pwd)
|
|
root=$(cd "$here/../.." && pwd)
|
|
cd "$root" || exit 1
|
|
|
|
# FLAN is how the dune @x86 alias hands this script a compiler that dune has
|
|
# already built. Building it here instead would be a second dune inside the
|
|
# first one's lock, which does not run at all; and the alias has bin/main.exe
|
|
# in its deps precisely so it does not have to. Standalone -- the way the
|
|
# baseline in every handoff was measured -- nothing sets it and the build
|
|
# below is what it always was.
|
|
if [ -n "${FLAN:-}" ]; then
|
|
# Resolved against the directory this was invoked from, not against $root:
|
|
# dune spells its deps relative to the dune file, and the cd above has
|
|
# already happened by the time this is read.
|
|
case $FLAN in /*) flan=$FLAN;; *) flan=$orig/$FLAN;; esac
|
|
else
|
|
dune build --root . bin/main.exe 2>&1 | head -30
|
|
flan=$root/_build/default/bin/main.exe
|
|
fi
|
|
test -x "$flan" || { echo "build failed"; exit 1; }
|
|
|
|
# Where the corpus is read from. Under dune the script runs from the build
|
|
# tree, where test/programs is present but spike/x86 is not, so the alias
|
|
# points this at the source tree and gets both.
|
|
corpus=${SURVEY_CORPUS:-$root}
|
|
|
|
out=$(mktemp -d); trap 'rm -rf "$out"' EXIT
|
|
|
|
# The ones that run until something stops them. Not a failure and not a match;
|
|
# they are excluded by name because a timeout cannot tell them apart from a
|
|
# backend that hung.
|
|
#
|
|
# dev-chatty is the third and is here for a sharper reason than the other two.
|
|
# It also outlives the timeout -- it is sized to outlast test_dev's checks and
|
|
# is killed with the connection -- but what makes it unusable here is that it
|
|
# *prints* while it does, 4K a frame. Two backends stopped by a clock stop at
|
|
# different lines, so the diff is a report about scheduling rather than about
|
|
# lowering, and it fails the alias every run. dev-repl outlives the timeout in
|
|
# the same way and is not listed, because it prints nothing and the two
|
|
# truncations are both empty. agent-auto is here for the dev-loop reason with a
|
|
# different clock: its ticks wait on an agent connection that never comes when
|
|
# it is run standalone, both backends sit until the timeout, and what each has
|
|
# printed by then (a randomized socket path among it) is not a lowering
|
|
# comparison. test_agent.ml is where that program's behaviour is asserted.
|
|
forever="dev-loop dev-watch dev-chatty agent-auto"
|
|
|
|
# There used to be a second exclusion list here, holding the five dyn
|
|
# programs, and its note said to take a name off it when the backend grew the
|
|
# lowering and the sweep would then say whether it works. The backend grew it,
|
|
# so the list is gone rather than empty: a dyn is one machine word in both
|
|
# calling conventions and every operation on one is an ordinary runtime call.
|
|
# What the lane cost was the collector's root discipline -- a zeroed frame
|
|
# slot per dyn local and per dyn-producing call, pushed at entry, and one pop
|
|
# in the epilogue that every exit already went through. The five are in the
|
|
# sweep now and they are five of the MATCHes.
|
|
|
|
# The ones whose whole point is a fault, and which therefore cannot be compared
|
|
# at this sweep's optimisation level. Both write through a bytes-view of a
|
|
# string literal, which is a store into .rodata: measured here, LLVM exits 0
|
|
# having printed the unmodified literal and this backend exits 139, because the
|
|
# store is undefined and the optimiser deleted it on one side and there is no
|
|
# optimiser on the other. That is not a lowering disagreement. At -O0 the two
|
|
# agree exactly -- 139, no output, both backends -- and test_acceptance.ml's
|
|
# dies_segv rows pin precisely that, on both backends, which is the coverage
|
|
# this sweep would otherwise be duplicating at the one level where, as that
|
|
# file's own comment puts it, there is nothing left to pin but the UB.
|
|
#
|
|
# Excluded by name rather than by building the whole corpus at -O0: the counts
|
|
# below are one measurement and every handoff's baseline was taken against it.
|
|
#
|
|
# dev-segv would not belong here even if the store survived. It calls
|
|
# agent/start, so it leaves a socket in /tmp on both runs, and under
|
|
# SURVEY_FLAGS=--dev it parks in the break loop instead of dying -- two
|
|
# truncations at the timeout, which is the dev-chatty failure by another road.
|
|
faults="bytes-view-write dev-segv"
|
|
|
|
TIMEOUT=${TIMEOUT:-20}
|
|
|
|
# Extra flags, given to *both* sides. SURVEY_FLAGS=--dev is the one that has a
|
|
# use: a dev build with nothing yet redefined must behave exactly like a
|
|
# release one -- the indirection cell is the only difference -- so the whole
|
|
# corpus is a test of the cells, and of nothing else changing beside them.
|
|
# Off by default, so the counts above the line stay the same measurement.
|
|
read -r -a extra <<<"${SURVEY_FLAGS:-}"
|
|
|
|
declare -a match=() differ=() refused=() nox86=() skip=()
|
|
|
|
for src in "$corpus"/test/programs/*.flan "$corpus"/spike/x86/*.flan \
|
|
"$corpus"/spike/js/*.flan; do
|
|
# An unmatched glob comes through as its own pattern; a corpus that is
|
|
# missing one of these directories is a smaller sweep, not an error.
|
|
[ -e "$src" ] || continue
|
|
name=$(basename "$src" .flan)
|
|
if [ $# -gt 0 ]; then
|
|
want=0
|
|
for pat in "$@"; do case "$name" in *"$pat"*) want=1;; esac; done
|
|
[ $want = 1 ] || continue
|
|
fi
|
|
case " $forever " in *" $name "*) skip+=("$name:runs-forever"); continue;; esac
|
|
case " $faults " in *" $name "*) skip+=("$name:faults-by-design"); continue;; esac
|
|
|
|
# LLVM first. A program that does not compile at all, or has no main, is not
|
|
# this backend's business -- the frontend refused it either way.
|
|
if ! "$flan" build "$src" "${extra[@]}" -o "$out/$name.llvm" \
|
|
>"$out/$name.llvm.err" 2>&1; then
|
|
if grep -q "in function \`_start\|undefined reference to \`main\|crt1.o" "$out/$name.llvm.err"; then
|
|
skip+=("$name:no-main")
|
|
else
|
|
skip+=("$name:does-not-compile")
|
|
fi
|
|
continue
|
|
fi
|
|
|
|
"$flan" build "$src" --x86 "${extra[@]}" -o "$out/$name.x86" \
|
|
>"$out/$name.x86.err" 2>&1
|
|
rc=$?
|
|
if [ $rc = 3 ]; then
|
|
why=$(head -1 "$out/$name.x86.err" | sed 's/^x86: //')
|
|
refused+=("$name:$why")
|
|
continue
|
|
fi
|
|
if [ $rc != 0 ]; then
|
|
nox86+=("$name:$(head -1 "$out/$name.x86.err")")
|
|
continue
|
|
fi
|
|
|
|
( cd "$out" && timeout "$TIMEOUT" "$out/$name.llvm" \
|
|
>"$out/$name.llvm.out" 2>"$out/$name.llvm.diag" )
|
|
a=$?
|
|
( cd "$out" && timeout "$TIMEOUT" "$out/$name.x86" \
|
|
>"$out/$name.x86.out" 2>"$out/$name.x86.diag" )
|
|
b=$?
|
|
if [ "$a" = "$b" ] && cmp -s "$out/$name.llvm.out" "$out/$name.x86.out" \
|
|
&& cmp -s "$out/$name.llvm.diag" "$out/$name.x86.diag"; then
|
|
match+=("$name")
|
|
else
|
|
differ+=("$name:llvm=$a/x86=$b")
|
|
if [ "${SURVEY_SHOW:-}" = 1 ]; then
|
|
echo "--- $name: llvm exit $a, x86 exit $b"
|
|
diff "$out/$name.llvm.out" "$out/$name.x86.out" | head -20
|
|
diff "$out/$name.llvm.diag" "$out/$name.x86.diag" | head -20
|
|
fi
|
|
fi
|
|
done
|
|
|
|
echo
|
|
echo "MATCH ${#match[@]}"
|
|
echo "DIFFER ${#differ[@]}"
|
|
[ "${#differ[@]}" = 0 ] || printf ' %s\n' "${differ[@]}"
|
|
echo "REFUSED ${#refused[@]}"
|
|
if [ "${#refused[@]}" != 0 ] && [ "${SURVEY_QUIET:-}" != 1 ]; then
|
|
printf '%s\n' "${refused[@]}" | sed 's/^[^:]*://' | sort | uniq -c | sort -rn \
|
|
| sed 's/^/ /'
|
|
fi
|
|
echo "NOX86 ${#nox86[@]}"
|
|
[ "${#nox86[@]}" = 0 ] || printf ' %s\n' "${nox86[@]}"
|
|
echo "SKIP ${#skip[@]}"
|
|
if [ "${#skip[@]}" != 0 ] && [ "${SURVEY_QUIET:-}" != 1 ]; then
|
|
printf '%s\n' "${skip[@]}" | sed 's/^[^:]*://' | sort | uniq -c \
|
|
| sed 's/^/ /'
|
|
fi
|
|
|
|
# Strict mode, for the @x86 alias: the counts above are a report, and a report
|
|
# nobody reads is how two refusals from another lane's new primitive sat in
|
|
# the tree for a month. A DIFFER is a wrong answer and a refusal by name is a
|
|
# node this backend has stopped lowering; either is a failure. NOX86 and SKIP
|
|
# are not: the first is usually a toolchain that is not installed here, and
|
|
# the second is the frontend refusing the program on both sides.
|
|
if [ "${SURVEY_STRICT:-}" = 1 ]; then
|
|
if [ "${#differ[@]}" != 0 ] || [ "${#refused[@]}" != 0 ]; then
|
|
echo
|
|
echo "x86 survey FAILED: ${#differ[@]} differ, ${#refused[@]} refused"
|
|
exit 1
|
|
fi
|
|
echo
|
|
echo "x86 survey ok: ${#match[@]} match"
|
|
fi
|