diff --git a/docs/v4.0.0/DECOMPOSITION.md b/docs/v4.0.0/DECOMPOSITION.md index 8ae8e135..a08fb356 100644 --- a/docs/v4.0.0/DECOMPOSITION.md +++ b/docs/v4.0.0/DECOMPOSITION.md @@ -170,6 +170,7 @@ definition below depends on one, it says so. | **D-6** | Which heat structures exist in hardware: per-opcode counters only, or also per-call-target and word-to-word transition counters. | **Deferred** to step 2. The golden model implements per-opcode and per-call-target heat in the meantime. | | **D-7** | `ROLL` semantics. | **Moot.** `ROLL` is retired under D-2. | | **D-8** | Q48.16 signedness. | **Signed.** v3's unsigned comparisons and `Q.FROM-INT` clamping are retired. | +| **D-9** | Instruction word width on a 64-bit-cell host (ruled 2026-10-02). | **32 bits at every cell width.** Six 5-bit slots plus 2 spare bits, as in §1.2. On a 64-bit host the instruction word is the low 32 bits of the cell and the high half is ignored, so compiled code is identical at both widths. Only data and `@p` literals are a full cell wide. | **Consequences of D-2 that every definition must respect.** The data stack holds 10 items and the return stack 9, and every `call`, `FOR`, `DO` loop frame and `push` uses return-stack slots. Nesting diff --git a/v4/.gitignore b/v4/.gitignore new file mode 100644 index 00000000..e757b735 --- /dev/null +++ b/v4/.gitignore @@ -0,0 +1,2 @@ +# Build output. Regenerated by `make test` / `make sanitize`. +build/ diff --git a/v4/Makefile b/v4/Makefile new file mode 100644 index 00000000..bd8a2905 --- /dev/null +++ b/v4/Makefile @@ -0,0 +1,105 @@ +# v4/Makefile -- host golden model build. +# +# JUSTIFICATION.md section 5: the host node runs the same mesh-defined +# instruction set as the hardware, so the C99 model is the reference the +# hardware is compared against -- not a host-specific reimplementation. +# There is consequently nothing v4-specific in the architecture here; the +# build is plain hosted C99 and the ISA is defined in DECOMPOSITION.md. +# +# The two builds that matter are the two cell widths. v4/Makefile builds and +# runs the tests at both, because a width-dependent bug that only appears at +# 32 bits is exactly the failure mode the F18's 32-bit cells invite and +# exactly the one a 64-bit-only test run would miss. +# +# C99, warnings fatal. A warning here is a defect in a model whose only job +# is to be trusted. + +CC ?= cc +CSTD := -std=c99 +WARN := -Wall -Wextra -Wpedantic -Werror +OPT ?= -O2 -g +CFLAGS ?= $(CSTD) $(WARN) $(OPT) + +# Anchor everything to this Makefile's directory so the build behaves the same +# whether it is invoked as `make -C v4 test` or `make -f v4/Makefile test` +# from the repository root. Without this the wildcards below silently expand +# to nothing from the root and the build "succeeds" having compiled no tests. +HERE := $(patsubst %/,%,$(dir $(abspath $(lastword $(MAKEFILE_LIST))))) + +SRCS := $(wildcard $(HERE)/src/*.c) +TESTS := $(wildcard $(HERE)/tests/*.c) + +# Cell widths to build and verify. 32 is the F18/Zynq case, 64 the aarch64 +# host case; see DECOMPOSITION.md D-5. +WIDTHS := 32 64 + +BINDIR := $(HERE)/build + +.PHONY: all test sanitize clean $(addprefix test-,$(WIDTHS)) + +all: test + +# A build that finds no tests must fail loudly, not report a vacuous pass. A +# loop over an empty list exits 0 having run nothing, which in CI is +# indistinguishable from a green run -- the worst possible failure mode for a +# test harness, and one this Makefile has already produced once. +ifeq ($(strip $(TESTS)),) +$(error No test sources under $(HERE)/tests -- refusing to report success) +endif + +# One test binary per (width, test source) pair. The cell width is a +# preprocessor parameter rather than a runtime one, so the two widths are +# separate compilations -- which is the point: there is no single code path +# that could paper over a width assumption. +# +# `make test` builds optimised; `make sanitize` builds the same sources under +# AddressSanitizer and UndefinedBehaviorSanitizer at both widths and runs them. +# The optimized run catches logic errors, this one catches the out-of-bounds +# ring index and the width-dependent shift that a logic-only test cannot see. +define TEST_RULE +$(BINDIR)/$(1)-$(notdir $(2)): $(2) $$(SRCS) $$(wildcard $(HERE)/include/v4/*.h) + @mkdir -p $(BINDIR) + $$(CC) $$(CFLAGS) -I$(HERE)/include -DV4_CELL_BITS=$(1) \ + $(2) $$(SRCS) -o $$@ + +$(BINDIR)/san-$(1)-$(notdir $(2)): $(2) $$(SRCS) $$(wildcard $(HERE)/include/v4/*.h) + @mkdir -p $(BINDIR) + $$(CC) $$(CSTD) $$(WARN) -O1 -g -fno-omit-frame-pointer \ + -fsanitize=address,undefined -I$(HERE)/include -DV4_CELL_BITS=$(1) \ + $(2) $$(SRCS) -o $$@ +endef + +# One run target per (width, test) and one prerequisite line per width, so +# adding a test file cannot redefine a recipe and the run cannot skip the +# build. A shell loop over `$^` is avoided deliberately: `$^` is make syntax +# that does not survive the define/eval expansion, and a silently-empty loop +# list is exactly the vacuous-pass failure this file already made once. +define RUN_RULE +run-$(1)-$(notdir $(2)): $(BINDIR)/$(1)-$(notdir $(2)) + @echo " [V4_CELL_BITS=$(1) $$@]" + @$(BINDIR)/$(1)-$(notdir $(2)) + +test-$(1): run-$(1)-$(notdir $(2)) +endef + +# The same sources and the same assertions, under ASan+UBSan. +define SAN_RULE +san-$(1)-$(notdir $(2)): $(BINDIR)/san-$(1)-$(notdir $(2)) + @echo " [ASan+UBSan V4_CELL_BITS=$(1) $$@]" + @$(BINDIR)/san-$(1)-$(notdir $(2)) +endef + +$(foreach w,$(WIDTHS),$(foreach t,$(TESTS),$(eval $(call TEST_RULE,$(w),$(t))))) +$(foreach w,$(WIDTHS),$(foreach t,$(TESTS),$(eval $(call RUN_RULE,$(w),$(t))))) +$(foreach w,$(WIDTHS),$(foreach t,$(TESTS),$(eval $(call SAN_RULE,$(w),$(t))))) + +sanitize: $(foreach w,$(WIDTHS),$(foreach t,$(TESTS),san-$(w)-$(notdir $(t)))) + @echo "all v4 tests passed under ASan+UBSan" + +.PHONY: $(foreach w,$(WIDTHS),$(foreach t,$(TESTS),run-$(w)-$(notdir $(t)) san-$(w)-$(notdir $(t)))) + +test: $(addprefix test-,$(WIDTHS)) + @echo "all v4 tests passed" + +clean: + rm -rf $(BINDIR) diff --git a/v4/README.md b/v4/README.md index d1beb6b1..3b1251dd 100644 --- a/v4/README.md +++ b/v4/README.md @@ -4,6 +4,12 @@ StarForth v4: the 32-instruction F18-derived core. The design lives in `docs/v4.0.0/JUSTIFICATION.md` (why) and `docs/v4.0.0/DECOMPOSITION.md` (every v3 word mapped to a v4 fate). -There is no code yet. The first deliverable is the hosted C99 golden model -(`JUSTIFICATION.md` §10, step 1). It must pass POST and hold K≡1.0 with -32- and 64-bit cells on all three host ISAs. +The first deliverable is the hosted C99 golden model (`JUSTIFICATION.md` §10, +step 1). It must pass POST and hold K≡1.0 with 32- and 64-bit cells on all +three host ISAs. + +What exists so far is the single node: registers, circular stacks, memory, all +32 opcodes, and per-opcode and per-call-target heat. `make -C v4 test` builds +and runs the tests at both cell widths; `make -C v4 sanitize` repeats them +under ASan and UBSan. There is no compiler capsule, no POST and no K +measurement yet. diff --git a/v4/include/v4/asm.h b/v4/include/v4/asm.h new file mode 100644 index 00000000..619cd9d0 --- /dev/null +++ b/v4/include/v4/asm.h @@ -0,0 +1,68 @@ +/* asm.h -- slot packer for hand-assembled test programs. + * + * This is not the compiler capsule (JUSTIFICATION.md section 9). It exists so + * that tests can write a DECOMPOSITION.md definition opcode by opcode and have + * the slots, literals and branch targets laid out the way section 1.2 and + * section 2 require, instead of packing words by hand in every test. + * + * Layout rules it applies: + * + * - Opcodes fill slots 0..5 left to right; a full word is written out and a + * new one started. Unused slots are padded with `nop`. + * - A literal is `@p` in a slot plus its value in the word after the + * instruction word. Several literals in one word follow it in order. + * - `;` and `ex` close the word, since nothing after them executes. + * - A branch closes the word, since the bits to its right are its address. + * If the next free slot is 4 or 5 the word is closed first, because + * branches are legal in slots 0-3 only. + * - A branch target must lie in the same page as P at the time the branch + * executes, for the reach of the slot it landed in. Otherwise the + * assembler sets its error flag rather than emit a branch that goes + * somewhere else. + */ +#ifndef V4_ASM_H +#define V4_ASM_H + +#include "v4/iword.h" +#include "v4/node.h" + +typedef struct { + v4_node *n; + v4_cell here; /* address of the word being built */ + v4_iword cur; + unsigned slot; /* next free slot, 0..V4_SLOT_COUNT */ + v4_cell lit[V4_SLOT_COUNT]; /* literals owed after this word */ + unsigned nlit; + int error; +} v4_asm; + +/* A branch whose target is not known yet. */ +typedef struct { + v4_cell addr; /* the word holding the branch */ + unsigned slot; + v4_cell p; /* P when the branch executes */ +} v4_asm_ref; + +void v4_asm_begin(v4_asm *as, v4_node *n, v4_cell origin); + +/* Any opcode that is neither a branch nor `@p`. */ +void v4_asm_op(v4_asm *as, unsigned op); + +/* `@p` and its value. */ +void v4_asm_lit(v4_asm *as, v4_cell value); + +/* jump, call, next, if, -if to a known address. */ +void v4_asm_branch(v4_asm *as, unsigned op, v4_cell target); + +/* The same with the target supplied later by v4_asm_resolve. */ +v4_asm_ref v4_asm_branch_fwd(v4_asm *as, unsigned op); +void v4_asm_resolve(v4_asm *as, v4_asm_ref ref, v4_cell target); + +/* Close the word being built and return the address of the next one. This is + * how a label is taken: branch targets are word addresses. */ +v4_cell v4_asm_label(v4_asm *as); + +/* 1 if nothing went wrong so far. Closes the word being built. */ +int v4_asm_ok(v4_asm *as); + +#endif /* V4_ASM_H */ diff --git a/v4/include/v4/cell.h b/v4/include/v4/cell.h new file mode 100644 index 00000000..9f5286b2 --- /dev/null +++ b/v4/include/v4/cell.h @@ -0,0 +1,60 @@ +/* cell.h -- v4 cell width as a build parameter. + * + * DECOMPOSITION.md D-5: the host node's cell width matches the host CPU + * (32 on a Zynq-7000 Cortex-A9, 64 on an aarch64 host). JUSTIFICATION.md + * section 5 makes this the second invariance axis: if K==1.0 holds across + * cell widths as well as across ISAs, the conservation law is shown not to + * depend on word size. + * + * The compiler capsule is written width-independent against these types. + * Nothing in v4 may assume a particular V4_CELL_BITS; if you find yourself + * wanting to, the code belongs in the host node, not the mesh. + * + * JUSTIFICATION.md section 5 also fixes the split that matters for the + * physics: heat and K arithmetic stay 64-bit on every host regardless of + * cell width, because they must not lose precision when the cell shrinks. + * That is why this header defines a separate v4_heat_t rather than reusing + * v4_cell. + */ +#ifndef V4_CELL_H +#define V4_CELL_H + +#include + +#ifndef V4_CELL_BITS +#error "V4_CELL_BITS must be defined to 32 or 64 (e.g. -DV4_CELL_BITS=32)." +#endif + +#if V4_CELL_BITS == 32 +typedef int32_t v4_cell; +typedef uint32_t v4_ucell; +#elif V4_CELL_BITS == 64 +typedef int64_t v4_cell; +typedef uint64_t v4_ucell; +#else +#error "V4_CELL_BITS must be 32 or 64." +#endif + +/* Physics/heat arithmetic width. Fixed at 64 by JUSTIFICATION.md section 5: + * "Heat and K arithmetic remain 64-bit (int64_t in C99 on every host; double + * cells on a 32-bit node). Changing only the payload width keeps the + * experiment clean: any difference in K can be attributed to cell width and + * not to lost precision." The C side is therefore always 64-bit; the + * double-cell framing is the 32-bit node's problem, not this type's. */ +typedef int64_t v4_heat_t; +typedef uint64_t v4_uheat_t; + +/* MSB -- the cell-width top-bit constant. + * + * DECOMPOSITION.md section 5.5 (RSHIFT) and section 5.7 (D2/) both call for + * "MSB, the cell-width top-bit constant" but never define it, which leaves a + * bit-width-dependent constant to be invented mid-arithmetic. It is defined + * here once, as a value with exactly the cell's sign bit set and nothing + * else, so that the sign-test idioms in those two words are width-correct by + * construction at 32 and at 64. */ +#define V4_MSB (((v4_ucell)1) << (V4_CELL_BITS - 1)) + +/* All-ones, the width-correct "true". FORTH-79 uses -1 for TRUE. */ +#define V4_ALL_ONES ((v4_cell)-1) + +#endif /* V4_CELL_H */ diff --git a/v4/include/v4/exec.h b/v4/include/v4/exec.h new file mode 100644 index 00000000..e238ab4a --- /dev/null +++ b/v4/include/v4/exec.h @@ -0,0 +1,56 @@ +/* exec.h -- instruction execution for v4. + * + * DECOMPOSITION.md section 1.2: "Slots execute left to right." A branch + * (jump, call, next, if, -if) takes its target from all bits to the right of + * its slot and replaces the same low bits of P (page-relative). Branches are + * legal in slots 0-3 only. Nothing to the right of a branch executes, taken + * or not, because those bits are its address. + * + * Section 1.4: "Every retired instruction increments that opcode's heat counter + * and advances the anti-clock; every `call` increments the heat counter of its + * target. None of this costs an instruction." + * + * The node owns P, A, B, the stacks and memory; heat and the anti-clock are + * separate physics state (section 7 names them as registers). + * + * THREE POINTS THE SPEC LEAVES TO THE F18, and which this model takes from it: + * + * - `ex` ends the word, like `;`. P has been replaced, so the remaining + * slots belong to a word that is no longer the current one. + * - `unext` restarts the *already fetched* word at slot 0. It does not + * refetch and does not move P, so `@p` and `!p` inside a unext loop step + * through the words that follow the loop word. When R reaches 0 the loop + * ends and execution continues with the slot after `unext`. + * - `+*` shifts T arithmetically (sign bit kept). Section 1.3 says only + * "shift T:A right one bit"; D-3 says plain F18 semantics. + * + * A branch opcode in slot 4 or 5 is outside the spec. The model does not trap + * it: it applies the 7- or 2-bit target the layout formula gives. + */ +#ifndef V4_EXEC_H +#define V4_EXEC_H + +#include "v4/heat.h" +#include "v4/iword.h" +#include "v4/node.h" + +typedef struct { + v4_uheat_t anticlock; +} v4_exec_state; + +void v4_exec_reset(v4_exec_state *es); + +/* Execute one instruction word: fetch at P, advance P, run the slots left to + * right until the word ends. A unext loop runs to completion inside one call, + * so the call does not return while R is nonzero at a `unext`. Returns the + * number of instructions retired. */ +unsigned v4_exec_step_word(v4_node *n, v4_exec_state *es, v4_heat *h); + +/* Execute a single opcode. The branch opcodes read their target from `w` and + * `slot`; every other opcode ignores both. `unext` here only updates R; the + * restart is v4_exec_step_word's. A value above 31 does nothing and retires + * nothing. */ +void v4_exec_op(v4_node *n, v4_exec_state *es, v4_heat *h, + unsigned op, v4_iword w, unsigned slot); + +#endif /* V4_EXEC_H */ diff --git a/v4/include/v4/guard.h b/v4/include/v4/guard.h new file mode 100644 index 00000000..9e9b5058 --- /dev/null +++ b/v4/include/v4/guard.h @@ -0,0 +1,92 @@ +/* guard.h -- 5% safety boundaries at each end of every bounded list. + * + * The convention used throughout this project is that a bounded list carries a + * safety boundary at each end rather than trusting its own bounds. The kernel + * arena is the reference instance (kernel/src/vm/arena.c): + * + * #define SK_VM_GUARD_SIZE PMM_PAGE_SIZE + * #define SK_VM_GUARD_PATTERN_HEAD 0x5ac0ffeed0c0ffeeULL + * #define SK_VM_GUARD_PATTERN_TAIL 0x0bad0badf00df00dULL + * + * with a guard_fill / guard_check pair, and total size = payload + 2 * guard. + * Two properties of that instance are kept here deliberately: + * + * - HEAD and TAIL carry *different* patterns, so a check that fails can say + * which end was overrun. A single pattern tells you that something ran + * off the end and nothing about where. + * - the boundary is placed immediately adjacent to the payload, because a + * boundary separated from what it protects catches nothing. + * + * This header supplies the same facility at a size that scales with the list + * (5% at each end) rather than a fixed page, and reduces it to two helpers so + * that no list in v4 has to hand-roll the arithmetic. + * + * WHY THIS IS LOAD-BEARING HERE, NOT DECORATIVE. AddressSanitizer already + * catches a genuine out-of-struct access, so a guard band cannot be justified + * as "catching overflow" -- ASan does that. What ASan does *not* catch is an + * index that is wrong but still lands inside the same object, which is exactly + * the failure mode of a circular buffer driven by unsigned index arithmetic: + * + * head = (head + V4_DATA_RING - 1u) % V4_DATA_RING; correct + * head = (head + V4_DATA_RING - 1u) & (V4_DATA_RING - 1u); correct today + * + * The second line is identical today *only because 8 is a power of two*. It + * is the sort of "simplification" that gets made, it is right for the current + * depths, and it silently breaks the first time a ring size stops being a power + * of two. Both forms stay in range, so ASan stays quiet. A guard band with a + * 5% margin either side is what turns "the tests happened to cover it" into a + * check that fires. The tests in tests/test_stack.c exercise that directly: + * they index past each end on purpose and require the check to catch it. + * + * The boundary is sized with ceiling division and a floor of one element, so a + * short list still gets a real boundary. A 5% margin on a 3-element list + * rounds to 0, and a zero-width guard is not a guard -- it is a comment. + */ +#ifndef V4_GUARD_H +#define V4_GUARD_H + +#include "v4/cell.h" + +/* Fraction of the payload reserved as a boundary at each end. */ +#define V4_GUARD_PCT 5u + +/* Boundary width, in elements, for a payload of n elements: 5% rounded up, + * never less than one. Written as a macro so that a struct can size its own + * inline boundary at compile time -- the boundary has to be storage the object + * owns, not something a helper allocates later, because a mesh node has no + * heap to allocate it from. + * + * n = 8 -> 1 (5% = 0.4, rounds up) + * n = 32 -> 2 (5% = 1.6, rounds up) + * n = 100 -> 5 + * n = 1000 -> 50 + * n = 3 -> 1 (5% = 0.15, rounds up to 1 -- not 0) + */ +#define V4_GUARD_ELEMS(n) \ + (((n) * V4_GUARD_PCT + 99u) / 100u) + +/* The kernel arena's patterns, carried over so that a hex dump from a mesh + * node and one from the kernel can be read with the same key. Casting to + * v4_ucell truncates cleanly at 32-bit cells: the arena's own constants are + * 64-bit, and their low halves are as distinct as the full words -- + * 0xd0c0ffee against 0xf00df00d -- so "which end" stays answerable at both + * widths. */ +#define V4_GUARD_PATTERN_HEAD ((v4_ucell)0x5ac0ffeed0c0ffeeULL) +#define V4_GUARD_PATTERN_TAIL ((v4_ucell)0x0bad0badf00df00dULL) + +/* If these ever compare equal the boundary has stopped saying which end was + * overrun, which is half its purpose. C99 has no _Static_assert. */ +typedef char v4_guard_patterns_are_distinct + [(V4_GUARD_PATTERN_HEAD != V4_GUARD_PATTERN_TAIL) ? 1 : -1]; + +/* Write one boundary band. `band` is the start of the band, `n` its width in + * elements. */ +void v4_guard_fill(v4_cell *band, unsigned n, v4_ucell pattern); + +/* 1 if the whole band still holds `pattern`, else 0. Checks every element, not + * just the ends: the band is 5% wide precisely so there is something to check + * in it, and a single word is what the kernel arena can afford at page + * granularity. */ +int v4_guard_intact(const v4_cell *band, unsigned n, v4_ucell pattern); + +#endif /* V4_GUARD_H */ diff --git a/v4/include/v4/heat.h b/v4/include/v4/heat.h new file mode 100644 index 00000000..96d8832a --- /dev/null +++ b/v4/include/v4/heat.h @@ -0,0 +1,79 @@ +/* heat.h -- heat and anti-clock, not instructions themselves (1.4). + * + * DECOMPOSITION.md section 1.4 states: "Execution itself drives the physics. + * Every retired instruction increments that opcode's heat counter and advances + * the anti-clock; every `call` increments the heat counter of its target. None + * of this costs an instruction." + * + * D-6 is deferred to step 2, but the note immediately adds: "The golden model + * implements per-opcode and per-call-target heat in the meantime." The + * counters here are therefore a two-dimensional heat structure (op × target) + * on a per-call basis, with counts stored in 64-bit (because heat is + * accumulated and must not roll over at 32-bit). The "per-call-target" + * counter is indexed by the address of the called target -- on a word-addressed + * node, that is a cell value equal to the address `a` passed to `call a`. + * + * The anti-clock is a virtual tick driven by execution -- one tick per retired + * instruction -- and the rest of §7 refers to it via `ANTICLOCK`. It is not a + * wall-clock time, it is a function of the execution stream, as specified. + * + * The MM names in §7: `HEAT-OP[0..31]` (R), `HEAT-CALL[...]` (R/W) with a + * freeze bit (D-6). D-4 (memory map) is deferred, so the model does not + * materialize those MM addresses yet. This header defines the backing storage + * that a later MM layer will expose; it keeps the counts separate from the node + * itself because heat is physics data, not the instruction word stream, and it + * needs to be accessible to assertions such as K conservation tests later. + * + * Counts are 64-bit. Even for a short run a call-heavy word can heat up many + * times, and the golden model must run POST and remain monotonic and + * comparable. They are stored as `v4_uheat_t` (which is uint64_t by cell-width + * parameterisation) rather than cell-sized so that a 32-bit node's heat is not + * width-truncated in the model. (JUSTIFICATION.md §5 keeps heat arithmetic + * 64-bit regardless of cell width.) + */ +#ifndef V4_HEAT_H +#define V4_HEAT_H + +#include "v4/cell.h" +#include "v4/opcode.h" + +#define V4_HEAT_MAX_CALL_TARGETS 4096u + +typedef struct { + v4_uheat_t op[V4_OPCODE_COUNT]; /* HEAT-OP[0..31] */ + + /* Per-call-target heat. Indexed by target address. The table is fixed-size + * for the golden model; the actual hardware table size is deferred with D-6, + * but "in the meantime" the model must have a per-call-target structure. + * 4096 entries is small (32KB if using 64-bit per entry), enough to cover a + * typical test program's call targets without needing to allocate on-the-fly + * -- and since the node has boundaries, over-indexing is caught. */ + v4_uheat_t call[V4_HEAT_MAX_CALL_TARGETS]; + v4_ucell call_freeze_mask[V4_HEAT_MAX_CALL_TARGETS / (sizeof(v4_ucell) * 8u)]; + /* freeze bits: one per call-target entry (D-6). Stored as a bitmask. */ +} v4_heat; + +void v4_heat_reset(v4_heat *h); + +/* Every retired instruction: +1 to the opcode's heat and +1 to ANTICLOCK. */ +void v4_heat_on_retire(v4_heat *h, unsigned op, v4_uheat_t *anticlock); + +/* Every call: +1 to the call-target's heat. The target is the address passed + * to call (word address). If the target is outside the table range, the model + * must not silently wrap -- a boundary-style check is the conservative choice + * here, because the table size is a model parameter. */ +void v4_heat_on_call(v4_heat *h, v4_cell target); + +/* Read/write helpers for per-call-target heat (R/W). Freeze bits are set per + * target when D-6's freeze bit is asserted; writes to a frozen target should be + * ignored in a real implementation, but the MM layer is not defined yet. The + * golden model stores freeze bits and the accessor can enforce them later. */ +v4_uheat_t v4_heat_call_get(const v4_heat *h, v4_cell target); +int v4_heat_call_set(v4_heat *h, v4_cell target, v4_uheat_t val); + +/* Freeze control (D-6). */ +void v4_heat_call_freeze(v4_heat *h, v4_cell target); +void v4_heat_call_unfreeze(v4_heat *h, v4_cell target); +int v4_heat_call_is_frozen(const v4_heat *h, v4_cell target); + +#endif /* V4_HEAT_H */ diff --git a/v4/include/v4/iword.h b/v4/include/v4/iword.h new file mode 100644 index 00000000..cf75b7b1 --- /dev/null +++ b/v4/include/v4/iword.h @@ -0,0 +1,101 @@ +/* iword.h -- the 32-bit instruction word and its decode. + * + * DECOMPOSITION.md section 1.2: + * + * 31 27 26 22 21 17 16 12 11 7 6 2 1 0 + * +--------+--------+--------+--------+--------+--------+----+ + * | slot 0 | slot 1 | slot 2 | slot 3 | slot 4 | slot 5 | xx | + * +--------+--------+--------+--------+--------+--------+----+ + * + * "Slots execute left to right. A branch (jump, call, next, if, -if) takes + * its target from all bits to the right of its slot, and that target replaces + * the same low bits of P (page-relative, as in the F18). Branches are + * therefore legal in slots 0-3 only", with reach 27 / 22 / 17 / 12 bits. + * + * That reach column is not a separate rule, it falls out of the layout: slot k + * sits at bits [31-5k : 27-5k], so the bits to its right are [26-5k : 0], + * which is 27-5k bits -- 27, 22, 17, 12 for slots 0..3. This header derives + * the masks from that one formula rather than hard-coding four of them, so the + * decode cannot drift from the reach table. + * + * THE SPARE BITS ARE ADDRESS BITS. "Spare" means only that they are never + * *executed*: they are below every slot, so "all bits to the right of its + * slot" includes them, and they are the low bits of the address of a branch in + * any slot. They are inert for opcode decode and live in a branch target. + * This is easy to assume otherwise and get wrong -- an earlier revision of + * this header subtracted them from the branch width, which silently cost two + * address bits off every branch and was caught by tests/test_iword.c. They + * exist because 6*5 = 30 does not fill a 32-bit word, and on the F18 those low + * bits are simply the bottom of the address space. + * + * The consequence for the compiler is worth stating: a non-branch slot's low + * bits are free to carry a compressed argument, but a branch slot's are not, + * because they are address. + * + * NOTE ON WIDTH. The word format is 32 bits regardless of V4_CELL_BITS: the + * diagram is 32 bits wide and 6*5+2 = 32 exactly. D-5 makes the *cell* width + * a build parameter, but the instruction word is not a payload and does not + * follow the cell. At 32-bit cells a word is a cell. At 64-bit cells it is + * the low 32 bits of the cell and the high half is ignored (D-9), so compiled + * code is identical at both widths. The decode below operates on a distinct + * 32-bit type; the fetch in exec.c does the narrowing. + */ +#ifndef V4_IWORD_H +#define V4_IWORD_H + +#include + +#include "v4/cell.h" +#include "v4/opcode.h" + +/* The instruction word is 32 bits at every cell width. */ +typedef uint32_t v4_iword; + +#define V4_SLOT_COUNT 6u /* six 5-bit slots, DECOMPOSITION.md 1.1 */ +#define V4_SPARE_BITS 2u /* "plus 2 spare bits" */ + +/* C99 has no _Static_assert, so the "6*5+2 == 32" identity is enforced with + * the negative-array-size idiom. If the slot count, the spare-bit count or the + * word width ever stop adding up to 32, this fails to compile. */ +typedef char v4_iword_layout_is_exact + [((V4_SLOT_COUNT * 5u + V4_SPARE_BITS) == 32u) ? 1 : -1]; + +/* Bit range of slot k: high bit 31-5k, low bit 27-5k. */ +#define V4_SLOT_HIGH_BIT(k) (31u - 5u * (k)) +#define V4_SLOT_LOW_BIT(k) (27u - 5u * (k)) + +/* Number of address bits a branch in slot k carries: all bits to the right of + * the slot. Slot k's low bit is 27-5k, so the bits below it are 0..(26-5k), + * which is 27-5k bits -- equal to the slot's own low bit. This is the single + * source of the reach column, and it is why the spare bits are *not* subtracted + * here: see the note on the spare bits below. */ +#define V4_BRANCH_ADDR_BITS(k) (V4_SLOT_LOW_BIT(k)) + +/* Mask of the low address bits a branch in slot k replaces in P. */ +uint32_t v4_iword_slot_mask(unsigned slot); + +/* The 5-bit opcode in slot k, 0..31. */ +unsigned v4_iword_op(v4_iword w, unsigned slot); + +/* The 2 spare low bits. Never executed. */ +unsigned v4_iword_spare(v4_iword w); + +/* The branch target carried by a branch in slot k: all bits to the right of + * the slot, zero-extended. Meaningful only for the five branch opcodes. */ +v4_iword v4_iword_target(v4_iword w, unsigned slot); + +/* Assemble a word from six 5-bit opcodes and the 2 spare bits. Opcode values + * above 31 or spare bits above 3 are rejected, because a golden model must not + * paper over an out-of-range encoding. */ +v4_iword v4_iword_assemble(const unsigned op[V4_SLOT_COUNT], unsigned spare); + +/* Apply a branch: replace the low V4_BRANCH_ADDR_BITS(slot) bits of P with the + * target, preserving the bits above (page-relative, F18-style). */ +v4_cell v4_iword_branch(v4_cell p, v4_iword w, unsigned slot); + +/* Branches are legal in slots 0-3 only (section 1.2). Slots 4 and 5 have 7 and + * 2 address bits respectively, which is not a useful reach and is not + * sanctioned by the table. */ +int v4_iword_branch_legal(unsigned slot); + +#endif /* V4_IWORD_H */ diff --git a/v4/include/v4/node.h b/v4/include/v4/node.h new file mode 100644 index 00000000..f794c30d --- /dev/null +++ b/v4/include/v4/node.h @@ -0,0 +1,114 @@ +/* node.h -- the v4 node: registers, stacks, and word-addressed memory. + * + * DECOMPOSITION.md section 1.1: + * + * | Cell | 32 bits (mesh node). Host node's width is a build + * | parameter; see D-5. | + * | Addressing | Word-addressed (D-1). | + * | Registers | T (top of data stack), S (second), R (top of return + * | stack), P (program counter), A and B (address + * | registers). | + * | Stacks | F18 circular hardware stacks, not visible to code + * | (D-2) ... data stack 10 deep, return stack 9 deep. | + * | Instruction | Six 5-bit slots, plus 2 spare bits. | + * + * T, S and R are not fields here: they live inside v4_dstack and v4_rstack, + * because on the F18 they are the topmost slots of the circular stacks rather + * than registers that happen to shadow them, and the golden model has to model + * the mechanism and not just the values. P, A and B are genuinely separate + * registers and live here. + * + * MEMORY SIZE IS A BUILD PARAMETER, not a constant of the design. + * JUSTIFICATION.md section 10 names cell width, node count and node memory as + * the three parameters of the golden model, and no word count is fixed anywhere + * in DECOMPOSITION.md -- D-4, which is where a node memory map would be pinned + * down, is deferred to step 2. The default here is a round number that suits + * a host; the datapoint worth knowing is JUSTIFICATION.md section 4, that a + * GA144 node had 64 words of RAM and 64 of ROM. v4 keeps the same split -- + * dictionary and interpreter live in the compiler capsule on the host and are + * streamed to nodes, so node memory is not sized by capsule size. + */ +#ifndef V4_NODE_H +#define V4_NODE_H + +#include "v4/cell.h" +#include "v4/guard.h" +#include "v4/stack.h" + +/* Words of node memory. Overridable at build time. */ +#ifndef V4_NODE_WORDS +#define V4_NODE_WORDS 1024u +#endif + +/* 5% safety boundary at each end of node memory, per guard.h. Note the + * arithmetic: at 1024 words that is 52 words either side, 5.2% of memory + * spent on a boundary. That is the cost of the convention on a resource that + * is scarce by construction, and it is worth naming rather than discovering + * later -- on a real mesh node the memory is 32-bit words on silicon with no + * second chance. A deployment that cannot afford it sets V4_GUARD_PCT to 0 at + * build time, which is why the percentage is a macro and not baked in. */ +#define V4_MEM_BOUND V4_GUARD_ELEMS(V4_NODE_WORDS) + +typedef struct { + v4_cell p; /* P -- program counter */ + v4_cell a; /* A -- address register */ + v4_cell b; /* B -- address register */ + + v4_dstack ds; /* holds T and S */ + v4_rstack rs; /* holds R */ + + v4_cell mem_guard_head[V4_MEM_BOUND]; + v4_cell mem[V4_NODE_WORDS]; + v4_cell mem_guard_tail[V4_MEM_BOUND]; +} v4_node; + +/* Zero P, A and B, empty the stacks, and clear memory. Installs every + * boundary (memory and both stacks) and leaves the node in a defined state. */ +void v4_node_reset(v4_node *n); + +/* 1 if every boundary in the node is intact -- memory and both stacks -- else + * 0. This is the single call a caller needs to detect that an index went + * wrong somewhere; see guard.h for what it catches that AddressSanitizer + * cannot. */ +int v4_node_guards_intact(const v4_node *n); + +/* Read and write one word. D-1: word-addressed, so `addr` counts words and + * CELLS is a no-op. `addr` is a v4_cell because A, B and P are all v4_cells + * and a narrower parameter would put a conversion between the register and the + * access on every one of the eight memory opcodes. + * + * PRECONDITION: 0 <= addr < V4_NODE_WORDS. + * + * Out-of-range addressing is not defined by DECOMPOSITION.md. D-1 fixes word + * addressing and says nothing about an address outside the node's memory; a + * F18 has no such case because its address space is the memory. Whether a + * conforming model should wrap, saturate, or fault is a real open question and + * is raised as a gap rather than decided here, because each answer changes the + * ISA's observable behaviour and inventing one silently would bake a guess into + * every later test. + * + * WHAT THE BOUNDARY HERE DOES AND DOES NOT COVER, since it is easy to + * overclaim and the distinction decides whether out-of-range addresses are + * actually protected: + * + * - It DOES cover a linear index error. An access to mem[-1] or + * mem[V4_NODE_WORDS] lands in the 5% band, because the band is immediately + * adjacent (tests/test_node.c asserts the adjacency with offsetof rather + * than assuming it). That access is inside the struct, so AddressSanitizer + * stays silent about it and v4_node_guards_intact() is what reports it. + * + * - It does NOT cover an out-of-range *word address*. A v4_cell address of + * -1 casts to unsigned 0xFFFFFFFF, which is 16 GB past mem at 32-bit cells + * -- far outside the struct and outside the band. So a P, A or B that runs + * off the end of memory is caught by neither the band nor, in a production + * build without sanitizers, by anything else here. It is a precondition + * violation and the range check that would catch it is the open question + * above, not the band. + * + * The band is worth its 10.2% for the first case regardless. The second case + * is a hole, and it is closed by ruling on out-of-range addressing, not by + * widening the band. */ +v4_cell v4_node_load(const v4_node *n, v4_cell addr); +void v4_node_store(v4_node *n, v4_cell addr, v4_cell value); + +#endif /* V4_NODE_H */ diff --git a/v4/include/v4/opcode.h b/v4/include/v4/opcode.h new file mode 100644 index 00000000..cbe67810 --- /dev/null +++ b/v4/include/v4/opcode.h @@ -0,0 +1,64 @@ +/* opcode.h -- the 32 core opcodes. + * + * DECOMPOSITION.md section 1.3. "Opcode numbering follows the F18. Two names + * differ from Moore's: F18 `-` is renamed `inv` and F18 `or` (which is + * exclusive-or) is renamed `xor`, so instruction names never collide with + * FORTH-79 word names." + * + * Those two renames are load-bearing, not cosmetic: FORTH-79 has words named + * `-` and `OR`, and the DECOMPOSITION table would otherwise have two distinct + * meanings per name. The renames are therefore asserted against the FORTH + * name set in tests/test_iword.c, so a future edit that reintroduces a + * collision fails a test rather than silently producing an ambiguous table. + * + * Branch opcodes are marked because the instruction-word format depends on + * them: only these five take a target out of the low bits of their slot, and + * only they are legal outside the low four slots (section 1.2). + */ +#ifndef V4_OPCODE_H +#define V4_OPCODE_H + +typedef enum { + V4_OP_SEMI = 0x00, /* ; return */ + V4_OP_EX = 0x01, /* ex swap P and R */ + V4_OP_JUMP = 0x02, /* jump a */ + V4_OP_CALL = 0x03, /* call a */ + V4_OP_UNEXT = 0x04, /* unext */ + V4_OP_NEXT = 0x05, /* next a */ + V4_OP_IF = 0x06, /* if a */ + V4_OP_MINUS_IF = 0x07, /* -if a */ + V4_OP_FETCH_P = 0x08, /* @p literal */ + V4_OP_FETCH_INC= 0x09, /* @+ */ + V4_OP_FETCH_B = 0x0A, /* @b */ + V4_OP_FETCH_A = 0x0B, /* @ */ + V4_OP_STORE_P = 0x0C, /* !p */ + V4_OP_STORE_INC= 0x0D, /* !+ */ + V4_OP_STORE_B = 0x0E, /* !b */ + V4_OP_STORE_A = 0x0F, /* ! */ + V4_OP_MUL_STEP = 0x10, /* +* multiply step, D-3 */ + V4_OP_TWO_STAR = 0x11, /* 2* */ + V4_OP_TWO_SLASH= 0x12, /* 2/ */ + V4_OP_INV = 0x13, /* inv (F18 `-`) */ + V4_OP_ADD = 0x14, /* + */ + V4_OP_AND = 0x15, /* and */ + V4_OP_XOR = 0x16, /* xor (F18 `or`) */ + V4_OP_DROP = 0x17, /* drop */ + V4_OP_DUP = 0x18, /* dup */ + V4_OP_RPOP = 0x19, /* pop R to data stack */ + V4_OP_OVER = 0x1A, /* over */ + V4_OP_PUSH_A = 0x1B, /* a */ + V4_OP_NOP = 0x1C, /* nop */ + V4_OP_PUSH = 0x1D, /* push T to return stack */ + V4_OP_BANG_B = 0x1E, /* b! B <- T */ + V4_OP_BANG_A = 0x1F, /* a! A <- T */ + + V4_OPCODE_COUNT = 0x20 +} v4_opcode; + +/* True for the five opcodes that take a branch target from the low bits of + * their slot: jump, call, next, if, -if. Everything else ignores the low + * bits of its slot, which is what makes a non-branch slot's low bits + * free for the compiler to use as a compressed argument. */ +int v4_op_is_branch(unsigned op); + +#endif /* V4_OPCODE_H */ diff --git a/v4/include/v4/stack.h b/v4/include/v4/stack.h new file mode 100644 index 00000000..556e7143 --- /dev/null +++ b/v4/include/v4/stack.h @@ -0,0 +1,89 @@ +/* stack.h -- F18 circular hardware stacks (DECOMPOSITION.md D-2). + * + * D-2 rules these explicitly: "F18 circular stacks, hidden. Data stack 10 + * deep (T, S + 8 circular), return stack 9 deep (R + 8 circular), exactly as + * the F18. No stack pointer is visible to code, so DEPTH, PICK, ROLL, .S, + * SP@ and SP! are retired everywhere, host node included." + * + * Two consequences that the ISA and the capsule code must both respect: + * + * 1. There is no overflow or underflow detection, and there never will be. + * Pushing past the bottom silently overwrites the oldest entry. This is + * hardware behaviour, reproduced faithfully here rather than defended + * against: a golden model that trapped on overflow would diverge from + * the thing it is a model of, which is the whole point of the exercise. + * + * 2. Nesting is therefore shallow (10 and 9), and exceeding either depth + * corrupts silently rather than faulting. DECOMPOSITION.md section 3 + * calls this out as the constraint that "every CAP definition in section + * 4 and 5 must be re-checked against ... before it is accepted as a POST + * target; the hand traces so far assumed unbounded stacks." + * + * The T/S + 8-ring decomposition is the hardware shape, not an optimisation: + * T and S are architectural registers (section 1.1 lists them), and the ring + * is the circular part behind them. It is provably equivalent to a flat + * 10-deep circular buffer, which is what tests/test_stack.c checks it + * against -- see the equivalence argument in stack.c. + */ +#ifndef V4_STACK_H +#define V4_STACK_H + +#include "v4/cell.h" +#include "v4/guard.h" + +/* D-2: data stack is T, S + 8 circular = 10 deep. */ +#define V4_DATA_RING 8 +#define V4_DATA_DEPTH (V4_DATA_RING + 2) + +/* D-2: return stack is R + 8 circular = 9 deep. */ +#define V4_RET_RING 8 +#define V4_RET_DEPTH (V4_RET_RING + 1) + +/* 5% safety boundary at each end of each ring, per guard.h. The bands sit + * immediately either side of `ring` -- a boundary separated from the payload + * by another field catches nothing -- and are storage the struct owns, because + * a mesh node has no heap. Widths are 1 element each at the current ring + * sizes; V4_GUARD_ELEMS scales them if a ring size ever changes. */ +#define V4_DATA_BOUND V4_GUARD_ELEMS(V4_DATA_RING) +#define V4_RET_BOUND V4_GUARD_ELEMS(V4_RET_RING) + +typedef struct { + v4_cell t; /* T -- top of data stack (a register) */ + v4_cell s; /* S -- second (a register) */ + unsigned head; + + v4_cell guard_head[V4_DATA_BOUND]; + v4_cell ring[V4_DATA_RING]; + v4_cell guard_tail[V4_DATA_BOUND]; +} v4_dstack; + +typedef struct { + v4_cell r; /* R -- top of return stack (a register) */ + unsigned head; + + v4_cell guard_head[V4_RET_BOUND]; + v4_cell ring[V4_RET_RING]; + v4_cell guard_tail[V4_RET_BOUND]; +} v4_rstack; + +void v4_dstack_reset(v4_dstack *st); +void v4_dstack_push(v4_dstack *st, v4_cell x); +v4_cell v4_dstack_pop(v4_dstack *st); + +/* Read the top without disturbing the stack. T is a register on hardware, so + * this is a plain read and needs no ring access. */ +v4_cell v4_dstack_peek(const v4_dstack *st); +/* Read the second element without disturbing the stack. */ +v4_cell v4_dstack_peek2(const v4_dstack *st); + +void v4_rstack_reset(v4_rstack *st); +void v4_rstack_push(v4_rstack *st, v4_cell x); +v4_cell v4_rstack_pop(v4_rstack *st); +v4_cell v4_rstack_peek(const v4_rstack *st); + +/* 1 if both boundaries of the data stack are intact, else 0. See guard.h for + * what this catches that AddressSanitizer does not. */ +int v4_dstack_guards_intact(const v4_dstack *st); +int v4_rstack_guards_intact(const v4_rstack *st); + +#endif /* V4_STACK_H */ diff --git a/v4/include/v4/testcode.h b/v4/include/v4/testcode.h new file mode 100644 index 00000000..1a94a330 --- /dev/null +++ b/v4/include/v4/testcode.h @@ -0,0 +1,19 @@ +/* testcode.h -- run a word on a node, for tests. */ +#ifndef V4_TESTCODE_H +#define V4_TESTCODE_H + +#include "v4/exec.h" + +/* Where a word called by v4_test_call returns to. Nothing is executed there: + * arriving is the stop condition. */ +#define V4_TEST_HALT ((v4_cell)(V4_NODE_WORDS - 1u)) + +/* Call the word at `entry` as a caller would: push a return address, set P, + * and step instruction words until the word returns. The stacks and A, B are + * left as the caller set them, so arguments go on the data stack first. + * Returns the number of instruction words stepped, or -1 if the word had not + * returned after `max_words`. */ +long v4_test_call(v4_node *n, v4_exec_state *es, v4_heat *h, + v4_cell entry, long max_words); + +#endif /* V4_TESTCODE_H */ diff --git a/v4/include/v4/umul.h b/v4/include/v4/umul.h new file mode 100644 index 00000000..38d8f5f0 --- /dev/null +++ b/v4/include/v4/umul.h @@ -0,0 +1,31 @@ +/* umul.h -- unsigned multiply fixing UM* under D-3 constraints. + * + * DECOMPOSITION.md line 205-207: "NEEDS REVISION: written for the carry-keeping + * +* that D-3 rejected. With plain F18 +* the carry out of T is lost, so this + * is only correct when the partial sums never overflow T (operands below 2^31). + * A full-range UM* needs a correction step; to be rewritten and proven on the + * golden model." + * + * D-3 states: "Plain F18 semantics. The carry out of T is not kept." + * The +* defined in the F18 shifts T:A right; the add is T <- T+S when bit 0 + * of A was set, but any carry out of that addition is discarded (not shifted + * into the high part). That means UM* as written only works for inputs whose + * partial products don't generate a carry beyond T's width during the loop. + * + * We need a correct full-range unsigned multiply (u1*u2 -> ulo, uhi) that is + * correct for all cell-width values of u1,u2, expressed in terms of the v4 + * primitives or at least specified precisely for the golden model. Since + * we're building the golden model in C first, the canonical implementation is + * the width-correct C multiplication split into high/low halves at V4_CELL_BITS. + * This header provides both: a reference C implementation (for verification) + * and a documented form that matches D-3's +* constraint. + */ +#ifndef V4_UMUL_H +#define V4_UMUL_H + +#include "v4/cell.h" + +/* Unsigned multiply: (u1, u2) -> (ulo, uhi). Full range, width-correct. */ +void v4_umul(v4_ucell u1, v4_ucell u2, v4_ucell *ulo, v4_ucell *uhi); + +#endif /* V4_UMUL_H */ diff --git a/v4/src/asm.c b/v4/src/asm.c new file mode 100644 index 00000000..0a2c7378 --- /dev/null +++ b/v4/src/asm.c @@ -0,0 +1,115 @@ +/* asm.c -- slot packer for test programs. See asm.h. */ +#include "v4/asm.h" + +static void put(v4_asm *as, unsigned op) +{ + as->cur |= (v4_iword)(op & 0x1Fu) << V4_SLOT_LOW_BIT(as->slot); + as->slot++; +} + +static void emit(v4_asm *as, v4_cell value) +{ + if (as->here < 0 || (v4_ucell)as->here >= V4_NODE_WORDS) { + as->error = 1; + return; + } + v4_node_store(as->n, as->here, value); + as->here++; +} + +/* Write out the word being built, then the literals it owes. */ +static void flush(v4_asm *as) +{ + if (as->slot == 0) return; + while (as->slot < V4_SLOT_COUNT) put(as, V4_OP_NOP); + emit(as, (v4_cell)as->cur); + for (unsigned i = 0; i < as->nlit; i++) emit(as, as->lit[i]); + as->cur = 0; + as->slot = 0; + as->nlit = 0; +} + +/* 1 if a branch in `slot`, executing with P = p, can reach `target`. */ +static int reachable(v4_cell p, unsigned slot, v4_cell target) +{ + v4_ucell mask = (v4_ucell)v4_iword_slot_mask(slot); + return (((v4_ucell)p ^ (v4_ucell)target) & ~mask) == 0; +} + +void v4_asm_begin(v4_asm *as, v4_node *n, v4_cell origin) +{ + as->n = n; + as->here = origin; + as->cur = 0; + as->slot = 0; + as->nlit = 0; + as->error = 0; +} + +void v4_asm_op(v4_asm *as, unsigned op) +{ + if (op >= V4_OPCODE_COUNT || v4_op_is_branch(op) || op == V4_OP_FETCH_P) { + as->error = 1; + return; + } + put(as, op); + if (op == V4_OP_SEMI || op == V4_OP_EX || as->slot == V4_SLOT_COUNT) flush(as); +} + +void v4_asm_lit(v4_asm *as, v4_cell value) +{ + put(as, V4_OP_FETCH_P); + as->lit[as->nlit++] = value; + if (as->slot == V4_SLOT_COUNT) flush(as); +} + +v4_asm_ref v4_asm_branch_fwd(v4_asm *as, unsigned op) +{ + v4_asm_ref ref; + + if (!v4_op_is_branch(op)) as->error = 1; + if (!v4_iword_branch_legal(as->slot)) flush(as); + + ref.addr = as->here; + ref.slot = as->slot; + /* The word has been fetched and every literal before the branch consumed + * by the time the branch executes. */ + ref.p = as->here + 1 + (v4_cell)as->nlit; + + put(as, op); + as->slot = V4_SLOT_COUNT; /* the rest of the word is address */ + flush(as); + return ref; +} + +void v4_asm_resolve(v4_asm *as, v4_asm_ref ref, v4_cell target) +{ + v4_ucell mask = (v4_ucell)v4_iword_slot_mask(ref.slot); + v4_ucell word; + + if (as->error) return; + if (!reachable(ref.p, ref.slot, target)) { + as->error = 1; + return; + } + word = (v4_ucell)v4_node_load(as->n, ref.addr); + word = (word & ~mask) | ((v4_ucell)target & mask); + v4_node_store(as->n, ref.addr, (v4_cell)word); +} + +void v4_asm_branch(v4_asm *as, unsigned op, v4_cell target) +{ + v4_asm_resolve(as, v4_asm_branch_fwd(as, op), target); +} + +v4_cell v4_asm_label(v4_asm *as) +{ + flush(as); + return as->here; +} + +int v4_asm_ok(v4_asm *as) +{ + flush(as); + return !as->error; +} diff --git a/v4/src/exec.c b/v4/src/exec.c new file mode 100644 index 00000000..8c634d50 --- /dev/null +++ b/v4/src/exec.c @@ -0,0 +1,209 @@ +/* exec.c -- instruction execution. See exec.h. + * + * All cell arithmetic is done in v4_ucell and converted back. Signed overflow + * is undefined behaviour in C99 and a right shift of a negative value is + * implementation-defined, and a golden model cannot have either: the hardware + * wraps, so the model wraps, on every host and at every optimisation level. + */ +#include "v4/exec.h" + +static v4_cell add_wrap(v4_cell x, v4_cell y) +{ + return (v4_cell)((v4_ucell)x + (v4_ucell)y); +} + +/* Arithmetic shift right by one: the sign bit is kept. */ +static v4_cell asr1(v4_cell x) +{ + v4_ucell u = (v4_ucell)x; + return (v4_cell)((u >> 1) | (u & V4_MSB)); +} + +/* An instruction word occupies the low 32 bits of a cell at every cell width + * (DECOMPOSITION.md D-9). */ +static v4_iword fetch_iword(const v4_node *n, v4_cell addr) +{ + return (v4_iword)((v4_ucell)v4_node_load(n, addr) & 0xFFFFFFFFu); +} + +/* Opcodes after which no further slot of the word executes. The five branch + * opcodes end it because everything to their right is address, taken or not. + * `;` and `ex` end it because P has been replaced, as on the F18. `unext` is + * not in this set: when its loop finishes it falls through to the next slot. */ +static int ends_word(unsigned op) +{ + return v4_op_is_branch(op) || op == V4_OP_SEMI || op == V4_OP_EX; +} + +void v4_exec_reset(v4_exec_state *es) +{ + es->anticlock = 0; +} + +void v4_exec_op(v4_node *n, v4_exec_state *es, v4_heat *h, + unsigned op, v4_iword w, unsigned slot) +{ + v4_dstack *ds = &n->ds; + v4_rstack *rs = &n->rs; + v4_cell x; + + switch (op) { + case V4_OP_SEMI: + n->p = v4_rstack_pop(rs); + break; + case V4_OP_EX: + x = rs->r; + rs->r = n->p; + n->p = x; + break; + case V4_OP_JUMP: + n->p = v4_iword_branch(n->p, w, slot); + break; + case V4_OP_CALL: + x = v4_iword_branch(n->p, w, slot); + v4_rstack_push(rs, n->p); + v4_heat_on_call(h, x); + n->p = x; + break; + case V4_OP_UNEXT: + /* The restart at slot 0 is v4_exec_step_word's business; this is only + * the effect on R. */ + if (rs->r != 0) rs->r = add_wrap(rs->r, -1); + else (void)v4_rstack_pop(rs); + break; + case V4_OP_NEXT: + if (rs->r != 0) { + rs->r = add_wrap(rs->r, -1); + n->p = v4_iword_branch(n->p, w, slot); + } else { + (void)v4_rstack_pop(rs); + } + break; + case V4_OP_IF: /* does not pop T */ + if (ds->t == 0) n->p = v4_iword_branch(n->p, w, slot); + break; + case V4_OP_MINUS_IF: /* does not pop T */ + if (((v4_ucell)ds->t & V4_MSB) == 0) n->p = v4_iword_branch(n->p, w, slot); + break; + + case V4_OP_FETCH_P: + v4_dstack_push(ds, v4_node_load(n, n->p)); + n->p = add_wrap(n->p, 1); + break; + case V4_OP_FETCH_INC: + v4_dstack_push(ds, v4_node_load(n, n->a)); + n->a = add_wrap(n->a, 1); + break; + case V4_OP_FETCH_B: + v4_dstack_push(ds, v4_node_load(n, n->b)); + break; + case V4_OP_FETCH_A: + v4_dstack_push(ds, v4_node_load(n, n->a)); + break; + case V4_OP_STORE_P: + v4_node_store(n, n->p, v4_dstack_pop(ds)); + n->p = add_wrap(n->p, 1); + break; + case V4_OP_STORE_INC: + v4_node_store(n, n->a, v4_dstack_pop(ds)); + n->a = add_wrap(n->a, 1); + break; + case V4_OP_STORE_B: + v4_node_store(n, n->b, v4_dstack_pop(ds)); + break; + case V4_OP_STORE_A: + v4_node_store(n, n->a, v4_dstack_pop(ds)); + break; + + case V4_OP_MUL_STEP: + /* D-3, plain F18: the carry out of T+S is not kept. T:A then shifts + * right one bit, T keeping its sign bit and T's bit 0 entering the top + * of A. S is not consumed. */ + x = ds->t; + if ((v4_ucell)n->a & 1u) x = add_wrap(x, ds->s); + n->a = (v4_cell)(((v4_ucell)n->a >> 1) + | (((v4_ucell)x & 1u) << (V4_CELL_BITS - 1))); + ds->t = asr1(x); + break; + case V4_OP_TWO_STAR: + ds->t = (v4_cell)((v4_ucell)ds->t << 1); + break; + case V4_OP_TWO_SLASH: + ds->t = asr1(ds->t); + break; + case V4_OP_INV: + ds->t = (v4_cell)~(v4_ucell)ds->t; + break; + case V4_OP_ADD: + x = v4_dstack_pop(ds); /* old T; T is now old S */ + ds->t = add_wrap(ds->t, x); + break; + case V4_OP_AND: + x = v4_dstack_pop(ds); + ds->t = (v4_cell)((v4_ucell)ds->t & (v4_ucell)x); + break; + case V4_OP_XOR: + x = v4_dstack_pop(ds); + ds->t = (v4_cell)((v4_ucell)ds->t ^ (v4_ucell)x); + break; + case V4_OP_DROP: + (void)v4_dstack_pop(ds); + break; + case V4_OP_DUP: + v4_dstack_push(ds, ds->t); + break; + case V4_OP_RPOP: + v4_dstack_push(ds, v4_rstack_pop(rs)); + break; + case V4_OP_OVER: + v4_dstack_push(ds, ds->s); + break; + case V4_OP_PUSH_A: + v4_dstack_push(ds, n->a); + break; + case V4_OP_NOP: + break; + case V4_OP_PUSH: + v4_rstack_push(rs, v4_dstack_pop(ds)); + break; + case V4_OP_BANG_B: + n->b = v4_dstack_pop(ds); + break; + case V4_OP_BANG_A: + n->a = v4_dstack_pop(ds); + break; + default: + return; /* not an opcode: nothing retires */ + } + + v4_heat_on_retire(h, op, &es->anticlock); +} + +unsigned v4_exec_step_word(v4_node *n, v4_exec_state *es, v4_heat *h) +{ + v4_iword iw = fetch_iword(n, n->p); + unsigned executed = 0; + unsigned slot = 0; + + n->p = add_wrap(n->p, 1); + + while (slot < V4_SLOT_COUNT) { + unsigned op = v4_iword_op(iw, slot); + /* Decided before the opcode runs, because running it changes R. */ + int again = (op == V4_OP_UNEXT) && (n->rs.r != 0); + + v4_exec_op(n, es, h, op, iw, slot); + executed++; + + if (again) { + /* Same word, slot 0, no refetch and P untouched: an `@p` or `!p` + * inside a unext loop therefore walks the words that follow, as + * it does on the F18. */ + slot = 0; + continue; + } + if (ends_word(op)) break; + slot++; + } + return executed; +} diff --git a/v4/src/guard.c b/v4/src/guard.c new file mode 100644 index 00000000..c0bad815 --- /dev/null +++ b/v4/src/guard.c @@ -0,0 +1,15 @@ +/* guard.c -- 5% safety boundaries. See guard.h. */ +#include "v4/guard.h" + +void v4_guard_fill(v4_cell *band, unsigned n, v4_ucell pattern) +{ + for (unsigned i = 0; i < n; i++) band[i] = (v4_cell)pattern; +} + +int v4_guard_intact(const v4_cell *band, unsigned n, v4_ucell pattern) +{ + for (unsigned i = 0; i < n; i++) { + if ((v4_ucell)band[i] != pattern) return 0; + } + return 1; +} diff --git a/v4/src/heat.c b/v4/src/heat.c new file mode 100644 index 00000000..8d80b1ba --- /dev/null +++ b/v4/src/heat.c @@ -0,0 +1,89 @@ +/* heat.c -- heat and anti-clock. See heat.h. */ +#include "v4/heat.h" +#include + +void v4_heat_reset(v4_heat *h) +{ + memset(h->op, 0, sizeof(h->op)); + memset(h->call, 0, sizeof(h->call)); + memset(h->call_freeze_mask, 0, sizeof(h->call_freeze_mask)); +} + +void v4_heat_on_retire(v4_heat *h, unsigned op, v4_uheat_t *anticlock) +{ + if (op < V4_OPCODE_COUNT) { + h->op[op]++; + } + if (anticlock) (*anticlock)++; +} + +static unsigned call_index(v4_cell target) +{ + /* Index by target address. The table is a fixed-size cache for the model. + * A real hardware table is deferred; here we hash-mod or just take low bits? + * Or simply if in range use it. The spec says "per-call-target heat table" + * indexed by target. For golden model, targets are small (node memory). */ + unsigned idx = (unsigned)(v4_ucell)target; + if (idx >= V4_HEAT_MAX_CALL_TARGETS) { + /* Out of range: ignore for now? Or clamp? D-6 deferred; but heat must + * be recorded for actual call targets exercised. Many tests will have + * targets < 4096. This is a conservative model choice: if target is + * outside the modeled table, we do not increment a modeled slot (and + * we could log, but not specified). The boundary philosophy says detect + * rather than corrupt: for the model, we just don't record in the fixed + * table -- acceptable until D-4/D-6 pin sizes. */ + return (unsigned)-1; /* invalid */ + } + return idx; +} + +void v4_heat_on_call(v4_heat *h, v4_cell target) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return; + if (v4_heat_call_is_frozen(h, target)) return; + h->call[idx]++; +} + +v4_uheat_t v4_heat_call_get(const v4_heat *h, v4_cell target) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return 0; + return h->call[idx]; +} + +int v4_heat_call_set(v4_heat *h, v4_cell target, v4_uheat_t val) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return 0; + if (v4_heat_call_is_frozen(h, target)) return 0; + h->call[idx] = val; + return 1; +} + +void v4_heat_call_freeze(v4_heat *h, v4_cell target) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return; + unsigned word = idx / (8u * sizeof(v4_ucell)); + unsigned bit = idx % (8u * sizeof(v4_ucell)); + h->call_freeze_mask[word] |= ((v4_ucell)1u << bit); +} + +void v4_heat_call_unfreeze(v4_heat *h, v4_cell target) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return; + unsigned word = idx / (8u * sizeof(v4_ucell)); + unsigned bit = idx % (8u * sizeof(v4_ucell)); + h->call_freeze_mask[word] &= ~((v4_ucell)1u << bit); +} + +int v4_heat_call_is_frozen(const v4_heat *h, v4_cell target) +{ + unsigned idx = call_index(target); + if (idx == (unsigned)-1) return 0; + unsigned word = idx / (8u * sizeof(v4_ucell)); + unsigned bit = idx % (8u * sizeof(v4_ucell)); + return (h->call_freeze_mask[word] & ((v4_ucell)1u << bit)) != 0; +} diff --git a/v4/src/iword.c b/v4/src/iword.c new file mode 100644 index 00000000..5f2b0a72 --- /dev/null +++ b/v4/src/iword.c @@ -0,0 +1,74 @@ +/* iword.c -- instruction-word decode. See iword.h for the layout. */ +#include "v4/iword.h" + +/* The five opcodes that consume a target from the low bits of their slot + * (section 1.2): jump, call, next, if, -if. A table rather than a switch so + * that the set is visible in one place and can be counted in the tests. */ +int v4_op_is_branch(unsigned op) +{ + return op == V4_OP_JUMP || op == V4_OP_CALL || op == V4_OP_NEXT + || op == V4_OP_IF || op == V4_OP_MINUS_IF; +} + +int v4_iword_branch_legal(unsigned slot) +{ + return slot < 4u; /* section 1.2: branches are legal in slots 0-3 only */ +} + +uint32_t v4_iword_slot_mask(unsigned slot) +{ + if (slot >= V4_SLOT_COUNT) return 0u; + /* 27-5k address bits, e.g. 0x07FFFFFF for slot 0 down to 0x00000FFF for + * slot 3. For slots 4 and 5 the formula still produces 7 and 2 bits; the + * legality check is separate because those slots are simply not usable. */ + return ((uint32_t)1u << V4_BRANCH_ADDR_BITS(slot)) - 1u; +} + +unsigned v4_iword_op(v4_iword w, unsigned slot) +{ + if (slot >= V4_SLOT_COUNT) return 0u; + return (unsigned)((w >> V4_SLOT_LOW_BIT(slot)) & 0x1Fu); +} + +unsigned v4_iword_spare(v4_iword w) +{ + return (unsigned)(w & ((1u << V4_SPARE_BITS) - 1u)); +} + +v4_iword v4_iword_target(v4_iword w, unsigned slot) +{ + if (slot >= V4_SLOT_COUNT) return 0u; + return w & v4_iword_slot_mask(slot); +} + +v4_iword v4_iword_assemble(const unsigned op[V4_SLOT_COUNT], unsigned spare) +{ + v4_iword w = 0u; + + for (unsigned k = 0; k < V4_SLOT_COUNT; k++) { + /* An opcode that does not fit in 5 bits is a bug in whoever built the + * word, not something to encode silently. Truncating here would let a + * malformed capsule execute as something else. */ + if (op[k] > 0x1Fu) return 0xFFFFFFFFu; + w |= (v4_iword)(op[k] & 0x1Fu) << V4_SLOT_LOW_BIT(k); + } + if (spare > ((1u << V4_SPARE_BITS) - 1u)) return 0xFFFFFFFFu; + w |= (v4_iword)(spare & ((1u << V4_SPARE_BITS) - 1u)); + + return w; +} + +v4_cell v4_iword_branch(v4_cell p, v4_iword w, unsigned slot) +{ + uint32_t mask = v4_iword_slot_mask(slot); + v4_ucell up = (v4_ucell)p; + + /* Page-relative, as in the F18: the target supplies the low + * V4_BRANCH_ADDR_BITS(slot) bits of P and the bits above them survive. The + * mask is computed in 32 bits and widened, so a 64-bit P keeps its high + * half -- which is the correct reading of "replaces the same low bits of P" + * at any cell width. */ + up = (up & ~(v4_ucell)mask) | (v4_ucell)(w & mask); + + return (v4_cell)up; +} diff --git a/v4/src/node.c b/v4/src/node.c new file mode 100644 index 00000000..d654f181 --- /dev/null +++ b/v4/src/node.c @@ -0,0 +1,35 @@ +/* node.c -- the v4 node. See node.h. */ +#include "v4/node.h" + +void v4_node_reset(v4_node *n) +{ + n->p = 0; + n->a = 0; + n->b = 0; + for (unsigned i = 0; i < V4_NODE_WORDS; i++) n->mem[i] = 0; + v4_guard_fill(n->mem_guard_head, V4_MEM_BOUND, V4_GUARD_PATTERN_HEAD); + v4_guard_fill(n->mem_guard_tail, V4_MEM_BOUND, V4_GUARD_PATTERN_TAIL); + v4_dstack_reset(&n->ds); + v4_rstack_reset(&n->rs); +} + +int v4_node_guards_intact(const v4_node *n) +{ + return v4_guard_intact(n->mem_guard_head, V4_MEM_BOUND, V4_GUARD_PATTERN_HEAD) + && v4_guard_intact(n->mem_guard_tail, V4_MEM_BOUND, V4_GUARD_PATTERN_TAIL) + && v4_dstack_guards_intact(&n->ds) + && v4_rstack_guards_intact(&n->rs); +} + +v4_cell v4_node_load(const v4_node *n, v4_cell addr) +{ + /* Precondition checked by the caller; see node.h. The unsigned compare is + * deliberate: `addr` is signed, and a negative address must not wrap into + * a large positive one and read the top of memory instead. */ + return n->mem[(unsigned)addr]; +} + +void v4_node_store(v4_node *n, v4_cell addr, v4_cell value) +{ + n->mem[(unsigned)addr] = value; +} diff --git a/v4/src/stack.c b/v4/src/stack.c new file mode 100644 index 00000000..e00c9f0e --- /dev/null +++ b/v4/src/stack.c @@ -0,0 +1,107 @@ +/* stack.c -- F18 circular hardware stacks. + * + * The two decompositions (T/S + 8-ring, R + 8-ring) are exactly equivalent to + * flat circular buffers of depth 10 and 9 respectively. The argument, since + * the ISA is specified in terms of the decomposition and the differential + * test in tests/test_stack.c is written against the flat form: + * + * Data stack, 10 deep. Read the stack top to bottom as + * + * T, S, ring[(head-1) mod 8], ring[(head-2) mod 8], ..., ring[head mod 8] + * + * -- that is, ring[(head-1) mod 8] sits immediately below S, and head names + * the oldest slot. push(x) does ring[head]=S; S=T; T=x; head=(head+1) mod 8, + * which appends at the top and rotates the oldest entry into the ring. + * pop() takes T, promotes S, and pulls ring[(head-1) mod 8] up into S while + * stepping head back one -- so the element that was third becomes second and + * the ring is re-seated one position earlier. That is a 10-slot circular + * buffer whose slot 0 is T and slot 1 is S, and the head arithmetic is + * bookkeeping for which ring slot is oldest. + * + * Return stack, 9 deep, same argument with one register instead of two: the + * stack reads R, ring[(head-1) mod 8], ..., ring[head mod 8]. + * + * "No overflow or underflow; pushing past the bottom silently overwrites the + * oldest entry" (D-2) is not an extra rule here -- it is what a fixed-depth + * circular buffer with no bounds check does on its own, once the pointer is + * allowed to wrap. Nothing below tests head against anything. + */ +#include "v4/stack.h" + +void v4_dstack_reset(v4_dstack *st) +{ + st->t = 0; + st->s = 0; + st->head = 0; + for (unsigned i = 0; i < V4_DATA_RING; i++) st->ring[i] = 0; + v4_guard_fill(st->guard_head, V4_DATA_BOUND, V4_GUARD_PATTERN_HEAD); + v4_guard_fill(st->guard_tail, V4_DATA_BOUND, V4_GUARD_PATTERN_TAIL); +} + +int v4_dstack_guards_intact(const v4_dstack *st) +{ + return v4_guard_intact(st->guard_head, V4_DATA_BOUND, V4_GUARD_PATTERN_HEAD) + && v4_guard_intact(st->guard_tail, V4_DATA_BOUND, V4_GUARD_PATTERN_TAIL); +} + +void v4_dstack_push(v4_dstack *st, v4_cell x) +{ + st->ring[st->head] = st->s; + st->s = st->t; + st->t = x; + st->head = (st->head + 1u) % V4_DATA_RING; +} + +v4_cell v4_dstack_pop(v4_dstack *st) +{ + v4_cell x = st->t; + st->t = st->s; + st->head = (st->head + V4_DATA_RING - 1u) % V4_DATA_RING; + st->s = st->ring[st->head]; + return x; +} + +v4_cell v4_dstack_peek(const v4_dstack *st) +{ + return st->t; +} + +v4_cell v4_dstack_peek2(const v4_dstack *st) +{ + return st->s; +} + +void v4_rstack_reset(v4_rstack *st) +{ + st->r = 0; + st->head = 0; + for (unsigned i = 0; i < V4_RET_RING; i++) st->ring[i] = 0; + v4_guard_fill(st->guard_head, V4_RET_BOUND, V4_GUARD_PATTERN_HEAD); + v4_guard_fill(st->guard_tail, V4_RET_BOUND, V4_GUARD_PATTERN_TAIL); +} + +int v4_rstack_guards_intact(const v4_rstack *st) +{ + return v4_guard_intact(st->guard_head, V4_RET_BOUND, V4_GUARD_PATTERN_HEAD) + && v4_guard_intact(st->guard_tail, V4_RET_BOUND, V4_GUARD_PATTERN_TAIL); +} + +void v4_rstack_push(v4_rstack *st, v4_cell x) +{ + st->ring[st->head] = st->r; + st->r = x; + st->head = (st->head + 1u) % V4_RET_RING; +} + +v4_cell v4_rstack_pop(v4_rstack *st) +{ + v4_cell x = st->r; + st->head = (st->head + V4_RET_RING - 1u) % V4_RET_RING; + st->r = st->ring[st->head]; + return x; +} + +v4_cell v4_rstack_peek(const v4_rstack *st) +{ + return st->r; +} diff --git a/v4/src/testcode.c b/v4/src/testcode.c new file mode 100644 index 00000000..90bda8f7 --- /dev/null +++ b/v4/src/testcode.c @@ -0,0 +1,20 @@ +/* testcode.c -- run a word on a node, for tests. See testcode.h. */ +#include "v4/testcode.h" + +long v4_test_call(v4_node *n, v4_exec_state *es, v4_heat *h, + v4_cell entry, long max_words) +{ + long words = 0; + + v4_rstack_push(&n->rs, V4_TEST_HALT); + n->p = entry; + while (n->p != V4_TEST_HALT) { + if (words >= max_words) return -1; + /* A golden model must not index outside its own memory on a runaway + * P; node.h leaves out-of-range addressing undefined. */ + if (n->p < 0 || (v4_ucell)n->p >= V4_NODE_WORDS) return -1; + (void)v4_exec_step_word(n, es, h); + words++; + } + return words; +} diff --git a/v4/src/umul.c b/v4/src/umul.c new file mode 100644 index 00000000..7acef151 --- /dev/null +++ b/v4/src/umul.c @@ -0,0 +1,27 @@ +/* umul.c -- reference unsigned multiply. See umul.h. + * + * Schoolbook multiplication on half-cells, in v4_ucell only. No wider type is + * used, so the same code is correct at both cell widths and builds on a host + * with no 128-bit integer (ARMv7 is one, and is a v4 host). + */ +#include "v4/umul.h" + +#define HALF_BITS (V4_CELL_BITS / 2) +#define HALF_MASK ((((v4_ucell)1) << HALF_BITS) - 1u) + +void v4_umul(v4_ucell u1, v4_ucell u2, v4_ucell *ulo, v4_ucell *uhi) +{ + v4_ucell a0 = u1 & HALF_MASK, a1 = u1 >> HALF_BITS; + v4_ucell b0 = u2 & HALF_MASK, b1 = u2 >> HALF_BITS; + + v4_ucell p00 = a0 * b0; + v4_ucell p10 = a1 * b0; + v4_ucell p01 = a0 * b1; + v4_ucell p11 = a1 * b1; + + /* Cannot overflow: at most (2^h - 1) + (2^h - 1) + (2^h - 1)^2 = 2^2h - 1. */ + v4_ucell mid = (p00 >> HALF_BITS) + (p10 & HALF_MASK) + p01; + + if (ulo) *ulo = (mid << HALF_BITS) | (p00 & HALF_MASK); + if (uhi) *uhi = p11 + (p10 >> HALF_BITS) + (mid >> HALF_BITS); +} diff --git a/v4/tests/test_asm.c b/v4/tests/test_asm.c new file mode 100644 index 00000000..b4d847f3 --- /dev/null +++ b/v4/tests/test_asm.c @@ -0,0 +1,110 @@ +/* test_asm.c -- the slot packer lays words out as asm.h says it does. */ +#include "v4/asm.h" +#include "v4/testcode.h" +#include + +static int failures = 0, checks = 0; +#define CHECK(c,...) do{checks++; if(!(c)){failures++; printf("FAIL %s:%d: ",__FILE__,__LINE__); printf(__VA_ARGS__); printf("\n");}}while(0) + +static v4_node n; +static v4_exec_state es; +static v4_heat h; +static v4_asm as; + +static void fresh(v4_cell origin) +{ + v4_node_reset(&n); + v4_exec_reset(&es); + v4_heat_reset(&h); + v4_asm_begin(&as, &n, origin); +} + +static v4_iword word_at(v4_cell addr) { return (v4_iword)(v4_ucell)n.mem[addr]; } + +int main(void) +{ + v4_asm_ref ref; + v4_cell loop, other; + + printf("v4 asm tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + /* Six opcodes fill a word; the seventh starts the next, padded with nop. */ + fresh(0); + for (unsigned k = 0; k < 7; k++) v4_asm_op(&as, V4_OP_DUP); + CHECK(v4_asm_label(&as) == 2, "seven opcodes take two words"); + for (unsigned k = 0; k < 6; k++) CHECK(v4_iword_op(word_at(0), k) == V4_OP_DUP, "word 0 slot %u", k); + CHECK(v4_iword_op(word_at(1), 0) == V4_OP_DUP, "word 1 slot 0"); + for (unsigned k = 1; k < 6; k++) CHECK(v4_iword_op(word_at(1), k) == V4_OP_NOP, "word 1 pad %u", k); + + /* `;` closes the word. */ + fresh(0); + v4_asm_op(&as, V4_OP_DUP); v4_asm_op(&as, V4_OP_SEMI); + CHECK(v4_asm_label(&as) == 1, "; closes the word"); + + /* Literals follow the instruction word, in order, and the code runs. */ + fresh(0); + v4_asm_lit(&as, 30); v4_asm_lit(&as, 12); v4_asm_op(&as, V4_OP_ADD); v4_asm_op(&as, V4_OP_SEMI); + CHECK(v4_asm_ok(&as), "literal program assembles"); + CHECK(n.mem[1] == 30 && n.mem[2] == 12 && v4_asm_label(&as) == 3, "literal layout"); + CHECK(v4_test_call(&n, &es, &h, 0, 10) == 1 && n.ds.t == 42, "30 12 + ;"); + + /* A branch takes the next free slot when that slot is legal. */ + fresh(0); + v4_asm_op(&as, V4_OP_DUP); v4_asm_op(&as, V4_OP_DROP); + v4_asm_branch(&as, V4_OP_JUMP, 0x155); + CHECK(v4_iword_op(word_at(0), 2) == V4_OP_JUMP, "branch in slot 2"); + CHECK(v4_iword_target(word_at(0), 2) == 0x155, "branch target in slot 2"); + CHECK(v4_asm_label(&as) == 1, "a branch closes the word"); + + /* With slots 0-3 used, the branch moves to slot 0 of a new word. */ + fresh(0); + for (unsigned k = 0; k < 4; k++) v4_asm_op(&as, V4_OP_DUP); + v4_asm_branch(&as, V4_OP_JUMP, 0x155); + CHECK(v4_iword_op(word_at(0), 4) == V4_OP_NOP && v4_iword_op(word_at(0), 5) == V4_OP_NOP, + "word before the branch is padded"); + CHECK(v4_iword_op(word_at(1), 0) == V4_OP_JUMP && v4_iword_target(word_at(1), 0) == 0x155, + "branch moved to slot 0"); + + /* Forward reference, section 2's IF ... ELSE ... THEN with the flag left + * on the stack by `if`: if L1 drop 111 ; L1: drop 222 ; */ + fresh(0); + ref = v4_asm_branch_fwd(&as, V4_OP_IF); + v4_asm_op(&as, V4_OP_DROP); v4_asm_lit(&as, 111); v4_asm_op(&as, V4_OP_SEMI); + other = v4_asm_label(&as); + v4_asm_resolve(&as, ref, other); + v4_asm_op(&as, V4_OP_DROP); v4_asm_lit(&as, 222); v4_asm_op(&as, V4_OP_SEMI); + CHECK(v4_asm_ok(&as), "forward branch assembles"); + v4_dstack_push(&n.ds, 5); + CHECK(v4_test_call(&n, &es, &h, 0, 10) > 0 && n.ds.t == 111, "flag nonzero takes the first arm"); + v4_dstack_reset(&n.ds); v4_rstack_reset(&n.rs); + v4_dstack_push(&n.ds, 0); + CHECK(v4_test_call(&n, &es, &h, 0, 10) > 0 && n.ds.t == 222, "flag zero takes the second arm"); + + /* Backward reference: 4 FOR 1 + NEXT runs the body n+1 times. */ + fresh(0); + v4_asm_lit(&as, 4); v4_asm_op(&as, V4_OP_PUSH); + loop = v4_asm_label(&as); + v4_asm_lit(&as, 1); v4_asm_op(&as, V4_OP_ADD); + v4_asm_branch(&as, V4_OP_NEXT, loop); + v4_asm_op(&as, V4_OP_SEMI); + CHECK(v4_asm_ok(&as), "loop assembles"); + v4_dstack_push(&n.ds, 0); + CHECK(v4_test_call(&n, &es, &h, 0, 50) > 0 && n.ds.t == 5, "4 FOR 1 + NEXT"); + + /* Errors are reported, not encoded. */ + fresh(0); v4_asm_op(&as, V4_OP_JUMP); + CHECK(!v4_asm_ok(&as), "a branch through v4_asm_op is refused"); + fresh(0); v4_asm_op(&as, V4_OPCODE_COUNT); + CHECK(!v4_asm_ok(&as), "an out-of-range opcode is refused"); + fresh(0); + for (unsigned k = 0; k < 3; k++) v4_asm_op(&as, V4_OP_DUP); + v4_asm_branch(&as, V4_OP_JUMP, 5000); /* slot 3 reaches 4096 words */ + CHECK(!v4_asm_ok(&as), "a target outside the page is refused"); + fresh((v4_cell)(V4_NODE_WORDS - 1u)); + v4_asm_lit(&as, 1); v4_asm_op(&as, V4_OP_SEMI); + CHECK(!v4_asm_ok(&as), "running off the end of memory is refused"); + CHECK(v4_node_guards_intact(&n), "and does not touch the guard band"); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_exec.c b/v4/tests/test_exec.c new file mode 100644 index 00000000..850670f1 --- /dev/null +++ b/v4/tests/test_exec.c @@ -0,0 +1,339 @@ +/* test_exec.c -- every opcode of DECOMPOSITION.md section 1.3, and the slot + * sequencing of section 1.2. + * + * Each expected value below is worked out from the table in section 1.3, not + * from the executor. Words are packed by hand with v4_iword_assemble so that + * this file does not depend on asm.c, which has its own test. + */ +#include "v4/exec.h" +#include + +static int failures = 0, checks = 0; +#define CHECK(c,...) do{checks++; if(!(c)){failures++; printf("FAIL %s:%d: ",__FILE__,__LINE__); printf(__VA_ARGS__); printf("\n");}}while(0) + +#define NOP V4_OP_NOP +#define MINC ((v4_cell)V4_MSB) /* most negative cell */ +#define MAXC ((v4_cell)(V4_MSB - 1u)) /* most positive cell */ + +static v4_node n; +static v4_exec_state es; +static v4_heat h; + +static void fresh(void) +{ + v4_node_reset(&n); + v4_exec_reset(&es); + v4_heat_reset(&h); +} + +static void dpush(v4_cell x) { v4_dstack_push(&n.ds, x); } +static void rpush(v4_cell x) { v4_rstack_push(&n.rs, x); } + +/* A word of six plain opcodes. */ +static v4_cell word6(unsigned a, unsigned b, unsigned c, + unsigned d, unsigned e, unsigned f) +{ + unsigned op[V4_SLOT_COUNT]; + op[0] = a; op[1] = b; op[2] = c; op[3] = d; op[4] = e; op[5] = f; + return (v4_cell)v4_iword_assemble(op, 0); +} + +/* A word with `npre` nops, then a branch in the next slot, then its target. */ +static v4_cell bword(unsigned npre, unsigned op, v4_iword target) +{ + v4_iword w = 0; + for (unsigned k = 0; k < npre; k++) w |= (v4_iword)NOP << V4_SLOT_LOW_BIT(k); + w |= (v4_iword)op << V4_SLOT_LOW_BIT(npre); + w |= target & v4_iword_slot_mask(npre); + return (v4_cell)w; +} + +/* Run one opcode, padded with nops, on the current stacks. Returns the count + * of instructions retired. */ +static unsigned run1(unsigned op) +{ + n.mem[0] = word6(op, NOP, NOP, NOP, NOP, NOP); + n.p = 0; + return v4_exec_step_word(&n, &es, &h); +} + +static void test_stack_ops(void) +{ + fresh(); dpush(1); dpush(2); run1(V4_OP_DUP); + CHECK(n.ds.t == 2 && n.ds.s == 2, "dup"); + (void)v4_dstack_pop(&n.ds); + CHECK(n.ds.t == 2 && n.ds.s == 1, "dup keeps what was below"); + + fresh(); dpush(1); dpush(2); run1(V4_OP_OVER); + CHECK(n.ds.t == 1 && n.ds.s == 2, "over"); + (void)v4_dstack_pop(&n.ds); + CHECK(n.ds.t == 2 && n.ds.s == 1, "over keeps what was below"); + + fresh(); dpush(1); dpush(2); dpush(3); run1(V4_OP_DROP); + CHECK(n.ds.t == 2 && n.ds.s == 1, "drop"); + + fresh(); dpush(1); dpush(2); CHECK(run1(NOP) == 6, "nop word retires six"); + CHECK(n.ds.t == 2 && n.ds.s == 1 && n.p == 1, "nop changes nothing but P"); +} + +static void test_alu(void) +{ + fresh(); dpush(9); dpush(3); dpush(4); run1(V4_OP_ADD); + CHECK(n.ds.t == 7 && n.ds.s == 9, "3 + 4, and 9 moves up to S"); + fresh(); dpush(MAXC); dpush(1); run1(V4_OP_ADD); + CHECK(n.ds.t == MINC, "+ wraps"); + + fresh(); dpush(9); dpush(12); dpush(10); run1(V4_OP_AND); + CHECK(n.ds.t == 8 && n.ds.s == 9, "and"); + fresh(); dpush(9); dpush(12); dpush(10); run1(V4_OP_XOR); + CHECK(n.ds.t == 6 && n.ds.s == 9, "xor"); + + fresh(); dpush(5); dpush(0); run1(V4_OP_INV); + CHECK(n.ds.t == -1 && n.ds.s == 5, "inv 0"); + fresh(); dpush(MAXC); run1(V4_OP_INV); + CHECK(n.ds.t == MINC, "inv max"); + + fresh(); dpush(3); run1(V4_OP_TWO_STAR); CHECK(n.ds.t == 6, "2* 3"); + fresh(); dpush(-1); run1(V4_OP_TWO_STAR); CHECK(n.ds.t == -2, "2* -1"); + fresh(); dpush(MINC); run1(V4_OP_TWO_STAR); CHECK(n.ds.t == 0, "2* drops the top bit"); + + fresh(); dpush(6); run1(V4_OP_TWO_SLASH); CHECK(n.ds.t == 3, "2/ 6"); + fresh(); dpush(-1); run1(V4_OP_TWO_SLASH); CHECK(n.ds.t == -1, "2/ -1 is -1, not 0"); + fresh(); dpush(-7); run1(V4_OP_TWO_SLASH); CHECK(n.ds.t == -4, "2/ -7 floors"); + fresh(); dpush(MINC); run1(V4_OP_TWO_SLASH); + CHECK((v4_ucell)n.ds.t == (V4_MSB | (V4_MSB >> 1)), "2/ keeps the sign bit"); +} + +static void test_mul_step(void) +{ + /* A even: no add. */ + fresh(); dpush(5); dpush(6); n.a = 4; run1(V4_OP_MUL_STEP); + CHECK(n.ds.t == 3 && n.ds.s == 5 && n.a == 2, "+* A even"); + + /* A odd: T+S = 11, shifted to 5, the 1 shifted out enters the top of A. */ + fresh(); dpush(5); dpush(6); n.a = 5; run1(V4_OP_MUL_STEP); + CHECK(n.ds.t == 5 && n.ds.s == 5, "+* A odd, T"); + CHECK((v4_ucell)n.a == (V4_MSB | 2u), "+* A odd, A"); + + fresh(); dpush(0); dpush(3); n.a = 0; run1(V4_OP_MUL_STEP); + CHECK(n.ds.t == 1 && (v4_ucell)n.a == V4_MSB, "+* T bit 0 enters A"); + + fresh(); dpush(0); dpush(-2); n.a = 0; run1(V4_OP_MUL_STEP); + CHECK(n.ds.t == -1 && n.a == 0, "+* keeps T's sign bit"); + + /* D-3: all-ones + 1 carries out of T and the carry is gone. */ + fresh(); dpush(1); dpush(-1); n.a = 1; run1(V4_OP_MUL_STEP); + CHECK(n.ds.t == 0 && n.a == 0, "+* does not keep the carry (D-3)"); +} + +static void test_registers_and_rstack(void) +{ + fresh(); dpush(4); dpush(5); run1(V4_OP_PUSH); + CHECK(n.rs.r == 5 && n.ds.t == 4, "push"); + run1(V4_OP_RPOP); + CHECK(n.ds.t == 5 && n.ds.s == 4, "pop"); + + fresh(); dpush(4); dpush(77); run1(V4_OP_BANG_A); + CHECK(n.a == 77 && n.ds.t == 4, "a!"); + run1(V4_OP_PUSH_A); + CHECK(n.ds.t == 77 && n.ds.s == 4 && n.a == 77, "a"); + + fresh(); dpush(4); dpush(88); run1(V4_OP_BANG_B); + CHECK(n.b == 88 && n.ds.t == 4 && n.a == 0, "b!"); +} + +static void test_memory(void) +{ + fresh(); n.a = 100; n.b = 101; n.mem[100] = 11; n.mem[101] = 22; + run1(V4_OP_FETCH_A); CHECK(n.ds.t == 11 && n.a == 100, "@"); + run1(V4_OP_FETCH_B); CHECK(n.ds.t == 22 && n.ds.s == 11 && n.b == 101, "@b"); + run1(V4_OP_FETCH_INC); CHECK(n.ds.t == 11 && n.a == 101, "@+ first"); + run1(V4_OP_FETCH_INC); CHECK(n.ds.t == 22 && n.a == 102, "@+ second"); + + fresh(); n.a = 100; n.b = 200; dpush(1); dpush(5); + run1(V4_OP_STORE_A); + CHECK(n.mem[100] == 5 && n.ds.t == 1 && n.a == 100, "!"); + dpush(6); run1(V4_OP_STORE_INC); + CHECK(n.mem[100] == 6 && n.ds.t == 1 && n.a == 101, "!+"); + dpush(7); run1(V4_OP_STORE_B); + CHECK(n.mem[200] == 7 && n.ds.t == 1 && n.b == 200, "!b"); + + /* @p takes the word after the instruction word and steps over it. */ + fresh(); n.mem[1] = (v4_cell)0x1122334455667788LL; + run1(V4_OP_FETCH_P); + CHECK(n.ds.t == n.mem[1] && n.p == 2, "@p is a full cell and advances P"); + + fresh(); dpush(1); dpush(9); run1(V4_OP_STORE_P); + CHECK(n.mem[1] == 9 && n.ds.t == 1 && n.p == 2, "!p"); + + /* Two literals in one word are the two words after it, in order. */ + fresh(); n.mem[0] = word6(V4_OP_FETCH_P, V4_OP_FETCH_P, V4_OP_ADD, NOP, NOP, NOP); + n.mem[1] = 30; n.mem[2] = 12; n.p = 0; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.ds.t == 42 && n.p == 3, "two literals"); +} + +static void test_return_and_ex(void) +{ + fresh(); rpush(7); rpush(50); + n.mem[0] = word6(V4_OP_SEMI, V4_OP_DUP, V4_OP_DUP, V4_OP_DUP, V4_OP_DUP, V4_OP_DUP); + n.p = 0; + CHECK(v4_exec_step_word(&n, &es, &h) == 1, "; ends the word"); + CHECK(n.p == 50 && n.rs.r == 7, "; returns and pops R"); + CHECK(h.op[V4_OP_DUP] == 0 && es.anticlock == 1, "nothing after ; retires"); + + fresh(); rpush(7); rpush(50); + n.mem[20] = word6(V4_OP_EX, V4_OP_DUP, NOP, NOP, NOP, NOP); + n.p = 20; + CHECK(v4_exec_step_word(&n, &es, &h) == 1, "ex ends the word"); + CHECK(n.p == 50 && n.rs.r == 21, "ex swaps P and R"); + (void)v4_rstack_pop(&n.rs); + CHECK(n.rs.r == 7, "ex does not change the return stack depth"); +} + +static void test_branches(void) +{ + /* jump in each legal slot; the slots before it execute. */ + for (unsigned slot = 0; slot < 4; slot++) { + fresh(); n.mem[10] = bword(slot, V4_OP_JUMP, 300); n.p = 10; + CHECK(v4_exec_step_word(&n, &es, &h) == slot + 1, "jump slot %u count", slot); + CHECK(n.p == 300, "jump slot %u", slot); + CHECK(h.op[NOP] == slot && h.op[V4_OP_JUMP] == 1, "jump slot %u heat", slot); + } + + fresh(); rpush(7); n.mem[10] = bword(0, V4_OP_CALL, 200); n.p = 10; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 200 && n.rs.r == 11, "call pushes the return address"); + CHECK(v4_heat_call_get(&h, 200) == 1 && h.op[V4_OP_CALL] == 1, "call heats its target"); + (void)v4_rstack_pop(&n.rs); + CHECK(n.rs.r == 7, "call pushes exactly one"); + + fresh(); rpush(99); rpush(2); n.mem[5] = bword(0, V4_OP_NEXT, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 40 && n.rs.r == 1, "next, R nonzero"); + fresh(); rpush(99); rpush(0); n.mem[5] = bword(0, V4_OP_NEXT, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 6 && n.rs.r == 99, "next, R zero"); + + fresh(); dpush(8); dpush(0); n.mem[5] = bword(0, V4_OP_IF, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 40 && n.ds.t == 0 && n.ds.s == 8, "if, T zero: taken, T kept"); + /* The target bits 0x0FFFF would decode as opcodes if they were executed. */ + fresh(); dpush(8); dpush(5); n.mem[5] = bword(0, V4_OP_IF, 0x0FFFF); n.p = 5; + CHECK(v4_exec_step_word(&n, &es, &h) == 1, "if not taken still ends the word"); + CHECK(n.p == 6 && n.ds.t == 5 && n.ds.s == 8, "if, T nonzero: not taken, T kept"); + + fresh(); dpush(8); dpush(5); n.mem[5] = bword(0, V4_OP_MINUS_IF, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 40 && n.ds.t == 5 && n.ds.s == 8, "-if, T positive: taken"); + fresh(); dpush(0); n.mem[5] = bword(0, V4_OP_MINUS_IF, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 40, "-if, T zero: taken"); + fresh(); dpush(8); dpush(-1); n.mem[5] = bword(0, V4_OP_MINUS_IF, 40); n.p = 5; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 6 && n.ds.t == -1 && n.ds.s == 8, "-if, T negative: not taken"); +} + +static void test_unext(void) +{ + /* R = 3: the body runs four times, then the rest of the word runs once. */ + fresh(); dpush(1); rpush(99); rpush(3); + n.mem[0] = word6(V4_OP_TWO_STAR, V4_OP_UNEXT, NOP, NOP, NOP, NOP); n.p = 0; + CHECK(v4_exec_step_word(&n, &es, &h) == 12, "unext retire count"); + CHECK(n.ds.t == 16, "unext body ran R+1 times"); + CHECK(n.rs.r == 99, "unext pops R when done"); + CHECK(n.p == 1, "unext leaves P at the next word"); + CHECK(h.op[V4_OP_TWO_STAR] == 4 && h.op[V4_OP_UNEXT] == 4 && h.op[NOP] == 4, + "unext heat"); + CHECK(es.anticlock == 12, "unext anticlock"); + + fresh(); dpush(1); rpush(99); rpush(0); + n.mem[0] = word6(V4_OP_TWO_STAR, V4_OP_UNEXT, NOP, NOP, NOP, NOP); n.p = 0; + CHECK(v4_exec_step_word(&n, &es, &h) == 6, "unext with R zero runs once"); + CHECK(n.ds.t == 2 && n.rs.r == 99, "unext with R zero"); + + /* The word is not refetched and P is not rewound, so @p walks forward. */ + fresh(); rpush(99); rpush(2); n.a = 100; + n.mem[0] = word6(V4_OP_FETCH_P, V4_OP_STORE_INC, V4_OP_UNEXT, NOP, NOP, NOP); + n.mem[1] = 11; n.mem[2] = 22; n.mem[3] = 33; n.p = 0; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.mem[100] == 11 && n.mem[101] == 22 && n.mem[102] == 33, + "@p in a unext loop copies successive words"); + CHECK(n.a == 103 && n.p == 4 && n.rs.r == 99, "@p in a unext loop, registers"); +} + +static void test_retirement(void) +{ + fresh(); + for (unsigned op = 0; op < V4_OPCODE_COUNT; op++) v4_exec_op(&n, &es, &h, op, 0, 0); + for (unsigned op = 0; op < V4_OPCODE_COUNT; op++) + CHECK(h.op[op] == 1, "opcode %u retires once", op); + CHECK(es.anticlock == V4_OPCODE_COUNT, "one anticlock tick per opcode"); + + v4_exec_op(&n, &es, &h, V4_OPCODE_COUNT, 0, 0); + CHECK(es.anticlock == V4_OPCODE_COUNT, "a non-opcode retires nothing"); +} + +#if V4_CELL_BITS == 64 +static void test_iword_in_wide_cell(void) +{ + /* D-9: the instruction word is the low 32 bits of the cell. The high half + * is not decoded and does not reach a branch target. */ + v4_cell junk = (v4_cell)0xA5A5A5A500000000ULL; + + fresh(); dpush(3); + n.mem[0] = junk | word6(V4_OP_DUP, V4_OP_ADD, NOP, NOP, NOP, NOP); n.p = 0; + CHECK(v4_exec_step_word(&n, &es, &h) == 6, "wide cell: six slots"); + CHECK(n.ds.t == 6, "wide cell: decode ignores the high half"); + + fresh(); n.mem[10] = junk | bword(0, V4_OP_JUMP, 300); n.p = 10; + (void)v4_exec_step_word(&n, &es, &h); + CHECK(n.p == 300, "wide cell: branch ignores the high half"); +} +#endif + +/* Many opcodes on arbitrary values. The point is the sanitizer build: no + * opcode may overflow a signed cell or shift out of range, whatever is on the + * stacks. Memory opcodes are left out because A, B and P would be arbitrary. */ +static void test_soak(void) +{ + static const unsigned safe[] = { + V4_OP_MUL_STEP, V4_OP_TWO_STAR, V4_OP_TWO_SLASH, V4_OP_INV, V4_OP_ADD, + V4_OP_AND, V4_OP_XOR, V4_OP_DROP, V4_OP_DUP, V4_OP_RPOP, V4_OP_OVER, + V4_OP_PUSH_A, V4_OP_NOP, V4_OP_PUSH, V4_OP_BANG_B, V4_OP_BANG_A + }; + uint64_t seed = 0x9E3779B97F4A7C15ULL; + const unsigned count = 200000; + + fresh(); + dpush(MINC); dpush(MAXC); dpush(-1); dpush(1); + for (unsigned i = 0; i < count; i++) { + seed = seed * 6364136223846793005ULL + 1442695040888963407ULL; + if ((seed >> 60) == 0) dpush((v4_cell)(seed >> 7)); + v4_exec_op(&n, &es, &h, safe[(seed >> 33) % (sizeof safe / sizeof safe[0])], 0, 0); + } + CHECK(es.anticlock == count, "soak retired every opcode"); + CHECK(v4_node_guards_intact(&n), "soak left the guards intact"); +} + +int main(void) +{ + printf("v4 exec tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + test_stack_ops(); + test_alu(); + test_mul_step(); + test_registers_and_rstack(); + test_memory(); + test_return_and_ex(); + test_branches(); + test_unext(); + test_retirement(); +#if V4_CELL_BITS == 64 + test_iword_in_wide_cell(); +#endif + test_soak(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_foundation.c b/v4/tests/test_foundation.c new file mode 100644 index 00000000..6e288e09 --- /dev/null +++ b/v4/tests/test_foundation.c @@ -0,0 +1,209 @@ +/* test_foundation.c -- the first DECOMPOSITION.md definitions, executed. + * + * Section 4 says of its colon definitions that each "has been traced by hand, + * but none has been executed". This file assembles the helpers, the sign and + * zero tests, U< and UM* exactly as written there (plus 2DUP and - from + * section 5, which U< needs) and runs them on the golden model against the C + * operation each one stands for. + * + * Every call is made with a canary under the arguments, and the canary must + * still be directly under the results afterwards: a definition that leaves + * the right answer but an unbalanced stack is wrong. + */ +#include "v4/asm.h" +#include "v4/testcode.h" +#include "v4/umul.h" +#include + +static int failures = 0, checks = 0; +#define CHECK(c,...) do{checks++; if(!(c)){failures++; printf("FAIL %s:%d: ",__FILE__,__LINE__); printf(__VA_ARGS__); printf("\n");}}while(0) + +#define CANARY ((v4_cell)0x0C0FFEE5) +#define MAXU ((v4_ucell)~(v4_ucell)0) +#define FLAG(c) ((c) ? V4_ALL_ONES : (v4_cell)0) + +static v4_node n; +static v4_exec_state es; +static v4_heat h; +static v4_asm as; + +static v4_cell w_nip, w_swap, w_or, w_negate, w_rot, w_zless, w_zequal, + w_2dup, w_minus, w_uless, w_umstar; + +#define O(name) v4_asm_op(&as, V4_OP_##name) +#define LIT(v) v4_asm_lit(&as, (v4_cell)(v)) +#define CALL(w) v4_asm_branch(&as, V4_OP_CALL, (w)) + +static void build(void) +{ + v4_asm_ref ref; + + v4_node_reset(&n); + v4_asm_begin(&as, &n, 16); + + /* : NIP push drop pop ; */ + w_nip = v4_asm_label(&as); + O(PUSH); O(DROP); O(RPOP); O(SEMI); + + /* : SWAP over push push drop pop pop ; */ + w_swap = v4_asm_label(&as); + O(OVER); O(PUSH); O(PUSH); O(DROP); O(RPOP); O(RPOP); O(SEMI); + + /* : OR over inv and xor ; */ + w_or = v4_asm_label(&as); + O(OVER); O(INV); O(AND); O(XOR); O(SEMI); + + /* : NEGATE inv 1 + ; */ + w_negate = v4_asm_label(&as); + O(INV); LIT(1); O(ADD); O(SEMI); + + /* : ROT push SWAP pop SWAP ; */ + w_rot = v4_asm_label(&as); + O(PUSH); CALL(w_swap); O(RPOP); CALL(w_swap); O(SEMI); + + /* : 0< -if L1 drop -1 ; L1: drop 0 ; */ + w_zless = v4_asm_label(&as); + ref = v4_asm_branch_fwd(&as, V4_OP_MINUS_IF); + O(DROP); LIT(-1); O(SEMI); + v4_asm_resolve(&as, ref, v4_asm_label(&as)); + O(DROP); LIT(0); O(SEMI); + + /* : 0= if L1 drop 0 ; L1: drop -1 ; */ + w_zequal = v4_asm_label(&as); + ref = v4_asm_branch_fwd(&as, V4_OP_IF); + O(DROP); LIT(0); O(SEMI); + v4_asm_resolve(&as, ref, v4_asm_label(&as)); + O(DROP); LIT(-1); O(SEMI); + + /* : 2DUP over over ; (section 5.1) */ + w_2dup = v4_asm_label(&as); + O(OVER); O(OVER); O(SEMI); + + /* : - NEGATE + ; (section 5.4), NEGATE in line */ + w_minus = v4_asm_label(&as); + O(INV); LIT(1); O(ADD); O(ADD); O(SEMI); + + /* : U< 2DUP xor 0< IF NIP 0< ELSE - 0< THEN ; + * with section 2's IF: the flag is dropped on both arms. */ + w_uless = v4_asm_label(&as); + CALL(w_2dup); O(XOR); CALL(w_zless); + ref = v4_asm_branch_fwd(&as, V4_OP_IF); + O(DROP); CALL(w_nip); CALL(w_zless); O(SEMI); + v4_asm_resolve(&as, ref, v4_asm_label(&as)); + O(DROP); CALL(w_minus); CALL(w_zless); O(SEMI); + + /* : UM* a! 0 31 FOR +* UNEXT push drop a pop ; + * 31 is the cell width less one; written for a 32-bit cell in section 4. + * The loop body is the start of its own word so that unext restarts it. */ + w_umstar = v4_asm_label(&as); + O(BANG_A); LIT(0); LIT(V4_CELL_BITS - 1); O(PUSH); + (void)v4_asm_label(&as); + O(MUL_STEP); O(UNEXT); O(PUSH); O(DROP); O(PUSH_A); O(RPOP); + O(SEMI); + + CHECK(v4_asm_ok(&as), "foundation words assemble"); +} + +/* Call `word` with a canary and up to three arguments on fresh stacks. */ +static int call(v4_cell word, unsigned argc, v4_cell a, v4_cell b, v4_cell c) +{ + v4_dstack_reset(&n.ds); + v4_rstack_reset(&n.rs); + v4_exec_reset(&es); + v4_heat_reset(&h); + v4_dstack_push(&n.ds, CANARY); + if (argc > 0) v4_dstack_push(&n.ds, a); + if (argc > 1) v4_dstack_push(&n.ds, b); + if (argc > 2) v4_dstack_push(&n.ds, c); + return v4_test_call(&n, &es, &h, word, 1000) > 0; +} + +/* The results, top first, then the canary. */ +static int left1(v4_cell t) +{ + return n.ds.t == t && n.ds.s == CANARY; +} +static int left2(v4_cell s, v4_cell t) +{ + if (n.ds.t != t || n.ds.s != s) return 0; + (void)v4_dstack_pop(&n.ds); + return n.ds.s == CANARY; +} +static int left3(v4_cell third, v4_cell s, v4_cell t) +{ + if (n.ds.t != t) return 0; + (void)v4_dstack_pop(&n.ds); + return left2(third, s); +} + +static const v4_cell vec[] = { + 0, 1, 2, 3, -1, -2, 12345, -12345, + (v4_cell)(V4_MSB - 1u), (v4_cell)V4_MSB, (v4_cell)(V4_MSB + 1u), + (v4_cell)(V4_MSB >> 1), (v4_cell)((V4_MSB >> 1) - 1u), + (v4_cell)(MAXU / 3u), (v4_cell)(MAXU / 3u * 2u) +}; +#define NVEC (sizeof vec / sizeof vec[0]) + +/* Does UM* as written give the true product? */ +static int umstar_exact(v4_ucell u1, v4_ucell u2) +{ + v4_ucell lo, hi; + v4_umul(u1, u2, &lo, &hi); + return call(w_umstar, 2, (v4_cell)u1, (v4_cell)u2, 0) && left2((v4_cell)lo, (v4_cell)hi); +} + +int main(void) +{ + printf("v4 foundation tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + build(); + + for (unsigned i = 0; i < NVEC; i++) { + v4_cell a = vec[i]; + v4_ucell ua = (v4_ucell)a; + + CHECK(call(w_negate, 1, a, 0, 0) && left1((v4_cell)(0u - ua)), "NEGATE [%u]", i); + CHECK(call(w_zless, 1, a, 0, 0) && left1(FLAG(ua & V4_MSB)), "0< [%u]", i); + CHECK(call(w_zequal, 1, a, 0, 0) && left1(FLAG(a == 0)), "0= [%u]", i); + + for (unsigned j = 0; j < NVEC; j++) { + v4_cell b = vec[j]; + v4_ucell ub = (v4_ucell)b; + + CHECK(call(w_nip, 2, a, b, 0) && left1(b), "NIP [%u,%u]", i, j); + CHECK(call(w_swap, 2, a, b, 0) && left2(b, a), "SWAP [%u,%u]", i, j); + CHECK(call(w_or, 2, a, b, 0) && left1((v4_cell)(ua | ub)), "OR [%u,%u]", i, j); + CHECK(call(w_2dup, 2, a, b, 0) && n.ds.t == b && n.ds.s == a + && (v4_dstack_pop(&n.ds), v4_dstack_pop(&n.ds), left2(a, b)), + "2DUP [%u,%u]", i, j); + CHECK(call(w_minus, 2, a, b, 0) && left1((v4_cell)(ua - ub)), "- [%u,%u]", i, j); + CHECK(call(w_uless, 2, a, b, 0) && left1(FLAG(ua < ub)), "U< [%u,%u]", i, j); + + for (unsigned k = 0; k < NVEC; k++) { + v4_cell c = vec[k]; + CHECK(call(w_rot, 3, a, b, c) && left3(b, c, a), "ROT [%u,%u,%u]", i, j, k); + } + + /* UM* as written is exact whenever the multiplicand (u1, which + * stays in S) is at most 2^(n-2): T is then always below S, so + * T+S stays below 2^(n-1) and neither the lost carry nor +*'s + * sign-keeping shift can touch the result. u2 is unrestricted. */ + if (ua <= (V4_MSB >> 1)) + CHECK(umstar_exact(ua, ub), "UM* [%u,%u]", i, j); + } + } + + /* KNOWN LIMITS of UM* as written, recorded so they cannot be forgotten. + * Section 4 marks UM* "NEEDS REVISION" and gives the limit as "operands + * below 2^31". The measured limit is tighter and is on u1 only. These + * two checks assert that the definition is still wrong where it is known + * to be wrong; when UM* is rewritten they must be turned into ordinary + * exactness checks. */ + CHECK(!umstar_exact(MAXU, MAXU), "UM* limit: carry out of T is lost (D-3)"); + CHECK(!umstar_exact(V4_MSB - 1u, 3u), "UM* limit: u1 below 2^(n-1) still fails"); + + CHECK(v4_node_guards_intact(&n), "guards intact"); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_guard.c b/v4/tests/test_guard.c new file mode 100644 index 00000000..c7068c6a --- /dev/null +++ b/v4/tests/test_guard.c @@ -0,0 +1,213 @@ +/* test_guard.c -- the 5% safety boundary. + * + * A guard band whose only assertion is "the patterns still look right" has not + * been shown to catch anything. These tests therefore do three things a + * structural test cannot: + * + * 1. pin the sizing arithmetic, including the case that makes a 5% margin + * evaporate -- a short list, where 5% rounds down to zero elements and a + * zero-width guard is a comment rather than a boundary; + * 2. index past each end of a real stack on purpose, and require the check to + * fail, and require it to say *which* end was hit; + * 3. run the real push/pop paths and require the boundary to survive them, so + * a boundary that fires on ordinary use is caught as a false alarm rather + * than being tuned into uselessness. + */ +#include "v4/guard.h" +#include "v4/stack.h" + +#include +#include +#include + +static int failures = 0; +static int checks = 0; + +#define CHECK(cond, ...) \ + do { \ + checks++; \ + if (!(cond)) { \ + failures++; \ + printf(" FAIL %s:%d: ", __FILE__, __LINE__); \ + printf(__VA_ARGS__); \ + printf("\n"); \ + } \ + } while (0) + +static void test_sizing_rounds_up(void) +{ + /* 5% of each, rounded up. */ + CHECK(V4_GUARD_ELEMS(20) == 1, "5%% of 20 is 1, got %u", V4_GUARD_ELEMS(20)); + CHECK(V4_GUARD_ELEMS(40) == 2, "5%% of 40 is 2, got %u", V4_GUARD_ELEMS(40)); + CHECK(V4_GUARD_ELEMS(100) == 5, "5%% of 100 is 5, got %u", V4_GUARD_ELEMS(100)); + CHECK(V4_GUARD_ELEMS(1000) == 50, "5%% of 1000 is 50, got %u", V4_GUARD_ELEMS(1000)); + CHECK(V4_GUARD_ELEMS(2000) == 100, "5%% of 2000 is 100, got %u", V4_GUARD_ELEMS(2000)); +} + +static void test_sizing_never_zero(void) +{ + /* The case that matters. 5% of anything under 20 rounds to 0 by + * truncation, and a guard of zero elements catches nothing at all while + * still reading as "5% boundary" in the source. */ + for (unsigned n = 1; n < 20u; n++) { + unsigned g = V4_GUARD_ELEMS(n); + CHECK(g >= 1u, "a %u-element list got a %u-element boundary", n, g); + } + CHECK(V4_GUARD_ELEMS(1) == 1, "1-element list boundary"); + CHECK(V4_GUARD_ELEMS(3) == 1, "3-element list boundary"); + CHECK(V4_GUARD_ELEMS(8) == 1, "8-element list boundary"); + CHECK(V4_GUARD_ELEMS(19) == 1, "19-element list boundary"); + CHECK(V4_GUARD_ELEMS(V4_DATA_RING) == 1, "data ring boundary at depth 8"); + CHECK(V4_GUARD_ELEMS(V4_RET_RING) == 1, "return ring boundary at depth 8"); +} + +static void test_patterns_are_distinct_at_this_width(void) +{ + /* The compile-time assertion in guard.h covers this, but the point of + * distinct patterns is "which end", and that is only meaningful if it + * survives truncation to the cell width. */ + CHECK(V4_GUARD_PATTERN_HEAD != V4_GUARD_PATTERN_TAIL, + "head and tail patterns are equal at V4_CELL_BITS=%d", V4_CELL_BITS); + CHECK((v4_ucell)V4_GUARD_PATTERN_HEAD != 0u, "head pattern truncated to 0"); + CHECK((v4_ucell)V4_GUARD_PATTERN_TAIL != 0u, "tail pattern truncated to 0"); +} + +static void test_fill_and_intact(void) +{ + v4_cell band[7]; + v4_guard_fill(band, 7, V4_GUARD_PATTERN_HEAD); + CHECK(v4_guard_intact(band, 7, V4_GUARD_PATTERN_HEAD), "filled band intact"); + CHECK(!v4_guard_intact(band, 7, V4_GUARD_PATTERN_TAIL), + "a head-filled band must not read as a tail band"); + + /* Every element is checked, not just the ends. */ + for (unsigned i = 0; i < 7; i++) { + band[i] = 0; + CHECK(!v4_guard_intact(band, 7, V4_GUARD_PATTERN_HEAD), + "band intact after corrupting interior element %u", i); + v4_guard_fill(band, 7, V4_GUARD_PATTERN_HEAD); + } +} + +static void test_reset_installs_boundaries(void) +{ + v4_dstack d; + v4_rstack r; + + /* A struct that has never been reset holds whatever was on the stack, so + * the boundary cannot be assumed present. Poison both first: that the + * check fails before reset is the point, since it is what proves the + * check is reading the boundary and not returning a constant. */ + memset(&d, 0xA5, sizeof d); + memset(&r, 0xA5, sizeof r); + + CHECK(!v4_dstack_guards_intact(&d), + "data boundaries must not read as intact before reset"); + CHECK(!v4_rstack_guards_intact(&r), + "return boundaries must not read as intact before reset"); + + v4_dstack_reset(&d); + v4_rstack_reset(&r); + CHECK(v4_dstack_guards_intact(&d), "data boundaries intact after reset"); + CHECK(v4_rstack_guards_intact(&r), "return boundaries intact after reset"); +} + +static void test_head_overrun_is_detected(void) +{ + /* Deliberately index one element before the ring: exactly what a lost + * modulo or a sign error on the head arithmetic would do. */ + v4_dstack d; + v4_dstack_reset(&d); + CHECK(v4_dstack_guards_intact(&d), "intact before the deliberate overrun"); + + d.guard_head[0] = (v4_cell)V4_GUARD_PATTERN_TAIL; /* wrong pattern too */ + CHECK(!v4_dstack_guards_intact(&d), + "head overrun past the data ring was not detected"); + /* Which end is answerable because the two patterns differ. */ + CHECK((v4_ucell)d.guard_head[0] != V4_GUARD_PATTERN_HEAD, + "head boundary no longer holds its own pattern"); + + v4_dstack_reset(&d); + CHECK(v4_dstack_guards_intact(&d), "intact after re-reset"); +} + +static void test_tail_overrun_is_detected(void) +{ + v4_dstack d; + v4_dstack_reset(&d); + + d.guard_tail[0] = 0; + CHECK(!v4_dstack_guards_intact(&d), + "tail overrun past the data ring was not detected"); + + v4_dstack_reset(&d); + CHECK(v4_dstack_guards_intact(&d), "intact after re-reset"); +} + +static void test_return_stack_boundaries(void) +{ + v4_rstack r; + v4_rstack_reset(&r); + CHECK(v4_rstack_guards_intact(&r), "return boundaries intact after reset"); + + r.guard_head[0] = 0; + CHECK(!v4_rstack_guards_intact(&r), "return head overrun not detected"); + v4_rstack_reset(&r); + + r.guard_tail[V4_RET_BOUND - 1u] = 0; + CHECK(!v4_rstack_guards_intact(&r), "return tail overrun not detected"); + v4_rstack_reset(&r); + CHECK(v4_rstack_guards_intact(&r), "return intact after re-reset"); +} + +static void test_ordinary_use_never_touches_the_boundaries(void) +{ + /* The other half of the contract: a boundary that fires on legitimate use + * is worse than none, because it trains you to ignore it. Drive both + * stacks hard, well past their depths, and require the boundaries to + * survive every operation. */ + v4_dstack d; + v4_rstack r; + v4_dstack_reset(&d); + v4_rstack_reset(&r); + + uint64_t s = 0x243F6A8885A308D3ull; + for (int i = 0; i < 50000; i++) { + s ^= s << 13; s ^= s >> 7; s ^= s << 17; + + v4_dstack_push(&d, (v4_cell)s); + v4_rstack_push(&r, (v4_cell)(s >> 11)); + if (i & 1) { + (void)v4_dstack_pop(&d); + (void)v4_rstack_pop(&r); + } + } + CHECK(v4_dstack_guards_intact(&d), + "data boundary damaged by ordinary push/pop"); + CHECK(v4_rstack_guards_intact(&r), + "return boundary damaged by ordinary push/pop"); + + /* head must still be in range, which is the invariant the boundary exists + * to police. */ + CHECK(d.head < V4_DATA_RING, "data head %u out of range", d.head); + CHECK(r.head < V4_RET_RING, "return head %u out of range", r.head); +} + +int main(void) +{ + printf("v4 guard tests: V4_CELL_BITS=%d, data bound %d, return bound %d\n", + V4_CELL_BITS, V4_DATA_BOUND, V4_RET_BOUND); + + test_sizing_rounds_up(); + test_sizing_never_zero(); + test_patterns_are_distinct_at_this_width(); + test_fill_and_intact(); + test_reset_installs_boundaries(); + test_head_overrun_is_detected(); + test_tail_overrun_is_detected(); + test_return_stack_boundaries(); + test_ordinary_use_never_touches_the_boundaries(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_heat.c b/v4/tests/test_heat.c new file mode 100644 index 00000000..914934df --- /dev/null +++ b/v4/tests/test_heat.c @@ -0,0 +1,109 @@ +/* test_heat.c -- heat and anti-clock. */ +#include "v4/heat.h" +#include +#include +#include + +static int failures = 0; +static int checks = 0; + +#define CHECK(cond, ...) \ + do { \ + checks++; \ + if (!(cond)) { \ + failures++; \ + printf(" FAIL %s:%d: ", __FILE__, __LINE__); \ + printf(__VA_ARGS__); \ + printf("\n"); \ + } \ + } while (0) + +static void test_reset(void) +{ + v4_heat h; + memset(&h, 0xFF, sizeof h); + v4_heat_reset(&h); + + for (unsigned i = 0; i < V4_OPCODE_COUNT; i++) { + CHECK(h.op[i] == 0, "op %u heat non-zero after reset", i); + } + for (unsigned i = 0; i < V4_HEAT_MAX_CALL_TARGETS; i++) { + CHECK(h.call[i] == 0, "call target %u heat non-zero after reset", i); + CHECK(!v4_heat_call_is_frozen(&h, (v4_cell)i), + "call target %u frozen after reset", i); + } + v4_uheat_t ac = 999; + v4_heat_on_retire(&h, V4_OP_NOP, &ac); + CHECK(ac == 1000, "anticlock not advanced"); + CHECK(h.op[V4_OP_NOP] == 1, "nop heat"); +} + +static void test_retire_advances_anticlock_and_opcode(void) +{ + v4_heat h; + v4_heat_reset(&h); + v4_uheat_t ac = 0; + for (int i = 0; i < 100; i++) { + v4_heat_on_retire(&h, V4_OP_ADD, &ac); + } + CHECK(ac == 100, "anticlock"); + CHECK(h.op[V4_OP_ADD] == 100, "add heat"); + for (unsigned i = 0; i < V4_OPCODE_COUNT; i++) { + if (i == V4_OP_ADD) continue; + CHECK(h.op[i] == 0, "other op %u", i); + } +} + +static void test_call_advances_target(void) +{ + v4_heat h; + v4_heat_reset(&h); + v4_heat_on_call(&h, 10); + v4_heat_on_call(&h, 10); + v4_heat_on_call(&h, 5); + CHECK(v4_heat_call_get(&h, 10) == 2, "target 10"); + CHECK(v4_heat_call_get(&h, 5) == 1, "target 5"); + CHECK(v4_heat_call_get(&h, 1) == 0, "target 1"); +} + +static void test_freeze(void) +{ + v4_heat h; + v4_heat_reset(&h); + v4_heat_on_call(&h, 7); + v4_heat_call_freeze(&h, 7); + CHECK(v4_heat_call_is_frozen(&h, 7), "frozen"); + v4_heat_on_call(&h, 7); + CHECK(v4_heat_call_get(&h, 7) == 1, "still 1"); + CHECK(v4_heat_call_set(&h, 7, 100) == 0, "set returns 0 when frozen"); + CHECK(v4_heat_call_get(&h, 7) == 1, "unchanged"); + v4_heat_call_unfreeze(&h, 7); + CHECK(!v4_heat_call_is_frozen(&h, 7), "unfrozen"); + CHECK(v4_heat_call_set(&h, 7, 42) == 1, "set allowed"); + CHECK(v4_heat_call_get(&h, 7) == 42, "changed"); +} + +static void test_out_of_range_ignored(void) +{ + v4_heat h; + v4_heat_reset(&h); + v4_heat_on_call(&h, (v4_cell)V4_HEAT_MAX_CALL_TARGETS + 5); + CHECK(v4_heat_call_get(&h, (v4_cell)V4_HEAT_MAX_CALL_TARGETS + 5) == 0, + "out of range call ignored"); + CHECK(v4_heat_call_set(&h, (v4_cell)-1, 123) == 0, "set out of range"); + CHECK(v4_heat_call_is_frozen(&h, (v4_cell)-1) == 0, "frozen check"); +} + +int main(void) +{ + printf("v4 heat tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + test_reset(); + test_retire_advances_anticlock_and_opcode(); + test_call_advances_target(); + test_freeze(); + test_out_of_range_ignored(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_iword.c b/v4/tests/test_iword.c new file mode 100644 index 00000000..7e87add7 --- /dev/null +++ b/v4/tests/test_iword.c @@ -0,0 +1,400 @@ +/* test_iword.c -- the instruction-word format. + * + * The properties pinned here are the ones the rest of the machine is built on + * and the ones most likely to be got wrong by a plausible-looking edit: + * + * - the six slots occupy the bit ranges the diagram in section 1.2 gives + * them, with slot 0 highest; + * - the reach column (27 / 22 / 17 / 12 for slots 0-3) is what the layout + * actually produces, rather than four independently-entered constants; + * - branches are illegal in slots 4 and 5; + * - the two spare bits are inert, so a word that differs only in them + * decodes identically; + * - a branch is page-relative: it replaces the low bits of P and preserves + * the rest; + * - no opcode name collides with a FORTH-79 word name, which is the stated + * reason for the `inv` and `xor` renames in section 1.3. + */ +#include "v4/iword.h" + +#include +#include + +static int failures = 0; +static int checks = 0; + +#define CHECK(cond, ...) \ + do { \ + checks++; \ + if (!(cond)) { \ + failures++; \ + printf(" FAIL %s:%d: ", __FILE__, __LINE__); \ + printf(__VA_ARGS__); \ + printf("\n"); \ + } \ + } while (0) + +static v4_iword word_of(const unsigned op[6], unsigned spare) +{ + return v4_iword_assemble(op, spare); +} + +static void test_slot_bit_ranges(void) +{ + /* The diagram's own boundaries, entered independently of the macros so the + * macros are checked rather than merely used. */ + static const unsigned want_high[6] = { 31, 26, 21, 16, 11, 6 }; + static const unsigned want_low[6] = { 27, 22, 17, 12, 7, 2 }; + + for (unsigned k = 0; k < 6; k++) { + CHECK(V4_SLOT_HIGH_BIT(k) == want_high[k], + "slot %u high bit: macro says %u, diagram says %u", k, + V4_SLOT_HIGH_BIT(k), want_high[k]); + CHECK(V4_SLOT_LOW_BIT(k) == want_low[k], + "slot %u low bit: macro says %u, diagram says %u", k, + V4_SLOT_LOW_BIT(k), want_low[k]); + } + + /* Each slot is 5 bits, and the six slots do not overlap. */ + for (unsigned k = 0; k < 6; k++) { + CHECK(V4_SLOT_HIGH_BIT(k) - V4_SLOT_LOW_BIT(k) == 4u, + "slot %u is not 5 bits wide", k); + } + CHECK(V4_SLOT_LOW_BIT(5) - V4_SPARE_BITS == 0u, + "slot 5 does not end just above the spare bits"); +} + +static void test_roundtrip_all_opcodes(void) +{ + /* Every one of the 32 opcodes, in every slot, survives assemble+decode. */ + for (unsigned op = 0; op < 32u; op++) { + for (unsigned k = 0; k < 6; k++) { + unsigned o[6]; + for (unsigned j = 0; j < 6; j++) o[j] = 0x1Du; /* distinct filler */ + o[k] = op; + v4_iword w = word_of(o, 0u); + CHECK(w != 0xFFFFFFFFu, "assemble rejected opcode %u", op); + CHECK(v4_iword_op(w, k) == op, + "opcode %u in slot %u decoded as %u", op, k, + v4_iword_op(w, k)); + /* And the other slots are undisturbed. */ + for (unsigned j = 0; j < 6; j++) { + if (j == k) continue; + CHECK(v4_iword_op(w, j) == 0x1Du, + "assembling %u into slot %u disturbed slot %u", op, k, j); + } + } + } +} + +static void test_slots_are_independent(void) +{ + /* All 32^6 words cannot be enumerated, but the packing is linear, so + * checking that each slot's bits land only in its own range is enough. */ + for (unsigned k = 0; k < 6; k++) { + unsigned o[6] = { 0, 0, 0, 0, 0, 0 }; + o[k] = 0x1Fu; /* all five bits set */ + v4_iword w = word_of(o, 0u); + v4_iword expect = (v4_iword)0x1Fu << V4_SLOT_LOW_BIT(k); + CHECK(w == expect, + "slot %u: word is %08lX, expected %08lX", k, + (unsigned long)w, (unsigned long)expect); + } +} + +static void test_spare_bits_are_inert_for_decode(void) +{ + /* The two spare low bits sit below every slot, so they are never executed. + * A word that differs only in them must decode to the same six opcodes -- + * otherwise the spare bits would have acquired decode meaning. */ + unsigned o[6] = { 2, 3, 5, 6, 7, 1 }; + v4_iword a = word_of(o, 0u); + v4_iword b = word_of(o, 3u); /* both spare patterns */ + + for (unsigned k = 0; k < 6; k++) { + CHECK(v4_iword_op(a, k) == v4_iword_op(b, k), + "spare bits changed the opcode in slot %u", k); + } + CHECK(v4_iword_spare(a) == 0u, "spare bits of word a"); + CHECK(v4_iword_spare(b) == 3u, "spare bits of word b"); +} + +static void test_spare_bits_are_branch_address_bits(void) +{ + /* The opposite property, and the one that is easy to get wrong: the spare + * bits are BELOW every slot, so "all bits to the right of its slot" -- the + * branch target, per section 1.2 -- includes them. They are the low bits + * of a branch address, not padding. Pinned here so that a later "cleanup" + * that treats them as unusable address bits fails a test instead of costing + * two bits of reach off every branch in the machine. */ + /* Only slot 0 is set, so every bit below it -- all 27 of them, spare bits + * included -- is zero and the branch target is the spare field alone. */ + unsigned o[6] = { 2, 0, 0, 0, 0, 0 }; + v4_iword a = word_of(o, 0u); + v4_iword b = word_of(o, 3u); /* both spare patterns */ + + CHECK(v4_iword_op(a, 0) == V4_OP_JUMP, "slot 0 should be jump"); + CHECK(v4_iword_target(a, 0u) == 0u, + "word a (spare 0) should have a zero low target field, got %u", + v4_iword_target(a, 0u)); + CHECK(v4_iword_target(b, 0u) == 3u, + "word b (spare 3) should carry 3 in the low address bits, got %u", + v4_iword_target(b, 0u)); + CHECK(v4_iword_branch((v4_cell)0, a, 0u) == 0, "branch with spare 0"); + CHECK(v4_iword_branch((v4_cell)0, b, 0u) == 3, + "branch with spare 3 should land on address 3, got %lld", + (long long)v4_iword_branch((v4_cell)0, b, 0u)); + + /* Every legal branch slot reaches bit 0, so the spare bits are address in + * all four of them. */ + for (unsigned k = 0; k < 4; k++) { + CHECK((v4_iword_slot_mask(k) & 3u) == 3u, + "slot %u branch target does not reach the spare bits", k); + } +} + +static void test_reach_column(void) +{ + /* The reach table from section 1.2, entered as literals. The + * implementation derives these from the layout; this asserts the two + * agree, which is the whole reason to derive rather than hard-code. */ + static const unsigned want_bits[4] = { 27, 22, 17, 12 }; + static const unsigned want_reach[4] = { + 0x8000000u, /* 2^27 words = whole address space */ + 0x400000u, /* 2^22 words = 4M words */ + 0x20000u, /* 2^17 words = 128K words */ + 0x1000u /* 2^12 words = 4K words */ + }; + + for (unsigned k = 0; k < 4; k++) { + CHECK(V4_BRANCH_ADDR_BITS(k) == want_bits[k], + "slot %u address bits: %u, spec says %u", k, + V4_BRANCH_ADDR_BITS(k), want_bits[k]); + CHECK(v4_iword_slot_mask(k) + 1u == want_reach[k], + "slot %u reach: %u words, spec table implies %u", k, + v4_iword_slot_mask(k) + 1u, want_reach[k]); + } +} + +static void test_branch_targets_take_all_low_bits(void) +{ + /* A target is every bit to the right of the slot, not just the low 5 or the + * low 12. Set a pattern of all-ones in the low region and confirm each + * slot recovers exactly its own low bits. */ + for (unsigned k = 0; k < 4; k++) { + v4_iword w = 0xFFFFFFFFu; /* every slot and spare bit set */ + uint32_t t = v4_iword_target(w, k); + uint32_t mask = v4_iword_slot_mask(k); + CHECK(t == mask, + "slot %u target from all-ones word is %08lX, expected %08lX", k, + (unsigned long)t, (unsigned long)mask); + + /* And the slot's own five bits must be excluded from its target. */ + unsigned op_in_slot = v4_iword_op(w, k); + CHECK(op_in_slot == 0x1Fu, "slot %u opcode from all-ones", k); + } +} + +static void test_branch_legality(void) +{ + for (unsigned k = 0; k < 4; k++) { + CHECK(v4_iword_branch_legal(k), "slot %u should allow branches", k); + } + for (unsigned k = 4; k < 6; k++) { + CHECK(!v4_iword_branch_legal(k), + "slot %u must not allow branches (section 1.2)", k); + } +} + +static void test_page_relative_branch(void) +{ + /* "That target replaces the same low bits of P (page-relative, as in the + * F18)." A branch inside one page must not disturb the bits above it, and + * must be able to reach every address in its page. */ + for (unsigned k = 0; k < 4; k++) { + uint32_t mask = v4_iword_slot_mask(k); + v4_ucell page = ~(v4_ucell)mask; /* a high value, low bits zero */ + v4_cell p = (v4_cell)page; + + for (uint32_t target = 0; target <= mask; target = target * 2u + 1u) { + v4_iword w = target; /* target lives in the low bits */ + v4_cell got = v4_iword_branch(p, w, k); + v4_ucell want = ((v4_ucell)page & ~(v4_ucell)mask) + | (v4_ucell)target; + CHECK((v4_ucell)got == want, + "slot %u: P=%lld target=%u gave %lld, expected %llu", k, + (long long)p, target, (long long)got, + (unsigned long long)want); + CHECK((v4_ucell)got == (v4_ucell)(page ^ target), + "slot %u: branch is not page-relative", k); + if (target > mask / 2u && target != mask) break; + } + } +} + +static void test_page_relative_branch_preserves_high_page(void) +{ + /* Same, from several distinct pages, to be sure nothing is hard-wired to + * page zero. */ + static const v4_ucell pages[] = { + (v4_ucell)0, + (v4_ucell)1, + (v4_ucell)0x80000000ull, + ~(v4_ucell)0 + }; + for (unsigned k = 0; k < 4; k++) { + uint32_t mask = v4_iword_slot_mask(k); + for (unsigned i = 0; i < sizeof(pages) / sizeof(pages[0]); i++) { + v4_cell p = (v4_cell)pages[i]; + v4_cell got = v4_iword_branch(p, 0u, k); + v4_ucell want = (pages[i] & ~(v4_ucell)mask); + CHECK((v4_ucell)got == want, + "slot %u: P=%llX with target 0 gave %llX, expected %llX", k, + (unsigned long long)pages[i], (unsigned long long)got, + (unsigned long long)want); + } + } +} + +static void test_branch_opcode_set(void) +{ + /* Exactly five opcodes take a target, and they are the five section 1.2 + * names. Everything else must not, so that a non-branch slot's low bits + * stay available to the compiler. */ + static const int want[32] = { + /* 0 */ 0, /* 1 */ 0, /* 2 jump */ 1, /* 3 call */ 1, /* 4 */ 0, + /* 5 next */ 1, /* 6 if */ 1, /* 7 -if */ 1, /* 8 */ 0, /* 9 */ 0, + /* A */ 0, /* B */ 0, /* C */ 0, /* D */ 0, /* E */ 0, /* F */ 0, + /* 10 */ 0, /* 11 */ 0, /* 12 */ 0, /* 13 */ 0, /* 14 */ 0, /* 15 */ 0, + /* 16 */ 0, /* 17 */ 0, /* 18 */ 0, /* 19 */ 0, /* 1A */ 0, + /* 1B */ 0, /* 1C */ 0, /* 1D */ 0, /* 1E */ 0, /* 1F */ 0 + }; + unsigned nbranch = 0; + for (unsigned op = 0; op < 32u; op++) { + int got = v4_op_is_branch(op); + CHECK(got == want[op], "opcode %02X: is_branch=%d, expected %d", op, + got, want[op]); + nbranch += (unsigned)got; + } + CHECK(nbranch == 5u, "expected exactly 5 branch opcodes, found %u", nbranch); +} + +static void test_opcode_enum_matches_spec_numbering(void) +{ + /* Each named opcode must carry the number section 1.3 gives it. Transposed + * or off-by-one enum entries are the classic way a 32-entry table rots. */ + CHECK(V4_OP_SEMI == 0x00, "opcode ;"); CHECK(V4_OP_EX == 0x01, "opcode ex"); + CHECK(V4_OP_JUMP == 0x02, "opcode jump"); CHECK(V4_OP_CALL == 0x03, "opcode call"); + CHECK(V4_OP_UNEXT == 0x04, "opcode unext"); CHECK(V4_OP_NEXT == 0x05, "opcode next"); + CHECK(V4_OP_IF == 0x06, "opcode if"); CHECK(V4_OP_MINUS_IF == 0x07, "opcode -if"); + CHECK(V4_OP_FETCH_P == 0x08, "opcode @p"); CHECK(V4_OP_FETCH_INC == 0x09, "opcode @+"); + CHECK(V4_OP_FETCH_B == 0x0A, "opcode @b"); CHECK(V4_OP_FETCH_A == 0x0B, "opcode @"); + CHECK(V4_OP_STORE_P == 0x0C, "opcode !p"); CHECK(V4_OP_STORE_INC == 0x0D, "opcode !+"); + CHECK(V4_OP_STORE_B == 0x0E, "opcode !b"); CHECK(V4_OP_STORE_A == 0x0F, "opcode !"); + CHECK(V4_OP_MUL_STEP == 0x10, "opcode +*"); CHECK(V4_OP_TWO_STAR == 0x11, "opcode 2*"); + CHECK(V4_OP_TWO_SLASH == 0x12, "opcode 2/"); CHECK(V4_OP_INV == 0x13, "opcode inv"); + CHECK(V4_OP_ADD == 0x14, "opcode +"); CHECK(V4_OP_AND == 0x15, "opcode and"); + CHECK(V4_OP_XOR == 0x16, "opcode xor"); CHECK(V4_OP_DROP == 0x17, "opcode drop"); + CHECK(V4_OP_DUP == 0x18, "opcode dup"); CHECK(V4_OP_RPOP == 0x19, "opcode pop"); + CHECK(V4_OP_OVER == 0x1A, "opcode over"); CHECK(V4_OP_PUSH_A == 0x1B, "opcode a"); + CHECK(V4_OP_NOP == 0x1C, "opcode nop"); CHECK(V4_OP_PUSH == 0x1D, "opcode push"); + CHECK(V4_OP_BANG_B == 0x1E, "opcode b!"); CHECK(V4_OP_BANG_A == 0x1F, "opcode a!"); + CHECK(V4_OPCODE_COUNT == 0x20, "opcode count"); +} + +static void test_ambiguous_f18_names_were_renamed(void) +{ + /* Section 1.3 says: "Opcode numbering follows the F18. Two names differ + * from Moore's: F18 `-` is renamed `inv` and F18 `or` (which is + * exclusive-or) is renamed `xor`, so instruction names never collide with + * FORTH-79 word names." + * + * The renames are the part that carries weight, and they are asserted here. + * `-` and `or` are exactly the dangerous case: Moore's F18 uses them with + * meanings that differ from the FORTH-79 words of the same name, so an + * instruction table containing either would be ambiguous rather than merely + * duplicated. `and` and `xor` are also FORTH-79 words, but `and` agrees + * with FORTH and `xor` is F18's spelling, so neither is ambiguous. + * + * NOTE: the trailing clause of that sentence is an overclaim, and the test + * deliberately does not assert it. `dup`, `drop`, `over`, `and`, `if` and + * `+` are all instruction names in section 1.3 and all FORTH-79 words, + * spelled identically and meaning the same thing. They coincide; they do + * not collide. What the renames prevent is a *divergent* meaning under a + * shared name, which is the property asserted above. Raised for the spec + * text rather than silently encoded here as if it were true. */ + static const char *opnames[32] = { + ";", "ex", "jump", "call", "unext", "next", "if", "-if", + "@p", "@+", "@b", "@", "!p", "!+", "!b", "!", + "+*", "2*", "2/", "inv", "+", "and", "xor", "drop", + "dup", "pop", "over", "a", "nop", "push", "b!", "a!" + }; + + CHECK(strcmp(opnames[V4_OP_INV], "inv") == 0, + "opcode 13 must be named inv, got \"%s\"", opnames[V4_OP_INV]); + CHECK(strcmp(opnames[V4_OP_XOR], "xor") == 0, + "opcode 16 must be named xor, got \"%s\"", opnames[V4_OP_XOR]); + + /* The two names the renames exist to remove must appear nowhere. */ + for (unsigned i = 0; i < 32u; i++) { + CHECK(strcmp(opnames[i], "-") != 0, + "F18 `-` must not survive as an instruction name (slot %u)", i); + CHECK(strcmp(opnames[i], "or") != 0, + "F18 `or` must not survive as an instruction name (slot %u)", i); + CHECK(strcmp(opnames[i], "OR") != 0, + "F18 `or` must not survive as an instruction name (slot %u)", i); + } + + /* The five branch opcodes, by the names section 1.2 lists. */ + CHECK(strcmp(opnames[V4_OP_JUMP], "jump") == 0, "opcode 2 name"); + CHECK(strcmp(opnames[V4_OP_CALL], "call") == 0, "opcode 3 name"); + CHECK(strcmp(opnames[V4_OP_NEXT], "next") == 0, "opcode 5 name"); + CHECK(strcmp(opnames[V4_OP_IF], "if") == 0, "opcode 6 name"); + CHECK(strcmp(opnames[V4_OP_MINUS_IF], "-if") == 0, "opcode 7 name"); +} + +static void test_assemble_rejects_out_of_range(void) +{ + unsigned o[6] = { 0, 0, 0, 0, 0, 0 }; + o[3] = 32u; /* will not fit in 5 bits */ + CHECK(v4_iword_assemble(o, 0u) == 0xFFFFFFFFu, + "assemble must reject a 6-bit opcode"); + o[3] = 0; + CHECK(v4_iword_assemble(o, 4u) == 0xFFFFFFFFu, + "assemble must reject spare bits above 3"); +} + +static void test_out_of_range_slots_are_inert(void) +{ + /* Slots past 5 do not exist. The decode must answer deterministically + * rather than shift by a nonsense amount. */ + v4_iword w = 0xFFFFFFFFu; + CHECK(v4_iword_op(w, 6u) == 0u, "op in slot 6 must be 0"); + CHECK(v4_iword_target(w, 6u) == 0u, "target in slot 6 must be 0"); + CHECK(v4_iword_slot_mask(6u) == 0u, "mask in slot 6 must be 0"); + CHECK(!v4_iword_branch_legal(6u), "slot 6 must not allow branches"); +} + +int main(void) +{ + printf("v4 iword tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + test_slot_bit_ranges(); + test_roundtrip_all_opcodes(); + test_slots_are_independent(); + test_spare_bits_are_inert_for_decode(); + test_spare_bits_are_branch_address_bits(); + test_reach_column(); + test_branch_targets_take_all_low_bits(); + test_branch_legality(); + test_page_relative_branch(); + test_page_relative_branch_preserves_high_page(); + test_branch_opcode_set(); + test_opcode_enum_matches_spec_numbering(); + test_ambiguous_f18_names_were_renamed(); + test_assemble_rejects_out_of_range(); + test_out_of_range_slots_are_inert(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_node.c b/v4/tests/test_node.c new file mode 100644 index 00000000..6a750856 --- /dev/null +++ b/v4/tests/test_node.c @@ -0,0 +1,266 @@ +/* test_node.c -- the node: registers, memory, and the boundary on memory. + * + * The memory boundary does a job the stack boundaries do not: it is the landing + * place for a *linear* index error -- mem[-1], mem[V4_NODE_WORDS] -- which is + * inside the struct and therefore invisible to AddressSanitizer. + * + * It explicitly does NOT cover an out-of-range *word address*, where P, A or B + * has left the address space. That lands gigabytes away, outside any band. + * DECOMPOSITION.md does not say what such an access should do, so no policy is + * invented here; instead the limitation is measured and asserted, so the claim + * in node.h cannot rot. Closing that hole is a ruling on out-of-range + * addressing, not a wider band. + */ +#include "v4/node.h" + +#include +#include +#include +#include +#include + +static int failures = 0; +static int checks = 0; + +#define CHECK(cond, ...) \ + do { \ + checks++; \ + if (!(cond)) { \ + failures++; \ + printf(" FAIL %s:%d: ", __FILE__, __LINE__); \ + printf(__VA_ARGS__); \ + printf("\n"); \ + } \ + } while (0) + +static void test_geometry(void) +{ + /* The boundary is 5% of memory at each end, rounded up, never zero. */ + CHECK(V4_MEM_BOUND == V4_GUARD_ELEMS(V4_NODE_WORDS), + "memory boundary is not 5%% of %u words", V4_NODE_WORDS); + CHECK(V4_MEM_BOUND >= 1u, "memory boundary must be at least one word"); + CHECK((unsigned long)V4_MEM_BOUND * 100u + >= (unsigned long)V4_NODE_WORDS * V4_GUARD_PCT, + "memory boundary %u is less than %u%% of %u", V4_MEM_BOUND, + V4_GUARD_PCT, V4_NODE_WORDS); + + /* At the default size that is a real cost, and naming it here means the + * number is on the record rather than discovered from a memory report. */ + printf(" node: %u words + %u head + %u tail boundary " + "(%.1f%% overhead)\n", + V4_NODE_WORDS, V4_MEM_BOUND, V4_MEM_BOUND, + 100.0 * 2.0 * (double)V4_MEM_BOUND / (double)V4_NODE_WORDS); +} + +static void test_reset_state(void) +{ + v4_node n; + v4_node_reset(&n); + + CHECK(n.p == 0, "P after reset"); + CHECK(n.a == 0, "A after reset"); + CHECK(n.b == 0, "B after reset"); + CHECK(v4_node_guards_intact(&n), "all boundaries intact after reset"); + + /* T, S and R live in the stacks, so they are checked through the stack + * API rather than as node fields. */ + CHECK(v4_dstack_peek(&n.ds) == 0, "T after reset"); + CHECK(v4_dstack_peek2(&n.ds) == 0, "S after reset"); + CHECK(v4_rstack_peek(&n.rs) == 0, "R after reset"); +} + +static void test_guards_absent_before_reset(void) +{ + /* A node that has never been reset holds whatever was on the stack. The + * check must notice, which is what proves it reads the boundary rather than + * returning a constant. */ + v4_node *n = (v4_node *)malloc(sizeof *n); + if (n == NULL) { + failures++; + printf(" FAIL %s:%d: out of memory\n", __FILE__, __LINE__); + return; + } + memset(n, 0xA5, sizeof *n); + CHECK(!v4_node_guards_intact(n), + "boundaries must not read as intact before reset"); + v4_node_reset(n); + CHECK(v4_node_guards_intact(n), "boundaries intact after reset"); + free(n); +} + +static void test_load_store_roundtrip(void) +{ + v4_node n; + v4_node_reset(&n); + + /* Every word, not a sample: a stray boundary in the middle of memory would + * not be found by spot checks. */ + for (v4_cell i = 0; i < (v4_cell)V4_NODE_WORDS; i++) { + v4_node_store(&n, i, i * 3 + 1); + } + for (v4_cell i = 0; i < (v4_cell)V4_NODE_WORDS; i++) { + CHECK(v4_node_load(&n, i) == i * 3 + 1, + "word %lld: got %lld want %lld", (long long)i, + (long long)v4_node_load(&n, i), (long long)(i * 3 + 1)); + } + CHECK(v4_node_guards_intact(&n), "memory boundary survived a full sweep"); +} + +static void test_load_store_edges(void) +{ + /* The first and last words, which are the ones adjacent to the boundary. + * A boundary that is a word too wide would quietly steal word 0 or the + * last word, and the memory would still pass every interior test. */ + v4_node n; + v4_node_reset(&n); + + v4_node_store(&n, 0, 0x11111111); + v4_node_store(&n, (v4_cell)V4_NODE_WORDS - 1, 0x22222222); + CHECK(v4_node_load(&n, 0) == 0x11111111, "word 0 did not take its value"); + CHECK(v4_node_load(&n, (v4_cell)V4_NODE_WORDS - 1) == 0x22222222, + "last word did not take its value"); + CHECK(v4_node_guards_intact(&n), "edge writes disturbed the boundary"); +} + +static void test_boundary_is_adjacent_to_memory(void) +{ + /* The band only catches a linear index error if it is immediately next to + * mem, so adjacency is asserted rather than assumed -- the compiler is free + * to reorder struct members, and a band that drifted to the far end of the + * struct would silently stop catching anything. */ + CHECK(offsetof(v4_node, mem_guard_tail) + == offsetof(v4_node, mem) + sizeof(((v4_node *)0)->mem), + "tail boundary does not start immediately after memory"); + CHECK(offsetof(v4_node, mem_guard_head) + + sizeof(((v4_node *)0)->mem_guard_head) + == offsetof(v4_node, mem), + "head boundary does not end immediately before memory"); +} + +static void test_linear_overrun_low_is_caught(void) +{ + /* A linear index error -- the kind an off-by-one in a loop bound or a + * pointer step produces. It lands in the head boundary, it is inside the + * struct so AddressSanitizer says nothing about it, and the boundary check + * is what reports it. Written through the boundary array rather than as + * mem[-1], because mem[-1] is out of bounds of a declared array and the + * standard lets the compiler do anything with it; the offsetof test above + * establishes that this *is* the word mem[-1] resolves to. */ + v4_node n; + v4_node_reset(&n); + CHECK(v4_node_guards_intact(&n), "intact before the deliberate overrun"); + + n.mem_guard_head[V4_MEM_BOUND - 1u] = (v4_cell)0xDEADBEEF; + CHECK(!v4_node_guards_intact(&n), + "a linear access before mem was not detected"); + + v4_node_reset(&n); + CHECK(v4_node_guards_intact(&n), "intact after re-reset"); +} + +static void test_linear_overrun_high_is_caught(void) +{ + v4_node n; + v4_node_reset(&n); + + n.mem_guard_tail[0] = (v4_cell)0xDEADBEEF; + CHECK(!v4_node_guards_intact(&n), + "a linear access past the last word was not detected"); + + v4_node_reset(&n); + CHECK(v4_node_guards_intact(&n), "intact after re-reset"); +} + +static void test_out_of_range_word_address_is_not_covered_by_the_band(void) +{ + /* What the band does NOT catch, asserted so the limitation stays visible + * and cannot be quietly forgotten. + * + * An out-of-range *word address* is a different animal from a linear index + * error. A v4_cell address of -1 becomes unsigned 0xFFFFFFFF, which is + * 2^32 words past mem -- 16 GB at 32-bit cells. That is far outside the + * struct, so no boundary can see it. The band was briefly documented as + * catching this; it does not, and the protection is a range check whose + * policy DECOMPOSITION.md does not define. This test records the gap by + * measuring it, so the number in node.h stays true. */ + v4_cell neg = (v4_cell)-1; + uint64_t words_out = (uint64_t)(v4_ucell)neg; + uint64_t far = words_out * (uint64_t)sizeof(v4_cell); + printf(" out-of-range word address -1 lands %llu bytes past mem " + "(%.1f GB at this width): outside the band, as documented\n", + (unsigned long long)far, (double)far / 1073741824.0); + CHECK(far > (uint64_t)sizeof(v4_node), + "an out-of-range address should land outside the node"); + + /* The band is only V4_MEM_BOUND words, so even a modest overrun is caught + * only while it is within the band. Past that it is not, which is why the + * open question is a range check and not a wider band. */ + CHECK(V4_MEM_BOUND < V4_NODE_WORDS / 2u, + "if the band were most of memory, this reasoning would be wrong"); +} + +static void test_ends_are_distinguishable(void) +{ + /* The two boundaries carry different patterns precisely so that a failure + * says which end. Checked here on memory as well as in test_guard.c on the + * ring, because "which end" is the property that makes a report actionable. */ + v4_node n; + v4_node_reset(&n); + CHECK((v4_ucell)n.mem_guard_head[0] == V4_GUARD_PATTERN_HEAD, + "memory head boundary pattern"); + CHECK((v4_ucell)n.mem_guard_tail[0] == V4_GUARD_PATTERN_TAIL, + "memory tail boundary pattern"); + CHECK(V4_GUARD_PATTERN_HEAD != V4_GUARD_PATTERN_TAIL, + "head and tail patterns are equal at V4_CELL_BITS=%d", V4_CELL_BITS); +} + +static void test_workload_never_false_alarms(void) +{ + /* A boundary that fires on legitimate use is worse than no boundary. Drive + * a workload across the whole address space, through both address registers + * and the program counter, and require every boundary to survive. */ + v4_node n; + v4_node_reset(&n); + + uint64_t s = 0xB5026F5AA96619E9ull; + for (int i = 0; i < 200000; i++) { + s ^= s << 13; s ^= s >> 7; s ^= s << 17; + + v4_cell addr = (v4_cell)(s % (uint64_t)V4_NODE_WORDS); + v4_node_store(&n, addr, (v4_cell)s); + + n.a = addr; + n.b = (v4_cell)((addr + 1u) % V4_NODE_WORDS); + n.p = (v4_cell)((addr + 2u) % V4_NODE_WORDS); + + v4_dstack_push(&n.ds, (v4_cell)s); + v4_rstack_push(&n.rs, (v4_cell)(s >> 13)); + if (i & 1) { (void)v4_dstack_pop(&n.ds); (void)v4_rstack_pop(&n.rs); } + } + CHECK(v4_node_guards_intact(&n), + "a boundary fired during ordinary workload"); + CHECK(n.p < (v4_cell)V4_NODE_WORDS, "P left the address space"); + CHECK(n.a < (v4_cell)V4_NODE_WORDS, "A left the address space"); + CHECK(n.b < (v4_cell)V4_NODE_WORDS, "B left the address space"); +} + +int main(void) +{ + printf("v4 node tests: V4_CELL_BITS=%d, %u words\n", + V4_CELL_BITS, V4_NODE_WORDS); + + test_geometry(); + test_reset_state(); + test_guards_absent_before_reset(); + test_load_store_roundtrip(); + test_load_store_edges(); + test_boundary_is_adjacent_to_memory(); + test_linear_overrun_low_is_caught(); + test_linear_overrun_high_is_caught(); + test_out_of_range_word_address_is_not_covered_by_the_band(); + test_ends_are_distinguishable(); + test_workload_never_false_alarms(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_stack.c b/v4/tests/test_stack.c new file mode 100644 index 00000000..ba166f10 --- /dev/null +++ b/v4/tests/test_stack.c @@ -0,0 +1,338 @@ +/* test_stack.c -- the F18 circular stacks, checked against an independent + * model of the same semantics. + * + * The implementation in stack.c is specified in DECOMPOSITION.md as + * "T, S + 8 circular" / "R + 8 circular". This test does not take that + * decomposition on trust. It builds a second, deliberately dumb model of + * what D-2 describes -- a flat fixed-depth circular buffer with no bounds + * check at all -- and drives both with identical operation sequences, + * requiring them to agree after every single operation. If the register/ring + * bookkeeping in stack.c has a wrong pointer direction or an off-by-one in + * the wrap, this catches it; an inspection-only check would not. + * + * Depth, ordering, and the "silently overwrites the oldest entry" behaviour + * are additionally pinned with explicit cases, because those are the three + * properties the ISA's push-heavy words actually depend on. + * + * Built and run at both V4_CELL_BITS=32 and 64; see v4/Makefile. + */ +#include "v4/stack.h" + +#include +#include + +static int failures = 0; +static int checks = 0; + +#define CHECK(cond, ...) \ + do { \ + checks++; \ + if (!(cond)) { \ + failures++; \ + printf(" FAIL %s:%d: ", __FILE__, __LINE__); \ + printf(__VA_ARGS__); \ + printf("\n"); \ + } \ + } while (0) + +/* ---- the independent reference model --------------------------------------- + * A flat circular buffer of depth N with no overflow or underflow detection, + * which is exactly what D-2 describes. top indexes the newest element. */ + +typedef struct { + v4_cell buf[64]; + unsigned top; + unsigned depth; +} flat_t; + +static void flat_reset(flat_t *f, unsigned depth) +{ + f->depth = depth; + f->top = 0; + for (unsigned i = 0; i < 64; i++) f->buf[i] = 0; +} + +static void flat_push(flat_t *f, v4_cell x) +{ + f->top = (f->top + 1u) % f->depth; + f->buf[f->top] = x; +} + +static v4_cell flat_pop(flat_t *f) +{ + v4_cell x = f->buf[f->top]; + f->top = (f->top + f->depth - 1u) % f->depth; + return x; +} + +static v4_cell flat_peek(const flat_t *f) +{ + return f->buf[f->top]; +} + +/* ---- deterministic PRNG --------------------------------------------------- + * xorshift64, so a failure is reproducible from the seed alone. No rand(), + * whose sequence is implementation-defined and would make a failure on one + * host unreproducible on another. */ + +static uint64_t rng_state = 0x9E3779B97F4A7C15ull; + +static uint64_t rng_next(void) +{ + uint64_t x = rng_state; + x ^= x << 13; + x ^= x >> 7; + x ^= x << 17; + rng_state = x; + return x; +} + +/* ---- tests ---------------------------------------------------------------- */ + +static void test_reset_is_zero(void) +{ + v4_dstack d; + v4_rstack r; + v4_dstack_reset(&d); + v4_rstack_reset(&r); + CHECK(v4_dstack_peek(&d) == 0, "data stack peek after reset != 0"); + CHECK(v4_dstack_peek2(&d) == 0, "data stack peek2 after reset != 0"); + CHECK(v4_rstack_peek(&r) == 0, "return stack peek after reset != 0"); +} + +static void test_exact_depth_data(void) +{ + /* Fill to exactly V4_DATA_DEPTH, then drain and confirm the order. */ + v4_dstack d; + flat_t f; + v4_dstack_reset(&d); + flat_reset(&f, V4_DATA_DEPTH); + + for (unsigned i = 0; i < V4_DATA_DEPTH; i++) { + v4_cell v = (v4_cell)(i + 1); + v4_dstack_push(&d, v); + flat_push(&f, v); + } + CHECK(v4_dstack_peek(&d) == (v4_cell)V4_DATA_DEPTH, + "peek at full depth should be the last pushed value (%d)", + (int)V4_DATA_DEPTH); + + for (unsigned i = V4_DATA_DEPTH; i > 0; i--) { + v4_cell got = v4_dstack_pop(&d); + v4_cell exp = flat_pop(&f); + CHECK(got == exp, "drain at depth %u: got %lld want %lld", i, + (long long)got, (long long)exp); + CHECK(got == (v4_cell)i, "drain at depth %u: got %lld want %d", i, + (long long)got, (int)i); + } +} + +static void test_overflow_overwrites_oldest(void) +{ + /* D-2: "pushing past the bottom silently overwrites the oldest entry." + * Push one past depth: the very first value pushed must be gone, and the + * remaining V4_DATA_DEPTH-1 must come back newest-first. */ + v4_dstack d; + flat_t f; + v4_dstack_reset(&d); + flat_reset(&f, V4_DATA_DEPTH); + + for (unsigned i = 0; i < V4_DATA_DEPTH + 1u; i++) { + v4_cell v = (v4_cell)(i + 1); + v4_dstack_push(&d, v); + flat_push(&f, v); + } + CHECK(v4_dstack_peek(&d) == (v4_cell)(V4_DATA_DEPTH + 1u), + "peek after overflow should be the newest value"); + + for (unsigned i = 0; i < V4_DATA_DEPTH; i++) { + v4_cell got = v4_dstack_pop(&d); + v4_cell exp = flat_pop(&f); + CHECK(got == exp, "post-overflow drain %u: got %lld want %lld", i, + (long long)got, (long long)exp); + CHECK(got != 1, + "post-overflow drain %u: value 1 should have been overwritten", + i); + } +} + +static void test_exact_depth_return(void) +{ + v4_rstack r; + flat_t f; + v4_rstack_reset(&r); + flat_reset(&f, V4_RET_DEPTH); + + for (unsigned i = 0; i < V4_RET_DEPTH + 3u; i++) { + v4_cell v = (v4_cell)(i + 1); + v4_rstack_push(&r, v); + flat_push(&f, v); + } + for (unsigned i = 0; i < V4_RET_DEPTH; i++) { + v4_cell got = v4_rstack_pop(&r); + v4_cell exp = flat_pop(&f); + CHECK(got == exp, "return drain %u: got %lld want %lld", i, + (long long)got, (long long)exp); + } +} + +static void test_underflow_wraps_rather_than_trapping(void) +{ + /* D-2 says there is no underflow detection. Popping an "empty" stack must + * therefore return a defined (stale) value, not fault and not abort. The + * golden model must diverge from hardware in neither direction, so this is + * asserted rather than left to chance. */ + v4_dstack d; + flat_t f; + v4_dstack_reset(&d); + flat_reset(&f, V4_DATA_DEPTH); + + for (unsigned i = 0; i < 5; i++) { + v4_dstack_pop(&d); + flat_pop(&f); + } + CHECK(1, "popping past empty must not trap"); +} + +/* ---- live-depth-aware comparison ------------------------------------------ + * D-2 specifies what these stacks do while they hold live entries: LIFO order + * within the depth, and oldest-entry-overwritten past it. It explicitly does + * NOT specify the contents once the stack has been popped empty, because + * "no overflow or underflow" means the residue is whatever the physical + * register file happened to hold. Two independent models of a circular buffer + * will legitimately disagree down there -- it is not a semantic difference. + * + * So the differential test tracks how many live entries each stack holds and + * asserts agreement only where the spec makes a claim: the value a pop returns + * and the top peek while depth > 0, and the second element while depth > 1. + * The stale region is still exercised (the sequence runs right through it) and + * is still required not to trap, it is simply not required to agree. */ + +typedef struct { + v4_dstack hw; + flat_t model; + int live; +} pair_t; + +static void pair_reset(pair_t *p) +{ + v4_dstack_reset(&p->hw); + flat_reset(&p->model, V4_DATA_DEPTH); + p->live = 0; +} + +static void pair_step(pair_t *p, int push, v4_cell v, int step, const char *tag) +{ + if (push) { + v4_dstack_push(&p->hw, v); + flat_push(&p->model, v); + if (p->live < V4_DATA_DEPTH) p->live++; + } else { + v4_cell got = v4_dstack_pop(&p->hw); + v4_cell exp = flat_pop(&p->model); + if (p->live > 0) { + CHECK(got == exp, "%s op %d: pop got %lld want %lld (live=%d)", + tag, step, (long long)got, (long long)exp, p->live); + p->live--; + } + } + if (p->live > 0) { + CHECK(v4_dstack_peek(&p->hw) == flat_peek(&p->model), + "%s op %d: peek mismatch (live=%d)", tag, step, p->live); + } + if (p->live > 1) { + /* Second element of a depth-N circular buffer whose top is at `top` is + * N-1 further along the fill direction, i.e. (top-1) mod N. */ + v4_cell want = p->model.buf[(p->model.top + p->model.depth - 1u) + % p->model.depth]; + CHECK(v4_dstack_peek2(&p->hw) == want, + "%s op %d: peek2 got %lld want %lld (live=%d)", tag, step, + (long long)v4_dstack_peek2(&p->hw), (long long)want, p->live); + } +} + +static void test_exhaustive_sequences(void) +{ + /* Every push/pop sequence up to length 10 -- 2046 of them -- driven through + * both models. Exhaustive over short sequences rather than random, because + * a wrap-direction bug shows up in a handful of specific short patterns + * (notably push x N+1 then pop x N, which is the only sequence that ever + * overwrites the oldest entry) and a random driver finds those only by + * luck. This finds all of them, every run, deterministically. */ + enum { MAXLEN = 10 }; + int total = 0; + for (int len = 1; len <= MAXLEN; len++) total += 1 << len; + + for (int len = 1; len <= MAXLEN; len++) { + for (int bits = 0; bits < (1 << len); bits++) { + pair_t p; + pair_reset(&p); + for (int i = 0; i < len; i++) { + int push = (bits >> i) & 1; + /* Distinct, non-zero values so a slot mix-up is visible. */ + v4_cell v = (v4_cell)(100 + i * 7); + pair_step(&p, push, v, i, "exhaustive"); + } + } + } + printf(" exhaustive: %d sequences of length <= %d\n", total, MAXLEN); +} + +static void test_differential_random(void) +{ + /* Long random walk across the wrap points, both widths, both stacks. */ + enum { OPS = 200000 }; + pair_t dp; + v4_rstack r; + flat_t rf; + int rlive; + + pair_reset(&dp); + v4_rstack_reset(&r); + flat_reset(&rf, V4_RET_DEPTH); + rlive = 0; + + for (int i = 0; i < OPS; i++) { + uint64_t bits = rng_next(); + + pair_step(&dp, (int)(bits & 1u), (v4_cell)bits, i, "differential data"); + + if (bits & 2u) { + v4_cell v = (v4_cell)(bits >> 8); + v4_rstack_push(&r, v); + flat_push(&rf, v); + if (rlive < V4_RET_DEPTH) rlive++; + } else { + v4_cell got = v4_rstack_pop(&r); + v4_cell exp = flat_pop(&rf); + if (rlive > 0) { + CHECK(got == exp, + "differential ret op %d: pop got %lld want %lld (live=%d)", + i, (long long)got, (long long)exp, rlive); + rlive--; + } + } + if (rlive > 0) { + CHECK(v4_rstack_peek(&r) == flat_peek(&rf), + "differential ret op %d: peek mismatch (live=%d)", i, rlive); + } + } + printf(" differential: %d ops\n", OPS); +} + +int main(void) +{ + printf("v4 stack tests: V4_CELL_BITS=%d, data depth %d, return depth %d\n", + V4_CELL_BITS, V4_DATA_DEPTH, V4_RET_DEPTH); + + test_reset_is_zero(); + test_exact_depth_data(); + test_overflow_overwrites_oldest(); + test_exact_depth_return(); + test_underflow_wraps_rather_than_trapping(); + test_exhaustive_sequences(); + test_differential_random(); + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +} diff --git a/v4/tests/test_umul.c b/v4/tests/test_umul.c new file mode 100644 index 00000000..2cec6aa4 --- /dev/null +++ b/v4/tests/test_umul.c @@ -0,0 +1,70 @@ +/* test_umul.c -- the reference unsigned multiply. + * + * v4_umul is what the capsule-level UM* is checked against, so it is checked + * here against things that do not depend on it: algebraic identities that hold + * at any width, and at 32-bit cells the host's own 64-bit product. + */ +#include "v4/umul.h" +#include + +static int failures = 0, checks = 0; +#define CHECK(c,...) do{checks++; if(!(c)){failures++; printf("FAIL %s:%d: ",__FILE__,__LINE__); printf(__VA_ARGS__); printf("\n");}}while(0) + +#define MAXU ((v4_ucell)~(v4_ucell)0) + +static const v4_ucell vec[] = { + 0u, 1u, 2u, 3u, 7u, 255u, 256u, 65535u, 65536u, 0x12345678u, + (v4_ucell)0x9E3779B97F4A7C15ULL, (v4_ucell)0xDEADBEEFCAFEF00DULL, + V4_MSB - 1u, V4_MSB, V4_MSB + 1u, MAXU - 1u, MAXU +}; +#define NVEC (sizeof vec / sizeof vec[0]) + +int main(void) +{ + v4_ucell lo, hi, lo2, hi2; + + printf("v4 umul tests: V4_CELL_BITS=%d\n", V4_CELL_BITS); + + v4_umul(3, 7, &lo, &hi); CHECK(lo == 21 && hi == 0, "3*7"); + /* (2^n - 1)^2 = 2^2n - 2^(n+1) + 1 */ + v4_umul(MAXU, MAXU, &lo, &hi); CHECK(lo == 1 && hi == MAXU - 1u, "max*max"); + /* 2^(n-1) * 2^(n-1) = 2^(2n-2) */ + v4_umul(V4_MSB, V4_MSB, &lo, &hi); + CHECK(lo == 0 && hi == (V4_MSB >> 1), "msb*msb"); + + for (unsigned i = 0; i < NVEC; i++) { + v4_ucell x = vec[i]; + + v4_umul(x, 0, &lo, &hi); CHECK(lo == 0 && hi == 0, "x*0 [%u]", i); + v4_umul(x, 1, &lo, &hi); CHECK(lo == x && hi == 0, "x*1 [%u]", i); + /* x * (2^n - 1) = x*2^n - x */ + v4_umul(x, MAXU, &lo, &hi); + CHECK(lo == (v4_ucell)(0u - x) && hi == (x ? x - 1u : 0u), "x*max [%u]", i); + + for (unsigned k = 0; k < V4_CELL_BITS; k++) { + /* x * 2^k is x shifted across the double cell */ + v4_umul(x, ((v4_ucell)1) << k, &lo, &hi); + CHECK(lo == (v4_ucell)(x << k), "x<<%u lo [%u]", k, i); + CHECK(hi == (k ? x >> (V4_CELL_BITS - k) : 0u), "x<<%u hi [%u]", k, i); + } + + for (unsigned j = 0; j < NVEC; j++) { + v4_ucell y = vec[j]; + + v4_umul(x, y, &lo, &hi); + v4_umul(y, x, &lo2, &hi2); + CHECK(lo == lo2 && hi == hi2, "commutes [%u,%u]", i, j); + CHECK(lo == (v4_ucell)(x * y), "low cell [%u,%u]", i, j); +#if V4_CELL_BITS == 32 + { + uint64_t p = (uint64_t)x * (uint64_t)y; + CHECK(lo == (uint32_t)p && hi == (uint32_t)(p >> 32), + "host product [%u,%u]", i, j); + } +#endif + } + } + + printf(" %d checks, %d failures\n", checks, failures); + return failures ? 1 : 0; +}