diff --git a/minzc/pkg/hir/exhaustive_judge_test.go b/minzc/pkg/hir/exhaustive_judge_test.go index 490dc5e4..1b011e53 100644 --- a/minzc/pkg/hir/exhaustive_judge_test.go +++ b/minzc/pkg/hir/exhaustive_judge_test.go @@ -317,9 +317,10 @@ func TestExhaustiveJudgeLIRSingleBlock(t *testing.T) { } // Word fallback judges assemble once and exhaust the u16 domain for >>3; -// add32 checks fallback provenance and MIR2 semantics: the current PBQP u32 -// ABI spills to memory and produces unassemblable code, so it cannot yet be -// judged on Z80. Keep that limitation explicit instead of altering the ABI. +// This judge writes memory arguments directly to callee spill slots; it does +// not test the caller/callee memory-parameter ABI (which currently mismatches). +// add32 checks 65,636 sums on MIR2 and Z80 arithmetic, including +// carries between main and shadow register banks. func TestExhaustiveJudgeLIRWideFallback(t *testing.T) { for _, name := range []string{"shr16", "add32"} { t.Run(name, func(t *testing.T) { @@ -368,27 +369,63 @@ func TestExhaustiveJudgeLIRWideFallback(t *testing.T) { } } } - // PBQP cannot assemble this production u32 ABI yet. A future - // repair must replace this source-VM check with a Z80 judge. - res, err := z80asm.NewAssembler().AssembleString(steps.Assembly) + // The production ABI now assembles; exercise both register banks + // and named memory parameters against the MIR2 boundary oracle. + var boot strings.Builder + fmt.Fprintf(&boot, "ORG 0x%04X\nLD SP,0xFF00\n", testLoadAddr) + var inputs [2]string + for i, p := range mf.Contract.Params { + loc := steps.Allocation.Locs[p.Reg] + if loc.Kind == mir2.LocMem { + inputs[i] = fmt.Sprintf("_spill_%s_r%d", name, p.Reg) + } else { + inputs[i] = fmt.Sprintf("judge_arg%d", i) + fmt.Fprintf(&boot, "LD %s,(%s)\nEXX\nLD %s,(%s+2)\nEXX\n", loc.Name, inputs[i], loc.Name, inputs[i]) + } + } + fmt.Fprintf(&boot, "CALL %s\nPUSH HL\nEXX\nPUSH HL\nEXX\nPOP BC\nPOP HL\nDI\nHALT\njudge_arg0: DB 0,0,0,0\njudge_arg1: DB 0,0,0,0\n", name) + res, err := z80asm.NewAssembler().AssembleString(boot.String() + steps.Assembly) if err != nil || len(res.Errors) > 0 { - plainOpts := pipeline.DefaultOptions() - plain, plainErr := pipeline.CompileHIRSteps(&hir.Module{Name: "judge_wide", Funcs: []*hir.Func{f}}, plainOpts) - if plainErr != nil { - t.Fatal(plainErr) + t.Fatalf("assemble: %v %v", err, res.Errors) + } + z := emulator.NewRemogattoZ80() + check := func(a, b int64) { + t.Helper() + z.Reset() + z.LoadMemory(testLoadAddr, res.Binary) + for i, v := range []int64{a, b} { + addr, ok := res.Symbols[inputs[i]] + if !ok { + t.Fatalf("undefined parameter slot %s", inputs[i]) + } + for j := 0; j < 4; j++ { + z.SetMemory(uint16(addr+j), byte(uint32(v)>>uint(j*8))) + } + } + z.SetRegisters(emulator.Registers{SP: 0xFF00, PC: testLoadAddr}) + for n := 0; !z.IsHalted(); n++ { + if n >= judgeStepBudget { + t.Fatal("no HALT") + } + z.Step() + } + r := z.GetRegisters() + got := uint32(r.HL) | uint32(r.BC)<<16 + if got != uint32(a+b) { + t.Fatalf("Z80 add32(%x,%x)=%x want %x", a, b, got, uint32(a+b)) } - plainRes, plainErr := z80asm.NewAssembler().AssembleString(plain.Assembly) - if len(plainRes.Errors) == 0 || len(plainRes.Errors) != len(res.Errors) { - t.Fatalf("known PBQP u32 record requires equal nonzero assembly error counts: --lir %d, plain %d (%v)", len(res.Errors), len(plainRes.Errors), plainErr) + } + for a := int64(0); a < 256; a++ { + for b := int64(0); b < 256; b++ { + check(a, b) } - // Re-measured on deterministic origin/main ed55c1c7. - const knownAssemblyErrors = 18 // 2026-10-02, both modes - if len(plainRes.Errors) != knownAssemblyErrors { - t.Fatalf("known PBQP u32 assembly error count changed: got %d, recorded %d; re-measure both modes", len(plainRes.Errors), knownAssemblyErrors) + } + for _, a := range boundaries { + for _, b := range boundaries { + check(a, b) } - t.Skipf("2026-10-02: 65,636 MIR2 sums checked; known PBQP u32 Z80 assembly errors: --lir %d, plain %d: %v %v", len(res.Errors), len(plainRes.Errors), plainErr, plainRes.Errors) } - t.Fatal("production u32 now assembles: replace this skip with a real judge") + return } boot := fmt.Sprintf(" ORG 0x%04X\n CALL %s\n DI\n HALT\n", testLoadAddr, name) res, err := z80asm.NewAssembler().AssembleString(boot + steps.Assembly) diff --git a/minzc/pkg/hir/p8_split_test.go b/minzc/pkg/hir/p8_split_test.go new file mode 100644 index 00000000..65f93aa2 --- /dev/null +++ b/minzc/pkg/hir/p8_split_test.go @@ -0,0 +1,103 @@ +package hir + +import ( + "fmt" + "github.com/minz/minzc/pkg/emulator" + "github.com/minz/minzc/pkg/mir2" + "github.com/minz/minzc/pkg/z80asm" + "testing" +) + +func TestP8SplitReturnDependency(t *testing.T) { + f := &Func{Name: "compute", RetTy: mir2.TyU16, Params: []Param{{Name: "a", Ty: mir2.TyU16}}, Body: &Block{Body: []Stmt{ + &VarDeclStmt{Name: "x", Ty: mir2.TyU16, Init: &VarRefExpr{Name: "a", Ty: mir2.TyU16}}, + &VarDeclStmt{Name: "y", Ty: mir2.TyU16, Init: &IntLitExpr{Val: 7, Ty: mir2.TyU16}}, + &VarDeclStmt{Name: "z", Ty: mir2.TyU16, Init: &IntLitExpr{Val: 8, Ty: mir2.TyU16}}, + &ReturnStmt{Val: &VarRefExpr{Name: "x", Ty: mir2.TyU16}}, + }}} + m := &Module{Name: "p8", Funcs: []*Func{f}} + candidates := FindSplitPoints(f, []int{9, 9, 9, 9}) + if len(candidates) == 0 { + t.Fatal("no split") + } + sub := ApplySplit(m, f, candidates[0]) + m.Funcs = append(m.Funcs, sub) + mm := LowerModule(m) + if mm.FuncByName(sub.Name) == nil { + t.Fatal("split callee omitted by free-variable detection") + } + result, err := mir2.NewVM(mm).Call("compute", []mir2.Value{{I: 1234}}) + if err != nil || len(result) != 1 || result[0].I != 1234 { + t.Fatalf("return lost across split: %v %v", result, err) + } + mm.RenumberRegs() + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{}} + for _, mf := range mm.Funcs { + a := mir2.PBQPAllocate(mf, mir2.ComputeLiveness(mf), mir2.Z80CostTable{}) + for r, l := range a.Locs { + ar.Locs[r] = l + } + ar.Spilled = append(ar.Spilled, a.Spilled...) + } + param := mm.FuncByName("compute").Contract.Params[0].Reg + // Use distinct parameter locations so this tests the forwarding caller + // rather than the codegen's identity-function EQU alias optimization. + ar.Locs[param] = mir2.PhysLoc{Kind: mir2.LocReg, Name: "BC"} + asm := fmt.Sprintf("ORG 0x8000\nLD SP,0xFF00\nLD %s,1234\nCALL compute\nDI\nHALT\n", ar.Loc(param).Name) + mir2.Z80Codegen(mm, ar) + res, e := z80asm.NewAssembler().AssembleString(asm) + if e != nil || len(res.Errors) > 0 { + t.Fatalf("assemble %v %v\n%s", e, res.Errors, asm) + } + z := emulator.NewRemogattoZ80() + z.Reset() + z.LoadMemory(0x8000, res.Binary) + z.SetRegisters(emulator.Registers{PC: 0x8000, SP: 0xff00}) + for n := 0; !z.IsHalted(); n++ { + if n > 10000 { + t.Fatalf("no HALT\n%s", asm) + } + z.Step() + } + if got := z.GetRegisters().HL; got != uint16(result[0].I) { + t.Fatalf("Z80 got %d MIR2 got %d\n%s", got, result[0].I, asm) + } +} + +func TestP8SplitConditionalDependency(t *testing.T) { + f := &Func{Name: "compute", RetTy: mir2.TyU16, Params: []Param{{Name: "a", Ty: mir2.TyU16}}, Body: &Block{Body: []Stmt{ + &VarDeclStmt{Name: "x", Ty: mir2.TyU16, Init: &VarRefExpr{Name: "a", Ty: mir2.TyU16}}, + &VarDeclStmt{Name: "y", Ty: mir2.TyU16, Init: &IntLitExpr{Val: 7, Ty: mir2.TyU16}}, + &VarDeclStmt{Name: "z", Ty: mir2.TyU16, Init: &CondExpr{Cond: &BoolLitExpr{Val: true}, Then: &VarRefExpr{Name: "x", Ty: mir2.TyU16}, Else: &IntLitExpr{Val: 0, Ty: mir2.TyU16}, Ty: mir2.TyU16}}, + &ReturnStmt{Val: &VarRefExpr{Name: "z", Ty: mir2.TyU16}}, + }}} + m := &Module{Name: "p8", Funcs: []*Func{f}} + c := FindSplitPoints(f, []int{9, 9, 9, 9}) + sub := ApplySplit(m, f, c[0]) + m.Funcs = append(m.Funcs, sub) + mm := LowerModule(m) + if mm.FuncByName(sub.Name) == nil { + t.Fatal("conditional dependencies omitted from split interface") + } + got, err := mir2.NewVM(mm).Call("compute", []mir2.Value{{I: 1234}}) + if err != nil || len(got) != 1 || got[0].I != 1234 { + t.Fatalf("got %v %v", got, err) + } +} + +func TestP8ReturnedSplitIsNotSplitAgain(t *testing.T) { + f := &Func{Name: "compute", RetTy: mir2.TyU16, Body: &Block{}} + var args []Expr + for _, name := range []string{"a", "b", "c", "d", "e", "f", "x", "y", "z"} { + f.Params = append(f.Params, Param{Name: name, Ty: mir2.TyU16}) + args = append(args, &VarRefExpr{Name: name, Ty: mir2.TyU16}) + } + for i := 0; i < 4; i++ { + f.Body.Body = append(f.Body.Body, &ExprStmt{Expr: &CallExpr{Fn: "side", Args: args, Ty: mir2.TyVoid}}) + } + f.Body.Body = append(f.Body.Body, &ReturnStmt{Val: &CallExpr{Fn: "compute$split_1", Args: args[:3], Ty: mir2.TyU16}}) + m := &Module{Name: "p8", Funcs: []*Func{f}} + var results []SplitResult + if subs := splitRecursive(m, f, &results, 0); len(subs) != 0 { + t.Fatalf("returned split call was split again: %v", results) + } +} diff --git a/minzc/pkg/hir/split.go b/minzc/pkg/hir/split.go index 8a1af052..938fa2d1 100644 --- a/minzc/pkg/hir/split.go +++ b/minzc/pkg/hir/split.go @@ -98,6 +98,11 @@ func splitRecursive(m *Module, f *Func, results *[]SplitResult, depth int) []*Fu } // Guard: don't re-split if last stmt is already a split call. if len(f.Body.Body) > 0 { + if rs, ok := f.Body.Body[len(f.Body.Body)-1].(*ReturnStmt); ok { + if ce, ok := rs.Val.(*CallExpr); ok && strings.Contains(ce.Fn, "$split_") { + return nil + } + } if es, ok := f.Body.Body[len(f.Body.Body)-1].(*ExprStmt); ok { if ce, ok := es.Expr.(*CallExpr); ok { if strings.Contains(ce.Fn, "$split_") { @@ -285,6 +290,10 @@ func (s splitCandidate) interfaceWidth() int { // FindSplitPoints returns viable split candidates for a function. func FindSplitPoints(f *Func, pressure []int) []splitCandidate { + // Multi-result calls need tuple forwarding; leave those functions intact. + if len(f.RetTys) > 1 { + return nil + } stmts := f.Body.Body if len(stmts) < 4 { return nil @@ -400,7 +409,8 @@ func ApplySplit(m *Module, f *Func, c splitCandidate) *Func { sub := &Func{ Name: subName, Params: params, - RetTy: mir2.TyVoid, + RetTy: f.RetTy, + RetTys: append([]mir2.Ty(nil), f.RetTys...), Body: subBody, } @@ -417,13 +427,19 @@ func ApplySplit(m *Module, f *Func, c splitCandidate) *Func { callExpr := &CallExpr{ Fn: subName, Args: args, - Ty: mir2.TyVoid, + Ty: f.RetTy, + } + if len(f.RetTys) == 1 { + callExpr.Ty = f.RetTys[0] } // New body = top half + call statement. newBody := make([]Stmt, c.splitAt+2) copy(newBody, stmts[:c.splitAt+1]) newBody[c.splitAt+1] = &ExprStmt{Expr: callExpr} + if countReturns(f) != 0 { + newBody[c.splitAt+1] = &ReturnStmt{Val: callExpr} + } f.Body.Body = newBody return sub @@ -434,6 +450,36 @@ func ApplySplit(m *Module, f *Func, c splitCandidate) *Func { // collectVarRefs walks a statement and collects variable references and definitions. func collectVarRefs(s Stmt, refs, defs map[string]bool) { switch s := s.(type) { + case *Block: + for _, inner := range s.Body { + collectVarRefs(inner, refs, defs) + } + case *StoreStmt: + collectExprRefs(s.Ptr, refs) + collectExprRefs(s.Val, refs) + case *SwitchStmt: + collectExprRefs(s.Val, refs) + for _, c := range s.Cases { + collectVarRefs(c.Body, refs, defs) + } + if s.Default != nil { + collectVarRefs(s.Default, refs, defs) + } + case *ForEachStmt: + defs[s.Var] = true + collectExprRefs(s.Ptr, refs) + collectExprRefs(s.Start, refs) + collectExprRefs(s.Len, refs) + if s.Body != nil { + collectVarRefs(s.Body, refs, defs) + } + case *AsmStmt: + for _, in := range s.Ins { + refs[in.Name] = true + } + for _, out := range s.Outs { + defs[out.Name] = true + } case *VarDeclStmt: defs[s.Name] = true if s.Init != nil { @@ -446,6 +492,7 @@ func collectVarRefs(s Stmt, refs, defs map[string]bool) { collectExprRefs(s.Target, refs) collectExprRefs(s.Val, refs) case *ReturnStmt: + collectExprRefs(s.Val, refs) for _, v := range s.Vals { collectExprRefs(v, refs) } @@ -496,6 +543,34 @@ func collectExprRefs(e Expr, refs map[string]bool) { return } switch e := e.(type) { + case *CondExpr: + collectExprRefs(e.Cond, refs) + collectExprRefs(e.Then, refs) + collectExprRefs(e.Else, refs) + case *LoadExpr: + collectExprRefs(e.Ptr, refs) + case *BitExpr: + collectExprRefs(e.X, refs) + case *CallIndirectExpr: + collectExprRefs(e.FnPtr, refs) + for _, a := range e.Args { + collectExprRefs(a, refs) + } + case *StructLitExpr: + for _, f := range e.Fields { + collectExprRefs(f.Val, refs) + } + case *RangeSourceExpr: + collectExprRefs(e.Lo, refs) + collectExprRefs(e.Hi, refs) + case *LetInExpr: + collectExprRefs(e.Init, refs) + bodyRefs := make(map[string]bool) + collectExprRefs(e.Body, bodyRefs) + delete(bodyRefs, e.Name) + for v := range bodyRefs { + refs[v] = true + } case *VarRefExpr: refs[e.Name] = true case *CallExpr: diff --git a/minzc/pkg/mir2/p8_fix1_internal_test.go b/minzc/pkg/mir2/p8_fix1_internal_test.go new file mode 100644 index 00000000..05ff11fb --- /dev/null +++ b/minzc/pkg/mir2/p8_fix1_internal_test.go @@ -0,0 +1,17 @@ +package mir2 + +import ( + "fmt" + "strings" + "testing" +) + +func TestP8WideSpillExhaustionFails(t *testing.T) { + defer func() { + err := recover() + if err == nil || !strings.Contains(fmt.Sprint(err), "needs 1 wide spill staging pairs, only 0 available") { + t.Fatalf("expected explicit staging failure, got %v", err) + } + }() + wideSpillPairs("full", OpAdd, map[string]bool{"HL": true, "DE": true, "BC": true}, 1) +} diff --git a/minzc/pkg/mir2/p8_fix1_test.go b/minzc/pkg/mir2/p8_fix1_test.go new file mode 100644 index 00000000..8dae633f --- /dev/null +++ b/minzc/pkg/mir2/p8_fix1_test.go @@ -0,0 +1,94 @@ +package mir2_test + +import ( + "fmt" + "strings" + "testing" + + "github.com/minz/minzc/pkg/mir2" +) + +func TestCriticU16ConstBlockParamSpill(t *testing.T) { + testConstBlockParamSpill(t, mir2.TyU16, 0x1234) +} + +func TestP8WideConstBlockParamSpill(t *testing.T) { + for _, tc := range []struct { + ty mir2.Ty + value int64 + }{{mir2.TyU24, 0x123456}, {mir2.TyU32, 0x12345678}} { + t.Run(tc.ty.String(), func(t *testing.T) { testConstBlockParamSpill(t, tc.ty, tc.value) }) + } +} + +func testConstBlockParamSpill(t *testing.T, ty mir2.Ty, value int64) { + m := &mir2.Module{Name: "c"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU16, Class: mir2.ClassPair}} + b := mir2.NewBuilder(f) + entry := b.SwitchToNewBlock("entry") + next := b.SwitchToNewBlock("next") + class := mir2.ClassPair + if ty.Width() > 16 { + class = mir2.ClassDWord + } + p := b.BlockParam(next, ty, class) + b.SwitchTo(entry) + x := b.Const(value, ty, class) + b.Jmp("next", x) + b.SwitchTo(next) + if ty.Width() > 16 { + f.Contract.Returns = nil + b.Ret() + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{p: {Kind: mir2.LocMem}, x: {Kind: mir2.LocDWord, Name: "DE"}}} + code := mir2.Z80Codegen(m, ar) + for i := 0; i < mir2.ByteWidth(ty); i++ { + // Poison every byte so an omitted store cannot pass via zero-initialised data. + pre := "ORG 0x8000\nLD SP,0xFF00\nLD A,255\n" + for j := 0; j < mir2.ByteWidth(ty); j++ { + pre += fmt.Sprintf("LD (_spill_arith_r%d+%d),A\n", p, j) + } + asm := pre + fmt.Sprintf("CALL arith\nLD A,(_spill_arith_r%d+%d)\nDI\nHALT\n", p, i) + code + if got := fix1Execute(t, asm).A; got != uint8(value>>(8*i)) { + t.Fatalf("byte %d got %x want %x\n%s", i, got, uint8(value>>(8*i)), asm) + } + } + } else { + r := b.Move(p, ty, class) + b.Ret(r) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{p: {Kind: mir2.LocMem}, x: {Kind: mir2.LocReg, Name: "DE"}, r: {Kind: mir2.LocReg, Name: "HL"}}} + asm := fmt.Sprintf("ORG 0x8000\nLD SP,0xFF00\nLD HL,0xFFFF\nLD (_spill_arith_r%d),HL\nCALL arith\nDI\nHALT\n", p) + mir2.Z80Codegen(m, ar) + if got := fix1Execute(t, asm).HL; got != uint16(value) { + t.Fatalf("got %x want %x\n%s", got, value, asm) + } + } +} + +func TestP8AddressSpillIsWord(t *testing.T) { + for _, ty := range []mir2.Ty{&mir2.StructTy{Name: "s", Fields: []mir2.StructField{{Name: "a", Ty: mir2.TyU32}, {Name: "b", Ty: mir2.TyU32}}}, mir2.PtrFor(32)} { + for _, explicit := range []bool{false, true} { + t.Run(fmt.Sprintf("%s/explicit=%v", ty, explicit), func(t *testing.T) { + m := &mir2.Module{Name: "c"} + f := m.AddFunc("arith") + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", ty, mir2.ClassPair) + r := b.Move(a, ty, mir2.ClassPair) + b.Ret() + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocReg, Name: "DE"}, r: {Kind: mir2.LocMem}}} + if explicit { + ar.Spilled = []mir2.Reg{r} + } + asm := mir2.Z80Codegen(m, ar) + slot := fmt.Sprintf("_spill_arith_r%d:", r) + if !strings.Contains(asm, slot+" DW 0") && !strings.Contains(asm, slot+" DB 0, 0\n") { + t.Fatalf("expected two-byte slot\n%s", asm) + } + if strings.Contains(asm, "EXX") { + t.Fatalf("address used wide staging\n%s", asm) + } + fix1Execute(t, "ORG 0x8000\nLD SP,0xFF00\nCALL arith\nDI\nHALT\n"+asm) + }) + } + } +} diff --git a/minzc/pkg/mir2/p8_invalid_asm_test.go b/minzc/pkg/mir2/p8_invalid_asm_test.go new file mode 100644 index 00000000..e86e4d2e --- /dev/null +++ b/minzc/pkg/mir2/p8_invalid_asm_test.go @@ -0,0 +1,212 @@ +package mir2_test + +import ( + "fmt" + "github.com/minz/minzc/pkg/mir2" + "testing" +) + +func TestP8SpilledByteExtension(t *testing.T) { + for _, signed := range []bool{false, true} { + t.Run(fmt.Sprint(signed), func(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU16, Class: mir2.ClassPair}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU8, mir2.ClassGeneral) + var out mir2.Reg + if signed { + out = b.Sext(a, mir2.TyU8, mir2.TyU16, mir2.ClassPair) + } else { + out = b.Ext(a, mir2.TyU8, mir2.TyU16, mir2.ClassPair) + } + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocReg, Name: "C"}, out: {Kind: mir2.LocMem}}} + asm := "ORG 0x8000\nLD SP,0xFF00\nLD C,254\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + want := uint16(254) + if signed { + want = 65534 + } + if got := fix1Execute(t, asm).HL; got != want { + t.Fatalf("got %x want %x\n%s", got, want, asm) + } + }) + } +} + +func TestP8ByteALUWithAccumulatorRHS(t *testing.T) { + for _, op := range []mir2.Op{mir2.OpAdd, mir2.OpSub} { + t.Run(op.String(), func(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU8, Class: mir2.ClassAcc}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU8, mir2.ClassGeneral) + c := b.Param("b", mir2.TyU8, mir2.ClassAcc) + out := b.BinOp(op, a, c, mir2.TyU8, mir2.ClassAcc) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocMem}, c: {Kind: mir2.LocReg, Name: "A"}, out: {Kind: mir2.LocReg, Name: "A"}}} + asm := fmt.Sprintf("ORG 0x8000\nLD SP,0xFF00\nLD A,29\nLD (_spill_arith_r%d),A\nLD A,7\nCALL arith\nDI\nHALT\n", a) + mir2.Z80Codegen(m, ar) + want := uint8(36) + if op == mir2.OpSub { + want = 22 + } + if got := fix1Execute(t, asm).A; got != want { + t.Fatalf("got %d want %d", got, want) + } + }) + } +} + +func TestP8SubtractIndexByte(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU16, Class: mir2.ClassPair}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU16, mir2.ClassPair) + c := b.Param("b", mir2.TyU8, mir2.ClassGeneral) + out := b.Sub(a, c, mir2.TyU16, mir2.ClassPair) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocReg, Name: "HL"}, c: {Kind: mir2.LocIXY8, Name: "IYL"}, out: {Kind: mir2.LocReg, Name: "HL"}}} + asm := "ORG 0x8000\nLD SP,0xFF00\nLD HL,1000\nLD IY,7\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + if got := fix1Execute(t, asm).HL; got != 993 { + t.Fatalf("got %d want 993", got) + } +} + +func TestP8SpilledConstantPreservesAccumulator(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU8, Class: mir2.ClassAcc}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU8, mir2.ClassAcc) + c := b.Const(7, mir2.TyU8, mir2.ClassGeneral) + out := b.Add(a, c, mir2.TyU8, mir2.ClassAcc) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocReg, Name: "A"}, c: {Kind: mir2.LocMem}, out: {Kind: mir2.LocReg, Name: "A"}}} + want, err := mir2.NewVM(m).Call("arith", []mir2.Value{{I: 29}}) + if err != nil { + t.Fatal(err) + } + asm := "ORG 0x8000\nLD SP,0xFF00\nLD A,29\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + if got := fix1Execute(t, asm).A; got != uint8(want[0].I) { + t.Fatalf("got %d want %d\n%s", got, want[0].I, asm) + } +} + +func TestP8WidenIndexByteToSpill(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU16, Class: mir2.ClassPair}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU8, mir2.ClassGeneral) + out := b.Move(a, mir2.TyU16, mir2.ClassPair) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocIXY8, Name: "IYL"}, out: {Kind: mir2.LocMem}}} + asm := "ORG 0x8000\nLD SP,0xFF00\nLD IY,7\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + if got := fix1Execute(t, asm).HL; got != 7 { + t.Fatalf("got %d want 7", got) + } +} + +func TestP8WideSpill(t *testing.T) { + for _, op := range []mir2.Op{mir2.OpAdd, mir2.OpSub, mir2.OpSar} { + t.Run(op.String(), func(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyI32, Class: mir2.ClassDWord}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Const(0x12345678, mir2.TyI32, mir2.ClassDWord) + c := b.Const(3, mir2.TyI32, mir2.ClassDWord) + out := b.BinOp(op, a, c, mir2.TyI32, mir2.ClassDWord) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocMem}, c: {Kind: mir2.LocMem}, out: {Kind: mir2.LocMem}}} + want, err := mir2.NewVM(m).Call("arith", nil) + if err != nil { + t.Fatal(err) + } + asm := "ORG 0x8000\nLD SP,0xFF00\nCALL arith\nPUSH HL\nEXX\nPUSH HL\nEXX\nPOP BC\nPOP HL\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + regs := fix1Execute(t, asm) + got := uint32(regs.HL) | uint32(regs.BC)<<16 + if got != uint32(want[0].I) { + t.Fatalf("got %x want %x\n%s", got, want[0].I, asm) + } + }) + } +} + +func TestP8ConditionalIntrinsicAndSanitizedCall(t *testing.T) { + for _, sym := range []string{"@mir.io.print.nl", "callee$with.dots"} { + t.Run(sym, func(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + f.Blocks[0].Insts = append(f.Blocks[0].Insts, &mir2.Inst{Op: mir2.OpCallCond, Sym: sym, Cond: mir2.CmpEq, Ty: mir2.TyVoid}) + b.Ret() + if sym != "@mir.io.print.nl" { + callee := m.AddFunc(sym) + c := mir2.NewBuilder(callee) + c.SwitchToNewBlock("entry") + c.Ret() + } + asm := "ORG 0x8000\nLD SP,0xFF00\nXOR A\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{}}) + fix1Execute(t, asm) + }) + } +} + +func TestP8WordSpillShift(t *testing.T) { + for _, op := range []mir2.Op{mir2.OpShl, mir2.OpShr, mir2.OpSar} { + t.Run(op.String(), func(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyI16, Class: mir2.ClassPair}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyI16, mir2.ClassPair) + c := b.Const(3, mir2.TyU8, mir2.ClassGeneral) + out := b.BinOp(op, a, c, mir2.TyI16, mir2.ClassPair) + b.Ret(out) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocReg, Name: "DE"}, c: {Kind: mir2.LocReg, Name: "B"}, out: {Kind: mir2.LocMem}}} + want, err := mir2.NewVM(m).Call("arith", []mir2.Value{{I: 0x9234}}) + if err != nil { + t.Fatal(err) + } + asm := "ORG 0x8000\nLD SP,0xFF00\nLD DE,0x9234\nCALL arith\nDI\nHALT\n" + mir2.Z80Codegen(m, ar) + if got := fix1Execute(t, asm).HL; got != uint16(want[0].I) { + t.Fatalf("got %x want %x", got, want[0].I) + } + }) + } +} + +func TestP8WideSpillInPlace(t *testing.T) { + m := &mir2.Module{Name: "p8"} + f := m.AddFunc("arith") + f.Contract.Returns = []mir2.Return{{Ty: mir2.TyU32, Class: mir2.ClassDWord}} + b := mir2.NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", mir2.TyU32, mir2.ClassDWord) + c := b.Const(3, mir2.TyU32, mir2.ClassDWord) + b.Add(a, c, mir2.TyU32, mir2.ClassDWord) + f.Blocks[0].Insts[1].Dst = a + b.Ret(a) + ar := &mir2.AllocResult{Locs: map[mir2.Reg]mir2.PhysLoc{a: {Kind: mir2.LocMem}, c: {Kind: mir2.LocDWord, Name: "DE"}}} + want, err := mir2.NewVM(m).Call("arith", []mir2.Value{{I: 0x12345678}}) + if err != nil { + t.Fatal(err) + } + asm := fmt.Sprintf("ORG 0x8000\nLD SP,0xFF00\nLD HL,0x5678\nLD (_spill_arith_r%d),HL\nLD HL,0x1234\nLD (_spill_arith_r%d+2),HL\nCALL arith\nPUSH HL\nEXX\nPUSH HL\nEXX\nPOP BC\nPOP HL\nDI\nHALT\n", a, a) + mir2.Z80Codegen(m, ar) + regs := fix1Execute(t, asm) + got := uint32(regs.HL) | uint32(regs.BC)<<16 + if got != uint32(want[0].I) { + t.Fatalf("got %x want %x", got, want[0].I) + } +} diff --git a/minzc/pkg/mir2/p8_spill_labels_test.go b/minzc/pkg/mir2/p8_spill_labels_test.go new file mode 100644 index 00000000..d3dda4ec --- /dev/null +++ b/minzc/pkg/mir2/p8_spill_labels_test.go @@ -0,0 +1,29 @@ +package mir2 + +import ( + "github.com/minz/minzc/pkg/z80asm" + "strings" + "testing" +) + +func TestP8OrphanSpillDefinition(t *testing.T) { + m := &Module{Name: "p8"} + f := m.AddFunc("arith") + b := NewBuilder(f) + b.SwitchToNewBlock("entry") + a := b.Param("a", TyU8, ClassAcc) + x := b.Const(7, TyU8, ClassGeneral) + out := b.Add(a, x, TyU8, ClassAcc) + b.Ret(out) + ar := &AllocResult{Locs: map[Reg]PhysLoc{a: {Kind: LocReg, Name: "A"}, x: {Kind: LocMem}, out: {Kind: LocReg, Name: "A"}}} + asm := Z80Codegen(m, ar) + res, err := z80asm.NewAssembler().AssembleString(asm) + if err != nil || len(res.Errors) > 0 { + t.Fatalf("%v %v\n%s", err, res.Errors, asm) + } + // The immediate ALU path does not emit the TSMC reload, so the + // constant's patch store must become an ordinary, defined spill store. + if strings.Contains(asm, "LD (_spill_") && !strings.Contains(asm, "_spill_arith_r2:") { + t.Fatalf("missing spill definition:\n%s", asm) + } +} diff --git a/minzc/pkg/mir2/z80codegen.go b/minzc/pkg/mir2/z80codegen.go index 023f884d..fbcc942c 100644 --- a/minzc/pkg/mir2/z80codegen.go +++ b/minzc/pkg/mir2/z80codegen.go @@ -329,6 +329,7 @@ type z80cg struct { // overriding the static allocator assignment g.ar.Loc(r).Name. // Cleared at every CALL (the call clobbers volatile state). physOverride map[Reg]string + regInfo map[Reg]RegInfo // liveness gives cross-block live-out sets so that block-local // physOverride relocations can be undone for values a successor reads. @@ -579,6 +580,7 @@ func computeDeadConsts(f *Func, ar *AllocResult) map[Reg]bool { func (g *z80cg) genFunc(f *Func) { g.fn = f + g.regInfo = collectRegInfo(f) g.cmpSwapped = make(map[Reg]bool) g.cmpAndZero = make(map[Reg]bool) g.cmpNeedsTwo = make(map[Reg]bool) @@ -664,6 +666,10 @@ func (g *z80cg) genFunc(f *Func) { g.emitf(" RET") } + // Redirect orphaned TSMC stores before collecting spill references: the + // rewrite can introduce new spill slots that must also be defined. + fixOrphanedTSMCStores(g.sb, label) + // Emit named spill data section — one label per spilled register. // DB for 8-bit, DW for 16-bit. Labels resolve forward references // from LD (label),A / LD A,(label) throughout the function. @@ -672,7 +678,7 @@ func (g *z80cg) genFunc(f *Func) { // 1. Emit from g.ar.Spilled for vregs in collectRegInfo (known width) // 2. Scan emitted asm for any _spill_ refs still missing a definition // (catches vregs not in Spilled or filtered by TSMC/regInfo) - regInfo := collectRegInfo(f) + regInfo := g.regInfo emittedSpills := make(map[string]bool) var spillCount int for _, r := range g.ar.Spilled { @@ -695,7 +701,7 @@ func (g *z80cg) genFunc(f *Func) { } w := 1 if info, ok := regInfo[r]; ok { - w = (info.Ty.Width() + 7) / 8 + w = z80SpillBytes(info.Ty) } slabel := g.spillLabel(r) emittedSpills[slabel] = true @@ -707,7 +713,7 @@ func (g *z80cg) genFunc(f *Func) { case 3: g.emitf("%s: DB 0, 0, 0", slabel) // eZ80 24-bit default: - g.emitf("%s: DB 0", slabel) + g.emitf("%s: DB %s", slabel, strings.TrimSuffix(strings.Repeat("0, ", w), ", ")) } } } @@ -755,15 +761,18 @@ func (g *z80cg) genFunc(f *Func) { if len(missingSpills) > 0 { g.emitf("; — extra spill slots for %s (%d rescued)", label, len(missingSpills)) for _, slabel := range missingSpills { - // Default to DW (16-bit) — most missing spills are ptr/u16 regs - g.emitf("%s: DW 0", slabel) + // Preserve the full storage width, including main/shadow words. + w := 2 + var r int + if _, err := fmt.Sscanf(slabel[len(prefix):], "%d", &r); err == nil { + if info, ok := regInfo[Reg(r)]; ok { + w = z80SpillBytes(info.Ty) + } + } + g.emitf("%s: DB %s", slabel, strings.TrimSuffix(strings.Repeat("0, ", w), ", ")) } } - // Pass 3: fix orphaned TSMC stores — when a _tsmc_ label is referenced - // (by store patches) but never defined (no reload instruction), redirect - // the stores to the corresponding _spill_ label so data flows correctly. - fixOrphanedTSMCStores(g.sb, label) } // ── Block ───────────────────────────────────────────────────────────────────── diff --git a/minzc/pkg/mir2/z80codegen_alu.go b/minzc/pkg/mir2/z80codegen_alu.go index 50520042..82214a10 100644 --- a/minzc/pkg/mir2/z80codegen_alu.go +++ b/minzc/pkg/mir2/z80codegen_alu.go @@ -49,7 +49,7 @@ func (g *z80cg) emitSBCHL(rhs string) { } else if isSimpleReg(rhs) && !isPairReg(rhs) { // 8-bit operand: promote to pair via promote8toPair. pair := g.promote8toPair(rhs) - g.emitf(" SBC HL, %s", pair) + g.emitSBCHL(pair) } else { g.emitf(" SBC HL, %s", rhs) } @@ -213,12 +213,12 @@ func (g *z80cg) genBinOp(mnem string, inst *Inst) { g.invalidate("A") // A about to hold a new result switch mnem { case "ADD": - g.emitf(" ADD A, %s", lhs) + g.emit8ALU("ADD", lhs) case "AND", "OR", "XOR": g.emit8ALU(mnem, lhs) case "SUB": g.emit(" NEG") - g.emitf(" ADD A, %s", lhs) + g.emit8ALU("ADD", lhs) default: g.comment(fmt.Sprintf("TODO: 8-bit %s %s, A → %s with A as rhs", mnem, lhs, dst)) g.emitLDA(lhs) @@ -819,6 +819,22 @@ func (g *z80cg) emit8ALUImm(mnem string, imm int64) { // ── Shifts ──────────────────────────────────────────────────────────────────── func (g *z80cg) genShift(mnem string, inst *Inst) { + if dst := g.loc(inst.Dst); isSpill(dst) && inst.Ty.Width() <= 16 { + // Z80 shifts accept registers or indirect memory, never absolute + // spill labels. Stage the result without disturbing a live HL/A. + pair, scratch := "HL", "HL" + if inst.Ty.Width() <= 8 { + pair, scratch = "AF", "A" + } + g.emitf(" PUSH %s", pair) + g.physOverride[inst.Dst] = scratch + g.genShift(mnem, inst) + delete(g.physOverride, inst.Dst) + g.emitMov(dst, scratch, inst.Ty.Width()) + g.emitf(" POP %s", pair) + g.invalidate(scratch) + return + } if _, constant := g.constVals[inst.Src[1]]; !constant { g.genVariableShift(mnem, inst) return diff --git a/minzc/pkg/mir2/z80codegen_copy.go b/minzc/pkg/mir2/z80codegen_copy.go index ff6f9482..6ebf7a3b 100644 --- a/minzc/pkg/mir2/z80codegen_copy.go +++ b/minzc/pkg/mir2/z80codegen_copy.go @@ -1,5 +1,7 @@ package mir2 +import "fmt" + // parallelCopy is a single register-to-register move in a parallel copy sequence. type parallelCopy struct { srcName string @@ -316,8 +318,14 @@ func (g *z80cg) emitParallelCopy(copies []parallelCopy) { } } else if isSpill(c.dstName) { g.emit(" EX AF, AF'") - g.emitf(" LD A, %d", c.immVal&0xFF) - g.emitf(" LD (%s), A", c.dstName) + for i := 0; i < z80SpillBytes(c.ty); i++ { + g.emitf(" LD A, %d", (c.immVal>>(8*i))&0xFF) + addr := c.dstName + if i > 0 { + addr += fmt.Sprintf("+%d", i) + } + g.emitf(" LD (%s), A", addr) + } g.emit(" EX AF, AF'") } else if c.ty.Width() <= 8 { g.emitf(" LD %s, %d", c.dstName, c.immVal&0xFF) @@ -333,6 +341,15 @@ func (g *z80cg) emitSingleCopy(src, dst string, ty Ty) { if src == dst { return } + if isZ80WideInt(ty) { + g.emitMov32(dst, src) + if ty.Width() == 24 && isPairReg(dst) { + g.emit(" EXX") + g.emitLD8(highByte(dst), "0") + g.emit(" EXX") + } + return + } // F register: cannot be accessed directly. Materialise flag→register or // register→flag via the same logic as emitMov. if src == "F" { diff --git a/minzc/pkg/mir2/z80codegen_inst.go b/minzc/pkg/mir2/z80codegen_inst.go index 1f3d5941..7b804afc 100644 --- a/minzc/pkg/mir2/z80codegen_inst.go +++ b/minzc/pkg/mir2/z80codegen_inst.go @@ -47,6 +47,10 @@ func (g *z80cg) genInst(inst *Inst) { dst := g.loc(inst.Dst) + if g.genWideSpills(inst) { + return + } + switch inst.Op { case OpConst: // Only record as constant if the Dst is NOT a block parameter. @@ -69,7 +73,7 @@ func (g *z80cg) genInst(inst *Inst) { } if !g.deadConsts[inst.Dst] { w := inst.Ty.Width() - if w >= 24 { + if isZ80WideInt(inst.Ty) { // 24/32-bit constant via shadow pair. // LD rr, lo16 (10T) // EXX (4T) @@ -84,6 +88,13 @@ func (g *z80cg) genInst(inst *Inst) { g.emit(" EXX") } else if isSpill(dst) { // LocMem spill destination. + // Loading an immediate for a spill must preserve unrelated live + // values in the staging register. + if w <= 8 { + g.emit(" PUSH AF") + } else { + g.emit(" PUSH HL") + } // TSMC: if eligible, patch reload sites instead of memory store. if pair := g.tsmcSpillPairFor(inst.Dst); pair != nil { if w <= 8 { @@ -107,11 +118,11 @@ func (g *z80cg) genInst(inst *Inst) { g.invalidate("HL") } } - } else if isSpill(dst) { - g.emit(" EX AF, AF'") - g.emitf(" LD A, %d", inst.Imm&0xFF) - g.emitf(" LD (%s), A", dst) - g.emit(" EX AF, AF'") + if w <= 8 { + g.emit(" POP AF") + } else { + g.emit(" POP HL") + } } else { g.emitf(" LD %s, %d", dst, inst.Imm) } @@ -122,7 +133,7 @@ func (g *z80cg) genInst(inst *Inst) { if dst == src { return // no-op } - g.emitMov(dst, src, inst.Ty.Width()) + g.emitMov(dst, src, 8*z80SpillBytes(inst.Ty)) case OpAdd: g.lastFlagsLhs = "" @@ -896,7 +907,13 @@ func (g *z80cg) genInst(inst *Inst) { g.lastFlagsRhs = "" cc := cmpCondCode(inst.Cond) g.comment(fmt.Sprintf("genCallCond: CALL %s, %s", cc, inst.Sym)) - g.emitf(" CALL %s, %s", cc, inst.Sym) + // Use the regular call lowering for intrinsics, fixed addresses, and + // sanitized symbols. Intrinsics have no callable assembly label. + skip := fmt.Sprintf(".%s_callcond%d", sanitizeIdent(g.fn.Name), g.trampIdx) + g.trampIdx++ + g.emitf(" JRS %s, %s", invertCC(cc), skip) + g.genCall(inst) + g.emitf("%s:", skip) clear(g.holdsPhys) // calls clobber all volatile registers case OpIn8: diff --git a/minzc/pkg/mir2/z80codegen_move.go b/minzc/pkg/mir2/z80codegen_move.go index 3d7eec0a..1c2639ab 100644 --- a/minzc/pkg/mir2/z80codegen_move.go +++ b/minzc/pkg/mir2/z80codegen_move.go @@ -416,7 +416,7 @@ func (g *z80cg) genExt(inst *Inst) { g.emitLD8(lo, src) } } - g.emitf(" LD %s, 0", hi) + g.emitLD8(hi, "0") return } // Fallback. @@ -529,6 +529,33 @@ func (g *z80cg) emitMov32(dst, src string) { if dst == src { return } + if isSpill(src) && isSpill(dst) { + g.emit(" PUSH HL") + g.emit(" EXX") + g.emit(" PUSH HL") + g.emit(" EXX") + g.emitMov32("HL", src) + g.emitMov32(dst, "HL") + g.emit(" EXX") + g.emit(" POP HL") + g.emit(" EXX") + g.emit(" POP HL") + return + } + if isSpill(src) && isPairReg(dst) { + g.emitf(" LD %s, (%s)", dst, src) + g.emit(" EXX") + g.emitf(" LD %s, (%s+2)", dst, src) + g.emit(" EXX") + return + } + if isSpill(dst) && isPairReg(src) { + g.emitf(" LD (%s), %s", dst, src) + g.emit(" EXX") + g.emitf(" LD (%s+2), %s", dst, src) + g.emit(" EXX") + return + } g.emitf(" PUSH %s", src) // save main src_lo g.emit(" EXX") // switch to shadow g.emitf(" PUSH %s", src) // save shadow src_hi @@ -726,6 +753,11 @@ func (g *z80cg) emitMov(dst, src string, widthBits int) { } // register pair → LocMem spill slot: use LD (nn), rr. if isSpill(dst) { + if !isPairReg(src) { + g.emitLD8(dst, src) + g.emitLD8(highByte(dst), "0") + break + } g.emitf(" LD (%s), %s", dst, src) break } @@ -805,6 +837,9 @@ func lowByte(rr string) string { // highByte returns the high-byte name of a 16-bit register. func highByte(rr string) string { + if isSpill(rr) { + return rr + "+1" + } switch rr { case "HL": return "H" diff --git a/minzc/pkg/mir2/z80codegen_spill.go b/minzc/pkg/mir2/z80codegen_spill.go new file mode 100644 index 00000000..64864c0c --- /dev/null +++ b/minzc/pkg/mir2/z80codegen_spill.go @@ -0,0 +1,115 @@ +package mir2 + +import "fmt" + +// Z80 carries aggregate and pointer values as addresses. Only 24/32-bit +// integers use the main/shadow register representation. +func isZ80WideInt(ty Ty) bool { + ty = BaseOf(ty) + return IsInt(ty) && (ty.Width() == 24 || ty.Width() == 32) +} + +func z80SpillBytes(ty Ty) int { + if isZ80WideInt(ty) { + return ByteWidth(ty) + } + if ty.Width() <= 8 { + return 1 + } + return 2 +} + +// genWideSpills stages memory-backed wide operands in main/shadow register +// pairs. Wide emitters operate on pairs; a spill label is never a PUSH, shift, +// or arithmetic operand. Preserve both banks of every temporary pair. +func (g *z80cg) genWideSpills(inst *Inst) bool { + if !isZ80WideInt(inst.Ty) && !isZ80WideInt(inst.SrcTy) { + return false + } + switch inst.Op { + case OpConst, OpMove, OpAdd, OpSub, OpAnd, OpOr, OpXor, OpShl, OpShr, OpSar, OpMul, OpExt, OpSext, OpTrunc: + default: + return false + } + info := g.regInfo + regs := append([]Reg{inst.Dst}, inst.Src[:]...) + var spilled []Reg + used := map[string]bool{} + seen := map[Reg]bool{} + for _, r := range regs { + if r == NoReg || seen[r] { + continue + } + seen[r] = true + loc := g.loc(r) + if ri, ok := info[r]; ok && isZ80WideInt(ri.Ty) && isSpill(loc) { + spilled = append(spilled, r) + } else if isPairReg(loc) { + used[loc] = true + } else if pair, ok := regParent[loc]; ok { + used[pair] = true + } + } + if len(spilled) == 0 { + return false + } + available := wideSpillPairs(g.fn.Name, inst.Op, used, len(spilled)) + labels := map[Reg]string{} + for i, r := range spilled { + pair := available[i] + labels[r] = g.loc(r) + g.emitf(" PUSH %s", pair) + g.emit(" EXX") + g.emitf(" PUSH %s", pair) + g.emit(" EXX") + if r != inst.Dst || inst.Src[0] == r || inst.Src[1] == r { + g.emitf(" LD %s, (%s)", pair, labels[r]) + g.emit(" EXX") + if info[r].Ty.Width() == 24 { + g.emitLD8(lowByte(pair), labels[r]+"+2") + g.emitLD8(highByte(pair), "0") + } else { + g.emitf(" LD %s, (%s+2)", pair, labels[r]) + } + g.emit(" EXX") + } + g.physOverride[r] = pair + } + g.genInst(inst) + if label, ok := labels[inst.Dst]; ok { + pair := g.loc(inst.Dst) + g.emitf(" LD (%s), %s", label, pair) + g.emit(" EXX") + if info[inst.Dst].Ty.Width() == 24 { + g.emitLD8(label+"+2", lowByte(pair)) + } else { + g.emitf(" LD (%s+2), %s", label, pair) + } + g.emit(" EXX") + } + for i := len(spilled) - 1; i >= 0; i-- { + r := spilled[i] + pair := available[i] + delete(g.physOverride, r) + g.emit(" EXX") + g.emitf(" POP %s", pair) + g.emit(" EXX") + g.emitf(" POP %s", pair) + g.invalidate(pair) + } + return true +} + +// Refuse code generation when staging cannot represent every spilled operand. +func wideSpillPairs(fn string, op Op, used map[string]bool, needed int) []string { + var available []string + for _, pair := range []string{"HL", "DE", "BC"} { + if !used[pair] { + available = append(available, pair) + } + } + if len(available) < needed { + panic(fmt.Sprintf("Z80 codegen %s: %s needs %d wide spill staging pairs, only %d available", fn, op, needed, len(available))) + } + return available +} diff --git a/reports/2026-10-02-P8-Invalid-Asm.md b/reports/2026-10-02-P8-Invalid-Asm.md new file mode 100644 index 00000000..1353bbc8 --- /dev/null +++ b/reports/2026-10-02-P8-Invalid-Asm.md @@ -0,0 +1,239 @@ +# Production Z80 assembly repair (P8) + +Baseline: `6b8fc396` (the worktree's origin/main base). Measurements use the +production PBQP pipeline, `--asserts none`, then the built-in MZA assembler. +The seeded generator is `.local/fuzz2.py`: seeds 0–1099, 1,100 generated +programs, no generation skips. No `scripts/fuzz_diff.py` or +`scripts/assert_matrix.py` existed at the baseline. + +All 522 tracked source examples were attempted, including archived examples +and Objective-C sources. 215 fail compilation on both versions; those are +reported separately from assembly failures. Both parser self-host examples +already assemble at this baseline, despite the older task observations. + +## Before and after + +| Measurement | Main | Repaired | +|---|---:|---:| +| Seeded programs failing assembly | 683 / 1,100 | 0 / 1,100 | +| Tracked sources failing assembly | 65 / 522 | 61 / 522 | +| Tracked sources failing compilation | 215 | 215 | +| Corpus files with validator diagnostics | 37 | 25 | +| Corpus validator instruction markers | 1,331 | 408 | +| Successfully assembled corpus binary bytes | 141,594 | 157,697 | +| Binary bytes for the same successfully assembled files | 141,594 | 141,630 | +| Emitted corpus assembly text bytes | 3,872,997 | 3,917,272 | + +The larger binary total includes four recovered programs; the common corpus +increases by 36 bytes (0.03%). FIX1 reduces the previous branch total by +1,060 bytes (158,757 → 157,697). No previously assembling corpus file stops +assembling. + +Four tracked sources newly assemble: + +- `examples/glsl_sphere_demo.minz` +- `examples/nanz/nc.nanz` +- `examples/nanz/rotozoomer.nanz` +- `examples/nanz/test_irc_minimal.nanz` + +## Root causes + +Counts below are unique assembler diagnostics with an emitted source-line +mapping; identical repeated pass diagnostics are counted once. Validator +markers are collected separately. A single file can contain several classes. +The assembler often reports an undefined symbol as “invalid operands.” + +| Class | Fuzzer before | Corpus before | Corpus after | Emitting path | +|---|---:|---:|---:|---| +| Spilled byte extension / zero high byte | 631 | 8 | 0 | `mir2/z80codegen_move.go`: `genExt`, `genSext` | +| Spill labels introduced after data emission | 449 | 31 | 0 | `mir2/z80codegen.go`: `genFunc`; `z80codegen_move.go`: `fixOrphanedTSMCStores` | +| Wide/word spill operands in register-only forms | 1 | 472 | 0 | `z80codegen_inst.go`: constants; `z80codegen_move.go`: copies; `z80codegen_alu.go`: shifts/arithmetic | +| Missing split-function definitions | 105 | 5 | 0 | `hir/split.go`: dependency walker, `ApplySplit`, `splitRecursive` | +| Byte ALU with spill or pair operand and A as rhs | 17 | 8 | 0 | `z80codegen_alu.go`: `genBinOp` | +| SBC with promoted index-register byte | 2 | 3 | 0 | `z80codegen_alu.go`: `emitSBCHL` | +| Conditional intrinsic call to a non-existent/raw label | 0 | 1 | 0 | `z80codegen_inst.go`: `OpCallCond` | +| Unresolved external/import/inline-assembly symbols | 0 | 214 | 214 | Frontend/import/stdlib symbol lowering; `genCall` and inline assembly | +| Address operands without backing local/global storage | 0 | 51 | 51 | HIR address lowering → `z80codegen_inst.go`: `OpAddrOf` | +| Missing MIR operands (`?`) | 0 | 38 | 38 | HIR lowering → `z80codegen_call.go`: `emitCallArgs` | +| Other invalid inline/legacy forms | 0 | 58 | 56 | Frontend inline assembly → `OpAsm` | +| Other assembler diagnostics | 0 | 3 | 3 | Unresolved EQU symbols and a duplicate inline label | + +Representative failures include `LD _spill_v_f_r29, 0`, +`LD (_spill_v_f_r48), A` without a definition, `PUSH _spill_fp_div_r14`, +`SRA _spill_fp_mul_r7`, `JP f_split_7`, `ADD A, _spill_v_f_r20`, +`SBC HL, IY`, and `CALL Z, @mir.io.print.nl`. + +Spill high bytes now address `slot+1`. Byte operations use the existing legal +memory/register helpers. Word shifts stage through a preserved register; +wide spill operations stage main and shadow words through preserved pairs, +and wide copies/returns carry both words. Spill storage uses three/four bytes only for 24/32-bit integer values; +struct and pointer values use two-byte address slots. Orphan TSMC stores are rewritten before +spill references are collected. + +Split dependencies include single returns and nested expressions/statements. +Split callees preserve return types and callers forward their return values. +A returned split call is protected against repeated splitting and duplicate +function definitions. Conditional calls use regular call lowering, which +already shares `sanitizeIdent` with function definitions and handles inline +intrinsics and fixed addresses. + +The remaining failures require frontend/runtime/import repairs or changes to +legacy inline assembly, rather than legalizing spill operands. Examples: +`CALL disk_read`, `CALL sql__sqlite__sqlite_query`, `LD HL, val`, `LD C, ?`, +inline references to `zx_console_con_attr`, and prose emitted as assembly. +They are retained as failures, not stubbed or silently accepted. The Zork VM +source also exceeds the assembler scanner line limit; this parse-level error +has no emitted instruction-line diagnostic and is retained separately in the +raw assembler log. + +## Correctness guard and regressions + +This correctness evidence covers **fuzz2-shaped programs only**. Assembly +success does not establish general compiler correctness. + +All 683 newly assembling seeds produce the same result on Z80 and the MIR2 +VM: **683 newly assemble + VM-correct; 0 newly assemble + VM-wrong**. +679 also agree with the generator's independent Python source oracle. +Four source-oracle/compiler disagreements remain and are reported as found +semantic discrepancies rather than counted as source-oracle passes: + +| Seed | Python expected | MIR2 result | Z80 result | +|---|---:|---:|---:| +| 56 | 15,271 | 15,015 | 15,015 | +| 784 | 50,813 | 50,809 | 50,809 | +| 812 | 55,334 | 55,590 | 55,590 | +| 967 | 56,519 | 56,775 | 56,775 | + +The isolated cause is literal-only expressions being typed u8 even in a u16 +context: `let y: u16 = 7 + 255` produces 6 instead of 262. The Python oracle +is right; MIR2 and Z80 share the frontend error. Seed 56 contains `255 + 3`. +These are not Z80-versus-VM mismatches. + +FIX1 checks all 1,100 seeds, including previously assembling ones: zero +assembly failures, 1,100 VM/Z80 agreements, 1,095 source-oracle passes. +The fifth shared oracle failure is seed 399 (expected 712, VM/Z80 456), which +was outside the earlier set of 683 newly assembling seeds. + +Regression files: + +- `minzc/pkg/mir2/p8_invalid_asm_test.go`: assembling and executing byte/word + spill extensions, accumulator-rhs arithmetic, index-byte subtraction, + preserved spilled-constant staging, widening to a spill, wide arithmetic and + shifts, an in-place wide spill, conditional intrinsic/sanitized calls, and + word spill shifts. +- `minzc/pkg/mir2/p8_spill_labels_test.go`: orphan-store data definition. +- `minzc/pkg/mir2/p8_fix1_test.go`: the critic u16 constant block-parameter + regression (`0xff34` versus `0x1234`), every stored u24/u32 byte checked + after poisoning, and two-byte struct/pointer slots without shadow staging, + for both explicit and rescued spills. +- `minzc/pkg/mir2/p8_fix1_internal_test.go`: exhausted staging pairs cause an + explicit codegen failure, rather than falling through to invalid assembly. +- `minzc/pkg/hir/p8_split_test.go`: return forwarding on VM and Z80, nested + conditional dependencies, and prevention of repeated returned-call splits. +- `minzc/pkg/hir/exhaustive_judge_test.go`: replaces the old known-u32- + assembly-failure skip with 65,636 executed Z80 sums, also checked in MIR2. + It writes callee spill slots directly and does not judge the real memory + parameter calling convention. + +Reverting the production changes gives exit 1 for the P8 regressions; +restoring them gives exit 0. The additional conditional-dependency, +returned-split, and in-place-wide regressions were also individually checked +red with their fixes reverted and green after restoration. + +The per-assert matrix reparses each source, forces Z80, and isolates each +module assert. Sandbox checks retain preceding assertions because their +shared state is a dependency. It covers imported assertions as well as the +source file's own assertions; it uses Nanz, Pascal, and Frill frontends. +The baseline has 4,016 checks: 3,070 pass and 946 fail. The repair has zero +newly failing checks and two newly passing checks. + +## Gates and retained evidence + +All required gates were run sequentially with `set -o pipefail`, +`GOCACHE=/tmp/minz-go-cache`, and `GOFLAGS=-buildvcs=false`: + +| Command (from `minzc`) | Final exit code | +|---|---:| +| `go build ./pkg/... ./cmd/...` | 0 | +| `go test ./pkg/hir ./pkg/mir2 -count=1` | 0 | +| `go test -short ./pkg/pipeline/... ./pkg/c89/... -count=1` | 0 | +| `go test ./pkg/nanz -skip '^TestShowcaseCompileAssemble$' -count=1` | 0 | + +The first HIR gate exited 1 because its deliberate old-u32-limitation marker +required replacing the skip when assembly began succeeding. The replacement +Z80 judge passes; the final gate exits 0. + +Raw generated sources, assembly, binary sizes, compiler/assembler logs, +validator lines, classified offending lines with codegen paths, oracle +results, assert matrices, and gate logs are retained under `.local/`. +Collection helpers are `.local/measure.py`, `.local/report_errors.py`, +`.local/check_values.py`, and `.local/assert_matrix.go`. That directory is +intentionally not committed. + +## FIX1 correctness limits and remaining bugs + +The critic reported gen3 programs with globals and u8 temporaries: 150/150 +assembled on the P8 branch but 0/150 were correct; main assembled 35/150 +and had 0 correct. These observations show pre-existing wrong-value bugs +surfacing once assembly errors are removed. They are separate from the +fuzz2 correctness evidence above. The retained gen3 script has several modes; +rerun results below identify the mode rather than assuming equal populations. + +Isolated pre-existing defects retained for follow-up: + +- Memory-parameter ABI mismatch: a caller uses `physName` to emit a store + such as `LD ($F072), DE`, but the callee reads `_spill_v_f_rN`. The add32 + judge bypasses this mismatch by poking callee slots directly; its comment + now describes that limitation. +- Byte XOR of truncated u16 values computes zero in the isolated repro. +- Tail-call parallel moves clobber B when setting up small constant arguments. +- Literal-only expressions use u8 arithmetic in a u16 context, including + `let y: u16 = 7 + 255` producing 6. The shared fuzz2 oracle failures are + compiler errors, not evidence against the source oracle. This also explains + the critic's four P8 oracle seeds plus three in its separate fuzz2 run. +- Remaining invalid assembly includes shifts on IX/IY half registers and + u32 spill-to-global stores. +- `emitMov32` still performs four-byte moves for u24 values; a neighbouring + three-byte spill can be overwritten. Wide move/truncation behavior also + remains outside the constant-store regression's coverage. + +P8 also fixed conditional-call argument setup (`OpCallCond` now uses normal +call lowering) and `highByte(spill)` addressing the low byte twice instead +of `slot+1`. FIX1 limits wide staging and spill sizes to 24/32-bit integers, +stores every byte of spilled constant block parameters, and caches register +information once per function. Insufficient staging pairs now abort codegen +with a diagnostic containing the function, operation, and pair counts +(the string-returning codegen API reports this as a panic). + +Binary size changes from the prior P8 commit: + +| Source | Before FIX1 | After FIX1 | Delta | +|---|---:|---:|---:| +| `examples/c89/fatfs_lowlevel.c` | 1,605 | 1,545 | -60 | +| `examples/glsl_sphere_demo.minz` | 3,775 | 2,878 | -897 | +| `examples/nanz/canvas_house.nanz` | 931 | 889 | -42 | +| `examples/nanz/typed_print.nanz` | 892 | 826 | -66 | +| `examples/nanz/rotozoomer.nanz` | 2,641 | 2,646 | +5 | +| All successfully assembled corpus files | 158,757 | 157,697 | -1,060 | + +Every FIX1 regression was observed failing with its repair removed (exit 1), +then passing after restoration (exit 0). The exhaustion regression removes +only the diagnostic guard so it tests a silent fallback, not a build failure. +Raw FIX1 evidence is retained in `.local/fix1-*`; `.local/` is not committed. + +FIX1 rerun of the retained `gen3.py`, seeds 0–149, default `all` mode: +150 compile, 118 assemble, 32 fail assembly (IX/IY-half shifts such as +`SRL IYL`), and 0 pass Z80 source-oracle assertions. Of the 118 assembling +programs, all 118 fail correctness; MIR2 passes 94/150 source assertions. +This population includes shifts and u32 operations and differs from the +critic's quoted 150/150-assembling population. + +FIX1 per-assert comparison against unchanged `origin/main` (`6b8fc396`): +4,016 checks, 3,070 → 3,072 passes, zero newly failing, two newly passing. +Corpus remains at 215 compilation failures and 61 assembly failures out of +522 tracked sources, with no new failures. The four Go gates all exit 0. + +A second gen3 run with the retained script's `noshift` mode (same seeds, +including u32 and conditional calls) compiles and assembles 150/150: +1 Z80 source-oracle pass and 149 wrong values; MIR2 passes 92/150. This +result does not establish correctness beyond the one passing program.