diff --git a/builder/bdwgc.go b/builder/bdwgc.go index b03b154203..508e9470b8 100644 --- a/builder/bdwgc.go +++ b/builder/bdwgc.go @@ -30,10 +30,11 @@ var BoehmGC = Library{ // Use a minimal environment. "-DNO_MSGBOX_ON_ERROR", // don't call MessageBoxA on Windows "-DDONT_USE_ATEXIT", - "-DNO_GETENV", // smaller binary, more predictable configuration - "-DNO_CLOCK", // don't use system clock - "-DNO_DEBUGGING", // reduce code size - "-DGC_NO_FINALIZATION", // finalization is not used at the moment + "-DNO_GETENV", // smaller binary, more predictable configuration + "-DNO_CLOCK", // don't use system clock + "-DNO_DEBUGGING", // reduce code size + "-DGC_NO_FINALIZATION", // finalization is not used at the moment + "-DGC_ATOMIC_UNCOLLECTABLE", // pointer-free storage retained until GC_free // Special flag to work around the lack of __data_start in ld.lld. // TODO: try to fix this in LLVM/lld directly so we don't have to diff --git a/builder/build.go b/builder/build.go index 104e0f3866..4b0d175b0e 100644 --- a/builder/build.go +++ b/builder/build.go @@ -1286,6 +1286,7 @@ func makeGlobalsModule(ctx llvm.Context, globals map[string]map[string]string, m global := llvm.AddGlobal(mod, stringType, globalName) global.SetInitializer(initializer) global.SetAlignment(targetData.PrefTypeAlignment(stringType)) + global.SetVisibility(llvm.HiddenVisibility) } } diff --git a/builder/sizes_test.go b/builder/sizes_test.go index 639c32b5c2..18cd43e40c 100644 --- a/builder/sizes_test.go +++ b/builder/sizes_test.go @@ -42,9 +42,9 @@ func TestBinarySize(t *testing.T) { // This is a small number of very diverse targets that we want to test. tests := []sizeTest{ // microcontrollers - {"hifive1b", "examples/echo", 4313, 323, 0, 2260}, - {"microbit", "examples/serial", 2838, 382, 8, 2256}, - {"wioterminal", "examples/pininterrupt", 8027, 1665, 132, 7488}, + {"hifive1b", "examples/echo", 4477, 323, 0, 2260}, + {"microbit", "examples/serial", 2946, 382, 8, 2256}, + {"wioterminal", "examples/pininterrupt", 8507, 1717, 148, 7488}, // TODO: also check wasm. Right now this is difficult, because // wasm binaries are run through wasm-opt and therefore the diff --git a/compileopts/config.go b/compileopts/config.go index 7786cd2178..7e7c4487d6 100644 --- a/compileopts/config.go +++ b/compileopts/config.go @@ -24,7 +24,7 @@ import ( // library path in advance in several places). var libVersions = map[string]int{ "musl": 3, - "bdwgc": 2, + "bdwgc": 3, } // Config keeps all configuration affecting the build in a single struct. diff --git a/compiler/channel.go b/compiler/channel.go index a562e97e3a..e03995b52e 100644 --- a/compiler/channel.go +++ b/compiler/channel.go @@ -14,8 +14,10 @@ import ( ) func (b *builder) createMakeChan(expr *ssa.MakeChan) llvm.Value { - elementSize := b.targetData.TypeAllocSize(b.getLLVMType(expr.Type().Underlying().(*types.Chan).Elem())) + elementType := b.getLLVMType(expr.Type().Underlying().(*types.Chan).Elem()) + elementSize := b.targetData.TypeAllocSize(elementType) elementSizeValue := llvm.ConstInt(b.uintptrType, elementSize, false) + elementLayout := b.createObjectLayout(elementType, expr.Pos()) bufSize := b.getValue(expr.Size, getPos(expr)) b.createChanBoundsCheck(elementSize, bufSize, expr.Size.Type().Underlying().(*types.Basic), expr.Pos()) if bufSize.Type().IntTypeWidth() < b.uintptrType.IntTypeWidth() { @@ -23,7 +25,7 @@ func (b *builder) createMakeChan(expr *ssa.MakeChan) llvm.Value { } else if bufSize.Type().IntTypeWidth() > b.uintptrType.IntTypeWidth() { bufSize = b.CreateTrunc(bufSize, b.uintptrType, "") } - return b.createRuntimeCall("chanMake", []llvm.Value{elementSizeValue, bufSize}, "") + return b.createRuntimeCall("chanMake", []llvm.Value{elementSizeValue, bufSize, elementLayout}, "") } // createChanSend emits a pseudo chan send operation. It is lowered to the diff --git a/compiler/interface.go b/compiler/interface.go index 84f91cc448..10bffd7993 100644 --- a/compiler/interface.go +++ b/compiler/interface.go @@ -284,6 +284,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { types.NewVar(token.NoPos, nil, "elementType", types.Typ[types.UnsafePointer]), types.NewVar(token.NoPos, nil, "length", types.Typ[types.Uintptr]), types.NewVar(token.NoPos, nil, "sliceOf", types.Typ[types.UnsafePointer]), + types.NewVar(token.NoPos, nil, "layout", types.Typ[types.UnsafePointer]), ) case *types.Map: typeFieldTypes = append(typeFieldTypes, @@ -291,6 +292,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { types.NewVar(token.NoPos, nil, "ptrTo", types.Typ[types.UnsafePointer]), types.NewVar(token.NoPos, nil, "elementType", types.Typ[types.UnsafePointer]), types.NewVar(token.NoPos, nil, "keyType", types.Typ[types.UnsafePointer]), + types.NewVar(token.NoPos, nil, "hashmapTypeInfo", types.Typ[types.UnsafePointer]), ) case *types.Struct: typeFieldTypes = append(typeFieldTypes, @@ -299,6 +301,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { types.NewVar(token.NoPos, nil, "pkgpath", types.Typ[types.UnsafePointer]), types.NewVar(token.NoPos, nil, "size", types.Typ[types.Uint32]), types.NewVar(token.NoPos, nil, "numFields", types.Typ[types.Uint16]), + types.NewVar(token.NoPos, nil, "layout", types.Typ[types.UnsafePointer]), types.NewVar(token.NoPos, nil, "fields", types.NewArray(c.getRuntimeType("structField"), int64(typ.NumFields()))), ) if len(methods) > 0 { @@ -418,6 +421,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { c.getTypeCode(typ.Elem()), // elementType llvm.ConstInt(c.uintptrType, uint64(typ.Len()), false), // length c.getTypeCode(types.NewSlice(typ.Elem())), // slicePtr + c.createObjectLayout(c.getLLVMType(typ), token.NoPos), // layout } case *types.Map: typeFields = []llvm.Value{ @@ -425,6 +429,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { c.getTypeCode(types.NewPointer(typ)), // ptrTo c.getTypeCode(typ.Elem()), // elem c.getTypeCode(typ.Key()), // key + c.getHashmapTypeInfo(typ, token.NoPos), // hashmapTypeInfo } case *types.Struct: var pkgpath string @@ -450,6 +455,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { pkgPathPtr, llvm.ConstInt(c.ctx.Int32Type(), uint64(size), false), // size llvm.ConstInt(c.ctx.Int16Type(), uint64(typ.NumFields()), false), // numFields + c.createObjectLayout(llvmStructType, token.NoPos), // layout } structFieldType := c.getLLVMRuntimeType("structField") @@ -510,7 +516,7 @@ func (c *compilerContext) getTypeCode(typ types.Type) llvm.Value { typeFields = []llvm.Value{c.getTypeCode(types.NewPointer(typ))} // TODO: params, return values, etc } - // Prepend metadata byte. + // Prepend the common RawType field. typeFields = append([]llvm.Value{ llvm.ConstInt(c.ctx.Int8Type(), uint64(metabyte), false), }, typeFields...) diff --git a/compiler/map.go b/compiler/map.go index 3481ae90ac..ee3c23821c 100644 --- a/compiler/map.go +++ b/compiler/map.go @@ -13,6 +13,12 @@ import ( const hashArrayUnrollLimit = 4 +const ( + hashmapBucketSlots = 8 + hashmapMaxKeySize = 128 + hashmapMaxValueSize = 128 +) + // createMakeMap creates a new map object (runtime.hashmap) by allocating and // initializing an appropriately sized object. func (b *builder) createMakeMap(expr *ssa.MakeMap) (llvm.Value, error) { @@ -25,6 +31,8 @@ func (b *builder) createMakeMap(expr *ssa.MakeMap) (llvm.Value, error) { valueSize := b.targetData.TypeAllocSize(llvmValueType) llvmKeySize := llvm.ConstInt(b.uintptrType, keySize, false) llvmValueSize := llvm.ConstInt(b.uintptrType, valueSize, false) + mapLayout := b.getHashmapTypeInfo(mapType, expr.Pos()) + sizeHint := llvm.ConstInt(b.uintptrType, 8, false) if expr.Reserve != nil { sizeHint = b.getValue(expr.Reserve, getPos(expr)) @@ -54,11 +62,72 @@ func (b *builder) createMakeMap(expr *ssa.MakeMap) (llvm.Value, error) { hashmap := b.createRuntimeCall("hashmapMakeGeneric", []llvm.Value{ llvmKeySize, llvmValueSize, sizeHint, + mapLayout, hashFn, equalFn, }, "") return hashmap, nil } +func (c *compilerContext) getHashmapTypeInfo(mapType *types.Map, pos token.Pos) llvm.Value { + llvmKeyType := c.getLLVMType(mapType.Key().Underlying()) + llvmValueType := c.getLLVMType(mapType.Elem().Underlying()) + keySize := c.targetData.TypeAllocSize(llvmKeyType) + valueSize := c.targetData.TypeAllocSize(llvmValueType) + keyLayout := c.createObjectLayout(llvmKeyType, pos) + valueLayout := c.createObjectLayout(llvmValueType, pos) + + llvmKeySlotType := llvmKeyType + if keySize > hashmapMaxKeySize { + llvmKeySlotType = c.dataPtrType + } + llvmValueSlotType := llvmValueType + if valueSize > hashmapMaxValueSize { + llvmValueSlotType = c.dataPtrType + } + + // Keep this in sync with runtime.hashmapBucket and + // runtime.hashmapBucketHeaderSize. + pointerSize := c.targetData.TypeAllocSize(c.dataPtrType) + headerSize := (uint64(8) + pointerSize + 7) &^ 7 + headerPadding := headerSize - uint64(8) - pointerSize + bucketFields := []llvm.Type{ + llvm.ArrayType(c.ctx.Int8Type(), hashmapBucketSlots), + c.dataPtrType, + } + if headerPadding != 0 { + bucketFields = append(bucketFields, llvm.ArrayType(c.ctx.Int8Type(), int(headerPadding))) + } + bucketFields = append(bucketFields, + llvm.ArrayType(llvmKeySlotType, hashmapBucketSlots), + llvm.ArrayType(llvmValueSlotType, hashmapBucketSlots), + ) + bucketType := c.ctx.StructType(bucketFields, true) + bucketSize := headerSize + + c.targetData.TypeAllocSize(llvmKeySlotType)*hashmapBucketSlots + + c.targetData.TypeAllocSize(llvmValueSlotType)*hashmapBucketSlots + if c.targetData.TypeAllocSize(bucketType) != bucketSize { + panic("compiler hashmap bucket layout does not match runtime") + } + bucketLayout := c.createObjectLayout(bucketType, pos) + mapLayoutName := "runtime.hashmapType:" + + hashmapCanonicalTypeName(mapType.Key()) + ":" + + hashmapCanonicalTypeName(mapType.Elem()) + mapLayout := c.mod.NamedGlobal(mapLayoutName) + if mapLayout.IsNil() { + initializer := c.ctx.ConstStruct([]llvm.Value{ + keyLayout, + valueLayout, + bucketLayout, + }, false) + mapLayout = llvm.AddGlobal(c.mod, initializer.Type(), mapLayoutName) + mapLayout.SetInitializer(initializer) + mapLayout.SetGlobalConstant(true) + mapLayout.SetUnnamedAddr(true) + mapLayout.SetLinkage(llvm.LinkOnceODRLinkage) + } + return mapLayout +} + // getRuntimeFunctionValue returns a TinyGo function value (with nil context) // for the named runtime function. func (b *builder) getRuntimeFunctionValue(name string, sig *types.Signature) llvm.Value { diff --git a/compiler/testdata/go1.21.ll b/compiler/testdata/go1.21.ll index 664309518e..00d7146dab 100644 --- a/compiler/testdata/go1.21.ll +++ b/compiler/testdata/go1.21.ll @@ -166,13 +166,13 @@ entry: } ; Function Attrs: nounwind -define hidden void @main.clearMap(ptr dereferenceable_or_null(48) %m, ptr %context) unnamed_addr #1 { +define hidden void @main.clearMap(ptr dereferenceable_or_null(52) %m, ptr %context) unnamed_addr #1 { entry: call void @runtime.hashmapClear(ptr %m, ptr undef) #4 ret void } -declare void @runtime.hashmapClear(ptr dereferenceable_or_null(48), ptr) #0 +declare void @runtime.hashmapClear(ptr dereferenceable_or_null(52), ptr) #0 attributes #0 = { "target-features"="+bulk-memory,+bulk-memory-opt,+call-indirect-overlong,+mutable-globals,+nontrapping-fptoint,+sign-ext,-multivalue,-reference-types" } attributes #1 = { nounwind "target-features"="+bulk-memory,+bulk-memory-opt,+call-indirect-overlong,+mutable-globals,+nontrapping-fptoint,+sign-ext,-multivalue,-reference-types" } diff --git a/compiler/testdata/go1.27.ll b/compiler/testdata/go1.27.ll index d52ba19a38..522cb54a93 100644 --- a/compiler/testdata/go1.27.ll +++ b/compiler/testdata/go1.27.ll @@ -14,7 +14,7 @@ target triple = "wasm32-unknown-wasi" @"main$string" = internal unnamed_addr constant [18 x i8] c"main.genericMethod", align 1 @"main$string.1" = internal unnamed_addr constant [7 x i8] c"Regular", align 1 @"pointer:named:main.genericMethod$methodset" = linkonce_odr unnamed_addr constant { i32, [1 x ptr], { ptr } } { i32 1, [1 x ptr] [ptr @"reflect/methods.Regular:func:{basic:int}{basic:int}"], { ptr } { ptr @"(*main.genericMethod).Regular" } } -@"reflect/types.type:struct:{}" = linkonce_odr constant { i8, i16, ptr, ptr, i32, i16, [0 x %runtime.structField] } { i8 90, i16 0, ptr @"reflect/types.type:pointer:struct:{}", ptr @"reflect/types.type.pkgpath.empty", i32 0, i16 0, [0 x %runtime.structField] zeroinitializer }, align 4 +@"reflect/types.type:struct:{}" = linkonce_odr constant { i8, i16, ptr, ptr, i32, i16, ptr, [0 x %runtime.structField] } { i8 90, i16 0, ptr @"reflect/types.type:pointer:struct:{}", ptr @"reflect/types.type.pkgpath.empty", i32 0, i16 0, ptr inttoptr (i32 3 to ptr), [0 x %runtime.structField] zeroinitializer }, align 4 @"reflect/types.type.pkgpath.empty" = linkonce_odr unnamed_addr constant [1 x i8] zeroinitializer, align 1 @"reflect/types.type:pointer:struct:{}" = linkonce_odr constant { i8, i16, ptr } { i8 -43, i16 0, ptr @"reflect/types.type:struct:{}" }, align 4 @"named:main.genericMethod$methodset" = linkonce_odr unnamed_addr constant { i32, [1 x ptr], { ptr } } { i32 1, [1 x ptr] [ptr @"reflect/methods.Regular:func:{basic:int}{basic:int}"], { ptr } { ptr @"(main.genericMethod).Regular$invoke" } } diff --git a/compiler/testdata/large.ll b/compiler/testdata/large.ll index 27da846060..8fa7c83f40 100644 --- a/compiler/testdata/large.ll +++ b/compiler/testdata/large.ll @@ -9,6 +9,7 @@ target triple = "wasm32-unknown-wasi" @"runtime/gc.layout:258-000000000000000000000000000000000000000000000000000000000000000002" = linkonce_odr unnamed_addr constant { i32, [33 x i8] } { i32 258, [33 x i8] c"\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\00\02" } @"reflect/types.typeid:named:main.largeValue" = external constant i8 +@"runtime.hashmapType:[1025]byte:[1025]byte" = linkonce_odr unnamed_addr constant { ptr, ptr, ptr } { ptr inttoptr (i32 3 to ptr), ptr inttoptr (i32 3 to ptr), ptr inttoptr (i32 67108137 to ptr) } @llvm.used = appending global [15 x ptr] [ptr @"(main.largeReceiver).makeLargeValue", ptr @"(main.largeReceiver).readLargeValue", ptr @main.makeLargeValue, ptr @main.makeZeroLargeValue, ptr @main.readLargeValue, ptr @main.deferLargeValue, ptr @main.goLargeValue, ptr @main.makeLargeResults, ptr @main.makeTwoLargeResults, ptr @main.makeMixedLargeResults, ptr @main.chooseLargeValue, ptr @main.makePointerLargeValue, ptr @main.useLargeMap, ptr @main.useLargeChannel, ptr @main.selectLargeChannel] @"main$string" = internal unnamed_addr constant [31 x i8] c"blocking select matched no case", align 1 @"main$pack" = internal unnamed_addr constant { %runtime._string } { %runtime._string { ptr @"main$string", i32 31 } } @@ -350,7 +351,7 @@ declare void @llvm.memset.p0.i32(ptr writeonly captures(none), i8, i32, i1 immar define hidden i8 @main.useLargeMap(ptr readonly dereferenceable_or_null(1025) %key, ptr readonly dereferenceable_or_null(1025) %value, ptr %context) unnamed_addr #1 { entry: %stackalloc = alloca i8, align 1 - %0 = call ptr @runtime.hashmapMakeGeneric(i32 1025, i32 1025, i32 1, ptr null, ptr nonnull @runtime.hash32, ptr null, ptr nonnull @runtime.memequal, ptr undef) #9 + %0 = call ptr @runtime.hashmapMakeGeneric(i32 1025, i32 1025, i32 1, ptr nonnull @"runtime.hashmapType:[1025]byte:[1025]byte", ptr null, ptr nonnull @runtime.hash32, ptr null, ptr nonnull @runtime.memequal, ptr undef) #9 call void @runtime.trackPointer(ptr %0, ptr nonnull %stackalloc, ptr undef) #9 call void @runtime.hashmapBinarySet(ptr %0, ptr %key, ptr %value, ptr undef) #9 %result = call align 1 dereferenceable(1025) ptr @runtime.alloc(i32 1025, ptr nonnull inttoptr (i32 3 to ptr), ptr undef) #9 @@ -381,11 +382,11 @@ declare i32 @runtime.hash32(ptr, i32, i32, ptr) #0 declare i1 @runtime.memequal(ptr, ptr, i32, ptr) #0 -declare ptr @runtime.hashmapMakeGeneric(i32, i32, i32, ptr, ptr, ptr, ptr, ptr) #0 +declare ptr @runtime.hashmapMakeGeneric(i32, i32, i32, ptr, ptr, ptr, ptr, ptr, ptr) #0 -declare void @runtime.hashmapBinarySet(ptr dereferenceable_or_null(48), ptr, ptr, ptr) #0 +declare void @runtime.hashmapBinarySet(ptr dereferenceable_or_null(52), ptr, ptr, ptr) #0 -declare i1 @runtime.hashmapBinaryGet(ptr dereferenceable_or_null(48), ptr, ptr, i32, ptr) #0 +declare i1 @runtime.hashmapBinaryGet(ptr dereferenceable_or_null(52), ptr, ptr, i32, ptr) #0 ; Function Attrs: nounwind define hidden i8 @main.useLargeChannel(ptr dereferenceable_or_null(36) %ch, ptr readonly dereferenceable_or_null(1025) %value, ptr %context) unnamed_addr #1 { diff --git a/compiler/testdata/zeromap.ll b/compiler/testdata/zeromap.ll index be8b924c21..b8c42bee38 100644 --- a/compiler/testdata/zeromap.ll +++ b/compiler/testdata/zeromap.ll @@ -14,7 +14,7 @@ entry: } ; Function Attrs: noinline nounwind -define hidden i32 @main.testZeroGet(ptr dereferenceable_or_null(48) %m, i1 %s.b1, i32 %s.i, i1 %s.b2, ptr %context) unnamed_addr #2 { +define hidden i32 @main.testZeroGet(ptr dereferenceable_or_null(52) %m, i1 %s.b1, i32 %s.i, i1 %s.b2, ptr %context) unnamed_addr #2 { entry: %hashmap.key = alloca %main.hasPadding, align 8 %hashmap.value = alloca i32, align 4 @@ -34,13 +34,13 @@ entry: ; Function Attrs: nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) declare void @llvm.lifetime.start.p0(i64 immarg, ptr nocapture) #3 -declare i1 @runtime.hashmapGenericGet(ptr dereferenceable_or_null(48), ptr nocapture, ptr nocapture, i32, ptr) #0 +declare i1 @runtime.hashmapGenericGet(ptr dereferenceable_or_null(52), ptr nocapture, ptr nocapture, i32, ptr) #0 ; Function Attrs: nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) declare void @llvm.lifetime.end.p0(i64 immarg, ptr nocapture) #3 ; Function Attrs: noinline nounwind -define hidden void @main.testZeroSet(ptr dereferenceable_or_null(48) %m, i1 %s.b1, i32 %s.i, i1 %s.b2, ptr %context) unnamed_addr #2 { +define hidden void @main.testZeroSet(ptr dereferenceable_or_null(52) %m, i1 %s.b1, i32 %s.i, i1 %s.b2, ptr %context) unnamed_addr #2 { entry: %hashmap.key = alloca %main.hasPadding, align 8 %hashmap.value = alloca i32, align 4 @@ -57,10 +57,10 @@ entry: ret void } -declare void @runtime.hashmapGenericSet(ptr dereferenceable_or_null(48), ptr nocapture, ptr nocapture, ptr) #0 +declare void @runtime.hashmapGenericSet(ptr dereferenceable_or_null(52), ptr nocapture, ptr nocapture, ptr) #0 ; Function Attrs: noinline nounwind -define hidden i32 @main.testZeroArrayGet(ptr dereferenceable_or_null(48) %m, [2 x %main.hasPadding] %s, ptr %context) unnamed_addr #2 { +define hidden i32 @main.testZeroArrayGet(ptr dereferenceable_or_null(52) %m, [2 x %main.hasPadding] %s, ptr %context) unnamed_addr #2 { entry: %hashmap.key = alloca [2 x %main.hasPadding], align 8 %hashmap.value = alloca i32, align 4 @@ -79,7 +79,7 @@ entry: } ; Function Attrs: noinline nounwind -define hidden void @main.testZeroArraySet(ptr dereferenceable_or_null(48) %m, [2 x %main.hasPadding] %s, ptr %context) unnamed_addr #2 { +define hidden void @main.testZeroArraySet(ptr dereferenceable_or_null(52) %m, [2 x %main.hasPadding] %s, ptr %context) unnamed_addr #2 { entry: %hashmap.key = alloca [2 x %main.hasPadding], align 8 %hashmap.value = alloca i32, align 4 diff --git a/interp/memory.go b/interp/memory.go index 7c1eb2d335..2f00e1bc79 100644 --- a/interp/memory.go +++ b/interp/memory.go @@ -1278,11 +1278,10 @@ func (r *runner) readObjectLayout(layoutValue value) (uint64, *big.Int) { // integer value, or can be nil. ptr, err := layoutValue.asPointer(r) if err == errIntegerAsPointer { - // It's an integer, which means it's a small object or unknown. + // It's an integer, which means it's a small object. layout := layoutValue.Uint(r) if layout == 0 { - // Nil pointer, which means the layout is unknown. - return 0, nil + panic("runtime.alloc called without a GC layout") } if layout%2 != 1 { // Sanity check: the least significant bit must be set. This is how @@ -1331,11 +1330,6 @@ func (r *runner) readObjectLayout(layoutValue value) (uint64, *big.Int) { // have some additional repetition, for example in the buffer of a slice. func (r *runner) getLLVMTypeFromLayout(layoutValue value) llvm.Type { objectSizeWords, bitmap := r.readObjectLayout(layoutValue) - if bitmap == nil { - // No information available. - return llvm.Type{} - } - if bitmap.BitLen() == 0 { // There are no pointers in this object, so treat this as a raw byte // buffer. This is important because objects without pointers may have diff --git a/interp/testdata/alloc.ll b/interp/testdata/alloc.ll index 82fbb5b276..3e3c1ca228 100644 --- a/interp/testdata/alloc.ll +++ b/interp/testdata/alloc.ll @@ -11,6 +11,7 @@ target triple = "wasm32--wasi" @layout3 = global ptr null @layout4 = global ptr null @bigobj1 = global ptr null +@pointerFree10 = global ptr null declare ptr @runtime.alloc(i32, ptr) unnamed_addr @@ -49,5 +50,9 @@ define internal void @main.init() unnamed_addr { ; Large object that needs to be stored in a separate global. %bigobj1 = call ptr @runtime.alloc(i32 248, ptr @"runtime/gc.layout:62-2000000000000001") store ptr %bigobj1, ptr @bigobj1 + + ; Another pointer-free object. + %pointerFree10 = call ptr @runtime.alloc(i32 10, ptr inttoptr (i32 3 to ptr)) + store ptr %pointerFree10, ptr @pointerFree10 ret void } diff --git a/interp/testdata/alloc.out.ll b/interp/testdata/alloc.out.ll index b9da6291f6..641bad4ddc 100644 --- a/interp/testdata/alloc.out.ll +++ b/interp/testdata/alloc.out.ll @@ -10,6 +10,7 @@ target triple = "wasm32--wasi" @layout3 = local_unnamed_addr global ptr @"main$alloc.6" @layout4 = local_unnamed_addr global ptr @"main$alloc.7" @bigobj1 = local_unnamed_addr global ptr @"main$alloc.8" +@pointerFree10 = local_unnamed_addr global ptr @"main$alloc.9" @"main$alloc" = internal global [12 x i8] zeroinitializer, align 4 @"main$alloc.1" = internal global [7 x i8] zeroinitializer, align 4 @"main$alloc.2" = internal global [3 x i8] zeroinitializer, align 4 @@ -19,6 +20,7 @@ target triple = "wasm32--wasi" @"main$alloc.6" = internal global { ptr, ptr, ptr, i32, i32, ptr, ptr, i32, i32, i32, i32, i32, i32, ptr, ptr, i32, i32, i32, ptr, ptr, i32, i32, ptr, i32, i32, ptr } zeroinitializer, align 4 @"main$alloc.7" = internal global [3 x { ptr, ptr, ptr, i32, i32, ptr, ptr, i32, i32, i32, i32, i32, i32, ptr, ptr, i32, i32, i32, ptr, ptr, i32, i32, ptr, i32, i32, ptr }] zeroinitializer, align 4 @"main$alloc.8" = internal global { ptr, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, ptr } zeroinitializer, align 4 +@"main$alloc.9" = internal global [10 x i8] zeroinitializer, align 4 define void @runtime.initAll() unnamed_addr { ret void diff --git a/src/internal/gclayout/gclayout.go b/src/internal/gclayout/gclayout.go index d6235889ff..3ed750d138 100644 --- a/src/internal/gclayout/gclayout.go +++ b/src/internal/gclayout/gclayout.go @@ -17,10 +17,15 @@ const ( sizeShift = sizeBits + 1 - NoPtrs = Layout((0 << sizeShift) | (1 << 1) | 1) - Pointer = Layout((1 << sizeShift) | ((unsafe.Sizeof(unsafe.Pointer(nil)) / ptrAlign) << 1) | 1) - String = Layout((1 << sizeShift) | ((unsafe.Sizeof("") / ptrAlign) << 1) | 1) - Slice = Layout((1 << sizeShift) | ((unsafe.Sizeof([]byte{}) / ptrAlign) << 1) | 1) + NoPtrs = Layout((0 << sizeShift) | (1 << 1) | 1) + Pointer = Layout((1 << sizeShift) | ((unsafe.Sizeof(unsafe.Pointer(nil)) / ptrAlign) << 1) | 1) + PointerPair = Layout((3 << sizeShift) | ((2 * unsafe.Sizeof(unsafe.Pointer(nil)) / ptrAlign) << 1) | 1) + String = Layout((1 << sizeShift) | ((unsafe.Sizeof("") / ptrAlign) << 1) | 1) + Slice = Layout((1 << sizeShift) | ((unsafe.Sizeof([]byte{}) / ptrAlign) << 1) | 1) + + // Conservative is reserved for stack storage, which does not have an + // ordinary Go object layout. + Conservative = Layout(2) ) func (l Layout) AsPtr() unsafe.Pointer { return unsafe.Pointer(l) } diff --git a/src/internal/reflectlite/type.go b/src/internal/reflectlite/type.go index 5ced5d3573..0189530a70 100644 --- a/src/internal/reflectlite/type.go +++ b/src/internal/reflectlite/type.go @@ -166,6 +166,11 @@ type RawType struct { meta uint8 // metadata byte, contains kind and flags (see constants above) } +type basicType struct { + RawType + ptrTo *RawType +} + // All types that have an element type: named, chan, slice, array, map (but not // pointer because it doesn't have ptrTo). type elemType struct { @@ -200,6 +205,7 @@ type arrayType struct { elem *RawType arrayLen uintptr slicePtr *RawType + layout unsafe.Pointer } type mapType struct { @@ -208,6 +214,7 @@ type mapType struct { ptrTo *RawType elem *RawType key *RawType + typeInfo unsafe.Pointer } // namedType is the type descriptor for named types. The numMethod field uses @@ -243,6 +250,7 @@ type structType struct { pkgpath *byte size uint32 numField uint16 + layout unsafe.Pointer fields [1]structField // the remaining fields are all of type structField // methods methodSet follows after fields, only when numMethod & numMethodHasMethodSet != 0 } @@ -298,6 +306,8 @@ func pointerTo(t *RawType) *RawType { } switch t.Kind() { + case Bool, Int, Int8, Int16, Int32, Int64, Uint, Uint8, Uint16, Uint32, Uint64, Uintptr, Complex64, Complex128, Float32, Float64, String, UnsafePointer: + return (*basicType)(unsafe.Pointer(t)).ptrTo case Pointer: if tag := t.ptrtag(); tag < 3 { return (*RawType)(unsafe.Add(unsafe.Pointer(t), 1)) @@ -306,6 +316,8 @@ func pointerTo(t *RawType) *RawType { // TODO(dgryski): This is blocking https://github.com/tinygo-org/tinygo/issues/3131 // We need to be able to create types that match existing types to prevent typecode equality. panic("reflect: cannot make *****T type") + case Interface, Func: + return (*interfaceType)(unsafe.Pointer(t)).ptrTo case Struct: return (*structType)(unsafe.Pointer(t)).ptrTo default: @@ -729,6 +741,7 @@ func (t *RawType) Align() int { } func (r *RawType) gcLayout() unsafe.Pointer { + r = r.underlying() kind := r.Kind() if kind < String { @@ -736,16 +749,26 @@ func (r *RawType) gcLayout() unsafe.Pointer { } switch kind { - case Pointer, UnsafePointer, Chan, Map: - return gclayout.Pointer.AsPtr() case String: return gclayout.String.AsPtr() + case UnsafePointer, Chan, Pointer, Map: + return gclayout.Pointer.AsPtr() + case Interface, Func: + return gclayout.PointerPair.AsPtr() case Slice: return gclayout.Slice.AsPtr() + case Array: + return (*arrayType)(unsafe.Pointer(r)).layout + case Struct: + return (*structType)(unsafe.Pointer(r)).layout + default: + panic("reflect: invalid GC layout kind") } +} - // Unknown (for now); let the conservative pointer scanning handle it - return nil +func (r *RawType) hashmapTypeInfo() unsafe.Pointer { + r = r.underlying() + return (*mapType)(unsafe.Pointer(r)).typeInfo } // FieldAlign returns the alignment if this type is used in a struct field. It diff --git a/src/internal/reflectlite/value.go b/src/internal/reflectlite/value.go index b353b1a92a..dc66b2cde7 100644 --- a/src/internal/reflectlite/value.go +++ b/src/internal/reflectlite/value.go @@ -1,6 +1,7 @@ package reflectlite import ( + "internal/gclayout" "math" "unsafe" ) @@ -1644,7 +1645,7 @@ func makeInt(flags valueFlags, bits uint64, t *RawType) Value { ptr := unsafe.Pointer(&v.value) if size > unsafe.Sizeof(uintptr(0)) { - ptr = alloc(size, nil) + ptr = alloc(size, gclayout.NoPtrs.AsPtr()) v.value = ptr } @@ -1671,7 +1672,7 @@ func makeFloat(flags valueFlags, f float64, t *RawType) Value { ptr := unsafe.Pointer(&v.value) if size > unsafe.Sizeof(uintptr(0)) { - ptr = alloc(size, nil) + ptr = alloc(size, gclayout.NoPtrs.AsPtr()) v.value = ptr } @@ -1703,7 +1704,7 @@ func makeComplex(flags valueFlags, f complex128, t *RawType) Value { ptr := unsafe.Pointer(&v.value) if size > unsafe.Sizeof(uintptr(0)) { - ptr = alloc(size, nil) + ptr = alloc(size, gclayout.NoPtrs.AsPtr()) v.value = ptr } @@ -1834,7 +1835,7 @@ func Zero(typ Type) Value { return Value{ typecode: typ.(*RawType), - value: alloc(size, nil), + value: alloc(size, typ.(*RawType).gcLayout()), flags: valueFlagExported | valueFlagRO, } } @@ -1844,7 +1845,7 @@ func Zero(typ Type) Value { func New(typ Type) Value { return Value{ typecode: pointerTo(typ.(*RawType)), - value: alloc(typ.Size(), nil), + value: alloc(typ.Size(), typ.(*RawType).gcLayout()), flags: valueFlagExported, } } @@ -2203,10 +2204,10 @@ func (v Value) FieldByNameFunc(match func(string) bool) Value { } //go:linkname hashmapMake runtime.hashmapMake -func hashmapMake(keySize, valueSize uintptr, sizeHint uintptr, alg uint8) unsafe.Pointer +func hashmapMake(keySize, valueSize uintptr, sizeHint uintptr, typeInfo unsafe.Pointer, alg uint8) unsafe.Pointer //go:linkname hashmapMakeReflect runtime.hashmapMakeReflect -func hashmapMakeReflect(keySize, valueSize, sizeHint uintptr, keyType unsafe.Pointer) unsafe.Pointer +func hashmapMakeReflect(keySize, valueSize, sizeHint uintptr, typeInfo, keyType unsafe.Pointer) unsafe.Pointer // MakeMapWithSize creates a new map with the specified type and initial space // for approximately n elements. @@ -2228,18 +2229,19 @@ func MakeMapWithSize(typ Type, n int) Value { key := typ.Key().(*RawType) val := typ.Elem().(*RawType) + typeInfo := typ.(*RawType).hashmapTypeInfo() var m unsafe.Pointer if key.Kind() == String { - m = hashmapMake(key.Size(), val.Size(), uintptr(n), hashmapAlgorithmString) + m = hashmapMake(key.Size(), val.Size(), uintptr(n), typeInfo, hashmapAlgorithmString) } else if key.isBinary() { - m = hashmapMake(key.Size(), val.Size(), uintptr(n), hashmapAlgorithmBinary) + m = hashmapMake(key.Size(), val.Size(), uintptr(n), typeInfo, hashmapAlgorithmBinary) } else { // Composite key type (struct with strings, floats, etc.). // Use runtime-generated hash/equal closures that walk the // type structure, matching the compiler-generated functions. - m = hashmapMakeReflect(key.Size(), val.Size(), uintptr(n), unsafe.Pointer(key)) + m = hashmapMakeReflect(key.Size(), val.Size(), uintptr(n), typeInfo, unsafe.Pointer(key)) } return Value{ diff --git a/src/internal/task/task_asyncify.go b/src/internal/task/task_asyncify.go index d7d9a6de42..8edb5b15e1 100644 --- a/src/internal/task/task_asyncify.go +++ b/src/internal/task/task_asyncify.go @@ -3,6 +3,7 @@ package task import ( + "internal/gclayout" "unsafe" ) @@ -66,7 +67,7 @@ func (s *state) initialize(fn uintptr, args unsafe.Pointer, stackSize uintptr) { s.args = args // Create a stack. - stack := runtime_alloc(stackSize, nil) + stack := runtime_alloc(stackSize, gclayout.Conservative.AsPtr()) // Set up the stack canary, a random number that should be checked when // switching from the task back to the scheduler. The stack canary pointer diff --git a/src/internal/task/task_stack.go b/src/internal/task/task_stack.go index 23f3b9097f..eaab211bbd 100644 --- a/src/internal/task/task_stack.go +++ b/src/internal/task/task_stack.go @@ -3,6 +3,7 @@ package task import ( + "internal/gclayout" "unsafe" ) @@ -36,7 +37,7 @@ func taskExit() { // initialize the state and prepare to call the specified function with the specified argument bundle. func (s *state) initialize(fn uintptr, args unsafe.Pointer, stackSize uintptr) { // Create a stack. - stack := runtime_alloc(stackSize, nil) + stack := runtime_alloc(stackSize, gclayout.Conservative.AsPtr()) // Set up the stack canary, a random number that should be checked when // switching from the task back to the scheduler. The stack canary pointer diff --git a/src/runtime/arch_tinygowasm_malloc.go b/src/runtime/arch_tinygowasm_malloc.go index df824881e2..9e2042b284 100644 --- a/src/runtime/arch_tinygowasm_malloc.go +++ b/src/runtime/arch_tinygowasm_malloc.go @@ -2,27 +2,30 @@ package runtime -import "unsafe" +import ( + "internal/task" + "unsafe" +) // The below functions override the default allocator of wasi-libc. This ensures // code linked from other languages can allocate memory without colliding with // our GC allocations. -// Map of allocations, where the key is the allocated pointer and the value is -// the size of the allocation. -// TODO: make this a map[unsafe.Pointer]uintptr, since that results in slightly -// smaller binaries. But for that to work, unsafe.Pointer needs to be seen as a -// binary key (which it is not at the moment). -// See https://github.com/tinygo-org/tinygo/pull/4898 for details. -var allocs = make(map[*byte]uintptr) +// Map of allocations, where the key is the allocation address and the value is +// its size. Integer keys intentionally do not act as GC roots: manual +// allocations are retained by the allocator until free. +var allocs = make(map[uintptr]uintptr) +var allocsLock task.PMutex //export malloc func libc_malloc(size uintptr) unsafe.Pointer { if size == 0 { return nil } - ptr := alloc(size, nil) - allocs[(*byte)(ptr)] = size + ptr := allocManual(size) + allocsLock.Lock() + allocs[uintptr(ptr)] = size + allocsLock.Unlock() return ptr } @@ -31,16 +34,22 @@ func libc_free(ptr unsafe.Pointer) { if ptr == nil { return } - if _, ok := allocs[(*byte)(ptr)]; ok { - delete(allocs, (*byte)(ptr)) + allocsLock.Lock() + if _, ok := allocs[uintptr(ptr)]; ok { + delete(allocs, uintptr(ptr)) + allocsLock.Unlock() + freeManual(ptr) } else { + allocsLock.Unlock() runtimeFatal("free: invalid pointer") } } //export calloc func libc_calloc(nmemb, size uintptr) unsafe.Pointer { - // No difference between calloc and malloc. + if size != 0 && nmemb > ^uintptr(0)/size { + return nil + } return libc_malloc(nmemb * size) } @@ -54,19 +63,24 @@ func libc_realloc(oldPtr unsafe.Pointer, size uintptr) unsafe.Pointer { // It's hard to optimize this to expand the current buffer with our GC, but // it is theoretically possible. For now, just always allocate fresh. // TODO: we could skip this if the new allocation is smaller than the old. - ptr := alloc(size, nil) + ptr := allocManual(size) + allocsLock.Lock() if oldPtr != nil { - if oldSize, ok := allocs[(*byte)(oldPtr)]; ok { + if oldSize, ok := allocs[uintptr(oldPtr)]; ok { oldBuf := unsafe.Slice((*byte)(oldPtr), oldSize) newBuf := unsafe.Slice((*byte)(ptr), size) copy(newBuf, oldBuf) - delete(allocs, (*byte)(oldPtr)) + delete(allocs, uintptr(oldPtr)) } else { + allocsLock.Unlock() runtimeFatal("realloc: invalid pointer") } } - - allocs[(*byte)(ptr)] = size + allocs[uintptr(ptr)] = size + allocsLock.Unlock() + if oldPtr != nil { + freeManual(oldPtr) + } return ptr } diff --git a/src/runtime/baremetal.go b/src/runtime/baremetal.go index 6dd29e490d..d2963a4964 100644 --- a/src/runtime/baremetal.go +++ b/src/runtime/baremetal.go @@ -11,18 +11,20 @@ import ( func libc_malloc(size uintptr) unsafe.Pointer { // Note: this zeroes the returned buffer which is not necessary. // The same goes for bytealg.MakeNoZero. - return alloc(size, nil) + return allocManual(size) } //export calloc func libc_calloc(nmemb, size uintptr) unsafe.Pointer { - // No difference between calloc and malloc. + if size != 0 && nmemb > ^uintptr(0)/size { + return nil + } return libc_malloc(nmemb * size) } //export free func libc_free(ptr unsafe.Pointer) { - free(ptr) + freeManual(ptr) } //export runtime_putchar diff --git a/src/runtime/chan.go b/src/runtime/chan.go index a85e9b6617..f425daf5d5 100644 --- a/src/runtime/chan.go +++ b/src/runtime/chan.go @@ -137,11 +137,11 @@ type chanSelectState struct { value unsafe.Pointer } -func chanMake(elementSize uintptr, bufSize uintptr) *channel { +func chanMake(elementSize uintptr, bufSize uintptr, elementLayout unsafe.Pointer) *channel { return &channel{ elementSize: elementSize, bufCap: bufSize, - buf: alloc(elementSize*bufSize, nil), + buf: alloc(elementSize*bufSize, elementLayout), } } diff --git a/src/runtime/gc_blocks.go b/src/runtime/gc_blocks.go index 78387cb312..fa0241d53f 100644 --- a/src/runtime/gc_blocks.go +++ b/src/runtime/gc_blocks.go @@ -31,6 +31,7 @@ package runtime // Moss. import ( + "internal/gclayout" "internal/reflectlite" "internal/task" "runtime/interrupt" @@ -203,8 +204,9 @@ func (b gcBlock) free() { // objHeader is a structure appended to every heap object to hold metadata. type objHeader struct { - // next is the next object to scan after this. - next *objHeader + // next links the GC scan list. Manual allocations remain permanently marked + // and use the otherwise invalid value 1 as an until-free marker. + next uintptr // layout holds the layout bitmap used to find pointers in the object. layout gcLayout @@ -480,6 +482,7 @@ func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer { // Create the object header. size -= unsafe.Sizeof(objHeader{}) header := (*objHeader)(unsafe.Add(pointer, size)) + header.next = 0 header.layout = parseGCLayout(layout) // We've claimed this allocation, now we can unlock the heap. @@ -498,42 +501,38 @@ func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer { return pointer } -func realloc(ptr unsafe.Pointer, size uintptr) unsafe.Pointer { +// allocManual allocates pointer-free memory that remains live until freeManual. +func allocManual(size uintptr) unsafe.Pointer { + if size == 0 { + return alloc_zero(size, gclayout.NoPtrs.AsPtr()) + } + ptr := alloc(size, gclayout.NoPtrs.AsPtr()) + + gcLock.Lock() + head := blockFromAddr(uintptr(ptr)).findHead() + head.setState(blockStateMark) + header := (*objHeader)(unsafe.Add(head.pointer(), bytesPerBlock-unsafe.Sizeof(objHeader{}))) + header.next = 1 + gcLock.Unlock() + return ptr +} + +func free(ptr unsafe.Pointer) { if ptr == nil { - return alloc(size, nil) + return } - // Find the first block of the original allocation. + gcLock.Lock() firstBlock := blockFromAddr(uintptr(ptr)) - - // Find the last block of the original allocation. lastBlock := firstBlock.findHead() - - // Calculate the size of the original allocation body. - oldSize := uintptr(lastBlock-firstBlock)*bytesPerBlock + (bytesPerBlock - unsafe.Sizeof(objHeader{})) - - if size <= oldSize { - // The requested size is less than the old size. - // There are likely scenarios for this: - // - The caller intended to grow the allocation, but the original size - // was rounded up by alloc to a multiple of the block size. - // The rounded size is already sufficient. - // - The caller intended to shrink the allocation. - // We currently ignore this case. - // Either way, the current allocation can be left alone. - return ptr + header := (*objHeader)(unsafe.Add(lastBlock.pointer(), bytesPerBlock-unsafe.Sizeof(objHeader{}))) + if header.next == 1 { + for block := firstBlock; block <= lastBlock; block++ { + block.free() + } + insertFreeRange(firstBlock.pointer(), uintptr(lastBlock-firstBlock+1)) } - - // Create a new allocation and copy the old data. - newAlloc := alloc(size, nil) - memcpy(newAlloc, ptr, oldSize) - free(ptr) - - return newAlloc -} - -func free(ptr unsafe.Pointer) { - // TODO: free blocks on request, when the compiler knows they're unused. + gcLock.Unlock() } // GC performs a garbage collection cycle. @@ -659,7 +658,7 @@ func finishMark() { if obj == nil { return } - scanList = obj.next + scanList = (*objHeader)(unsafe.Pointer(obj.next)) // Check if the object may contain pointers. if obj.layout.pointerFree() { @@ -717,7 +716,7 @@ func markRoot(addr, root uintptr) { // Add the object to the scan list. header := (*objHeader)(unsafe.Add(head.pointer(), bytesPerBlock-unsafe.Sizeof(objHeader{}))) - header.next = scanList + header.next = uintptr(unsafe.Pointer(scanList)) scanList = header } @@ -751,7 +750,10 @@ func sweep() uintptr { // Unmark the next head. block-- - block.unmark() + header := (*objHeader)(unsafe.Add(block.pointer(), bytesPerBlock-unsafe.Sizeof(objHeader{}))) + if header.next != 1 { + block.unmark() + } // Skip the tail. for block > 0 && (block-1).state() == blockStateTail { @@ -893,5 +895,19 @@ func SetFinalizer(obj interface{}, finalizer interface{}) { // A nil pointer has nothing to finalize. return } + + gcLock.Lock() + addr := uintptr(objPtr) + manual := false + if isOnHeap(addr) { + head := blockFromAddr(addr).findHead() + header := (*objHeader)(unsafe.Add(head.pointer(), bytesPerBlock-unsafe.Sizeof(objHeader{}))) + manual = header.next == 1 + } + gcLock.Unlock() + if manual && finalizer != nil { + runtimeFatal("runtime.SetFinalizer: manual allocation") + } + registerFinalizer(uintptr(objPtr), finalizer) } diff --git a/src/runtime/gc_boehm.go b/src/runtime/gc_boehm.go index e0fc16a67f..1fb4e654f1 100644 --- a/src/runtime/gc_boehm.go +++ b/src/runtime/gc_boehm.go @@ -98,8 +98,28 @@ func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer { return ptr } +func allocManual(size uintptr) unsafe.Pointer { + if size == 0 { + return alloc_zero(size, gclayout.NoPtrs.AsPtr()) + } + + gcLock.Lock() + ptr := libgc_malloc_atomic_uncollectable(size) + gcResumeWorld() + gcLock.Unlock() + if ptr == nil { + runtimeFatal("gc: out of memory") + return nil + } + memzero(ptr, size) + return ptr +} + func free(ptr unsafe.Pointer) { + gcLock.Lock() libgc_free(ptr) + gcResumeWorld() + gcLock.Unlock() } func GC() { @@ -152,6 +172,9 @@ func libgc_malloc(uintptr) unsafe.Pointer //export GC_malloc_atomic func libgc_malloc_atomic(uintptr) unsafe.Pointer +//export GC_malloc_atomic_uncollectable +func libgc_malloc_atomic_uncollectable(uintptr) unsafe.Pointer + //export GC_free func libgc_free(unsafe.Pointer) @@ -164,6 +187,9 @@ func libgc_size(ptr uintptr) uintptr //export GC_push_all func libgc_push_all(bottom, top uintptr) +//export GC_push_all_eager +func libgc_push_all_eager(bottom, top uintptr) + //export GC_push_all_stack func libgc_push_all_stack(bottom, top uintptr) diff --git a/src/runtime/gc_custom.go b/src/runtime/gc_custom.go index 0125f1688b..12f1b2e12d 100644 --- a/src/runtime/gc_custom.go +++ b/src/runtime/gc_custom.go @@ -48,7 +48,6 @@ func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer func free(ptr unsafe.Pointer) // markRoots is called with the start and end addresses to scan for references. -// It is currently only called with the top and bottom of the stack. func markRoots(start, end uintptr) // GC is called to explicitly run garbage collection. diff --git a/src/runtime/gc_globals_blocks.go b/src/runtime/gc_globals_blocks.go new file mode 100644 index 0000000000..ec9641a0b4 --- /dev/null +++ b/src/runtime/gc_globals_blocks.go @@ -0,0 +1,17 @@ +//go:build gc.conservative || gc.precise + +package runtime + +import "unsafe" + +func markGlobals() { + for i := uintptr(0); i < gcGlobalRootCount(); i++ { + addr := gcGlobalRoot(i) + markRoot(uintptr(addr), *(*uintptr)(addr)) + } +} + +// These functions are generated by the compiler from the pointer layouts of +// mutable globals. +func gcGlobalRootCount() uintptr +func gcGlobalRoot(index uintptr) unsafe.Pointer diff --git a/src/runtime/gc_globals_boehm.go b/src/runtime/gc_globals_boehm.go new file mode 100644 index 0000000000..19c45c0ad3 --- /dev/null +++ b/src/runtime/gc_globals_boehm.go @@ -0,0 +1,20 @@ +//go:build gc.boehm + +package runtime + +import "unsafe" + +func markGlobals() { + for i := uintptr(0); i < gcGlobalRootCount(); i++ { + addr := gcGlobalRoot(uintptr(i)) + // GC_push_all queues one range per call and overflows Boehm's mark + // stack for programs with thousands of roots. Scan each sparse slot + // immediately instead. + libgc_push_all_eager(uintptr(addr), uintptr(addr)+unsafe.Sizeof(uintptr(0))) + } +} + +// These functions are generated by the compiler from the pointer layouts of +// mutable globals. +func gcGlobalRootCount() uintptr +func gcGlobalRoot(index uintptr) unsafe.Pointer diff --git a/src/runtime/gc_globals_custom.go b/src/runtime/gc_globals_custom.go new file mode 100644 index 0000000000..08b59768af --- /dev/null +++ b/src/runtime/gc_globals_custom.go @@ -0,0 +1,7 @@ +//go:build gc.custom + +package runtime + +func markGlobals() { + findGlobals(markRoots) +} diff --git a/src/runtime/gc_globals_none.go b/src/runtime/gc_globals_none.go new file mode 100644 index 0000000000..264406ae08 --- /dev/null +++ b/src/runtime/gc_globals_none.go @@ -0,0 +1,6 @@ +//go:build gc.leaking || gc.none + +package runtime + +func markGlobals() { +} diff --git a/src/runtime/gc_leaking.go b/src/runtime/gc_leaking.go index 3f1595ff63..3469b13d24 100644 --- a/src/runtime/gc_leaking.go +++ b/src/runtime/gc_leaking.go @@ -68,18 +68,6 @@ func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer { return pointer } -func realloc(ptr unsafe.Pointer, size uintptr) unsafe.Pointer { - newAlloc := alloc(size, nil) - if ptr == nil { - return newAlloc - } - // according to POSIX everything beyond the previous pointer's - // size will have indeterminate values so we can just copy garbage - memcpy(newAlloc, ptr, size) - - return newAlloc -} - func free(ptr unsafe.Pointer) { // Memory is never freed. } diff --git a/src/runtime/gc_manual.go b/src/runtime/gc_manual.go new file mode 100644 index 0000000000..25f7adc54c --- /dev/null +++ b/src/runtime/gc_manual.go @@ -0,0 +1,11 @@ +//go:build !gc.custom + +package runtime + +import "unsafe" + +func freeManual(ptr unsafe.Pointer) { + if ptr != nil && ptr != unsafe.Pointer(zeroSizeAllocPtr) { + free(ptr) + } +} diff --git a/src/runtime/gc_manual_custom.go b/src/runtime/gc_manual_custom.go new file mode 100644 index 0000000000..4f6adc6488 --- /dev/null +++ b/src/runtime/gc_manual_custom.go @@ -0,0 +1,35 @@ +//go:build gc.custom + +package runtime + +import ( + "internal/gclayout" + "internal/task" + "unsafe" +) + +// Custom collectors retain manual allocations through ordinary typed roots so +// the custom GC interface does not need an additional allocation primitive. +var manualAllocs = make(map[*byte]struct{}) +var manualAllocsLock task.PMutex + +func allocManual(size uintptr) unsafe.Pointer { + if size == 0 { + return alloc_zero(size, gclayout.NoPtrs.AsPtr()) + } + ptr := alloc(size, gclayout.NoPtrs.AsPtr()) + manualAllocsLock.Lock() + manualAllocs[(*byte)(ptr)] = struct{}{} + manualAllocsLock.Unlock() + return ptr +} + +func freeManual(ptr unsafe.Pointer) { + if ptr == nil || ptr == unsafe.Pointer(zeroSizeAllocPtr) { + return + } + manualAllocsLock.Lock() + delete(manualAllocs, (*byte)(ptr)) + manualAllocsLock.Unlock() + free(ptr) +} diff --git a/src/runtime/gc_manual_leaking.go b/src/runtime/gc_manual_leaking.go new file mode 100644 index 0000000000..5813587383 --- /dev/null +++ b/src/runtime/gc_manual_leaking.go @@ -0,0 +1,15 @@ +//go:build gc.leaking || gc.none + +package runtime + +import ( + "internal/gclayout" + "unsafe" +) + +func allocManual(size uintptr) unsafe.Pointer { + if size == 0 { + return alloc_zero(size, gclayout.NoPtrs.AsPtr()) + } + return alloc(size, gclayout.NoPtrs.AsPtr()) +} diff --git a/src/runtime/gc_none.go b/src/runtime/gc_none.go index ce9649c719..8634308d9d 100644 --- a/src/runtime/gc_none.go +++ b/src/runtime/gc_none.go @@ -22,8 +22,6 @@ func scanCurrentStack() {} func alloc(size uintptr, layout unsafe.Pointer) unsafe.Pointer -func realloc(ptr unsafe.Pointer, size uintptr) unsafe.Pointer - func free(ptr unsafe.Pointer) { // Nothing to free when nothing gets allocated. } diff --git a/src/runtime/gc_precise.go b/src/runtime/gc_precise.go index 062cc46afa..7f05c4f094 100644 --- a/src/runtime/gc_precise.go +++ b/src/runtime/gc_precise.go @@ -55,7 +55,10 @@ package runtime -import "unsafe" +import ( + "internal/gclayout" + "unsafe" +) const sizeFieldBits = 4 + (unsafe.Sizeof(uintptr(0)) / 4) @@ -76,9 +79,7 @@ func (layout gcLayout) pointerFree() bool { // The length is rounded down to a multiple of the element size. func (layout gcLayout) scan(start, len uintptr) { switch { - case layout == 0: - // This is an unknown layout. - // Scan conservatively. + case layout == gcLayout(gclayout.Conservative): // NOTE: This is *NOT* equivalent to a slice of pointers on AVR. scanConservative(start, len) diff --git a/src/runtime/gc_stack_cores.go b/src/runtime/gc_stack_cores.go index 9100109a2c..66aff00879 100644 --- a/src/runtime/gc_stack_cores.go +++ b/src/runtime/gc_stack_cores.go @@ -26,7 +26,7 @@ func gcMarkReachable() { } // Scan globals. - findGlobals(markRoots) + markGlobals() // Nothing more to do: the other cores haven't started yet. return @@ -57,7 +57,7 @@ func gcMarkReachable() { } // Scan globals. - findGlobals(markRoots) + markGlobals() // Signal each core in turn that they can scan the stack. for i := uint32(0); i < numCPU; i++ { diff --git a/src/runtime/gc_stack_portable.go b/src/runtime/gc_stack_portable.go index 04162bb07a..fdf0a7cae9 100644 --- a/src/runtime/gc_stack_portable.go +++ b/src/runtime/gc_stack_portable.go @@ -10,7 +10,7 @@ import ( func gcMarkReachable() { markStack() - findGlobals(markRoots) + markGlobals() } //go:extern runtime.stackChainStart diff --git a/src/runtime/gc_stack_raw.go b/src/runtime/gc_stack_raw.go index 03c37696a9..95d4b0a59f 100644 --- a/src/runtime/gc_stack_raw.go +++ b/src/runtime/gc_stack_raw.go @@ -12,7 +12,7 @@ var gcScanState atomic.Uint32 func gcMarkReachable() { markStack() - findGlobals(markRoots) + markGlobals() } // markStack marks all root pointers found on the stack. diff --git a/src/runtime/gc_stack_threads.go b/src/runtime/gc_stack_threads.go index a2b06486f5..0e58644a84 100644 --- a/src/runtime/gc_stack_threads.go +++ b/src/runtime/gc_stack_threads.go @@ -13,7 +13,7 @@ func gcMarkReachable() { // //go:linkname gcScanGlobals internal/task.gcScanGlobals func gcScanGlobals() { - findGlobals(markRoots) + markGlobals() } // Function called from assembly with all registers pushed, to actually scan the diff --git a/src/runtime/hashmap.go b/src/runtime/hashmap.go index ba864db305..ac33ee7512 100644 --- a/src/runtime/hashmap.go +++ b/src/runtime/hashmap.go @@ -14,6 +14,7 @@ import ( // The underlying hashmap structure for Go. type hashmap struct { buckets unsafe.Pointer // pointer to array of buckets + typeInfo *hashmapTypeInfo seed uintptr count uintptr keySize uintptr @@ -26,6 +27,17 @@ type hashmap struct { keyHash func(key unsafe.Pointer, size, seed uintptr) uint32 } +type hashmapTypeInfo struct { + keyLayout unsafe.Pointer + valueLayout unsafe.Pointer + bucketLayout unsafe.Pointer +} + +//go:inline +func hashmapType(m *hashmap) *hashmapTypeInfo { + return m.typeInfo +} + const ( hashmapMaxKeySize = 128 hashmapMaxValueSize = 128 @@ -113,7 +125,7 @@ func hashmapTopHash(hash uint32) uint8 { } // Create a new hashmap with the given keySize and valueSize. -func hashmapMake(keySize, valueSize uintptr, sizeHint uintptr, alg uint8) *hashmap { +func hashmapMake(keySize, valueSize uintptr, sizeHint uintptr, typeInfo unsafe.Pointer, alg uint8) *hashmap { bucketBits := uint8(0) for hashmapHasSpaceToGrow(bucketBits) && hashmapOverLoadFactor(sizeHint, bucketBits) { bucketBits++ @@ -132,13 +144,14 @@ func hashmapMake(keySize, valueSize uintptr, sizeHint uintptr, alg uint8) *hashm } bucketBufSize := hashmapBucketHeaderSize + keySlotSize*8 + valueSlotSize*8 - buckets := alloc(bucketBufSize*(1< size { + copySize = size + } + memcpy(newPtr, ptr, copySize) + freeManual(ptr) + } + return newPtr } func ticksToNanoseconds(ticks timeUnit) int64 { diff --git a/testdata/cgo/main.c b/testdata/cgo/main.c index 94b338dda2..82718eb6fb 100644 --- a/testdata/cgo/main.c +++ b/testdata/cgo/main.c @@ -1,6 +1,7 @@ #include #include "main.h" #include +#include int global = 3; bool globalBool = 1; @@ -82,3 +83,98 @@ int set_errno(int err) { errno = err; return -1; } + +typedef struct malloc_node { + struct malloc_node *next; + int value; +} malloc_node; + +void *makeMallocChain(void) { + malloc_node *tail = malloc(sizeof(malloc_node)); + tail->next = NULL; + tail->value = 42; + + malloc_node *head = malloc(sizeof(malloc_node)); + head->next = tail; + head->value = 1; + return head; +} + +void clobberMalloc(void) { +#if defined(__AVR__) + return; +#else + malloc_node *nodes[64]; + for (int i = 0; i < 64; i++) { + nodes[i] = malloc(sizeof(malloc_node)); + nodes[i]->next = NULL; + nodes[i]->value = 0; + } + for (int i = 0; i < 64; i++) { + free(nodes[i]); + } +#endif +} + +int mallocChainValue(void *ptr) { + return ((malloc_node *)ptr)->next->value; +} + +#define MALLOC_HIDE_MASK ((uintptr_t)0x5a5a5a5a) + +__attribute__((noinline)) uintptr_t makeHiddenMalloc(void) { + malloc_node *node = malloc(sizeof(malloc_node)); + node->next = NULL; + node->value = 84; + return (uintptr_t)node ^ MALLOC_HIDE_MASK; +} + +int hiddenMallocValue(uintptr_t hidden) { + malloc_node *node = (malloc_node *)(hidden ^ MALLOC_HIDE_MASK); + return node->value; +} + +void freeHiddenMalloc(uintptr_t hidden) { + free((void *)(hidden ^ MALLOC_HIDE_MASK)); +} + +void mallocFreeStress(void) { +#if defined(__AVR__) + const int count = 32; +#else + const int count = 1024; +#endif + for (int i = 0; i < count; i++) { + char *ptr = malloc(1024); + ptr[0] = (char)i; + free(ptr); + } +} + +void mallocZero(void) { + free(malloc(0)); +} + +__attribute__((noinline)) void *callCalloc(size_t nmemb, size_t size) { + return calloc(nmemb, size); +} + +int callocOverflowReturnsNull(void) { +#if defined(__linux__) || defined(_WIN32) || defined(__APPLE__) + return 1; +#else + volatile size_t nmemb = (size_t)-1; + return callCalloc(nmemb, 2) == NULL; +#endif +} + +__attribute__((noinline)) void clobberStack(void) { +#if defined(__AVR__) + return; +#else + volatile uintptr_t values[128]; + for (int i = 0; i < 128; i++) { + values[i] = 0; + } +#endif +} diff --git a/testdata/cgo/main.go b/testdata/cgo/main.go index 38d11386a9..55992d81fe 100644 --- a/testdata/cgo/main.go +++ b/testdata/cgo/main.go @@ -19,6 +19,7 @@ import "C" import "C" import ( + "runtime" "syscall" "unsafe" ) @@ -171,6 +172,35 @@ func main() { println("len(C.GoBytes(C.CBytes(nil),0)):", len(C.GoBytes(C.CBytes(nil), 0))) println(`rountrip CBytes:`, C.GoString((*C.char)(C.CBytes([]byte("hello\000"))))) + // malloc allocations remain live until free, even when C pointers are the + // only links between them. + mallocChain := C.makeMallocChain() + runtime.GC() + C.clobberMalloc() + println("malloc chain:", C.mallocChainValue(mallocChain)) + + // malloc lifetime ends at free, not when the allocation becomes invisible + // to the GC. Encode the address so neither Go nor C exposes a pointer root. + hiddenMallocChan := make(chan C.uintptr_t, 1) + hiddenMallocDone := make(chan struct{}) + go func() { + hiddenMallocChan <- C.makeHiddenMalloc() + close(hiddenMallocDone) + }() + hiddenMalloc := <-hiddenMallocChan + <-hiddenMallocDone + C.clobberStack() + runtime.GC() + C.clobberMalloc() + println("hidden malloc:", C.hiddenMallocValue(hiddenMalloc)) + C.freeHiddenMalloc(hiddenMalloc) + + C.mallocFreeStress() + println("malloc/free stress: ok") + C.mallocZero() + println("malloc zero: ok") + println("calloc overflow:", C.callocOverflowReturnsNull() != 0) + // Check that errno is returned from the second return value, and that it // matches the errno value that was just set. _, errno := C.set_errno(C.EINVAL) diff --git a/testdata/cgo/main.h b/testdata/cgo/main.h index 3942497f23..0ae9e5fbce 100644 --- a/testdata/cgo/main.h +++ b/testdata/cgo/main.h @@ -1,4 +1,5 @@ #include +#include #include #include @@ -157,3 +158,15 @@ double doSqrt(double); void printf_single_int(char *format, int arg); int set_errno(int err); + +void *makeMallocChain(void); +void clobberMalloc(void); +int mallocChainValue(void *ptr); +uintptr_t makeHiddenMalloc(void); +int hiddenMallocValue(uintptr_t hidden); +void freeHiddenMalloc(uintptr_t hidden); +void mallocFreeStress(void); +void mallocZero(void); +void *callCalloc(size_t nmemb, size_t size); +int callocOverflowReturnsNull(void); +void clobberStack(void); diff --git a/testdata/cgo/out.txt b/testdata/cgo/out.txt index 1d63f5e82f..bd70fae2cd 100644 --- a/testdata/cgo/out.txt +++ b/testdata/cgo/out.txt @@ -75,6 +75,11 @@ len(C.GoStringN(nil, 0)): 0 len(C.GoBytes(nil, 0)): 0 len(C.GoBytes(C.CBytes(nil),0)): 0 rountrip CBytes: hello +malloc chain: 42 +hidden malloc: 84 +malloc/free stress: ok +malloc zero: ok +calloc overflow: true EINVAL: true EAGAIN: true copied string: foobar diff --git a/testdata/gc.go b/testdata/gc.go index 456d763b4c..445290cbb4 100644 --- a/testdata/gc.go +++ b/testdata/gc.go @@ -1,6 +1,9 @@ package main -import "runtime" +import ( + "reflect" + "runtime" +) var xorshift32State uint32 = 1 @@ -19,6 +22,9 @@ func randuint32() uint32 { func main() { testNonPointerHeap() + testGlobalMapRoots() + testGlobalChannelRoots() + testReflectRoots() testKeepAlive() } @@ -74,3 +80,146 @@ func testKeepAlive() { var x int runtime.KeepAlive(&x) } + +type globalMapObject struct { + marker int + data [64]byte +} + +var globalMap = make(map[int]*globalMapObject) +var globalChannel chan *globalMapObject +var globalPointerSlice []*globalMapObject +var globalGCClobber any + +type globalMapLargeKey struct { + object *globalMapObject + data [129]byte +} + +type globalMapLargeValue struct { + object *globalMapObject + data [129]byte +} + +var globalLargeKeyMap = make(map[globalMapLargeKey]int) +var globalLargeValueMap = make(map[int]globalMapLargeValue) + +//go:noinline +func populateGlobalMaps() { + for i := 0; i < 32; i++ { + globalMap[i] = &globalMapObject{marker: 100 + i} + globalPointerSlice = append(globalPointerSlice, &globalMapObject{marker: 800 + i}) + } + globalLargeKeyMap[globalMapLargeKey{ + object: &globalMapObject{marker: 200}, + }] = 1 + globalLargeValueMap[0] = globalMapLargeValue{ + object: &globalMapObject{marker: 300}, + } +} + +func testGlobalMapRoots() { + populateGlobalMaps() + + runtime.GC() + for i := 0; i < 100; i++ { + globalGCClobber = new(globalMapObject) + } + runtime.GC() + + for i := 0; i < 32; i++ { + if globalMap[i].marker != 100+i { + panic("global map value was collected") + } + } + for key := range globalLargeKeyMap { + if key.object.marker != 200 { + panic("indirect global map key was collected") + } + } + if globalLargeValueMap[0].object.marker != 300 { + panic("indirect global map value was collected") + } + for i, object := range globalPointerSlice { + if object.marker != 800+i { + panic("global slice value was collected") + } + } +} + +//go:noinline +func populateGlobalChannel() { + globalChannel = make(chan *globalMapObject, 4) + globalChannel <- &globalMapObject{marker: 400} +} + +func testGlobalChannelRoots() { + populateGlobalChannel() + + runtime.GC() + for i := 0; i < 100; i++ { + globalGCClobber = new(globalMapObject) + } + runtime.GC() + + if (<-globalChannel).marker != 400 { + panic("global channel value was collected") + } +} + +type reflectRootObject struct { + marker int + child *reflectRootObject + data [128]byte +} + +type reflectMapKey struct { + object *reflectRootObject + data [129]byte +} + +type reflectMapValue struct { + object *reflectRootObject + data [129]byte +} + +type reflectRootMap map[reflectMapKey]reflectMapValue + +var globalReflectObject *reflectRootObject +var globalReflectMap reflectRootMap + +//go:noinline +func populateReflectRoots() { + value := reflect.New(reflect.TypeOf(reflectRootObject{})) + globalReflectObject = value.Interface().(*reflectRootObject) + globalReflectObject.child = &reflectRootObject{marker: 500} + + mapValue := reflect.MakeMapWithSize(reflect.TypeOf(globalReflectMap), 1) + mapValue.SetMapIndex( + reflect.ValueOf(reflectMapKey{object: &reflectRootObject{marker: 600}}), + reflect.ValueOf(reflectMapValue{object: &reflectRootObject{marker: 700}}), + ) + globalReflectMap = mapValue.Interface().(reflectRootMap) +} + +func testReflectRoots() { + populateReflectRoots() + + runtime.GC() + for i := 0; i < 100; i++ { + globalGCClobber = new(reflectRootObject) + } + runtime.GC() + + if globalReflectObject.child.marker != 500 { + panic("reflected object field was collected") + } + for key, value := range globalReflectMap { + if key.object.marker != 600 { + panic("reflected map key was collected") + } + if value.object.marker != 700 { + panic("reflected map value was collected") + } + } +} diff --git a/testdata/map.go b/testdata/map.go index f5be02ab06..0f19ed0976 100644 --- a/testdata/map.go +++ b/testdata/map.go @@ -123,8 +123,15 @@ func main() { println(`structMap[{"tau", 6.28}]:`, structMap[namedFloat{"tau", 6.28}]) // test preallocated map - squares := make(map[int]int, 200) - testBigMap(squares, 100) + mapSize := 200 + mapEntries := 100 + if unsafe.Sizeof(uintptr(0)) < 4 { + // Leave enough heap for the rest of this test on low-memory devices. + mapSize = 100 + mapEntries = 50 + } + squares := make(map[int]int, mapSize) + testBigMap(squares, mapEntries) println("tested preallocated map") // test growing maps diff --git a/transform/gc.go b/transform/gc.go index abbe3cb7bb..6f207e3ed8 100644 --- a/transform/gc.go +++ b/transform/gc.go @@ -1,6 +1,8 @@ package transform import ( + "strings" + "tinygo.org/x/go-llvm" ) @@ -12,6 +14,8 @@ const shiftExcludeArgMem = 2 // MakeGCStackSlots converts all calls to runtime.trackPointer to explicit // stores to stack slots that are scannable by the GC. func MakeGCStackSlots(mod llvm.Module) bool { + hasGlobalRoots := makeGCGlobalRoots(mod) + // Check whether there are allocations at all. alloc := mod.NamedFunction("runtime.alloc") if alloc.IsNil() { @@ -26,12 +30,12 @@ func MakeGCStackSlots(mod llvm.Module) bool { stackChainStart.SetInitializer(llvm.ConstNull(stackChainStart.GlobalValueType())) stackChainStart.SetGlobalConstant(true) } - return false + return hasGlobalRoots } trackPointer := mod.NamedFunction("runtime.trackPointer") if trackPointer.IsNil() || trackPointer.FirstUse().IsNil() { - return false // nothing to do + return hasGlobalRoots } ctx := mod.Context() @@ -107,7 +111,7 @@ func MakeGCStackSlots(mod llvm.Module) bool { for _, use := range getUses(trackPointer) { use.EraseFromParentAsInstruction() } - return false + return hasGlobalRoots } stackChainStart.SetLinkage(llvm.InternalLinkage) stackChainStartType := stackChainStart.GlobalValueType() @@ -285,6 +289,110 @@ func MakeGCStackSlots(mod llvm.Module) bool { return true } +func makeGCGlobalRoots(mod llvm.Module) bool { + rootCount := mod.NamedFunction("runtime.gcGlobalRootCount") + rootAt := mod.NamedFunction("runtime.gcGlobalRoot") + rootValues := mod.NamedFunction("runtime.gcGlobalRootValues") + if rootCount.IsNil() || rootAt.IsNil() || + !rootCount.FirstBasicBlock().IsNil() || !rootAt.FirstBasicBlock().IsNil() { + return false + } + if !rootValues.IsNil() && !rootValues.FirstBasicBlock().IsNil() { + return false + } + + ctx := mod.Context() + uintptrType := rootCount.GlobalValueType().ReturnType() + var roots []llvm.Value + for global := mod.FirstGlobal(); !global.IsNil(); global = llvm.NextGlobal(global) { + if strings.HasPrefix(global.Name(), "llvm.") || + global.IsGlobalConstant() || + global.Initializer().IsNil() || + !gcTypeHasPointers(global.GlobalValueType()) { + continue + } + roots = appendGCGlobalRoots(roots, global, global.GlobalValueType(), ctx.Int32Type()) + } + + ptrType := rootAt.GlobalValueType().ReturnType() + rootArray := llvm.AddGlobal(mod, llvm.ArrayType(ptrType, len(roots)), "runtime.gcGlobalRoots") + rootArray.SetInitializer(llvm.ConstArray(ptrType, roots)) + rootArray.SetGlobalConstant(true) + rootArray.SetLinkage(llvm.InternalLinkage) + + builder := ctx.NewBuilder() + defer builder.Dispose() + + entry := ctx.AddBasicBlock(rootCount, "entry") + builder.SetInsertPointAtEnd(entry) + builder.CreateRet(llvm.ConstInt(rootCount.GlobalValueType().ReturnType(), uint64(len(roots)), false)) + + entry = ctx.AddBasicBlock(rootAt, "entry") + builder.SetInsertPointAtEnd(entry) + index := rootAt.FirstParam() + addr := builder.CreateInBoundsGEP(rootArray.GlobalValueType(), rootArray, []llvm.Value{ + llvm.ConstInt(ctx.Int32Type(), 0, false), + index, + }, "") + builder.CreateRet(builder.CreateLoad(ptrType, addr, "")) + + if !rootValues.IsNil() { + rootValueArray := llvm.AddGlobal(mod, llvm.ArrayType(uintptrType, len(roots)), "runtime.gcGlobalRootValueArray") + rootValueArray.SetInitializer(llvm.ConstNull(rootValueArray.GlobalValueType())) + rootValueArray.SetLinkage(llvm.InternalLinkage) + + entry = ctx.AddBasicBlock(rootValues, "entry") + builder.SetInsertPointAtEnd(entry) + builder.CreateRet(rootValueArray) + } + return true +} + +func appendGCGlobalRoots(roots []llvm.Value, addr llvm.Value, typ llvm.Type, i32Type llvm.Type) []llvm.Value { + switch typ.TypeKind() { + case llvm.PointerTypeKind: + return append(roots, addr) + case llvm.StructTypeKind: + for i, fieldType := range typ.StructElementTypes() { + if gcTypeHasPointers(fieldType) { + field := llvm.ConstGEP(typ, addr, []llvm.Value{ + llvm.ConstInt(i32Type, 0, false), + llvm.ConstInt(i32Type, uint64(i), false), + }) + roots = appendGCGlobalRoots(roots, field, fieldType, i32Type) + } + } + case llvm.ArrayTypeKind: + elemType := typ.ElementType() + if gcTypeHasPointers(elemType) { + for i := 0; i < typ.ArrayLength(); i++ { + elem := llvm.ConstGEP(typ, addr, []llvm.Value{ + llvm.ConstInt(i32Type, 0, false), + llvm.ConstInt(i32Type, uint64(i), false), + }) + roots = appendGCGlobalRoots(roots, elem, elemType, i32Type) + } + } + } + return roots +} + +func gcTypeHasPointers(typ llvm.Type) bool { + switch typ.TypeKind() { + case llvm.PointerTypeKind: + return true + case llvm.StructTypeKind: + for _, field := range typ.StructElementTypes() { + if gcTypeHasPointers(field) { + return true + } + } + case llvm.ArrayTypeKind: + return typ.ArrayLength() != 0 && gcTypeHasPointers(typ.ElementType()) + } + return false +} + // markParentFunctions traverses all parent function calls (recursively) and // adds them to the set of marked functions. It only considers function calls: // any other uses of such a function is ignored. diff --git a/transform/optimizer.go b/transform/optimizer.go index 150a9a77cb..f04b213e92 100644 --- a/transform/optimizer.go +++ b/transform/optimizer.go @@ -58,7 +58,9 @@ func Optimize(mod llvm.Module, config *compileopts.Config) []error { // LLVM 17 doesn't have the no-verify-fixpoint flag. optPasses = "globaldce,globalopt,ipsccp,instcombine,adce,function-attrs" } + blockGlobalAllocPromotion(mod) err := mod.RunPasses(optPasses, llvm.TargetMachine{}, po) + removeGlobalAllocPromotionMarker(mod) if err != nil { return []error{fmt.Errorf("could not build pass pipeline: %w", err)} } @@ -80,7 +82,9 @@ func Optimize(mod llvm.Module, config *compileopts.Config) []error { // After interfaces are lowered, there are many more opportunities for // interprocedural optimizations. To get them to work, function // attributes have to be updated first. + blockGlobalAllocPromotion(mod) err = mod.RunPasses(optPasses, llvm.TargetMachine{}, po) + removeGlobalAllocPromotionMarker(mod) if err != nil { return []error{fmt.Errorf("could not build pass pipeline: %w", err)} } @@ -162,7 +166,9 @@ func Optimize(mod llvm.Module, config *compileopts.Config) []error { po := llvm.NewPassBuilderOptions() defer po.Dispose() passes := fmt.Sprintf("thinlto-pre-link<%s>", optLevel) + blockGlobalAllocPromotion(mod) err := mod.RunPasses(passes, llvm.TargetMachine{}, po) + removeGlobalAllocPromotionMarker(mod) if err != nil { return []error{fmt.Errorf("could not build pass pipeline: %w", err)} } @@ -177,6 +183,61 @@ func Optimize(mod llvm.Module, config *compileopts.Config) []error { return nil } +func blockGlobalAllocPromotion(mod llvm.Module) { + ctx := mod.Context() + ptrType := llvm.PointerType(ctx.Int8Type(), 0) + marker := llvm.AddFunction(mod, "tinygo.gc.alloc.marker", llvm.FunctionType(ctx.VoidType(), []llvm.Type{ptrType}, false)) + + builder := ctx.NewBuilder() + defer builder.Dispose() + var marked bool + for _, name := range []string{"runtime.alloc", "runtime.alloc_noheap"} { + alloc := mod.NamedFunction(name) + if alloc.IsNil() { + continue + } + for _, call := range getUses(alloc) { + if isPointerFreeAllocation(call) { + continue + } + + // GlobalOpt may otherwise turn this allocation into an untyped + // global, hiding its pointer fields from makeGCGlobalRoots. + next := llvm.NextInstruction(call) + if next.IsNil() { + continue + } + builder.SetInsertPointBefore(next) + builder.CreateCall(marker.GlobalValueType(), marker, []llvm.Value{call}, "") + marked = true + } + } + if !marked { + marker.EraseFromParentAsFunction() + } +} + +func isPointerFreeAllocation(call llvm.Value) bool { + const noPointerLayout = 3 + + layout := call.Operand(1) + return !layout.IsAConstantExpr().IsNil() && + layout.Opcode() == llvm.IntToPtr && + !layout.Operand(0).IsAConstantInt().IsNil() && + layout.Operand(0).ZExtValue() == noPointerLayout +} + +func removeGlobalAllocPromotionMarker(mod llvm.Module) { + marker := mod.NamedFunction("tinygo.gc.alloc.marker") + if marker.IsNil() { + return + } + for _, call := range getUses(marker) { + call.EraseFromParentAsInstruction() + } + marker.EraseFromParentAsFunction() +} + // functionsUsedInTransform is a list of function symbols that may be used // during TinyGo optimization passes so they have to be marked as external // linkage until all TinyGo passes have finished. diff --git a/transform/testdata/allocs.ll b/transform/testdata/allocs.ll index 4f6960ef99..65d941adf2 100644 --- a/transform/testdata/allocs.ll +++ b/transform/testdata/allocs.ll @@ -7,7 +7,7 @@ declare nonnull ptr @runtime.alloc(i32, ptr) ; Test allocating a single int (i32) that should be allocated on the stack. define void @testInt() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) store i32 5, ptr %alloc ret void } @@ -15,7 +15,7 @@ define void @testInt() { ; Test allocating an array of 3 i16 values that should be allocated on the ; stack. define i16 @testArray() { - %alloc = call align 2 ptr @runtime.alloc(i32 6, ptr null) + %alloc = call align 2 ptr @runtime.alloc(i32 6, ptr inttoptr (i32 3 to ptr)) %alloc.1 = getelementptr i16, ptr %alloc, i32 1 store i16 5, ptr %alloc.1 %alloc.2 = getelementptr i16, ptr %alloc, i32 2 @@ -25,15 +25,15 @@ define i16 @testArray() { ; Test allocating objects with an unknown alignment. define void @testUnknownAlign() { - %alloc32 = call ptr @runtime.alloc(i32 32, ptr null) + %alloc32 = call ptr @runtime.alloc(i32 32, ptr inttoptr (i32 3 to ptr)) store i8 5, ptr %alloc32 - %alloc24 = call ptr @runtime.alloc(i32 24, ptr null) + %alloc24 = call ptr @runtime.alloc(i32 24, ptr inttoptr (i32 3 to ptr)) store i16 5, ptr %alloc24 - %alloc12 = call ptr @runtime.alloc(i32 12, ptr null) + %alloc12 = call ptr @runtime.alloc(i32 12, ptr inttoptr (i32 3 to ptr)) store i16 5, ptr %alloc12 - %alloc6 = call ptr @runtime.alloc(i32 6, ptr null) + %alloc6 = call ptr @runtime.alloc(i32 6, ptr inttoptr (i32 3 to ptr)) store i16 5, ptr %alloc6 - %alloc3 = call ptr @runtime.alloc(i32 3, ptr null) + %alloc3 = call ptr @runtime.alloc(i32 3, ptr inttoptr (i32 3 to ptr)) store i16 5, ptr %alloc3 ret void } @@ -41,27 +41,27 @@ define void @testUnknownAlign() { ; Call a function that will let the pointer escape, so the heap-to-stack ; transform shouldn't be applied. define void @testEscapingCall() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %val = call ptr @escapeIntPtr(ptr %alloc) ret void } define void @testEscapingCall2() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %val = call ptr @escapeIntPtrSometimes(ptr %alloc, ptr %alloc) ret void } ; Call a function that doesn't let the pointer escape. define void @testNonEscapingCall() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %val = call ptr @noescapeIntPtr(ptr %alloc) ret void } ; Return the allocated value, which lets it escape. define ptr @testEscapingReturn() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) ret ptr %alloc } @@ -70,7 +70,7 @@ define void @testNonEscapingLoop() { entry: br label %loop loop: - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %ptr = call ptr @noescapeIntPtr(ptr %alloc) %result = icmp eq ptr null, %ptr br i1 %result, label %loop, label %end @@ -80,7 +80,7 @@ end: ; Test a zero-sized allocation. define void @testZeroSizedAlloc() { - %alloc = call align 1 ptr @runtime.alloc(i32 0, ptr null) + %alloc = call align 1 ptr @runtime.alloc(i32 0, ptr inttoptr (i32 3 to ptr)) %ptr = call ptr @noescapeIntPtr(ptr %alloc) ret void } diff --git a/transform/testdata/allocs.out.ll b/transform/testdata/allocs.out.ll index e4a5e4f7e9..1353d927b6 100644 --- a/transform/testdata/allocs.out.ll +++ b/transform/testdata/allocs.out.ll @@ -42,13 +42,13 @@ define void @testUnknownAlign() { } define void @testEscapingCall() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %val = call ptr @escapeIntPtr(ptr %alloc) ret void } define void @testEscapingCall2() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %val = call ptr @escapeIntPtrSometimes(ptr %alloc, ptr %alloc) ret void } @@ -61,7 +61,7 @@ define void @testNonEscapingCall() { } define ptr @testEscapingReturn() { - %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr null) + %alloc = call align 4 ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) ret ptr %alloc } diff --git a/transform/testdata/gc-stackslots.ll b/transform/testdata/gc-stackslots.ll index 58fa3eede2..9be77cf3d8 100644 --- a/transform/testdata/gc-stackslots.ll +++ b/transform/testdata/gc-stackslots.ll @@ -5,11 +5,17 @@ target triple = "wasm32-unknown-unknown-wasm" @someGlobal = global i8 3 @ptrGlobal = global ptr null @arrGlobal = global [8 x i8] zeroinitializer +@structGlobal = global {ptr, i32, [2 x ptr]} zeroinitializer +@constantPtrGlobal = constant ptr @someGlobal declare void @runtime.trackPointer(ptr nocapture readonly) declare noalias nonnull ptr @runtime.alloc(i32, ptr) +declare i32 @runtime.gcGlobalRootCount() + +declare ptr @runtime.gcGlobalRoot(i32) + ; Generic function that returns a pointer (that must be tracked). define ptr @getPointer() { ret ptr @someGlobal @@ -18,7 +24,7 @@ define ptr @getPointer() { define ptr @needsStackSlots() { ; Tracked pointer. Although, in this case the value is immediately returned ; so tracking it is not really necessary. - %ptr = call ptr @runtime.alloc(i32 4, ptr null) + %ptr = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %ptr) call void @someArbitraryFunction() %val = load i8, ptr @someGlobal @@ -39,7 +45,7 @@ define ptr @needsStackSlots2() { call void @runtime.trackPointer(ptr %ptr2) ; Here is finally the point where an allocation happens. - %unused = call ptr @runtime.alloc(i32 4, ptr null) + %unused = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %unused) ret ptr %ptr1 @@ -57,7 +63,7 @@ define ptr @fibNext(ptr %x, ptr %y) { %x.val = load i8, ptr %x %y.val = load i8, ptr %y %out.val = add i8 %x.val, %y.val - %out.alloc = call ptr @runtime.alloc(i32 1, ptr null) + %out.alloc = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %out.alloc) store i8 %out.val, ptr %out.alloc ret ptr %out.alloc @@ -65,9 +71,9 @@ define ptr @fibNext(ptr %x, ptr %y) { define ptr @allocLoop() { entry: - %entry.x = call ptr @runtime.alloc(i32 1, ptr null) + %entry.x = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %entry.x) - %entry.y = call ptr @runtime.alloc(i32 1, ptr null) + %entry.y = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %entry.y) store i8 1, ptr %entry.y br label %loop @@ -93,7 +99,7 @@ define void @testGEPBitcast() { %arr = call ptr @arrayAlloc() %arr.bitcast = getelementptr [32 x i8], ptr %arr, i32 0, i32 0 call void @runtime.trackPointer(ptr %arr.bitcast) - %other = call ptr @runtime.alloc(i32 1, ptr null) + %other = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %other) ret void } @@ -103,7 +109,7 @@ define void @someArbitraryFunction() { } define void @earlyPopRegression() { - %x.alloc = call ptr @runtime.alloc(i32 4, ptr null) + %x.alloc = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %x.alloc) ; At this point the pass used to pop the stack chain, resulting in a potential use-after-free during allocAndSave. musttail call void @allocAndSave(ptr %x.alloc) @@ -111,7 +117,7 @@ define void @earlyPopRegression() { } define void @allocAndSave(ptr %x) { - %y = call ptr @runtime.alloc(i32 4, ptr null) + %y = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) call void @runtime.trackPointer(ptr %y) store ptr %y, ptr %x store ptr %x, ptr @ptrGlobal diff --git a/transform/testdata/gc-stackslots.out.ll b/transform/testdata/gc-stackslots.out.ll index 8a257352bc..3a86a32a7d 100644 --- a/transform/testdata/gc-stackslots.out.ll +++ b/transform/testdata/gc-stackslots.out.ll @@ -5,11 +5,26 @@ target triple = "wasm32-unknown-unknown-wasm" @someGlobal = global i8 3 @ptrGlobal = global ptr null @arrGlobal = global [8 x i8] zeroinitializer +@structGlobal = global { ptr, i32, [2 x ptr] } zeroinitializer +@constantPtrGlobal = constant ptr @someGlobal +@runtime.gcGlobalRoots = internal constant [4 x ptr] [ptr @ptrGlobal, ptr @structGlobal, ptr getelementptr ({ ptr, i32, [2 x ptr] }, ptr @structGlobal, i32 0, i32 2), ptr getelementptr ([2 x ptr], ptr getelementptr ({ ptr, i32, [2 x ptr] }, ptr @structGlobal, i32 0, i32 2), i32 0, i32 1)] declare void @runtime.trackPointer(ptr nocapture readonly) declare noalias nonnull ptr @runtime.alloc(i32, ptr) +define i32 @runtime.gcGlobalRootCount() { +entry: + ret i32 4 +} + +define ptr @runtime.gcGlobalRoot(i32 %0) { +entry: + %1 = getelementptr inbounds [4 x ptr], ptr @runtime.gcGlobalRoots, i32 0, i32 %0 + %2 = load ptr, ptr %1, align 4 + ret ptr %2 +} + define ptr @getPointer() { ret ptr @someGlobal } @@ -21,7 +36,7 @@ define ptr @needsStackSlots() { %2 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 0 store ptr %1, ptr %2, align 4 store ptr %gc.stackobject, ptr @runtime.stackChainStart, align 4 - %ptr = call ptr @runtime.alloc(i32 4, ptr null) + %ptr = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %3 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %ptr, ptr %3, align 4 call void @someArbitraryFunction() @@ -47,7 +62,7 @@ define ptr @needsStackSlots2() { %ptr2 = getelementptr i8, ptr @someGlobal, i32 0 %6 = getelementptr { ptr, i32, ptr, ptr, ptr, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 5 store ptr %ptr2, ptr %6, align 4 - %unused = call ptr @runtime.alloc(i32 4, ptr null) + %unused = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %7 = getelementptr { ptr, i32, ptr, ptr, ptr, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 6 store ptr %unused, ptr %7, align 4 store ptr %1, ptr @runtime.stackChainStart, align 4 @@ -69,7 +84,7 @@ define ptr @fibNext(ptr %x, ptr %y) { %x.val = load i8, ptr %x, align 1 %y.val = load i8, ptr %y, align 1 %out.val = add i8 %x.val, %y.val - %out.alloc = call ptr @runtime.alloc(i32 1, ptr null) + %out.alloc = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) %3 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %out.alloc, ptr %3, align 4 store i8 %out.val, ptr %out.alloc, align 1 @@ -85,10 +100,10 @@ entry: %1 = getelementptr { ptr, i32, ptr, ptr, ptr, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 0 store ptr %0, ptr %1, align 4 store ptr %gc.stackobject, ptr @runtime.stackChainStart, align 4 - %entry.x = call ptr @runtime.alloc(i32 1, ptr null) + %entry.x = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) %2 = getelementptr { ptr, i32, ptr, ptr, ptr, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %entry.x, ptr %2, align 4 - %entry.y = call ptr @runtime.alloc(i32 1, ptr null) + %entry.y = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) %3 = getelementptr { ptr, i32, ptr, ptr, ptr, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 3 store ptr %entry.y, ptr %3, align 4 store i8 1, ptr %entry.y, align 1 @@ -126,7 +141,7 @@ define void @testGEPBitcast() { %arr.bitcast = getelementptr [32 x i8], ptr %arr, i32 0, i32 0 %3 = getelementptr { ptr, i32, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %arr.bitcast, ptr %3, align 4 - %other = call ptr @runtime.alloc(i32 1, ptr null) + %other = call ptr @runtime.alloc(i32 1, ptr inttoptr (i32 3 to ptr)) %4 = getelementptr { ptr, i32, ptr, ptr }, ptr %gc.stackobject, i32 0, i32 3 store ptr %other, ptr %4, align 4 store ptr %1, ptr @runtime.stackChainStart, align 4 @@ -144,7 +159,7 @@ define void @earlyPopRegression() { %2 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 0 store ptr %1, ptr %2, align 4 store ptr %gc.stackobject, ptr @runtime.stackChainStart, align 4 - %x.alloc = call ptr @runtime.alloc(i32 4, ptr null) + %x.alloc = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %3 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %x.alloc, ptr %3, align 4 call void @allocAndSave(ptr %x.alloc) @@ -159,7 +174,7 @@ define void @allocAndSave(ptr %x) { %2 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 0 store ptr %1, ptr %2, align 4 store ptr %gc.stackobject, ptr @runtime.stackChainStart, align 4 - %y = call ptr @runtime.alloc(i32 4, ptr null) + %y = call ptr @runtime.alloc(i32 4, ptr inttoptr (i32 3 to ptr)) %3 = getelementptr { ptr, i32, ptr }, ptr %gc.stackobject, i32 0, i32 2 store ptr %y, ptr %3, align 4 store ptr %y, ptr %x, align 4