diff options
Diffstat (limited to 'src/runtime')
| -rw-r--r-- | src/runtime/memmove_arm64.s | 42 | ||||
| -rw-r--r-- | src/runtime/os_windows.go | 94 | ||||
| -rw-r--r-- | src/runtime/syscall_windows.go | 12 | ||||
| -rw-r--r-- | src/runtime/type.go | 2 |
4 files changed, 131 insertions, 19 deletions
diff --git a/src/runtime/memmove_arm64.s b/src/runtime/memmove_arm64.s index dcbead8cf4..4b6b4965af 100644 --- a/src/runtime/memmove_arm64.s +++ b/src/runtime/memmove_arm64.s @@ -22,7 +22,7 @@ check: CMP R3, R4 BLT backward - // Copying forward proceeds by copying R7/8 words then copying R6 bytes. + // Copying forward proceeds by copying R7/32 quadwords then R6 <= 31 tail bytes. // R3 and R4 are advanced as we copy. // (There may be implementations of armv8 where copying by bytes until @@ -30,11 +30,12 @@ check: // optimization, but the on the one tested so far (xgene) it did not // make a significance difference.) - CBZ R7, noforwardlarge // Do we need to do any doubleword-by-doubleword copying? + CBZ R7, noforwardlarge // Do we need to do any quadword copying? ADD R3, R7, R9 // R9 points just past where we copy by word forwardlargeloop: + // Copy 32 bytes at a time. LDP.P 32(R4), (R8, R10) STP.P (R8, R10), 32(R3) LDP -16(R4), (R11, R12) @@ -43,10 +44,26 @@ forwardlargeloop: CBNZ R7, forwardlargeloop noforwardlarge: - CBNZ R6, forwardtail // Do we need to do any byte-by-byte copying? + CBNZ R6, forwardtail // Do we need to copy any tail bytes? RET forwardtail: + // There are R6 <= 31 bytes remaining to copy. + // This is large enough to still contain pointers, + // which must be copied atomically. + // Copy the next 16 bytes, then 8 bytes, then any remaining bytes. + TBZ $4, R6, 3(PC) // write 16 bytes if R6&16 != 0 + LDP.P 16(R4), (R8, R10) + STP.P (R8, R10), 16(R3) + + TBZ $3, R6, 3(PC) // write 8 bytes if R6&8 != 0 + MOVD.P 8(R4), R8 + MOVD.P R8, 8(R3) + + AND $7, R6 + CBNZ R6, 2(PC) + RET + ADD R3, R6, R9 // R9 points just past the destination memory forwardtailloop: @@ -90,7 +107,7 @@ copy1: RET backward: - // Copying backwards proceeds by copying R6 bytes then copying R7/8 words. + // Copying backwards first copies R6 <= 31 tail bytes, then R7/32 quadwords. // R3 and R4 are advanced to the end of the destination/source buffers // respectively and moved back as we copy. @@ -99,13 +116,28 @@ backward: CBZ R6, nobackwardtail // Do we need to do any byte-by-byte copying? - SUB R6, R3, R9 // R9 points at the lowest destination byte that should be copied by byte. + AND $7, R6, R12 + CBZ R12, backwardtaillarge + + SUB R12, R3, R9 // R9 points at the lowest destination byte that should be copied by byte. backwardtailloop: + // Copy sub-pointer-size tail. MOVBU.W -1(R4), R8 MOVBU.W R8, -1(R3) CMP R9, R3 BNE backwardtailloop +backwardtaillarge: + // Do 8/16-byte write if possible. + // See comment at forwardtail. + TBZ $3, R6, 3(PC) + MOVD.W -8(R4), R8 + MOVD.W R8, -8(R3) + + TBZ $4, R6, 3(PC) + LDP.W -16(R4), (R8, R10) + STP.W (R8, R10), -16(R3) + nobackwardtail: CBNZ R7, backwardlarge // Do we need to do any doubleword-by-doubleword copying? RET diff --git a/src/runtime/os_windows.go b/src/runtime/os_windows.go index d3e84fe3dc..a278dddc57 100644 --- a/src/runtime/os_windows.go +++ b/src/runtime/os_windows.go @@ -49,6 +49,7 @@ const ( //go:cgo_import_dynamic runtime._VirtualFree VirtualFree%3 "kernel32.dll" //go:cgo_import_dynamic runtime._VirtualQuery VirtualQuery%3 "kernel32.dll" //go:cgo_import_dynamic runtime._WaitForSingleObject WaitForSingleObject%2 "kernel32.dll" +//go:cgo_import_dynamic runtime._WaitForMultipleObjects WaitForMultipleObjects%4 "kernel32.dll" //go:cgo_import_dynamic runtime._WriteConsoleW WriteConsoleW%5 "kernel32.dll" //go:cgo_import_dynamic runtime._WriteFile WriteFile%5 "kernel32.dll" @@ -96,6 +97,7 @@ var ( _VirtualFree, _VirtualQuery, _WaitForSingleObject, + _WaitForMultipleObjects, _WriteConsoleW, _WriteFile, _ stdFunction @@ -138,7 +140,8 @@ func tstart_stdcall(newm *m) uint32 func ctrlhandler(_type uint32) uint32 type mOS struct { - waitsema uintptr // semaphore for parking on locks + waitsema uintptr // semaphore for parking on locks + resumesema uintptr // semaphore to indicate suspend/resume } //go:linkname os_sigpipe os.sigpipe @@ -257,6 +260,53 @@ func loadOptionalSyscalls() { } } +func monitorSuspendResume() { + const ( + _DEVICE_NOTIFY_CALLBACK = 2 + _ERROR_FILE_NOT_FOUND = 2 + ) + type _DEVICE_NOTIFY_SUBSCRIBE_PARAMETERS struct { + callback uintptr + context uintptr + } + + powrprof := windowsLoadSystemLib([]byte("powrprof.dll\000")) + if powrprof == 0 { + return // Running on Windows 7, where we don't need it anyway. + } + powerRegisterSuspendResumeNotification := windowsFindfunc(powrprof, []byte("PowerRegisterSuspendResumeNotification\000")) + if powerRegisterSuspendResumeNotification == nil { + return // Running on Windows 7, where we don't need it anyway. + } + var fn interface{} = func(context uintptr, changeType uint32, setting uintptr) uintptr { + for mp := (*m)(atomic.Loadp(unsafe.Pointer(&allm))); mp != nil; mp = mp.alllink { + if mp.resumesema != 0 { + stdcall1(_SetEvent, mp.resumesema) + } + } + return 0 + } + params := _DEVICE_NOTIFY_SUBSCRIBE_PARAMETERS{ + callback: compileCallback(*efaceOf(&fn), true), + } + handle := uintptr(0) + ret := stdcall3(powerRegisterSuspendResumeNotification, _DEVICE_NOTIFY_CALLBACK, + uintptr(unsafe.Pointer(¶ms)), uintptr(unsafe.Pointer(&handle))) + // This function doesn't use GetLastError(), so we use the return value directly. + switch ret { + case 0: + return // Successful, nothing more to do. + case _ERROR_FILE_NOT_FOUND: + // Systems without access to the suspend/resume notifier + // also have their clock on "program time", and therefore + // don't want or need this anyway. + return + default: + println("runtime: PowerRegisterSuspendResumeNotification failed with errno=", ret) + throw("runtime: PowerRegisterSuspendResumeNotification failure") + } +} + //go:nosplit func getLoadLibrary() uintptr { return uintptr(unsafe.Pointer(_LoadLibraryW)) @@ -487,6 +537,10 @@ func goenvs() { } stdcall1(_FreeEnvironmentStringsW, uintptr(strings)) + + // We call this all the way here, late in init, so that malloc works + // for the callback function this generates. + monitorSuspendResume() } // exiting is set to non-zero when the process is exiting. @@ -605,19 +659,32 @@ func semasleep(ns int64) int32 { _WAIT_FAILED = 0xFFFFFFFF ) - // store ms in ns to save stack space + var result uintptr if ns < 0 { - ns = _INFINITE + result = stdcall2(_WaitForSingleObject, getg().m.waitsema, uintptr(_INFINITE)) } else { - ns = int64(timediv(ns, 1000000, nil)) - if ns == 0 { - ns = 1 + start := nanotime() + elapsed := int64(0) + for { + ms := int64(timediv(ns-elapsed, 1000000, nil)) + if ms == 0 { + ms = 1 + } + result = stdcall4(_WaitForMultipleObjects, 2, + uintptr(unsafe.Pointer(&[2]uintptr{getg().m.waitsema, getg().m.resumesema})), + 0, uintptr(ms)) + if result != _WAIT_OBJECT_0+1 { + // Not a suspend/resume event + break + } + elapsed = nanotime() - start + if elapsed >= ns { + return -1 + } } } - - result := stdcall2(_WaitForSingleObject, getg().m.waitsema, uintptr(ns)) switch result { - case _WAIT_OBJECT_0: //signaled + case _WAIT_OBJECT_0: // Signaled return 0 case _WAIT_TIMEOUT: @@ -666,6 +733,15 @@ func semacreate(mp *m) { throw("runtime.semacreate") }) } + mp.resumesema = stdcall4(_CreateEventA, 0, 0, 0, 0) + if mp.resumesema == 0 { + systemstack(func() { + print("runtime: createevent failed; errno=", getlasterror(), "\n") + throw("runtime.semacreate") + }) + stdcall1(_CloseHandle, mp.waitsema) + mp.waitsema = 0 + } } // May run with m.p==nil, so write barriers are not allowed. This diff --git a/src/runtime/syscall_windows.go b/src/runtime/syscall_windows.go index 36ad7511af..920468286b 100644 --- a/src/runtime/syscall_windows.go +++ b/src/runtime/syscall_windows.go @@ -74,16 +74,18 @@ func compileCallback(fn eface, cleanstack bool) (code uintptr) { argsize += uintptrSize } - lock(&cbs.lock) - defer unlock(&cbs.lock) + lock(&cbs.lock) // We don't unlock this in a defer because this is used from the system stack. n := cbs.n for i := 0; i < n; i++ { if cbs.ctxt[i].gobody == fn.data && cbs.ctxt[i].isCleanstack() == cleanstack { - return callbackasmAddr(i) + r := callbackasmAddr(i) + unlock(&cbs.lock) + return r } } if n >= cb_max { + unlock(&cbs.lock) throw("too many callback functions") } @@ -99,7 +101,9 @@ func compileCallback(fn eface, cleanstack bool) (code uintptr) { cbs.ctxt[n] = c cbs.n++ - return callbackasmAddr(n) + r := callbackasmAddr(n) + unlock(&cbs.lock) + return r } const _LOAD_LIBRARY_SEARCH_SYSTEM32 = 0x00000800 diff --git a/src/runtime/type.go b/src/runtime/type.go index f7f99924ea..a393da19f7 100644 --- a/src/runtime/type.go +++ b/src/runtime/type.go @@ -290,7 +290,7 @@ func (t *_type) textOff(off textOff) unsafe.Pointer { for i := range md.textsectmap { sectaddr := md.textsectmap[i].vaddr sectlen := md.textsectmap[i].length - if uintptr(off) >= sectaddr && uintptr(off) <= sectaddr+sectlen { + if uintptr(off) >= sectaddr && uintptr(off) < sectaddr+sectlen { res = md.textsectmap[i].baseaddr + uintptr(off) - uintptr(md.textsectmap[i].vaddr) break } |
