816 lines
30 KiB
Go
816 lines
30 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
|
|
// SPDX-License-Identifier: MIT
|
|||
|
|
|
|||
|
|
package signal
|
|||
|
|
|
|||
|
|
import (
|
|||
|
|
"math"
|
|||
|
|
|
|||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
|||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/engine"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
// Pooling operations for 2-D feature maps. All functions take
|
|||
|
|
// NCHW input (N, C, H, W) and return (N, C, H_out, W_out). Padding is
|
|||
|
|
// a single int applied to both spatial dims; stride is a single int.
|
|||
|
|
// Padding must stay below the kernel: a window entirely inside the
|
|||
|
|
// padding holds no data, and a max would answer −Inf there, an
|
|||
|
|
// average 0.
|
|||
|
|
//
|
|||
|
|
// The kernels hold two constraints fixed:
|
|||
|
|
//
|
|||
|
|
// - Element order. Average pooling adds its window elements in the
|
|||
|
|
// (kernel height, kernel width) nesting and divides once, with the
|
|||
|
|
// countIncludePad divisor chosen exactly as before; any other
|
|||
|
|
// order or divisor changes bits. Max pooling is order free but
|
|||
|
|
// keeps the same walk for a single shared code path, and a NaN in
|
|||
|
|
// a window wins: v > best is false for NaN, so the comparison
|
|||
|
|
// would silently drop it and answer the largest finite neighbour
|
|||
|
|
// instead.
|
|||
|
|
// - Parallel split. Work is distributed over output rows, which own
|
|||
|
|
// disjoint output elements, so no split can influence a result.
|
|||
|
|
//
|
|||
|
|
// Non-float inputs widen once into a scratch payload (see
|
|||
|
|
// widenFloats): FloatAt widens exactly, so the values
|
|||
|
|
// compared and summed are bit-identical to the per-element accessor
|
|||
|
|
// reads.
|
|||
|
|
|
|||
|
|
// parallelWorkBudget is the number of element updates a worker should
|
|||
|
|
// carry before splitting a kernel pays for the worker's spawn. Every
|
|||
|
|
// scheduling floor in the package is derived from it: an item body
|
|||
|
|
// costing a few hundred nanoseconds still leaves a worker below the
|
|||
|
|
// budget on a tiny input, where the split would spawn goroutines to do
|
|||
|
|
// less work than the spawn itself costs.
|
|||
|
|
const parallelWorkBudget = 1 << 12
|
|||
|
|
|
|||
|
|
// workFloorFor returns the smallest number of items a worker should
|
|||
|
|
// carry when one item costs about workPerItem element updates. It is
|
|||
|
|
// the floor argument engine.ParallelMin wants: below it the kernel runs
|
|||
|
|
// whole on the calling goroutine, and above it the split is the one
|
|||
|
|
// engine.Parallel computes, so the worker choice and the chunk
|
|||
|
|
// boundaries of a split kernel are unchanged.
|
|||
|
|
func workFloorFor(workPerItem int) int {
|
|||
|
|
if workPerItem < 1 {
|
|||
|
|
return parallelWorkBudget
|
|||
|
|
}
|
|||
|
|
return max(1, parallelWorkBudget/workPerItem)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// MaxPool2D returns the maximum over each kernel-sized window. A NaN
|
|||
|
|
// in a window propagates: the maximum answers NaN.
|
|||
|
|
func MaxPool2D(input *core.Array, kernel, stride, padding int) (*core.Array, error) {
|
|||
|
|
return pool2D(input, kernel, stride, padding, true, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AvgPool2D returns the average over each kernel-sized window. When
|
|||
|
|
// countIncludePad is true the divisor is kH*kW; otherwise it's the
|
|||
|
|
// number of non-padded elements in the window.
|
|||
|
|
func AvgPool2D(input *core.Array, kernel, stride, padding int, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
return pool2D(input, kernel, stride, padding, false, countIncludePad)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// MaxPool1D returns the maximum over each kernel-sized window of a
|
|||
|
|
// 3-D tensor (N, C, L). A NaN in a window propagates: the maximum
|
|||
|
|
// answers NaN.
|
|||
|
|
func MaxPool1D(input *core.Array, kernel, stride, padding int) (*core.Array, error) {
|
|||
|
|
return pool1D(input, kernel, stride, padding, true, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AvgPool1D returns the average over each kernel-sized window of a
|
|||
|
|
// 3-D tensor (N, C, L). When countIncludePad is true the divisor is
|
|||
|
|
// the full kernel size; otherwise it is the number of non-padded
|
|||
|
|
// elements in the window.
|
|||
|
|
func AvgPool1D(input *core.Array, kernel, stride, padding int, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
return pool1D(input, kernel, stride, padding, false, countIncludePad)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// MaxPool3D returns the maximum over each kernel-sized window of a
|
|||
|
|
// 5-D tensor (N, C, D, H, W). A NaN in a window propagates: the
|
|||
|
|
// maximum answers NaN.
|
|||
|
|
func MaxPool3D(input *core.Array, kernel [3]int, stride [3]int, padding [3]int) (*core.Array, error) {
|
|||
|
|
return pool3D(input, kernel, stride, padding, true, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AvgPool3D returns the average over each kernel-sized window of a
|
|||
|
|
// 5-D tensor (N, C, D, H, W). When countIncludePad is true the
|
|||
|
|
// divisor is the full kernel size; otherwise it is the number of
|
|||
|
|
// non-padded elements in the window.
|
|||
|
|
func AvgPool3D(input *core.Array, kernel [3]int, stride [3]int, padding [3]int, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
return pool3D(input, kernel, stride, padding, false, countIncludePad)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveMaxPool2D pools to the requested output size by taking the
|
|||
|
|
// maximum over each output window. Window o covers the input rows
|
|||
|
|
// from floor(o·H_in/H_out) to ceil((o+1)·H_in/H_out), so every input
|
|||
|
|
// sample lands in some window. A NaN in a window propagates: the
|
|||
|
|
// maximum answers NaN.
|
|||
|
|
func AdaptiveMaxPool2D(input *core.Array, outputH, outputW int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 4 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool2D: input must be 4-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool2D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputH < 1 || outputW < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool2D: output size must be at least 1")
|
|||
|
|
}
|
|||
|
|
n, c, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3]
|
|||
|
|
// A zero spatial dimension leaves every window empty: max would
|
|||
|
|
// answer -Inf, so it is an error like every other shape refusal.
|
|||
|
|
if hIn < 1 || wIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool2D: the spatial dimensions must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputH, outputW}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output row; rows are disjoint. A row walks
|
|||
|
|
// about one window height of input rows per window column.
|
|||
|
|
items := n * c * outputH
|
|||
|
|
floor := workFloorFor(max(1, (hIn+outputH-1)/outputH) * wIn)
|
|||
|
|
engine.ParallelMin(items, floor, func(bs, be int) {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oy := item % outputH
|
|||
|
|
row := item / outputH
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*hIn*wIn + ch*hIn*wIn
|
|||
|
|
yStart := oy * hIn / outputH
|
|||
|
|
yEnd := ((oy+1)*hIn + outputH - 1) / outputH
|
|||
|
|
for ox := range outputW {
|
|||
|
|
xStart := ox * wIn / outputW
|
|||
|
|
xEnd := ((ox+1)*wIn + outputW - 1) / outputW
|
|||
|
|
best := math.Inf(-1)
|
|||
|
|
for y := yStart; y < yEnd; y++ {
|
|||
|
|
inRow := chanBase + y*wIn
|
|||
|
|
for x := xStart; x < xEnd; x++ {
|
|||
|
|
if v := inF[inRow+x]; v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
outF[batch*c*outputH*outputW+ch*outputH*outputW+oy*outputW+ox] = best
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalAvgPool2D returns AdaptiveAvgPool2D reduced to (1, 1), the
|
|||
|
|
// global average pooling operation common in modern CNNs.
|
|||
|
|
func GlobalAvgPool2D(input *core.Array) (*core.Array, error) {
|
|||
|
|
return AdaptiveAvgPool2D(input, 1, 1)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalMaxPool2D returns the global maximum of a 4-D feature map.
|
|||
|
|
func GlobalMaxPool2D(input *core.Array) (*core.Array, error) {
|
|||
|
|
return AdaptiveMaxPool2D(input, 1, 1)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalAvgPool1D reduces a 3-D input (N, C, L) to shape (N, C, 1).
|
|||
|
|
func GlobalAvgPool1D(input *core.Array) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 3 {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool1D: input must be 3-D (N, C, L), got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool1D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if input.Shape()[2] == 0 {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool1D: the spatial dimension must not be empty")
|
|||
|
|
}
|
|||
|
|
return pool1D(input, input.Shape()[2], 1, 0, false, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalMaxPool1D reduces a 3-D input (N, C, L) to shape (N, C, 1).
|
|||
|
|
func GlobalMaxPool1D(input *core.Array) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 3 {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool1D: input must be 3-D (N, C, L), got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool1D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if input.Shape()[2] == 0 {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool1D: the spatial dimension must not be empty")
|
|||
|
|
}
|
|||
|
|
return pool1D(input, input.Shape()[2], 1, 0, true, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalAvgPool3D reduces a 5-D input (N, C, D, H, W) to (N, C, 1, 1, 1).
|
|||
|
|
func GlobalAvgPool3D(input *core.Array) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 5 {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool3D: input must be 5-D (N, C, D, H, W), got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool3D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if input.Shape()[2] == 0 || input.Shape()[3] == 0 || input.Shape()[4] == 0 {
|
|||
|
|
return nil, base.Errf("GlobalAvgPool3D: the spatial dimensions must not be empty")
|
|||
|
|
}
|
|||
|
|
return pool3D(input, [3]int{input.Shape()[2], input.Shape()[3], input.Shape()[4]}, [3]int{1, 1, 1}, [3]int{0, 0, 0}, false, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// GlobalMaxPool3D reduces a 5-D input (N, C, D, H, W) to (N, C, 1, 1, 1).
|
|||
|
|
func GlobalMaxPool3D(input *core.Array) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 5 {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool3D: input must be 5-D (N, C, D, H, W), got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool3D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if input.Shape()[2] == 0 || input.Shape()[3] == 0 || input.Shape()[4] == 0 {
|
|||
|
|
return nil, base.Errf("GlobalMaxPool3D: the spatial dimensions must not be empty")
|
|||
|
|
}
|
|||
|
|
return pool3D(input, [3]int{input.Shape()[2], input.Shape()[3], input.Shape()[4]}, [3]int{1, 1, 1}, [3]int{0, 0, 0}, true, false)
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveMaxPool1D pools the input to the requested output size by
|
|||
|
|
// taking the maximum over each output window. Window o covers the
|
|||
|
|
// input positions from floor(o·L_in/L_out) to ceil((o+1)·L_in/L_out),
|
|||
|
|
// so every input sample lands in some window. Input is (N, C, L). A
|
|||
|
|
// NaN in a window propagates: the maximum answers NaN.
|
|||
|
|
func AdaptiveMaxPool1D(input *core.Array, outputL int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 3 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool1D: input must be 3-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool1D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputL < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool1D: output size must be at least 1")
|
|||
|
|
}
|
|||
|
|
n, c, lIn := input.Shape()[0], input.Shape()[1], input.Shape()[2]
|
|||
|
|
if lIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool1D: the spatial dimension must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputL}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per (batch, channel) row; rows are disjoint.
|
|||
|
|
engine.ParallelMin(n*c, workFloorFor(lIn), func(bs, be int) {
|
|||
|
|
for row := bs; row < be; row++ {
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*lIn + ch*lIn
|
|||
|
|
for ol := range outputL {
|
|||
|
|
start := ol * lIn / outputL
|
|||
|
|
end := ((ol+1)*lIn + outputL - 1) / outputL
|
|||
|
|
best := math.Inf(-1)
|
|||
|
|
for l := start; l < end; l++ {
|
|||
|
|
if v := inF[chanBase+l]; v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
outF[batch*c*outputL+ch*outputL+ol] = best
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveMaxPool3D pools the input to the requested output size by
|
|||
|
|
// taking the maximum over each output window, with the same
|
|||
|
|
// floor-start, ceil-end convention as AdaptiveMaxPool1D. Input is
|
|||
|
|
// (N, C, D, H, W). A NaN in a window propagates: the maximum answers
|
|||
|
|
// NaN.
|
|||
|
|
func AdaptiveMaxPool3D(input *core.Array, outputD, outputH, outputW int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 5 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool3D: input must be 5-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool3D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputD < 1 || outputH < 1 || outputW < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool3D: output sizes must be at least 1")
|
|||
|
|
}
|
|||
|
|
n, c, dIn, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3], input.Shape()[4]
|
|||
|
|
if dIn < 1 || hIn < 1 || wIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveMaxPool3D: the spatial dimensions must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputD, outputH, outputW}...)
|
|||
|
|
strideHW := wIn
|
|||
|
|
strideDHW := hIn * wIn
|
|||
|
|
strideCDHW := dIn * hIn * wIn
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output depth-height row; rows are disjoint.
|
|||
|
|
items := n * c * outputD * outputH
|
|||
|
|
floor := workFloorFor(wIn * max(1, (hIn+outputH-1)/outputH) * max(1, (dIn+outputD-1)/outputD))
|
|||
|
|
engine.ParallelMin(items, floor, func(bs, be int) {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oh := item % outputH
|
|||
|
|
t := item / outputH
|
|||
|
|
od := t % outputD
|
|||
|
|
t /= outputD
|
|||
|
|
ch := t % c
|
|||
|
|
batch := t / c
|
|||
|
|
chanBase := batch*c*strideCDHW + ch*strideCDHW
|
|||
|
|
dStart := od * dIn / outputD
|
|||
|
|
dEnd := ((od+1)*dIn + outputD - 1) / outputD
|
|||
|
|
hStart := oh * hIn / outputH
|
|||
|
|
hEnd := ((oh+1)*hIn + outputH - 1) / outputH
|
|||
|
|
for ow := range outputW {
|
|||
|
|
wStart := ow * wIn / outputW
|
|||
|
|
wEnd := ((ow+1)*wIn + outputW - 1) / outputW
|
|||
|
|
best := math.Inf(-1)
|
|||
|
|
for d := dStart; d < dEnd; d++ {
|
|||
|
|
for h := hStart; h < hEnd; h++ {
|
|||
|
|
inRow := chanBase + d*strideDHW + h*strideHW
|
|||
|
|
for w := wStart; w < wEnd; w++ {
|
|||
|
|
if v := inF[inRow+w]; v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
outF[batch*c*outputD*outputH*outputW+ch*outputD*outputH*outputW+od*outputH*outputW+oh*outputW+ow] = best
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveAvgPool1D pools to the requested output size by averaging
|
|||
|
|
// each output window, with the same floor-start, ceil-end convention
|
|||
|
|
// as AdaptiveMaxPool1D. Input is (N, C, L).
|
|||
|
|
func AdaptiveAvgPool1D(input *core.Array, outputL int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 3 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool1D: input must be 3-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool1D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputL < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool1D: output size must be at least 1")
|
|||
|
|
}
|
|||
|
|
n, c, lIn := input.Shape()[0], input.Shape()[1], input.Shape()[2]
|
|||
|
|
if lIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool1D: the spatial dimension must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputL}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per (batch, channel) row; rows are disjoint and
|
|||
|
|
// every window sums its elements in ascending order.
|
|||
|
|
engine.ParallelMin(n*c, workFloorFor(lIn), func(bs, be int) {
|
|||
|
|
for row := bs; row < be; row++ {
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*lIn + ch*lIn
|
|||
|
|
for ol := range outputL {
|
|||
|
|
start := ol * lIn / outputL
|
|||
|
|
end := ((ol+1)*lIn + outputL - 1) / outputL
|
|||
|
|
var sum float64
|
|||
|
|
for l := start; l < end; l++ {
|
|||
|
|
sum += inF[chanBase+l]
|
|||
|
|
}
|
|||
|
|
outF[batch*c*outputL+ch*outputL+ol] = sum / float64(end-start)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveAvgPool3D pools to the requested output size by averaging
|
|||
|
|
// each output window, with the same floor-start, ceil-end convention
|
|||
|
|
// as AdaptiveMaxPool1D. Input is (N, C, D, H, W).
|
|||
|
|
func AdaptiveAvgPool3D(input *core.Array, outputD, outputH, outputW int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 5 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool3D: input must be 5-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool3D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputD < 1 || outputH < 1 || outputW < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool3D: output sizes must be at least 1")
|
|||
|
|
}
|
|||
|
|
n, c, dIn, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3], input.Shape()[4]
|
|||
|
|
if dIn < 1 || hIn < 1 || wIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool3D: the spatial dimensions must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputD, outputH, outputW}...)
|
|||
|
|
strideHW := wIn
|
|||
|
|
strideDHW := hIn * wIn
|
|||
|
|
strideCDHW := dIn * hIn * wIn
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output element here: the 3-D windows share no
|
|||
|
|
// row structure worth exploiting, and every element sums its own
|
|||
|
|
// window in ascending order.
|
|||
|
|
items := n * c * outputD * outputH * outputW
|
|||
|
|
floor := workFloorFor(max(1, (dIn+outputD-1)/outputD) * max(1, (hIn+outputH-1)/outputH) * max(1, (wIn+outputW-1)/outputW))
|
|||
|
|
engine.ParallelMin(items, floor, func(bs, be int) {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
ow := item % outputW
|
|||
|
|
t := item / outputW
|
|||
|
|
oh := t % outputH
|
|||
|
|
t /= outputH
|
|||
|
|
od := t % outputD
|
|||
|
|
t /= outputD
|
|||
|
|
ch := t % c
|
|||
|
|
batch := t / c
|
|||
|
|
chanBase := batch*c*strideCDHW + ch*strideCDHW
|
|||
|
|
dStart := od * dIn / outputD
|
|||
|
|
dEnd := ((od+1)*dIn + outputD - 1) / outputD
|
|||
|
|
hStart := oh * hIn / outputH
|
|||
|
|
hEnd := ((oh+1)*hIn + outputH - 1) / outputH
|
|||
|
|
wStart := ow * wIn / outputW
|
|||
|
|
wEnd := ((ow+1)*wIn + outputW - 1) / outputW
|
|||
|
|
var sum float64
|
|||
|
|
count := 0
|
|||
|
|
for d := dStart; d < dEnd; d++ {
|
|||
|
|
for h := hStart; h < hEnd; h++ {
|
|||
|
|
inRow := chanBase + d*strideDHW + h*strideHW
|
|||
|
|
for w := wStart; w < wEnd; w++ {
|
|||
|
|
sum += inF[inRow+w]
|
|||
|
|
count++
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if count > 0 {
|
|||
|
|
outF[batch*c*outputD*outputH*outputW+ch*outputD*outputH*outputW+od*outputH*outputW+oh*outputW+ow] = sum / float64(count)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// AdaptiveAvgPool2D pools the input to the requested output size by
|
|||
|
|
// averaging each output window, with the same floor-start, ceil-end
|
|||
|
|
// convention as AdaptiveMaxPool2D. outputH and outputW must both be
|
|||
|
|
// at least 1.
|
|||
|
|
func AdaptiveAvgPool2D(input *core.Array, outputH, outputW int) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 4 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool2D: input must be 4-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool2D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if outputH < 1 || outputW < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool2D: output size must be at least 1, got %dx%d", outputH, outputW)
|
|||
|
|
}
|
|||
|
|
n, c, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3]
|
|||
|
|
// A zero spatial dimension leaves every window empty: avg would
|
|||
|
|
// divide by zero, so it is an error like every other shape refusal.
|
|||
|
|
if hIn < 1 || wIn < 1 {
|
|||
|
|
return nil, base.Errf("AdaptiveAvgPool2D: the spatial dimensions must not be empty, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, outputH, outputW}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output row; rows are disjoint and every window
|
|||
|
|
// sums its elements in ascending order.
|
|||
|
|
items := n * c * outputH
|
|||
|
|
engine.ParallelMin(items, workFloorFor(max(1, (hIn+outputH-1)/outputH)*wIn), func(bs, be int) {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oy := item % outputH
|
|||
|
|
row := item / outputH
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*hIn*wIn + ch*hIn*wIn
|
|||
|
|
// Window in the input: floor start, ceil end, so
|
|||
|
|
// every input sample lands in some window.
|
|||
|
|
yStart := oy * hIn / outputH
|
|||
|
|
yEnd := ((oy+1)*hIn + outputH - 1) / outputH
|
|||
|
|
for ox := range outputW {
|
|||
|
|
xStart := ox * wIn / outputW
|
|||
|
|
xEnd := ((ox+1)*wIn + outputW - 1) / outputW
|
|||
|
|
var sum float64
|
|||
|
|
for y := yStart; y < yEnd; y++ {
|
|||
|
|
inRow := chanBase + y*wIn
|
|||
|
|
for x := xStart; x < xEnd; x++ {
|
|||
|
|
sum += inF[inRow+x]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
count := float64((yEnd - yStart) * (xEnd - xStart))
|
|||
|
|
off := batch*c*outputH*outputW + ch*outputH*outputW + oy*outputW + ox
|
|||
|
|
outF[off] = sum / count
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// pool2D is the shared implementation behind MaxPool2D and AvgPool2D.
|
|||
|
|
func pool2D(input *core.Array, kernel, stride, padding int, isMax, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 4 {
|
|||
|
|
return nil, base.Errf("Pool2D: input must be 4-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("Pool2D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if kernel < 1 {
|
|||
|
|
return nil, base.Errf("Pool2D: kernel must be at least 1, got %d", kernel)
|
|||
|
|
}
|
|||
|
|
if stride < 1 {
|
|||
|
|
return nil, base.Errf("Pool2D: stride must be at least 1, got %d", stride)
|
|||
|
|
}
|
|||
|
|
if padding < 0 {
|
|||
|
|
return nil, base.Errf("Pool2D: padding must be non-negative, got %d", padding)
|
|||
|
|
}
|
|||
|
|
// Padding from the kernel upward leaves output windows entirely
|
|||
|
|
// inside the padding (the first window covers −padding ..
|
|||
|
|
// −padding+kernel−1): a max would answer −Inf and an average 0,
|
|||
|
|
// so such configurations are refused, in the spirit of the
|
|||
|
|
// common pooling implementations' padding bound.
|
|||
|
|
if padding >= kernel {
|
|||
|
|
return nil, base.Errf("Pool2D: padding %d must stay below the kernel %d, some windows would hold no data", padding, kernel)
|
|||
|
|
}
|
|||
|
|
n, c, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3]
|
|||
|
|
// Guard the numerators before the division: a negative numerator
|
|||
|
|
// truncates toward zero and masquerades as a 1-wide output.
|
|||
|
|
hNum := hIn + 2*padding - kernel
|
|||
|
|
wNum := wIn + 2*padding - kernel
|
|||
|
|
if hNum < 0 || wNum < 0 {
|
|||
|
|
return nil, base.Errf("Pool2D: kernel %d with padding %d does not fit the %dx%d input", kernel, padding, hIn, wIn)
|
|||
|
|
}
|
|||
|
|
hOut := hNum/stride + 1
|
|||
|
|
wOut := wNum/stride + 1
|
|||
|
|
if hOut < 1 || wOut < 1 {
|
|||
|
|
return nil, base.Errf("Pool2D: output is empty (H_out=%d, W_out=%d)", hOut, wOut)
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, hOut, wOut}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output row (batch, channel, oy). Rows are
|
|||
|
|
// disjoint. The max and average variants walk the same windows in
|
|||
|
|
// the same (kernel height, kernel width) order, but as two loop
|
|||
|
|
// bodies: the max/sum choice leaves the innermost loop, which would
|
|||
|
|
// otherwise pay for a perfectly predicted branch on every tap, and
|
|||
|
|
// neither variant's arithmetic changes a bit.
|
|||
|
|
items := n * c * hOut
|
|||
|
|
engine.ParallelMin(items, workFloorFor(wOut*kernel*kernel), func(bs, be int) {
|
|||
|
|
if isMax {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oy := item % hOut
|
|||
|
|
row := item / hOut
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*hIn*wIn + ch*hIn*wIn
|
|||
|
|
for ox := range wOut {
|
|||
|
|
best := math.Inf(-1)
|
|||
|
|
for kh := range kernel {
|
|||
|
|
iy := oy*stride + kh - padding
|
|||
|
|
if iy < 0 || iy >= hIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
inRow := chanBase + iy*wIn
|
|||
|
|
for kw := range kernel {
|
|||
|
|
ix := ox*stride + kw - padding
|
|||
|
|
if ix < 0 || ix >= wIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
if v := inF[inRow+ix]; v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
off := batch*c*hOut*wOut + ch*hOut*wOut + oy*wOut + ox
|
|||
|
|
outF[off] = best
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
return
|
|||
|
|
}
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oy := item % hOut
|
|||
|
|
row := item / hOut
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*hIn*wIn + ch*hIn*wIn
|
|||
|
|
for ox := range wOut {
|
|||
|
|
var acc float64
|
|||
|
|
var count int
|
|||
|
|
for kh := range kernel {
|
|||
|
|
iy := oy*stride + kh - padding
|
|||
|
|
if iy < 0 || iy >= hIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
inRow := chanBase + iy*wIn
|
|||
|
|
for kw := range kernel {
|
|||
|
|
ix := ox*stride + kw - padding
|
|||
|
|
if ix < 0 || ix >= wIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
acc += inF[inRow+ix]
|
|||
|
|
count++
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if countIncludePad {
|
|||
|
|
acc /= float64(kernel * kernel)
|
|||
|
|
} else if count > 0 {
|
|||
|
|
acc /= float64(count)
|
|||
|
|
}
|
|||
|
|
off := batch*c*hOut*wOut + ch*hOut*wOut + oy*wOut + ox
|
|||
|
|
outF[off] = acc
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// pool1D is the shared implementation behind MaxPool1D and AvgPool1D.
|
|||
|
|
// countIncludePad is honoured only by the average variant.
|
|||
|
|
func pool1D(input *core.Array, kernel, stride, padding int, isMax, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 3 {
|
|||
|
|
return nil, base.Errf("Pool1D: input must be 3-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("Pool1D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
if kernel < 1 || stride < 1 || padding < 0 {
|
|||
|
|
return nil, base.Errf("Pool1D: kernel/stride ≥1, padding ≥0")
|
|||
|
|
}
|
|||
|
|
// Padding from the kernel upward leaves output windows entirely
|
|||
|
|
// inside the padding (the first window covers −padding ..
|
|||
|
|
// −padding+kernel−1): a max would answer −Inf and an average 0,
|
|||
|
|
// so such configurations are refused, in the spirit of the
|
|||
|
|
// common pooling implementations' padding bound.
|
|||
|
|
if padding >= kernel {
|
|||
|
|
return nil, base.Errf("Pool1D: padding %d must stay below the kernel %d, some windows would hold no data", padding, kernel)
|
|||
|
|
}
|
|||
|
|
n, c, lIn := input.Shape()[0], input.Shape()[1], input.Shape()[2]
|
|||
|
|
lNum := lIn + 2*padding - kernel
|
|||
|
|
if lNum < 0 {
|
|||
|
|
return nil, base.Errf("Pool1D: kernel %d with padding %d does not fit the length-%d input", kernel, padding, lIn)
|
|||
|
|
}
|
|||
|
|
lOut := lNum/stride + 1
|
|||
|
|
if lOut < 1 {
|
|||
|
|
return nil, base.Errf("Pool1D: output is empty (L_out=%d)", lOut)
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, lOut}...)
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per (batch, channel) row; rows are disjoint and
|
|||
|
|
// every window walks its elements in ascending order.
|
|||
|
|
engine.ParallelMin(n*c, workFloorFor(lOut*kernel), func(bs, be int) {
|
|||
|
|
for row := bs; row < be; row++ {
|
|||
|
|
ch := row % c
|
|||
|
|
batch := row / c
|
|||
|
|
chanBase := batch*c*lIn + ch*lIn
|
|||
|
|
for ol := range lOut {
|
|||
|
|
var acc float64
|
|||
|
|
var best float64
|
|||
|
|
if isMax {
|
|||
|
|
best = math.Inf(-1)
|
|||
|
|
}
|
|||
|
|
var count int
|
|||
|
|
for kl := range kernel {
|
|||
|
|
il := ol*stride + kl - padding
|
|||
|
|
if il < 0 || il >= lIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
v := inF[chanBase+il]
|
|||
|
|
if isMax {
|
|||
|
|
if v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
acc += v
|
|||
|
|
}
|
|||
|
|
count++
|
|||
|
|
}
|
|||
|
|
if isMax {
|
|||
|
|
outF[batch*c*lOut+ch*lOut+ol] = best
|
|||
|
|
} else if countIncludePad {
|
|||
|
|
outF[batch*c*lOut+ch*lOut+ol] = acc / float64(kernel)
|
|||
|
|
} else if count > 0 {
|
|||
|
|
outF[batch*c*lOut+ch*lOut+ol] = acc / float64(count)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// pool3D is the shared implementation behind MaxPool3D and AvgPool3D.
|
|||
|
|
// countIncludePad is honoured only by the average variant.
|
|||
|
|
func pool3D(input *core.Array, kernel, stride, padding [3]int, isMax, countIncludePad bool) (*core.Array, error) {
|
|||
|
|
if input.NDim() != 5 {
|
|||
|
|
return nil, base.Errf("Pool3D: input must be 5-D, got shape %s", base.ShapeText(input.Shape()))
|
|||
|
|
}
|
|||
|
|
if input.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("Pool3D: complex arrays are not supported")
|
|||
|
|
}
|
|||
|
|
for d := range kernel {
|
|||
|
|
if kernel[d] < 1 || stride[d] < 1 || padding[d] < 0 {
|
|||
|
|
return nil, base.Errf("Pool3D: kernel/stride ≥1, padding ≥0")
|
|||
|
|
}
|
|||
|
|
// Padding from the kernel upward leaves output windows
|
|||
|
|
// entirely inside the padding: a max would answer −Inf and
|
|||
|
|
// an average 0, so such configurations are refused, in the
|
|||
|
|
// spirit of the common pooling implementations' padding
|
|||
|
|
// bound.
|
|||
|
|
if padding[d] >= kernel[d] {
|
|||
|
|
return nil, base.Errf("Pool3D: padding %d must stay below the kernel %d in dimension %d, some windows would hold no data", padding[d], kernel[d], d)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
n, c, dIn, hIn, wIn := input.Shape()[0], input.Shape()[1], input.Shape()[2], input.Shape()[3], input.Shape()[4]
|
|||
|
|
// Guard the numerators before the division: a negative numerator
|
|||
|
|
// truncates toward zero and masquerades as a 1-wide output.
|
|||
|
|
dNum := dIn + 2*padding[0] - kernel[0]
|
|||
|
|
hNum := hIn + 2*padding[1] - kernel[1]
|
|||
|
|
wNum := wIn + 2*padding[2] - kernel[2]
|
|||
|
|
if dNum < 0 || hNum < 0 || wNum < 0 {
|
|||
|
|
return nil, base.Errf("Pool3D: kernel %v with padding %v does not fit the input", kernel, padding)
|
|||
|
|
}
|
|||
|
|
dOut := dNum/stride[0] + 1
|
|||
|
|
hOut := hNum/stride[1] + 1
|
|||
|
|
wOut := wNum/stride[2] + 1
|
|||
|
|
if dOut < 1 || hOut < 1 || wOut < 1 {
|
|||
|
|
return nil, base.Errf("Pool3D: output is empty")
|
|||
|
|
}
|
|||
|
|
out := core.New(core.Float, []int{n, c, dOut, hOut, wOut}...)
|
|||
|
|
strideHW := wIn
|
|||
|
|
strideDHW := hIn * wIn
|
|||
|
|
strideCDHW := dIn * hIn * wIn
|
|||
|
|
inF := input.RawFloats()
|
|||
|
|
if inF == nil || input.Strided() {
|
|||
|
|
inF = widenFloats(input)
|
|||
|
|
}
|
|||
|
|
outF := out.RawFloats()
|
|||
|
|
// One work item per output depth-height row (batch, channel, od,
|
|||
|
|
// oh); rows are disjoint and each window walks its elements in the
|
|||
|
|
// (kernel depth, kernel height, kernel width) order.
|
|||
|
|
items := n * c * dOut * hOut
|
|||
|
|
engine.ParallelMin(items, workFloorFor(wOut*kernel[0]*kernel[1]*kernel[2]), func(bs, be int) {
|
|||
|
|
for item := bs; item < be; item++ {
|
|||
|
|
oh := item % hOut
|
|||
|
|
t := item / hOut
|
|||
|
|
od := t % dOut
|
|||
|
|
t /= dOut
|
|||
|
|
ch := t % c
|
|||
|
|
batch := t / c
|
|||
|
|
chanBase := batch*c*strideCDHW + ch*strideCDHW
|
|||
|
|
for ow := range wOut {
|
|||
|
|
var acc float64
|
|||
|
|
var best float64
|
|||
|
|
if isMax {
|
|||
|
|
best = math.Inf(-1)
|
|||
|
|
}
|
|||
|
|
var count int
|
|||
|
|
for kd := range kernel[0] {
|
|||
|
|
id := od*stride[0] + kd - padding[0]
|
|||
|
|
if id < 0 || id >= dIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
inDepth := chanBase + id*strideDHW
|
|||
|
|
for kh := range kernel[1] {
|
|||
|
|
ih := oh*stride[1] + kh - padding[1]
|
|||
|
|
if ih < 0 || ih >= hIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
inRow := inDepth + ih*strideHW
|
|||
|
|
for kw := range kernel[2] {
|
|||
|
|
iw := ow*stride[2] + kw - padding[2]
|
|||
|
|
if iw < 0 || iw >= wIn {
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
v := inF[inRow+iw]
|
|||
|
|
if isMax {
|
|||
|
|
if v > best || math.IsNaN(v) {
|
|||
|
|
best = v
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
acc += v
|
|||
|
|
}
|
|||
|
|
count++
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
off := batch*c*dOut*hOut*wOut + ch*dOut*hOut*wOut + od*hOut*wOut + oh*wOut + ow
|
|||
|
|
if isMax {
|
|||
|
|
outF[off] = best
|
|||
|
|
} else if countIncludePad {
|
|||
|
|
outF[off] = acc / float64(kernel[0]*kernel[1]*kernel[2])
|
|||
|
|
} else if count > 0 {
|
|||
|
|
outF[off] = acc / float64(count)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
})
|
|||
|
|
return out, nil
|
|||
|
|
}
|