You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Go实现Python正则表达式split函数的等效输出?

Replicating Python's Regex Split Behavior in Go

First, let's break down why the Python code produces that verbose output:

  • The regex ([0-9eE.+]*) uses a capturing group that matches zero or more of the specified characters.
  • Python's re.split includes all capturing groups in the result—even empty matches.
  • Since the regex can match zero-length strings (thanks to *), it inserts empty strings between every character that doesn't fit the pattern, leading to the fragmented output you see.

Exact Replication of Python's Output in Go

Go's regexp.Split doesn't handle capturing groups or empty matches the same way as Python. To replicate the exact output, use FindAllStringSubmatch to collect all matches (including empty ones) and reconstruct the split array manually:

package main

import (
	"fmt"
	"regexp"
	"strings"
)

func main() {
	input := "Hello, where are you 1.1?"
	re := regexp.MustCompile(`([0-9eE.+]*)`)

	matches := re.FindAllStringSubmatch(input, -1)
	var result []string
	lastPos := 0

	for _, match := range matches {
		fullMatch := match[0]
		startIdx := strings.Index(input[lastPos:], fullMatch) + lastPos

		// Add non-matching characters as individual elements with trailing empty strings
		if startIdx > lastPos {
			nonMatch := input[lastPos:startIdx]
			for _, c := range nonMatch {
				result = append(result, string(c))
				result = append(result, "")
			}
		}

		// Add the capturing group value (even empty)
		result = append(result, match[1])
		lastPos = startIdx + len(fullMatch)
	}

	// Handle remaining characters after the last match
	if lastPos < len(input) {
		remaining := input[lastPos:]
		for _, c := range remaining {
			result = append(result, string(c))
			result = append(result, "")
		}
	}

	fmt.Printf("%#v\n", result)
}

This code will produce the identical output to your Python script.

Better Approach for Your End Goal

The original Python regex isn't ideal for detecting valid floats—it matches invalid sequences like "++e" or "..". For your actual goal (separating valid float strings from other characters to compute moving averages), use a more precise regex and cleaner logic:

Step 1: Split Input into Valid Float and Non-Float Segments

Use a regex that properly matches float literals, then split the input into alternating non-float and float parts:

package main

import (
	"fmt"
	"regexp"
	"strconv"
)

func splitFloatAndNonFloat(input string) ([]interface{}, error) {
	// Regex for valid float literals (supports integers, decimals, exponents, signs)
	floatRegex := regexp.MustCompile(`[-+]?\d*\.?\d+(?:[eE][-+]?\d+)?`)
	var segments []interface{}
	lastPos := 0

	for _, match := range floatRegex.FindAllStringIndex(input, -1) {
		start, end := match[0], match[1]
		// Add non-float segment
		if start > lastPos {
			segments = append(segments, input[lastPos:start])
		}
		// Parse and add float value
		floatStr := input[start:end]
		f, err := strconv.ParseFloat(floatStr, 64)
		if err != nil {
			return nil, err
		}
		segments = append(segments, f)
		lastPos = end
	}

	// Add remaining non-float content
	if lastPos < len(input) {
		segments = append(segments, input[lastPos:])
	}

	return segments, nil
}

Step 2: Compute Moving Average on Float Values

Track float values over time and calculate moving averages with a specified window size:

func computeMovingAverage(currentSegments []interface{}, previousFloats []float64, windowSize int) ([]interface{}, []float64) {
	var newSegments []interface{}
	var currentFloats []float64

	for _, seg := range currentSegments {
		switch v := seg.(type) {
		case string:
			newSegments = append(newSegments, v)
		case float64:
			currentFloats = append(currentFloats, v)
			// Calculate average if we have enough data points
			allFloats := append(previousFloats, currentFloats...)
			if len(allFloats) >= windowSize {
				window := allFloats[len(allFloats)-windowSize:]
				sum := 0.0
				for _, f := range window {
					sum += f
				}
				newSegments = append(newSegments, sum/float64(windowSize))
			} else {
				// Fall back to current value if window isn't filled
				newSegments = append(newSegments, v)
			}
		}
	}

	// Keep only the most recent windowSize floats to save memory
	previousFloats = append(previousFloats, currentFloats...)
	if len(previousFloats) > windowSize {
		previousFloats = previousFloats[len(previousFloats)-windowSize:]
	}

	return newSegments, previousFloats
}

Step 3: Reconstruct and Display Like watch

Reassemble the segments into a string and refresh it periodically, mimicking the Linux watch command:

import (
	"fmt"
	"strings"
	"time"
)

func displayLikeWatch(updateFunc func() string, interval time.Duration) {
	for {
		// Clear screen (works on Unix-like systems)
		fmt.Print("\033[H\033[2J")
		fmt.Println(updateFunc())
		time.Sleep(interval)
	}
}

func main() {
	windowSize := 3
	var previousFloats []float64

	update := func() string {
		// Replace with your actual input source (file, stdin, etc.)
		input := "Hello, where are you 1.1? The value is 2.5 and another 3.0"
		segments, _ := splitFloatAndNonFloat(input)
		newSegments, prev := computeMovingAverage(segments, previousFloats, windowSize)
		previousFloats = prev

		// Reconstruct the final string
		var result strings.Builder
		for _, seg := range newSegments {
			switch v := seg.(type) {
			case string:
				result.WriteString(v)
			case float64:
				result.WriteString(fmt.Sprintf("%.2f", v))
			}
		}
		return result.String()
	}

	displayLikeWatch(update, 2*time.Second)
}

This approach is far more robust for your intended use case than replicating the original Python's fragmented split output.

内容的提问来源于stack exchange,提问作者harshavmb

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 16:00:52