You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Python中可视化含尖峰的密集间隙型传感器时间序列?

可视化带间隙、尖峰的密集传感器时间序列:6种方法对比

针对你描述的12小时高采样率(约25万数据点)、含周期性离线间隙(11个连续段,段间30-40分钟)、存在大幅尖峰的传感器时间序列,我整理了6种不同的可视化方案,包含颜色编码和阈值线展示,附带完整可复现代码。


完整可复现代码

import numpy as np
import matplotlib
matplotlib.use('Agg')
import matplotlib.pyplot as plt

np.random.seed(42)

# 模拟间隙型传感器时间序列:
# 约12小时的记录分为约11个连续段
# 段间存在30-40分钟的间隙(传感器周期性离线)
segments = []
t_start = 0
for i in range(11):
    duration = np.random.uniform(2000, 3000)
    n_points = int(duration / 0.1)
    t = np.linspace(t_start, t_start + duration, n_points)
    
    base = 70 + 20 * np.sin(2 * np.pi * t / 5000)
    noise = np.cumsum(np.random.randn(n_points)) * 0.3
    noise -= np.mean(noise)
    rate = base + noise + np.random.poisson(base) - base
    
    n_spikes = np.random.poisson(3)
    for _ in range(n_spikes):
        spike_pos = np.random.randint(0, n_points)
        spike_amp = np.random.exponential(100)
        spike_width = np.random.uniform(5, 50)
        spike = spike_amp * np.exp(-0.5 * ((t - t[spike_pos]) / spike_width) ** 2)
        rate += spike
    
    rate = np.maximum(rate, 0)
    segments.append((t / 3600, rate))  # 以小时为单位存储
    gap = np.random.uniform(1800, 2400)
    t_start += duration + gap

# 颜色编码用阈值
all_rates = np.concatenate([s[1] for s in segments])
med = np.median(all_rates[all_rates > 0])
thresh1 = med
thresh2 = med * 2

print(f"总数据点数: {len(all_rates)}")
print(f"连续段数量: {len(segments)}")
print(f"中位数: {med:.1f}, 2倍中位数: {thresh2:.1f}, 最大值: {np.max(all_rates):.1f}")

def color_for_rate(r, t1=thresh1, t2=thresh2):
    return np.where(r < t1, '#2166ac', np.where(r < t2, '#ff8c00', '#cc0000'))

def bin_segment(t, r, bin_width_hrs=100/3600):
    """对单个连续段进行分箱处理。"""
    if len(t) < 2:
        return np.array([]), np.array([]), np.array([])
    t_bins, r_bins, e_bins = [], [], []
    t_min, t_max = t[0], t[-1]
    edges = np.arange(t_min, t_max, bin_width_hrs)
    for j in range(len(edges) - 1):
        mask = (t >= edges[j]) & (t < edges[j+1])
        if np.sum(mask) > 0:
            t_bins.append(np.mean(t[mask]))
            r_bins.append(np.mean(r[mask]))
            e_bins.append(np.std(r[mask]) / np.sqrt(np.sum(mask)))
    return np.array(t_bins), np.array(r_bins), np.array(e_bins)

# =====================================================================
# 6种可视化方法
# =====================================================================
fig, axes = plt.subplots(6, 1, figsize=(20, 30), sharex=True)

# --- 1. 朴素ax.plot(缺点:跨间隙连接)---
ax = axes[0]
t_all = np.concatenate([s[0] for s in segments])
r_all = np.concatenate([s[1] for s in segments])
ax.plot(t_all, r_all, 'k-', linewidth=0.3, alpha=0.5)
# 添加阈值线
ax.axhline(thresh1, color='#2166ac', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='#cc0000', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
ax.set_ylabel("Intensity")
ax.set_title("1. ax.plot() — 跨间隙连接(效果差)", fontsize=14, fontweight='bold')
ax.set_ylim(0, thresh2 * 3)

# --- 2. 分段折线图(修复间隙问题,但数据过密)---
ax = axes[1]
for t_seg, r_seg in segments:
    ax.plot(t_seg, r_seg, 'k-', linewidth=0.2, alpha=0.4)
# 添加阈值线
ax.axhline(thresh1, color='#2166ac', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='#cc0000', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
ax.set_ylabel("Intensity")
ax.set_title("2. 分段ax.plot() — 间隙展示正确,但数据过密难以读取", fontsize=14, fontweight='bold')
ax.set_ylim(0, thresh2 * 3)

# --- 3. 100秒分箱+误差棒图 ---
ax = axes[2]
colors = []
for t_seg, r_seg in segments:
    t_bin, r_bin, e_bin = bin_segment(t_seg, r_seg)
    seg_colors = color_for_rate(r_bin)
    colors.extend(seg_colors)
    ax.errorbar(t_bin, r_bin, yerr=e_bin, fmt='o', markersize=3, capsize=2, color=seg_colors, alpha=0.8)
# 添加阈值线
ax.axhline(thresh1, color='#2166ac', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='#cc0000', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
ax.set_ylabel("Mean Intensity (100s bin)")
ax.set_title("3. 分箱误差棒图 — 降低密度,展示统计特征,颜色编码生效", fontsize=14, fontweight='bold')
ax.set_ylim(0, thresh2 * 3)

# --- 4. 散点图(颜色编码)---
ax = axes[3]
for t_seg, r_seg in segments:
    ax.scatter(t_seg, r_seg, s=1, color=color_for_rate(r_seg), alpha=0.5)
# 添加阈值线
ax.axhline(thresh1, color='#2166ac', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='#cc0000', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
ax.set_ylabel("Intensity")
ax.set_title("4. 散点图 — 颜色编码清晰,但数据过密导致重叠严重", fontsize=14, fontweight='bold')
ax.set_ylim(0, thresh2 * 3)

# --- 5. 带颜色编码的分段折线图 ---
ax = axes[4]
for t_seg, r_seg in segments:
    # 按阈值拆分线段,实现颜色区分
    mask_low = r_seg < thresh1
    mask_mid = (r_seg >= thresh1) & (r_seg < thresh2)
    mask_high = r_seg >= thresh2
    
    if np.any(mask_low):
        ax.plot(t_seg[mask_low], r_seg[mask_low], color='#2166ac', linewidth=0.2, alpha=0.6)
    if np.any(mask_mid):
        ax.plot(t_seg[mask_mid], r_seg[mask_mid], color='#ff8c00', linewidth=0.2, alpha=0.6)
    if np.any(mask_high):
        ax.plot(t_seg[mask_high], r_seg[mask_high], color='#cc0000', linewidth=0.2, alpha=0.6)
# 添加阈值线
ax.axhline(thresh1, color='#2166ac', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='#cc0000', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
ax.set_ylabel("Intensity")
ax.set_title("5. 颜色编码分段折线图 — 保留时序细节,颜色区分强度区间", fontsize=14, fontweight='bold')
ax.set_ylim(0, thresh2 * 3)

# --- 6. 热力图(时间分箱)---
ax = axes[5]
# 构建时间网格和强度矩阵
t_all = np.concatenate([s[0] for s in segments])
r_all = np.concatenate([s[1] for s in segments])
# 按10分钟分箱
bin_width_hrs = 10/60
t_min, t_max = t_all.min(), t_all.max()
t_bins = np.arange(t_min, t_max + bin_width_hrs, bin_width_hrs)
r_bins = np.arange(0, thresh2 * 3 + 50, 50)

# 生成2D直方图(热力图数据)
heatmap, _, _ = np.histogram2d(t_all, r_all, bins=[t_bins, r_bins])
# 转置并取对数(避免尖峰掩盖细节)
heatmap = np.log1p(heatmap.T)

im = ax.imshow(heatmap, extent=[t_min, t_max, r_bins[0], r_bins[-1]], aspect='auto', origin='lower', cmap='viridis')
# 添加阈值线
ax.axhline(thresh1, color='white', linestyle='--', label=f'Median ({med:.1f})')
ax.axhline(thresh2, color='white', linestyle='--', label=f'2x Median ({thresh2:.1f})')
ax.legend()
plt.colorbar(im, ax=ax, label='Log(Count + 1)')
ax.set_xlabel("Time (Hours)")
ax.set_ylabel("Intensity")
ax.set_title("6. 时间分箱热力图 — 展示数据分布密度,突出尖峰高频区域", fontsize=14, fontweight='bold')

plt.tight_layout()
plt.savefig('sensor_timeseries_comparison.png')

各方法优缺点分析

  • 方法1:朴素折线图
    实现最简单,但会把离线间隙的前后数据错误连接,完全丢失间隙信息,不建议使用。
  • 方法2:分段折线图
    修复了间隙问题,但25万点的高密度导致线条重叠严重,无法看清时序细节和强度分布。
  • 方法3:分箱误差棒图
    通过分箱(示例用100秒)降低数据密度,同时展示均值和标准误,颜色编码清晰区分强度区间,适合快速观察统计趋势。
  • 方法4:散点图
    颜色编码直观,但数据过密时大量点重叠,无法分辨局部细节,仅适合极低密度数据。
  • 方法5:颜色编码分段折线图
    保留完整时序细节,同时用颜色区分不同强度区间,能清晰看到尖峰出现的时间点,是平衡细节和可读性的较好选择。
  • 方法6:时间分箱热力图
    从密度角度展示数据分布,能突出尖峰出现的高频时间段,适合分析数据的整体分布特征,但丢失了精确的时序数值。

内容的提问来源于stack exchange,提问作者requiemman

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.27 09:17:28