You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何修复对数刻度Hexplot调整图尺寸后出现的Bin重叠问题

对数刻度Hexplot缩小尺寸后Bin重叠混乱的原因与解决办法

我复现了Zipf定律可视化的代码,使用预设的figsize=(20,24)时图表显示正常,但将图尺寸改为(5.9055,7.0866)适配论文页面后,对数刻度的Hexplot出现Bin重叠混乱的问题。

正常效果(figsize=(20,24)):
正常效果

重叠问题效果(figsize=(5.9055,7.0866)):下方图中Bin混乱重叠
Bin重叠效果

可复现代码如下:

from typing import Tuple
import nltk
import urllib.request
from collections import Counter
import numpy as np
import re
import random
import pandas as pd
import matplotlib.pyplot as plt


def plot_zipf(text: str, title, alpha=1, cmap='inferno', include_counts: bool = False, figsize: Tuple[float, float]=(20, 24)):
    tokenizer = nltk.tokenize.RegexpTokenizer('\w+')
    tokens = tokenizer.tokenize(text.lower())
    # tokens = list(map(lambda x: x.lower(), text.split()))

    # Randomly split the tokens into two corpora
    random.shuffle(tokens)
    half = len(tokens) // 2
    corpus1, corpus2 = tokens[:half], tokens[half:]

    # Count word frequencies in each corpus
    counter1 = Counter(corpus1)
    counter2 = Counter(corpus2)

    # Find common words in both corpora
    common_words = set(counter1.keys()) & set(counter2.keys())

    # Create a dictionary of word, frequency pairs where frequency is the average of frequencies in two corpora 
    freqs = {word: (counter1[word] + counter2[word]) / 2 for word in common_words}

    # Create a dictionary of word, rank pairs where rank is determined from the frequencies in the second corpus
    ranks = {word: rank
             for rank, (word, freq) in enumerate(
                sorted([(w, counter2[w]) for w in common_words], key=lambda x: -x[1]),
                1)
             }

    # Create arrays for ranks and frequencies, ordered by rank
    words = sorted(freqs.keys(), key=ranks.get)
    ranks = np.array([ranks[word] for word in words])
    counts = np.array([freqs[word] for word in words])

    # plt.rcParams.update({'font.size': 30})
    fig = plt.figure(layout='constrained', figsize=figsize)
    axes = fig.subplot_mosaic('A;B', sharex=True)

    # Plot A: Zipf's Law with fitted line
    ax = axes['A']
    ax.set_title(f"Zipf's Law on {title}")
    ax.set_ylabel('Frequency')

    # Zipf law fit 
    indices = (ranks >= 10) & (ranks <= 1000)
    slope, intercept = np.polyfit(np.log(ranks[indices]), np.log(counts[indices]), 1)
    ax.plot(
        ranks,
        np.exp(intercept) * ranks ** slope,
        color='red',
        alpha=1,
        label=f"Zipf law ($f = 1/(r+{alpha})^{-slope:.2f}$)"
    )

    # Create the hexagonal heatmap using hexbin, on the log scale data 
    hb = ax.hexbin(
        ranks+alpha,
        counts,
        gridsize=100,
        cmap=cmap,
        bins='log',
        xscale='log',
        yscale='log',
    ) 
    if include_counts:
        cb = plt.colorbar(hb, ax=ax)
        cb.set_label('Counts in log10(N)')
    ax.legend()
    ax.grid()
    
    # Plot B: Zipf's Law with Zipf removed 
    ax = axes['B'] 
    ax.set_title('with Zipf removed')
    ax.set_xlabel('Frequency rank')
    ax.set_ylabel('Frequency/Zipf law')
    
    # Create the hexagonal heatmap using hexbin, on the log scale data
    hb = ax.hexbin(
        ranks+alpha,
        counts / (np.exp(intercept) * (ranks+alpha) ** slope),
        gridsize=100,
        cmap=cmap,
        bins='log',
        xscale='log',
        yscale='log',
    ) 
    if include_counts:
        cb = plt.colorbar(hb, ax=ax)
        cb.set_label('Counts in log10(N)')
    # ax.legend()
    ax.grid()
    # plt.show()
    return fig


def gutenberg(txt_url, title, alpha=1, figsize: Tuple[float, float]=(20, 24)): 
    response = urllib.request.urlopen(txt_url)
    print('response received')
    long_txt = response.read().decode('utf8') # Remove the Project Gutenberg boilerplate

    pattern = r'\*\*\* START OF THE PROJECT GUTENBERG EBOOK [^\*]* \*\*\*'
    match = re.search(pattern, long_txt)

    if match:
        start_index = match.end()
        end_marker = '*** END OF THE PROJECT GUTENBERG EBOOK'
        end_index = long_txt.find(end_marker, start_index)
        long_txt = long_txt[start_index:end_index].strip()
    else:
        print('Pattern not found in the text.') # Tokenize the text

    return plot_zipf(long_txt, title, alpha=alpha, figsize=figsize)


# works fine here
gutenberg(
    'https://www.gutenberg.org/cache/epub/2600/pg2600.txt', 
    'War and Peace', 
    alpha=2, 
    figsize=(20, 24),
)

# overlapping bins here
gutenberg(
    'https://www.gutenberg.org/cache/epub/2600/pg2600.txt', 
    'War and Peace', 
    alpha=2, 
    figsize=(5.9055, 7.0866),
)

问题原因

  • 固定gridsize未适配尺寸:代码中gridsize=100是固定值,当图表尺寸大幅缩小时,六边形网格密度未同步降低,导致小画布上六边形过度拥挤,出现视觉重叠。
  • 对数轴非线性特性:对数轴的坐标分布是非线性的,低秩区域坐标更密集,小尺寸画布下固定数量的网格会在该区域产生更严重的挤压。

解决办法

1. 动态调整gridsize

根据画布尺寸计算合适的网格数量,避免固定值带来的适配问题:

# 在plot_zipf函数中替换固定gridsize
gridsize = int(min(figsize[0], figsize[1]) * 5)  # 系数可根据效果调整
# 后续hexbin调用使用该变量
hb = ax.hexbin(
    ranks+alpha,
    counts,
    gridsize=gridsize,
    cmap=cmap,
    bins='log',
    xscale='log',
    yscale='log',
)

2. 过滤空Bin并限制绘图范围

用mincnt过滤无数据的Bin,extent限定绘图区域,减少无效网格:

hb = ax.hexbin(
    ranks+alpha,
    counts,
    gridsize=60,  # 针对小尺寸设置合适固定值
    cmap=cmap,
    bins='log',
    xscale='log',
    yscale='log',
    mincnt=1,  # 仅显示有数据的Bin
    extent=[np.min(ranks), np.max(ranks), np.min(counts), np.max(counts)]
)

3. 适配小尺寸的全局样式调整

缩小图尺寸后同步调整字体、元素大小,避免挤压Hexplot空间:

# 在plot_zipf函数开头添加
plt.rcParams.update({
    'font.size': 8,
    'axes.titlesize': 10,
    'axes.labelsize': 8,
    'legend.fontsize': 7
})

内容的提问来源于stack exchange,提问作者mxcx

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 10:46:02