如何修复对数刻度Hexplot调整图尺寸后出现的Bin重叠问题
对数刻度Hexplot缩小尺寸后Bin重叠混乱的原因与解决办法
我复现了Zipf定律可视化的代码,使用预设的figsize=(20,24)时图表显示正常,但将图尺寸改为(5.9055,7.0866)适配论文页面后,对数刻度的Hexplot出现Bin重叠混乱的问题。
正常效果(figsize=(20,24)):
重叠问题效果(figsize=(5.9055,7.0866)):下方图中Bin混乱重叠
可复现代码如下:
from typing import Tuple import nltk import urllib.request from collections import Counter import numpy as np import re import random import pandas as pd import matplotlib.pyplot as plt def plot_zipf(text: str, title, alpha=1, cmap='inferno', include_counts: bool = False, figsize: Tuple[float, float]=(20, 24)): tokenizer = nltk.tokenize.RegexpTokenizer('\w+') tokens = tokenizer.tokenize(text.lower()) # tokens = list(map(lambda x: x.lower(), text.split())) # Randomly split the tokens into two corpora random.shuffle(tokens) half = len(tokens) // 2 corpus1, corpus2 = tokens[:half], tokens[half:] # Count word frequencies in each corpus counter1 = Counter(corpus1) counter2 = Counter(corpus2) # Find common words in both corpora common_words = set(counter1.keys()) & set(counter2.keys()) # Create a dictionary of word, frequency pairs where frequency is the average of frequencies in two corpora freqs = {word: (counter1[word] + counter2[word]) / 2 for word in common_words} # Create a dictionary of word, rank pairs where rank is determined from the frequencies in the second corpus ranks = {word: rank for rank, (word, freq) in enumerate( sorted([(w, counter2[w]) for w in common_words], key=lambda x: -x[1]), 1) } # Create arrays for ranks and frequencies, ordered by rank words = sorted(freqs.keys(), key=ranks.get) ranks = np.array([ranks[word] for word in words]) counts = np.array([freqs[word] for word in words]) # plt.rcParams.update({'font.size': 30}) fig = plt.figure(layout='constrained', figsize=figsize) axes = fig.subplot_mosaic('A;B', sharex=True) # Plot A: Zipf's Law with fitted line ax = axes['A'] ax.set_title(f"Zipf's Law on {title}") ax.set_ylabel('Frequency') # Zipf law fit indices = (ranks >= 10) & (ranks <= 1000) slope, intercept = np.polyfit(np.log(ranks[indices]), np.log(counts[indices]), 1) ax.plot( ranks, np.exp(intercept) * ranks ** slope, color='red', alpha=1, label=f"Zipf law ($f = 1/(r+{alpha})^{-slope:.2f}$)" ) # Create the hexagonal heatmap using hexbin, on the log scale data hb = ax.hexbin( ranks+alpha, counts, gridsize=100, cmap=cmap, bins='log', xscale='log', yscale='log', ) if include_counts: cb = plt.colorbar(hb, ax=ax) cb.set_label('Counts in log10(N)') ax.legend() ax.grid() # Plot B: Zipf's Law with Zipf removed ax = axes['B'] ax.set_title('with Zipf removed') ax.set_xlabel('Frequency rank') ax.set_ylabel('Frequency/Zipf law') # Create the hexagonal heatmap using hexbin, on the log scale data hb = ax.hexbin( ranks+alpha, counts / (np.exp(intercept) * (ranks+alpha) ** slope), gridsize=100, cmap=cmap, bins='log', xscale='log', yscale='log', ) if include_counts: cb = plt.colorbar(hb, ax=ax) cb.set_label('Counts in log10(N)') # ax.legend() ax.grid() # plt.show() return fig def gutenberg(txt_url, title, alpha=1, figsize: Tuple[float, float]=(20, 24)): response = urllib.request.urlopen(txt_url) print('response received') long_txt = response.read().decode('utf8') # Remove the Project Gutenberg boilerplate pattern = r'\*\*\* START OF THE PROJECT GUTENBERG EBOOK [^\*]* \*\*\*' match = re.search(pattern, long_txt) if match: start_index = match.end() end_marker = '*** END OF THE PROJECT GUTENBERG EBOOK' end_index = long_txt.find(end_marker, start_index) long_txt = long_txt[start_index:end_index].strip() else: print('Pattern not found in the text.') # Tokenize the text return plot_zipf(long_txt, title, alpha=alpha, figsize=figsize) # works fine here gutenberg( 'https://www.gutenberg.org/cache/epub/2600/pg2600.txt', 'War and Peace', alpha=2, figsize=(20, 24), ) # overlapping bins here gutenberg( 'https://www.gutenberg.org/cache/epub/2600/pg2600.txt', 'War and Peace', alpha=2, figsize=(5.9055, 7.0866), )
问题原因
- 固定
gridsize未适配尺寸:代码中gridsize=100是固定值,当图表尺寸大幅缩小时,六边形网格密度未同步降低,导致小画布上六边形过度拥挤,出现视觉重叠。 - 对数轴非线性特性:对数轴的坐标分布是非线性的,低秩区域坐标更密集,小尺寸画布下固定数量的网格会在该区域产生更严重的挤压。
解决办法
1. 动态调整gridsize
根据画布尺寸计算合适的网格数量,避免固定值带来的适配问题:
# 在plot_zipf函数中替换固定gridsize gridsize = int(min(figsize[0], figsize[1]) * 5) # 系数可根据效果调整 # 后续hexbin调用使用该变量 hb = ax.hexbin( ranks+alpha, counts, gridsize=gridsize, cmap=cmap, bins='log', xscale='log', yscale='log', )
2. 过滤空Bin并限制绘图范围
用mincnt过滤无数据的Bin,extent限定绘图区域,减少无效网格:
hb = ax.hexbin( ranks+alpha, counts, gridsize=60, # 针对小尺寸设置合适固定值 cmap=cmap, bins='log', xscale='log', yscale='log', mincnt=1, # 仅显示有数据的Bin extent=[np.min(ranks), np.max(ranks), np.min(counts), np.max(counts)] )
3. 适配小尺寸的全局样式调整
缩小图尺寸后同步调整字体、元素大小,避免挤压Hexplot空间:
# 在plot_zipf函数开头添加 plt.rcParams.update({ 'font.size': 8, 'axes.titlesize': 10, 'axes.labelsize': 8, 'legend.fontsize': 7 })
内容的提问来源于stack exchange,提问作者mxcx
相关产品推荐
相关产品推荐

