You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R中根据原行名对已拆分的dataframe进行二次子集筛选?

按原行名筛选子DataFrame的问题

我有一个包含70000+行的大型DataFrame,拆分后得到的子DataFrame保留了原DataFrame的连续行名,但子DataFrame的行号默认从1开始。尝试按原行名进行二次筛选时,始终得到全为NA的新DataFrame。

原DataFrame结构

structure(list(ID = c("S25", "S25", "S25", "S25", "S25", "S25", 
"S25", "S25", "S25", "S25"), `Step length (km)` = c(0.0199109278453956, 
0.0164543610381346, 0.0132836337125505, 0.0136162564636374, 0.0107277708964608, 
0.0164161750767766, 0.0293870854019551, 0.0124045174064797, 0.0142961706697052, 
0), Angle = c(NA, -2.64417813085975, -1.53209954875884, 2.5154982969622, 
2.38871510555124, -2.29873359207977, 2.54239785024155, 1.69163871104124, 
2.10364018105916, NA), Longitude = c(1.54147, 1.54147, 1.54136, 
1.54152, 1.54133, 1.54143, 1.5412, 1.54154, 1.54162, 1.54142), 
    Latitude = c(50.215488, 50.215309, 50.215439, 50.2155, 50.215511, 
    50.215439, 50.215439, 50.21529, 50.215389, 50.215382), Date = c("07.10.2019", 
    "07.10.2019", "07.10.2019", "07.10.2019", "07.10.2019", "07.10.2019", 
    "07.10.2019", "07.10.2019", "07.10.2019", "07.10.2019"), 
    Time = c("18:01", "18:21", "19:01", "19:21", "19:41", "20:01", 
    "20:21", "21:01", "21:21", "21:41"), Depth = c(6, 6, 6, 6, 
    6, 6, 6, 6, 6, 6), land_sea = c("land", "land", "land", "land", 
    "land", "land", "land", "land", "land", "land")), row.names = 69400:69409, class = c("moveData", 
"data.frame"))

尝试的代码

pv_ID_S25_ft = pv_ID_S25[c('69402':'69407'),]

得到的错误结果

structure(list(ID = c(NA_character_, NA_character_, NA_character_, 
NA_character_, NA_character_), `Step length (km)` = c(NA_real_, 
NA_real_, NA_real_, NA_real_, NA_real_), Angle = c(NA_real_, 
NA_real_, NA_real_, NA_real_, NA_real_), Longitude = c(NA_real_, 
NA_real_, NA_real_, NA_real_, NA_real_), Latitude = c(NA_real_, 
NA_real_, NA_real_, NA_real_, NA_real_), Date = c(NA_character_, 
NA_character_, NA_character_, NA_character_, NA_character_), 
    Time = c(NA_character_, NA_character_, NA_character_, NA_character_, 
    NA_character_), Depth = c(NA_real_, NA_real_, NA_real_, NA_real_, 
    NA_real_), land_sea = c(NA_character_, NA_character_, NA_character_, 
    NA_character_, NA_character_)), row.names = c("NA", "NA.1", 
"NA.2", "NA.3", "NA.4"), class = c("moveData", "data.frame"))

期望结果

structure(list(ID = c("S25", "S25", "S25", "S25", 
"S25", "S25"), `Step length (km)` = c(0.0132836337125505, 0.0136162564636374, 0.0107277708964608, 
0.0164161750767766, 0.0293870854019551, 0.0124045174064797), Angle = c(-1.53209954875884, 2.5154982969622, 
2.38871510555124, -2.29873359207977, 2.54239785024155, 1.69163871104124), Longitude = c(1.54136, 
1.54152, 1.54133, 1.54143, 1.5412, 1.54154), 
    Latitude = c(50.215439, 50.2155, 50.215511, 
    50.215439, 50.215439, 50.21529), Date = c("07.10.2019", "07.10.2019", "07.10.2019", "07.10.2019", 
    "07.10.2019", "07.10.2019"), 
    Time = c("19:01", "19:21", "19:41", "20:01", 
    "20:21", "21:01"), Depth = c(6, 6, 
    6, 6, 6, 6), land_sea = c("land", "land", 
    "land", "land", "land", "land")), row.names = 69402:69407, class = c("moveData", 
"data.frame"))

解决方法

方法1:直接使用数值型行名范围

子DataFrame保留了原数值行名,直接用数值范围筛选即可,无需加引号:

pv_ID_S25_ft = pv_ID_S25[69402:69407, ]

方法2:匹配字符型行名(若行名被转为字符)

如果子DataFrame的行名是字符类型,用which()匹配目标行名:

target_rows = which(rownames(pv_ID_S25) %in% as.character(69402:69407))
pv_ID_S25_ft = pv_ID_S25[target_rows, ]

方法3:使用dplyr包筛选

若习惯tidyverse语法,可使用slice()或filter():

library(dplyr)
# 用slice筛选匹配的行
pv_ID_S25_ft = pv_ID_S25 %>% slice(which(rownames(.) %in% as.character(69402:69407)))

错误原因说明

你之前的代码c('69402':'69407')存在语法错误:字符型向量不能用冒号生成范围,R无法识别这个无效的索引,因此返回全NA的结果。直接使用原数值行名或匹配字符行名即可解决问题。

内容的提问来源于stack exchange,提问作者Kingpin96

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 23:27:35